fix(asm): close the oracle parity gaps in frame addressing and calls
This commit is contained in:
+60
-32
@@ -222,7 +222,7 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
return a64wordLE(uint32(immFromOperand(ops[0]))), nil
|
||||
case "B":
|
||||
case "B", "JMP":
|
||||
return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve)
|
||||
case "BL", "CALL":
|
||||
return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve)
|
||||
@@ -337,8 +337,9 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
||||
}
|
||||
op := ops[0]
|
||||
|
||||
// External symbol reference: BL sym(SB).
|
||||
if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
|
||||
// Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
|
||||
// relocation (R_CALLARM64 either way).
|
||||
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
|
||||
if relocs != nil {
|
||||
*relocs = append(*relocs, Reloc{
|
||||
Off: 0,
|
||||
@@ -348,8 +349,12 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
||||
Kind: RelArm64Branch,
|
||||
})
|
||||
}
|
||||
// Emit BL with zero offset; the linker fills in the target.
|
||||
return a64wordLE(a64Branch(1, 0)), nil
|
||||
// Emit B/BL with zero offset; the linker fills in the target.
|
||||
bop := uint32(0) // B
|
||||
if link {
|
||||
bop = 1 // BL
|
||||
}
|
||||
return a64wordLE(a64Branch(bop, 0)), nil
|
||||
}
|
||||
|
||||
target := resolve(arm64Label(op))
|
||||
@@ -593,9 +598,9 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
}
|
||||
_, off := arm64MemWithFrame(mem, fi)
|
||||
// Scaled unsigned offset fits if aligned and in range.
|
||||
lt := a64LoadTable[mnem]
|
||||
if lt.size == 0 {
|
||||
lt.size = 3 // default to64-bit for MOV
|
||||
lt, ok := a64LoadTable[mnem]
|
||||
if !ok {
|
||||
lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access
|
||||
}
|
||||
scale := int32(1) << uint(lt.size)
|
||||
if off >= 0 && off%scale == 0 && off/scale < 4096 {
|
||||
@@ -604,7 +609,10 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if off >= -256 && off <= 255 {
|
||||
return 4 // unscaled
|
||||
}
|
||||
return 12 // materialise offset + LDR/STR
|
||||
if _, _, _, ok := arm64SplitOffset(off, scale); ok {
|
||||
return 8 // ADD base, REGTMP + access
|
||||
}
|
||||
return 12 // literal pool range: encoding reports it as unsupported
|
||||
default:
|
||||
return 4 // register move
|
||||
}
|
||||
@@ -824,33 +832,53 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
|
||||
}
|
||||
|
||||
scale := int32(1) << uint(lt.size)
|
||||
if load {
|
||||
// Try scaled unsigned offset first.
|
||||
if off >= 0 && off%scale == 0 {
|
||||
imm12 := uint32(off / scale)
|
||||
if imm12 < 4096 {
|
||||
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil
|
||||
}
|
||||
}
|
||||
// Try unscaled (9-bit signed).
|
||||
if off >= -256 && off <= 255 {
|
||||
return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil
|
||||
}
|
||||
// Large offset: materialise in R20 (TMP) and use register-offset.
|
||||
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
|
||||
}
|
||||
// Store: same encoding but opc bits indicate store.
|
||||
storeOpc := a64StoreOpc(lt)
|
||||
if off >= 0 && off%scale == 0 {
|
||||
imm12 := uint32(off / scale)
|
||||
if imm12 < 4096 {
|
||||
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil
|
||||
}
|
||||
var opc int
|
||||
if load {
|
||||
opc = lt.opc
|
||||
} else {
|
||||
opc = storeOpc
|
||||
}
|
||||
// Scaled unsigned offset first, then the unscaled ±255 form.
|
||||
if off >= 0 && off%scale == 0 && off/scale < 4096 {
|
||||
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil
|
||||
}
|
||||
if off >= -256 && off <= 255 {
|
||||
return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil
|
||||
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, off, rn, reg)), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
|
||||
// Large offset: materialise the base in REGTMP (R27) the way the
|
||||
// toolchain does and access what remains.
|
||||
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, 0, 0, addShift, uint32(addImm), 31, 27), // ADD $addImm<<shift, SP, R27
|
||||
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
|
||||
), nil
|
||||
}
|
||||
|
||||
// arm64SplitOffset decomposes an out-of-range frame offset for a REGTMP
|
||||
// base: an ADD (plain, or shifted left by 12) brings SP near the target and
|
||||
// the access covers what remains. ok is false when no decomposition exists
|
||||
// (offsets at or beyond 16 MiB, where the toolchain falls back to a literal
|
||||
// pool).
|
||||
func arm64SplitOffset(off int32, scale int32) (addImm, addShift uint32, access int32, ok bool) {
|
||||
if off < 0 {
|
||||
return 0, 0, 0, false
|
||||
}
|
||||
// Plain ADD: bring SP to within the largest scaled access.
|
||||
l := min(off, 4095*scale)
|
||||
l -= l % scale
|
||||
if a := off - l; a <= 4095 {
|
||||
return uint32(a), 0, l, true
|
||||
}
|
||||
// Shifted ADD: cover everything but the bits the access imm12 carries.
|
||||
rest := off &^ (0xFFF * scale)
|
||||
if rest >= 0 && rest>>12 <= 4095 {
|
||||
return uint32(rest >> 12), 1, off - rest, true
|
||||
}
|
||||
return 0, 0, 0, false
|
||||
}
|
||||
|
||||
// ---- static symbol references (ADRP + offset) ----
|
||||
|
||||
+21
-6
@@ -86,9 +86,16 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
|
||||
|
||||
if fi.frame != 0 || !fi.leaf {
|
||||
fi.autosize = fi.frame + 8 // space for the saved LR
|
||||
if fi.autosize%16 != 0 {
|
||||
// The toolchain aligns to 16: if autosize%16 == 8, add 8;
|
||||
// otherwise add whatever is needed.
|
||||
// The toolchain always adds an extrasize: 8 when the total leaves a
|
||||
// 16-byte alignment gap, another 16 when already aligned.
|
||||
switch fi.autosize % 16 {
|
||||
case 8:
|
||||
fi.autosize += 8
|
||||
case 0:
|
||||
fi.autosize += 16
|
||||
default:
|
||||
// The toolchain rejects unaligned frames; round up so such
|
||||
// sources still assemble.
|
||||
fi.autosize += 16 - (fi.autosize % 16)
|
||||
}
|
||||
}
|
||||
@@ -181,12 +188,16 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
|
||||
}
|
||||
|
||||
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
|
||||
// fits the imm12 field, otherwise the toolchain materialises it into REGTMP
|
||||
// (R27) and subtracts the register in the extended-register form.
|
||||
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
|
||||
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
|
||||
// subtracts the register in the extended-register form.
|
||||
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 0xFFF {
|
||||
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
||||
}
|
||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
||||
}
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
@@ -194,11 +205,15 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
||||
}
|
||||
|
||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same REGTMP fallback.
|
||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
|
||||
// and REGTMP fallback ladder.
|
||||
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 0xFFF {
|
||||
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
||||
}
|
||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
||||
}
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
|
||||
+12
-4
@@ -515,6 +515,9 @@ func addSP(size int) []byte { // ADDQ $size, SP
|
||||
func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) {
|
||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||
if isJumpMnemonic(mnem) {
|
||||
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
||||
return 5, nil // opcode + rel32, always the long form
|
||||
}
|
||||
return jumpSize(mnem, long), nil
|
||||
}
|
||||
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
||||
@@ -564,9 +567,10 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
||||
var ps []sbPatch
|
||||
var err error
|
||||
if isJumpMnemonic(mnem) {
|
||||
if mnem == "CALL" && isSBCall(s) {
|
||||
// CALL sym(SB): a rel32 call against a static or external
|
||||
// symbol, resolved by the file-level layout or the linker.
|
||||
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
||||
// CALL/JMP sym(SB): a rel32 call (or tail call) against a
|
||||
// static or external symbol, resolved by the file-level layout
|
||||
// or the linker.
|
||||
code, ps, err = encodeSBCall(s, link)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
@@ -678,8 +682,12 @@ func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||
if !ok {
|
||||
return nil, nil, fmt.Errorf("CALL: unsupported operand")
|
||||
}
|
||||
opcode := []byte{0xE8}
|
||||
if strings.ToUpper(s.Mnemonic.Text) == "JMP" {
|
||||
opcode = []byte{0xE9} // a tail call, no return address pushed
|
||||
}
|
||||
e := &enc{}
|
||||
if err := e.emit(&instr{opcode: []byte{0xE8}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil {
|
||||
if err := e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
ps := make([]sbPatch, len(e.patches))
|
||||
|
||||
@@ -527,13 +527,18 @@ func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[stri
|
||||
}
|
||||
return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil
|
||||
}
|
||||
// Direct symbol: sym+off(SB) → bl with an R_CALLLOONG64 relocation (the
|
||||
// linker fills the offset), as the toolchain does for CALL/BL/JAL.
|
||||
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" && link {
|
||||
// Direct symbol: sym+off(SB) → b/bl with an R_CALLLOONG64 relocation
|
||||
// (the linker fills the offset), as the toolchain does for CALL/BL/JAL
|
||||
// and for tail-calling JMP.
|
||||
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
|
||||
opc := l64jumpTable["B"]
|
||||
if link {
|
||||
opc = l64jumpTable["BL"]
|
||||
}
|
||||
if relocs != nil {
|
||||
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelLoong64Branch, Addend: op.Addr.Sym.Offset})
|
||||
}
|
||||
return l64wordLE(l64bbl(l64jumpTable["BL"], 0)), nil
|
||||
return l64wordLE(l64bbl(opc, 0)), nil
|
||||
}
|
||||
// Direct: label → b/bl.
|
||||
target := resolve(l64Label(op))
|
||||
|
||||
+56
-4
@@ -162,6 +162,14 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
|
||||
if isImmOperand(ops[0]) && ops[0].Imm.Sym == nil {
|
||||
return riscvMovImmSize(regFromOperand(ops[1]), immFromOperand(ops[0]))
|
||||
}
|
||||
// Frame-relative loads and stores: a frame offset beyond the signed
|
||||
// 12-bit range materialises the address in X31 first.
|
||||
if isMemOperand(ops[0]) && !isMemOperand(ops[1]) {
|
||||
return riscvFrameMemSize(ops[0], fi)
|
||||
}
|
||||
if isMemOperand(ops[1]) && !isMemOperand(ops[0]) {
|
||||
return riscvFrameMemSize(ops[1], fi)
|
||||
}
|
||||
}
|
||||
// I-type arithmetic with a large immediate expands to several instructions.
|
||||
if (mnem == "ADDI" || mnem == "ANDI" || mnem == "ORI" || mnem == "XORI") && len(ops) >= 1 && isImmOperand(ops[0]) {
|
||||
@@ -212,6 +220,14 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
|
||||
// C.J, so always emit the 32-bit JAL.
|
||||
var target string
|
||||
if len(ops) >= 1 {
|
||||
// JMP sym(SB): a tail call, JAL X0 against a symbol relocation.
|
||||
if ops[0].Addr.Sym != nil && ops[0].Addr.Sym.Pseudo == "SB" {
|
||||
if relocs != nil {
|
||||
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: ops[0].Addr.Sym.Name, Kind: RelRISCVJal, Addend: ops[0].Addr.Sym.Offset})
|
||||
}
|
||||
word = riscvJType(0, 0)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
}
|
||||
target = labelFromOperand(ops[0])
|
||||
}
|
||||
targetOff, ok := offsets[target]
|
||||
@@ -586,8 +602,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
|
||||
if rd < 0 || rs1 < 0 {
|
||||
return nil, fmt.Errorf("MOV load: invalid operand")
|
||||
}
|
||||
word := riscvIType(riscvEnc{0x03, 0x3, 0x00}, rd, rs1, off)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
return riscvFrameMemOp(riscvEnc{0x03, 0x3, 0x00}, false, rd, rs1, off), nil
|
||||
}
|
||||
|
||||
// Register → memory (store).
|
||||
@@ -604,8 +619,7 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
|
||||
if rs2 < 0 || rs1 < 0 {
|
||||
return nil, fmt.Errorf("MOV store: invalid operand")
|
||||
}
|
||||
word := riscvSType(riscvEnc{0x23, 0x3, 0x00}, rs1, rs2, off)
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
|
||||
return riscvFrameMemOp(riscvEnc{0x23, 0x3, 0x00}, true, rs2, rs1, off), nil
|
||||
}
|
||||
|
||||
// Register → register (ADDI $0, src, dst).
|
||||
@@ -620,6 +634,44 @@ func encodeRISCVMov(instr *ast.Instr, fi riscvFrameInfo, relocs *[]Reloc) ([]byt
|
||||
}
|
||||
}
|
||||
|
||||
// riscvFrameMemOp encodes a register-relative load (store=false, I-type
|
||||
// width 0x03) or store (store=true, S-type width 0x23) of the 64-bit width
|
||||
// at off(rs1). Offsets beyond the signed 12-bit range materialise the
|
||||
// address in X31 first: LUI hi (the rounding split), then ADD X31, rs1,
|
||||
// matching the toolchain's large-frame addressing; the access uses the
|
||||
// sign-extended low part, which always fits.
|
||||
func riscvFrameMemOp(enc riscvEnc, store bool, reg, rs1 int, off int32) []byte {
|
||||
if fits12(off) {
|
||||
var word uint32
|
||||
if store {
|
||||
word = riscvSType(enc, rs1, reg, off)
|
||||
} else {
|
||||
word = riscvIType(enc, reg, rs1, off)
|
||||
}
|
||||
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}
|
||||
}
|
||||
lo := off - (splitHi(off) << 12)
|
||||
out := riscvAddressInX31WithBase(off, rs1)
|
||||
var word uint32
|
||||
if store {
|
||||
word = riscvSType(enc, 31, reg, lo)
|
||||
} else {
|
||||
word = riscvIType(enc, reg, 31, lo)
|
||||
}
|
||||
return append(out, wordLE(word)...)
|
||||
}
|
||||
|
||||
// riscvFrameMemSize returns the encoded size of a frame-relative MOV for the
|
||||
// layout pass: 4 bytes when the offset fits, otherwise the X31
|
||||
// materialisation plus the access.
|
||||
func riscvFrameMemSize(op *ast.Operand, fi riscvFrameInfo) int {
|
||||
rs1, off := memFromOperandWithFrame(op, fi)
|
||||
if fits12(off) {
|
||||
return 4
|
||||
}
|
||||
return len(riscvAddressInX31WithBase(off, rs1)) + 4
|
||||
}
|
||||
|
||||
// encodeRISCVLoadImm encodes loading an immediate into a register (MOV $imm,
|
||||
// rd), matching the toolchain's instructionsForMOVConst. For 12-bit
|
||||
// immediates it emits ADDI $imm, ZERO, rd (compressed to C.LI when it fits
|
||||
|
||||
+13
-10
@@ -146,10 +146,18 @@ func splitHi(v int32) int32 {
|
||||
return high
|
||||
}
|
||||
|
||||
// riscvAddressInX31 materialises hi(v) into X31 and leaves the caller to add
|
||||
// the low part, matching the toolchain's large-frame addressing: C.LUI (or
|
||||
// LUI) X31, hi; C.ADD (or ADD) X31, SP.
|
||||
// riscvAddressInX31 materialises hi(v) into X31 against the stack pointer,
|
||||
// matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi;
|
||||
// C.ADD (or ADD) X31, SP.
|
||||
func riscvAddressInX31(v int32) []byte {
|
||||
return riscvAddressInX31WithBase(v, 2)
|
||||
}
|
||||
|
||||
// riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary
|
||||
// base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field
|
||||
// carries the full 5-bit register, so the compressed form is always
|
||||
// available.
|
||||
func riscvAddressInX31WithBase(v int32, rs1 int) []byte {
|
||||
hi := splitHi(v)
|
||||
var out []byte
|
||||
if hi >= -32 && hi <= 31 {
|
||||
@@ -158,13 +166,8 @@ func riscvAddressInX31(v int32) []byte {
|
||||
} else {
|
||||
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
|
||||
}
|
||||
if hi >= -32 && hi <= 31 {
|
||||
c := rvcCR(0x9, 31, 2)
|
||||
out = append(out, byte(c), byte(c>>8))
|
||||
} else {
|
||||
out = append(out, wordLE(riscvRType(riscvEnc{0x33, 0x0, 0x00}, 31, 2, 31))...)
|
||||
}
|
||||
return out
|
||||
c := rvcCR(0x9, 31, uint32(rs1))
|
||||
return append(out, byte(c), byte(c>>8))
|
||||
}
|
||||
|
||||
// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry:
|
||||
|
||||
Reference in New Issue
Block a user