fix(arm64): store-exclusive operand order and large-frame parity
Assisted-by: GLM 5.3
This commit is contained in:
+75
-21
@@ -187,10 +187,32 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
|
||||
return a64WordsLE(ws...)
|
||||
}
|
||||
|
||||
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
|
||||
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
|
||||
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
|
||||
// subtracts the register in the extended-register form.
|
||||
// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
|
||||
// two imm12 instructions instead of materialising it into REGTMP
|
||||
// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
|
||||
// unsigned and be neither encodable as one imm12 (checked by the callers
|
||||
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
|
||||
// logical immediate, because conclass tests all three before C_ADDCON2.
|
||||
func arm64SplitImm12(imm uint32) bool {
|
||||
if imm > 0xFFFFFF {
|
||||
return false
|
||||
}
|
||||
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
|
||||
return false
|
||||
}
|
||||
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
|
||||
}
|
||||
|
||||
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
|
||||
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
|
||||
// immediate form when the value fits imm12 (plain, or shifted left by 12
|
||||
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
|
||||
// logical immediate, or one wider than 24 bits is materialised into REGTMP
|
||||
// (R27) and subtracted in the extended-register form; everything else up to
|
||||
// 0xFFFFFF is split into two imm12 instructions:
|
||||
//
|
||||
// SUB $(imm&0xfff), SP, Rd
|
||||
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
|
||||
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 0xFFF {
|
||||
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
||||
@@ -198,15 +220,21 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
||||
}
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
if !arm64SplitImm12(imm) {
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
||||
}
|
||||
return []uint32{
|
||||
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
|
||||
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
||||
}
|
||||
|
||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
|
||||
// and REGTMP fallback ladder.
|
||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
|
||||
// split and REGTMP ladder as arm64SubImmWords.
|
||||
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 0xFFF {
|
||||
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
||||
@@ -214,11 +242,35 @@ func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
||||
}
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if !arm64SplitImm12(imm) {
|
||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
||||
}
|
||||
return []uint32{
|
||||
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
|
||||
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||
}
|
||||
}
|
||||
|
||||
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
|
||||
// large frame. The toolchain adds the frame back with a single instruction:
|
||||
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
|
||||
// materialised into REGTMP and added as a register, so the epilogue never
|
||||
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
|
||||
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
|
||||
// the leaf epilogue's plain ADD instructions.
|
||||
func arm64RetAddWords(autosize uint32) []uint32 {
|
||||
if autosize < 1<<12 {
|
||||
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
|
||||
}
|
||||
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
|
||||
if err != nil {
|
||||
mov = nil
|
||||
}
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
|
||||
}
|
||||
|
||||
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
||||
@@ -237,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte {
|
||||
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
||||
)
|
||||
} else {
|
||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
||||
// Large frame: LDP -8(SP), (FP, LR), then deallocate.
|
||||
ws = append(ws,
|
||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||
)
|
||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
||||
ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
|
||||
}
|
||||
}
|
||||
// RET: BR LR (0xd65f03c0)
|
||||
@@ -260,15 +312,16 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
|
||||
}
|
||||
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
|
||||
// position depends on how many words the SUB itself took (immediate,
|
||||
// shifted immediate, or a materialised REGTMP sequence).
|
||||
// shifted immediate, the two-word imm12 split, or a materialised REGTMP
|
||||
// sequence).
|
||||
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
|
||||
}
|
||||
|
||||
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||
// (but not including) the final RET instruction. The ADD sequences share the
|
||||
// prologue's immediate ladder, so their length is read from the same helper
|
||||
// rather than assumed: a materialised autosize costs its MOV words plus the
|
||||
// ADD itself.
|
||||
// (but not including) the final RET instruction. The lengths are read from
|
||||
// the same word-emitting helpers the epilogue uses rather than assumed: the
|
||||
// leaf path shares the prologue's immediate ladder, and a materialised
|
||||
// autosize costs its MOV words plus the ADD itself.
|
||||
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
||||
if fi.autosize == 0 {
|
||||
return 0
|
||||
@@ -280,8 +333,9 @@ func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
||||
if fi.autosize <= 0xf0 {
|
||||
return 8 // LDR + LDR.P
|
||||
}
|
||||
// LDP + the ADD ladder that deallocates the frame.
|
||||
return 4 + 4*len(arm64AddImmWords(uint32(fi.autosize), 31))
|
||||
// LDP + the deallocation emitted by arm64RetAddWords, so the length
|
||||
// tracks whatever the MOVD ladder needs.
|
||||
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
|
||||
}
|
||||
|
||||
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
||||
|
||||
Reference in New Issue
Block a user