fix(asm): place the arm64 literal pool the way the toolchain flushes it

Assisted-by: GLM 5.3
This commit is contained in:
petrbalvin committed 2026-10-07 13:49:49 +02:00
1 parent 4fc96decc4
commit 405e2427ed
3 files changed
+215 -68

No files matched your search

+59 -46
View File
@@ -89,11 +89,20 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
out = append(out, prologue...)
pc := guardLen + len(prologue)
// The offset literal pool lands after the last instruction (and after
// the morestack block); a function whose last instruction does not
// branch gets an UNDEF first, the toolchain's flushpool guard against
// falling through into the words. The base decides the PC-relative
// distances the pool loads encode, so it is fixed before pass 2.
poolBase := guardLen + len(prologue) + bodyLen + arm64PoolPadLen(t)
// the morestack block, whose trailing branch closes the function); a
// function whose last instruction does not branch gets an UNDEF first,
// the toolchain's flushpool guard against falling through into the
// words. The base decides the PC-relative distances the pool loads
// encode, so it is fixed before pass 2.
poolGuard := arm64PoolPadLen(t)
poolBase := guardLen + len(prologue) + bodyLen + poolGuard
if fi.needSplit {
// The morestack block's B back to the entry is the toolchain's last
// Prog, an unconditional branch: the pool follows the block itself,
// with no guard before it.
poolGuard = 0
poolBase += arm64MoreStackBlockLen
}
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
@@ -144,12 +153,13 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
relocs = append(relocs, blReloc)
pc += len(block)
}
// The pool itself: the UNDEF guard word when the body does not end in a
// branch, then the pooled constants in first-use order. The guard is
// the toolchain's word-zero UNDEF, not the BRK the UNDEF statement
// spells: it only has to be a faulting word nothing jumps to.
// The pool itself: the UNDEF guard word when the function does not end
// in a branch (the morestack block's B counts as one), then the pooled
// constants in first-use order. The guard is the toolchain's word-zero
// UNDEF, not the BRK the UNDEF statement spells: it only has to be a
// faulting word nothing jumps to.
if pool.size > 0 {
if arm64PoolPadLen(t) > 0 {
if poolGuard > 0 {
out = append(out, a64wordLE(0)...)
pc += 4
}
@@ -5211,7 +5221,10 @@ func moviLitName(mnem string, data []byte) string {
// arm64PoolPadLen returns the UNDEF word the pool guard needs: four bytes
// when the body's last instruction does not branch (the toolchain's
// flushpool inserts one so execution cannot fall through into the words),
// zero otherwise.
// zero otherwise. END closes the body without becoming an instruction, so
// it is skipped the way a trailing label is; FUNCDATA and PCDATA stay real
// statements, exactly the Progs the toolchain's flushpool sees as the last
// one.
func arm64PoolPadLen(t *ast.Text) int {
for _, v := range slices.Backward(t.Body) {
in, ok := v.(*ast.Instr)
@@ -5219,6 +5232,8 @@ func arm64PoolPadLen(t *ast.Text) int {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "END":
continue
case "RET", "B", "JMP", "ERET":
return 0
}
@@ -5258,9 +5273,11 @@ func (l *arm64Literals) list() []Arm64Literal { return l.order }
// arm64Pool collects the out-of-range load/store offsets a function pools.
// The toolchain appends them after the last instruction (asm7.go addpool and
// flushpool) and reaches them with PC-relative literal loads into REGTMP;
// equal values deduplicate to one entry regardless of which instruction
// pooled them first.
// flushpool) and reaches them with PC-relative literal loads into REGTMP.
// Entries deduplicate by value alone, whatever width the first referrer
// selected, and concatenate in first-use order with no alignment padding:
// the toolchain's roundUp touches its size accounting alone, never the byte
// stream.
type arm64Pool struct {
order []arm64PoolEntry
seen map[int64]int // pooled value → entry index
@@ -5268,37 +5285,50 @@ type arm64Pool struct {
}
// arm64PoolEntry is one pooled constant: its bytes, its offset from the pool
// start and the literal-load width the first referrer selected (0 = LDR W,
// 2 = LDRSW for a negative word, 1 = LDR X for an 8-byte entry).
// start and the literal-load width its bytes select (0 = LDR W zero-extended,
// 1 = LDR X for an 8-byte entry). omovlit reads the width off the entry
// itself, so every referrer of a value loads with the first referrer's
// width; negative values always take the 8-byte entry, which makes the
// sign-extended LDRSW load unreachable for this pool.
type arm64PoolEntry struct {
data []byte
off int
w uint32
}
// add interns a pooled value and returns its offset from the pool start and
// the literal-load width. A value beyond the 32-bit reach takes an
// eight-byte entry aligned to eight; a negative word takes the sign-extended
// load, the toolchain's omovlit choice for its AMOVD pool reference.
// add interns a pooled load/store offset and returns its offset from the
// pool start and the literal-load width (asm7.go addpool): a value inside
// [0, 0x7FFFFFFF] takes a four-byte word loaded zero-extended, anything
// else the eight-byte slot a full LDR X reads.
func (p *arm64Pool) add(v int64) (int, uint32) {
return p.addEntry(v, false)
}
// add64 interns a pooled displacement of the MOVD $con(R) lowering (asm7.go
// case 34): the entry takes the eight-byte slot even when the value fits a
// word, but an existing entry of the same value is shared as it stands, the
// toolchain's value-only dedup.
func (p *arm64Pool) add64(v int64) (int, uint32) {
return p.addEntry(v, true)
}
// addEntry creates or reuses the pool entry for v. Reuse is by value alone;
// at creation, lacon forces the eight-byte slot and every other requestor
// takes it only for a value no 32-bit load can carry: omovlit's ADWORD rule
// `lit != int32(lit) || uint64(lit) != uint32(lit)`.
func (p *arm64Pool) addEntry(v int64, lacon bool) (int, uint32) {
if i, ok := p.seen[v]; ok {
return p.order[i].off, p.order[i].w
}
wide := v != int64(int32(v)) || uint64(v) != uint64(uint32(v))
off := p.size
var data []byte
var w uint32
switch {
case wide:
if lacon || v < 0 || v > 0x7FFFFFFF {
w = 1 // LDR X
off = (p.size + 7) &^ 7
data = a64WordsLE(uint32(v), uint32(v>>32))
p.size = off + 8
case v < 0:
w = 2 // LDRSW, sign-extended to 64
data = a64wordLE(uint32(v))
p.size = off + 4
default:
} else {
w = 0 // LDR W, zero-extended
data = a64wordLE(uint32(v))
p.size = off + 4
}
@@ -5310,23 +5340,6 @@ func (p *arm64Pool) add(v int64) (int, uint32) {
return off, w
}
// add64 reserves an 8-byte slot for v loaded by a full LDR X: the lacon
// pool path always reads 64 bits, even when the value fits 32 (asm7.go case
// 34's omovlit(AMOVD)).
func (p *arm64Pool) add64(v int64) (int, uint32) {
if i, ok := p.seen[v]; ok && p.order[i].w == 1 {
return p.order[i].off, p.order[i].w
}
off := (p.size + 7) &^ 7
p.size = off + 8
if p.seen == nil {
p.seen = map[int64]int{}
}
p.seen[v] = len(p.order)
p.order = append(p.order, arm64PoolEntry{data: a64WordsLE(uint32(v), uint32(v>>32)), off: off, w: 1})
return off, 1
}
// AssembleFileARM64 assembles every TEXT function of a parsed arm64 file
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
// code. SB references in the code are encoded as ADRP pairs with zero