feat(asm): drain the arm64 literal pool mid-function at the distance bound

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 19:54:22 +02:00
1 parent 4632ac1bb9
commit 5a5936d222
1 file changed
+328 -137
+328 -137
View File
@@ -53,99 +53,89 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
spadj = append(spadj, SpadjStep{PC: guardLen + arm64PrologueSpadjPC(fi), Value: fi.autosize})
}
// Pass 1: label offsets from the instruction sizes.
// Pass 1 plans the whole layout in one walk, the way the toolchain's
// span7 runs its own single linear pass: the label offsets come out of
// the same positions the encoder lays down, and the literal pool's
// references are harvested per statement with a probe encoding, so
// checkpool's flush condition is evaluated exactly as the toolchain's
// is and a reference that would leave the load-literal displacement
// bound drains the pool inside the body, at the point the toolchain
// would drain it.
offsets := map[string]int{}
pos := guardLen + len(prologue)
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pos += arm64PCAlignPad(pos, s)
} else {
pos += arm64InstrSize(s, fi, pos)
}
}
}
layout := arm64PlanPool(t, fi, guardLen, len(prologue), pool, offsets, resolve)
pool.probe = false
// Pass 2: encode. The guard prefix precedes the prologue; its branches
// target the morestack block at the end of the function, whose position
// the first pass has settled.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += arm64InstrSize(in, fi, p)
}
}
bodyLen = p - (guardLen + len(prologue))
}
// the plan has settled.
var out []byte
if fi.needSplit {
out = append(out, arm64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
out = append(out, arm64GuardBytes(fi, layout.blockStart)...)
}
out = append(out, prologue...)
pc := guardLen + len(prologue)
// The offset literal pool lands after the last instruction (and after
// the morestack block, whose trailing branch closes the function); a
// function whose last instruction does not branch gets an UNDEF first,
// the toolchain's flushpool guard against falling through into the
// words. The base decides the PC-relative distances the pool loads
// encode, so it is fixed before pass 2.
poolGuard := arm64PoolPadLen(t)
poolBase := guardLen + len(prologue) + bodyLen + poolGuard
if fi.needSplit {
// The morestack block's B back to the entry is the toolchain's last
// Prog, an unconditional branch: the pool follows the block itself,
// with no guard before it.
poolGuard = 0
poolBase += arm64MoreStackBlockLen
}
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
// The drained segments replay in the plan's own order: the segment open
// at a statement holds its references, and the flush event after it
// emits the guard word and the words the segment drained.
segIdx := 0
for i, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" {
flushAt := -1
if len(pool.segs) > 0 {
for segIdx < len(pool.segs)-1 && pool.segs[segIdx].after < i {
segIdx++
}
pool.active = segIdx
if pool.segs[segIdx].after == i {
flushAt = segIdx
}
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "PCALIGN":
pad := arm64PCAlignPad(pc, in)
for i := 0; i < pad/4; i++ {
for j := 0; j < pad/4; j++ {
out = append(out, a64wordLE(a64NOP)...)
pc += 4
}
continue
}
if strings.ToUpper(in.Mnemonic.Text) == "BYTE" {
case "BYTE":
for _, op := range in.Operands {
out = append(out, byte(arm64Imm64(op)))
pc++
}
continue
default:
code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve, lits, pool, pool.wordsBase())
if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
for j := preCount; j < len(relocs); j++ {
// Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
// The RET's epilogue closes the frame: the SP delta returns to zero.
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
epi := arm64ReturnEpilogueLen(fi)
spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0})
}
out = append(out, code...)
pc += len(code)
}
code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve, lits, pool, poolBase)
if err != nil {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
if flushAt >= 0 {
seg := pool.segs[flushAt]
out = appendARM64PoolSeg(out, seg)
pc += seg.guard + seg.size
segIdx++
}
for j := preCount; j < len(relocs); j++ {
// Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
// The RET's epilogue closes the frame: the SP delta returns to zero.
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
epi := arm64ReturnEpilogueLen(fi)
spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0})
}
out = append(out, code...)
pc += len(code)
}
if fi.needSplit {
block, blReloc := arm64MoreStackBlock(pc)
@@ -153,24 +143,185 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
relocs = append(relocs, blReloc)
pc += len(block)
}
// The pool itself: the UNDEF guard word when the function does not end
// in a branch (the morestack block's B counts as one), then the pooled
// constants in first-use order. The guard is the toolchain's word-zero
// UNDEF, not the BRK the UNDEF statement spells: it only has to be a
// faulting word nothing jumps to.
if pool.size > 0 {
if poolGuard > 0 {
out = append(out, a64wordLE(0)...)
pc += 4
}
for _, e := range pool.order {
out = append(out, e.data...)
pc += len(e.data)
}
// The plan's closing segment, the toolchain's end-of-function flush: the
// function's last Prog branches away (the morestack block's B when the
// function splits, the RET otherwise), so the words follow bare, and a
// body that would fall through gets the UNDEF word between.
if n := len(pool.segs); n > 0 && pool.segs[n-1].after == len(t.Body) {
seg := pool.segs[n-1]
out = appendARM64PoolSeg(out, seg)
pc += seg.guard + seg.size
}
if pc != layout.total {
return nil, nil, nil, nil, nil, nil, fmt.Errorf("internal: layout diverged from the literal-pool plan (%d bytes, planned %d)", pc, layout.total)
}
return out, offsets, relocs, lines, spadj, lits.list(), nil
}
// arm64PoolLayout carries what the encode pass needs from the plan: the
// whole function image's length, to hold the encoder to the plan, and the
// position the trailing morestack block lands at for a splitting function,
// which the guard prefix's conditional branches target.
type arm64PoolLayout struct {
total int
blockStart int
}
// a64MaxPCDisp is the toolchain's conservative bound on a PC-relative
// literal displacement (asm7.go's maxPCDisp): a load literal reaches ±1 MiB,
// and the flush points sit at half that, so the span-dependent branch
// enlargements of the later passes cannot push a reference out of reach.
const a64MaxPCDisp = 512 * 1024
// a64IsPCDisp ports the toolchain's ispcdisp.
func a64IsPCDisp(v int) bool {
return -a64MaxPCDisp < v && v < a64MaxPCDisp && v&3 == 0
}
// arm64EndsBlock reports whether a mnemonic hands control away
// unconditionally, the toolchain's flushpool exemption (AB, ARET and AERET):
// a pool drained after one needs no guard word, execution cannot fall into
// it.
func arm64EndsBlock(mnem string) bool {
switch mnem {
case "RET", "B", "JMP", "ERET":
return true
}
return false
}
// arm64PlanPool walks the body once, planning the layout and the literal
// pool's flush points together. The label offsets are the positions the
// encode pass lays down, flush bytes included; the pool references are
// harvested per statement by a probe encoding, whose band decisions hang on
// the operands alone and never on the positions, so the flush condition can
// run exactly as the toolchain's checkpool does:
//
// - the segment's accounting size has reached 0xffff0, or
// - the segment's far side has left the conservative displacement bound
// from the statement that would reference it, or
// - the statement is the function's last.
//
// A flush drains the open segment after the triggering statement: bare
// after a statement that branches away unconditionally, behind the UNDEF
// word at the function's end, behind a branch over the words inside the
// body. Every drained word carries the triggering statement's source line,
// the toolchain's own choice, so the pc-line tables see no deltas across
// the words.
func arm64PlanPool(t *ast.Text, fi arm64FrameInfo, guardLen, prologueLen int, pool *arm64Pool, offsets map[string]int, resolve func(string) string) arm64PoolLayout {
pool.probe = true
pos := guardLen + prologueLen
// The function's last real statement: END closes the body without
// becoming one, skipped the way a trailing label is, and this is the
// statement the toolchain's p.Link == nil check lands on.
last := -1
for i, v := range slices.Backward(t.Body) {
in, ok := v.(*ast.Instr)
if !ok || strings.ToUpper(in.Mnemonic.Text) == "END" {
continue
}
last = i
break
}
for i, stmt := range t.Body {
var in *ast.Instr
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
in = s
}
if in == nil {
continue
}
mnem := strings.ToUpper(in.Mnemonic.Text)
pc := pos
opened := pool.activeSeg() // the segment before the probe
switch mnem {
case "PCALIGN":
pos += arm64PCAlignPad(pc, in)
case "BYTE":
pos += len(in.Operands)
default:
// The probe: the encoder's own band decisions name the pool
// references into the segment open here. Its errors carry
// nothing: a forward branch cannot resolve yet, and the reach
// check would run against a base that is not final.
var probeRelocs []Reloc
_, _ = encodeARM64Instr(in, pc, offsets, fi, &probeRelocs, resolve, &arm64Literals{}, pool, pool.wordsBase())
pos += arm64InstrSize(in, fi, pc)
}
seg := pool.activeSeg()
if seg == nil {
continue
}
if opened == nil {
// The statement that opened the segment, the toolchain's
// pool.start: the first reference decides the distances the
// flush condition measures.
seg.start = pc
}
v := pc + 4 + seg.acct - seg.start + 8
end := !fi.needSplit && i == last
if seg.acct >= 0xffff0 || !a64IsPCDisp(v) || end {
seg.base = pos
seg.line = in.Pos().Line
seg.after = i
switch {
case arm64EndsBlock(mnem):
case end:
seg.guard = 4 // the UNDEF word, execution must not fall in
default:
seg.guard, seg.branch = 4, true // a branch over the words
}
pool.open()
pos += seg.guard + seg.size
}
}
var layout arm64PoolLayout
// The morestack block trails the body whatever the pool holds, and the
// guard prefix's conditional branches target it.
if fi.needSplit {
layout.blockStart = pos
pos += arm64MoreStackBlockLen
}
// The closing segment, the toolchain's flush at the function's last
// Prog. For a splitting function that Prog is the morestack block's
// trailing branch, so the words follow the block bare.
if seg := pool.activeSeg(); seg != nil {
seg.after = len(t.Body)
seg.base = pos
if last >= 0 {
seg.line = t.Body[last].(*ast.Instr).Pos().Line
}
if !fi.needSplit && !arm64EndsBlock(strings.ToUpper(t.Body[last].(*ast.Instr).Mnemonic.Text)) {
seg.guard = 4
}
pos += seg.guard + seg.size
}
layout.total = pos
return layout
}
// appendARM64PoolSeg emits a drained pool segment: the guard word, the
// toolchain's word-zero UNDEF or the branch over the words, when one is
// planned, then the words in first-use order.
func appendARM64PoolSeg(out []byte, seg *arm64PoolSeg) []byte {
if seg.guard > 0 {
if seg.branch {
out = append(out, a64wordLE(a64Branch(0, int32((seg.guard+seg.size)>>2)))...)
} else {
// The UNDEF word: not the BRK the UNDEF statement spells, only a
// faulting word nothing jumps to.
out = append(out, a64wordLE(0)...)
}
}
for _, e := range seg.order {
out = append(out, e.data...)
}
return out
}
// arm64JumpChain precomputes jump-to-jump folding: a label whose first
// instruction is an unconditional local jump redirects its own jumpers to
// the ultimate target. The Go toolchain chases these chains before it
@@ -2325,7 +2476,9 @@ func arm64PoolAccess(mnem string, lt a64LSType, opc int, off int64, rn, reg, pc
}
entryOff, w := pool.add(off)
dist := (poolBase + entryOff - pc) >> 2
if dist < -(1<<18) || dist >= 1<<18 {
// The plan pass probes before the segments' bases are final, so its
// distances carry no meaning.
if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) {
return nil, fmt.Errorf("%s: literal pool %d out of 19-bit reach", mnem, dist<<2)
}
return a64WordsLE(
@@ -2591,7 +2744,7 @@ func encodeARM64ConRn(rn int, con int64, rd int, pool *arm64Pool, poolBase, pc i
}
entryOff, w := pool.add64(con)
dist := (poolBase + entryOff - pc) >> 2
if dist < -(1<<18) || dist >= 1<<18 {
if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) {
return nil, fmt.Errorf("MOVD $%d(R%d): literal pool %d out of 19-bit reach", con, rn, dist<<2)
}
return a64WordsLE(
@@ -4008,7 +4161,7 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, pc int, fi
}
entryOff, w := pool.add(off)
dist := (poolBase + entryOff - pc) >> 2
if dist < -(1<<18) || dist >= 1<<18 {
if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) {
return nil, fmt.Errorf("%s: literal pool %d out of 19-bit reach", mnem, dist<<2)
}
return a64WordsLE(
@@ -5378,30 +5531,6 @@ func moviLitName(mnem string, data []byte) string {
}
}
// arm64PoolPadLen returns the UNDEF word the pool guard needs: four bytes
// when the body's last instruction does not branch (the toolchain's
// flushpool inserts one so execution cannot fall through into the words),
// zero otherwise. END closes the body without becoming an instruction, so
// it is skipped the way a trailing label is; FUNCDATA and PCDATA stay real
// statements, exactly the Progs the toolchain's flushpool sees as the last
// one.
func arm64PoolPadLen(t *ast.Text) int {
for _, v := range slices.Backward(t.Body) {
in, ok := v.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "END":
continue
case "RET", "B", "JMP", "ERET":
return 0
}
return 4
}
return 4
}
// arm64Literals collects the read-only constants the VMOVS/VMOVD/VMOVQ
// loads refer to. Names follow the toolchain's $i32/$i64/$i128 spellings so
// equal constants deduplicate to one literal.
@@ -5433,15 +5562,36 @@ func (l *arm64Literals) list() []Arm64Literal { return l.order }
// arm64Pool collects the out-of-range load/store offsets a function pools.
// The toolchain appends them after the last instruction (asm7.go addpool and
// flushpool) and reaches them with PC-relative literal loads into REGTMP.
// Entries deduplicate by value alone, whatever width the first referrer
// selected, and concatenate in first-use order with no alignment padding:
// the toolchain's roundUp touches its size accounting alone, never the byte
// stream.
// flushpool) and reaches them with PC-relative literal loads into REGTMP, and
// when a reference would leave the displacement bound it drains the pool
// mid-function behind a branch. The pool therefore holds segments: each
// holds the entries drained together, an entry resolves against the segment
// open at its referrer. Entries deduplicate by value alone, whatever width
// the first referrer selected, and concatenate in first-use order with no
// alignment padding: the toolchain's roundUp touches its size accounting
// alone, never the byte stream.
type arm64Pool struct {
order []arm64PoolEntry
seen map[int64]int // pooled value → entry index
size int // bytes the pool occupies so far
segs []*arm64PoolSeg // the drained segments in layout order, the open one last
active int // the segment the current statement's references resolve against
probe bool // the plan pass probes: harvest the requests, suppress the reach check
}
// arm64PoolSeg is one drained pool segment: the words with their offsets
// from the segment start and their literal-load widths (0 = LDR W
// zero-extended, 1 = LDR X for an 8-byte entry), the accounting size the
// flush condition measures, the image position and the guard word, and the
// body index the flush follows.
type arm64PoolSeg struct {
order []arm64PoolEntry
seen map[int64]int // pooled value → entry index
size int // bytes the words occupy
acct int // the toolchain's size accounting: an 8-byte entry rounds the total up to 8 first, the byte stream never
start int // the pc of the statement that opened the segment
base int // image offset of the segment's first byte
guard int // bytes of the guard word before the words (0 or 4)
branch bool // the guard word is a branch over the words, not the UNDEF
line int // the source line the words carry, the flushing statement's
after int // the body index the flush follows
}
// arm64PoolEntry is one pooled constant: its bytes, its offset from the pool
@@ -5456,12 +5606,39 @@ type arm64PoolEntry struct {
w uint32
}
// add interns a pooled load/store offset and returns its offset from the
// pool start and the literal-load width (asm7.go addpool): a value inside
// [0, 0x7FFFFFFF] takes a four-byte word loaded zero-extended, anything
// else the eight-byte slot a full LDR X reads.
// open starts a fresh segment and makes it the active one. The fresh
// segment's after is unassigned: a flush and the closing segment set it, and
// a segment left open but empty carries no words and no flush.
func (p *arm64Pool) open() {
p.segs = append(p.segs, &arm64PoolSeg{after: -1})
p.active = len(p.segs) - 1
}
// activeSeg returns the segment open now, nil when it holds no entries: the
// toolchain's checkpool runs only while the pool is open, blitrl non-nil.
func (p *arm64Pool) activeSeg() *arm64PoolSeg {
if p.active < len(p.segs) && len(p.segs[p.active].order) > 0 {
return p.segs[p.active]
}
return nil
}
// wordsBase returns the image offset of the active segment's first word, the
// base the PC-relative literal loads resolve against.
func (p *arm64Pool) wordsBase() int {
if p.active >= len(p.segs) {
return 0
}
s := p.segs[p.active]
return s.base + s.guard
}
// add interns a pooled load/store offset in the active segment and returns
// its offset from the segment start and the literal-load width (asm7.go
// addpool): a value inside [0, 0x7FFFFFFF] takes a four-byte word loaded
// zero-extended, anything else the eight-byte slot a full LDR X reads.
func (p *arm64Pool) add(v int64) (int, uint32) {
return p.addEntry(v, false)
return p.seg().addEntry(v, false)
}
// add64 interns a pooled displacement of the MOVD $con(R) lowering (asm7.go
@@ -5469,34 +5646,48 @@ func (p *arm64Pool) add(v int64) (int, uint32) {
// word, but an existing entry of the same value is shared as it stands, the
// toolchain's value-only dedup.
func (p *arm64Pool) add64(v int64) (int, uint32) {
return p.addEntry(v, true)
return p.seg().addEntry(v, true)
}
// addEntry creates or reuses the pool entry for v. Reuse is by value alone;
// at creation, lacon forces the eight-byte slot and every other requestor
// takes it only for a value no 32-bit load can carry: omovlit's ADWORD rule
// `lit != int32(lit) || uint64(lit) != uint32(lit)`.
func (p *arm64Pool) addEntry(v int64, lacon bool) (int, uint32) {
if i, ok := p.seen[v]; ok {
return p.order[i].off, p.order[i].w
// seg returns the active segment, opening one on demand: the first reference
// of a function opens the first segment.
func (p *arm64Pool) seg() *arm64PoolSeg {
if p.active >= len(p.segs) {
p.open()
}
off := p.size
return p.segs[p.active]
}
// addEntry creates or reuses the segment's entry for v. Reuse is by value
// alone; at creation, lacon forces the eight-byte slot and every other
// requestor takes it only for a value no 32-bit load can carry: omovlit's
// ADWORD rule `lit != int32(lit) || uint64(lit) != uint32(lit)`.
func (s *arm64PoolSeg) addEntry(v int64, lacon bool) (int, uint32) {
if i, ok := s.seen[v]; ok {
return s.order[i].off, s.order[i].w
}
off := s.size
var data []byte
var w uint32
if lacon || v < 0 || v > 0x7FFFFFFF {
// The toolchain's roundUp before a DWORD: the accounting rounds the
// total up to eight, the byte stream stays unpadded.
s.acct = (s.acct + 7) &^ 7
w = 1 // LDR X
data = a64WordsLE(uint32(v), uint32(v>>32))
p.size = off + 8
s.size = off + 8
s.acct += 8
} else {
w = 0 // LDR W, zero-extended
data = a64wordLE(uint32(v))
p.size = off + 4
s.size = off + 4
s.acct += 4
}
if p.seen == nil {
p.seen = map[int64]int{}
if s.seen == nil {
s.seen = map[int64]int{}
}
p.seen[v] = len(p.order)
p.order = append(p.order, arm64PoolEntry{data: data, off: off, w: w})
s.seen[v] = len(s.order)
s.order = append(s.order, arm64PoolEntry{data: data, off: off, w: w})
return off, w
}