diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index beea5d0..efa408e 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -53,99 +53,89 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ spadj = append(spadj, SpadjStep{PC: guardLen + arm64PrologueSpadjPC(fi), Value: fi.autosize}) } - // Pass 1: label offsets from the instruction sizes. + // Pass 1 plans the whole layout in one walk, the way the toolchain's + // span7 runs its own single linear pass: the label offsets come out of + // the same positions the encoder lays down, and the literal pool's + // references are harvested per statement with a probe encoding, so + // checkpool's flush condition is evaluated exactly as the toolchain's + // is and a reference that would leave the load-literal displacement + // bound drains the pool inside the body, at the point the toolchain + // would drain it. offsets := map[string]int{} - pos := guardLen + len(prologue) - for _, stmt := range t.Body { - switch s := stmt.(type) { - case *ast.Label: - offsets[s.Name.Text] = pos - case *ast.Instr: - if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" { - pos += arm64PCAlignPad(pos, s) - } else { - pos += arm64InstrSize(s, fi, pos) - } - } - } + layout := arm64PlanPool(t, fi, guardLen, len(prologue), pool, offsets, resolve) + pool.probe = false // Pass 2: encode. The guard prefix precedes the prologue; its branches // target the morestack block at the end of the function, whose position - // the first pass has settled. - bodyLen := 0 - { - p := guardLen + len(prologue) - for _, stmt := range t.Body { - if in, ok := stmt.(*ast.Instr); ok { - p += arm64InstrSize(in, fi, p) - } - } - bodyLen = p - (guardLen + len(prologue)) - } + // the plan has settled. var out []byte if fi.needSplit { - out = append(out, arm64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...) + out = append(out, arm64GuardBytes(fi, layout.blockStart)...) } out = append(out, prologue...) pc := guardLen + len(prologue) - // The offset literal pool lands after the last instruction (and after - // the morestack block, whose trailing branch closes the function); a - // function whose last instruction does not branch gets an UNDEF first, - // the toolchain's flushpool guard against falling through into the - // words. The base decides the PC-relative distances the pool loads - // encode, so it is fixed before pass 2. - poolGuard := arm64PoolPadLen(t) - poolBase := guardLen + len(prologue) + bodyLen + poolGuard - if fi.needSplit { - // The morestack block's B back to the entry is the toolchain's last - // Prog, an unconditional branch: the pool follows the block itself, - // with no guard before it. - poolGuard = 0 - poolBase += arm64MoreStackBlockLen - } preCount := len(relocs) var lines []LineEntry - for _, stmt := range t.Body { + // The drained segments replay in the plan's own order: the segment open + // at a statement holds its references, and the flush event after it + // emits the guard word and the words the segment drained. + segIdx := 0 + for i, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } - if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" { + flushAt := -1 + if len(pool.segs) > 0 { + for segIdx < len(pool.segs)-1 && pool.segs[segIdx].after < i { + segIdx++ + } + pool.active = segIdx + if pool.segs[segIdx].after == i { + flushAt = segIdx + } + } + switch strings.ToUpper(in.Mnemonic.Text) { + case "PCALIGN": pad := arm64PCAlignPad(pc, in) - for i := 0; i < pad/4; i++ { + for j := 0; j < pad/4; j++ { out = append(out, a64wordLE(a64NOP)...) pc += 4 } - continue - } - if strings.ToUpper(in.Mnemonic.Text) == "BYTE" { + case "BYTE": for _, op := range in.Operands { out = append(out, byte(arm64Imm64(op))) pc++ } - continue + default: + code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve, lits, pool, pool.wordsBase()) + if err != nil { + return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) + } + for j := preCount; j < len(relocs); j++ { + // Make the relocation offsets function-relative: each instruction + // records its reloc offset relative to its own start, and pc is + // that instruction's offset from the function start (prologue + // included). After shifts by the same amount. + relocs[j].Off += pc + relocs[j].After += pc + } + preCount = len(relocs) + lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) + // The RET's epilogue closes the frame: the SP delta returns to zero. + if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 { + epi := arm64ReturnEpilogueLen(fi) + spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0}) + } + out = append(out, code...) + pc += len(code) } - code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve, lits, pool, poolBase) - if err != nil { - return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) + if flushAt >= 0 { + seg := pool.segs[flushAt] + out = appendARM64PoolSeg(out, seg) + pc += seg.guard + seg.size + segIdx++ } - for j := preCount; j < len(relocs); j++ { - // Make the relocation offsets function-relative: each instruction - // records its reloc offset relative to its own start, and pc is - // that instruction's offset from the function start (prologue - // included). After shifts by the same amount. - relocs[j].Off += pc - relocs[j].After += pc - } - preCount = len(relocs) - lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) - // The RET's epilogue closes the frame: the SP delta returns to zero. - if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 { - epi := arm64ReturnEpilogueLen(fi) - spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0}) - } - out = append(out, code...) - pc += len(code) } if fi.needSplit { block, blReloc := arm64MoreStackBlock(pc) @@ -153,24 +143,185 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ relocs = append(relocs, blReloc) pc += len(block) } - // The pool itself: the UNDEF guard word when the function does not end - // in a branch (the morestack block's B counts as one), then the pooled - // constants in first-use order. The guard is the toolchain's word-zero - // UNDEF, not the BRK the UNDEF statement spells: it only has to be a - // faulting word nothing jumps to. - if pool.size > 0 { - if poolGuard > 0 { - out = append(out, a64wordLE(0)...) - pc += 4 - } - for _, e := range pool.order { - out = append(out, e.data...) - pc += len(e.data) - } + // The plan's closing segment, the toolchain's end-of-function flush: the + // function's last Prog branches away (the morestack block's B when the + // function splits, the RET otherwise), so the words follow bare, and a + // body that would fall through gets the UNDEF word between. + if n := len(pool.segs); n > 0 && pool.segs[n-1].after == len(t.Body) { + seg := pool.segs[n-1] + out = appendARM64PoolSeg(out, seg) + pc += seg.guard + seg.size + } + if pc != layout.total { + return nil, nil, nil, nil, nil, nil, fmt.Errorf("internal: layout diverged from the literal-pool plan (%d bytes, planned %d)", pc, layout.total) } return out, offsets, relocs, lines, spadj, lits.list(), nil } +// arm64PoolLayout carries what the encode pass needs from the plan: the +// whole function image's length, to hold the encoder to the plan, and the +// position the trailing morestack block lands at for a splitting function, +// which the guard prefix's conditional branches target. +type arm64PoolLayout struct { + total int + blockStart int +} + +// a64MaxPCDisp is the toolchain's conservative bound on a PC-relative +// literal displacement (asm7.go's maxPCDisp): a load literal reaches ±1 MiB, +// and the flush points sit at half that, so the span-dependent branch +// enlargements of the later passes cannot push a reference out of reach. +const a64MaxPCDisp = 512 * 1024 + +// a64IsPCDisp ports the toolchain's ispcdisp. +func a64IsPCDisp(v int) bool { + return -a64MaxPCDisp < v && v < a64MaxPCDisp && v&3 == 0 +} + +// arm64EndsBlock reports whether a mnemonic hands control away +// unconditionally, the toolchain's flushpool exemption (AB, ARET and AERET): +// a pool drained after one needs no guard word, execution cannot fall into +// it. +func arm64EndsBlock(mnem string) bool { + switch mnem { + case "RET", "B", "JMP", "ERET": + return true + } + return false +} + +// arm64PlanPool walks the body once, planning the layout and the literal +// pool's flush points together. The label offsets are the positions the +// encode pass lays down, flush bytes included; the pool references are +// harvested per statement by a probe encoding, whose band decisions hang on +// the operands alone and never on the positions, so the flush condition can +// run exactly as the toolchain's checkpool does: +// +// - the segment's accounting size has reached 0xffff0, or +// - the segment's far side has left the conservative displacement bound +// from the statement that would reference it, or +// - the statement is the function's last. +// +// A flush drains the open segment after the triggering statement: bare +// after a statement that branches away unconditionally, behind the UNDEF +// word at the function's end, behind a branch over the words inside the +// body. Every drained word carries the triggering statement's source line, +// the toolchain's own choice, so the pc-line tables see no deltas across +// the words. +func arm64PlanPool(t *ast.Text, fi arm64FrameInfo, guardLen, prologueLen int, pool *arm64Pool, offsets map[string]int, resolve func(string) string) arm64PoolLayout { + pool.probe = true + pos := guardLen + prologueLen + // The function's last real statement: END closes the body without + // becoming one, skipped the way a trailing label is, and this is the + // statement the toolchain's p.Link == nil check lands on. + last := -1 + for i, v := range slices.Backward(t.Body) { + in, ok := v.(*ast.Instr) + if !ok || strings.ToUpper(in.Mnemonic.Text) == "END" { + continue + } + last = i + break + } + for i, stmt := range t.Body { + var in *ast.Instr + switch s := stmt.(type) { + case *ast.Label: + offsets[s.Name.Text] = pos + case *ast.Instr: + in = s + } + if in == nil { + continue + } + mnem := strings.ToUpper(in.Mnemonic.Text) + pc := pos + opened := pool.activeSeg() // the segment before the probe + switch mnem { + case "PCALIGN": + pos += arm64PCAlignPad(pc, in) + case "BYTE": + pos += len(in.Operands) + default: + // The probe: the encoder's own band decisions name the pool + // references into the segment open here. Its errors carry + // nothing: a forward branch cannot resolve yet, and the reach + // check would run against a base that is not final. + var probeRelocs []Reloc + _, _ = encodeARM64Instr(in, pc, offsets, fi, &probeRelocs, resolve, &arm64Literals{}, pool, pool.wordsBase()) + pos += arm64InstrSize(in, fi, pc) + } + seg := pool.activeSeg() + if seg == nil { + continue + } + if opened == nil { + // The statement that opened the segment, the toolchain's + // pool.start: the first reference decides the distances the + // flush condition measures. + seg.start = pc + } + v := pc + 4 + seg.acct - seg.start + 8 + end := !fi.needSplit && i == last + if seg.acct >= 0xffff0 || !a64IsPCDisp(v) || end { + seg.base = pos + seg.line = in.Pos().Line + seg.after = i + switch { + case arm64EndsBlock(mnem): + case end: + seg.guard = 4 // the UNDEF word, execution must not fall in + default: + seg.guard, seg.branch = 4, true // a branch over the words + } + pool.open() + pos += seg.guard + seg.size + } + } + var layout arm64PoolLayout + // The morestack block trails the body whatever the pool holds, and the + // guard prefix's conditional branches target it. + if fi.needSplit { + layout.blockStart = pos + pos += arm64MoreStackBlockLen + } + // The closing segment, the toolchain's flush at the function's last + // Prog. For a splitting function that Prog is the morestack block's + // trailing branch, so the words follow the block bare. + if seg := pool.activeSeg(); seg != nil { + seg.after = len(t.Body) + seg.base = pos + if last >= 0 { + seg.line = t.Body[last].(*ast.Instr).Pos().Line + } + if !fi.needSplit && !arm64EndsBlock(strings.ToUpper(t.Body[last].(*ast.Instr).Mnemonic.Text)) { + seg.guard = 4 + } + pos += seg.guard + seg.size + } + layout.total = pos + return layout +} + +// appendARM64PoolSeg emits a drained pool segment: the guard word, the +// toolchain's word-zero UNDEF or the branch over the words, when one is +// planned, then the words in first-use order. +func appendARM64PoolSeg(out []byte, seg *arm64PoolSeg) []byte { + if seg.guard > 0 { + if seg.branch { + out = append(out, a64wordLE(a64Branch(0, int32((seg.guard+seg.size)>>2)))...) + } else { + // The UNDEF word: not the BRK the UNDEF statement spells, only a + // faulting word nothing jumps to. + out = append(out, a64wordLE(0)...) + } + } + for _, e := range seg.order { + out = append(out, e.data...) + } + return out +} + // arm64JumpChain precomputes jump-to-jump folding: a label whose first // instruction is an unconditional local jump redirects its own jumpers to // the ultimate target. The Go toolchain chases these chains before it @@ -2325,7 +2476,9 @@ func arm64PoolAccess(mnem string, lt a64LSType, opc int, off int64, rn, reg, pc } entryOff, w := pool.add(off) dist := (poolBase + entryOff - pc) >> 2 - if dist < -(1<<18) || dist >= 1<<18 { + // The plan pass probes before the segments' bases are final, so its + // distances carry no meaning. + if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) { return nil, fmt.Errorf("%s: literal pool %d out of 19-bit reach", mnem, dist<<2) } return a64WordsLE( @@ -2591,7 +2744,7 @@ func encodeARM64ConRn(rn int, con int64, rd int, pool *arm64Pool, poolBase, pc i } entryOff, w := pool.add64(con) dist := (poolBase + entryOff - pc) >> 2 - if dist < -(1<<18) || dist >= 1<<18 { + if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) { return nil, fmt.Errorf("MOVD $%d(R%d): literal pool %d out of 19-bit reach", con, rn, dist<<2) } return a64WordsLE( @@ -4008,7 +4161,7 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, pc int, fi } entryOff, w := pool.add(off) dist := (poolBase + entryOff - pc) >> 2 - if dist < -(1<<18) || dist >= 1<<18 { + if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) { return nil, fmt.Errorf("%s: literal pool %d out of 19-bit reach", mnem, dist<<2) } return a64WordsLE( @@ -5378,30 +5531,6 @@ func moviLitName(mnem string, data []byte) string { } } -// arm64PoolPadLen returns the UNDEF word the pool guard needs: four bytes -// when the body's last instruction does not branch (the toolchain's -// flushpool inserts one so execution cannot fall through into the words), -// zero otherwise. END closes the body without becoming an instruction, so -// it is skipped the way a trailing label is; FUNCDATA and PCDATA stay real -// statements, exactly the Progs the toolchain's flushpool sees as the last -// one. -func arm64PoolPadLen(t *ast.Text) int { - for _, v := range slices.Backward(t.Body) { - in, ok := v.(*ast.Instr) - if !ok { - continue - } - switch strings.ToUpper(in.Mnemonic.Text) { - case "END": - continue - case "RET", "B", "JMP", "ERET": - return 0 - } - return 4 - } - return 4 -} - // arm64Literals collects the read-only constants the VMOVS/VMOVD/VMOVQ // loads refer to. Names follow the toolchain's $i32/$i64/$i128 spellings so // equal constants deduplicate to one literal. @@ -5433,15 +5562,36 @@ func (l *arm64Literals) list() []Arm64Literal { return l.order } // arm64Pool collects the out-of-range load/store offsets a function pools. // The toolchain appends them after the last instruction (asm7.go addpool and -// flushpool) and reaches them with PC-relative literal loads into REGTMP. -// Entries deduplicate by value alone, whatever width the first referrer -// selected, and concatenate in first-use order with no alignment padding: -// the toolchain's roundUp touches its size accounting alone, never the byte -// stream. +// flushpool) and reaches them with PC-relative literal loads into REGTMP, and +// when a reference would leave the displacement bound it drains the pool +// mid-function behind a branch. The pool therefore holds segments: each +// holds the entries drained together, an entry resolves against the segment +// open at its referrer. Entries deduplicate by value alone, whatever width +// the first referrer selected, and concatenate in first-use order with no +// alignment padding: the toolchain's roundUp touches its size accounting +// alone, never the byte stream. type arm64Pool struct { - order []arm64PoolEntry - seen map[int64]int // pooled value → entry index - size int // bytes the pool occupies so far + segs []*arm64PoolSeg // the drained segments in layout order, the open one last + active int // the segment the current statement's references resolve against + probe bool // the plan pass probes: harvest the requests, suppress the reach check +} + +// arm64PoolSeg is one drained pool segment: the words with their offsets +// from the segment start and their literal-load widths (0 = LDR W +// zero-extended, 1 = LDR X for an 8-byte entry), the accounting size the +// flush condition measures, the image position and the guard word, and the +// body index the flush follows. +type arm64PoolSeg struct { + order []arm64PoolEntry + seen map[int64]int // pooled value → entry index + size int // bytes the words occupy + acct int // the toolchain's size accounting: an 8-byte entry rounds the total up to 8 first, the byte stream never + start int // the pc of the statement that opened the segment + base int // image offset of the segment's first byte + guard int // bytes of the guard word before the words (0 or 4) + branch bool // the guard word is a branch over the words, not the UNDEF + line int // the source line the words carry, the flushing statement's + after int // the body index the flush follows } // arm64PoolEntry is one pooled constant: its bytes, its offset from the pool @@ -5456,12 +5606,39 @@ type arm64PoolEntry struct { w uint32 } -// add interns a pooled load/store offset and returns its offset from the -// pool start and the literal-load width (asm7.go addpool): a value inside -// [0, 0x7FFFFFFF] takes a four-byte word loaded zero-extended, anything -// else the eight-byte slot a full LDR X reads. +// open starts a fresh segment and makes it the active one. The fresh +// segment's after is unassigned: a flush and the closing segment set it, and +// a segment left open but empty carries no words and no flush. +func (p *arm64Pool) open() { + p.segs = append(p.segs, &arm64PoolSeg{after: -1}) + p.active = len(p.segs) - 1 +} + +// activeSeg returns the segment open now, nil when it holds no entries: the +// toolchain's checkpool runs only while the pool is open, blitrl non-nil. +func (p *arm64Pool) activeSeg() *arm64PoolSeg { + if p.active < len(p.segs) && len(p.segs[p.active].order) > 0 { + return p.segs[p.active] + } + return nil +} + +// wordsBase returns the image offset of the active segment's first word, the +// base the PC-relative literal loads resolve against. +func (p *arm64Pool) wordsBase() int { + if p.active >= len(p.segs) { + return 0 + } + s := p.segs[p.active] + return s.base + s.guard +} + +// add interns a pooled load/store offset in the active segment and returns +// its offset from the segment start and the literal-load width (asm7.go +// addpool): a value inside [0, 0x7FFFFFFF] takes a four-byte word loaded +// zero-extended, anything else the eight-byte slot a full LDR X reads. func (p *arm64Pool) add(v int64) (int, uint32) { - return p.addEntry(v, false) + return p.seg().addEntry(v, false) } // add64 interns a pooled displacement of the MOVD $con(R) lowering (asm7.go @@ -5469,34 +5646,48 @@ func (p *arm64Pool) add(v int64) (int, uint32) { // word, but an existing entry of the same value is shared as it stands, the // toolchain's value-only dedup. func (p *arm64Pool) add64(v int64) (int, uint32) { - return p.addEntry(v, true) + return p.seg().addEntry(v, true) } -// addEntry creates or reuses the pool entry for v. Reuse is by value alone; -// at creation, lacon forces the eight-byte slot and every other requestor -// takes it only for a value no 32-bit load can carry: omovlit's ADWORD rule -// `lit != int32(lit) || uint64(lit) != uint32(lit)`. -func (p *arm64Pool) addEntry(v int64, lacon bool) (int, uint32) { - if i, ok := p.seen[v]; ok { - return p.order[i].off, p.order[i].w +// seg returns the active segment, opening one on demand: the first reference +// of a function opens the first segment. +func (p *arm64Pool) seg() *arm64PoolSeg { + if p.active >= len(p.segs) { + p.open() } - off := p.size + return p.segs[p.active] +} + +// addEntry creates or reuses the segment's entry for v. Reuse is by value +// alone; at creation, lacon forces the eight-byte slot and every other +// requestor takes it only for a value no 32-bit load can carry: omovlit's +// ADWORD rule `lit != int32(lit) || uint64(lit) != uint32(lit)`. +func (s *arm64PoolSeg) addEntry(v int64, lacon bool) (int, uint32) { + if i, ok := s.seen[v]; ok { + return s.order[i].off, s.order[i].w + } + off := s.size var data []byte var w uint32 if lacon || v < 0 || v > 0x7FFFFFFF { + // The toolchain's roundUp before a DWORD: the accounting rounds the + // total up to eight, the byte stream stays unpadded. + s.acct = (s.acct + 7) &^ 7 w = 1 // LDR X data = a64WordsLE(uint32(v), uint32(v>>32)) - p.size = off + 8 + s.size = off + 8 + s.acct += 8 } else { w = 0 // LDR W, zero-extended data = a64wordLE(uint32(v)) - p.size = off + 4 + s.size = off + 4 + s.acct += 4 } - if p.seen == nil { - p.seen = map[int64]int{} + if s.seen == nil { + s.seen = map[int64]int{} } - p.seen[v] = len(p.order) - p.order = append(p.order, arm64PoolEntry{data: data, off: off, w: w}) + s.seen[v] = len(s.order) + s.order = append(s.order, arm64PoolEntry{data: data, off: off, w: w}) return off, w }