fix(asm): place the arm64 literal pool the way the toolchain flushes it

Assisted-by: GLM 5.3
This commit is contained in:
petrbalvin committed 2026-10-07 13:49:49 +02:00
1 parent 4fc96decc4
commit 405e2427ed
3 files changed
+215 -68

No files matched your search

+59 -46
View File
@@ -89,11 +89,20 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
out = append(out, prologue...)
pc := guardLen + len(prologue)
// The offset literal pool lands after the last instruction (and after
// the morestack block); a function whose last instruction does not
// branch gets an UNDEF first, the toolchain's flushpool guard against
// falling through into the words. The base decides the PC-relative
// distances the pool loads encode, so it is fixed before pass 2.
poolBase := guardLen + len(prologue) + bodyLen + arm64PoolPadLen(t)
// the morestack block, whose trailing branch closes the function); a
// function whose last instruction does not branch gets an UNDEF first,
// the toolchain's flushpool guard against falling through into the
// words. The base decides the PC-relative distances the pool loads
// encode, so it is fixed before pass 2.
poolGuard := arm64PoolPadLen(t)
poolBase := guardLen + len(prologue) + bodyLen + poolGuard
if fi.needSplit {
// The morestack block's B back to the entry is the toolchain's last
// Prog, an unconditional branch: the pool follows the block itself,
// with no guard before it.
poolGuard = 0
poolBase += arm64MoreStackBlockLen
}
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
@@ -144,12 +153,13 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [
relocs = append(relocs, blReloc)
pc += len(block)
}
// The pool itself: the UNDEF guard word when the body does not end in a
// branch, then the pooled constants in first-use order. The guard is
// the toolchain's word-zero UNDEF, not the BRK the UNDEF statement
// spells: it only has to be a faulting word nothing jumps to.
// The pool itself: the UNDEF guard word when the function does not end
// in a branch (the morestack block's B counts as one), then the pooled
// constants in first-use order. The guard is the toolchain's word-zero
// UNDEF, not the BRK the UNDEF statement spells: it only has to be a
// faulting word nothing jumps to.
if pool.size > 0 {
if arm64PoolPadLen(t) > 0 {
if poolGuard > 0 {
out = append(out, a64wordLE(0)...)
pc += 4
}
@@ -5211,7 +5221,10 @@ func moviLitName(mnem string, data []byte) string {
// arm64PoolPadLen returns the UNDEF word the pool guard needs: four bytes
// when the body's last instruction does not branch (the toolchain's
// flushpool inserts one so execution cannot fall through into the words),
// zero otherwise.
// zero otherwise. END closes the body without becoming an instruction, so
// it is skipped the way a trailing label is; FUNCDATA and PCDATA stay real
// statements, exactly the Progs the toolchain's flushpool sees as the last
// one.
func arm64PoolPadLen(t *ast.Text) int {
for _, v := range slices.Backward(t.Body) {
in, ok := v.(*ast.Instr)
@@ -5219,6 +5232,8 @@ func arm64PoolPadLen(t *ast.Text) int {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "END":
continue
case "RET", "B", "JMP", "ERET":
return 0
}
@@ -5258,9 +5273,11 @@ func (l *arm64Literals) list() []Arm64Literal { return l.order }
// arm64Pool collects the out-of-range load/store offsets a function pools.
// The toolchain appends them after the last instruction (asm7.go addpool and
// flushpool) and reaches them with PC-relative literal loads into REGTMP;
// equal values deduplicate to one entry regardless of which instruction
// pooled them first.
// flushpool) and reaches them with PC-relative literal loads into REGTMP.
// Entries deduplicate by value alone, whatever width the first referrer
// selected, and concatenate in first-use order with no alignment padding:
// the toolchain's roundUp touches its size accounting alone, never the byte
// stream.
type arm64Pool struct {
order []arm64PoolEntry
seen map[int64]int // pooled value → entry index
@@ -5268,37 +5285,50 @@ type arm64Pool struct {
}
// arm64PoolEntry is one pooled constant: its bytes, its offset from the pool
// start and the literal-load width the first referrer selected (0 = LDR W,
// 2 = LDRSW for a negative word, 1 = LDR X for an 8-byte entry).
// start and the literal-load width its bytes select (0 = LDR W zero-extended,
// 1 = LDR X for an 8-byte entry). omovlit reads the width off the entry
// itself, so every referrer of a value loads with the first referrer's
// width; negative values always take the 8-byte entry, which makes the
// sign-extended LDRSW load unreachable for this pool.
type arm64PoolEntry struct {
data []byte
off int
w uint32
}
// add interns a pooled value and returns its offset from the pool start and
// the literal-load width. A value beyond the 32-bit reach takes an
// eight-byte entry aligned to eight; a negative word takes the sign-extended
// load, the toolchain's omovlit choice for its AMOVD pool reference.
// add interns a pooled load/store offset and returns its offset from the
// pool start and the literal-load width (asm7.go addpool): a value inside
// [0, 0x7FFFFFFF] takes a four-byte word loaded zero-extended, anything
// else the eight-byte slot a full LDR X reads.
func (p *arm64Pool) add(v int64) (int, uint32) {
return p.addEntry(v, false)
}
// add64 interns a pooled displacement of the MOVD $con(R) lowering (asm7.go
// case 34): the entry takes the eight-byte slot even when the value fits a
// word, but an existing entry of the same value is shared as it stands, the
// toolchain's value-only dedup.
func (p *arm64Pool) add64(v int64) (int, uint32) {
return p.addEntry(v, true)
}
// addEntry creates or reuses the pool entry for v. Reuse is by value alone;
// at creation, lacon forces the eight-byte slot and every other requestor
// takes it only for a value no 32-bit load can carry: omovlit's ADWORD rule
// `lit != int32(lit) || uint64(lit) != uint32(lit)`.
func (p *arm64Pool) addEntry(v int64, lacon bool) (int, uint32) {
if i, ok := p.seen[v]; ok {
return p.order[i].off, p.order[i].w
}
wide := v != int64(int32(v)) || uint64(v) != uint64(uint32(v))
off := p.size
var data []byte
var w uint32
switch {
case wide:
if lacon || v < 0 || v > 0x7FFFFFFF {
w = 1 // LDR X
off = (p.size + 7) &^ 7
data = a64WordsLE(uint32(v), uint32(v>>32))
p.size = off + 8
case v < 0:
w = 2 // LDRSW, sign-extended to 64
data = a64wordLE(uint32(v))
p.size = off + 4
default:
} else {
w = 0 // LDR W, zero-extended
data = a64wordLE(uint32(v))
p.size = off + 4
}
@@ -5310,23 +5340,6 @@ func (p *arm64Pool) add(v int64) (int, uint32) {
return off, w
}
// add64 reserves an 8-byte slot for v loaded by a full LDR X: the lacon
// pool path always reads 64 bits, even when the value fits 32 (asm7.go case
// 34's omovlit(AMOVD)).
func (p *arm64Pool) add64(v int64) (int, uint32) {
if i, ok := p.seen[v]; ok && p.order[i].w == 1 {
return p.order[i].off, p.order[i].w
}
off := (p.size + 7) &^ 7
p.size = off + 8
if p.seen == nil {
p.seen = map[int64]int{}
}
p.seen[v] = len(p.order)
p.order = append(p.order, arm64PoolEntry{data: a64WordsLE(uint32(v), uint32(v>>32)), off: off, w: 1})
return off, 1
}
// AssembleFileARM64 assembles every TEXT function of a parsed arm64 file
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
// code. SB references in the code are encoded as ADRP pairs with zero
+152 -22
View File
@@ -2070,15 +2070,11 @@ func TestArm64ConRnRejections(t *testing.T) {
}
}
// TestArm64LogicalMaterialisationBranch pins a forward branch over the
// three-word logical materialisation against `go tool asm -S` (Go 1.27,
// arm64): the size pass must count the MOVZ/MOVK pair the encoder lays down,
// or the label offsets desynchronise from the bytes and the branch lands a
// word early.
func TestArm64LogicalMaterialisationBranch(t *testing.T) {
// The body closes with its own RET under the end label, so the file is
// parsed as written rather than through arm64Words' appended RET.
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tCBZ\tR2, end\n\tTST\t$0x4900000049, R0\nend:\tRET\n")
// arm64WordsTail assembles a NOSPLIT leaf body exactly as written, adding no
// RET: the pool guard tests need bodies whose last statement is not a branch.
func arm64WordsTail(t *testing.T, body string) []uint32 {
t.Helper()
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
@@ -2086,21 +2082,155 @@ func TestArm64LogicalMaterialisationBranch(t *testing.T) {
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
want := []uint32{
0xb4000082, // CBZ R2, +16 (four words ahead)
0xd280093b, // MOVZ $0x49, R27
0xf2c0093b, // MOVK $(0x49<<32), R27
0xea1b001f, // TST R27, R0
0xd65f03c0, // RET
return leWords(img.Code)
}
// TestArm64LiteralPool pins the offset literal pool against `go tool asm -S`
// output (Go 1.27, arm64): the PC-relative literal loads into REGTMP, the
// register-offset accesses, first-use ordering with value-only dedup, the
// entry widths (four-byte words for [0, 0x7FFFFFFF], eight-byte slots for
// the lacon displacements, the negatives and the beyond-32-bit values, with
// no alignment padding between entries) and the shared entries' widths
// following the entry rather than the referrer.
func TestArm64LiteralPool(t *testing.T) {
tests := []struct {
name string
body string
want []uint32
}{
{
name: "dedup and first-use order",
body: "\tMOVD\tR1, 0x1007000(R2)\n\tMOVD\tR1, 0x44332211(R2)\n\tMOVD\tR1, 0x1007000(R2)\n",
want: []uint32{
0x180000fb, 0xf83b6841, // LDR W27, pool0; MOVD R1, (R2)(R27)
0x180000db, 0xf83b6841, // LDR W27, pool1; MOVD R1, (R2)(R27)
0x1800007b, 0xf83b6841, // LDR W27, pool0; MOVD R1, (R2)(R27)
0xd65f03c0, // RET
0x01007000, 0x44332211, // WORD 0x1007000, WORD 0x44332211
},
},
{
name: "mixed widths, no padding, cross-width dedup",
body: "\tMOVB\tR1, 0x1000000(R2)\n\tMOVB\tR1, -0x1000000(R3)\n\tMOVB\tR1, 0x1001000(R4)\n\tMOVD\t$0x1000000(R7), R1\n",
want: []uint32{
0x1800013b, 0x383b6841, // LDR W27, pool0; MOVB R1, (R2)(R27)
0x5800011b, 0x383b6861, // LDR X27, pool1; MOVB R1, (R3)(R27)
0x1800011b, 0x383b6881, // LDR W27, pool2; MOVB R1, (R4)(R27)
0x1800007b, 0x8b3b60e1, // LDR W27, pool0 (lacon reuse); ADD R27.UXTX, R7, R1
0xd65f03c0, // RET
0x01000000, // WORD 0x1000000 (off 0)
0xff000000, 0xffffffff, // DWORD -0x1000000 (off 4, unpadded)
0x01001000, // WORD 0x1001000 (off 12)
},
},
{
name: "lacon entry takes the eight-byte slot",
body: "\tMOVD\t$0x1000000(R7), R1\n",
want: []uint32{
0x5800007b, 0x8b3b60e1, // LDR X27, pool; ADD R27.UXTX, R7, R1
0xd65f03c0, // RET
0x01000000, 0x00000000, // DWORD 0x1000000
},
},
{
name: "negative offsets pool as DWORD with LDR X",
body: "\tMOVB\tR1, -0x1000000(R2)\n\tMOVD\t$-0x1000000(R7), R1\n",
want: []uint32{
0x580000bb, 0x383b6841, // LDR X27, pool; MOVB R1, (R2)(R27)
0x5800007b, 0x8b3b60e1, // LDR X27, pool; ADD R27.UXTX, R7, R1
0xd65f03c0, // RET
0xff000000, 0xffffffff, // DWORD -0x1000000
},
},
{
name: "beyond 32-bit offsets",
body: "\tMOVD\tR1, 0x12345678901(R2)\n\tMOVB\tR2, 0x12345678901(R3)\n",
want: []uint32{
0x580000bb, 0xf83b6841, // LDR X27, pool; MOVD R1, (R2)(R27)
0x5800007b, 0x383b6862, // LDR X27, pool; MOVB R2, (R3)(R27)
0xd65f03c0, // RET
0x45678901, 0x00000123, // DWORD 0x12345678901
},
},
{
name: "pair offsets ride the pool",
body: "\tMOVD\tR1, 0x1000000(R2)\n\tLDP\t0x1000000(R2), (R1, R3)\n",
want: []uint32{
0x917ffc5b, 0xf9080361, // ADD $(4095<<12), R2, R27; MOVD R1, 64(R27)
0x1800009b, 0x8b3b605b, // LDR W27, pool; ADD R27.UXTX, R2, R27
0xa9400f61, // LDP (R27), (R1, R3)
0xd65f03c0, // RET
0x01000000, // WORD 0x1000000
},
},
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got := arm64Words(t, tt.body)
if len(got) != len(tt.want) {
t.Fatalf("word count = %d, want %d (got %08x)", len(got), len(tt.want), got)
}
for i := range tt.want {
if got[i] != tt.want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], tt.want[i])
}
}
})
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
// TestArm64LiteralPoolGuard pins the flushpool guard: the word-zero UNDEF
// that keeps execution from falling into the pool when the last statement is
// not a branch. END closes the body without becoming an instruction, so it
// does not count; a trailing PCDATA is a real statement and takes the guard;
// RET and the morestack block's branch need none.
func TestArm64LiteralPoolGuard(t *testing.T) {
tests := []struct {
name string
body string
want []uint32
}{
{
name: "END without RET still guards",
body: "\tMOVB\tR1, 0x1000000(R2)\n\tEND\n",
want: []uint32{
0x1800007b, 0x383b6841, // LDR W27, pool; MOVB R1, (R2)(R27)
0x00000000, // UNDEF guard
0x01000000, // WORD 0x1000000
},
},
{
name: "trailing PCDATA keeps the guard",
body: "\tMOVD\tR1, 0x1007000(R2)\n\tRET\n\tPCDATA\t$0, $-1\n",
want: []uint32{
0x1800009b, 0xf83b6841, // LDR W27, pool; MOVD R1, (R2)(R27)
0xd65f03c0, // RET
0x00000000, // UNDEF guard
0x01007000, // WORD 0x1007000
},
},
{
name: "RET closes without a guard",
body: "\tMOVD\tR1, 0x1007000(R2)\n\tRET\n\tEND\n",
want: []uint32{
0x1800007b, 0xf83b6841, // LDR W27, pool; MOVD R1, (R2)(R27)
0xd65f03c0, // RET
0x01007000, // WORD 0x1007000
},
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got := arm64WordsTail(t, tt.body)
if len(got) != len(tt.want) {
t.Fatalf("word count = %d, want %d (got %08x)", len(got), len(tt.want), got)
}
for i := range tt.want {
if got[i] != tt.want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], tt.want[i])
}
}
})
}
}
+4
View File
@@ -495,6 +495,10 @@ func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
return a64WordsLE(ws...)
}
// arm64MoreStackBlockLen is the byte length of arm64MoreStackBlock: the
// saved LR, the BL and the branch back, three words whatever the target.
const arm64MoreStackBlockLen = 12
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
// the R_CALLARM64 relocation.