diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index cd9c219..add36b5 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -89,11 +89,20 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ out = append(out, prologue...) pc := guardLen + len(prologue) // The offset literal pool lands after the last instruction (and after - // the morestack block); a function whose last instruction does not - // branch gets an UNDEF first, the toolchain's flushpool guard against - // falling through into the words. The base decides the PC-relative - // distances the pool loads encode, so it is fixed before pass 2. - poolBase := guardLen + len(prologue) + bodyLen + arm64PoolPadLen(t) + // the morestack block, whose trailing branch closes the function); a + // function whose last instruction does not branch gets an UNDEF first, + // the toolchain's flushpool guard against falling through into the + // words. The base decides the PC-relative distances the pool loads + // encode, so it is fixed before pass 2. + poolGuard := arm64PoolPadLen(t) + poolBase := guardLen + len(prologue) + bodyLen + poolGuard + if fi.needSplit { + // The morestack block's B back to the entry is the toolchain's last + // Prog, an unconditional branch: the pool follows the block itself, + // with no guard before it. + poolGuard = 0 + poolBase += arm64MoreStackBlockLen + } preCount := len(relocs) var lines []LineEntry for _, stmt := range t.Body { @@ -144,12 +153,13 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ relocs = append(relocs, blReloc) pc += len(block) } - // The pool itself: the UNDEF guard word when the body does not end in a - // branch, then the pooled constants in first-use order. The guard is - // the toolchain's word-zero UNDEF, not the BRK the UNDEF statement - // spells: it only has to be a faulting word nothing jumps to. + // The pool itself: the UNDEF guard word when the function does not end + // in a branch (the morestack block's B counts as one), then the pooled + // constants in first-use order. The guard is the toolchain's word-zero + // UNDEF, not the BRK the UNDEF statement spells: it only has to be a + // faulting word nothing jumps to. if pool.size > 0 { - if arm64PoolPadLen(t) > 0 { + if poolGuard > 0 { out = append(out, a64wordLE(0)...) pc += 4 } @@ -5211,7 +5221,10 @@ func moviLitName(mnem string, data []byte) string { // arm64PoolPadLen returns the UNDEF word the pool guard needs: four bytes // when the body's last instruction does not branch (the toolchain's // flushpool inserts one so execution cannot fall through into the words), -// zero otherwise. +// zero otherwise. END closes the body without becoming an instruction, so +// it is skipped the way a trailing label is; FUNCDATA and PCDATA stay real +// statements, exactly the Progs the toolchain's flushpool sees as the last +// one. func arm64PoolPadLen(t *ast.Text) int { for _, v := range slices.Backward(t.Body) { in, ok := v.(*ast.Instr) @@ -5219,6 +5232,8 @@ func arm64PoolPadLen(t *ast.Text) int { continue } switch strings.ToUpper(in.Mnemonic.Text) { + case "END": + continue case "RET", "B", "JMP", "ERET": return 0 } @@ -5258,9 +5273,11 @@ func (l *arm64Literals) list() []Arm64Literal { return l.order } // arm64Pool collects the out-of-range load/store offsets a function pools. // The toolchain appends them after the last instruction (asm7.go addpool and -// flushpool) and reaches them with PC-relative literal loads into REGTMP; -// equal values deduplicate to one entry regardless of which instruction -// pooled them first. +// flushpool) and reaches them with PC-relative literal loads into REGTMP. +// Entries deduplicate by value alone, whatever width the first referrer +// selected, and concatenate in first-use order with no alignment padding: +// the toolchain's roundUp touches its size accounting alone, never the byte +// stream. type arm64Pool struct { order []arm64PoolEntry seen map[int64]int // pooled value → entry index @@ -5268,37 +5285,50 @@ type arm64Pool struct { } // arm64PoolEntry is one pooled constant: its bytes, its offset from the pool -// start and the literal-load width the first referrer selected (0 = LDR W, -// 2 = LDRSW for a negative word, 1 = LDR X for an 8-byte entry). +// start and the literal-load width its bytes select (0 = LDR W zero-extended, +// 1 = LDR X for an 8-byte entry). omovlit reads the width off the entry +// itself, so every referrer of a value loads with the first referrer's +// width; negative values always take the 8-byte entry, which makes the +// sign-extended LDRSW load unreachable for this pool. type arm64PoolEntry struct { data []byte off int w uint32 } -// add interns a pooled value and returns its offset from the pool start and -// the literal-load width. A value beyond the 32-bit reach takes an -// eight-byte entry aligned to eight; a negative word takes the sign-extended -// load, the toolchain's omovlit choice for its AMOVD pool reference. +// add interns a pooled load/store offset and returns its offset from the +// pool start and the literal-load width (asm7.go addpool): a value inside +// [0, 0x7FFFFFFF] takes a four-byte word loaded zero-extended, anything +// else the eight-byte slot a full LDR X reads. func (p *arm64Pool) add(v int64) (int, uint32) { + return p.addEntry(v, false) +} + +// add64 interns a pooled displacement of the MOVD $con(R) lowering (asm7.go +// case 34): the entry takes the eight-byte slot even when the value fits a +// word, but an existing entry of the same value is shared as it stands, the +// toolchain's value-only dedup. +func (p *arm64Pool) add64(v int64) (int, uint32) { + return p.addEntry(v, true) +} + +// addEntry creates or reuses the pool entry for v. Reuse is by value alone; +// at creation, lacon forces the eight-byte slot and every other requestor +// takes it only for a value no 32-bit load can carry: omovlit's ADWORD rule +// `lit != int32(lit) || uint64(lit) != uint32(lit)`. +func (p *arm64Pool) addEntry(v int64, lacon bool) (int, uint32) { if i, ok := p.seen[v]; ok { return p.order[i].off, p.order[i].w } - wide := v != int64(int32(v)) || uint64(v) != uint64(uint32(v)) off := p.size var data []byte var w uint32 - switch { - case wide: + if lacon || v < 0 || v > 0x7FFFFFFF { w = 1 // LDR X - off = (p.size + 7) &^ 7 data = a64WordsLE(uint32(v), uint32(v>>32)) p.size = off + 8 - case v < 0: - w = 2 // LDRSW, sign-extended to 64 - data = a64wordLE(uint32(v)) - p.size = off + 4 - default: + } else { + w = 0 // LDR W, zero-extended data = a64wordLE(uint32(v)) p.size = off + 4 } @@ -5310,23 +5340,6 @@ func (p *arm64Pool) add(v int64) (int, uint32) { return off, w } -// add64 reserves an 8-byte slot for v loaded by a full LDR X: the lacon -// pool path always reads 64 bits, even when the value fits 32 (asm7.go case -// 34's omovlit(AMOVD)). -func (p *arm64Pool) add64(v int64) (int, uint32) { - if i, ok := p.seen[v]; ok && p.order[i].w == 1 { - return p.order[i].off, p.order[i].w - } - off := (p.size + 7) &^ 7 - p.size = off + 8 - if p.seen == nil { - p.seen = map[int64]int{} - } - p.seen[v] = len(p.order) - p.order = append(p.order, arm64PoolEntry{data: a64WordsLE(uint32(v), uint32(v>>32)), off: off, w: 1}) - return off, 1 -} - // AssembleFileARM64 assembles every TEXT function of a parsed arm64 file // and lays out its static symbols (GLOBL/DATA) in a data section behind the // code. SB references in the code are encoded as ADRP pairs with zero diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index a9d8919..86cdbb0 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -2070,15 +2070,11 @@ func TestArm64ConRnRejections(t *testing.T) { } } -// TestArm64LogicalMaterialisationBranch pins a forward branch over the -// three-word logical materialisation against `go tool asm -S` (Go 1.27, -// arm64): the size pass must count the MOVZ/MOVK pair the encoder lays down, -// or the label offsets desynchronise from the bytes and the branch lands a -// word early. -func TestArm64LogicalMaterialisationBranch(t *testing.T) { - // The body closes with its own RET under the end label, so the file is - // parsed as written rather than through arm64Words' appended RET. - f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tCBZ\tR2, end\n\tTST\t$0x4900000049, R0\nend:\tRET\n") +// arm64WordsTail assembles a NOSPLIT leaf body exactly as written, adding no +// RET: the pool guard tests need bodies whose last statement is not a branch. +func arm64WordsTail(t *testing.T, body string) []uint32 { + t.Helper() + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body) if len(errs) > 0 { t.Fatalf("parse: %v", errs) } @@ -2086,21 +2082,155 @@ func TestArm64LogicalMaterialisationBranch(t *testing.T) { if err != nil { t.Fatalf("AssembleFileARM64: %v", err) } - got := leWords(img.Code) - want := []uint32{ - 0xb4000082, // CBZ R2, +16 (four words ahead) - 0xd280093b, // MOVZ $0x49, R27 - 0xf2c0093b, // MOVK $(0x49<<32), R27 - 0xea1b001f, // TST R27, R0 - 0xd65f03c0, // RET + return leWords(img.Code) +} + +// TestArm64LiteralPool pins the offset literal pool against `go tool asm -S` +// output (Go 1.27, arm64): the PC-relative literal loads into REGTMP, the +// register-offset accesses, first-use ordering with value-only dedup, the +// entry widths (four-byte words for [0, 0x7FFFFFFF], eight-byte slots for +// the lacon displacements, the negatives and the beyond-32-bit values, with +// no alignment padding between entries) and the shared entries' widths +// following the entry rather than the referrer. +func TestArm64LiteralPool(t *testing.T) { + tests := []struct { + name string + body string + want []uint32 + }{ + { + name: "dedup and first-use order", + body: "\tMOVD\tR1, 0x1007000(R2)\n\tMOVD\tR1, 0x44332211(R2)\n\tMOVD\tR1, 0x1007000(R2)\n", + want: []uint32{ + 0x180000fb, 0xf83b6841, // LDR W27, pool0; MOVD R1, (R2)(R27) + 0x180000db, 0xf83b6841, // LDR W27, pool1; MOVD R1, (R2)(R27) + 0x1800007b, 0xf83b6841, // LDR W27, pool0; MOVD R1, (R2)(R27) + 0xd65f03c0, // RET + 0x01007000, 0x44332211, // WORD 0x1007000, WORD 0x44332211 + }, + }, + { + name: "mixed widths, no padding, cross-width dedup", + body: "\tMOVB\tR1, 0x1000000(R2)\n\tMOVB\tR1, -0x1000000(R3)\n\tMOVB\tR1, 0x1001000(R4)\n\tMOVD\t$0x1000000(R7), R1\n", + want: []uint32{ + 0x1800013b, 0x383b6841, // LDR W27, pool0; MOVB R1, (R2)(R27) + 0x5800011b, 0x383b6861, // LDR X27, pool1; MOVB R1, (R3)(R27) + 0x1800011b, 0x383b6881, // LDR W27, pool2; MOVB R1, (R4)(R27) + 0x1800007b, 0x8b3b60e1, // LDR W27, pool0 (lacon reuse); ADD R27.UXTX, R7, R1 + 0xd65f03c0, // RET + 0x01000000, // WORD 0x1000000 (off 0) + 0xff000000, 0xffffffff, // DWORD -0x1000000 (off 4, unpadded) + 0x01001000, // WORD 0x1001000 (off 12) + }, + }, + { + name: "lacon entry takes the eight-byte slot", + body: "\tMOVD\t$0x1000000(R7), R1\n", + want: []uint32{ + 0x5800007b, 0x8b3b60e1, // LDR X27, pool; ADD R27.UXTX, R7, R1 + 0xd65f03c0, // RET + 0x01000000, 0x00000000, // DWORD 0x1000000 + }, + }, + { + name: "negative offsets pool as DWORD with LDR X", + body: "\tMOVB\tR1, -0x1000000(R2)\n\tMOVD\t$-0x1000000(R7), R1\n", + want: []uint32{ + 0x580000bb, 0x383b6841, // LDR X27, pool; MOVB R1, (R2)(R27) + 0x5800007b, 0x8b3b60e1, // LDR X27, pool; ADD R27.UXTX, R7, R1 + 0xd65f03c0, // RET + 0xff000000, 0xffffffff, // DWORD -0x1000000 + }, + }, + { + name: "beyond 32-bit offsets", + body: "\tMOVD\tR1, 0x12345678901(R2)\n\tMOVB\tR2, 0x12345678901(R3)\n", + want: []uint32{ + 0x580000bb, 0xf83b6841, // LDR X27, pool; MOVD R1, (R2)(R27) + 0x5800007b, 0x383b6862, // LDR X27, pool; MOVB R2, (R3)(R27) + 0xd65f03c0, // RET + 0x45678901, 0x00000123, // DWORD 0x12345678901 + }, + }, + { + name: "pair offsets ride the pool", + body: "\tMOVD\tR1, 0x1000000(R2)\n\tLDP\t0x1000000(R2), (R1, R3)\n", + want: []uint32{ + 0x917ffc5b, 0xf9080361, // ADD $(4095<<12), R2, R27; MOVD R1, 64(R27) + 0x1800009b, 0x8b3b605b, // LDR W27, pool; ADD R27.UXTX, R2, R27 + 0xa9400f61, // LDP (R27), (R1, R3) + 0xd65f03c0, // RET + 0x01000000, // WORD 0x1000000 + }, + }, } - if len(got) != len(want) { - t.Fatalf("word count = %d, want %d", len(got), len(want)) + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := arm64Words(t, tt.body) + if len(got) != len(tt.want) { + t.Fatalf("word count = %d, want %d (got %08x)", len(got), len(tt.want), got) + } + for i := range tt.want { + if got[i] != tt.want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], tt.want[i]) + } + } + }) } - for i := range want { - if got[i] != want[i] { - t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) - } +} + +// TestArm64LiteralPoolGuard pins the flushpool guard: the word-zero UNDEF +// that keeps execution from falling into the pool when the last statement is +// not a branch. END closes the body without becoming an instruction, so it +// does not count; a trailing PCDATA is a real statement and takes the guard; +// RET and the morestack block's branch need none. +func TestArm64LiteralPoolGuard(t *testing.T) { + tests := []struct { + name string + body string + want []uint32 + }{ + { + name: "END without RET still guards", + body: "\tMOVB\tR1, 0x1000000(R2)\n\tEND\n", + want: []uint32{ + 0x1800007b, 0x383b6841, // LDR W27, pool; MOVB R1, (R2)(R27) + 0x00000000, // UNDEF guard + 0x01000000, // WORD 0x1000000 + }, + }, + { + name: "trailing PCDATA keeps the guard", + body: "\tMOVD\tR1, 0x1007000(R2)\n\tRET\n\tPCDATA\t$0, $-1\n", + want: []uint32{ + 0x1800009b, 0xf83b6841, // LDR W27, pool; MOVD R1, (R2)(R27) + 0xd65f03c0, // RET + 0x00000000, // UNDEF guard + 0x01007000, // WORD 0x1007000 + }, + }, + { + name: "RET closes without a guard", + body: "\tMOVD\tR1, 0x1007000(R2)\n\tRET\n\tEND\n", + want: []uint32{ + 0x1800007b, 0xf83b6841, // LDR W27, pool; MOVD R1, (R2)(R27) + 0xd65f03c0, // RET + 0x01007000, // WORD 0x1007000 + }, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := arm64WordsTail(t, tt.body) + if len(got) != len(tt.want) { + t.Fatalf("word count = %d, want %d (got %08x)", len(got), len(tt.want), got) + } + for i := range tt.want { + if got[i] != tt.want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], tt.want[i]) + } + } + }) } } diff --git a/asm/arm64_frame.go b/asm/arm64_frame.go index b92378e..231f70c 100644 --- a/asm/arm64_frame.go +++ b/asm/arm64_frame.go @@ -495,6 +495,10 @@ func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte { return a64WordsLE(ws...) } +// arm64MoreStackBlockLen is the byte length of arm64MoreStackBlock: the +// saved LR, the BL and the branch back, three words whatever the target. +const arm64MoreStackBlockLen = 12 + // arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR), // BL runtime.morestack_noctxt, B back to the function start. The BL carries // the R_CALLARM64 relocation.