fix(arm64): store-exclusive operand order and large-frame parity

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-20 00:38:24 +02:00
parent a6f3828c02
commit d315a998ce
6 changed files with 382 additions and 48 deletions
+60 -18
View File
@@ -314,7 +314,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64CRC32(mnem, enc.op, ops) return encodeARM64CRC32(mnem, enc.op, ops)
} }
// Exclusive load/store (LDXR, STXR, LDAXR, STLXR). // Exclusive load/store (LDXR, STXR, LDAXR, STLXR and the register-pair
// forms LDXP, STXP).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl { if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl {
return encodeARM64Excl(mnem, enc.op, ops) return encodeARM64Excl(mnem, enc.op, ops)
} }
@@ -771,14 +772,17 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
} }
// The Go toolchain classifies immediates: // The Go toolchain classifies immediates (asm7.go conclass):
// - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values // - inside the imm12/shifted-imm12 "addcon" band (C_ABCON0/C_ABCON,
// - Negative values: MOVN first, then bitmask // 0 < v ≤ 4095 or a 4096 multiple up to 0xFFF000): bitmask first, so
// - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first // `MOVD $4096, R27` is ORR $4096, not MOVZ $(1<<12)
tryBitmaskFirst := d > 0 && d <= 0xFFF // - outside that band: MOVZ/MOVN first (C_MOVCON before C_BITCON), and
// negative values reach MOVN before the bitmask test
tryBitmaskFirst := d > 0 && (d <= 0xFFF || (d&0xFFF == 0 && d <= 0xFFF000))
if tryBitmaskFirst { if tryBitmaskFirst {
// Small immediate: try bitmask first (Go uses ORR for values like $1, $256). // Addcon-band immediate: try bitmask first (Go uses ORR for values
// like $1, $256 and $65536).
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf)) N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
if ok { if ok {
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
@@ -1354,14 +1358,43 @@ func arm64ExclMem(mnem string, op *ast.Operand) (int, error) {
return rn, nil return rn, nil
} }
// encodeARM64Excl encodes an exclusive load/store instruction. // arm64PairOf parses a register-pair operand `(R1, R2)`, reporting false
// LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg) // when the operand is not a pair. The toolchain takes the second register of
// STXR Rs, (Rn), Rt → STXR Rs, Rt, [Rn] (3 operands: Rs, mem, Rt-status) // the pair from the operand's Offset (its C_PAIR class,
// cmd/internal/obj/arm64/asm7.go cases 58/59).
func arm64PairOf(op *ast.Operand) (int, int, bool) {
raw := strings.TrimSpace(op.Raw)
if !strings.HasPrefix(raw, "(") || !strings.HasSuffix(raw, ")") {
return -1, -1, false
}
parts := strings.Split(raw[1:len(raw)-1], ",")
if len(parts) != 2 {
return -1, -1, false
}
r1 := arm64RegNum(strings.TrimSpace(parts[0]))
r2 := arm64RegNum(strings.TrimSpace(parts[1]))
if r1 < 0 || r2 < 0 {
return -1, -1, false
}
return r1, r2, true
}
// encodeARM64Excl encodes the exclusive load/store family with the operand
// order the toolchain parses (cmd/internal/obj/arm64/asm7.go cases 58 and 59,
// and its own spellings in arm64enc.s):
//
// STXR Rt, (Rn), Rs store, single register
// STXP (Rt1, Rt2), (Rn), Rs store, register pair
// LDXR (Rn), Rt load, single register
// LDXP (Rn), (Rt1, Rt2) load, register pair
//
// Decoded toolchain evidence: `STXR R1, (R2), R3` assembles to 0xc8037c41,
// whose fields are Rs=3, Rn=2, Rt=1: the FIRST register operand is the data
// register and the LAST the status register.
func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
// LDXR/STXR have different operand forms.
isLoad := strings.HasPrefix(mnem, "LD") isLoad := strings.HasPrefix(mnem, "LD")
if isLoad { if isLoad {
// LDXR (Rn), Rt → 2 operands: mem, reg // LDXR (Rn), Rt / LDXP (Rn), (Rt1, Rt2): 2 operands.
if len(ops) != 2 { if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
} }
@@ -1369,25 +1402,34 @@ func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
if err != nil { if err != nil {
return nil, err return nil, err
} }
if rt1, rt2, ok := arm64PairOf(ops[1]); ok {
// The single-register opcodes pre-set the unused Rs (bits 20:16)
// and Rt2 (bits 14:10) fields to 31; the pair forms carry a real
// Rt2 and keep Rs at 31.
return a64wordLE(baseOp | 0x1F<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
}
rt := arm64RegNum(operandRegName(ops[1])) rt := arm64RegNum(operandRegName(ops[1]))
if rt < 0 { if rt < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem) return nil, fmt.Errorf("invalid operand in %s", mnem)
} }
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil
} }
// STXR Rs, (Rn), Rt → 3 operands: Rs, mem, Rt // STXR Rt, (Rn), Rs / STXP (Rt1, Rt2), (Rn), Rs: 3 operands.
if len(ops) != 3 { if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
} }
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
rn, err := arm64ExclMem(mnem, ops[1]) rn, err := arm64ExclMem(mnem, ops[1])
if err != nil { if err != nil {
return nil, err return nil, err
} }
rt := arm64RegNum(operandRegName(ops[2])) rs := arm64RegNum(operandRegName(ops[2]))
if rs < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
if rt1, rt2, ok := arm64PairOf(ops[0]); ok {
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
}
rt := arm64RegNum(operandRegName(ops[0]))
if rt < 0 { if rt < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem) return nil, fmt.Errorf("invalid operand in %s", mnem)
} }
+23 -3
View File
@@ -98,8 +98,12 @@ func arm64RegNum(name string) int {
return 30 return 30
case "R31", "ZR": case "R31", "ZR":
return 31 return 31
case "SP": case "SP", "RSP":
return 31 // SP and ZR share encoding 31; context determines meaning // RSP is the toolchain's spelling for register 31 (it rejects
// R31 in an operand); SP stays for sources that spell it the
// amd64 way. SP and ZR share encoding 31; context determines
// the meaning.
return 31
} }
// F0-F31. // F0-F31.
if len(name) >= 1 && name[0] == 'F' { if len(name) >= 1 && name[0] == 'F' {
@@ -282,7 +286,7 @@ const (
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FCRC32 // CRC32 a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
) )
@@ -575,6 +579,10 @@ func init() {
} }
// ---- exclusive load/store ---- // ---- exclusive load/store ----
// Single-register forms pre-set the unused Rs and Rt2 fields to 31 (the
// 0x7c00/0x1f0000 halves of the constants below); the register-pair
// forms carry a real Rt2 in bits 14:10, so their opcodes pre-set
// neither field.
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00} a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00} a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00} a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
@@ -583,6 +591,12 @@ func init() {
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00} a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00} a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00} a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
// Pair loads, LDSTX(sz, 0, l=1, o1=1, o0) in asm7.go: LDXP/ LDXPW have
// o0=0, LDAXP/LDAXPW o0=1 (bit 15). Rs (bits 20:16) stays 31.
a64InstrTable["LDXP"] = a64Enc{format: a64FExcl, op: 0xc8600000}
a64InstrTable["LDXPW"] = a64Enc{format: a64FExcl, op: 0x88600000}
a64InstrTable["LDAXP"] = a64Enc{format: a64FExcl, op: 0xc8608000}
a64InstrTable["LDAXPW"] = a64Enc{format: a64FExcl, op: 0x88608000}
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00} a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00} a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00} a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
@@ -591,6 +605,12 @@ func init() {
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00} a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00} a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00} a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
// Pair stores, LDSTX(sz, 0, l=0, o1=1, o0): STXP/STXPW have o0=0,
// STLXP/STLXPW o0=1 (bit 15). Both Rs and Rt2 are real fields.
a64InstrTable["STXP"] = a64Enc{format: a64FExcl, op: 0xc8200000}
a64InstrTable["STXPW"] = a64Enc{format: a64FExcl, op: 0x88200000}
a64InstrTable["STLXP"] = a64Enc{format: a64FExcl, op: 0xc8208000}
a64InstrTable["STLXPW"] = a64Enc{format: a64FExcl, op: 0x88208000}
// ---- LSE atomics ---- // ---- LSE atomics ----
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
+74 -6
View File
@@ -820,15 +820,28 @@ func TestArm64ExclOffsetErrors(t *testing.T) {
} }
} }
// TestArm64ExclNoOffset pins the plain (Rn) forms. gasm parses the store // TestArm64ExclNoOffset pins the plain (Rn) forms, byte-for-byte against
// with the status register first (ARM ARM order); go tool asm parses the // go tool asm. The toolchain parses the FIRST register of a store as the
// same text with the data register first, so the two spellings differ and // data register and the LAST as the status register (asm7.go case 59), and
// the store word below is gasm's own. // the pair forms as (Rt1, Rt2) (case 58/59):
//
// STXR R3, (R1), R4 → c8047c23 (Rt=3, Rn=1, Rs=4)
// STXP (R3, R4), (R1), R5 → c8251023 (Rt=3, Rt2=4, Rn=1, Rs=5)
// LDXP (R1), (R3, R4) → c87f1023 (Rn=1, Rt=3, Rt2=4)
func TestArm64ExclNoOffset(t *testing.T) { func TestArm64ExclNoOffset(t *testing.T) {
got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n") got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n"+
"\tSTXP (R3, R4), (R1), R5\n\tSTXPW (R3, R4), (R1), R5\n"+
"\tLDXP (R1), (R3, R4)\n\tLDXPW (R1), (R3, R4)\n"+
"\tSTXR R3, (RSP), R4\n\tLDXR (RSP), R2\n")
want := []uint32{ want := []uint32{
0xc85f7c22, // LDXR X2, [X1] 0xc85f7c22, // LDXR X2, [X1]
0xc8037c24, // STXR W3, X4, [X1] with Rs = R3, Rt = R4 0xc8047c23, // STXR W3, [X1], W4 with Rt = R3, Rs = R4
0xc8251023, // STXP (R3, R4), [X1], R5
0x88251023, // STXPW (R3, R4), [X1], R5
0xc87f1023, // LDXP [X1], (R3, R4)
0x887f1023, // LDXPW [X1], (R3, R4)
0xc8047fe3, // STXR R3, [SP], R4
0xc85f7fe2, // LDXR [SP], R2
0xd65f03c0, // RET 0xd65f03c0, // RET
} }
for i := range want { for i := range want {
@@ -920,3 +933,58 @@ func TestArm64LargeFrameSpadj(t *testing.T) {
t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got) t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got)
} }
} }
// TestArm64SplitFrameSpadj pins the addcon2 band, where neither imm12 form
// nor a single MOVZ carries the autosize and the toolchain splits the
// prologue SUB into two imm12 instructions (asm7.go case 48) while the
// non-leaf RET still materialises the value into REGTMP (obj7.go ARET,
// issue 73259). $65664 rounds the autosize to 65680 = 144 + 16<<12:
//
// [SUB $144, RSP, R20][SUB $(16<<12), R20, R20][STP][MOVD R20, SP][SUB $8]
// [CALL]
// [LDP][MOVD $144, R27][MOVK $(1<<16), R27][ADD R27, RSP, RSP][RET]
//
// SP moves at the fourth word (byte 12) and returns to zero at the final
// RET (byte 40); the words are go tool asm's own for the same source.
func TestArm64SplitFrameSpadj(t *testing.T) {
f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), NOSPLIT, $65664-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
fn := img.Funcs[0]
wantSpadj := []SpadjStep{{PC: 12, Value: 65680}, {PC: 40, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
want := []uint32{
0xd10243f4, // SUB $144, RSP, R20
0xd1404294, // SUB $(16<<12), R20, R20
0xa93ffa9d, // STP (R29, R30), -8(R20)
0x9100029f, // MOVD R20, RSP
0xd10023fd, // SUB $8, RSP, R29
0x94000000, // CALL (relocation masked at link time)
0xa97ffbfd, // LDP -8(RSP), (R29, R30)
0xd280121b, // MOVD $144, R27
0xf2a0003b, // MOVK $(1<<16), R27
0x8b3b63ff, // ADD R27, RSP, RSP
0xd65f03c0, // RET
}
words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
if len(words) != len(want) {
t.Fatalf("framed = %d words, want %d", len(words), len(want))
}
for i, w := range want {
if words[i] != w {
t.Errorf("word %d = %08x, want %08x", i, words[i], w)
}
}
}
+75 -21
View File
@@ -187,10 +187,32 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
return a64WordsLE(ws...) return a64WordsLE(ws...)
} }
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value // arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple // two imm12 instructions instead of materialising it into REGTMP
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and // (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
// subtracts the register in the extended-register form. // unsigned and be neither encodable as one imm12 (checked by the callers
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
// logical immediate, because conclass tests all three before C_ADDCON2.
func arm64SplitImm12(imm uint32) bool {
if imm > 0xFFFFFF {
return false
}
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
return false
}
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
}
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
// immediate form when the value fits imm12 (plain, or shifted left by 12
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
// logical immediate, or one wider than 24 bits is materialised into REGTMP
// (R27) and subtracted in the extended-register form; everything else up to
// 0xFFFFFF is split into two imm12 instructions:
//
// SUB $(imm&0xfff), SP, Rd
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
func arm64SubImmWords(imm uint32, rd uint32) []uint32 { func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF { if imm <= 0xFFF {
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)} return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
@@ -198,15 +220,21 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 4095<<12 && imm&0xFFF == 0 { if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)} return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
} }
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") if !arm64SplitImm12(imm) {
if err != nil { mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
mov = nil if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
}
return []uint32{
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
} }
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
} }
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12 // arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
// and REGTMP fallback ladder. // split and REGTMP ladder as arm64SubImmWords.
func arm64AddImmWords(imm uint32, rd uint32) []uint32 { func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF { if imm <= 0xFFF {
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)} return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
@@ -214,11 +242,35 @@ func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 4095<<12 && imm&0xFFF == 0 { if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)} return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
} }
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") if !arm64SplitImm12(imm) {
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
}
return []uint32{
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
}
}
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
// large frame. The toolchain adds the frame back with a single instruction:
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
// materialised into REGTMP and added as a register, so the epilogue never
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
// the leaf epilogue's plain ADD instructions.
func arm64RetAddWords(autosize uint32) []uint32 {
if autosize < 1<<12 {
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
}
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
if err != nil { if err != nil {
mov = nil mov = nil
} }
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd)) return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
} }
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and // arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
@@ -237,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte {
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
) )
} else { } else {
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP // Large frame: LDP -8(SP), (FP, LR), then deallocate.
ws = append(ws, ws = append(ws,
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair) a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
) )
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...) ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
} }
} }
// RET: BR LR (0xd65f03c0) // RET: BR LR (0xd65f03c0)
@@ -260,15 +312,16 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
} }
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose // Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
// position depends on how many words the SUB itself took (immediate, // position depends on how many words the SUB itself took (immediate,
// shifted immediate, or a materialised REGTMP sequence). // shifted immediate, the two-word imm12 split, or a materialised REGTMP
// sequence).
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1) return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
} }
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to // arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final RET instruction. The ADD sequences share the // (but not including) the final RET instruction. The lengths are read from
// prologue's immediate ladder, so their length is read from the same helper // the same word-emitting helpers the epilogue uses rather than assumed: the
// rather than assumed: a materialised autosize costs its MOV words plus the // leaf path shares the prologue's immediate ladder, and a materialised
// ADD itself. // autosize costs its MOV words plus the ADD itself.
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int { func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
if fi.autosize == 0 { if fi.autosize == 0 {
return 0 return 0
@@ -280,8 +333,9 @@ func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
if fi.autosize <= 0xf0 { if fi.autosize <= 0xf0 {
return 8 // LDR + LDR.P return 8 // LDR + LDR.P
} }
// LDP + the ADD ladder that deallocates the frame. // LDP + the deallocation emitted by arm64RetAddWords, so the length
return 4 + 4*len(arm64AddImmWords(uint32(fi.autosize), 31)) // tracks whatever the MOVD ladder needs.
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
} }
// arm64ResolvePseudo translates a pseudo-register memory reference into a // arm64ResolvePseudo translates a pseudo-register memory reference into a
+69
View File
@@ -0,0 +1,69 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Large frames across every immediate band of the prologue SUB and the RET
// epilogue, byte-parity-checked against go tool asm:
//
// $5000 autosize 5024 one 16-bit chunk, materialised into REGTMP
// $65664 autosize 65680 split into two imm12 instructions
// $70000 autosize 70016 split into two imm12 instructions
// $65520 autosize 65536 one shifted imm12 in the prologue, a logical
// immediate (ORR) in the non-leaf epilogue
// $16777232 autosize 16777248 wider than 24 bits, MOVZ/MOVK into REGTMP
#include "textflag.h"
TEXT ·leaf5000(SB), NOSPLIT, $5000-0
MOVD R0, R1
MOVD R1, R2
RET
TEXT ·leaf65664(SB), NOSPLIT, $65664-0
MOVD R0, R1
MOVD R1, R2
RET
TEXT ·leaf70000(SB), NOSPLIT, $70000-0
MOVD R0, R1
MOVD R1, R2
RET
TEXT ·leaf65520(SB), NOSPLIT, $65520-0
MOVD R0, R1
MOVD R1, R2
MOVD R2, R3
MOVD R3, R4
RET
TEXT ·nl5000(SB), NOSPLIT, $5000-0
MOVD R0, R1
MOVD R1, R2
CALL ·other(SB)
RET
TEXT ·nl65664(SB), NOSPLIT, $65664-0
MOVD R0, R1
CALL ·other(SB)
RET
TEXT ·nl70000(SB), NOSPLIT, $70000-0
MOVD R0, R1
CALL ·other(SB)
RET
TEXT ·nl65520(SB), NOSPLIT, $65520-0
MOVD R0, R1
MOVD R1, R2
MOVD R2, R3
CALL ·other(SB)
RET
TEXT ·nlhuge(SB), NOSPLIT, $16777232-0
CALL ·other(SB)
RET
TEXT ·other(SB), NOSPLIT, $0-0
MOVD R0, R1
MOVD R1, R2
MOVD R2, R3
RET
+81
View File
@@ -0,0 +1,81 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Exclusive load/store family, LSE atomics and their memory operands,
// byte-parity-checked against go tool asm. The toolchain parses the FIRST
// register of an exclusive store as the data register and the LAST as the
// status register, and takes register pairs as (Rt1, Rt2) operands.
#include "textflag.h"
TEXT ·loads(SB), NOSPLIT, $0-0
LDXR (R1), R2
LDXRB (R1), R3
LDXRH (R1), R4
LDXRW (R1), R5
LDAXR (R1), R2
LDAXRB (R1), R3
LDAXRH (R1), R4
LDAXRW (R1), R5
LDXR (RSP), R2
LDXRB (RSP), R3
LDXRH (RSP), R4
LDXRW (RSP), R5
LDAXR (RSP), R2
LDAXRB (RSP), R3
LDAXRH (RSP), R4
RET
TEXT ·stores(SB), NOSPLIT, $0-0
STXR R2, (R1), R6
STXRB R3, (R1), R6
STXRH R4, (R1), R6
STXRW R5, (R1), R6
STLXR R2, (R1), R6
STLXRB R3, (R1), R6
STLXRH R4, (R1), R6
STLXRW R5, (R1), R6
STXR R2, (RSP), R6
STXRB R3, (RSP), R6
STXRH R4, (RSP), R6
STXRW R5, (RSP), R6
STLXR R2, (RSP), R6
STLXRB R3, (RSP), R6
STLXRH R4, (RSP), R6
RET
TEXT ·pairs(SB), NOSPLIT, $0-0
LDXP (R1), (R2, R3)
LDXPW (R1), (R2, R3)
LDAXP (R1), (R2, R3)
LDAXPW (R1), (R2, R3)
STXP (R2, R3), (R1), R6
STXPW (R2, R3), (R1), R6
STLXP (R2, R3), (R1), R6
STLXPW (R2, R3), (R1), R6
LDXP (RSP), (R2, R3)
LDXPW (RSP), (R2, R3)
LDAXP (RSP), (R4, R5)
LDAXPW (RSP), (R4, R5)
STXP (R2, R3), (RSP), R6
STXPW (R2, R3), (RSP), R6
STLXP (R4, R5), (RSP), R7
RET
TEXT ·atomics(SB), NOSPLIT, $0-0
LDADDB R2, (R1), R3
LDADDH R2, (R1), R3
LDADDW R2, (R1), R3
LDADDD R2, (R1), R3
LDADDB R2, (R1), ZR
LDADDH R2, (R1), ZR
LDADDW R2, (R1), ZR
LDADDD R2, (R1), ZR
CASW R2, (R1), R3
CASD R2, (R1), R3
CASW R2, (R1), ZR
CASD R2, (R1), ZR
SWPW R2, (R1), R3
SWPD R2, (R1), R3
SWPD R2, (R1), ZR
RET