From d315a998ce969ea3281d430fc4cff4ae1cb13dbe Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 00:38:24 +0200 Subject: [PATCH] fix(arm64): store-exclusive operand order and large-frame parity Assisted-by: GLM 5.3 --- asm/arm64_assemble.go | 78 +++++++++++++++++++------ asm/arm64_encode.go | 26 ++++++++- asm/arm64_encode_test.go | 80 ++++++++++++++++++++++++-- asm/arm64_frame.go | 96 ++++++++++++++++++++++++------- testdata/verify/bigframe2_arm64.s | 69 ++++++++++++++++++++++ testdata/verify/exclusive_arm64.s | 81 ++++++++++++++++++++++++++ 6 files changed, 382 insertions(+), 48 deletions(-) create mode 100644 testdata/verify/bigframe2_arm64.s create mode 100644 testdata/verify/exclusive_arm64.s diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 8efcfaa..d754088 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -314,7 +314,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64CRC32(mnem, enc.op, ops) } - // Exclusive load/store (LDXR, STXR, LDAXR, STLXR). + // Exclusive load/store (LDXR, STXR, LDAXR, STLXR and the register-pair + // forms LDXP, STXP). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl { return encodeARM64Excl(mnem, enc.op, ops) } @@ -771,14 +772,17 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil } - // The Go toolchain classifies immediates: - // - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values - // - Negative values: MOVN first, then bitmask - // - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first - tryBitmaskFirst := d > 0 && d <= 0xFFF + // The Go toolchain classifies immediates (asm7.go conclass): + // - inside the imm12/shifted-imm12 "addcon" band (C_ABCON0/C_ABCON, + // 0 < v ≤ 4095 or a 4096 multiple up to 0xFFF000): bitmask first, so + // `MOVD $4096, R27` is ORR $4096, not MOVZ $(1<<12) + // - outside that band: MOVZ/MOVN first (C_MOVCON before C_BITCON), and + // negative values reach MOVN before the bitmask test + tryBitmaskFirst := d > 0 && (d <= 0xFFF || (d&0xFFF == 0 && d <= 0xFFF000)) if tryBitmaskFirst { - // Small immediate: try bitmask first (Go uses ORR for values like $1, $256). + // Addcon-band immediate: try bitmask first (Go uses ORR for values + // like $1, $256 and $65536). N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf)) if ok { return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil @@ -1354,14 +1358,43 @@ func arm64ExclMem(mnem string, op *ast.Operand) (int, error) { return rn, nil } -// encodeARM64Excl encodes an exclusive load/store instruction. -// LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg) -// STXR Rs, (Rn), Rt → STXR Rs, Rt, [Rn] (3 operands: Rs, mem, Rt-status) +// arm64PairOf parses a register-pair operand `(R1, R2)`, reporting false +// when the operand is not a pair. The toolchain takes the second register of +// the pair from the operand's Offset (its C_PAIR class, +// cmd/internal/obj/arm64/asm7.go cases 58/59). +func arm64PairOf(op *ast.Operand) (int, int, bool) { + raw := strings.TrimSpace(op.Raw) + if !strings.HasPrefix(raw, "(") || !strings.HasSuffix(raw, ")") { + return -1, -1, false + } + parts := strings.Split(raw[1:len(raw)-1], ",") + if len(parts) != 2 { + return -1, -1, false + } + r1 := arm64RegNum(strings.TrimSpace(parts[0])) + r2 := arm64RegNum(strings.TrimSpace(parts[1])) + if r1 < 0 || r2 < 0 { + return -1, -1, false + } + return r1, r2, true +} + +// encodeARM64Excl encodes the exclusive load/store family with the operand +// order the toolchain parses (cmd/internal/obj/arm64/asm7.go cases 58 and 59, +// and its own spellings in arm64enc.s): +// +// STXR Rt, (Rn), Rs store, single register +// STXP (Rt1, Rt2), (Rn), Rs store, register pair +// LDXR (Rn), Rt load, single register +// LDXP (Rn), (Rt1, Rt2) load, register pair +// +// Decoded toolchain evidence: `STXR R1, (R2), R3` assembles to 0xc8037c41, +// whose fields are Rs=3, Rn=2, Rt=1: the FIRST register operand is the data +// register and the LAST the status register. func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { - // LDXR/STXR have different operand forms. isLoad := strings.HasPrefix(mnem, "LD") if isLoad { - // LDXR (Rn), Rt → 2 operands: mem, reg + // LDXR (Rn), Rt / LDXP (Rn), (Rt1, Rt2): 2 operands. if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } @@ -1369,25 +1402,34 @@ func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er if err != nil { return nil, err } + if rt1, rt2, ok := arm64PairOf(ops[1]); ok { + // The single-register opcodes pre-set the unused Rs (bits 20:16) + // and Rt2 (bits 14:10) fields to 31; the pair forms carry a real + // Rt2 and keep Rs at 31. + return a64wordLE(baseOp | 0x1F<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil + } rt := arm64RegNum(operandRegName(ops[1])) if rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil } - // STXR Rs, (Rn), Rt → 3 operands: Rs, mem, Rt + // STXR Rt, (Rn), Rs / STXP (Rt1, Rt2), (Rn), Rs: 3 operands. if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } - rs := arm64RegNum(operandRegName(ops[0])) - if rs < 0 { - return nil, fmt.Errorf("invalid operand in %s", mnem) - } rn, err := arm64ExclMem(mnem, ops[1]) if err != nil { return nil, err } - rt := arm64RegNum(operandRegName(ops[2])) + rs := arm64RegNum(operandRegName(ops[2])) + if rs < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + if rt1, rt2, ok := arm64PairOf(ops[0]); ok { + return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil + } + rt := arm64RegNum(operandRegName(ops[0])) if rt < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index d6a68c2..e91f77a 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -98,8 +98,12 @@ func arm64RegNum(name string) int { return 30 case "R31", "ZR": return 31 - case "SP": - return 31 // SP and ZR share encoding 31; context determines meaning + case "SP", "RSP": + // RSP is the toolchain's spelling for register 31 (it rejects + // R31 in an operand); SP stays for sources that spell it the + // amd64 way. SP and ZR share encoding 31; context determines + // the meaning. + return 31 } // F0-F31. if len(name) >= 1 && name[0] == 'F' { @@ -282,7 +286,7 @@ const ( a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL a64FCRC32 // CRC32 a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG - a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR + a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP a64FLSE // LSE atomics: LDADD, CAS, SWP a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL ) @@ -575,6 +579,10 @@ func init() { } // ---- exclusive load/store ---- + // Single-register forms pre-set the unused Rs and Rt2 fields to 31 (the + // 0x7c00/0x1f0000 halves of the constants below); the register-pair + // forms carry a real Rt2 in bits 14:10, so their opcodes pre-set + // neither field. a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00} a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00} a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00} @@ -583,6 +591,12 @@ func init() { a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00} a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00} a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00} + // Pair loads, LDSTX(sz, 0, l=1, o1=1, o0) in asm7.go: LDXP/ LDXPW have + // o0=0, LDAXP/LDAXPW o0=1 (bit 15). Rs (bits 20:16) stays 31. + a64InstrTable["LDXP"] = a64Enc{format: a64FExcl, op: 0xc8600000} + a64InstrTable["LDXPW"] = a64Enc{format: a64FExcl, op: 0x88600000} + a64InstrTable["LDAXP"] = a64Enc{format: a64FExcl, op: 0xc8608000} + a64InstrTable["LDAXPW"] = a64Enc{format: a64FExcl, op: 0x88608000} a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00} a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00} a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00} @@ -591,6 +605,12 @@ func init() { a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00} a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00} a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00} + // Pair stores, LDSTX(sz, 0, l=0, o1=1, o0): STXP/STXPW have o0=0, + // STLXP/STLXPW o0=1 (bit 15). Both Rs and Rt2 are real fields. + a64InstrTable["STXP"] = a64Enc{format: a64FExcl, op: 0xc8200000} + a64InstrTable["STXPW"] = a64Enc{format: a64FExcl, op: 0x88200000} + a64InstrTable["STLXP"] = a64Enc{format: a64FExcl, op: 0xc8208000} + a64InstrTable["STLXPW"] = a64Enc{format: a64FExcl, op: 0x88208000} // ---- LSE atomics ---- a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10} diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index 9e72da1..980a8da 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -820,15 +820,28 @@ func TestArm64ExclOffsetErrors(t *testing.T) { } } -// TestArm64ExclNoOffset pins the plain (Rn) forms. gasm parses the store -// with the status register first (ARM ARM order); go tool asm parses the -// same text with the data register first, so the two spellings differ and -// the store word below is gasm's own. +// TestArm64ExclNoOffset pins the plain (Rn) forms, byte-for-byte against +// go tool asm. The toolchain parses the FIRST register of a store as the +// data register and the LAST as the status register (asm7.go case 59), and +// the pair forms as (Rt1, Rt2) (case 58/59): +// +// STXR R3, (R1), R4 → c8047c23 (Rt=3, Rn=1, Rs=4) +// STXP (R3, R4), (R1), R5 → c8251023 (Rt=3, Rt2=4, Rn=1, Rs=5) +// LDXP (R1), (R3, R4) → c87f1023 (Rn=1, Rt=3, Rt2=4) func TestArm64ExclNoOffset(t *testing.T) { - got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n") + got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n"+ + "\tSTXP (R3, R4), (R1), R5\n\tSTXPW (R3, R4), (R1), R5\n"+ + "\tLDXP (R1), (R3, R4)\n\tLDXPW (R1), (R3, R4)\n"+ + "\tSTXR R3, (RSP), R4\n\tLDXR (RSP), R2\n") want := []uint32{ 0xc85f7c22, // LDXR X2, [X1] - 0xc8037c24, // STXR W3, X4, [X1] with Rs = R3, Rt = R4 + 0xc8047c23, // STXR W3, [X1], W4 with Rt = R3, Rs = R4 + 0xc8251023, // STXP (R3, R4), [X1], R5 + 0x88251023, // STXPW (R3, R4), [X1], R5 + 0xc87f1023, // LDXP [X1], (R3, R4) + 0x887f1023, // LDXPW [X1], (R3, R4) + 0xc8047fe3, // STXR R3, [SP], R4 + 0xc85f7fe2, // LDXR [SP], R2 0xd65f03c0, // RET } for i := range want { @@ -920,3 +933,58 @@ func TestArm64LargeFrameSpadj(t *testing.T) { t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got) } } + +// TestArm64SplitFrameSpadj pins the addcon2 band, where neither imm12 form +// nor a single MOVZ carries the autosize and the toolchain splits the +// prologue SUB into two imm12 instructions (asm7.go case 48) while the +// non-leaf RET still materialises the value into REGTMP (obj7.go ARET, +// issue 73259). $65664 rounds the autosize to 65680 = 144 + 16<<12: +// +// [SUB $144, RSP, R20][SUB $(16<<12), R20, R20][STP][MOVD R20, SP][SUB $8] +// [CALL] +// [LDP][MOVD $144, R27][MOVK $(1<<16), R27][ADD R27, RSP, RSP][RET] +// +// SP moves at the fourth word (byte 12) and returns to zero at the final +// RET (byte 40); the words are go tool asm's own for the same source. +func TestArm64SplitFrameSpadj(t *testing.T) { + f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), NOSPLIT, $65664-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + fn := img.Funcs[0] + wantSpadj := []SpadjStep{{PC: 12, Value: 65680}, {PC: 40, Value: 0}} + if len(fn.Spadj) != len(wantSpadj) { + t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj) + } + for i := range wantSpadj { + if fn.Spadj[i] != wantSpadj[i] { + t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i]) + } + } + want := []uint32{ + 0xd10243f4, // SUB $144, RSP, R20 + 0xd1404294, // SUB $(16<<12), R20, R20 + 0xa93ffa9d, // STP (R29, R30), -8(R20) + 0x9100029f, // MOVD R20, RSP + 0xd10023fd, // SUB $8, RSP, R29 + 0x94000000, // CALL (relocation masked at link time) + 0xa97ffbfd, // LDP -8(RSP), (R29, R30) + 0xd280121b, // MOVD $144, R27 + 0xf2a0003b, // MOVK $(1<<16), R27 + 0x8b3b63ff, // ADD R27, RSP, RSP + 0xd65f03c0, // RET + } + words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size]) + if len(words) != len(want) { + t.Fatalf("framed = %d words, want %d", len(words), len(want)) + } + for i, w := range want { + if words[i] != w { + t.Errorf("word %d = %08x, want %08x", i, words[i], w) + } + } +} diff --git a/asm/arm64_frame.go b/asm/arm64_frame.go index fc1c636..e9b7af7 100644 --- a/asm/arm64_frame.go +++ b/asm/arm64_frame.go @@ -187,10 +187,32 @@ func arm64Prologue(fi arm64FrameInfo) []byte { return a64WordsLE(ws...) } -// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value -// fits the imm12 field (plain, or shifted left by 12 when it is a multiple -// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and -// subtracts the register in the extended-register form. +// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into +// two imm12 instructions instead of materialising it into REGTMP +// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits +// unsigned and be neither encodable as one imm12 (checked by the callers +// first), nor loadable into a register in a single MOVZ/MOVN word, nor a +// logical immediate, because conclass tests all three before C_ADDCON2. +func arm64SplitImm12(imm uint32) bool { + if imm > 0xFFFFFF { + return false + } + if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok { + return false + } + return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0 +} + +// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an +// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the +// immediate form when the value fits imm12 (plain, or shifted left by 12 +// when it is a multiple of 4096); a value with a single 16-bit chunk, a +// logical immediate, or one wider than 24 bits is materialised into REGTMP +// (R27) and subtracted in the extended-register form; everything else up to +// 0xFFFFFF is split into two imm12 instructions: +// +// SUB $(imm&0xfff), SP, Rd +// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd func arm64SubImmWords(imm uint32, rd uint32) []uint32 { if imm <= 0xFFF { return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)} @@ -198,15 +220,21 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 { if imm <= 4095<<12 && imm&0xFFF == 0 { return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)} } - mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") - if err != nil { - mov = nil + if !arm64SplitImm12(imm) { + mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") + if err != nil { + mov = nil + } + return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd)) + } + return []uint32{ + a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd), + a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd), } - return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd)) } -// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12 -// and REGTMP fallback ladder. +// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12, +// split and REGTMP ladder as arm64SubImmWords. func arm64AddImmWords(imm uint32, rd uint32) []uint32 { if imm <= 0xFFF { return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)} @@ -214,11 +242,35 @@ func arm64AddImmWords(imm uint32, rd uint32) []uint32 { if imm <= 4095<<12 && imm&0xFFF == 0 { return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)} } - mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") + if !arm64SplitImm12(imm) { + mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD") + if err != nil { + mov = nil + } + return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd)) + } + return []uint32{ + a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd), + a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd), + } +} + +// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a +// large frame. The toolchain adds the frame back with a single instruction: +// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is +// materialised into REGTMP and added as a register, so the epilogue never +// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The +// shifted-imm12 and split-imm12 forms are therefore never used here, unlike +// the leaf epilogue's plain ADD instructions. +func arm64RetAddWords(autosize uint32) []uint32 { + if autosize < 1<<12 { + return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)} + } + mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD") if err != nil { mov = nil } - return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd)) + return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31)) } // arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and @@ -237,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte { arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize ) } else { - // Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP + // Large frame: LDP -8(SP), (FP, LR), then deallocate. ws = append(ws, a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair) ) - ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...) + ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...) } } // RET: BR LR (0xd65f03c0) @@ -260,15 +312,16 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int { } // Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose // position depends on how many words the SUB itself took (immediate, - // shifted immediate, or a materialised REGTMP sequence). + // shifted immediate, the two-word imm12 split, or a materialised REGTMP + // sequence). return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1) } // arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to -// (but not including) the final RET instruction. The ADD sequences share the -// prologue's immediate ladder, so their length is read from the same helper -// rather than assumed: a materialised autosize costs its MOV words plus the -// ADD itself. +// (but not including) the final RET instruction. The lengths are read from +// the same word-emitting helpers the epilogue uses rather than assumed: the +// leaf path shares the prologue's immediate ladder, and a materialised +// autosize costs its MOV words plus the ADD itself. func arm64ReturnEpilogueLen(fi arm64FrameInfo) int { if fi.autosize == 0 { return 0 @@ -280,8 +333,9 @@ func arm64ReturnEpilogueLen(fi arm64FrameInfo) int { if fi.autosize <= 0xf0 { return 8 // LDR + LDR.P } - // LDP + the ADD ladder that deallocates the frame. - return 4 + 4*len(arm64AddImmWords(uint32(fi.autosize), 31)) + // LDP + the deallocation emitted by arm64RetAddWords, so the length + // tracks whatever the MOVD ladder needs. + return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize))) } // arm64ResolvePseudo translates a pseudo-register memory reference into a diff --git a/testdata/verify/bigframe2_arm64.s b/testdata/verify/bigframe2_arm64.s new file mode 100644 index 0000000..831348b --- /dev/null +++ b/testdata/verify/bigframe2_arm64.s @@ -0,0 +1,69 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Large frames across every immediate band of the prologue SUB and the RET +// epilogue, byte-parity-checked against go tool asm: +// +// $5000 autosize 5024 one 16-bit chunk, materialised into REGTMP +// $65664 autosize 65680 split into two imm12 instructions +// $70000 autosize 70016 split into two imm12 instructions +// $65520 autosize 65536 one shifted imm12 in the prologue, a logical +// immediate (ORR) in the non-leaf epilogue +// $16777232 autosize 16777248 wider than 24 bits, MOVZ/MOVK into REGTMP + +#include "textflag.h" + +TEXT ·leaf5000(SB), NOSPLIT, $5000-0 + MOVD R0, R1 + MOVD R1, R2 + RET + +TEXT ·leaf65664(SB), NOSPLIT, $65664-0 + MOVD R0, R1 + MOVD R1, R2 + RET + +TEXT ·leaf70000(SB), NOSPLIT, $70000-0 + MOVD R0, R1 + MOVD R1, R2 + RET + +TEXT ·leaf65520(SB), NOSPLIT, $65520-0 + MOVD R0, R1 + MOVD R1, R2 + MOVD R2, R3 + MOVD R3, R4 + RET + +TEXT ·nl5000(SB), NOSPLIT, $5000-0 + MOVD R0, R1 + MOVD R1, R2 + CALL ·other(SB) + RET + +TEXT ·nl65664(SB), NOSPLIT, $65664-0 + MOVD R0, R1 + CALL ·other(SB) + RET + +TEXT ·nl70000(SB), NOSPLIT, $70000-0 + MOVD R0, R1 + CALL ·other(SB) + RET + +TEXT ·nl65520(SB), NOSPLIT, $65520-0 + MOVD R0, R1 + MOVD R1, R2 + MOVD R2, R3 + CALL ·other(SB) + RET + +TEXT ·nlhuge(SB), NOSPLIT, $16777232-0 + CALL ·other(SB) + RET + +TEXT ·other(SB), NOSPLIT, $0-0 + MOVD R0, R1 + MOVD R1, R2 + MOVD R2, R3 + RET diff --git a/testdata/verify/exclusive_arm64.s b/testdata/verify/exclusive_arm64.s new file mode 100644 index 0000000..b848c36 --- /dev/null +++ b/testdata/verify/exclusive_arm64.s @@ -0,0 +1,81 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Exclusive load/store family, LSE atomics and their memory operands, +// byte-parity-checked against go tool asm. The toolchain parses the FIRST +// register of an exclusive store as the data register and the LAST as the +// status register, and takes register pairs as (Rt1, Rt2) operands. + +#include "textflag.h" + +TEXT ·loads(SB), NOSPLIT, $0-0 + LDXR (R1), R2 + LDXRB (R1), R3 + LDXRH (R1), R4 + LDXRW (R1), R5 + LDAXR (R1), R2 + LDAXRB (R1), R3 + LDAXRH (R1), R4 + LDAXRW (R1), R5 + LDXR (RSP), R2 + LDXRB (RSP), R3 + LDXRH (RSP), R4 + LDXRW (RSP), R5 + LDAXR (RSP), R2 + LDAXRB (RSP), R3 + LDAXRH (RSP), R4 + RET + +TEXT ·stores(SB), NOSPLIT, $0-0 + STXR R2, (R1), R6 + STXRB R3, (R1), R6 + STXRH R4, (R1), R6 + STXRW R5, (R1), R6 + STLXR R2, (R1), R6 + STLXRB R3, (R1), R6 + STLXRH R4, (R1), R6 + STLXRW R5, (R1), R6 + STXR R2, (RSP), R6 + STXRB R3, (RSP), R6 + STXRH R4, (RSP), R6 + STXRW R5, (RSP), R6 + STLXR R2, (RSP), R6 + STLXRB R3, (RSP), R6 + STLXRH R4, (RSP), R6 + RET + +TEXT ·pairs(SB), NOSPLIT, $0-0 + LDXP (R1), (R2, R3) + LDXPW (R1), (R2, R3) + LDAXP (R1), (R2, R3) + LDAXPW (R1), (R2, R3) + STXP (R2, R3), (R1), R6 + STXPW (R2, R3), (R1), R6 + STLXP (R2, R3), (R1), R6 + STLXPW (R2, R3), (R1), R6 + LDXP (RSP), (R2, R3) + LDXPW (RSP), (R2, R3) + LDAXP (RSP), (R4, R5) + LDAXPW (RSP), (R4, R5) + STXP (R2, R3), (RSP), R6 + STXPW (R2, R3), (RSP), R6 + STLXP (R4, R5), (RSP), R7 + RET + +TEXT ·atomics(SB), NOSPLIT, $0-0 + LDADDB R2, (R1), R3 + LDADDH R2, (R1), R3 + LDADDW R2, (R1), R3 + LDADDD R2, (R1), R3 + LDADDB R2, (R1), ZR + LDADDH R2, (R1), ZR + LDADDW R2, (R1), ZR + LDADDD R2, (R1), ZR + CASW R2, (R1), R3 + CASD R2, (R1), R3 + CASW R2, (R1), ZR + CASD R2, (R1), ZR + SWPW R2, (R1), R3 + SWPD R2, (R1), R3 + SWPD R2, (R1), ZR + RET