fix(arm64): store-exclusive operand order and large-frame parity
Assisted-by: GLM 5.3
This commit is contained in:
+60
-18
@@ -314,7 +314,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
|||||||
return encodeARM64CRC32(mnem, enc.op, ops)
|
return encodeARM64CRC32(mnem, enc.op, ops)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Exclusive load/store (LDXR, STXR, LDAXR, STLXR).
|
// Exclusive load/store (LDXR, STXR, LDAXR, STLXR and the register-pair
|
||||||
|
// forms LDXP, STXP).
|
||||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl {
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl {
|
||||||
return encodeARM64Excl(mnem, enc.op, ops)
|
return encodeARM64Excl(mnem, enc.op, ops)
|
||||||
}
|
}
|
||||||
@@ -771,14 +772,17 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
|
|||||||
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
|
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// The Go toolchain classifies immediates:
|
// The Go toolchain classifies immediates (asm7.go conclass):
|
||||||
// - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values
|
// - inside the imm12/shifted-imm12 "addcon" band (C_ABCON0/C_ABCON,
|
||||||
// - Negative values: MOVN first, then bitmask
|
// 0 < v ≤ 4095 or a 4096 multiple up to 0xFFF000): bitmask first, so
|
||||||
// - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first
|
// `MOVD $4096, R27` is ORR $4096, not MOVZ $(1<<12)
|
||||||
tryBitmaskFirst := d > 0 && d <= 0xFFF
|
// - outside that band: MOVZ/MOVN first (C_MOVCON before C_BITCON), and
|
||||||
|
// negative values reach MOVN before the bitmask test
|
||||||
|
tryBitmaskFirst := d > 0 && (d <= 0xFFF || (d&0xFFF == 0 && d <= 0xFFF000))
|
||||||
|
|
||||||
if tryBitmaskFirst {
|
if tryBitmaskFirst {
|
||||||
// Small immediate: try bitmask first (Go uses ORR for values like $1, $256).
|
// Addcon-band immediate: try bitmask first (Go uses ORR for values
|
||||||
|
// like $1, $256 and $65536).
|
||||||
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
|
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
|
||||||
if ok {
|
if ok {
|
||||||
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
|
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
|
||||||
@@ -1354,14 +1358,43 @@ func arm64ExclMem(mnem string, op *ast.Operand) (int, error) {
|
|||||||
return rn, nil
|
return rn, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeARM64Excl encodes an exclusive load/store instruction.
|
// arm64PairOf parses a register-pair operand `(R1, R2)`, reporting false
|
||||||
// LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg)
|
// when the operand is not a pair. The toolchain takes the second register of
|
||||||
// STXR Rs, (Rn), Rt → STXR Rs, Rt, [Rn] (3 operands: Rs, mem, Rt-status)
|
// the pair from the operand's Offset (its C_PAIR class,
|
||||||
|
// cmd/internal/obj/arm64/asm7.go cases 58/59).
|
||||||
|
func arm64PairOf(op *ast.Operand) (int, int, bool) {
|
||||||
|
raw := strings.TrimSpace(op.Raw)
|
||||||
|
if !strings.HasPrefix(raw, "(") || !strings.HasSuffix(raw, ")") {
|
||||||
|
return -1, -1, false
|
||||||
|
}
|
||||||
|
parts := strings.Split(raw[1:len(raw)-1], ",")
|
||||||
|
if len(parts) != 2 {
|
||||||
|
return -1, -1, false
|
||||||
|
}
|
||||||
|
r1 := arm64RegNum(strings.TrimSpace(parts[0]))
|
||||||
|
r2 := arm64RegNum(strings.TrimSpace(parts[1]))
|
||||||
|
if r1 < 0 || r2 < 0 {
|
||||||
|
return -1, -1, false
|
||||||
|
}
|
||||||
|
return r1, r2, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeARM64Excl encodes the exclusive load/store family with the operand
|
||||||
|
// order the toolchain parses (cmd/internal/obj/arm64/asm7.go cases 58 and 59,
|
||||||
|
// and its own spellings in arm64enc.s):
|
||||||
|
//
|
||||||
|
// STXR Rt, (Rn), Rs store, single register
|
||||||
|
// STXP (Rt1, Rt2), (Rn), Rs store, register pair
|
||||||
|
// LDXR (Rn), Rt load, single register
|
||||||
|
// LDXP (Rn), (Rt1, Rt2) load, register pair
|
||||||
|
//
|
||||||
|
// Decoded toolchain evidence: `STXR R1, (R2), R3` assembles to 0xc8037c41,
|
||||||
|
// whose fields are Rs=3, Rn=2, Rt=1: the FIRST register operand is the data
|
||||||
|
// register and the LAST the status register.
|
||||||
func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||||
// LDXR/STXR have different operand forms.
|
|
||||||
isLoad := strings.HasPrefix(mnem, "LD")
|
isLoad := strings.HasPrefix(mnem, "LD")
|
||||||
if isLoad {
|
if isLoad {
|
||||||
// LDXR (Rn), Rt → 2 operands: mem, reg
|
// LDXR (Rn), Rt / LDXP (Rn), (Rt1, Rt2): 2 operands.
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||||
}
|
}
|
||||||
@@ -1369,25 +1402,34 @@ func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
if rt1, rt2, ok := arm64PairOf(ops[1]); ok {
|
||||||
|
// The single-register opcodes pre-set the unused Rs (bits 20:16)
|
||||||
|
// and Rt2 (bits 14:10) fields to 31; the pair forms carry a real
|
||||||
|
// Rt2 and keep Rs at 31.
|
||||||
|
return a64wordLE(baseOp | 0x1F<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
||||||
|
}
|
||||||
rt := arm64RegNum(operandRegName(ops[1]))
|
rt := arm64RegNum(operandRegName(ops[1]))
|
||||||
if rt < 0 {
|
if rt < 0 {
|
||||||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||||||
}
|
}
|
||||||
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil
|
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil
|
||||||
}
|
}
|
||||||
// STXR Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
|
// STXR Rt, (Rn), Rs / STXP (Rt1, Rt2), (Rn), Rs: 3 operands.
|
||||||
if len(ops) != 3 {
|
if len(ops) != 3 {
|
||||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||||
}
|
}
|
||||||
rs := arm64RegNum(operandRegName(ops[0]))
|
|
||||||
if rs < 0 {
|
|
||||||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
|
||||||
}
|
|
||||||
rn, err := arm64ExclMem(mnem, ops[1])
|
rn, err := arm64ExclMem(mnem, ops[1])
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
rt := arm64RegNum(operandRegName(ops[2]))
|
rs := arm64RegNum(operandRegName(ops[2]))
|
||||||
|
if rs < 0 {
|
||||||
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||||||
|
}
|
||||||
|
if rt1, rt2, ok := arm64PairOf(ops[0]); ok {
|
||||||
|
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
||||||
|
}
|
||||||
|
rt := arm64RegNum(operandRegName(ops[0]))
|
||||||
if rt < 0 {
|
if rt < 0 {
|
||||||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||||||
}
|
}
|
||||||
|
|||||||
+23
-3
@@ -98,8 +98,12 @@ func arm64RegNum(name string) int {
|
|||||||
return 30
|
return 30
|
||||||
case "R31", "ZR":
|
case "R31", "ZR":
|
||||||
return 31
|
return 31
|
||||||
case "SP":
|
case "SP", "RSP":
|
||||||
return 31 // SP and ZR share encoding 31; context determines meaning
|
// RSP is the toolchain's spelling for register 31 (it rejects
|
||||||
|
// R31 in an operand); SP stays for sources that spell it the
|
||||||
|
// amd64 way. SP and ZR share encoding 31; context determines
|
||||||
|
// the meaning.
|
||||||
|
return 31
|
||||||
}
|
}
|
||||||
// F0-F31.
|
// F0-F31.
|
||||||
if len(name) >= 1 && name[0] == 'F' {
|
if len(name) >= 1 && name[0] == 'F' {
|
||||||
@@ -282,7 +286,7 @@ const (
|
|||||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||||
a64FCRC32 // CRC32
|
a64FCRC32 // CRC32
|
||||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
|
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
||||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||||
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
|
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
|
||||||
)
|
)
|
||||||
@@ -575,6 +579,10 @@ func init() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// ---- exclusive load/store ----
|
// ---- exclusive load/store ----
|
||||||
|
// Single-register forms pre-set the unused Rs and Rt2 fields to 31 (the
|
||||||
|
// 0x7c00/0x1f0000 halves of the constants below); the register-pair
|
||||||
|
// forms carry a real Rt2 in bits 14:10, so their opcodes pre-set
|
||||||
|
// neither field.
|
||||||
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
|
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
|
||||||
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
|
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
|
||||||
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
|
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
|
||||||
@@ -583,6 +591,12 @@ func init() {
|
|||||||
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
|
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
|
||||||
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
|
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
|
||||||
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
|
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
|
||||||
|
// Pair loads, LDSTX(sz, 0, l=1, o1=1, o0) in asm7.go: LDXP/ LDXPW have
|
||||||
|
// o0=0, LDAXP/LDAXPW o0=1 (bit 15). Rs (bits 20:16) stays 31.
|
||||||
|
a64InstrTable["LDXP"] = a64Enc{format: a64FExcl, op: 0xc8600000}
|
||||||
|
a64InstrTable["LDXPW"] = a64Enc{format: a64FExcl, op: 0x88600000}
|
||||||
|
a64InstrTable["LDAXP"] = a64Enc{format: a64FExcl, op: 0xc8608000}
|
||||||
|
a64InstrTable["LDAXPW"] = a64Enc{format: a64FExcl, op: 0x88608000}
|
||||||
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
|
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
|
||||||
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
|
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
|
||||||
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
|
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
|
||||||
@@ -591,6 +605,12 @@ func init() {
|
|||||||
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
|
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
|
||||||
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
|
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
|
||||||
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
|
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
|
||||||
|
// Pair stores, LDSTX(sz, 0, l=0, o1=1, o0): STXP/STXPW have o0=0,
|
||||||
|
// STLXP/STLXPW o0=1 (bit 15). Both Rs and Rt2 are real fields.
|
||||||
|
a64InstrTable["STXP"] = a64Enc{format: a64FExcl, op: 0xc8200000}
|
||||||
|
a64InstrTable["STXPW"] = a64Enc{format: a64FExcl, op: 0x88200000}
|
||||||
|
a64InstrTable["STLXP"] = a64Enc{format: a64FExcl, op: 0xc8208000}
|
||||||
|
a64InstrTable["STLXPW"] = a64Enc{format: a64FExcl, op: 0x88208000}
|
||||||
|
|
||||||
// ---- LSE atomics ----
|
// ---- LSE atomics ----
|
||||||
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
|
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
|
||||||
|
|||||||
@@ -820,15 +820,28 @@ func TestArm64ExclOffsetErrors(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestArm64ExclNoOffset pins the plain (Rn) forms. gasm parses the store
|
// TestArm64ExclNoOffset pins the plain (Rn) forms, byte-for-byte against
|
||||||
// with the status register first (ARM ARM order); go tool asm parses the
|
// go tool asm. The toolchain parses the FIRST register of a store as the
|
||||||
// same text with the data register first, so the two spellings differ and
|
// data register and the LAST as the status register (asm7.go case 59), and
|
||||||
// the store word below is gasm's own.
|
// the pair forms as (Rt1, Rt2) (case 58/59):
|
||||||
|
//
|
||||||
|
// STXR R3, (R1), R4 → c8047c23 (Rt=3, Rn=1, Rs=4)
|
||||||
|
// STXP (R3, R4), (R1), R5 → c8251023 (Rt=3, Rt2=4, Rn=1, Rs=5)
|
||||||
|
// LDXP (R1), (R3, R4) → c87f1023 (Rn=1, Rt=3, Rt2=4)
|
||||||
func TestArm64ExclNoOffset(t *testing.T) {
|
func TestArm64ExclNoOffset(t *testing.T) {
|
||||||
got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n")
|
got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n"+
|
||||||
|
"\tSTXP (R3, R4), (R1), R5\n\tSTXPW (R3, R4), (R1), R5\n"+
|
||||||
|
"\tLDXP (R1), (R3, R4)\n\tLDXPW (R1), (R3, R4)\n"+
|
||||||
|
"\tSTXR R3, (RSP), R4\n\tLDXR (RSP), R2\n")
|
||||||
want := []uint32{
|
want := []uint32{
|
||||||
0xc85f7c22, // LDXR X2, [X1]
|
0xc85f7c22, // LDXR X2, [X1]
|
||||||
0xc8037c24, // STXR W3, X4, [X1] with Rs = R3, Rt = R4
|
0xc8047c23, // STXR W3, [X1], W4 with Rt = R3, Rs = R4
|
||||||
|
0xc8251023, // STXP (R3, R4), [X1], R5
|
||||||
|
0x88251023, // STXPW (R3, R4), [X1], R5
|
||||||
|
0xc87f1023, // LDXP [X1], (R3, R4)
|
||||||
|
0x887f1023, // LDXPW [X1], (R3, R4)
|
||||||
|
0xc8047fe3, // STXR R3, [SP], R4
|
||||||
|
0xc85f7fe2, // LDXR [SP], R2
|
||||||
0xd65f03c0, // RET
|
0xd65f03c0, // RET
|
||||||
}
|
}
|
||||||
for i := range want {
|
for i := range want {
|
||||||
@@ -920,3 +933,58 @@ func TestArm64LargeFrameSpadj(t *testing.T) {
|
|||||||
t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got)
|
t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestArm64SplitFrameSpadj pins the addcon2 band, where neither imm12 form
|
||||||
|
// nor a single MOVZ carries the autosize and the toolchain splits the
|
||||||
|
// prologue SUB into two imm12 instructions (asm7.go case 48) while the
|
||||||
|
// non-leaf RET still materialises the value into REGTMP (obj7.go ARET,
|
||||||
|
// issue 73259). $65664 rounds the autosize to 65680 = 144 + 16<<12:
|
||||||
|
//
|
||||||
|
// [SUB $144, RSP, R20][SUB $(16<<12), R20, R20][STP][MOVD R20, SP][SUB $8]
|
||||||
|
// [CALL]
|
||||||
|
// [LDP][MOVD $144, R27][MOVK $(1<<16), R27][ADD R27, RSP, RSP][RET]
|
||||||
|
//
|
||||||
|
// SP moves at the fourth word (byte 12) and returns to zero at the final
|
||||||
|
// RET (byte 40); the words are go tool asm's own for the same source.
|
||||||
|
func TestArm64SplitFrameSpadj(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), NOSPLIT, $65664-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
fn := img.Funcs[0]
|
||||||
|
wantSpadj := []SpadjStep{{PC: 12, Value: 65680}, {PC: 40, Value: 0}}
|
||||||
|
if len(fn.Spadj) != len(wantSpadj) {
|
||||||
|
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
|
||||||
|
}
|
||||||
|
for i := range wantSpadj {
|
||||||
|
if fn.Spadj[i] != wantSpadj[i] {
|
||||||
|
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []uint32{
|
||||||
|
0xd10243f4, // SUB $144, RSP, R20
|
||||||
|
0xd1404294, // SUB $(16<<12), R20, R20
|
||||||
|
0xa93ffa9d, // STP (R29, R30), -8(R20)
|
||||||
|
0x9100029f, // MOVD R20, RSP
|
||||||
|
0xd10023fd, // SUB $8, RSP, R29
|
||||||
|
0x94000000, // CALL (relocation masked at link time)
|
||||||
|
0xa97ffbfd, // LDP -8(RSP), (R29, R30)
|
||||||
|
0xd280121b, // MOVD $144, R27
|
||||||
|
0xf2a0003b, // MOVK $(1<<16), R27
|
||||||
|
0x8b3b63ff, // ADD R27, RSP, RSP
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
|
||||||
|
if len(words) != len(want) {
|
||||||
|
t.Fatalf("framed = %d words, want %d", len(words), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
if words[i] != w {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, words[i], w)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+75
-21
@@ -187,10 +187,32 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
|
|||||||
return a64WordsLE(ws...)
|
return a64WordsLE(ws...)
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64SubImmWords emits SUB $imm, SP, Rd: the immediate form when the value
|
// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
|
||||||
// fits the imm12 field (plain, or shifted left by 12 when it is a multiple
|
// two imm12 instructions instead of materialising it into REGTMP
|
||||||
// of 4096); otherwise the toolchain materialises it into REGTMP (R27) and
|
// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
|
||||||
// subtracts the register in the extended-register form.
|
// unsigned and be neither encodable as one imm12 (checked by the callers
|
||||||
|
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
|
||||||
|
// logical immediate, because conclass tests all three before C_ADDCON2.
|
||||||
|
func arm64SplitImm12(imm uint32) bool {
|
||||||
|
if imm > 0xFFFFFF {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
|
||||||
|
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
|
||||||
|
// immediate form when the value fits imm12 (plain, or shifted left by 12
|
||||||
|
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
|
||||||
|
// logical immediate, or one wider than 24 bits is materialised into REGTMP
|
||||||
|
// (R27) and subtracted in the extended-register form; everything else up to
|
||||||
|
// 0xFFFFFF is split into two imm12 instructions:
|
||||||
|
//
|
||||||
|
// SUB $(imm&0xfff), SP, Rd
|
||||||
|
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
|
||||||
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
||||||
if imm <= 0xFFF {
|
if imm <= 0xFFF {
|
||||||
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
|
||||||
@@ -198,15 +220,21 @@ func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
|
|||||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||||
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
|
||||||
}
|
}
|
||||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
if !arm64SplitImm12(imm) {
|
||||||
if err != nil {
|
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||||
mov = nil
|
if err != nil {
|
||||||
|
mov = nil
|
||||||
|
}
|
||||||
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
||||||
|
}
|
||||||
|
return []uint32{
|
||||||
|
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
|
||||||
|
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||||
}
|
}
|
||||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12
|
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
|
||||||
// and REGTMP fallback ladder.
|
// split and REGTMP ladder as arm64SubImmWords.
|
||||||
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
||||||
if imm <= 0xFFF {
|
if imm <= 0xFFF {
|
||||||
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
|
||||||
@@ -214,11 +242,35 @@ func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
|
|||||||
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
if imm <= 4095<<12 && imm&0xFFF == 0 {
|
||||||
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
|
||||||
}
|
}
|
||||||
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
if !arm64SplitImm12(imm) {
|
||||||
|
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
|
||||||
|
if err != nil {
|
||||||
|
mov = nil
|
||||||
|
}
|
||||||
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
||||||
|
}
|
||||||
|
return []uint32{
|
||||||
|
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
|
||||||
|
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
|
||||||
|
// large frame. The toolchain adds the frame back with a single instruction:
|
||||||
|
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
|
||||||
|
// materialised into REGTMP and added as a register, so the epilogue never
|
||||||
|
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
|
||||||
|
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
|
||||||
|
// the leaf epilogue's plain ADD instructions.
|
||||||
|
func arm64RetAddWords(autosize uint32) []uint32 {
|
||||||
|
if autosize < 1<<12 {
|
||||||
|
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
|
||||||
|
}
|
||||||
|
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
mov = nil
|
mov = nil
|
||||||
}
|
}
|
||||||
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
|
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
|
||||||
@@ -237,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte {
|
|||||||
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
// Large frame: LDP -8(SP), (FP, LR), then deallocate.
|
||||||
ws = append(ws,
|
ws = append(ws,
|
||||||
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
|
||||||
)
|
)
|
||||||
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
|
ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// RET: BR LR (0xd65f03c0)
|
// RET: BR LR (0xd65f03c0)
|
||||||
@@ -260,15 +312,16 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
|
|||||||
}
|
}
|
||||||
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
|
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
|
||||||
// position depends on how many words the SUB itself took (immediate,
|
// position depends on how many words the SUB itself took (immediate,
|
||||||
// shifted immediate, or a materialised REGTMP sequence).
|
// shifted immediate, the two-word imm12 split, or a materialised REGTMP
|
||||||
|
// sequence).
|
||||||
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
|
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
|
||||||
// (but not including) the final RET instruction. The ADD sequences share the
|
// (but not including) the final RET instruction. The lengths are read from
|
||||||
// prologue's immediate ladder, so their length is read from the same helper
|
// the same word-emitting helpers the epilogue uses rather than assumed: the
|
||||||
// rather than assumed: a materialised autosize costs its MOV words plus the
|
// leaf path shares the prologue's immediate ladder, and a materialised
|
||||||
// ADD itself.
|
// autosize costs its MOV words plus the ADD itself.
|
||||||
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
||||||
if fi.autosize == 0 {
|
if fi.autosize == 0 {
|
||||||
return 0
|
return 0
|
||||||
@@ -280,8 +333,9 @@ func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
|
|||||||
if fi.autosize <= 0xf0 {
|
if fi.autosize <= 0xf0 {
|
||||||
return 8 // LDR + LDR.P
|
return 8 // LDR + LDR.P
|
||||||
}
|
}
|
||||||
// LDP + the ADD ladder that deallocates the frame.
|
// LDP + the deallocation emitted by arm64RetAddWords, so the length
|
||||||
return 4 + 4*len(arm64AddImmWords(uint32(fi.autosize), 31))
|
// tracks whatever the MOVD ladder needs.
|
||||||
|
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
|
||||||
}
|
}
|
||||||
|
|
||||||
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
// arm64ResolvePseudo translates a pseudo-register memory reference into a
|
||||||
|
|||||||
Vendored
+69
@@ -0,0 +1,69 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// Large frames across every immediate band of the prologue SUB and the RET
|
||||||
|
// epilogue, byte-parity-checked against go tool asm:
|
||||||
|
//
|
||||||
|
// $5000 autosize 5024 one 16-bit chunk, materialised into REGTMP
|
||||||
|
// $65664 autosize 65680 split into two imm12 instructions
|
||||||
|
// $70000 autosize 70016 split into two imm12 instructions
|
||||||
|
// $65520 autosize 65536 one shifted imm12 in the prologue, a logical
|
||||||
|
// immediate (ORR) in the non-leaf epilogue
|
||||||
|
// $16777232 autosize 16777248 wider than 24 bits, MOVZ/MOVK into REGTMP
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·leaf5000(SB), NOSPLIT, $5000-0
|
||||||
|
MOVD R0, R1
|
||||||
|
MOVD R1, R2
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·leaf65664(SB), NOSPLIT, $65664-0
|
||||||
|
MOVD R0, R1
|
||||||
|
MOVD R1, R2
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·leaf70000(SB), NOSPLIT, $70000-0
|
||||||
|
MOVD R0, R1
|
||||||
|
MOVD R1, R2
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·leaf65520(SB), NOSPLIT, $65520-0
|
||||||
|
MOVD R0, R1
|
||||||
|
MOVD R1, R2
|
||||||
|
MOVD R2, R3
|
||||||
|
MOVD R3, R4
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·nl5000(SB), NOSPLIT, $5000-0
|
||||||
|
MOVD R0, R1
|
||||||
|
MOVD R1, R2
|
||||||
|
CALL ·other(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·nl65664(SB), NOSPLIT, $65664-0
|
||||||
|
MOVD R0, R1
|
||||||
|
CALL ·other(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·nl70000(SB), NOSPLIT, $70000-0
|
||||||
|
MOVD R0, R1
|
||||||
|
CALL ·other(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·nl65520(SB), NOSPLIT, $65520-0
|
||||||
|
MOVD R0, R1
|
||||||
|
MOVD R1, R2
|
||||||
|
MOVD R2, R3
|
||||||
|
CALL ·other(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·nlhuge(SB), NOSPLIT, $16777232-0
|
||||||
|
CALL ·other(SB)
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·other(SB), NOSPLIT, $0-0
|
||||||
|
MOVD R0, R1
|
||||||
|
MOVD R1, R2
|
||||||
|
MOVD R2, R3
|
||||||
|
RET
|
||||||
Vendored
+81
@@ -0,0 +1,81 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// Exclusive load/store family, LSE atomics and their memory operands,
|
||||||
|
// byte-parity-checked against go tool asm. The toolchain parses the FIRST
|
||||||
|
// register of an exclusive store as the data register and the LAST as the
|
||||||
|
// status register, and takes register pairs as (Rt1, Rt2) operands.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
TEXT ·loads(SB), NOSPLIT, $0-0
|
||||||
|
LDXR (R1), R2
|
||||||
|
LDXRB (R1), R3
|
||||||
|
LDXRH (R1), R4
|
||||||
|
LDXRW (R1), R5
|
||||||
|
LDAXR (R1), R2
|
||||||
|
LDAXRB (R1), R3
|
||||||
|
LDAXRH (R1), R4
|
||||||
|
LDAXRW (R1), R5
|
||||||
|
LDXR (RSP), R2
|
||||||
|
LDXRB (RSP), R3
|
||||||
|
LDXRH (RSP), R4
|
||||||
|
LDXRW (RSP), R5
|
||||||
|
LDAXR (RSP), R2
|
||||||
|
LDAXRB (RSP), R3
|
||||||
|
LDAXRH (RSP), R4
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·stores(SB), NOSPLIT, $0-0
|
||||||
|
STXR R2, (R1), R6
|
||||||
|
STXRB R3, (R1), R6
|
||||||
|
STXRH R4, (R1), R6
|
||||||
|
STXRW R5, (R1), R6
|
||||||
|
STLXR R2, (R1), R6
|
||||||
|
STLXRB R3, (R1), R6
|
||||||
|
STLXRH R4, (R1), R6
|
||||||
|
STLXRW R5, (R1), R6
|
||||||
|
STXR R2, (RSP), R6
|
||||||
|
STXRB R3, (RSP), R6
|
||||||
|
STXRH R4, (RSP), R6
|
||||||
|
STXRW R5, (RSP), R6
|
||||||
|
STLXR R2, (RSP), R6
|
||||||
|
STLXRB R3, (RSP), R6
|
||||||
|
STLXRH R4, (RSP), R6
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·pairs(SB), NOSPLIT, $0-0
|
||||||
|
LDXP (R1), (R2, R3)
|
||||||
|
LDXPW (R1), (R2, R3)
|
||||||
|
LDAXP (R1), (R2, R3)
|
||||||
|
LDAXPW (R1), (R2, R3)
|
||||||
|
STXP (R2, R3), (R1), R6
|
||||||
|
STXPW (R2, R3), (R1), R6
|
||||||
|
STLXP (R2, R3), (R1), R6
|
||||||
|
STLXPW (R2, R3), (R1), R6
|
||||||
|
LDXP (RSP), (R2, R3)
|
||||||
|
LDXPW (RSP), (R2, R3)
|
||||||
|
LDAXP (RSP), (R4, R5)
|
||||||
|
LDAXPW (RSP), (R4, R5)
|
||||||
|
STXP (R2, R3), (RSP), R6
|
||||||
|
STXPW (R2, R3), (RSP), R6
|
||||||
|
STLXP (R4, R5), (RSP), R7
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·atomics(SB), NOSPLIT, $0-0
|
||||||
|
LDADDB R2, (R1), R3
|
||||||
|
LDADDH R2, (R1), R3
|
||||||
|
LDADDW R2, (R1), R3
|
||||||
|
LDADDD R2, (R1), R3
|
||||||
|
LDADDB R2, (R1), ZR
|
||||||
|
LDADDH R2, (R1), ZR
|
||||||
|
LDADDW R2, (R1), ZR
|
||||||
|
LDADDD R2, (R1), ZR
|
||||||
|
CASW R2, (R1), R3
|
||||||
|
CASD R2, (R1), R3
|
||||||
|
CASW R2, (R1), ZR
|
||||||
|
CASD R2, (R1), ZR
|
||||||
|
SWPW R2, (R1), R3
|
||||||
|
SWPD R2, (R1), R3
|
||||||
|
SWPD R2, (R1), ZR
|
||||||
|
RET
|
||||||
Reference in New Issue
Block a user