From d786b90fa163c3f7831fc6498c1a19a356638de3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 00:43:06 +0200 Subject: [PATCH] feat(asm): lower the arm64 con(register) form to the ADD chain Assisted-by: GLM 5.3 Flash --- asm/arm64_assemble.go | 133 ++++++++++++++++++++++++++++++++++++++- asm/arm64_encode_test.go | 62 ++++++++++++++++++ 2 files changed, 193 insertions(+), 2 deletions(-) diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 199d911..901c453 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -1602,6 +1602,24 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo, } return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil } + rd := arm64RegNum(operandRegName(dst)) + // The con(register) form: MOVD $con(Rn), Rd adds the displacement to + // the base register (asm7.go case 4). The toolchain rejects every + // other width and the ZR destination outright (RSP is a real register + // here, the C_RSP row). + if rn, ok := arm64ImmWithBase(src); ok { + if mnem != "MOVD" && mnem != "MOV" { + return nil, fmt.Errorf("%s: illegal combination: the con(register) form exists for MOVD only", mnem) + } + if strings.EqualFold(operandRegName(dst), "ZR") { + return nil, fmt.Errorf("%s: illegal combination: the con(register) form needs a real destination register", mnem) + } + if rd < 0 { + return nil, fmt.Errorf("%s $con(Rn): invalid destination register", mnem) + } + con, _ := arm64ImmOperandValue(src) + return encodeARM64ConRn(rn, con, rd, pool, poolBase, pc) + } // Immediate → memory: only storing zero is encodable (the ZR // register); the toolchain rejects any other immediate-to-memory // combination ("illegal combination"). @@ -1611,7 +1629,6 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo, } return encodeARM64MemOp(mnem, dst, 31, false, fi, "", pool, poolBase, pc) } - rd := arm64RegNum(operandRegName(dst)) if rd < 0 { return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) } @@ -1712,6 +1729,13 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { return 8 // ADRP + ADD } + // The con(register) form lowers to the toolchain's ADD/SUB chain: + // one word in the addcon band, two in the 24-bit band, and the two + // pool words (LDR X plus the UXTX add) beyond it. + if _, ok := arm64ImmWithBase(src); ok { + con, _ := arm64ImmOperandValue(src) + return arm64ConRnSize(con) + } if isMemOperand(dst) { // Only the $0 (ZR store) immediate reaches memory, in one word. return 4 @@ -2207,11 +2231,99 @@ func arm64OffsetSplitReach(off int64, lt a64LSType) bool { // arm64AddImmWord encodes ADD $v, Rn, Rd the way the toolchain's oaddi // does: a non-zero multiple of 0x1000 encodes shifted left by twelve. func arm64AddImmWord(v int64, rn, rd uint32) uint32 { + return arm64AddSubImmWord(0, v, rn, rd) +} + +// arm64AddSubImmWord encodes ADD (op 0) or SUB (op 1) $v, Rn, Rd the way the +// toolchain's oaddi does: a non-zero multiple of 0x1000 encodes shifted left +// by twelve. +func arm64AddSubImmWord(op uint32, v int64, rn, rd uint32) uint32 { sh := arm64AddShift(v) if sh == 1 { v >>= 12 } - return a64AddSub(1, 0, 0, sh, uint32(v), rn, rd) + return a64AddSub(1, op, 0, sh, uint32(v), rn, rd) +} + +// arm64ImmWithBase reports whether an immediate operand spells the +// con(register) form, $con(REG): the parser leaves it unstructured (an +// immediate whose raw spelling carries the parenthesised register), so the +// base register comes off the raw text while the constant rides the parsed +// immediate. ok is false for every other shape. +func arm64ImmWithBase(op *ast.Operand) (rn int, ok bool) { + if op.Kind != 0 || op.Imm.Sym != nil { + return 0, false + } + s := strings.Join(strings.Fields(op.Raw), " ") + if !strings.HasSuffix(s, ")") { + return 0, false + } + i := strings.LastIndex(s, "(") + if i < 0 { + return 0, false + } + reg := strings.TrimSpace(s[i+1 : len(s)-1]) + rn = arm64RegNum(reg) + if rn < 0 { + return 0, false + } + return rn, true +} + +// arm64ConRnSize sizes the con(register) lowering of encodeARM64ConRn +// without touching the pool: the bands mirror the encoder exactly. +func arm64ConRnSize(con int64) int { + a := con + if a < 0 { + a = -a + } + if a <= 0xfff || (a&0xfff == 0 && a <= 0xfff000) { + return 4 + } + if con >= 0 && a <= 0xffffff { + return 8 + } + return 8 // LDR X, pool + the UXTX add +} + +// encodeARM64ConRn lowers MOVD $con(Rn), Rd the way the toolchain's case 4 +// does: a single ADD/SUB immediate inside the addcon band (±4095 or a +// multiple of 4096 up to 0xfff<<12), the hi<<12 plus lo pair for a positive +// displacement inside the 24-bit band, and the literal pool plus a UXTX add +// beyond either (a negative displacement outside the addcon band pools +// straight away: isaddcon2 only takes non-negative values). +func encodeARM64ConRn(rn int, con int64, rd int, pool *arm64Pool, poolBase, pc int) ([]byte, error) { + op := uint32(0) // ADD + a := con + if a < 0 { + op = 1 // SUB + a = -a + } + if a <= 0xfff || (a&0xfff == 0 && a <= 0xfff000) { + return a64wordLE(arm64AddSubImmWord(op, a, uint32(rn), uint32(rd))), nil + } + if con >= 0 && a <= 0xffffff { + hi := a & 0xfff000 + lo := a & 0xfff + return a64WordsLE( + arm64AddSubImmWord(op, hi, uint32(rn), uint32(rd)), + arm64AddSubImmWord(op, lo, uint32(rd), uint32(rd)), + ), nil + } + // Beyond the bands the toolchain pools the displacement (a full 64-bit + // slot) and adds it back with a UXTX-extended register add (case 34). + if pool == nil { + return nil, fmt.Errorf("MOVD $%d(R%d): displacement out of range (literal pool not supported)", con, rn) + } + entryOff, w := pool.add64(con) + dist := (poolBase + entryOff - pc) >> 2 + if dist < -(1<<18) || dist >= 1<<18 { + return nil, fmt.Errorf("MOVD $%d(R%d): literal pool %d out of 19-bit reach", con, rn, dist<<2) + } + return a64WordsLE( + w<<30|3<<27|uint32(dist)&0x7FFFF<<5|27, // LDR R27, pool + a64AddSubReg(1, 27, uint32(rn), uint32(rd)), + ), nil } // ---- static symbol references (ADRP + offset) ---- @@ -4911,6 +5023,23 @@ func (p *arm64Pool) add(v int64) (int, uint32) { return off, w } +// add64 reserves an 8-byte slot for v loaded by a full LDR X: the lacon +// pool path always reads 64 bits, even when the value fits 32 (asm7.go case +// 34's omovlit(AMOVD)). +func (p *arm64Pool) add64(v int64) (int, uint32) { + if i, ok := p.seen[v]; ok && p.order[i].w == 1 { + return p.order[i].off, p.order[i].w + } + off := (p.size + 7) &^ 7 + p.size = off + 8 + if p.seen == nil { + p.seen = map[int64]int{} + } + p.seen[v] = len(p.order) + p.order = append(p.order, arm64PoolEntry{data: a64WordsLE(uint32(v), uint32(v>>32)), off: off, w: 1}) + return off, 1 +} + // AssembleFileARM64 assembles every TEXT function of a parsed arm64 file // and lays out its static symbols (GLOBL/DATA) in a data section behind the // code. SB references in the code are encoded as ADRP pairs with zero diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index 9d96285..767486a 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -1978,3 +1978,65 @@ func TestArm64SimdArrangementRejections(t *testing.T) { } } } + + +// TestArm64ConRn pins the MOVD $con(Rn), Rd lowering against `go tool asm` +// words: the single ADD/SUB inside the addcon band, the hi<<12 plus lo pair +// in the 24-bit band, and the pool plus UXTX add beyond it. +func TestArm64ConRn(t *testing.T) { + got := arm64Words(t, + "\tMOVD $0x1002(RSP), R1\n"+ + "\tMOVD $0x1708(RSP), RSP\n"+ + "\tMOVD $0x2001(R7), R1\n"+ + "\tMOVD $0xffffff(R7), R1\n"+ + "\tMOVD $-1(R7), R1\n"+ + "\tMOVD $-0x30(R7), R1\n"+ + "\tMOVD $-0x2000(RSP), R1\n"+ + "\tMOVD $-0x10000(RSP), RSP\n"+ + "\tMOVD $0(R7), R1\n"+ + "\tMOVD $4096(R7), R1\n"+ + "\tMOVD $5(R3), R1\n") + want := []uint32{ + 0x914007e1, // ADD $(1<<12), RSP, R1 + 0x91000821, // ADD $2, R1, R1 + 0x914007ff, // ADD $(1<<12), RSP, RSP + 0x911c23ff, // ADD $0x708, RSP, RSP + 0x914008e1, // ADD $(2<<12), R7, R1 + 0x91000421, // ADD $1, R1, R1 + 0x917ffce1, // ADD $(4095<<12), R7, R1 + 0x913ffc21, // ADD $4095, R1, R1 + 0xd10004e1, // SUB $1, R7, R1 + 0xd100c0e1, // SUB $0x30, R7, R1 + 0xd1400be1, // SUB $(2<<12), RSP, R1 + 0xd14043ff, // SUB $(16<<12), RSP, RSP + 0x910000e1, // ADD $0, R7, R1 + 0x914004e1, // ADD $(1<<12), R7, R1 + 0x91001461, // ADD $5, R3, R1 + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64ConRnRejections pins the shapes the toolchain refuses for the +// con(register) form: every other width and the ZR destination. +func TestArm64ConRnRejections(t *testing.T) { + for _, src := range []string{ + "\tMOVW\t$5(R3), R1\n", + "\tMOVD\t$5(R7), ZR\n", + } { + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n") + if len(errs) > 0 { + continue // a parse rejection is a rejection + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("expected rejection for %q, got nil", strings.TrimSpace(src)) + } + } +}