From 5f6be4584da14fddd99dc608588cd37e6d57ff89 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 20:59:14 +0200 Subject: [PATCH] fix(asm): encode the flag-setting logicals to ZR and fold immediate expressions Assisted-by: GLM 5.3 Flash --- asm/arm64_assemble.go | 67 +++++++++++++++++++++------------------- asm/arm64_encode_test.go | 46 +++++++++++++++++++++++++++ 2 files changed, 82 insertions(+), 31 deletions(-) diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 591e9a0..17b6d28 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -477,10 +477,11 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { return arm64MovSize(mnem, ops, fi) } // The logical-immediate family: one word on the bitmask fast path (a - // real destination, or the flags-only TST spellings), otherwise the - // constant materialisation into REGTMP plus the register-form tail. - // Mirrors encodeARM64DPSR's decision exactly, the materialisation word - // count included: a MOVZ plus up to three MOVKs makes five words. + // real destination, the flags-only TST spellings, or the flag-setting + // forms to ZR), otherwise the constant materialisation into REGTMP plus + // the register-form tail. Mirrors encodeARM64DPSR's decision exactly, + // the materialisation word count included: a MOVZ plus up to three MOVKs + // makes five words. if len(ops) >= 2 && len(ops) <= 3 && isImmOperand(ops[0]) { switch mnem { case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW", @@ -498,11 +499,12 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { width = 32 mwMnem = "MOVW" } + sLogical := mnem == "ANDS" || mnem == "ANDSW" || mnem == "BICS" || mnem == "BICSW" zrDest := false if mnem != "TST" && mnem != "TSTW" { zrDest = strings.EqualFold(operandRegName(ops[len(ops)-1]), "ZR") } - if _, _, _, bc := a64LogicalImm(v, width); bc && (mnem == "TST" || mnem == "TSTW" || !zrDest) { + if _, _, _, bc := a64LogicalImm(v, width); bc && (sLogical || mnem == "TST" || mnem == "TSTW" || !zrDest) { return 4 } if mw, err := encodeARM64LoadImm(27, written, mwMnem); err == nil { @@ -1110,20 +1112,22 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er return nil, fmt.Errorf("%s: illegal combination: the destination cannot be RSP", mnem) } n, immr, imms, ok := a64LogicalImm(v, width) - // The toolchain.s logical-immediate rows take a real destination - // only (omovconst guards the bitmask path with rt != REGZERO): a - // non-flag-setting logical to ZR materialises the constant into - // REGTMP (R27) and takes the register form. RSP is a real - // register here, but keeps the fast bitmask path inside the - // addcon band alone, and the flags-only TST spellings keep the - // fast path: ANDS ZR, Rn, #imm is their form. + // The toolchain's logical-immediate rows take a real destination + // only for the non-flag-setting forms (omovconst guards the + // bitmask path with rt != REGZERO): a plain AND/ORR/EOR to ZR + // materialises the constant into REGTMP (R27) and takes the + // register form. The flag-setting forms (ANDS, BICS, the TST + // spellings) keep the fast path with a ZR destination: case 53 + // encodes ANDS ZR, Rn, #imm directly. RSP is a real register + // here, but keeps the fast bitmask path inside the addcon band + // alone. if rspDest { inBand := writtenImm > 0 && (writtenImm <= 0xFFF || (writtenImm&0xFFF == 0 && writtenImm>>12 <= 0xFFF)) if !ok || !inBand { return nil, fmt.Errorf("%s: illegal combination: the destination cannot be RSP", mnem) } } - if ok && (isCmp || !zrDest) { + if ok && (isCmp || !zrDest || sLogical) { opc := (baseOp >> 29) & 7 sf := (baseOp >> 31) & 1 return a64wordLE(sf<<31 | opc<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | @@ -2895,16 +2899,11 @@ func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ( // ---- operand helpers ---- -// arm64Imm64 returns the full 64-bit immediate value of an operand. +// arm64Imm64 returns the full 64-bit immediate value of an operand, the +// raw-spelling-aware value arm64ImmOperandValue recovers. func arm64Imm64(op *ast.Operand) int64 { - if op.Imm.HasVal { - v := op.Imm.Val - if op.Imm.Neg { - v = -v - } - return v - } - return 0 + v, _ := arm64ImmOperandValue(op) + return v } // arm64ImmOperandValue returns the immediate an operand stands for, falling @@ -2917,6 +2916,15 @@ func arm64ImmOperandValue(op *ast.Operand) (int64, bool) { if op.Imm.Neg { v = -v } + // The parser folds a leading literal and drops the arithmetic that + // trails it ($14*16 parses as 14), so an operator-bearing spelling + // re-evaluates in full: the raw text is the statement's truth. + s := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$")) + if s != "" && strings.ContainsAny(s, "+-*/^|") { + if e, ok := arm64EvalExpr(s); ok { + return e, true + } + } return v, true } s := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$")) @@ -3069,7 +3077,7 @@ func arm64EvalExpr(s string) (int64, bool) { if t[0] < '0' || t[0] > '9' { return 0, false } - v, err := strconv.ParseInt(strings.TrimPrefix(strings.TrimPrefix(t, "0X"), "0x"), 0, 64) + v, err := strconv.ParseInt(t, 0, 64) if err != nil { return 0, false } @@ -3078,16 +3086,13 @@ func arm64EvalExpr(s string) (int64, bool) { } var binop func(minLevel int) (int64, bool) level := func(op string) int { + // Go's own precedence (the toolchain's evaluator folds with Go + // semantics): multiply, shift and AND bind tighter than add, OR and + // XOR. switch op { - case "|", "^": - return 1 - case "&": - return 2 - case "<<", ">>": - return 3 - case "*": + case "|", "^", "+", "-": return 4 - case "+", "-": + case "&", "<<", ">>", "*": return 5 } return 0 diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index ba0be4c..90bdef7 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -2596,3 +2596,49 @@ func TestArm64VTBLShapes(t *testing.T) { } } } + +// TestArm64LogicalToZR pins the toolchain's split between the flag-setting +// logicals, whose ZR destination keeps the bitmask fast path (asm7.go case +// 53), and the plain forms, which materialise into REGTMP: `go tool asm` +// encodes ANDSW $0x100, R13, ZR as one TSTW word and ANDW $1, R5, ZR as the +// ORRW-plus-register pair. +func TestArm64LogicalToZR(t *testing.T) { + got := arm64Words(t, "\tANDSW $0x100, R13, ZR\n\tBICSW $1, R5, ZR\n\tANDSW $0x101, R13, ZR\n\tANDW $1, R5, ZR\n") + want := []uint32{ + 0x721801bf, // ANDS (bitmask) R13, ZR: the TSTW word + 0x721f78bf, // BICS (bitmask of $1) R5, ZR + 0x5280203b, // MOVW $257, R27: not a bitmask, materialised + 0x6a1b01bf, // ANDSW R27, R13 + 0x320003fb, // ORRW $1, ZR, R27: the plain form materialises too + 0x0a1b00bf, // ANDW R27, R5, ZR + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64ImmediateExpression pins the folded arithmetic spellings the +// parser leaves half-parsed: `$14*16` parses as the leading literal 14, and +// the encoder must read the raw expression's 224 (`go tool asm` folds it). +func TestArm64ImmediateExpression(t *testing.T) { + got := arm64Words(t, "\tADD $14*16, R0\n\tMOVD $4*8+1, R1\n") + want := []uint32{ + 0x91038000, // ADD $224, R0 + 0xd2800421, // MOVZ $33, R1: MOVD $con rides MOVZ for a movcon value + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +}