diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 08faa69..d8a087d 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -330,11 +330,11 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { if strings.HasSuffix(mnem, "W") { width = 32 } - rd := 31 + zrDest := false if mnem != "TST" && mnem != "TSTW" { - rd = arm64RegNum(operandRegName(ops[len(ops)-1])) + zrDest = strings.EqualFold(operandRegName(ops[len(ops)-1]), "ZR") } - if _, _, _, bc := a64LogicalImm(v, width); bc && (rd != 31 || strings.HasPrefix(mnem, "TST")) { + if _, _, _, bc := a64LogicalImm(v, width); bc && (mnem == "TST" || mnem == "TSTW" || !zrDest) { return 4 } return 8 @@ -871,13 +871,16 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er width = 32 } var rn, rd int + zrDest := false switch len(ops) { case 3: rn = arm64RegNum(operandRegName(ops[1])) rd = arm64RegNum(operandRegName(ops[2])) + zrDest = strings.EqualFold(operandRegName(ops[2]), "ZR") default: rd = arm64RegNum(operandRegName(ops[1])) rn = rd + zrDest = strings.EqualFold(operandRegName(ops[1]), "ZR") } if isCmp { // CMP/CMN/TST write the flags alone: the destination is ZR @@ -891,9 +894,10 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er // The toolchain's logical-immediate rows take a real destination // only (omovconst guards the bitmask path with rt != REGZERO): a // non-flag-setting logical to ZR materialises the constant into - // REGTMP (R27) and takes the register form. The flags-only TST - // spellings keep the fast path: ANDS ZR, Rn, #imm is their form. - if ok && (rd != 31 || isCmp) { + // REGTMP (R27) and takes the register form. RSP is a real + // register here, and the flags-only TST spellings keep the fast + // path: ANDS ZR, Rn, #imm is their form. + if ok && (isCmp || !zrDest) { opc := (baseOp >> 29) & 7 sf := (baseOp >> 31) & 1 return a64wordLE(sf<<31 | opc<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | @@ -1694,6 +1698,11 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo, if rd < 0 { return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) } + if strings.EqualFold(operandRegName(dst), "ZR") { + // The destination is ZR: omovconst's bitmask path needs a real + // register, so the value rides MOVZ/MOVN. + return encodeARM64LoadImmClass(rd, arm64Imm64(src), mnem, false) + } return encodeARM64LoadImm(rd, arm64Imm64(src), mnem) } @@ -1806,7 +1815,7 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { // values expand to up to four words and the W forms truncate first. // Anything else would desynchronise the label offsets of pass 1 from // the bytes pass 2 lays down, corrupting every later branch. - b, err := encodeARM64LoadImm(31, arm64Imm64(src), mnem) + b, err := encodeARM64LoadImmClass(31, arm64Imm64(src), mnem, !strings.EqualFold(operandRegName(dst), "ZR")) if err != nil { return 4 } @@ -1860,6 +1869,14 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { // truncates to 0xFFFFFFFF, whose complement is a single zero chunk, and encodes // as MOVN W, #0. func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { + return encodeARM64LoadImmClass(rd, v, mnem, true) +} + +// encodeARM64LoadImmClass is encodeARM64LoadImm with the classification order +// in hand: bitmaskOK false skips the logical-immediate paths, which the +// toolchain's omovconst only takes for a real register (rt != REGZERO); an +// immediate to ZR rides the MOVZ/MOVN sequence carrying the value. +func encodeARM64LoadImmClass(rd int, v int64, mnem string, bitmaskOK bool) ([]byte, error) { d := v sf := uint32(1) // 64-bit if mnem == "MOVW" || mnem == "MOVWU" { @@ -1882,10 +1899,7 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { // `MOVD $4096, R27` is ORR $4096, not MOVZ $(1<<12) // - outside that band: MOVZ/MOVN first (C_MOVCON before C_BITCON), and // negative values reach MOVN before the bitmask test - // The bitmask path exists only for a real register: omovconst guards it - // with rt != REGZERO, so an immediate to ZR always rides the MOVZ/MOVN - // sequence carrying the value. - tryBitmaskFirst := rd != 31 && d > 0 && (d <= 0xFFF || (d&0xFFF == 0 && d <= 0xFFF000)) + tryBitmaskFirst := bitmaskOK && d > 0 && (d <= 0xFFF || (d&0xFFF == 0 && d <= 0xFFF000)) if tryBitmaskFirst { // Addcon-band immediate: try bitmask first (Go uses ORR for values @@ -1914,7 +1928,7 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { } // For values outside the bitmask-first range that are not movcon: try bitmask. - if !tryBitmaskFirst && rd != 31 { + if !tryBitmaskFirst && bitmaskOK { N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf)) if ok { return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index 6861fab..5512a6e 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -2157,3 +2157,31 @@ func TestArm64LogicalImmZR(t *testing.T) { } } } + +// TestArm64ZRNameNotNumber pins the spellings where ZR and RSP share the +// register number but not the class: the toolchain's omovconst guards the +// bitmask path with rt != REGZERO, so RSP keeps the fast logical and ORR +// forms while ZR materialises or rides MOVZ. +func TestArm64ZRNameNotNumber(t *testing.T) { + got := arm64Words(t, + "\tAND $8, R0, RSP\n"+ + "\tORR $8, R0, RSP\n"+ + "\tMOVW $0x10001000, RSP\n"+ + "\tADDW $0x10001000, R1\n") + want := []uint32{ + 0x927d001f, // AND X31(SP), X0, #bitmask: the fast path for RSP + 0xb27d001f, // ORR X31(SP), X0, #bitmask + 0x320483ff, // ORR W31(SP), WZR, #0x10001000 (the MOVW bitmask) + 0x320483fb, // ORR W27, WZR, #0x10001000 + 0x0b1b0021, // ADDW W1, W1, W27 + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +}