From b0f9071bf530c4ac478eb4b57d959b4dc191d283 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 21:17:20 +0200 Subject: [PATCH] feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering Assisted-by: GLM 5.3 Flash --- arch/arm64.go | 1 + asm/arm64_assemble.go | 160 +++++++++++++++++++++++++------ asm/arm64_encode.go | 5 + asm/arm64_encode_test.go | 94 ++++++++++++++++++ testdata/verify/bookkeep_arm64.s | 31 ++++++ testdata/verify/simdmove_arm64.s | 57 +++++++++++ 6 files changed, 318 insertions(+), 30 deletions(-) create mode 100644 testdata/verify/bookkeep_arm64.s create mode 100644 testdata/verify/simdmove_arm64.s diff --git a/arch/arm64.go b/arch/arm64.go index 0b1d7dd..4deb84b 100644 --- a/arch/arm64.go +++ b/arch/arm64.go @@ -33,6 +33,7 @@ func arm64Registers() []Register { for i := 0; i <= 30; i++ { add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register") } + add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)") add("ZR", Special, "zero register (reads as 0)") add("SP", Special, "stack pointer") add("LR", Special, "link register (alias of R30)") diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index e121887..48b58cc 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { return 4 } } - // The funcdata pseudo-statements contribute no bytes. + // The funcdata pseudo-statements contribute no bytes, the expanded + // FUNCDATA/PCDATA forms included. switch mnem { - case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED": + case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA": return 0 } switch mnem { @@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 // The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS, // GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine // code: the toolchain emits zero instruction bytes for them, and so does - // the encoder here. + // the encoder here. Files that include funcdata.h spell them after + // macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are + // bookkeeping too (the same treatment the loong64 encoder applies). switch mnem { case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED": return nil, nil + case "END": + if len(ops) != 0 { + return nil, fmt.Errorf("END expects no operands, got %d", len(ops)) + } + return nil, nil + case "FUNCDATA": + if len(ops) != 2 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)") + } + return nil, nil + case "PCDATA": + if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) { + return nil, fmt.Errorf("PCDATA expects $n, $n") + } + return nil, nil } // Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.). @@ -520,8 +538,17 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 // SIMD element moves (VDUP, VMOV with lane indices) take precedence // over the plain arrangement paths, which carry no index. - if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) { - return encodeARM64Dup(mnem, ops) + if mnem == "VDUP" || mnem == "VMOV" { + if arm64SimdHasElement(ops) { + return encodeARM64Dup(mnem, ops) + } + // VMOV/VDUP Rn, Vd.: a general register into an arranged whole + // vector (asm7.go case 82, shared by both mnemonics). The element + // paths above only run when a lane index is spelled, so this is the + // whole-vector shape's only route. + if b, ok, err := encodeARM64GPToVec(mnem, ops); ok { + return b, err + } } // Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1, @@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) { return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil } - // Integer → integer: ORR Rd, ZR, Rs. + // Integer → integer. Every truncating register move lowers to an + // extend in the toolchain (asm7.go case 45): the signed forms to SBFM + // (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM + // (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays + // ORR Xd, XZR, Xm. + if rs != 31 { + switch mnem { + case "MOVB": + return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil + case "MOVH": + return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil + case "MOVW": + return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil + case "MOVBU": + return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil + case "MOVHU": + return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil + } + } sf := uint32(1) // 64-bit - if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" || - mnem == "MOVH" || mnem == "MOVHU" { + if mnem == "MOVWU" { + sf = 0 + } + // A narrow move out of the zero register loses its width: the + // toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against + // WZR. MOVD and MOV keep the 64-bit form. + if rs == 31 && mnem != "MOVD" && mnem != "MOV" { sf = 0 } op := uint32(1<<29 | 0x0a<<24) // ORR @@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt } return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil } + // The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn + // with Vm in place (asm7.go case 89, r defaulting to rt). They exist + // for bare V registers alone: the arranged forms and every other + // three-register mnemonic are rejected outright. + if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") { + vm, ok1 := arm64VecOf(ops[0]) + vn, ok2 := arm64VecOf(ops[1]) + if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" { + return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem) + } + base := uint32(0x5ee08400) // VADD + if mnem == "VSUB" { + base = 0x7ee08400 + } + return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil + } if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } @@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) { return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil } +// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.: a +// general register into an arranged vector, the spelling asm7.go's case 82 +// calls vmov/vdup Rn, Vd.. ok is false for anything that is not that +// shape, so the caller falls through to the arrangement and element paths; +// the toolchain rejects the bare spellings outright, and the reverse +// Vd., Rs with them. +func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) { + if len(ops) != 2 { + return nil, false, nil + } + if ops[0].Addr.Base != "" || isImmOperand(ops[0]) { + return nil, false, nil + } + rs := arm64RegNum(operandRegName(ops[0])) + if rs < 0 { + return nil, false, nil + } + dst, ok := arm64VecOf(ops[1]) + if !ok || dst.hasIdx || dst.arr == "" { + return nil, false, nil + } + b, err := a64GPVecWhole(mnem, rs, dst) + return b, true, err +} + +// a64GPVecWhole lays down the general-register-into-a-whole-vector move: +// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the +// lane width and Q the vector length. Both VMOV and VDUP take this form +// (asm7.go case 82); INS-into-one-lane is encoded elsewhere. +func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) { + var imm5, q uint32 + switch dst.arr { + case "B8": + imm5, q = 1, 0 + case "B16": + imm5, q = 1, 1<<30 + case "H4": + imm5, q = 2, 0 + case "H8": + imm5, q = 2, 1<<30 + case "S2": + imm5, q = 4, 0 + case "S4": + imm5, q = 4, 1<<30 + case "D2": + imm5, q = 8, 1<<30 + default: + // D1 rides no case-82 row: the toolchain rejects the one-doubleword + // spelling for this form, so the encoder refuses it too. + return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr) + } + return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil +} + // encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with // lane indices: // @@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) { return nil, fmt.Errorf("%s: source must be a general register", mnem) } if !dst.hasIdx { - var imm5, q uint32 - switch dst.arr { - case "B8": - imm5, q = 1, 0 - case "B16": - imm5, q = 1, 1<<30 - case "H4": - imm5, q = 2, 0 - case "H8": - imm5, q = 2, 1<<30 - case "S2": - imm5, q = 4, 0 - case "S4": - imm5, q = 4, 1<<30 - case "D1": - imm5, q = 8, 0 - case "D2": - imm5, q = 8, 1<<30 - default: - return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr) - } - return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil + // Duplicates the register across every lane (DUP Vd.T, Rn). + return a64GPVecWhole(mnem, rs, dst) } f, ok := a64ElemField(dst.arr, dst.idx) if !ok { diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index b8b6843..19ca963 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -79,6 +79,11 @@ func arm64RegNum(name string) int { return 17 case "R18": return 18 + case "R18_PLATFORM": + // The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in + // cmd/asm/internal/arch so assembly cannot use it by accident, and + // sys_windows_arm64.s references it only through this name. + return 18 case "R19": return 19 case "R20": diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index 374a7de..772eb8c 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) { }{ {"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31}, {"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31}, + {"R18_PLATFORM", 18}, {"F0", 0}, {"F4", 4}, {"F31", 31}, {"INVALID", -1}, {"X0", -1}, {"", -1}, } @@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) { } } +// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd. +// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 | +// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form +// sys_windows_arm64.s and the bytealg loops use. The D1 destination is +// rejected, as the toolchain rejects it. +func TestArm64GPIntoVector(t *testing.T) { + got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+ + "\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+ + "\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n") + want := []uint32{ + 0x4e010ca5, // VMOV R5, V5.B16 + 0x0e010c22, // VMOV R1, V2.B8 + 0x0e020c64, // VMOV R3, V4.H4 + 0x4e040d2a, // VMOV R9, V10.S4 + 0x4e020cff, // VMOV R7, V31.H8 + 0x4e080d6c, // VMOV R11, V12.D2 + 0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV) + 0x4e020d2a, // VDUP R9, V10.H8 + 0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR) + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error") + } +} + +// TestArm64SimdTwoOperand pins the two-operand accumulate spellings +// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64): +// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with +// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89). +func TestArm64SimdTwoOperand(t *testing.T) { + got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n") + want := []uint32{ + 0x5ee78508, // VADD V7, V8 + 0x7ee78508, // VSUB V7, V8 + 0x5ee18442, // VADD V1, V2 + 0x4e208422, // VADD arranged: the ordinary three-register path + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64TruncMove pins the truncating register moves against +// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB, +// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and +// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register +// drops to the W ORR too (asm7.go case 45). +func TestArm64TruncMove(t *testing.T) { + got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+ + "\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+ + "\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n") + want := []uint32{ + 0x93401c64, // MOVB = SXTB + 0x93403ca6, // MOVH = SXTH + 0x93407d2a, // MOVW = SXTW + 0xd3401c64, // MOVBU = UXTB + 0xd3403c64, // MOVHU = UXTH + 0x2a0303e4, // MOVWU = ORR W + 0xaa0303e4, // MOVD = ORR X + 0xaa1f03e4, // MOVD ZR, R4 keeps the X form + 0x2a1f03e4, // MOVB ZR, R4 drops to the W form + 0x2a1f03e5, // MOVWU ZR, R5 + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + // TestArm64SIMDLoadStore pins the structure loads and stores. func TestArm64SIMDLoadStore(t *testing.T) { got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+ diff --git a/testdata/verify/bookkeep_arm64.s b/testdata/verify/bookkeep_arm64.s new file mode 100644 index 0000000..3777570 --- /dev/null +++ b/testdata/verify/bookkeep_arm64.s @@ -0,0 +1,31 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 bookkeeping statements: the funcdata.h +// pseudo-directives (GO_ARGS, NO_LOCAL_POINTERS, FUNCDATA, PCDATA) contribute +// no instruction bytes, and every function is byte-compared against +// go tool asm. + +#include "textflag.h" +#include "funcdata.h" + +// func bookkeep() +TEXT ·bookkeep(SB), NOSPLIT, $8-0 + GO_ARGS + FUNCDATA $3, inline_tree(SB) + PCDATA $1, $2 + MOVD R1, 0(RSP) + RET + +// func bookkeepNoLocals() +TEXT ·bookkeepNoLocals(SB), NOSPLIT, $16-0 + NO_LOCAL_POINTERS + PCDATA $0, $0 + PCDATA $1, $1 + MOVD R2, 8(RSP) + RET + +// func bookkeepPlain() +TEXT ·bookkeepPlain(SB), NOSPLIT, $0-0 + MOVD R3, R4 + RET diff --git a/testdata/verify/simdmove_arm64.s b/testdata/verify/simdmove_arm64.s new file mode 100644 index 0000000..6976234 --- /dev/null +++ b/testdata/verify/simdmove_arm64.s @@ -0,0 +1,57 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 whole-vector moves between a general +// register and an arranged vector (VMOV/VDUP Rs, Vd.), the two-operand +// accumulate spellings VADD/VSUB Vm, Vn, and the toolchain-reserved +// R18_PLATFORM register name. Every function is byte-compared against +// go tool asm. + +#include "textflag.h" + +// func gpIntoVector() +TEXT ·gpIntoVector(SB), NOSPLIT, $0-0 + VMOV R1, V2.B8 + VMOV R3, V4.B16 + VMOV R5, V6.H4 + VMOV R7, V8.H8 + VMOV R9, V10.S2 + VMOV R11, V12.S4 + VMOV R13, V14.D2 + VDUP R15, V16.B8 + VDUP R17, V18.B16 + VDUP R19, V20.H8 + VDUP R21, V22.S4 + VDUP R23, V24.D2 + RET + +// func simdAccumulate() +TEXT ·simdAccumulate(SB), NOSPLIT, $0-0 + VADD V7, V8 + VSUB V7, V8 + VADD V1, V2 + VSUB V30, V31 + VADD V0.B16, V1.B16, V2.B16 + VSUB V0.S4, V1.S4, V2.S4 + RET + +// func truncMove() +TEXT ·truncMove(SB), NOSPLIT, $0-0 + MOVB R3, R4 + MOVH R5, R6 + MOVW R9, R10 + MOVBU R3, R4 + MOVHU R3, R4 + MOVWU R3, R4 + MOVD R3, R4 + RET + +// func platformRegister() +TEXT ·platformRegister(SB), NOSPLIT, $0-0 + MOVD R18_PLATFORM, R3 + MOVW R18_PLATFORM, R4 + MOVD R3, R18_PLATFORM + MOVD 0x68(R18_PLATFORM), R5 + MOVD R5, 0x68(R18_PLATFORM) + MOVW 8(R18_PLATFORM), R6 + RET