feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 21:17:20 +02:00
parent 81e2673923
commit b0f9071bf5
6 changed files with 318 additions and 30 deletions
+1
View File
@@ -33,6 +33,7 @@ func arm64Registers() []Register {
for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
}
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
add("ZR", Special, "zero register (reads as 0)")
add("SP", Special, "stack pointer")
add("LR", Special, "link register (alias of R30)")
+129 -29
View File
@@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
return 4
}
}
// The funcdata pseudo-statements contribute no bytes.
// The funcdata pseudo-statements contribute no bytes, the expanded
// FUNCDATA/PCDATA forms included.
switch mnem {
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
return 0
}
switch mnem {
@@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
// code: the toolchain emits zero instruction bytes for them, and so does
// the encoder here.
// the encoder here. Files that include funcdata.h spell them after
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
// bookkeeping too (the same treatment the loong64 encoder applies).
switch mnem {
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
return nil, nil
case "END":
if len(ops) != 0 {
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
}
return nil, nil
case "FUNCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
}
return nil, nil
case "PCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
return nil, fmt.Errorf("PCDATA expects $n, $n")
}
return nil, nil
}
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
@@ -520,9 +538,18 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
// over the plain arrangement paths, which carry no index.
if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) {
if mnem == "VDUP" || mnem == "VMOV" {
if arm64SimdHasElement(ops) {
return encodeARM64Dup(mnem, ops)
}
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
// vector (asm7.go case 82, shared by both mnemonics). The element
// paths above only run when a lane index is spelled, so this is the
// whole-vector shape's only route.
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
return b, err
}
}
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
@@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
}
// Integer → integer: ORR Rd, ZR, Rs.
// Integer → integer. Every truncating register move lowers to an
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
// ORR Xd, XZR, Xm.
if rs != 31 {
switch mnem {
case "MOVB":
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVH":
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVW":
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVBU":
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVHU":
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
}
}
sf := uint32(1) // 64-bit
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" ||
mnem == "MOVH" || mnem == "MOVHU" {
if mnem == "MOVWU" {
sf = 0
}
// A narrow move out of the zero register loses its width: the
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
// WZR. MOVD and MOV keep the 64-bit form.
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
sf = 0
}
op := uint32(1<<29 | 0x0a<<24) // ORR
@@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
// for bare V registers alone: the arranged forms and every other
// three-register mnemonic are rejected outright.
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
vm, ok1 := arm64VecOf(ops[0])
vn, ok2 := arm64VecOf(ops[1])
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
}
base := uint32(0x5ee08400) // VADD
if mnem == "VSUB" {
base = 0x7ee08400
}
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
}
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
@@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
// general register into an arranged vector, the spelling asm7.go's case 82
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
// shape, so the caller falls through to the arrangement and element paths;
// the toolchain rejects the bare spellings outright, and the reverse
// Vd.<T>, Rs with them.
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
if len(ops) != 2 {
return nil, false, nil
}
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
return nil, false, nil
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, false, nil
}
dst, ok := arm64VecOf(ops[1])
if !ok || dst.hasIdx || dst.arr == "" {
return nil, false, nil
}
b, err := a64GPVecWhole(mnem, rs, dst)
return b, true, err
}
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
// lane width and Q the vector length. Both VMOV and VDUP take this form
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
var imm5, q uint32
switch dst.arr {
case "B8":
imm5, q = 1, 0
case "B16":
imm5, q = 1, 1<<30
case "H4":
imm5, q = 2, 0
case "H8":
imm5, q = 2, 1<<30
case "S2":
imm5, q = 4, 0
case "S4":
imm5, q = 4, 1<<30
case "D2":
imm5, q = 8, 1<<30
default:
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
// spelling for this form, so the encoder refuses it too.
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
}
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
}
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
// lane indices:
//
@@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: source must be a general register", mnem)
}
if !dst.hasIdx {
var imm5, q uint32
switch dst.arr {
case "B8":
imm5, q = 1, 0
case "B16":
imm5, q = 1, 1<<30
case "H4":
imm5, q = 2, 0
case "H8":
imm5, q = 2, 1<<30
case "S2":
imm5, q = 4, 0
case "S4":
imm5, q = 4, 1<<30
case "D1":
imm5, q = 8, 0
case "D2":
imm5, q = 8, 1<<30
default:
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
}
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
// Duplicates the register across every lane (DUP Vd.T, Rn).
return a64GPVecWhole(mnem, rs, dst)
}
f, ok := a64ElemField(dst.arr, dst.idx)
if !ok {
+5
View File
@@ -79,6 +79,11 @@ func arm64RegNum(name string) int {
return 17
case "R18":
return 18
case "R18_PLATFORM":
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
// cmd/asm/internal/arch so assembly cannot use it by accident, and
// sys_windows_arm64.s references it only through this name.
return 18
case "R19":
return 19
case "R20":
+94
View File
@@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
}{
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
{"R18_PLATFORM", 18},
{"F0", 0}, {"F4", 4}, {"F31", 31},
{"INVALID", -1}, {"X0", -1}, {"", -1},
}
@@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) {
}
}
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
// rejected, as the toolchain rejects it.
func TestArm64GPIntoVector(t *testing.T) {
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
want := []uint32{
0x4e010ca5, // VMOV R5, V5.B16
0x0e010c22, // VMOV R1, V2.B8
0x0e020c64, // VMOV R3, V4.H4
0x4e040d2a, // VMOV R9, V10.S4
0x4e020cff, // VMOV R7, V31.H8
0x4e080d6c, // VMOV R11, V12.D2
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
0x4e020d2a, // VDUP R9, V10.H8
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
}
}
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
func TestArm64SimdTwoOperand(t *testing.T) {
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
want := []uint32{
0x5ee78508, // VADD V7, V8
0x7ee78508, // VSUB V7, V8
0x5ee18442, // VADD V1, V2
0x4e208422, // VADD arranged: the ordinary three-register path
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64TruncMove pins the truncating register moves against
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
// drops to the W ORR too (asm7.go case 45).
func TestArm64TruncMove(t *testing.T) {
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
want := []uint32{
0x93401c64, // MOVB = SXTB
0x93403ca6, // MOVH = SXTH
0x93407d2a, // MOVW = SXTW
0xd3401c64, // MOVBU = UXTB
0xd3403c64, // MOVHU = UXTH
0x2a0303e4, // MOVWU = ORR W
0xaa0303e4, // MOVD = ORR X
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
0x2a1f03e5, // MOVWU ZR, R5
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLoadStore pins the structure loads and stores.
func TestArm64SIMDLoadStore(t *testing.T) {
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
+31
View File
@@ -0,0 +1,31 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 bookkeeping statements: the funcdata.h
// pseudo-directives (GO_ARGS, NO_LOCAL_POINTERS, FUNCDATA, PCDATA) contribute
// no instruction bytes, and every function is byte-compared against
// go tool asm.
#include "textflag.h"
#include "funcdata.h"
// func bookkeep()
TEXT ·bookkeep(SB), NOSPLIT, $8-0
GO_ARGS
FUNCDATA $3, inline_tree(SB)
PCDATA $1, $2
MOVD R1, 0(RSP)
RET
// func bookkeepNoLocals()
TEXT ·bookkeepNoLocals(SB), NOSPLIT, $16-0
NO_LOCAL_POINTERS
PCDATA $0, $0
PCDATA $1, $1
MOVD R2, 8(RSP)
RET
// func bookkeepPlain()
TEXT ·bookkeepPlain(SB), NOSPLIT, $0-0
MOVD R3, R4
RET
+57
View File
@@ -0,0 +1,57 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 whole-vector moves between a general
// register and an arranged vector (VMOV/VDUP Rs, Vd.<T>), the two-operand
// accumulate spellings VADD/VSUB Vm, Vn, and the toolchain-reserved
// R18_PLATFORM register name. Every function is byte-compared against
// go tool asm.
#include "textflag.h"
// func gpIntoVector()
TEXT ·gpIntoVector(SB), NOSPLIT, $0-0
VMOV R1, V2.B8
VMOV R3, V4.B16
VMOV R5, V6.H4
VMOV R7, V8.H8
VMOV R9, V10.S2
VMOV R11, V12.S4
VMOV R13, V14.D2
VDUP R15, V16.B8
VDUP R17, V18.B16
VDUP R19, V20.H8
VDUP R21, V22.S4
VDUP R23, V24.D2
RET
// func simdAccumulate()
TEXT ·simdAccumulate(SB), NOSPLIT, $0-0
VADD V7, V8
VSUB V7, V8
VADD V1, V2
VSUB V30, V31
VADD V0.B16, V1.B16, V2.B16
VSUB V0.S4, V1.S4, V2.S4
RET
// func truncMove()
TEXT ·truncMove(SB), NOSPLIT, $0-0
MOVB R3, R4
MOVH R5, R6
MOVW R9, R10
MOVBU R3, R4
MOVHU R3, R4
MOVWU R3, R4
MOVD R3, R4
RET
// func platformRegister()
TEXT ·platformRegister(SB), NOSPLIT, $0-0
MOVD R18_PLATFORM, R3
MOVW R18_PLATFORM, R4
MOVD R3, R18_PLATFORM
MOVD 0x68(R18_PLATFORM), R5
MOVD R5, 0x68(R18_PLATFORM)
MOVW 8(R18_PLATFORM), R6
RET