feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 21:17:20 +02:00
parent 81e2673923
commit b0f9071bf5
6 changed files with 318 additions and 30 deletions
+94
View File
@@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
}{
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
{"R18_PLATFORM", 18},
{"F0", 0}, {"F4", 4}, {"F31", 31},
{"INVALID", -1}, {"X0", -1}, {"", -1},
}
@@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) {
}
}
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
// rejected, as the toolchain rejects it.
func TestArm64GPIntoVector(t *testing.T) {
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
want := []uint32{
0x4e010ca5, // VMOV R5, V5.B16
0x0e010c22, // VMOV R1, V2.B8
0x0e020c64, // VMOV R3, V4.H4
0x4e040d2a, // VMOV R9, V10.S4
0x4e020cff, // VMOV R7, V31.H8
0x4e080d6c, // VMOV R11, V12.D2
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
0x4e020d2a, // VDUP R9, V10.H8
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
}
}
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
func TestArm64SimdTwoOperand(t *testing.T) {
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
want := []uint32{
0x5ee78508, // VADD V7, V8
0x7ee78508, // VSUB V7, V8
0x5ee18442, // VADD V1, V2
0x4e208422, // VADD arranged: the ordinary three-register path
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64TruncMove pins the truncating register moves against
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
// drops to the W ORR too (asm7.go case 45).
func TestArm64TruncMove(t *testing.T) {
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
want := []uint32{
0x93401c64, // MOVB = SXTB
0x93403ca6, // MOVH = SXTH
0x93407d2a, // MOVW = SXTW
0xd3401c64, // MOVBU = UXTB
0xd3403c64, // MOVHU = UXTH
0x2a0303e4, // MOVWU = ORR W
0xaa0303e4, // MOVD = ORR X
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
0x2a1f03e5, // MOVWU ZR, R5
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLoadStore pins the structure loads and stores.
func TestArm64SIMDLoadStore(t *testing.T) {
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+