feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 21:17:20 +02:00
parent 81e2673923
commit b0f9071bf5
6 changed files with 318 additions and 30 deletions
+130 -30
View File
@@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
return 4
}
}
// The funcdata pseudo-statements contribute no bytes.
// The funcdata pseudo-statements contribute no bytes, the expanded
// FUNCDATA/PCDATA forms included.
switch mnem {
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
return 0
}
switch mnem {
@@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
// code: the toolchain emits zero instruction bytes for them, and so does
// the encoder here.
// the encoder here. Files that include funcdata.h spell them after
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
// bookkeeping too (the same treatment the loong64 encoder applies).
switch mnem {
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
return nil, nil
case "END":
if len(ops) != 0 {
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
}
return nil, nil
case "FUNCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
}
return nil, nil
case "PCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
return nil, fmt.Errorf("PCDATA expects $n, $n")
}
return nil, nil
}
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
@@ -520,8 +538,17 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
// over the plain arrangement paths, which carry no index.
if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) {
return encodeARM64Dup(mnem, ops)
if mnem == "VDUP" || mnem == "VMOV" {
if arm64SimdHasElement(ops) {
return encodeARM64Dup(mnem, ops)
}
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
// vector (asm7.go case 82, shared by both mnemonics). The element
// paths above only run when a lane index is spelled, so this is the
// whole-vector shape's only route.
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
return b, err
}
}
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
@@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
}
// Integer → integer: ORR Rd, ZR, Rs.
// Integer → integer. Every truncating register move lowers to an
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
// ORR Xd, XZR, Xm.
if rs != 31 {
switch mnem {
case "MOVB":
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVH":
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVW":
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVBU":
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
case "MOVHU":
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
}
}
sf := uint32(1) // 64-bit
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" ||
mnem == "MOVH" || mnem == "MOVHU" {
if mnem == "MOVWU" {
sf = 0
}
// A narrow move out of the zero register loses its width: the
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
// WZR. MOVD and MOV keep the 64-bit form.
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
sf = 0
}
op := uint32(1<<29 | 0x0a<<24) // ORR
@@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
// for bare V registers alone: the arranged forms and every other
// three-register mnemonic are rejected outright.
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
vm, ok1 := arm64VecOf(ops[0])
vn, ok2 := arm64VecOf(ops[1])
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
}
base := uint32(0x5ee08400) // VADD
if mnem == "VSUB" {
base = 0x7ee08400
}
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
}
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
@@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
// general register into an arranged vector, the spelling asm7.go's case 82
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
// shape, so the caller falls through to the arrangement and element paths;
// the toolchain rejects the bare spellings outright, and the reverse
// Vd.<T>, Rs with them.
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
if len(ops) != 2 {
return nil, false, nil
}
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
return nil, false, nil
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, false, nil
}
dst, ok := arm64VecOf(ops[1])
if !ok || dst.hasIdx || dst.arr == "" {
return nil, false, nil
}
b, err := a64GPVecWhole(mnem, rs, dst)
return b, true, err
}
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
// lane width and Q the vector length. Both VMOV and VDUP take this form
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
var imm5, q uint32
switch dst.arr {
case "B8":
imm5, q = 1, 0
case "B16":
imm5, q = 1, 1<<30
case "H4":
imm5, q = 2, 0
case "H8":
imm5, q = 2, 1<<30
case "S2":
imm5, q = 4, 0
case "S4":
imm5, q = 4, 1<<30
case "D2":
imm5, q = 8, 1<<30
default:
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
// spelling for this form, so the encoder refuses it too.
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
}
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
}
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
// lane indices:
//
@@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: source must be a general register", mnem)
}
if !dst.hasIdx {
var imm5, q uint32
switch dst.arr {
case "B8":
imm5, q = 1, 0
case "B16":
imm5, q = 1, 1<<30
case "H4":
imm5, q = 2, 0
case "H8":
imm5, q = 2, 1<<30
case "S2":
imm5, q = 4, 0
case "S4":
imm5, q = 4, 1<<30
case "D1":
imm5, q = 8, 0
case "D2":
imm5, q = 8, 1<<30
default:
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
}
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
// Duplicates the register across every lane (DUP Vd.T, Rn).
return a64GPVecWhole(mnem, rs, dst)
}
f, ok := a64ElemField(dst.arr, dst.idx)
if !ok {
+5
View File
@@ -79,6 +79,11 @@ func arm64RegNum(name string) int {
return 17
case "R18":
return 18
case "R18_PLATFORM":
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
// cmd/asm/internal/arch so assembly cannot use it by accident, and
// sys_windows_arm64.s references it only through this name.
return 18
case "R19":
return 19
case "R20":
+94
View File
@@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
}{
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
{"R18_PLATFORM", 18},
{"F0", 0}, {"F4", 4}, {"F31", 31},
{"INVALID", -1}, {"X0", -1}, {"", -1},
}
@@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) {
}
}
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
// rejected, as the toolchain rejects it.
func TestArm64GPIntoVector(t *testing.T) {
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
want := []uint32{
0x4e010ca5, // VMOV R5, V5.B16
0x0e010c22, // VMOV R1, V2.B8
0x0e020c64, // VMOV R3, V4.H4
0x4e040d2a, // VMOV R9, V10.S4
0x4e020cff, // VMOV R7, V31.H8
0x4e080d6c, // VMOV R11, V12.D2
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
0x4e020d2a, // VDUP R9, V10.H8
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
}
}
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
func TestArm64SimdTwoOperand(t *testing.T) {
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
want := []uint32{
0x5ee78508, // VADD V7, V8
0x7ee78508, // VSUB V7, V8
0x5ee18442, // VADD V1, V2
0x4e208422, // VADD arranged: the ordinary three-register path
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64TruncMove pins the truncating register moves against
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
// drops to the W ORR too (asm7.go case 45).
func TestArm64TruncMove(t *testing.T) {
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
want := []uint32{
0x93401c64, // MOVB = SXTB
0x93403ca6, // MOVH = SXTH
0x93407d2a, // MOVW = SXTW
0xd3401c64, // MOVBU = UXTB
0xd3403c64, // MOVHU = UXTH
0x2a0303e4, // MOVWU = ORR W
0xaa0303e4, // MOVD = ORR X
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
0x2a1f03e5, // MOVWU ZR, R5
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLoadStore pins the structure loads and stores.
func TestArm64SIMDLoadStore(t *testing.T) {
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+