feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+130
-30
@@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
||||
return 4
|
||||
}
|
||||
}
|
||||
// The funcdata pseudo-statements contribute no bytes.
|
||||
// The funcdata pseudo-statements contribute no bytes, the expanded
|
||||
// FUNCDATA/PCDATA forms included.
|
||||
switch mnem {
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
|
||||
return 0
|
||||
}
|
||||
switch mnem {
|
||||
@@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
|
||||
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
|
||||
// code: the toolchain emits zero instruction bytes for them, and so does
|
||||
// the encoder here.
|
||||
// the encoder here. Files that include funcdata.h spell them after
|
||||
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
|
||||
// bookkeeping too (the same treatment the loong64 encoder applies).
|
||||
switch mnem {
|
||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
||||
return nil, nil
|
||||
case "END":
|
||||
if len(ops) != 0 {
|
||||
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
|
||||
}
|
||||
return nil, nil
|
||||
case "FUNCDATA":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
|
||||
}
|
||||
return nil, nil
|
||||
case "PCDATA":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
|
||||
return nil, fmt.Errorf("PCDATA expects $n, $n")
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
|
||||
@@ -520,8 +538,17 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
|
||||
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
|
||||
// over the plain arrangement paths, which carry no index.
|
||||
if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) {
|
||||
return encodeARM64Dup(mnem, ops)
|
||||
if mnem == "VDUP" || mnem == "VMOV" {
|
||||
if arm64SimdHasElement(ops) {
|
||||
return encodeARM64Dup(mnem, ops)
|
||||
}
|
||||
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
|
||||
// vector (asm7.go case 82, shared by both mnemonics). The element
|
||||
// paths above only run when a lane index is spelled, so this is the
|
||||
// whole-vector shape's only route.
|
||||
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
|
||||
return b, err
|
||||
}
|
||||
}
|
||||
|
||||
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
||||
@@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
|
||||
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// Integer → integer: ORR Rd, ZR, Rs.
|
||||
// Integer → integer. Every truncating register move lowers to an
|
||||
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
|
||||
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
|
||||
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
|
||||
// ORR Xd, XZR, Xm.
|
||||
if rs != 31 {
|
||||
switch mnem {
|
||||
case "MOVB":
|
||||
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVH":
|
||||
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVW":
|
||||
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVBU":
|
||||
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
case "MOVHU":
|
||||
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||
}
|
||||
}
|
||||
sf := uint32(1) // 64-bit
|
||||
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" ||
|
||||
mnem == "MOVH" || mnem == "MOVHU" {
|
||||
if mnem == "MOVWU" {
|
||||
sf = 0
|
||||
}
|
||||
// A narrow move out of the zero register loses its width: the
|
||||
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
|
||||
// WZR. MOVD and MOV keep the 64-bit form.
|
||||
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
|
||||
sf = 0
|
||||
}
|
||||
op := uint32(1<<29 | 0x0a<<24) // ORR
|
||||
@@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
}
|
||||
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
|
||||
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
|
||||
// for bare V registers alone: the arranged forms and every other
|
||||
// three-register mnemonic are rejected outright.
|
||||
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
|
||||
vm, ok1 := arm64VecOf(ops[0])
|
||||
vn, ok2 := arm64VecOf(ops[1])
|
||||
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
|
||||
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
|
||||
}
|
||||
base := uint32(0x5ee08400) // VADD
|
||||
if mnem == "VSUB" {
|
||||
base = 0x7ee08400
|
||||
}
|
||||
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
|
||||
}
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
@@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
|
||||
// general register into an arranged vector, the spelling asm7.go's case 82
|
||||
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
|
||||
// shape, so the caller falls through to the arrangement and element paths;
|
||||
// the toolchain rejects the bare spellings outright, and the reverse
|
||||
// Vd.<T>, Rs with them.
|
||||
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
|
||||
if len(ops) != 2 {
|
||||
return nil, false, nil
|
||||
}
|
||||
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
|
||||
return nil, false, nil
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
if rs < 0 {
|
||||
return nil, false, nil
|
||||
}
|
||||
dst, ok := arm64VecOf(ops[1])
|
||||
if !ok || dst.hasIdx || dst.arr == "" {
|
||||
return nil, false, nil
|
||||
}
|
||||
b, err := a64GPVecWhole(mnem, rs, dst)
|
||||
return b, true, err
|
||||
}
|
||||
|
||||
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
|
||||
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
|
||||
// lane width and Q the vector length. Both VMOV and VDUP take this form
|
||||
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
|
||||
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
|
||||
var imm5, q uint32
|
||||
switch dst.arr {
|
||||
case "B8":
|
||||
imm5, q = 1, 0
|
||||
case "B16":
|
||||
imm5, q = 1, 1<<30
|
||||
case "H4":
|
||||
imm5, q = 2, 0
|
||||
case "H8":
|
||||
imm5, q = 2, 1<<30
|
||||
case "S2":
|
||||
imm5, q = 4, 0
|
||||
case "S4":
|
||||
imm5, q = 4, 1<<30
|
||||
case "D2":
|
||||
imm5, q = 8, 1<<30
|
||||
default:
|
||||
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
|
||||
// spelling for this form, so the encoder refuses it too.
|
||||
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
||||
}
|
||||
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
||||
// lane indices:
|
||||
//
|
||||
@@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("%s: source must be a general register", mnem)
|
||||
}
|
||||
if !dst.hasIdx {
|
||||
var imm5, q uint32
|
||||
switch dst.arr {
|
||||
case "B8":
|
||||
imm5, q = 1, 0
|
||||
case "B16":
|
||||
imm5, q = 1, 1<<30
|
||||
case "H4":
|
||||
imm5, q = 2, 0
|
||||
case "H8":
|
||||
imm5, q = 2, 1<<30
|
||||
case "S2":
|
||||
imm5, q = 4, 0
|
||||
case "S4":
|
||||
imm5, q = 4, 1<<30
|
||||
case "D1":
|
||||
imm5, q = 8, 0
|
||||
case "D2":
|
||||
imm5, q = 8, 1<<30
|
||||
default:
|
||||
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
||||
}
|
||||
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
||||
// Duplicates the register across every lane (DUP Vd.T, Rn).
|
||||
return a64GPVecWhole(mnem, rs, dst)
|
||||
}
|
||||
f, ok := a64ElemField(dst.arr, dst.idx)
|
||||
if !ok {
|
||||
|
||||
@@ -79,6 +79,11 @@ func arm64RegNum(name string) int {
|
||||
return 17
|
||||
case "R18":
|
||||
return 18
|
||||
case "R18_PLATFORM":
|
||||
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
|
||||
// cmd/asm/internal/arch so assembly cannot use it by accident, and
|
||||
// sys_windows_arm64.s references it only through this name.
|
||||
return 18
|
||||
case "R19":
|
||||
return 19
|
||||
case "R20":
|
||||
|
||||
@@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
|
||||
}{
|
||||
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
|
||||
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
|
||||
{"R18_PLATFORM", 18},
|
||||
{"F0", 0}, {"F4", 4}, {"F31", 31},
|
||||
{"INVALID", -1}, {"X0", -1}, {"", -1},
|
||||
}
|
||||
@@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
|
||||
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
|
||||
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
|
||||
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
|
||||
// rejected, as the toolchain rejects it.
|
||||
func TestArm64GPIntoVector(t *testing.T) {
|
||||
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
|
||||
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
|
||||
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
|
||||
want := []uint32{
|
||||
0x4e010ca5, // VMOV R5, V5.B16
|
||||
0x0e010c22, // VMOV R1, V2.B8
|
||||
0x0e020c64, // VMOV R3, V4.H4
|
||||
0x4e040d2a, // VMOV R9, V10.S4
|
||||
0x4e020cff, // VMOV R7, V31.H8
|
||||
0x4e080d6c, // VMOV R11, V12.D2
|
||||
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
|
||||
0x4e020d2a, // VDUP R9, V10.H8
|
||||
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
|
||||
0xd65f03c0,
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if _, err := AssembleFileARM64(f); err == nil {
|
||||
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
|
||||
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
|
||||
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
|
||||
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
|
||||
func TestArm64SimdTwoOperand(t *testing.T) {
|
||||
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
|
||||
want := []uint32{
|
||||
0x5ee78508, // VADD V7, V8
|
||||
0x7ee78508, // VSUB V7, V8
|
||||
0x5ee18442, // VADD V1, V2
|
||||
0x4e208422, // VADD arranged: the ordinary three-register path
|
||||
0xd65f03c0,
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64TruncMove pins the truncating register moves against
|
||||
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
|
||||
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
|
||||
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
|
||||
// drops to the W ORR too (asm7.go case 45).
|
||||
func TestArm64TruncMove(t *testing.T) {
|
||||
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
|
||||
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
|
||||
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
|
||||
want := []uint32{
|
||||
0x93401c64, // MOVB = SXTB
|
||||
0x93403ca6, // MOVH = SXTH
|
||||
0x93407d2a, // MOVW = SXTW
|
||||
0xd3401c64, // MOVBU = UXTB
|
||||
0xd3403c64, // MOVHU = UXTH
|
||||
0x2a0303e4, // MOVWU = ORR W
|
||||
0xaa0303e4, // MOVD = ORR X
|
||||
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
|
||||
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
|
||||
0x2a1f03e5, // MOVWU ZR, R5
|
||||
0xd65f03c0,
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SIMDLoadStore pins the structure loads and stores.
|
||||
func TestArm64SIMDLoadStore(t *testing.T) {
|
||||
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
|
||||
|
||||
Reference in New Issue
Block a user