feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering
Assisted-by: GLM 5.3 Flash
This commit is contained in:
@@ -33,6 +33,7 @@ func arm64Registers() []Register {
|
|||||||
for i := 0; i <= 30; i++ {
|
for i := 0; i <= 30; i++ {
|
||||||
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
||||||
}
|
}
|
||||||
|
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
|
||||||
add("ZR", Special, "zero register (reads as 0)")
|
add("ZR", Special, "zero register (reads as 0)")
|
||||||
add("SP", Special, "stack pointer")
|
add("SP", Special, "stack pointer")
|
||||||
add("LR", Special, "link register (alias of R30)")
|
add("LR", Special, "link register (alias of R30)")
|
||||||
|
|||||||
+129
-29
@@ -263,9 +263,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
|||||||
return 4
|
return 4
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// The funcdata pseudo-statements contribute no bytes.
|
// The funcdata pseudo-statements contribute no bytes, the expanded
|
||||||
|
// FUNCDATA/PCDATA forms included.
|
||||||
switch mnem {
|
switch mnem {
|
||||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
switch mnem {
|
switch mnem {
|
||||||
@@ -348,10 +349,27 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
|||||||
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
|
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
|
||||||
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
|
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
|
||||||
// code: the toolchain emits zero instruction bytes for them, and so does
|
// code: the toolchain emits zero instruction bytes for them, and so does
|
||||||
// the encoder here.
|
// the encoder here. Files that include funcdata.h spell them after
|
||||||
|
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
|
||||||
|
// bookkeeping too (the same treatment the loong64 encoder applies).
|
||||||
switch mnem {
|
switch mnem {
|
||||||
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
||||||
return nil, nil
|
return nil, nil
|
||||||
|
case "END":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return nil, nil
|
||||||
|
case "FUNCDATA":
|
||||||
|
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
||||||
|
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
|
||||||
|
}
|
||||||
|
return nil, nil
|
||||||
|
case "PCDATA":
|
||||||
|
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
|
||||||
|
return nil, fmt.Errorf("PCDATA expects $n, $n")
|
||||||
|
}
|
||||||
|
return nil, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
|
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
|
||||||
@@ -520,9 +538,18 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
|||||||
|
|
||||||
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
|
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
|
||||||
// over the plain arrangement paths, which carry no index.
|
// over the plain arrangement paths, which carry no index.
|
||||||
if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) {
|
if mnem == "VDUP" || mnem == "VMOV" {
|
||||||
|
if arm64SimdHasElement(ops) {
|
||||||
return encodeARM64Dup(mnem, ops)
|
return encodeARM64Dup(mnem, ops)
|
||||||
}
|
}
|
||||||
|
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
|
||||||
|
// vector (asm7.go case 82, shared by both mnemonics). The element
|
||||||
|
// paths above only run when a lane index is spelled, so this is the
|
||||||
|
// whole-vector shape's only route.
|
||||||
|
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
|
||||||
|
return b, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
||||||
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
||||||
@@ -1733,10 +1760,33 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
|
|||||||
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
|
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// Integer → integer: ORR Rd, ZR, Rs.
|
// Integer → integer. Every truncating register move lowers to an
|
||||||
|
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
|
||||||
|
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
|
||||||
|
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
|
||||||
|
// ORR Xd, XZR, Xm.
|
||||||
|
if rs != 31 {
|
||||||
|
switch mnem {
|
||||||
|
case "MOVB":
|
||||||
|
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||||
|
case "MOVH":
|
||||||
|
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||||
|
case "MOVW":
|
||||||
|
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||||
|
case "MOVBU":
|
||||||
|
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||||
|
case "MOVHU":
|
||||||
|
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
||||||
|
}
|
||||||
|
}
|
||||||
sf := uint32(1) // 64-bit
|
sf := uint32(1) // 64-bit
|
||||||
if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" ||
|
if mnem == "MOVWU" {
|
||||||
mnem == "MOVH" || mnem == "MOVHU" {
|
sf = 0
|
||||||
|
}
|
||||||
|
// A narrow move out of the zero register loses its width: the
|
||||||
|
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
|
||||||
|
// WZR. MOVD and MOV keep the 64-bit form.
|
||||||
|
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
|
||||||
sf = 0
|
sf = 0
|
||||||
}
|
}
|
||||||
op := uint32(1<<29 | 0x0a<<24) // ORR
|
op := uint32(1<<29 | 0x0a<<24) // ORR
|
||||||
@@ -3185,6 +3235,22 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
|||||||
}
|
}
|
||||||
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||||
}
|
}
|
||||||
|
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
|
||||||
|
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
|
||||||
|
// for bare V registers alone: the arranged forms and every other
|
||||||
|
// three-register mnemonic are rejected outright.
|
||||||
|
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
|
||||||
|
vm, ok1 := arm64VecOf(ops[0])
|
||||||
|
vn, ok2 := arm64VecOf(ops[1])
|
||||||
|
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
|
||||||
|
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
|
||||||
|
}
|
||||||
|
base := uint32(0x5ee08400) // VADD
|
||||||
|
if mnem == "VSUB" {
|
||||||
|
base = 0x7ee08400
|
||||||
|
}
|
||||||
|
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
|
||||||
|
}
|
||||||
if len(ops) != 3 {
|
if len(ops) != 3 {
|
||||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||||
}
|
}
|
||||||
@@ -3378,6 +3444,60 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|||||||
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
|
||||||
|
// general register into an arranged vector, the spelling asm7.go's case 82
|
||||||
|
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
|
||||||
|
// shape, so the caller falls through to the arrangement and element paths;
|
||||||
|
// the toolchain rejects the bare spellings outright, and the reverse
|
||||||
|
// Vd.<T>, Rs with them.
|
||||||
|
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return nil, false, nil
|
||||||
|
}
|
||||||
|
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
|
||||||
|
return nil, false, nil
|
||||||
|
}
|
||||||
|
rs := arm64RegNum(operandRegName(ops[0]))
|
||||||
|
if rs < 0 {
|
||||||
|
return nil, false, nil
|
||||||
|
}
|
||||||
|
dst, ok := arm64VecOf(ops[1])
|
||||||
|
if !ok || dst.hasIdx || dst.arr == "" {
|
||||||
|
return nil, false, nil
|
||||||
|
}
|
||||||
|
b, err := a64GPVecWhole(mnem, rs, dst)
|
||||||
|
return b, true, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
|
||||||
|
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
|
||||||
|
// lane width and Q the vector length. Both VMOV and VDUP take this form
|
||||||
|
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
|
||||||
|
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
|
||||||
|
var imm5, q uint32
|
||||||
|
switch dst.arr {
|
||||||
|
case "B8":
|
||||||
|
imm5, q = 1, 0
|
||||||
|
case "B16":
|
||||||
|
imm5, q = 1, 1<<30
|
||||||
|
case "H4":
|
||||||
|
imm5, q = 2, 0
|
||||||
|
case "H8":
|
||||||
|
imm5, q = 2, 1<<30
|
||||||
|
case "S2":
|
||||||
|
imm5, q = 4, 0
|
||||||
|
case "S4":
|
||||||
|
imm5, q = 4, 1<<30
|
||||||
|
case "D2":
|
||||||
|
imm5, q = 8, 1<<30
|
||||||
|
default:
|
||||||
|
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
|
||||||
|
// spelling for this form, so the encoder refuses it too.
|
||||||
|
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
||||||
|
}
|
||||||
|
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
||||||
|
}
|
||||||
|
|
||||||
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
||||||
// lane indices:
|
// lane indices:
|
||||||
//
|
//
|
||||||
@@ -3413,28 +3533,8 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|||||||
return nil, fmt.Errorf("%s: source must be a general register", mnem)
|
return nil, fmt.Errorf("%s: source must be a general register", mnem)
|
||||||
}
|
}
|
||||||
if !dst.hasIdx {
|
if !dst.hasIdx {
|
||||||
var imm5, q uint32
|
// Duplicates the register across every lane (DUP Vd.T, Rn).
|
||||||
switch dst.arr {
|
return a64GPVecWhole(mnem, rs, dst)
|
||||||
case "B8":
|
|
||||||
imm5, q = 1, 0
|
|
||||||
case "B16":
|
|
||||||
imm5, q = 1, 1<<30
|
|
||||||
case "H4":
|
|
||||||
imm5, q = 2, 0
|
|
||||||
case "H8":
|
|
||||||
imm5, q = 2, 1<<30
|
|
||||||
case "S2":
|
|
||||||
imm5, q = 4, 0
|
|
||||||
case "S4":
|
|
||||||
imm5, q = 4, 1<<30
|
|
||||||
case "D1":
|
|
||||||
imm5, q = 8, 0
|
|
||||||
case "D2":
|
|
||||||
imm5, q = 8, 1<<30
|
|
||||||
default:
|
|
||||||
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
|
||||||
}
|
|
||||||
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
|
||||||
}
|
}
|
||||||
f, ok := a64ElemField(dst.arr, dst.idx)
|
f, ok := a64ElemField(dst.arr, dst.idx)
|
||||||
if !ok {
|
if !ok {
|
||||||
|
|||||||
@@ -79,6 +79,11 @@ func arm64RegNum(name string) int {
|
|||||||
return 17
|
return 17
|
||||||
case "R18":
|
case "R18":
|
||||||
return 18
|
return 18
|
||||||
|
case "R18_PLATFORM":
|
||||||
|
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
|
||||||
|
// cmd/asm/internal/arch so assembly cannot use it by accident, and
|
||||||
|
// sys_windows_arm64.s references it only through this name.
|
||||||
|
return 18
|
||||||
case "R19":
|
case "R19":
|
||||||
return 19
|
return 19
|
||||||
case "R20":
|
case "R20":
|
||||||
|
|||||||
@@ -105,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
|
|||||||
}{
|
}{
|
||||||
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
|
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
|
||||||
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
|
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
|
||||||
|
{"R18_PLATFORM", 18},
|
||||||
{"F0", 0}, {"F4", 4}, {"F31", 31},
|
{"F0", 0}, {"F4", 4}, {"F31", 31},
|
||||||
{"INVALID", -1}, {"X0", -1}, {"", -1},
|
{"INVALID", -1}, {"X0", -1}, {"", -1},
|
||||||
}
|
}
|
||||||
@@ -861,6 +862,99 @@ func TestArm64SIMDElement(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
|
||||||
|
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
|
||||||
|
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
|
||||||
|
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
|
||||||
|
// rejected, as the toolchain rejects it.
|
||||||
|
func TestArm64GPIntoVector(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
|
||||||
|
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
|
||||||
|
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x4e010ca5, // VMOV R5, V5.B16
|
||||||
|
0x0e010c22, // VMOV R1, V2.B8
|
||||||
|
0x0e020c64, // VMOV R3, V4.H4
|
||||||
|
0x4e040d2a, // VMOV R9, V10.S4
|
||||||
|
0x4e020cff, // VMOV R7, V31.H8
|
||||||
|
0x4e080d6c, // VMOV R11, V12.D2
|
||||||
|
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
|
||||||
|
0x4e020d2a, // VDUP R9, V10.H8
|
||||||
|
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
|
||||||
|
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
|
||||||
|
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
|
||||||
|
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
|
||||||
|
func TestArm64SimdTwoOperand(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x5ee78508, // VADD V7, V8
|
||||||
|
0x7ee78508, // VSUB V7, V8
|
||||||
|
0x5ee18442, // VADD V1, V2
|
||||||
|
0x4e208422, // VADD arranged: the ordinary three-register path
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64TruncMove pins the truncating register moves against
|
||||||
|
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
|
||||||
|
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
|
||||||
|
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
|
||||||
|
// drops to the W ORR too (asm7.go case 45).
|
||||||
|
func TestArm64TruncMove(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
|
||||||
|
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
|
||||||
|
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x93401c64, // MOVB = SXTB
|
||||||
|
0x93403ca6, // MOVH = SXTH
|
||||||
|
0x93407d2a, // MOVW = SXTW
|
||||||
|
0xd3401c64, // MOVBU = UXTB
|
||||||
|
0xd3403c64, // MOVHU = UXTH
|
||||||
|
0x2a0303e4, // MOVWU = ORR W
|
||||||
|
0xaa0303e4, // MOVD = ORR X
|
||||||
|
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
|
||||||
|
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
|
||||||
|
0x2a1f03e5, // MOVWU ZR, R5
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestArm64SIMDLoadStore pins the structure loads and stores.
|
// TestArm64SIMDLoadStore pins the structure loads and stores.
|
||||||
func TestArm64SIMDLoadStore(t *testing.T) {
|
func TestArm64SIMDLoadStore(t *testing.T) {
|
||||||
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
|
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
|
||||||
|
|||||||
Vendored
+31
@@ -0,0 +1,31 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// Differential kernel for the arm64 bookkeeping statements: the funcdata.h
|
||||||
|
// pseudo-directives (GO_ARGS, NO_LOCAL_POINTERS, FUNCDATA, PCDATA) contribute
|
||||||
|
// no instruction bytes, and every function is byte-compared against
|
||||||
|
// go tool asm.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
#include "funcdata.h"
|
||||||
|
|
||||||
|
// func bookkeep()
|
||||||
|
TEXT ·bookkeep(SB), NOSPLIT, $8-0
|
||||||
|
GO_ARGS
|
||||||
|
FUNCDATA $3, inline_tree(SB)
|
||||||
|
PCDATA $1, $2
|
||||||
|
MOVD R1, 0(RSP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func bookkeepNoLocals()
|
||||||
|
TEXT ·bookkeepNoLocals(SB), NOSPLIT, $16-0
|
||||||
|
NO_LOCAL_POINTERS
|
||||||
|
PCDATA $0, $0
|
||||||
|
PCDATA $1, $1
|
||||||
|
MOVD R2, 8(RSP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func bookkeepPlain()
|
||||||
|
TEXT ·bookkeepPlain(SB), NOSPLIT, $0-0
|
||||||
|
MOVD R3, R4
|
||||||
|
RET
|
||||||
Vendored
+57
@@ -0,0 +1,57 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// Differential kernel for the arm64 whole-vector moves between a general
|
||||||
|
// register and an arranged vector (VMOV/VDUP Rs, Vd.<T>), the two-operand
|
||||||
|
// accumulate spellings VADD/VSUB Vm, Vn, and the toolchain-reserved
|
||||||
|
// R18_PLATFORM register name. Every function is byte-compared against
|
||||||
|
// go tool asm.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func gpIntoVector()
|
||||||
|
TEXT ·gpIntoVector(SB), NOSPLIT, $0-0
|
||||||
|
VMOV R1, V2.B8
|
||||||
|
VMOV R3, V4.B16
|
||||||
|
VMOV R5, V6.H4
|
||||||
|
VMOV R7, V8.H8
|
||||||
|
VMOV R9, V10.S2
|
||||||
|
VMOV R11, V12.S4
|
||||||
|
VMOV R13, V14.D2
|
||||||
|
VDUP R15, V16.B8
|
||||||
|
VDUP R17, V18.B16
|
||||||
|
VDUP R19, V20.H8
|
||||||
|
VDUP R21, V22.S4
|
||||||
|
VDUP R23, V24.D2
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func simdAccumulate()
|
||||||
|
TEXT ·simdAccumulate(SB), NOSPLIT, $0-0
|
||||||
|
VADD V7, V8
|
||||||
|
VSUB V7, V8
|
||||||
|
VADD V1, V2
|
||||||
|
VSUB V30, V31
|
||||||
|
VADD V0.B16, V1.B16, V2.B16
|
||||||
|
VSUB V0.S4, V1.S4, V2.S4
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func truncMove()
|
||||||
|
TEXT ·truncMove(SB), NOSPLIT, $0-0
|
||||||
|
MOVB R3, R4
|
||||||
|
MOVH R5, R6
|
||||||
|
MOVW R9, R10
|
||||||
|
MOVBU R3, R4
|
||||||
|
MOVHU R3, R4
|
||||||
|
MOVWU R3, R4
|
||||||
|
MOVD R3, R4
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func platformRegister()
|
||||||
|
TEXT ·platformRegister(SB), NOSPLIT, $0-0
|
||||||
|
MOVD R18_PLATFORM, R3
|
||||||
|
MOVW R18_PLATFORM, R4
|
||||||
|
MOVD R3, R18_PLATFORM
|
||||||
|
MOVD 0x68(R18_PLATFORM), R5
|
||||||
|
MOVD R5, 0x68(R18_PLATFORM)
|
||||||
|
MOVW 8(R18_PLATFORM), R6
|
||||||
|
RET
|
||||||
Reference in New Issue
Block a user