From 9ef14bdb710e4288d5e2b30d8e46198277b0f13f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 00:21:27 +0200 Subject: [PATCH] feat(asm): encode the arm64 SIMD arrangement bits Assisted-by: GLM 5.3 Flash --- asm/arm64_assemble.go | 77 +++++++-- asm/arm64_encode.go | 260 ++++++++++++++++--------------- asm/arm64_encode_test.go | 84 ++++++++++ testdata/verify/simdarr_arm64.s | 118 ++++++++++++++ verify/arm64_groundtruth_test.go | 1 + 5 files changed, 407 insertions(+), 133 deletions(-) create mode 100644 testdata/verify/simdarr_arm64.s diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index faf7130..199d911 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -3028,8 +3028,10 @@ func arm64SimdNarrowPair(mnem, src, dst string, two bool) error { } // arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide -// instruction contributes: the driving arrangement's size and Q bits, or for -// the FCVT family only the Q bit, whose size field is fixed in the base. +// instruction contributes: the driving arrangement's size and Q bits, for +// the FCVT family only the Q bit (whose size field is fixed in the base), +// and for the long extend family the immh shift field the long forms imply +// (immh = esize/8) plus the Q bit for the .2 spellings. func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 { if spec.qonly { if two { @@ -3037,6 +3039,14 @@ func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 { } return 0 } + if spec.form == a64NLTwoLong { + se, _, _ := arm64SimdNLArr(drive) + bits := uint32(se) << 19 + if two { + bits |= 1 << 30 + } + return bits + } return a64ArrBits[a64ArrIndex(drive)] } @@ -3101,8 +3111,10 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b var pairErr error if spec.form == a64NLThreeWide { // UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one; - // the arrangement bits follow the wide side. - drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two) + // the size bits follow the narrow side (Vm) and the 128-bit flag + // follows the spelling: the plain form keeps Q clear, the .2 + // form sets it. + drive, pairErr = vm.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two) } else { // MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the // wide one; the arrangement bits follow the narrow source. @@ -3118,6 +3130,14 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr) } arrBits := arm64SimdNLArrBits(spec, drive, two) + if spec.form == a64NLThreeWide { + // The size bits ride the narrow side's letter with Q forced by + // the spelling alone. + arrBits = a64ArrBits[a64ArrIndex(drive)] &^ (1 << 30) + if two { + arrBits |= 1 << 30 + } + } return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil case a64NLThreeLongShift, a64NLThreeNarrowShift: if len(ops) != 3 || !isImmOperand(ops[0]) { @@ -4082,7 +4102,7 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt } allowed := uint16(0x7f) if strings.HasPrefix(mnem, "VFCM") { - allowed = 1<>1<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil + var imm4 uint32 + switch src.arr { + case "B": + imm4 = uint32(src.idx) + case "H": + imm4 = uint32(src.idx) << 1 + case "S": + imm4 = uint32(src.idx) << 2 + case "D": + imm4 = uint32(src.idx) << 3 + default: + return nil, fmt.Errorf("%s: invalid element operand", mnem) + } + return a64wordLE(0x6e000400 | df<<16 | imm4&0xf<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil } if dstGP { // Element to a general register: UMOV, with the D form setting bit diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index 18b6ca0..e8dd062 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -1010,13 +1010,16 @@ func init() { } // a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word, -// the set of arrangements it accepts as a bitmask over the a64Arr index and, -// for instructions that exist at a single arrangement and carry that -// arrangement's bits inside the base already, the fixed flag. +// the set of arrangements it accepts as a bitmask over the a64Arr index, +// the fixed flag for instructions that exist at a single arrangement and +// carry that arrangement's bits inside the base already, and the fp flag for +// the FP rows, whose size field is the single FP bit (a64FPArrBits) instead +// of the integer size. type a64SimdVSpec struct { base uint32 arrs uint16 fixed bool + fp bool } // a64Arr names the vector arrangements the encoders deal with, indexed by @@ -1063,12 +1066,25 @@ func a64ElemLetter(s string) bool { return false } -// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms -// accept: H, S and D widths for the pairwise data-processing, H and S for -// the across-vector reductions. -var fpSimdArrs = uint16(1< 0 { + continue // a parse rejection is a rejection + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("expected rejection for %q, got nil", strings.TrimSpace(src)) + } + } +} diff --git a/testdata/verify/simdarr_arm64.s b/testdata/verify/simdarr_arm64.s new file mode 100644 index 0000000..12a40ad --- /dev/null +++ b/testdata/verify/simdarr_arm64.s @@ -0,0 +1,118 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 SIMD arrangement bits: the FP one-bit +// size field (S=0, D=1 at bit 22), the immh field of the long extends, the +// narrow family's fixed bit 16, the SSHL/USHL size bits, the scalar D forms +// of the bare VADD/VSUB spellings and the INS lane packing. Every function +// is byte-compared against go tool asm. + +#include "textflag.h" + +// func fpthree() +TEXT ·fpthree(SB), NOSPLIT, $0-0 + VFADD V0.S4, V0.S4, V1.S4 + VFADD V0.D2, V0.D2, V1.D2 + VFMUL V0.S4, V0.S4, V1.S4 + VFMUL V0.D2, V0.D2, V1.D2 + VFDIV V0.S4, V0.S4, V1.S4 + VFDIV V0.D2, V0.D2, V1.D2 + VFMLA V1.D2, V12.D2, V1.D2 + VFMLA V1.S2, V12.S2, V1.S2 + VFMLA V1.S4, V12.S4, V1.S4 + VFMAX V3.S4, V2.S4, V1.S4 + VFMAXNM V3.S4, V2.S4, V1.S4 + VFADDP V3.D2, V2.D2, V1.D2 + VFCMEQ V1.S4, V2.S4, V3.S4 + VFCMGE V1.S4, V2.S4, V3.S4 + RET + +// func fpunary() +TEXT ·fpunary(SB), NOSPLIT, $0-0 + VFABS V0.D2, V1.D2 + VFNEG V0.D2, V1.D2 + VFSQRT V0.D2, V1.D2 + VFRINTN V0.D2, V1.D2 + VFRINTP V0.D2, V1.D2 + VFRINTM V0.D2, V1.D2 + VFRINTZ V0.D2, V1.D2 + VFCVTZS V1.D2, V2.D2 + VFCVTZU V1.D2, V2.D2 + VSCVTF V1.D2, V2.D2 + VUCVTF V1.D2, V2.D2 + VFABS V0.S4, V1.S4 + VSCVTF V1.S4, V2.S4 + VFNEG V0.S2, V1.S2 + RET + +// func shifts() +TEXT ·shifts(SB), NOSPLIT, $0-0 + VSSHL V1.S4, V2.S4, V3.S4 + VSSHL V1.S2, V2.S2, V3.S2 + VSSHL V1.H4, V2.H4, V3.H4 + VSSHL V1.H8, V2.H8, V3.H8 + VSSHL V1.B8, V2.B8, V3.B8 + VSSHL V1.B16, V2.B16, V3.B16 + VUSHL V1.S4, V2.S4, V3.S4 + VUSHL V1.S2, V2.S2, V3.S2 + VUSHL V1.H4, V2.H4, V3.H4 + VUSHL V1.H8, V2.H8, V3.H8 + VUSHL V1.B8, V2.B8, V3.B8 + VUSHL V1.B16, V2.B16, V3.B16 + VRBIT V24.B8, V24.B8 + RET + +// func widen() +TEXT ·widen(SB), NOSPLIT, $0-0 + VUXTL V30.B8, V30.H8 + VUXTL V30.H4, V29.S4 + VUXTL V29.S2, V2.D2 + VUXTL2 V30.H8, V30.S4 + VUXTL2 V29.S4, V2.D2 + VUXTL2 V30.B16, V2.H8 + VSXTL V1.B8, V2.H8 + VSXTL V1.H4, V2.S4 + VSXTL V1.S2, V2.D2 + VSXTL2 V1.B16, V2.H8 + VSXTL2 V1.H8, V2.S4 + VSXTL2 V1.S4, V2.D2 + VXTN V1.H8, V2.B8 + VXTN V1.S4, V2.H4 + VXTN V1.D2, V2.S2 + VXTN2 V1.H8, V2.B16 + VSQXTN V1.D2, V2.S2 + VSQXTN2 V1.S4, V2.H8 + VSQXTUN V1.H8, V2.B8 + VUQXTN V1.S4, V2.H4 + VUQXTN2 V1.D2, V2.S4 + RET + +// func uaddw() +TEXT ·uaddw(SB), NOSPLIT, $0-0 + VUADDW V9.B8, V12.H8, V14.H8 + VUADDW V13.H4, V10.S4, V11.S4 + VUADDW V21.S2, V24.D2, V29.D2 + VUADDW2 V9.B16, V12.H8, V14.H8 + VUADDW2 V13.H8, V20.S4, V30.S4 + VUADDW2 V21.S4, V24.D2, V29.D2 + RET + +// func bareadd() +TEXT ·bareadd(SB), NOSPLIT, $0-0 + VADD V1, V2, V3 + VADD V1, V3, V3 + VSUB V12, V30, V30 + VSUB V12, V20, V30 + VADD V1.B8, V2.B8, V3.B8 + VSUB V12.B8, V30.B8, V30.B8 + RET + +// func lanes() +TEXT ·lanes(SB), NOSPLIT, $0-0 + VMOV V12.D[0], V12.D[1] + VMOV V10.S[0], V12.S[1] + VMOV V9.H[0], V12.H[1] + VMOV V12.B[0], V12.B[1] + VMOV V12.S[2], V12.S[3] + VMOV V12.H[3], V12.H[5] + RET diff --git a/verify/arm64_groundtruth_test.go b/verify/arm64_groundtruth_test.go index 025e936..7536518 100644 --- a/verify/arm64_groundtruth_test.go +++ b/verify/arm64_groundtruth_test.go @@ -38,6 +38,7 @@ func TestGroundTruthARM64(t *testing.T) { "../testdata/verify/qmov_arm64.s", "../testdata/verify/splits_arm64.s", "../testdata/verify/regoffset_arm64.s", + "../testdata/verify/simdarr_arm64.s", "../testdata/verify/crypto_arm64.s", "../testdata/verify/integer_arm64.s", "../testdata/verify/simd_arm64.s",