feat(asm): encode the arm64 SIMD arrangement bits

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:36:24 +02:00
1 parent 458cdd2066
commit 9ef14bdb71
5 files changed
+407 -133

No files matched your search

+138 -122
View File
@@ -1010,13 +1010,16 @@ func init() {
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
// for instructions that exist at a single arrangement and carry that
// arrangement's bits inside the base already, the fixed flag.
// the set of arrangements it accepts as a bitmask over the a64Arr index,
// the fixed flag for instructions that exist at a single arrangement and
// carry that arrangement's bits inside the base already, and the fp flag for
// the FP rows, whose size field is the single FP bit (a64FPArrBits) instead
// of the integer size.
type a64SimdVSpec struct {
base uint32
arrs uint16
fixed bool
fp bool
}
// a64Arr names the vector arrangements the encoders deal with, indexed by
@@ -1063,12 +1066,25 @@ func a64ElemLetter(s string) bool {
return false
}
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
// accept: H, S and D widths for the pairwise data-processing, H and S for
// the across-vector reductions.
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
// fpSimdArrs bounds the arrangements the FP SIMD forms accept: S and D only,
// the toolchain rejecting the half-width spellings outright ("invalid
// arrangement"). fpAcrossArrs bounds the across-vector reductions, which do
// take the half width.
var fpSimdArrs = uint16(1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
// a64FPArrBits carries the bits an arrangement contributes to the FP SIMD
// words: the FP size field is a single bit at bit 22 (0 for the S widths, 1
// for the D widths; the toolchain carries no half-width FP rows) and the
// 128-bit flag sits at bit 30. Word-verified against go tool asm.
var a64FPArrBits = [a64ArrCount]uint32{
a64Arr2S: 0,
a64Arr4S: 1 << 30,
a64Arr2D: 1<<30 | 1<<22,
a64Arr4H: 0,
a64Arr8H: 1 << 30,
}
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
// flag alone, without the size bits: the FP converts, the FP round-to-integral
// and pairwise compares among them. Word-verified against go tool asm.
@@ -1099,81 +1115,81 @@ var a64ArrBits = [a64ArrCount]uint32{
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VCMGE": {0x0e203c00, 0x7f, false},
"VCMGT": {0x0e203400, 0x7f, false},
"VCMHI": {0x2e203400, 0x7f, false},
"VCMHS": {0x2e203c00, 0x7f, false},
// FP compares take H, S and D arrangements only (the toolchain rejects
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VADD": {0x0e208400, 0x7f, false, false},
"VSUB": {0x2e208400, 0x7f, false, false},
"VMUL": {0x0e209c00, 0x3f, false, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false, false},
"VORR": {0x0ea01c00, 0x03, false, false},
"VADDP": {0x0e20bc00, 0x7f, false, false},
"VZIP1": {0x0e003800, 0x7f, false, false},
"VZIP2": {0x0e007800, 0x7f, false, false},
"VCMEQ": {0x2e208c00, 0x7f, false, false},
"VCMGE": {0x0e203c00, 0x7f, false, false},
"VCMGT": {0x0e203400, 0x7f, false, false},
"VCMHI": {0x2e203400, 0x7f, false, false},
"VCMHS": {0x2e203c00, 0x7f, false, false},
// FP compares take S and D arrangements only (the toolchain rejects the
// byte and half forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, fpSimdArrs, false, true},
"VFCMGE": {0x2e20e400, fpSimdArrs, false, true},
"VFCMGT": {0x2ea0e400, fpSimdArrs, false, true},
// FP arithmetic shares the same arrangement restriction.
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFADD": {0x0e20d400, fpSimdArrs, false, true},
"VFSUB": {0x0ea0d400, fpSimdArrs, false, true},
"VFMUL": {0x2e20dc00, fpSimdArrs, false, true},
"VFDIV": {0x2e20fc00, fpSimdArrs, false, true},
"VFMAX": {0x0e20f400, fpSimdArrs, false, true},
"VFMIN": {0x0ea0f400, fpSimdArrs, false, true},
"VFMAXNM": {0x0e20c400, fpSimdArrs, false, true},
"VFMINNM": {0x0ea0c400, fpSimdArrs, false, true},
"VFMLA": {0x0e20cc00, fpSimdArrs, false, true},
"VFMLS": {0x0ea0cc00, fpSimdArrs, false, true},
// Saturating, halving, polynomial and pairwise arithmetic, the logical
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
// tool asm.
"VBIC": {0x0e601c00, 0x7f, false},
"VBIF": {0x2ee01c00, 0x7f, false},
"VBIT": {0x6ea01c00, 0x7f, false},
"VBSL": {0x6e601c00, 0x7f, false},
"VCMTST": {0x0e208c00, 0x7f, false},
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VMLA": {0x4ea09400, 0x7f, false},
"VMLS": {0x6ea09400, 0x7f, false},
"VORN": {0x4ee01c00, 0x7f, false},
"VSHADD": {0x4ea00400, 0x7f, false},
"VSRHADD": {0x4ea01400, 0x7f, false},
"VUHADD": {0x6ea00400, 0x7f, false},
"VURHADD": {0x6ea01400, 0x7f, false},
"VSMAX": {0x4ea06400, 0x7f, false},
"VSMIN": {0x4ea06c00, 0x7f, false},
"VSMAXP": {0x4ea0a400, 0x7f, false},
"VSMINP": {0x4ea0ac00, 0x7f, false},
"VUMAX": {0x2e206400, 0x7f, false},
"VUMIN": {0x2e206c00, 0x7f, false},
"VUMAXP": {0x6ea0a400, 0x7f, false},
"VUMINP": {0x6ea0ac00, 0x7f, false},
"VSQADD": {0x4ea00c00, 0x7f, false},
"VUQADD": {0x6ea00c00, 0x7f, false},
"VSQSUB": {0x4ea02c00, 0x7f, false},
"VUQSUB": {0x6ea02c00, 0x7f, false},
"VSSHL": {0x4ee04400, 0x7f, false},
"VUSHL": {0x6ee04400, 0x7f, false},
"VUZP1": {0x0e001800, 0x7f, false},
"VUZP2": {0x4ec05800, 0x7f, false},
"VTRN1": {0x4ec02800, 0x7f, false},
"VTRN2": {0x4ec06800, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
"VBIC": {0x0e601c00, 0x7f, false, false},
"VBIF": {0x2ee01c00, 0x7f, false, false},
"VBIT": {0x6ea01c00, 0x7f, false, false},
"VBSL": {0x6e601c00, 0x7f, false, false},
"VCMTST": {0x0e208c00, 0x7f, false, false},
"VFADDP": {0x2e20d400, fpSimdArrs, false, true},
"VFMAXP": {0x2e20f400, fpSimdArrs, false, true},
"VFMINP": {0x6ea0f400, fpSimdArrs, false, true},
"VFMAXNMP": {0x2e20c400, fpSimdArrs, false, true},
"VFMINNMP": {0x6ea0c400, fpSimdArrs, false, true},
"VMLA": {0x4ea09400, 0x7f, false, false},
"VMLS": {0x6ea09400, 0x7f, false, false},
"VORN": {0x4ee01c00, 0x7f, false, false},
"VSHADD": {0x4ea00400, 0x7f, false, false},
"VSRHADD": {0x4ea01400, 0x7f, false, false},
"VUHADD": {0x6ea00400, 0x7f, false, false},
"VURHADD": {0x6ea01400, 0x7f, false, false},
"VSMAX": {0x4ea06400, 0x7f, false, false},
"VSMIN": {0x4ea06c00, 0x7f, false, false},
"VSMAXP": {0x4ea0a400, 0x7f, false, false},
"VSMINP": {0x4ea0ac00, 0x7f, false, false},
"VUMAX": {0x2e206400, 0x7f, false, false},
"VUMIN": {0x2e206c00, 0x7f, false, false},
"VUMAXP": {0x6ea0a400, 0x7f, false, false},
"VUMINP": {0x6ea0ac00, 0x7f, false, false},
"VSQADD": {0x4ea00c00, 0x7f, false, false},
"VUQADD": {0x6ea00c00, 0x7f, false, false},
"VSQSUB": {0x4ea02c00, 0x7f, false, false},
"VUQSUB": {0x6ea02c00, 0x7f, false, false},
"VSSHL": {0x0e204400, 0x7f, false, false},
"VUSHL": {0x2e204400, 0x7f, false, false},
"VUZP1": {0x0e001800, 0x7f, false, false},
"VUZP2": {0x4ec05800, 0x7f, false, false},
"VTRN1": {0x4ec02800, 0x7f, false, false},
"VTRN2": {0x4ec06800, 0x7f, false, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true, false}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false, false},
// Saturating shifts, register forms (the immediate spellings route to
// a64FShiftImm).
"VSQSHL": {0x0e204c00, 0x7f, false},
"VUQSHL": {0x2e204c00, 0x7f, false},
"VSQSHL": {0x0e204c00, 0x7f, false, false},
"VUQSHL": {0x2e204c00, 0x7f, false, false},
}
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
@@ -1212,18 +1228,18 @@ var a64SimdNLTable = map[string]a64SimdNLSpec{
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
"VXTN": {0x0e212800, a64NLTwoNarrow, false},
"VXTN2": {0x0e212800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e214800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e214800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e212800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e212800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e214800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e214800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e616800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e616800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e617800, a64NLTwoLong, true},
"VFCVTL2": {0x0e617800, a64NLTwoLong, true},
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
@@ -1267,43 +1283,43 @@ var a64SimdVZero = map[string]uint32{
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false, false},
"VREV64": {0x0e200800, 0x3f, false, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
"VUADDLV": {0x2e303800, 0x3f, false, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
// Two-register data-processing across one arrangement.
"VABS": {0x0e20b800, 0x7f, false},
"VNEG": {0x2e20b800, 0x7f, false},
"VCLS": {0x0e204800, 0x7f, false},
"VCLZ": {0x2e204800, 0x7f, false},
"VCNT": {0x0e205800, 0x7f, false},
"VNOT": {0x2e205800, 0x7f, false},
"VSQABS": {0x0e207800, 0x7f, false},
"VSQNEG": {0x2e207800, 0x7f, false},
"VRBIT": {0x6e605800, 0x7f, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
"VFABS": {0x0ea0f800, fpSimdArrs, false},
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
"VFRINTN": {0x0e218800, fpSimdArrs, false},
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
"VFRINTM": {0x0e219800, fpSimdArrs, false},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
"VABS": {0x0e20b800, 0x7f, false, false},
"VNEG": {0x2e20b800, 0x7f, false, false},
"VCLS": {0x0e204800, 0x7f, false, false},
"VCLZ": {0x2e204800, 0x7f, false, false},
"VCNT": {0x0e205800, 0x7f, false, false},
"VNOT": {0x2e205800, 0x7f, false, false},
"VSQABS": {0x0e207800, 0x7f, false, false},
"VSQNEG": {0x2e207800, 0x7f, false, false},
"VRBIT": {0x2e605800, 0x7f, false, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false, true},
"VUCVTF": {0x6e21d800, fpSimdArrs, false, true},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false, true},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false, true},
"VFABS": {0x0ea0f800, fpSimdArrs, false, true},
"VFNEG": {0x2ea0f800, fpSimdArrs, false, true},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false, true},
"VFRINTN": {0x0e218800, fpSimdArrs, false, true},
"VFRINTP": {0x0ea18800, fpSimdArrs, false, true},
"VFRINTM": {0x0e219800, fpSimdArrs, false, true},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false, true},
// Across-vector reductions: the operand arrangement rides as usual and
// the destination stays a bare V register.
"VADDV": {0x0e31b800, 0x3f, false},
"VSMAXV": {0x0e30a800, 0x3f, false},
"VSMINV": {0x0e31a800, 0x3f, false},
"VUMAXV": {0x2e30a800, 0x3f, false},
"VUMINV": {0x2e31a800, 0x3f, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
"VADDV": {0x0e31b800, 0x3f, false, false},
"VSMAXV": {0x0e30a800, 0x3f, false, false},
"VSMINV": {0x0e31a800, 0x3f, false, false},
"VUMAXV": {0x2e30a800, 0x3f, false, false},
"VUMINV": {0x2e31a800, 0x3f, false, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must