feat(asm): encode the arm64 SIMD arrangement bits

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:36:24 +02:00
1 parent 458cdd2066
commit 9ef14bdb71
5 files changed
+407 -133

No files matched your search

+66 -11
View File
@@ -3028,8 +3028,10 @@ func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
}
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
// instruction contributes: the driving arrangement's size and Q bits, or for
// the FCVT family only the Q bit, whose size field is fixed in the base.
// instruction contributes: the driving arrangement's size and Q bits, for
// the FCVT family only the Q bit (whose size field is fixed in the base),
// and for the long extend family the immh shift field the long forms imply
// (immh = esize/8) plus the Q bit for the .2 spellings.
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
if spec.qonly {
if two {
@@ -3037,6 +3039,14 @@ func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
}
return 0
}
if spec.form == a64NLTwoLong {
se, _, _ := arm64SimdNLArr(drive)
bits := uint32(se) << 19
if two {
bits |= 1 << 30
}
return bits
}
return a64ArrBits[a64ArrIndex(drive)]
}
@@ -3101,8 +3111,10 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
var pairErr error
if spec.form == a64NLThreeWide {
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
// the arrangement bits follow the wide side.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
// the size bits follow the narrow side (Vm) and the 128-bit flag
// follows the spelling: the plain form keeps Q clear, the .2
// form sets it.
drive, pairErr = vm.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
} else {
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
// wide one; the arrangement bits follow the narrow source.
@@ -3118,6 +3130,14 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
if spec.form == a64NLThreeWide {
// The size bits ride the narrow side's letter with Q forced by
// the spelling alone.
arrBits = a64ArrBits[a64ArrIndex(drive)] &^ (1 << 30)
if two {
arrBits |= 1 << 30
}
}
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongShift, a64NLThreeNarrowShift:
if len(ops) != 3 || !isImmOperand(ops[0]) {
@@ -4082,7 +4102,7 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
allowed := uint16(0x7f)
if strings.HasPrefix(mnem, "VFCM") {
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
allowed = fpSimdArrs
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
if err != nil {
@@ -4118,15 +4138,28 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
vs[i], arrs[i] = v, v.arr
}
// The bare three-register spellings of VADD/VSUB are the scalar D forms
// (the toolchain's ADD/SUB scalar rows), not the 8B vector rows.
if (mnem == "VADD" || mnem == "VSUB") &&
arrs[0] == "" && arrs[1] == "" && arrs[2] == "" {
base := uint32(0x5ee08400) // VADD scalar
if mnem == "VSUB" {
base = 0x7ee08400
}
return a64wordLE(base | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
}
arr, err := arm64SimdArrs(mnem, arrs, spec.arrs)
if err != nil {
return nil, err
}
arrBits := a64ArrBits[arr]
if spec.fixed {
if spec.fp {
// The FP rows carry a one-bit size field (S=0, D=1) instead of the
// integer size, and no Q-only masking applies to them.
arrBits = a64FPArrBits[arr]
} else if spec.fixed {
arrBits = 0
}
if a64SimdQOnly[mnem] {
} else if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
@@ -4175,7 +4208,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
return nil, err
}
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
if spec.fp {
// The FP rows carry the one-bit FP size field instead of the
// integer size, and no Q-only masking applies to them.
arrBits = a64FPArrBits[arr]
} else if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
@@ -4402,12 +4439,30 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
if dst.hasIdx {
// Element to element.
// Element to element: the toolchain requires the two element letters
// to match (asm7.go case 92), packs the destination index into imm5
// and the source index, in units of the element size, into imm4.
if src.arr != dst.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s elements", mnem, src.arr, dst.arr)
}
df, ok := a64ElemField(dst.arr, dst.idx)
if !ok {
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
return a64wordLE(0x6e000400 | df<<16 | sf>>1<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
var imm4 uint32
switch src.arr {
case "B":
imm4 = uint32(src.idx)
case "H":
imm4 = uint32(src.idx) << 1
case "S":
imm4 = uint32(src.idx) << 2
case "D":
imm4 = uint32(src.idx) << 3
default:
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
return a64wordLE(0x6e000400 | df<<16 | imm4&0xf<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
}
if dstGP {
// Element to a general register: UMOV, with the D form setting bit
+138 -122
View File
@@ -1010,13 +1010,16 @@ func init() {
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
// for instructions that exist at a single arrangement and carry that
// arrangement's bits inside the base already, the fixed flag.
// the set of arrangements it accepts as a bitmask over the a64Arr index,
// the fixed flag for instructions that exist at a single arrangement and
// carry that arrangement's bits inside the base already, and the fp flag for
// the FP rows, whose size field is the single FP bit (a64FPArrBits) instead
// of the integer size.
type a64SimdVSpec struct {
base uint32
arrs uint16
fixed bool
fp bool
}
// a64Arr names the vector arrangements the encoders deal with, indexed by
@@ -1063,12 +1066,25 @@ func a64ElemLetter(s string) bool {
return false
}
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
// accept: H, S and D widths for the pairwise data-processing, H and S for
// the across-vector reductions.
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
// fpSimdArrs bounds the arrangements the FP SIMD forms accept: S and D only,
// the toolchain rejecting the half-width spellings outright ("invalid
// arrangement"). fpAcrossArrs bounds the across-vector reductions, which do
// take the half width.
var fpSimdArrs = uint16(1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
// a64FPArrBits carries the bits an arrangement contributes to the FP SIMD
// words: the FP size field is a single bit at bit 22 (0 for the S widths, 1
// for the D widths; the toolchain carries no half-width FP rows) and the
// 128-bit flag sits at bit 30. Word-verified against go tool asm.
var a64FPArrBits = [a64ArrCount]uint32{
a64Arr2S: 0,
a64Arr4S: 1 << 30,
a64Arr2D: 1<<30 | 1<<22,
a64Arr4H: 0,
a64Arr8H: 1 << 30,
}
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
// flag alone, without the size bits: the FP converts, the FP round-to-integral
// and pairwise compares among them. Word-verified against go tool asm.
@@ -1099,81 +1115,81 @@ var a64ArrBits = [a64ArrCount]uint32{
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VCMGE": {0x0e203c00, 0x7f, false},
"VCMGT": {0x0e203400, 0x7f, false},
"VCMHI": {0x2e203400, 0x7f, false},
"VCMHS": {0x2e203c00, 0x7f, false},
// FP compares take H, S and D arrangements only (the toolchain rejects
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VADD": {0x0e208400, 0x7f, false, false},
"VSUB": {0x2e208400, 0x7f, false, false},
"VMUL": {0x0e209c00, 0x3f, false, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false, false},
"VORR": {0x0ea01c00, 0x03, false, false},
"VADDP": {0x0e20bc00, 0x7f, false, false},
"VZIP1": {0x0e003800, 0x7f, false, false},
"VZIP2": {0x0e007800, 0x7f, false, false},
"VCMEQ": {0x2e208c00, 0x7f, false, false},
"VCMGE": {0x0e203c00, 0x7f, false, false},
"VCMGT": {0x0e203400, 0x7f, false, false},
"VCMHI": {0x2e203400, 0x7f, false, false},
"VCMHS": {0x2e203c00, 0x7f, false, false},
// FP compares take S and D arrangements only (the toolchain rejects the
// byte and half forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, fpSimdArrs, false, true},
"VFCMGE": {0x2e20e400, fpSimdArrs, false, true},
"VFCMGT": {0x2ea0e400, fpSimdArrs, false, true},
// FP arithmetic shares the same arrangement restriction.
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFADD": {0x0e20d400, fpSimdArrs, false, true},
"VFSUB": {0x0ea0d400, fpSimdArrs, false, true},
"VFMUL": {0x2e20dc00, fpSimdArrs, false, true},
"VFDIV": {0x2e20fc00, fpSimdArrs, false, true},
"VFMAX": {0x0e20f400, fpSimdArrs, false, true},
"VFMIN": {0x0ea0f400, fpSimdArrs, false, true},
"VFMAXNM": {0x0e20c400, fpSimdArrs, false, true},
"VFMINNM": {0x0ea0c400, fpSimdArrs, false, true},
"VFMLA": {0x0e20cc00, fpSimdArrs, false, true},
"VFMLS": {0x0ea0cc00, fpSimdArrs, false, true},
// Saturating, halving, polynomial and pairwise arithmetic, the logical
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
// tool asm.
"VBIC": {0x0e601c00, 0x7f, false},
"VBIF": {0x2ee01c00, 0x7f, false},
"VBIT": {0x6ea01c00, 0x7f, false},
"VBSL": {0x6e601c00, 0x7f, false},
"VCMTST": {0x0e208c00, 0x7f, false},
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VMLA": {0x4ea09400, 0x7f, false},
"VMLS": {0x6ea09400, 0x7f, false},
"VORN": {0x4ee01c00, 0x7f, false},
"VSHADD": {0x4ea00400, 0x7f, false},
"VSRHADD": {0x4ea01400, 0x7f, false},
"VUHADD": {0x6ea00400, 0x7f, false},
"VURHADD": {0x6ea01400, 0x7f, false},
"VSMAX": {0x4ea06400, 0x7f, false},
"VSMIN": {0x4ea06c00, 0x7f, false},
"VSMAXP": {0x4ea0a400, 0x7f, false},
"VSMINP": {0x4ea0ac00, 0x7f, false},
"VUMAX": {0x2e206400, 0x7f, false},
"VUMIN": {0x2e206c00, 0x7f, false},
"VUMAXP": {0x6ea0a400, 0x7f, false},
"VUMINP": {0x6ea0ac00, 0x7f, false},
"VSQADD": {0x4ea00c00, 0x7f, false},
"VUQADD": {0x6ea00c00, 0x7f, false},
"VSQSUB": {0x4ea02c00, 0x7f, false},
"VUQSUB": {0x6ea02c00, 0x7f, false},
"VSSHL": {0x4ee04400, 0x7f, false},
"VUSHL": {0x6ee04400, 0x7f, false},
"VUZP1": {0x0e001800, 0x7f, false},
"VUZP2": {0x4ec05800, 0x7f, false},
"VTRN1": {0x4ec02800, 0x7f, false},
"VTRN2": {0x4ec06800, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
"VBIC": {0x0e601c00, 0x7f, false, false},
"VBIF": {0x2ee01c00, 0x7f, false, false},
"VBIT": {0x6ea01c00, 0x7f, false, false},
"VBSL": {0x6e601c00, 0x7f, false, false},
"VCMTST": {0x0e208c00, 0x7f, false, false},
"VFADDP": {0x2e20d400, fpSimdArrs, false, true},
"VFMAXP": {0x2e20f400, fpSimdArrs, false, true},
"VFMINP": {0x6ea0f400, fpSimdArrs, false, true},
"VFMAXNMP": {0x2e20c400, fpSimdArrs, false, true},
"VFMINNMP": {0x6ea0c400, fpSimdArrs, false, true},
"VMLA": {0x4ea09400, 0x7f, false, false},
"VMLS": {0x6ea09400, 0x7f, false, false},
"VORN": {0x4ee01c00, 0x7f, false, false},
"VSHADD": {0x4ea00400, 0x7f, false, false},
"VSRHADD": {0x4ea01400, 0x7f, false, false},
"VUHADD": {0x6ea00400, 0x7f, false, false},
"VURHADD": {0x6ea01400, 0x7f, false, false},
"VSMAX": {0x4ea06400, 0x7f, false, false},
"VSMIN": {0x4ea06c00, 0x7f, false, false},
"VSMAXP": {0x4ea0a400, 0x7f, false, false},
"VSMINP": {0x4ea0ac00, 0x7f, false, false},
"VUMAX": {0x2e206400, 0x7f, false, false},
"VUMIN": {0x2e206c00, 0x7f, false, false},
"VUMAXP": {0x6ea0a400, 0x7f, false, false},
"VUMINP": {0x6ea0ac00, 0x7f, false, false},
"VSQADD": {0x4ea00c00, 0x7f, false, false},
"VUQADD": {0x6ea00c00, 0x7f, false, false},
"VSQSUB": {0x4ea02c00, 0x7f, false, false},
"VUQSUB": {0x6ea02c00, 0x7f, false, false},
"VSSHL": {0x0e204400, 0x7f, false, false},
"VUSHL": {0x2e204400, 0x7f, false, false},
"VUZP1": {0x0e001800, 0x7f, false, false},
"VUZP2": {0x4ec05800, 0x7f, false, false},
"VTRN1": {0x4ec02800, 0x7f, false, false},
"VTRN2": {0x4ec06800, 0x7f, false, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true, false}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false, false},
// Saturating shifts, register forms (the immediate spellings route to
// a64FShiftImm).
"VSQSHL": {0x0e204c00, 0x7f, false},
"VUQSHL": {0x2e204c00, 0x7f, false},
"VSQSHL": {0x0e204c00, 0x7f, false, false},
"VUQSHL": {0x2e204c00, 0x7f, false, false},
}
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
@@ -1212,18 +1228,18 @@ var a64SimdNLTable = map[string]a64SimdNLSpec{
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
"VXTN": {0x0e212800, a64NLTwoNarrow, false},
"VXTN2": {0x0e212800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e214800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e214800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e212800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e212800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e214800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e214800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e616800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e616800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e617800, a64NLTwoLong, true},
"VFCVTL2": {0x0e617800, a64NLTwoLong, true},
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
@@ -1267,43 +1283,43 @@ var a64SimdVZero = map[string]uint32{
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false, false},
"VREV64": {0x0e200800, 0x3f, false, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
"VUADDLV": {0x2e303800, 0x3f, false, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
// Two-register data-processing across one arrangement.
"VABS": {0x0e20b800, 0x7f, false},
"VNEG": {0x2e20b800, 0x7f, false},
"VCLS": {0x0e204800, 0x7f, false},
"VCLZ": {0x2e204800, 0x7f, false},
"VCNT": {0x0e205800, 0x7f, false},
"VNOT": {0x2e205800, 0x7f, false},
"VSQABS": {0x0e207800, 0x7f, false},
"VSQNEG": {0x2e207800, 0x7f, false},
"VRBIT": {0x6e605800, 0x7f, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
"VFABS": {0x0ea0f800, fpSimdArrs, false},
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
"VFRINTN": {0x0e218800, fpSimdArrs, false},
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
"VFRINTM": {0x0e219800, fpSimdArrs, false},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
"VABS": {0x0e20b800, 0x7f, false, false},
"VNEG": {0x2e20b800, 0x7f, false, false},
"VCLS": {0x0e204800, 0x7f, false, false},
"VCLZ": {0x2e204800, 0x7f, false, false},
"VCNT": {0x0e205800, 0x7f, false, false},
"VNOT": {0x2e205800, 0x7f, false, false},
"VSQABS": {0x0e207800, 0x7f, false, false},
"VSQNEG": {0x2e207800, 0x7f, false, false},
"VRBIT": {0x2e605800, 0x7f, false, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false, true},
"VUCVTF": {0x6e21d800, fpSimdArrs, false, true},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false, true},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false, true},
"VFABS": {0x0ea0f800, fpSimdArrs, false, true},
"VFNEG": {0x2ea0f800, fpSimdArrs, false, true},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false, true},
"VFRINTN": {0x0e218800, fpSimdArrs, false, true},
"VFRINTP": {0x0ea18800, fpSimdArrs, false, true},
"VFRINTM": {0x0e219800, fpSimdArrs, false, true},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false, true},
// Across-vector reductions: the operand arrangement rides as usual and
// the destination stays a bare V register.
"VADDV": {0x0e31b800, 0x3f, false},
"VSMAXV": {0x0e30a800, 0x3f, false},
"VSMINV": {0x0e31a800, 0x3f, false},
"VUMAXV": {0x2e30a800, 0x3f, false},
"VUMINV": {0x2e31a800, 0x3f, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
"VADDV": {0x0e31b800, 0x3f, false, false},
"VSMAXV": {0x0e30a800, 0x3f, false, false},
"VSMINV": {0x0e31a800, 0x3f, false, false},
"VUMAXV": {0x2e30a800, 0x3f, false, false},
"VUMINV": {0x2e31a800, 0x3f, false, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
+84
View File
@@ -1894,3 +1894,87 @@ func TestArm64RegOffsetRejections(t *testing.T) {
}
}
}
// TestArm64SimdArrangementBits pins the arrangement bits the first SIMD pass
// got wrong, word-verified against `go tool asm` (Go 1.27, arm64): the FP
// one-bit size field (S=0, D=1 at bit 22), the SSHL/USHL size bits, the long
// extends' immh field, the narrow family's fixed bit 16, the UADDW size bits
// off the narrow side with Q from the spelling, the scalar D forms of the
// bare VADD/VSUB spellings and the INS lane packing.
func TestArm64SimdArrangementBits(t *testing.T) {
got := arm64Words(t,
"\tVFADD V0.S4, V0.S4, V1.S4\n"+
"\tVFADD V0.D2, V0.D2, V1.D2\n"+
"\tVFABS V0.D2, V1.D2\n"+
"\tVSCVTF V1.D2, V2.D2\n"+
"\tVSSHL V1.B8, V2.B8, V3.B8\n"+
"\tVSSHL V1.S4, V2.S4, V3.S4\n"+
"\tVUSHL V1.H4, V2.H4, V3.H4\n"+
"\tVRBIT V24.B8, V24.B8\n"+
"\tVUXTL V30.B8, V30.H8\n"+
"\tVUXTL V29.S2, V2.D2\n"+
"\tVUXTL2 V30.H8, V30.S4\n"+
"\tVXTN V1.H8, V2.B8\n"+
"\tVFCVTN V1.D2, V2.S2\n"+
"\tVFCVTL V1.S2, V2.D2\n"+
"\tVUADDW V13.H4, V10.S4, V11.S4\n"+
"\tVUADDW2 V13.H8, V20.S4, V30.S4\n"+
"\tVADD V1, V2, V3\n"+
"\tVSUB V12, V20, V30\n"+
"\tVMOV V12.S[2], V12.S[3]\n"+
"\tVMOV V12.H[3], V12.H[5]\n")
want := []uint32{
0x4e20d401, // VFADD V0.4S: FP size field clear for S
0x4e60d401, // VFADD V0.2D: FP size bit 22, not bit 23
0x4ee0f801, // VFABS V1.2D: one-bit FP size
0x4e61d822, // VSCVTF V2.2D: bit 22, the Q-only mask must not strip it
0x0e214443, // VSSHL V3.8B: base without the pre-set size and Q bits
0x4ea14443, // VSSHL V3.4S: integer size bits from the arrangement
0x2e614443, // VUSHL V3.4H: U bit plus the H size
0x2e605b18, // VRBIT V24.8B: no Q bit in the base
0x2f08a7de, // VUXTL: immh = 1 at bit 19 for the byte extend
0x2f20a7a2, // VUXTL: immh = 4 at bit 21 for the word extend
0x6f10a7de, // VUXTL2: immh = 2 plus the 128-bit flag
0x0e212822, // VXTN: fixed bit 16 in the base
0x0e616822, // VFCVTN: bits 16 and 22, Q rides the spelling
0x0e617822, // VFCVTL: bit 22, Q rides the spelling
0x2e6d114b, // VUADDW: size bits off the narrow side, Q clear
0x6e6d129e, // VUADDW2: size off the narrow side, Q from the spelling
0x5ee18443, // VADD scalar D form for the bare spelling
0x7eec869e, // VSUB scalar D form for the bare spelling
0x6e1c458c, // INS: imm4 = 2<<2 for the word source lane 2
0x6e16358c, // INS: imm4 = 3<<1 for the halfword source lane 3
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SimdArrangementRejections pins the arrangements the toolchain
// refuses on the FP SIMD rows and the element-to-element moves: the
// half-width FP spellings, the Q1 spelling on the integer shifts, and the
// mixed element letters of INS.
func TestArm64SimdArrangementRejections(t *testing.T) {
for _, src := range []string{
"\tVFADD\tV1.H4, V2.H4, V3.H4\n",
"\tVFADD\tV1.H8, V2.H8, V3.H8\n",
"\tVFABS\tV1.H4, V2.H4\n",
"\tVSCVTF\tV1.H4, V2.H4\n",
"\tVSSHL\tV1.Q1, V2.Q1, V3.Q1\n",
"\tVMOV\tV12.S[0], V12.D[1]\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n")
if len(errs) > 0 {
continue // a parse rejection is a rejection
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("expected rejection for %q, got nil", strings.TrimSpace(src))
}
}
}