feat(asm): encode the arm64 SIMD arrangement bits

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:36:24 +02:00
1 parent 458cdd2066
commit 9ef14bdb71
5 files changed
+407 -133

No files matched your search

+66 -11
View File
@@ -3028,8 +3028,10 @@ func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
}
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
// instruction contributes: the driving arrangement's size and Q bits, or for
// the FCVT family only the Q bit, whose size field is fixed in the base.
// instruction contributes: the driving arrangement's size and Q bits, for
// the FCVT family only the Q bit (whose size field is fixed in the base),
// and for the long extend family the immh shift field the long forms imply
// (immh = esize/8) plus the Q bit for the .2 spellings.
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
if spec.qonly {
if two {
@@ -3037,6 +3039,14 @@ func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
}
return 0
}
if spec.form == a64NLTwoLong {
se, _, _ := arm64SimdNLArr(drive)
bits := uint32(se) << 19
if two {
bits |= 1 << 30
}
return bits
}
return a64ArrBits[a64ArrIndex(drive)]
}
@@ -3101,8 +3111,10 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
var pairErr error
if spec.form == a64NLThreeWide {
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
// the arrangement bits follow the wide side.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
// the size bits follow the narrow side (Vm) and the 128-bit flag
// follows the spelling: the plain form keeps Q clear, the .2
// form sets it.
drive, pairErr = vm.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
} else {
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
// wide one; the arrangement bits follow the narrow source.
@@ -3118,6 +3130,14 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
if spec.form == a64NLThreeWide {
// The size bits ride the narrow side's letter with Q forced by
// the spelling alone.
arrBits = a64ArrBits[a64ArrIndex(drive)] &^ (1 << 30)
if two {
arrBits |= 1 << 30
}
}
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongShift, a64NLThreeNarrowShift:
if len(ops) != 3 || !isImmOperand(ops[0]) {
@@ -4082,7 +4102,7 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
allowed := uint16(0x7f)
if strings.HasPrefix(mnem, "VFCM") {
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
allowed = fpSimdArrs
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
if err != nil {
@@ -4118,15 +4138,28 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
vs[i], arrs[i] = v, v.arr
}
// The bare three-register spellings of VADD/VSUB are the scalar D forms
// (the toolchain's ADD/SUB scalar rows), not the 8B vector rows.
if (mnem == "VADD" || mnem == "VSUB") &&
arrs[0] == "" && arrs[1] == "" && arrs[2] == "" {
base := uint32(0x5ee08400) // VADD scalar
if mnem == "VSUB" {
base = 0x7ee08400
}
return a64wordLE(base | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
}
arr, err := arm64SimdArrs(mnem, arrs, spec.arrs)
if err != nil {
return nil, err
}
arrBits := a64ArrBits[arr]
if spec.fixed {
if spec.fp {
// The FP rows carry a one-bit size field (S=0, D=1) instead of the
// integer size, and no Q-only masking applies to them.
arrBits = a64FPArrBits[arr]
} else if spec.fixed {
arrBits = 0
}
if a64SimdQOnly[mnem] {
} else if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
@@ -4175,7 +4208,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
return nil, err
}
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
if spec.fp {
// The FP rows carry the one-bit FP size field instead of the
// integer size, and no Q-only masking applies to them.
arrBits = a64FPArrBits[arr]
} else if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
@@ -4402,12 +4439,30 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
if dst.hasIdx {
// Element to element.
// Element to element: the toolchain requires the two element letters
// to match (asm7.go case 92), packs the destination index into imm5
// and the source index, in units of the element size, into imm4.
if src.arr != dst.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s elements", mnem, src.arr, dst.arr)
}
df, ok := a64ElemField(dst.arr, dst.idx)
if !ok {
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
return a64wordLE(0x6e000400 | df<<16 | sf>>1<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
var imm4 uint32
switch src.arr {
case "B":
imm4 = uint32(src.idx)
case "H":
imm4 = uint32(src.idx) << 1
case "S":
imm4 = uint32(src.idx) << 2
case "D":
imm4 = uint32(src.idx) << 3
default:
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
return a64wordLE(0x6e000400 | df<<16 | imm4&0xf<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
}
if dstGP {
// Element to a general register: UMOV, with the D form setting bit
+138 -122
View File
@@ -1010,13 +1010,16 @@ func init() {
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
// for instructions that exist at a single arrangement and carry that
// arrangement's bits inside the base already, the fixed flag.
// the set of arrangements it accepts as a bitmask over the a64Arr index,
// the fixed flag for instructions that exist at a single arrangement and
// carry that arrangement's bits inside the base already, and the fp flag for
// the FP rows, whose size field is the single FP bit (a64FPArrBits) instead
// of the integer size.
type a64SimdVSpec struct {
base uint32
arrs uint16
fixed bool
fp bool
}
// a64Arr names the vector arrangements the encoders deal with, indexed by
@@ -1063,12 +1066,25 @@ func a64ElemLetter(s string) bool {
return false
}
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
// accept: H, S and D widths for the pairwise data-processing, H and S for
// the across-vector reductions.
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
// fpSimdArrs bounds the arrangements the FP SIMD forms accept: S and D only,
// the toolchain rejecting the half-width spellings outright ("invalid
// arrangement"). fpAcrossArrs bounds the across-vector reductions, which do
// take the half width.
var fpSimdArrs = uint16(1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
// a64FPArrBits carries the bits an arrangement contributes to the FP SIMD
// words: the FP size field is a single bit at bit 22 (0 for the S widths, 1
// for the D widths; the toolchain carries no half-width FP rows) and the
// 128-bit flag sits at bit 30. Word-verified against go tool asm.
var a64FPArrBits = [a64ArrCount]uint32{
a64Arr2S: 0,
a64Arr4S: 1 << 30,
a64Arr2D: 1<<30 | 1<<22,
a64Arr4H: 0,
a64Arr8H: 1 << 30,
}
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
// flag alone, without the size bits: the FP converts, the FP round-to-integral
// and pairwise compares among them. Word-verified against go tool asm.
@@ -1099,81 +1115,81 @@ var a64ArrBits = [a64ArrCount]uint32{
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VCMGE": {0x0e203c00, 0x7f, false},
"VCMGT": {0x0e203400, 0x7f, false},
"VCMHI": {0x2e203400, 0x7f, false},
"VCMHS": {0x2e203c00, 0x7f, false},
// FP compares take H, S and D arrangements only (the toolchain rejects
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VADD": {0x0e208400, 0x7f, false, false},
"VSUB": {0x2e208400, 0x7f, false, false},
"VMUL": {0x0e209c00, 0x3f, false, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false, false},
"VORR": {0x0ea01c00, 0x03, false, false},
"VADDP": {0x0e20bc00, 0x7f, false, false},
"VZIP1": {0x0e003800, 0x7f, false, false},
"VZIP2": {0x0e007800, 0x7f, false, false},
"VCMEQ": {0x2e208c00, 0x7f, false, false},
"VCMGE": {0x0e203c00, 0x7f, false, false},
"VCMGT": {0x0e203400, 0x7f, false, false},
"VCMHI": {0x2e203400, 0x7f, false, false},
"VCMHS": {0x2e203c00, 0x7f, false, false},
// FP compares take S and D arrangements only (the toolchain rejects the
// byte and half forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, fpSimdArrs, false, true},
"VFCMGE": {0x2e20e400, fpSimdArrs, false, true},
"VFCMGT": {0x2ea0e400, fpSimdArrs, false, true},
// FP arithmetic shares the same arrangement restriction.
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFADD": {0x0e20d400, fpSimdArrs, false, true},
"VFSUB": {0x0ea0d400, fpSimdArrs, false, true},
"VFMUL": {0x2e20dc00, fpSimdArrs, false, true},
"VFDIV": {0x2e20fc00, fpSimdArrs, false, true},
"VFMAX": {0x0e20f400, fpSimdArrs, false, true},
"VFMIN": {0x0ea0f400, fpSimdArrs, false, true},
"VFMAXNM": {0x0e20c400, fpSimdArrs, false, true},
"VFMINNM": {0x0ea0c400, fpSimdArrs, false, true},
"VFMLA": {0x0e20cc00, fpSimdArrs, false, true},
"VFMLS": {0x0ea0cc00, fpSimdArrs, false, true},
// Saturating, halving, polynomial and pairwise arithmetic, the logical
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
// tool asm.
"VBIC": {0x0e601c00, 0x7f, false},
"VBIF": {0x2ee01c00, 0x7f, false},
"VBIT": {0x6ea01c00, 0x7f, false},
"VBSL": {0x6e601c00, 0x7f, false},
"VCMTST": {0x0e208c00, 0x7f, false},
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VMLA": {0x4ea09400, 0x7f, false},
"VMLS": {0x6ea09400, 0x7f, false},
"VORN": {0x4ee01c00, 0x7f, false},
"VSHADD": {0x4ea00400, 0x7f, false},
"VSRHADD": {0x4ea01400, 0x7f, false},
"VUHADD": {0x6ea00400, 0x7f, false},
"VURHADD": {0x6ea01400, 0x7f, false},
"VSMAX": {0x4ea06400, 0x7f, false},
"VSMIN": {0x4ea06c00, 0x7f, false},
"VSMAXP": {0x4ea0a400, 0x7f, false},
"VSMINP": {0x4ea0ac00, 0x7f, false},
"VUMAX": {0x2e206400, 0x7f, false},
"VUMIN": {0x2e206c00, 0x7f, false},
"VUMAXP": {0x6ea0a400, 0x7f, false},
"VUMINP": {0x6ea0ac00, 0x7f, false},
"VSQADD": {0x4ea00c00, 0x7f, false},
"VUQADD": {0x6ea00c00, 0x7f, false},
"VSQSUB": {0x4ea02c00, 0x7f, false},
"VUQSUB": {0x6ea02c00, 0x7f, false},
"VSSHL": {0x4ee04400, 0x7f, false},
"VUSHL": {0x6ee04400, 0x7f, false},
"VUZP1": {0x0e001800, 0x7f, false},
"VUZP2": {0x4ec05800, 0x7f, false},
"VTRN1": {0x4ec02800, 0x7f, false},
"VTRN2": {0x4ec06800, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
"VBIC": {0x0e601c00, 0x7f, false, false},
"VBIF": {0x2ee01c00, 0x7f, false, false},
"VBIT": {0x6ea01c00, 0x7f, false, false},
"VBSL": {0x6e601c00, 0x7f, false, false},
"VCMTST": {0x0e208c00, 0x7f, false, false},
"VFADDP": {0x2e20d400, fpSimdArrs, false, true},
"VFMAXP": {0x2e20f400, fpSimdArrs, false, true},
"VFMINP": {0x6ea0f400, fpSimdArrs, false, true},
"VFMAXNMP": {0x2e20c400, fpSimdArrs, false, true},
"VFMINNMP": {0x6ea0c400, fpSimdArrs, false, true},
"VMLA": {0x4ea09400, 0x7f, false, false},
"VMLS": {0x6ea09400, 0x7f, false, false},
"VORN": {0x4ee01c00, 0x7f, false, false},
"VSHADD": {0x4ea00400, 0x7f, false, false},
"VSRHADD": {0x4ea01400, 0x7f, false, false},
"VUHADD": {0x6ea00400, 0x7f, false, false},
"VURHADD": {0x6ea01400, 0x7f, false, false},
"VSMAX": {0x4ea06400, 0x7f, false, false},
"VSMIN": {0x4ea06c00, 0x7f, false, false},
"VSMAXP": {0x4ea0a400, 0x7f, false, false},
"VSMINP": {0x4ea0ac00, 0x7f, false, false},
"VUMAX": {0x2e206400, 0x7f, false, false},
"VUMIN": {0x2e206c00, 0x7f, false, false},
"VUMAXP": {0x6ea0a400, 0x7f, false, false},
"VUMINP": {0x6ea0ac00, 0x7f, false, false},
"VSQADD": {0x4ea00c00, 0x7f, false, false},
"VUQADD": {0x6ea00c00, 0x7f, false, false},
"VSQSUB": {0x4ea02c00, 0x7f, false, false},
"VUQSUB": {0x6ea02c00, 0x7f, false, false},
"VSSHL": {0x0e204400, 0x7f, false, false},
"VUSHL": {0x2e204400, 0x7f, false, false},
"VUZP1": {0x0e001800, 0x7f, false, false},
"VUZP2": {0x4ec05800, 0x7f, false, false},
"VTRN1": {0x4ec02800, 0x7f, false, false},
"VTRN2": {0x4ec06800, 0x7f, false, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true, false}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false, false},
// Saturating shifts, register forms (the immediate spellings route to
// a64FShiftImm).
"VSQSHL": {0x0e204c00, 0x7f, false},
"VUQSHL": {0x2e204c00, 0x7f, false},
"VSQSHL": {0x0e204c00, 0x7f, false, false},
"VUQSHL": {0x2e204c00, 0x7f, false, false},
}
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
@@ -1212,18 +1228,18 @@ var a64SimdNLTable = map[string]a64SimdNLSpec{
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
"VXTN": {0x0e212800, a64NLTwoNarrow, false},
"VXTN2": {0x0e212800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e214800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e214800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e212800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e212800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e214800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e214800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e616800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e616800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e617800, a64NLTwoLong, true},
"VFCVTL2": {0x0e617800, a64NLTwoLong, true},
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
@@ -1267,43 +1283,43 @@ var a64SimdVZero = map[string]uint32{
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false, false},
"VREV64": {0x0e200800, 0x3f, false, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
"VUADDLV": {0x2e303800, 0x3f, false, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
// Two-register data-processing across one arrangement.
"VABS": {0x0e20b800, 0x7f, false},
"VNEG": {0x2e20b800, 0x7f, false},
"VCLS": {0x0e204800, 0x7f, false},
"VCLZ": {0x2e204800, 0x7f, false},
"VCNT": {0x0e205800, 0x7f, false},
"VNOT": {0x2e205800, 0x7f, false},
"VSQABS": {0x0e207800, 0x7f, false},
"VSQNEG": {0x2e207800, 0x7f, false},
"VRBIT": {0x6e605800, 0x7f, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
"VFABS": {0x0ea0f800, fpSimdArrs, false},
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
"VFRINTN": {0x0e218800, fpSimdArrs, false},
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
"VFRINTM": {0x0e219800, fpSimdArrs, false},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
"VABS": {0x0e20b800, 0x7f, false, false},
"VNEG": {0x2e20b800, 0x7f, false, false},
"VCLS": {0x0e204800, 0x7f, false, false},
"VCLZ": {0x2e204800, 0x7f, false, false},
"VCNT": {0x0e205800, 0x7f, false, false},
"VNOT": {0x2e205800, 0x7f, false, false},
"VSQABS": {0x0e207800, 0x7f, false, false},
"VSQNEG": {0x2e207800, 0x7f, false, false},
"VRBIT": {0x2e605800, 0x7f, false, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false, true},
"VUCVTF": {0x6e21d800, fpSimdArrs, false, true},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false, true},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false, true},
"VFABS": {0x0ea0f800, fpSimdArrs, false, true},
"VFNEG": {0x2ea0f800, fpSimdArrs, false, true},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false, true},
"VFRINTN": {0x0e218800, fpSimdArrs, false, true},
"VFRINTP": {0x0ea18800, fpSimdArrs, false, true},
"VFRINTM": {0x0e219800, fpSimdArrs, false, true},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false, true},
// Across-vector reductions: the operand arrangement rides as usual and
// the destination stays a bare V register.
"VADDV": {0x0e31b800, 0x3f, false},
"VSMAXV": {0x0e30a800, 0x3f, false},
"VSMINV": {0x0e31a800, 0x3f, false},
"VUMAXV": {0x2e30a800, 0x3f, false},
"VUMINV": {0x2e31a800, 0x3f, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
"VADDV": {0x0e31b800, 0x3f, false, false},
"VSMAXV": {0x0e30a800, 0x3f, false, false},
"VSMINV": {0x0e31a800, 0x3f, false, false},
"VUMAXV": {0x2e30a800, 0x3f, false, false},
"VUMINV": {0x2e31a800, 0x3f, false, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
+84
View File
@@ -1894,3 +1894,87 @@ func TestArm64RegOffsetRejections(t *testing.T) {
}
}
}
// TestArm64SimdArrangementBits pins the arrangement bits the first SIMD pass
// got wrong, word-verified against `go tool asm` (Go 1.27, arm64): the FP
// one-bit size field (S=0, D=1 at bit 22), the SSHL/USHL size bits, the long
// extends' immh field, the narrow family's fixed bit 16, the UADDW size bits
// off the narrow side with Q from the spelling, the scalar D forms of the
// bare VADD/VSUB spellings and the INS lane packing.
func TestArm64SimdArrangementBits(t *testing.T) {
got := arm64Words(t,
"\tVFADD V0.S4, V0.S4, V1.S4\n"+
"\tVFADD V0.D2, V0.D2, V1.D2\n"+
"\tVFABS V0.D2, V1.D2\n"+
"\tVSCVTF V1.D2, V2.D2\n"+
"\tVSSHL V1.B8, V2.B8, V3.B8\n"+
"\tVSSHL V1.S4, V2.S4, V3.S4\n"+
"\tVUSHL V1.H4, V2.H4, V3.H4\n"+
"\tVRBIT V24.B8, V24.B8\n"+
"\tVUXTL V30.B8, V30.H8\n"+
"\tVUXTL V29.S2, V2.D2\n"+
"\tVUXTL2 V30.H8, V30.S4\n"+
"\tVXTN V1.H8, V2.B8\n"+
"\tVFCVTN V1.D2, V2.S2\n"+
"\tVFCVTL V1.S2, V2.D2\n"+
"\tVUADDW V13.H4, V10.S4, V11.S4\n"+
"\tVUADDW2 V13.H8, V20.S4, V30.S4\n"+
"\tVADD V1, V2, V3\n"+
"\tVSUB V12, V20, V30\n"+
"\tVMOV V12.S[2], V12.S[3]\n"+
"\tVMOV V12.H[3], V12.H[5]\n")
want := []uint32{
0x4e20d401, // VFADD V0.4S: FP size field clear for S
0x4e60d401, // VFADD V0.2D: FP size bit 22, not bit 23
0x4ee0f801, // VFABS V1.2D: one-bit FP size
0x4e61d822, // VSCVTF V2.2D: bit 22, the Q-only mask must not strip it
0x0e214443, // VSSHL V3.8B: base without the pre-set size and Q bits
0x4ea14443, // VSSHL V3.4S: integer size bits from the arrangement
0x2e614443, // VUSHL V3.4H: U bit plus the H size
0x2e605b18, // VRBIT V24.8B: no Q bit in the base
0x2f08a7de, // VUXTL: immh = 1 at bit 19 for the byte extend
0x2f20a7a2, // VUXTL: immh = 4 at bit 21 for the word extend
0x6f10a7de, // VUXTL2: immh = 2 plus the 128-bit flag
0x0e212822, // VXTN: fixed bit 16 in the base
0x0e616822, // VFCVTN: bits 16 and 22, Q rides the spelling
0x0e617822, // VFCVTL: bit 22, Q rides the spelling
0x2e6d114b, // VUADDW: size bits off the narrow side, Q clear
0x6e6d129e, // VUADDW2: size off the narrow side, Q from the spelling
0x5ee18443, // VADD scalar D form for the bare spelling
0x7eec869e, // VSUB scalar D form for the bare spelling
0x6e1c458c, // INS: imm4 = 2<<2 for the word source lane 2
0x6e16358c, // INS: imm4 = 3<<1 for the halfword source lane 3
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SimdArrangementRejections pins the arrangements the toolchain
// refuses on the FP SIMD rows and the element-to-element moves: the
// half-width FP spellings, the Q1 spelling on the integer shifts, and the
// mixed element letters of INS.
func TestArm64SimdArrangementRejections(t *testing.T) {
for _, src := range []string{
"\tVFADD\tV1.H4, V2.H4, V3.H4\n",
"\tVFADD\tV1.H8, V2.H8, V3.H8\n",
"\tVFABS\tV1.H4, V2.H4\n",
"\tVSCVTF\tV1.H4, V2.H4\n",
"\tVSSHL\tV1.Q1, V2.Q1, V3.Q1\n",
"\tVMOV\tV12.S[0], V12.D[1]\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n")
if len(errs) > 0 {
continue // a parse rejection is a rejection
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("expected rejection for %q, got nil", strings.TrimSpace(src))
}
}
}
+118
View File
@@ -0,0 +1,118 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 SIMD arrangement bits: the FP one-bit
// size field (S=0, D=1 at bit 22), the immh field of the long extends, the
// narrow family's fixed bit 16, the SSHL/USHL size bits, the scalar D forms
// of the bare VADD/VSUB spellings and the INS lane packing. Every function
// is byte-compared against go tool asm.
#include "textflag.h"
// func fpthree()
TEXT ·fpthree(SB), NOSPLIT, $0-0
VFADD V0.S4, V0.S4, V1.S4
VFADD V0.D2, V0.D2, V1.D2
VFMUL V0.S4, V0.S4, V1.S4
VFMUL V0.D2, V0.D2, V1.D2
VFDIV V0.S4, V0.S4, V1.S4
VFDIV V0.D2, V0.D2, V1.D2
VFMLA V1.D2, V12.D2, V1.D2
VFMLA V1.S2, V12.S2, V1.S2
VFMLA V1.S4, V12.S4, V1.S4
VFMAX V3.S4, V2.S4, V1.S4
VFMAXNM V3.S4, V2.S4, V1.S4
VFADDP V3.D2, V2.D2, V1.D2
VFCMEQ V1.S4, V2.S4, V3.S4
VFCMGE V1.S4, V2.S4, V3.S4
RET
// func fpunary()
TEXT ·fpunary(SB), NOSPLIT, $0-0
VFABS V0.D2, V1.D2
VFNEG V0.D2, V1.D2
VFSQRT V0.D2, V1.D2
VFRINTN V0.D2, V1.D2
VFRINTP V0.D2, V1.D2
VFRINTM V0.D2, V1.D2
VFRINTZ V0.D2, V1.D2
VFCVTZS V1.D2, V2.D2
VFCVTZU V1.D2, V2.D2
VSCVTF V1.D2, V2.D2
VUCVTF V1.D2, V2.D2
VFABS V0.S4, V1.S4
VSCVTF V1.S4, V2.S4
VFNEG V0.S2, V1.S2
RET
// func shifts()
TEXT ·shifts(SB), NOSPLIT, $0-0
VSSHL V1.S4, V2.S4, V3.S4
VSSHL V1.S2, V2.S2, V3.S2
VSSHL V1.H4, V2.H4, V3.H4
VSSHL V1.H8, V2.H8, V3.H8
VSSHL V1.B8, V2.B8, V3.B8
VSSHL V1.B16, V2.B16, V3.B16
VUSHL V1.S4, V2.S4, V3.S4
VUSHL V1.S2, V2.S2, V3.S2
VUSHL V1.H4, V2.H4, V3.H4
VUSHL V1.H8, V2.H8, V3.H8
VUSHL V1.B8, V2.B8, V3.B8
VUSHL V1.B16, V2.B16, V3.B16
VRBIT V24.B8, V24.B8
RET
// func widen()
TEXT ·widen(SB), NOSPLIT, $0-0
VUXTL V30.B8, V30.H8
VUXTL V30.H4, V29.S4
VUXTL V29.S2, V2.D2
VUXTL2 V30.H8, V30.S4
VUXTL2 V29.S4, V2.D2
VUXTL2 V30.B16, V2.H8
VSXTL V1.B8, V2.H8
VSXTL V1.H4, V2.S4
VSXTL V1.S2, V2.D2
VSXTL2 V1.B16, V2.H8
VSXTL2 V1.H8, V2.S4
VSXTL2 V1.S4, V2.D2
VXTN V1.H8, V2.B8
VXTN V1.S4, V2.H4
VXTN V1.D2, V2.S2
VXTN2 V1.H8, V2.B16
VSQXTN V1.D2, V2.S2
VSQXTN2 V1.S4, V2.H8
VSQXTUN V1.H8, V2.B8
VUQXTN V1.S4, V2.H4
VUQXTN2 V1.D2, V2.S4
RET
// func uaddw()
TEXT ·uaddw(SB), NOSPLIT, $0-0
VUADDW V9.B8, V12.H8, V14.H8
VUADDW V13.H4, V10.S4, V11.S4
VUADDW V21.S2, V24.D2, V29.D2
VUADDW2 V9.B16, V12.H8, V14.H8
VUADDW2 V13.H8, V20.S4, V30.S4
VUADDW2 V21.S4, V24.D2, V29.D2
RET
// func bareadd()
TEXT ·bareadd(SB), NOSPLIT, $0-0
VADD V1, V2, V3
VADD V1, V3, V3
VSUB V12, V30, V30
VSUB V12, V20, V30
VADD V1.B8, V2.B8, V3.B8
VSUB V12.B8, V30.B8, V30.B8
RET
// func lanes()
TEXT ·lanes(SB), NOSPLIT, $0-0
VMOV V12.D[0], V12.D[1]
VMOV V10.S[0], V12.S[1]
VMOV V9.H[0], V12.H[1]
VMOV V12.B[0], V12.B[1]
VMOV V12.S[2], V12.S[3]
VMOV V12.H[3], V12.H[5]
RET
+1
View File
@@ -38,6 +38,7 @@ func TestGroundTruthARM64(t *testing.T) {
"../testdata/verify/qmov_arm64.s",
"../testdata/verify/splits_arm64.s",
"../testdata/verify/regoffset_arm64.s",
"../testdata/verify/simdarr_arm64.s",
"../testdata/verify/crypto_arm64.s",
"../testdata/verify/integer_arm64.s",
"../testdata/verify/simd_arm64.s",