feat(asm): encode the arm64 SIMD arrangement bits
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
458cdd2066
commit
9ef14bdb71
5 files changed
+407
-133
No files matched your search
+66
-11
@@ -3028,8 +3028,10 @@ func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
|
||||
}
|
||||
|
||||
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
|
||||
// instruction contributes: the driving arrangement's size and Q bits, or for
|
||||
// the FCVT family only the Q bit, whose size field is fixed in the base.
|
||||
// instruction contributes: the driving arrangement's size and Q bits, for
|
||||
// the FCVT family only the Q bit (whose size field is fixed in the base),
|
||||
// and for the long extend family the immh shift field the long forms imply
|
||||
// (immh = esize/8) plus the Q bit for the .2 spellings.
|
||||
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
|
||||
if spec.qonly {
|
||||
if two {
|
||||
@@ -3037,6 +3039,14 @@ func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
|
||||
}
|
||||
return 0
|
||||
}
|
||||
if spec.form == a64NLTwoLong {
|
||||
se, _, _ := arm64SimdNLArr(drive)
|
||||
bits := uint32(se) << 19
|
||||
if two {
|
||||
bits |= 1 << 30
|
||||
}
|
||||
return bits
|
||||
}
|
||||
return a64ArrBits[a64ArrIndex(drive)]
|
||||
}
|
||||
|
||||
@@ -3101,8 +3111,10 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
|
||||
var pairErr error
|
||||
if spec.form == a64NLThreeWide {
|
||||
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
|
||||
// the arrangement bits follow the wide side.
|
||||
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
|
||||
// the size bits follow the narrow side (Vm) and the 128-bit flag
|
||||
// follows the spelling: the plain form keeps Q clear, the .2
|
||||
// form sets it.
|
||||
drive, pairErr = vm.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
|
||||
} else {
|
||||
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
|
||||
// wide one; the arrangement bits follow the narrow source.
|
||||
@@ -3118,6 +3130,14 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
|
||||
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
|
||||
}
|
||||
arrBits := arm64SimdNLArrBits(spec, drive, two)
|
||||
if spec.form == a64NLThreeWide {
|
||||
// The size bits ride the narrow side's letter with Q forced by
|
||||
// the spelling alone.
|
||||
arrBits = a64ArrBits[a64ArrIndex(drive)] &^ (1 << 30)
|
||||
if two {
|
||||
arrBits |= 1 << 30
|
||||
}
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
case a64NLThreeLongShift, a64NLThreeNarrowShift:
|
||||
if len(ops) != 3 || !isImmOperand(ops[0]) {
|
||||
@@ -4082,7 +4102,7 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
}
|
||||
allowed := uint16(0x7f)
|
||||
if strings.HasPrefix(mnem, "VFCM") {
|
||||
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
|
||||
allowed = fpSimdArrs
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
|
||||
if err != nil {
|
||||
@@ -4118,15 +4138,28 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
}
|
||||
vs[i], arrs[i] = v, v.arr
|
||||
}
|
||||
// The bare three-register spellings of VADD/VSUB are the scalar D forms
|
||||
// (the toolchain's ADD/SUB scalar rows), not the 8B vector rows.
|
||||
if (mnem == "VADD" || mnem == "VSUB") &&
|
||||
arrs[0] == "" && arrs[1] == "" && arrs[2] == "" {
|
||||
base := uint32(0x5ee08400) // VADD scalar
|
||||
if mnem == "VSUB" {
|
||||
base = 0x7ee08400
|
||||
}
|
||||
return a64wordLE(base | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, arrs, spec.arrs)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
arrBits := a64ArrBits[arr]
|
||||
if spec.fixed {
|
||||
if spec.fp {
|
||||
// The FP rows carry a one-bit size field (S=0, D=1) instead of the
|
||||
// integer size, and no Q-only masking applies to them.
|
||||
arrBits = a64FPArrBits[arr]
|
||||
} else if spec.fixed {
|
||||
arrBits = 0
|
||||
}
|
||||
if a64SimdQOnly[mnem] {
|
||||
} else if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
||||
@@ -4175,7 +4208,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
||||
return nil, err
|
||||
}
|
||||
arrBits := a64ArrBits[arr]
|
||||
if a64SimdQOnly[mnem] {
|
||||
if spec.fp {
|
||||
// The FP rows carry the one-bit FP size field instead of the
|
||||
// integer size, and no Q-only masking applies to them.
|
||||
arrBits = a64FPArrBits[arr]
|
||||
} else if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
@@ -4402,12 +4439,30 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
||||
}
|
||||
if dst.hasIdx {
|
||||
// Element to element.
|
||||
// Element to element: the toolchain requires the two element letters
|
||||
// to match (asm7.go case 92), packs the destination index into imm5
|
||||
// and the source index, in units of the element size, into imm4.
|
||||
if src.arr != dst.arr {
|
||||
return nil, fmt.Errorf("%s: operand mismatch: %s and %s elements", mnem, src.arr, dst.arr)
|
||||
}
|
||||
df, ok := a64ElemField(dst.arr, dst.idx)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
||||
}
|
||||
return a64wordLE(0x6e000400 | df<<16 | sf>>1<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
|
||||
var imm4 uint32
|
||||
switch src.arr {
|
||||
case "B":
|
||||
imm4 = uint32(src.idx)
|
||||
case "H":
|
||||
imm4 = uint32(src.idx) << 1
|
||||
case "S":
|
||||
imm4 = uint32(src.idx) << 2
|
||||
case "D":
|
||||
imm4 = uint32(src.idx) << 3
|
||||
default:
|
||||
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
||||
}
|
||||
return a64wordLE(0x6e000400 | df<<16 | imm4&0xf<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
|
||||
}
|
||||
if dstGP {
|
||||
// Element to a general register: UMOV, with the D form setting bit
|
||||
|
||||
+138
-122
@@ -1010,13 +1010,16 @@ func init() {
|
||||
}
|
||||
|
||||
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
|
||||
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
|
||||
// for instructions that exist at a single arrangement and carry that
|
||||
// arrangement's bits inside the base already, the fixed flag.
|
||||
// the set of arrangements it accepts as a bitmask over the a64Arr index,
|
||||
// the fixed flag for instructions that exist at a single arrangement and
|
||||
// carry that arrangement's bits inside the base already, and the fp flag for
|
||||
// the FP rows, whose size field is the single FP bit (a64FPArrBits) instead
|
||||
// of the integer size.
|
||||
type a64SimdVSpec struct {
|
||||
base uint32
|
||||
arrs uint16
|
||||
fixed bool
|
||||
fp bool
|
||||
}
|
||||
|
||||
// a64Arr names the vector arrangements the encoders deal with, indexed by
|
||||
@@ -1063,12 +1066,25 @@ func a64ElemLetter(s string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
|
||||
// accept: H, S and D widths for the pairwise data-processing, H and S for
|
||||
// the across-vector reductions.
|
||||
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
|
||||
// fpSimdArrs bounds the arrangements the FP SIMD forms accept: S and D only,
|
||||
// the toolchain rejecting the half-width spellings outright ("invalid
|
||||
// arrangement"). fpAcrossArrs bounds the across-vector reductions, which do
|
||||
// take the half width.
|
||||
var fpSimdArrs = uint16(1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
|
||||
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
|
||||
|
||||
// a64FPArrBits carries the bits an arrangement contributes to the FP SIMD
|
||||
// words: the FP size field is a single bit at bit 22 (0 for the S widths, 1
|
||||
// for the D widths; the toolchain carries no half-width FP rows) and the
|
||||
// 128-bit flag sits at bit 30. Word-verified against go tool asm.
|
||||
var a64FPArrBits = [a64ArrCount]uint32{
|
||||
a64Arr2S: 0,
|
||||
a64Arr4S: 1 << 30,
|
||||
a64Arr2D: 1<<30 | 1<<22,
|
||||
a64Arr4H: 0,
|
||||
a64Arr8H: 1 << 30,
|
||||
}
|
||||
|
||||
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
|
||||
// flag alone, without the size bits: the FP converts, the FP round-to-integral
|
||||
// and pairwise compares among them. Word-verified against go tool asm.
|
||||
@@ -1099,81 +1115,81 @@ var a64ArrBits = [a64ArrCount]uint32{
|
||||
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
|
||||
// word and arrangement bit was read off go tool asm.
|
||||
var a64SimdVTable = map[string]a64SimdVSpec{
|
||||
"VADD": {0x0e208400, 0x7f, false},
|
||||
"VSUB": {0x2e208400, 0x7f, false},
|
||||
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
|
||||
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
|
||||
"VEOR": {0x2e201c00, 0x03, false},
|
||||
"VORR": {0x0ea01c00, 0x03, false},
|
||||
"VADDP": {0x0e20bc00, 0x7f, false},
|
||||
"VZIP1": {0x0e003800, 0x7f, false},
|
||||
"VZIP2": {0x0e007800, 0x7f, false},
|
||||
"VCMEQ": {0x2e208c00, 0x7f, false},
|
||||
"VCMGE": {0x0e203c00, 0x7f, false},
|
||||
"VCMGT": {0x0e203400, 0x7f, false},
|
||||
"VCMHI": {0x2e203400, 0x7f, false},
|
||||
"VCMHS": {0x2e203c00, 0x7f, false},
|
||||
// FP compares take H, S and D arrangements only (the toolchain rejects
|
||||
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
|
||||
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VADD": {0x0e208400, 0x7f, false, false},
|
||||
"VSUB": {0x2e208400, 0x7f, false, false},
|
||||
"VMUL": {0x0e209c00, 0x3f, false, false}, // no 2D: integer multiply stops at 4S
|
||||
"VAND": {0x0e201c00, 0x03, false, false}, // logical ops accept 8B and 16B only
|
||||
"VEOR": {0x2e201c00, 0x03, false, false},
|
||||
"VORR": {0x0ea01c00, 0x03, false, false},
|
||||
"VADDP": {0x0e20bc00, 0x7f, false, false},
|
||||
"VZIP1": {0x0e003800, 0x7f, false, false},
|
||||
"VZIP2": {0x0e007800, 0x7f, false, false},
|
||||
"VCMEQ": {0x2e208c00, 0x7f, false, false},
|
||||
"VCMGE": {0x0e203c00, 0x7f, false, false},
|
||||
"VCMGT": {0x0e203400, 0x7f, false, false},
|
||||
"VCMHI": {0x2e203400, 0x7f, false, false},
|
||||
"VCMHS": {0x2e203c00, 0x7f, false, false},
|
||||
// FP compares take S and D arrangements only (the toolchain rejects the
|
||||
// byte and half forms), and VFCMLE/VFCMLT have no register form at all.
|
||||
"VFCMEQ": {0x0e20e400, fpSimdArrs, false, true},
|
||||
"VFCMGE": {0x2e20e400, fpSimdArrs, false, true},
|
||||
"VFCMGT": {0x2ea0e400, fpSimdArrs, false, true},
|
||||
// FP arithmetic shares the same arrangement restriction.
|
||||
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFADD": {0x0e20d400, fpSimdArrs, false, true},
|
||||
"VFSUB": {0x0ea0d400, fpSimdArrs, false, true},
|
||||
"VFMUL": {0x2e20dc00, fpSimdArrs, false, true},
|
||||
"VFDIV": {0x2e20fc00, fpSimdArrs, false, true},
|
||||
"VFMAX": {0x0e20f400, fpSimdArrs, false, true},
|
||||
"VFMIN": {0x0ea0f400, fpSimdArrs, false, true},
|
||||
"VFMAXNM": {0x0e20c400, fpSimdArrs, false, true},
|
||||
"VFMINNM": {0x0ea0c400, fpSimdArrs, false, true},
|
||||
"VFMLA": {0x0e20cc00, fpSimdArrs, false, true},
|
||||
"VFMLS": {0x0ea0cc00, fpSimdArrs, false, true},
|
||||
// Saturating, halving, polynomial and pairwise arithmetic, the logical
|
||||
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
|
||||
// tool asm.
|
||||
"VBIC": {0x0e601c00, 0x7f, false},
|
||||
"VBIF": {0x2ee01c00, 0x7f, false},
|
||||
"VBIT": {0x6ea01c00, 0x7f, false},
|
||||
"VBSL": {0x6e601c00, 0x7f, false},
|
||||
"VCMTST": {0x0e208c00, 0x7f, false},
|
||||
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VMLA": {0x4ea09400, 0x7f, false},
|
||||
"VMLS": {0x6ea09400, 0x7f, false},
|
||||
"VORN": {0x4ee01c00, 0x7f, false},
|
||||
"VSHADD": {0x4ea00400, 0x7f, false},
|
||||
"VSRHADD": {0x4ea01400, 0x7f, false},
|
||||
"VUHADD": {0x6ea00400, 0x7f, false},
|
||||
"VURHADD": {0x6ea01400, 0x7f, false},
|
||||
"VSMAX": {0x4ea06400, 0x7f, false},
|
||||
"VSMIN": {0x4ea06c00, 0x7f, false},
|
||||
"VSMAXP": {0x4ea0a400, 0x7f, false},
|
||||
"VSMINP": {0x4ea0ac00, 0x7f, false},
|
||||
"VUMAX": {0x2e206400, 0x7f, false},
|
||||
"VUMIN": {0x2e206c00, 0x7f, false},
|
||||
"VUMAXP": {0x6ea0a400, 0x7f, false},
|
||||
"VUMINP": {0x6ea0ac00, 0x7f, false},
|
||||
"VSQADD": {0x4ea00c00, 0x7f, false},
|
||||
"VUQADD": {0x6ea00c00, 0x7f, false},
|
||||
"VSQSUB": {0x4ea02c00, 0x7f, false},
|
||||
"VUQSUB": {0x6ea02c00, 0x7f, false},
|
||||
"VSSHL": {0x4ee04400, 0x7f, false},
|
||||
"VUSHL": {0x6ee04400, 0x7f, false},
|
||||
"VUZP1": {0x0e001800, 0x7f, false},
|
||||
"VUZP2": {0x4ec05800, 0x7f, false},
|
||||
"VTRN1": {0x4ec02800, 0x7f, false},
|
||||
"VTRN2": {0x4ec06800, 0x7f, false},
|
||||
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||
"VBIC": {0x0e601c00, 0x7f, false, false},
|
||||
"VBIF": {0x2ee01c00, 0x7f, false, false},
|
||||
"VBIT": {0x6ea01c00, 0x7f, false, false},
|
||||
"VBSL": {0x6e601c00, 0x7f, false, false},
|
||||
"VCMTST": {0x0e208c00, 0x7f, false, false},
|
||||
"VFADDP": {0x2e20d400, fpSimdArrs, false, true},
|
||||
"VFMAXP": {0x2e20f400, fpSimdArrs, false, true},
|
||||
"VFMINP": {0x6ea0f400, fpSimdArrs, false, true},
|
||||
"VFMAXNMP": {0x2e20c400, fpSimdArrs, false, true},
|
||||
"VFMINNMP": {0x6ea0c400, fpSimdArrs, false, true},
|
||||
"VMLA": {0x4ea09400, 0x7f, false, false},
|
||||
"VMLS": {0x6ea09400, 0x7f, false, false},
|
||||
"VORN": {0x4ee01c00, 0x7f, false, false},
|
||||
"VSHADD": {0x4ea00400, 0x7f, false, false},
|
||||
"VSRHADD": {0x4ea01400, 0x7f, false, false},
|
||||
"VUHADD": {0x6ea00400, 0x7f, false, false},
|
||||
"VURHADD": {0x6ea01400, 0x7f, false, false},
|
||||
"VSMAX": {0x4ea06400, 0x7f, false, false},
|
||||
"VSMIN": {0x4ea06c00, 0x7f, false, false},
|
||||
"VSMAXP": {0x4ea0a400, 0x7f, false, false},
|
||||
"VSMINP": {0x4ea0ac00, 0x7f, false, false},
|
||||
"VUMAX": {0x2e206400, 0x7f, false, false},
|
||||
"VUMIN": {0x2e206c00, 0x7f, false, false},
|
||||
"VUMAXP": {0x6ea0a400, 0x7f, false, false},
|
||||
"VUMINP": {0x6ea0ac00, 0x7f, false, false},
|
||||
"VSQADD": {0x4ea00c00, 0x7f, false, false},
|
||||
"VUQADD": {0x6ea00c00, 0x7f, false, false},
|
||||
"VSQSUB": {0x4ea02c00, 0x7f, false, false},
|
||||
"VUQSUB": {0x6ea02c00, 0x7f, false, false},
|
||||
"VSSHL": {0x0e204400, 0x7f, false, false},
|
||||
"VUSHL": {0x2e204400, 0x7f, false, false},
|
||||
"VUZP1": {0x0e001800, 0x7f, false, false},
|
||||
"VUZP2": {0x4ec05800, 0x7f, false, false},
|
||||
"VTRN1": {0x4ec02800, 0x7f, false, false},
|
||||
"VTRN2": {0x4ec06800, 0x7f, false, false},
|
||||
"VRAX1": {0xce608c00, 1 << a64Arr2D, true, false}, // SHA3 group, D2 only
|
||||
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false, false},
|
||||
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false, false},
|
||||
// Saturating shifts, register forms (the immediate spellings route to
|
||||
// a64FShiftImm).
|
||||
"VSQSHL": {0x0e204c00, 0x7f, false},
|
||||
"VUQSHL": {0x2e204c00, 0x7f, false},
|
||||
"VSQSHL": {0x0e204c00, 0x7f, false, false},
|
||||
"VUQSHL": {0x2e204c00, 0x7f, false, false},
|
||||
}
|
||||
|
||||
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
|
||||
@@ -1212,18 +1228,18 @@ var a64SimdNLTable = map[string]a64SimdNLSpec{
|
||||
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
|
||||
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
|
||||
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
|
||||
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
|
||||
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
|
||||
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
|
||||
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
|
||||
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
|
||||
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
|
||||
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
|
||||
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
|
||||
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
|
||||
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
|
||||
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
|
||||
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
|
||||
"VXTN": {0x0e212800, a64NLTwoNarrow, false},
|
||||
"VXTN2": {0x0e212800, a64NLTwoNarrow, false},
|
||||
"VSQXTN": {0x0e214800, a64NLTwoNarrow, false},
|
||||
"VSQXTN2": {0x0e214800, a64NLTwoNarrow, false},
|
||||
"VSQXTUN": {0x2e212800, a64NLTwoNarrow, false},
|
||||
"VSQXTUN2": {0x2e212800, a64NLTwoNarrow, false},
|
||||
"VUQXTN": {0x2e214800, a64NLTwoNarrow, false},
|
||||
"VUQXTN2": {0x2e214800, a64NLTwoNarrow, false},
|
||||
"VFCVTN": {0x0e616800, a64NLTwoNarrow, true},
|
||||
"VFCVTN2": {0x0e616800, a64NLTwoNarrow, true},
|
||||
"VFCVTL": {0x0e617800, a64NLTwoLong, true},
|
||||
"VFCVTL2": {0x0e617800, a64NLTwoLong, true},
|
||||
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
|
||||
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
|
||||
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
|
||||
@@ -1267,43 +1283,43 @@ var a64SimdVZero = map[string]uint32{
|
||||
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
|
||||
// the register pair spelling ORR Vd, Vn, Vm.
|
||||
var a64SimdV2Table = map[string]a64SimdVSpec{
|
||||
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
||||
"VREV64": {0x0e200800, 0x3f, false},
|
||||
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||
"VUADDLV": {0x2e303800, 0x3f, false},
|
||||
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false, false},
|
||||
"VREV64": {0x0e200800, 0x3f, false, false},
|
||||
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
|
||||
"VUADDLV": {0x2e303800, 0x3f, false, false},
|
||||
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
|
||||
// Two-register data-processing across one arrangement.
|
||||
"VABS": {0x0e20b800, 0x7f, false},
|
||||
"VNEG": {0x2e20b800, 0x7f, false},
|
||||
"VCLS": {0x0e204800, 0x7f, false},
|
||||
"VCLZ": {0x2e204800, 0x7f, false},
|
||||
"VCNT": {0x0e205800, 0x7f, false},
|
||||
"VNOT": {0x2e205800, 0x7f, false},
|
||||
"VSQABS": {0x0e207800, 0x7f, false},
|
||||
"VSQNEG": {0x2e207800, 0x7f, false},
|
||||
"VRBIT": {0x6e605800, 0x7f, false},
|
||||
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
|
||||
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
|
||||
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
|
||||
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
|
||||
"VFABS": {0x0ea0f800, fpSimdArrs, false},
|
||||
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
|
||||
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
|
||||
"VFRINTN": {0x0e218800, fpSimdArrs, false},
|
||||
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
|
||||
"VFRINTM": {0x0e219800, fpSimdArrs, false},
|
||||
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
|
||||
"VABS": {0x0e20b800, 0x7f, false, false},
|
||||
"VNEG": {0x2e20b800, 0x7f, false, false},
|
||||
"VCLS": {0x0e204800, 0x7f, false, false},
|
||||
"VCLZ": {0x2e204800, 0x7f, false, false},
|
||||
"VCNT": {0x0e205800, 0x7f, false, false},
|
||||
"VNOT": {0x2e205800, 0x7f, false, false},
|
||||
"VSQABS": {0x0e207800, 0x7f, false, false},
|
||||
"VSQNEG": {0x2e207800, 0x7f, false, false},
|
||||
"VRBIT": {0x2e605800, 0x7f, false, false},
|
||||
"VSCVTF": {0x4e21d800, fpSimdArrs, false, true},
|
||||
"VUCVTF": {0x6e21d800, fpSimdArrs, false, true},
|
||||
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false, true},
|
||||
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false, true},
|
||||
"VFABS": {0x0ea0f800, fpSimdArrs, false, true},
|
||||
"VFNEG": {0x2ea0f800, fpSimdArrs, false, true},
|
||||
"VFSQRT": {0x2ea1f800, fpSimdArrs, false, true},
|
||||
"VFRINTN": {0x0e218800, fpSimdArrs, false, true},
|
||||
"VFRINTP": {0x0ea18800, fpSimdArrs, false, true},
|
||||
"VFRINTM": {0x0e219800, fpSimdArrs, false, true},
|
||||
"VFRINTZ": {0x0ea19800, fpSimdArrs, false, true},
|
||||
// Across-vector reductions: the operand arrangement rides as usual and
|
||||
// the destination stays a bare V register.
|
||||
"VADDV": {0x0e31b800, 0x3f, false},
|
||||
"VSMAXV": {0x0e30a800, 0x3f, false},
|
||||
"VSMINV": {0x0e31a800, 0x3f, false},
|
||||
"VUMAXV": {0x2e30a800, 0x3f, false},
|
||||
"VUMINV": {0x2e31a800, 0x3f, false},
|
||||
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
|
||||
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
|
||||
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
|
||||
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
|
||||
"VADDV": {0x0e31b800, 0x3f, false, false},
|
||||
"VSMAXV": {0x0e30a800, 0x3f, false, false},
|
||||
"VSMINV": {0x0e31a800, 0x3f, false, false},
|
||||
"VUMAXV": {0x2e30a800, 0x3f, false, false},
|
||||
"VUMINV": {0x2e31a800, 0x3f, false, false},
|
||||
"VFMAXV": {0x2e30f800, fpAcrossArrs, false, false},
|
||||
"VFMINV": {0x2eb0f800, fpAcrossArrs, false, false},
|
||||
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false, false},
|
||||
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false, false},
|
||||
}
|
||||
|
||||
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
||||
|
||||
@@ -1894,3 +1894,87 @@ func TestArm64RegOffsetRejections(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SimdArrangementBits pins the arrangement bits the first SIMD pass
|
||||
// got wrong, word-verified against `go tool asm` (Go 1.27, arm64): the FP
|
||||
// one-bit size field (S=0, D=1 at bit 22), the SSHL/USHL size bits, the long
|
||||
// extends' immh field, the narrow family's fixed bit 16, the UADDW size bits
|
||||
// off the narrow side with Q from the spelling, the scalar D forms of the
|
||||
// bare VADD/VSUB spellings and the INS lane packing.
|
||||
func TestArm64SimdArrangementBits(t *testing.T) {
|
||||
got := arm64Words(t,
|
||||
"\tVFADD V0.S4, V0.S4, V1.S4\n"+
|
||||
"\tVFADD V0.D2, V0.D2, V1.D2\n"+
|
||||
"\tVFABS V0.D2, V1.D2\n"+
|
||||
"\tVSCVTF V1.D2, V2.D2\n"+
|
||||
"\tVSSHL V1.B8, V2.B8, V3.B8\n"+
|
||||
"\tVSSHL V1.S4, V2.S4, V3.S4\n"+
|
||||
"\tVUSHL V1.H4, V2.H4, V3.H4\n"+
|
||||
"\tVRBIT V24.B8, V24.B8\n"+
|
||||
"\tVUXTL V30.B8, V30.H8\n"+
|
||||
"\tVUXTL V29.S2, V2.D2\n"+
|
||||
"\tVUXTL2 V30.H8, V30.S4\n"+
|
||||
"\tVXTN V1.H8, V2.B8\n"+
|
||||
"\tVFCVTN V1.D2, V2.S2\n"+
|
||||
"\tVFCVTL V1.S2, V2.D2\n"+
|
||||
"\tVUADDW V13.H4, V10.S4, V11.S4\n"+
|
||||
"\tVUADDW2 V13.H8, V20.S4, V30.S4\n"+
|
||||
"\tVADD V1, V2, V3\n"+
|
||||
"\tVSUB V12, V20, V30\n"+
|
||||
"\tVMOV V12.S[2], V12.S[3]\n"+
|
||||
"\tVMOV V12.H[3], V12.H[5]\n")
|
||||
want := []uint32{
|
||||
0x4e20d401, // VFADD V0.4S: FP size field clear for S
|
||||
0x4e60d401, // VFADD V0.2D: FP size bit 22, not bit 23
|
||||
0x4ee0f801, // VFABS V1.2D: one-bit FP size
|
||||
0x4e61d822, // VSCVTF V2.2D: bit 22, the Q-only mask must not strip it
|
||||
0x0e214443, // VSSHL V3.8B: base without the pre-set size and Q bits
|
||||
0x4ea14443, // VSSHL V3.4S: integer size bits from the arrangement
|
||||
0x2e614443, // VUSHL V3.4H: U bit plus the H size
|
||||
0x2e605b18, // VRBIT V24.8B: no Q bit in the base
|
||||
0x2f08a7de, // VUXTL: immh = 1 at bit 19 for the byte extend
|
||||
0x2f20a7a2, // VUXTL: immh = 4 at bit 21 for the word extend
|
||||
0x6f10a7de, // VUXTL2: immh = 2 plus the 128-bit flag
|
||||
0x0e212822, // VXTN: fixed bit 16 in the base
|
||||
0x0e616822, // VFCVTN: bits 16 and 22, Q rides the spelling
|
||||
0x0e617822, // VFCVTL: bit 22, Q rides the spelling
|
||||
0x2e6d114b, // VUADDW: size bits off the narrow side, Q clear
|
||||
0x6e6d129e, // VUADDW2: size off the narrow side, Q from the spelling
|
||||
0x5ee18443, // VADD scalar D form for the bare spelling
|
||||
0x7eec869e, // VSUB scalar D form for the bare spelling
|
||||
0x6e1c458c, // INS: imm4 = 2<<2 for the word source lane 2
|
||||
0x6e16358c, // INS: imm4 = 3<<1 for the halfword source lane 3
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64SimdArrangementRejections pins the arrangements the toolchain
|
||||
// refuses on the FP SIMD rows and the element-to-element moves: the
|
||||
// half-width FP spellings, the Q1 spelling on the integer shifts, and the
|
||||
// mixed element letters of INS.
|
||||
func TestArm64SimdArrangementRejections(t *testing.T) {
|
||||
for _, src := range []string{
|
||||
"\tVFADD\tV1.H4, V2.H4, V3.H4\n",
|
||||
"\tVFADD\tV1.H8, V2.H8, V3.H8\n",
|
||||
"\tVFABS\tV1.H4, V2.H4\n",
|
||||
"\tVSCVTF\tV1.H4, V2.H4\n",
|
||||
"\tVSSHL\tV1.Q1, V2.Q1, V3.Q1\n",
|
||||
"\tVMOV\tV12.S[0], V12.D[1]\n",
|
||||
} {
|
||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
continue // a parse rejection is a rejection
|
||||
}
|
||||
if _, err := AssembleFileARM64(f); err == nil {
|
||||
t.Errorf("expected rejection for %q, got nil", strings.TrimSpace(src))
|
||||
}
|
||||
}
|
||||
}
|
||||
Vendored
+118
@@ -0,0 +1,118 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 SIMD arrangement bits: the FP one-bit
|
||||
// size field (S=0, D=1 at bit 22), the immh field of the long extends, the
|
||||
// narrow family's fixed bit 16, the SSHL/USHL size bits, the scalar D forms
|
||||
// of the bare VADD/VSUB spellings and the INS lane packing. Every function
|
||||
// is byte-compared against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func fpthree()
|
||||
TEXT ·fpthree(SB), NOSPLIT, $0-0
|
||||
VFADD V0.S4, V0.S4, V1.S4
|
||||
VFADD V0.D2, V0.D2, V1.D2
|
||||
VFMUL V0.S4, V0.S4, V1.S4
|
||||
VFMUL V0.D2, V0.D2, V1.D2
|
||||
VFDIV V0.S4, V0.S4, V1.S4
|
||||
VFDIV V0.D2, V0.D2, V1.D2
|
||||
VFMLA V1.D2, V12.D2, V1.D2
|
||||
VFMLA V1.S2, V12.S2, V1.S2
|
||||
VFMLA V1.S4, V12.S4, V1.S4
|
||||
VFMAX V3.S4, V2.S4, V1.S4
|
||||
VFMAXNM V3.S4, V2.S4, V1.S4
|
||||
VFADDP V3.D2, V2.D2, V1.D2
|
||||
VFCMEQ V1.S4, V2.S4, V3.S4
|
||||
VFCMGE V1.S4, V2.S4, V3.S4
|
||||
RET
|
||||
|
||||
// func fpunary()
|
||||
TEXT ·fpunary(SB), NOSPLIT, $0-0
|
||||
VFABS V0.D2, V1.D2
|
||||
VFNEG V0.D2, V1.D2
|
||||
VFSQRT V0.D2, V1.D2
|
||||
VFRINTN V0.D2, V1.D2
|
||||
VFRINTP V0.D2, V1.D2
|
||||
VFRINTM V0.D2, V1.D2
|
||||
VFRINTZ V0.D2, V1.D2
|
||||
VFCVTZS V1.D2, V2.D2
|
||||
VFCVTZU V1.D2, V2.D2
|
||||
VSCVTF V1.D2, V2.D2
|
||||
VUCVTF V1.D2, V2.D2
|
||||
VFABS V0.S4, V1.S4
|
||||
VSCVTF V1.S4, V2.S4
|
||||
VFNEG V0.S2, V1.S2
|
||||
RET
|
||||
|
||||
// func shifts()
|
||||
TEXT ·shifts(SB), NOSPLIT, $0-0
|
||||
VSSHL V1.S4, V2.S4, V3.S4
|
||||
VSSHL V1.S2, V2.S2, V3.S2
|
||||
VSSHL V1.H4, V2.H4, V3.H4
|
||||
VSSHL V1.H8, V2.H8, V3.H8
|
||||
VSSHL V1.B8, V2.B8, V3.B8
|
||||
VSSHL V1.B16, V2.B16, V3.B16
|
||||
VUSHL V1.S4, V2.S4, V3.S4
|
||||
VUSHL V1.S2, V2.S2, V3.S2
|
||||
VUSHL V1.H4, V2.H4, V3.H4
|
||||
VUSHL V1.H8, V2.H8, V3.H8
|
||||
VUSHL V1.B8, V2.B8, V3.B8
|
||||
VUSHL V1.B16, V2.B16, V3.B16
|
||||
VRBIT V24.B8, V24.B8
|
||||
RET
|
||||
|
||||
// func widen()
|
||||
TEXT ·widen(SB), NOSPLIT, $0-0
|
||||
VUXTL V30.B8, V30.H8
|
||||
VUXTL V30.H4, V29.S4
|
||||
VUXTL V29.S2, V2.D2
|
||||
VUXTL2 V30.H8, V30.S4
|
||||
VUXTL2 V29.S4, V2.D2
|
||||
VUXTL2 V30.B16, V2.H8
|
||||
VSXTL V1.B8, V2.H8
|
||||
VSXTL V1.H4, V2.S4
|
||||
VSXTL V1.S2, V2.D2
|
||||
VSXTL2 V1.B16, V2.H8
|
||||
VSXTL2 V1.H8, V2.S4
|
||||
VSXTL2 V1.S4, V2.D2
|
||||
VXTN V1.H8, V2.B8
|
||||
VXTN V1.S4, V2.H4
|
||||
VXTN V1.D2, V2.S2
|
||||
VXTN2 V1.H8, V2.B16
|
||||
VSQXTN V1.D2, V2.S2
|
||||
VSQXTN2 V1.S4, V2.H8
|
||||
VSQXTUN V1.H8, V2.B8
|
||||
VUQXTN V1.S4, V2.H4
|
||||
VUQXTN2 V1.D2, V2.S4
|
||||
RET
|
||||
|
||||
// func uaddw()
|
||||
TEXT ·uaddw(SB), NOSPLIT, $0-0
|
||||
VUADDW V9.B8, V12.H8, V14.H8
|
||||
VUADDW V13.H4, V10.S4, V11.S4
|
||||
VUADDW V21.S2, V24.D2, V29.D2
|
||||
VUADDW2 V9.B16, V12.H8, V14.H8
|
||||
VUADDW2 V13.H8, V20.S4, V30.S4
|
||||
VUADDW2 V21.S4, V24.D2, V29.D2
|
||||
RET
|
||||
|
||||
// func bareadd()
|
||||
TEXT ·bareadd(SB), NOSPLIT, $0-0
|
||||
VADD V1, V2, V3
|
||||
VADD V1, V3, V3
|
||||
VSUB V12, V30, V30
|
||||
VSUB V12, V20, V30
|
||||
VADD V1.B8, V2.B8, V3.B8
|
||||
VSUB V12.B8, V30.B8, V30.B8
|
||||
RET
|
||||
|
||||
// func lanes()
|
||||
TEXT ·lanes(SB), NOSPLIT, $0-0
|
||||
VMOV V12.D[0], V12.D[1]
|
||||
VMOV V10.S[0], V12.S[1]
|
||||
VMOV V9.H[0], V12.H[1]
|
||||
VMOV V12.B[0], V12.B[1]
|
||||
VMOV V12.S[2], V12.S[3]
|
||||
VMOV V12.H[3], V12.H[5]
|
||||
RET
|
||||
@@ -38,6 +38,7 @@ func TestGroundTruthARM64(t *testing.T) {
|
||||
"../testdata/verify/qmov_arm64.s",
|
||||
"../testdata/verify/splits_arm64.s",
|
||||
"../testdata/verify/regoffset_arm64.s",
|
||||
"../testdata/verify/simdarr_arm64.s",
|
||||
"../testdata/verify/crypto_arm64.s",
|
||||
"../testdata/verify/integer_arm64.s",
|
||||
"../testdata/verify/simd_arm64.s",
|
||||
|
||||
Reference in new issue
Block a user