feat(asm): encode the arm64 SIMD arrangement bits
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
458cdd2066
commit
9ef14bdb71
5 files changed
+407
-133
No files matched your search
+66
-11
@@ -3028,8 +3028,10 @@ func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
|
||||
}
|
||||
|
||||
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
|
||||
// instruction contributes: the driving arrangement's size and Q bits, or for
|
||||
// the FCVT family only the Q bit, whose size field is fixed in the base.
|
||||
// instruction contributes: the driving arrangement's size and Q bits, for
|
||||
// the FCVT family only the Q bit (whose size field is fixed in the base),
|
||||
// and for the long extend family the immh shift field the long forms imply
|
||||
// (immh = esize/8) plus the Q bit for the .2 spellings.
|
||||
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
|
||||
if spec.qonly {
|
||||
if two {
|
||||
@@ -3037,6 +3039,14 @@ func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
|
||||
}
|
||||
return 0
|
||||
}
|
||||
if spec.form == a64NLTwoLong {
|
||||
se, _, _ := arm64SimdNLArr(drive)
|
||||
bits := uint32(se) << 19
|
||||
if two {
|
||||
bits |= 1 << 30
|
||||
}
|
||||
return bits
|
||||
}
|
||||
return a64ArrBits[a64ArrIndex(drive)]
|
||||
}
|
||||
|
||||
@@ -3101,8 +3111,10 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
|
||||
var pairErr error
|
||||
if spec.form == a64NLThreeWide {
|
||||
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
|
||||
// the arrangement bits follow the wide side.
|
||||
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
|
||||
// the size bits follow the narrow side (Vm) and the 128-bit flag
|
||||
// follows the spelling: the plain form keeps Q clear, the .2
|
||||
// form sets it.
|
||||
drive, pairErr = vm.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
|
||||
} else {
|
||||
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
|
||||
// wide one; the arrangement bits follow the narrow source.
|
||||
@@ -3118,6 +3130,14 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
|
||||
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
|
||||
}
|
||||
arrBits := arm64SimdNLArrBits(spec, drive, two)
|
||||
if spec.form == a64NLThreeWide {
|
||||
// The size bits ride the narrow side's letter with Q forced by
|
||||
// the spelling alone.
|
||||
arrBits = a64ArrBits[a64ArrIndex(drive)] &^ (1 << 30)
|
||||
if two {
|
||||
arrBits |= 1 << 30
|
||||
}
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
case a64NLThreeLongShift, a64NLThreeNarrowShift:
|
||||
if len(ops) != 3 || !isImmOperand(ops[0]) {
|
||||
@@ -4082,7 +4102,7 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
}
|
||||
allowed := uint16(0x7f)
|
||||
if strings.HasPrefix(mnem, "VFCM") {
|
||||
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
|
||||
allowed = fpSimdArrs
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
|
||||
if err != nil {
|
||||
@@ -4118,15 +4138,28 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
}
|
||||
vs[i], arrs[i] = v, v.arr
|
||||
}
|
||||
// The bare three-register spellings of VADD/VSUB are the scalar D forms
|
||||
// (the toolchain's ADD/SUB scalar rows), not the 8B vector rows.
|
||||
if (mnem == "VADD" || mnem == "VSUB") &&
|
||||
arrs[0] == "" && arrs[1] == "" && arrs[2] == "" {
|
||||
base := uint32(0x5ee08400) // VADD scalar
|
||||
if mnem == "VSUB" {
|
||||
base = 0x7ee08400
|
||||
}
|
||||
return a64wordLE(base | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, arrs, spec.arrs)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
arrBits := a64ArrBits[arr]
|
||||
if spec.fixed {
|
||||
if spec.fp {
|
||||
// The FP rows carry a one-bit size field (S=0, D=1) instead of the
|
||||
// integer size, and no Q-only masking applies to them.
|
||||
arrBits = a64FPArrBits[arr]
|
||||
} else if spec.fixed {
|
||||
arrBits = 0
|
||||
}
|
||||
if a64SimdQOnly[mnem] {
|
||||
} else if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
||||
@@ -4175,7 +4208,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
||||
return nil, err
|
||||
}
|
||||
arrBits := a64ArrBits[arr]
|
||||
if a64SimdQOnly[mnem] {
|
||||
if spec.fp {
|
||||
// The FP rows carry the one-bit FP size field instead of the
|
||||
// integer size, and no Q-only masking applies to them.
|
||||
arrBits = a64FPArrBits[arr]
|
||||
} else if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
@@ -4402,12 +4439,30 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
||||
}
|
||||
if dst.hasIdx {
|
||||
// Element to element.
|
||||
// Element to element: the toolchain requires the two element letters
|
||||
// to match (asm7.go case 92), packs the destination index into imm5
|
||||
// and the source index, in units of the element size, into imm4.
|
||||
if src.arr != dst.arr {
|
||||
return nil, fmt.Errorf("%s: operand mismatch: %s and %s elements", mnem, src.arr, dst.arr)
|
||||
}
|
||||
df, ok := a64ElemField(dst.arr, dst.idx)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
||||
}
|
||||
return a64wordLE(0x6e000400 | df<<16 | sf>>1<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
|
||||
var imm4 uint32
|
||||
switch src.arr {
|
||||
case "B":
|
||||
imm4 = uint32(src.idx)
|
||||
case "H":
|
||||
imm4 = uint32(src.idx) << 1
|
||||
case "S":
|
||||
imm4 = uint32(src.idx) << 2
|
||||
case "D":
|
||||
imm4 = uint32(src.idx) << 3
|
||||
default:
|
||||
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
||||
}
|
||||
return a64wordLE(0x6e000400 | df<<16 | imm4&0xf<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
|
||||
}
|
||||
if dstGP {
|
||||
// Element to a general register: UMOV, with the D form setting bit
|
||||
|
||||
Reference in new issue
Block a user