feat(asm): encode the arm64 SIMD arrangement bits

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:36:24 +02:00
1 parent 458cdd2066
commit 9ef14bdb71
5 files changed
+407 -133

No files matched your search

+66 -11
View File
@@ -3028,8 +3028,10 @@ func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
}
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
// instruction contributes: the driving arrangement's size and Q bits, or for
// the FCVT family only the Q bit, whose size field is fixed in the base.
// instruction contributes: the driving arrangement's size and Q bits, for
// the FCVT family only the Q bit (whose size field is fixed in the base),
// and for the long extend family the immh shift field the long forms imply
// (immh = esize/8) plus the Q bit for the .2 spellings.
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
if spec.qonly {
if two {
@@ -3037,6 +3039,14 @@ func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
}
return 0
}
if spec.form == a64NLTwoLong {
se, _, _ := arm64SimdNLArr(drive)
bits := uint32(se) << 19
if two {
bits |= 1 << 30
}
return bits
}
return a64ArrBits[a64ArrIndex(drive)]
}
@@ -3101,8 +3111,10 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
var pairErr error
if spec.form == a64NLThreeWide {
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
// the arrangement bits follow the wide side.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
// the size bits follow the narrow side (Vm) and the 128-bit flag
// follows the spelling: the plain form keeps Q clear, the .2
// form sets it.
drive, pairErr = vm.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
} else {
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
// wide one; the arrangement bits follow the narrow source.
@@ -3118,6 +3130,14 @@ func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]b
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
if spec.form == a64NLThreeWide {
// The size bits ride the narrow side's letter with Q forced by
// the spelling alone.
arrBits = a64ArrBits[a64ArrIndex(drive)] &^ (1 << 30)
if two {
arrBits |= 1 << 30
}
}
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongShift, a64NLThreeNarrowShift:
if len(ops) != 3 || !isImmOperand(ops[0]) {
@@ -4082,7 +4102,7 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
allowed := uint16(0x7f)
if strings.HasPrefix(mnem, "VFCM") {
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
allowed = fpSimdArrs
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
if err != nil {
@@ -4118,15 +4138,28 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
}
vs[i], arrs[i] = v, v.arr
}
// The bare three-register spellings of VADD/VSUB are the scalar D forms
// (the toolchain's ADD/SUB scalar rows), not the 8B vector rows.
if (mnem == "VADD" || mnem == "VSUB") &&
arrs[0] == "" && arrs[1] == "" && arrs[2] == "" {
base := uint32(0x5ee08400) // VADD scalar
if mnem == "VSUB" {
base = 0x7ee08400
}
return a64wordLE(base | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
}
arr, err := arm64SimdArrs(mnem, arrs, spec.arrs)
if err != nil {
return nil, err
}
arrBits := a64ArrBits[arr]
if spec.fixed {
if spec.fp {
// The FP rows carry a one-bit size field (S=0, D=1) instead of the
// integer size, and no Q-only masking applies to them.
arrBits = a64FPArrBits[arr]
} else if spec.fixed {
arrBits = 0
}
if a64SimdQOnly[mnem] {
} else if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
@@ -4175,7 +4208,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
return nil, err
}
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
if spec.fp {
// The FP rows carry the one-bit FP size field instead of the
// integer size, and no Q-only masking applies to them.
arrBits = a64FPArrBits[arr]
} else if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
@@ -4402,12 +4439,30 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
if dst.hasIdx {
// Element to element.
// Element to element: the toolchain requires the two element letters
// to match (asm7.go case 92), packs the destination index into imm5
// and the source index, in units of the element size, into imm4.
if src.arr != dst.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s elements", mnem, src.arr, dst.arr)
}
df, ok := a64ElemField(dst.arr, dst.idx)
if !ok {
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
return a64wordLE(0x6e000400 | df<<16 | sf>>1<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
var imm4 uint32
switch src.arr {
case "B":
imm4 = uint32(src.idx)
case "H":
imm4 = uint32(src.idx) << 1
case "S":
imm4 = uint32(src.idx) << 2
case "D":
imm4 = uint32(src.idx) << 3
default:
return nil, fmt.Errorf("%s: invalid element operand", mnem)
}
return a64wordLE(0x6e000400 | df<<16 | imm4&0xf<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
}
if dstGP {
// Element to a general register: UMOV, with the D form setting bit