feat(asm): add the EVEX FP helper tail and gather/scatter with VSIB
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
+230
-4
@@ -234,6 +234,80 @@ var evexTable = map[string]evexSpec{
|
||||
// EVEX W1 qword shifts.
|
||||
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||||
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src).
|
||||
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
// EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is
|
||||
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
|
||||
// forms, these take the 66 prefix; W selects double/single.
|
||||
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VRCP14SS": {2, 0x4D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VRSQRT14SD": {2, 0x4F, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
// EVEX.66.0F38 — scale by a power of two (NDS form).
|
||||
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
|
||||
// EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst,
|
||||
// rm=src, imm8).
|
||||
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
"VGETMANTPD": {3, 0x26, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||
// EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS +
|
||||
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
|
||||
// prefix; W selects double/single.
|
||||
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||
"VRNDSCALESS": {3, 0x0A, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||
"VGETMANTSD": {3, 0x27, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||
"VGETMANTSS": {3, 0x27, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||
"VREDUCESD": {3, 0x57, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||
"VREDUCESS": {3, 0x57, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||
"VFIXUPIMMPD": {3, 0x54, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VFIXUPIMMPS": {3, 0x54, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VFIXUPIMMSD": {3, 0x55, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||
"VFIXUPIMMSS": {3, 0x55, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||
"VRANGEPD": {3, 0x50, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VRANGEPS": {3, 0x50, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||
|
||||
// EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the
|
||||
// reg field carries the opmask destination. The packed forms carry an
|
||||
// explicit length in the mnemonic (X/Y/Z).
|
||||
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
|
||||
"VFPCLASSPDY": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 32, 0}},
|
||||
"VFPCLASSPDZ": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 0, 64}},
|
||||
"VFPCLASSPSX": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{16, 0, 0}},
|
||||
"VFPCLASSPSY": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 32, 0}},
|
||||
"VFPCLASSPSZ": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 0, 64}},
|
||||
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
|
||||
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
|
||||
|
||||
// EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit
|
||||
// source sets the length); the rest follow the destination.
|
||||
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||||
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
||||
// EVEX.66.0F38 — half-precision convert (half-width source).
|
||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
|
||||
// rm=dst, imm8 — the extract layout).
|
||||
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
|
||||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||||
// the xmm/ymm/zmm destination lengths).
|
||||
@@ -466,6 +540,9 @@ var evexRound = map[string]bool{
|
||||
"VMINSD": true, "VMAXSD": true,
|
||||
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
|
||||
"VMINSS": true, "VMAXSS": true,
|
||||
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
|
||||
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
|
||||
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
|
||||
}
|
||||
|
||||
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
||||
@@ -475,6 +552,16 @@ var evexBcstN = map[string]int{
|
||||
"VMINPD": 8, "VMAXPD": 8,
|
||||
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
|
||||
"VMINPS": 4, "VMAXPS": 4,
|
||||
"VRCP14PD": 8, "VRCP14PS": 4, "VRSQRT14PD": 8, "VRSQRT14PS": 4,
|
||||
"VGETEXPPD": 8, "VGETEXPPS": 4,
|
||||
"VSCALEFPD": 8, "VSCALEFPS": 4,
|
||||
"VRNDSCALEPD": 8, "VRNDSCALEPS": 4,
|
||||
"VGETMANTPD": 8, "VGETMANTPS": 4,
|
||||
"VREDUCEPD": 8, "VREDUCEPS": 4,
|
||||
"VFIXUPIMMPD": 8, "VFIXUPIMMPS": 4,
|
||||
"VRANGEPD": 8, "VRANGEPS": 4,
|
||||
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
|
||||
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
|
||||
}
|
||||
|
||||
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
||||
@@ -547,6 +634,10 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
||||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||
return kdst(e.encodeEvexNDS3Imm)
|
||||
}
|
||||
case vexImmRM:
|
||||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||
return kdst(e.encodeEvexImmRM)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -636,7 +727,9 @@ func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffi
|
||||
}
|
||||
|
||||
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
||||
// (reg = dst, rm = src, imm8), e.g. VPSHUFD.
|
||||
// (reg = dst, rm = src, imm8), e.g. VPSHUFD. The destination may be an
|
||||
// opmask register (VFPCLASS*), in which case the vector length comes from
|
||||
// the source.
|
||||
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
@@ -647,11 +740,15 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
|
||||
return fmt.Errorf("shuffle control must be an immediate")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("shuffle destination must be a vector register")
|
||||
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||||
return fmt.Errorf("shuffle destination must be a vector or mask register")
|
||||
}
|
||||
ll := dstReg.vecLenBit()
|
||||
if r, ok := src.(Reg); ok && r.isVec() {
|
||||
if dstReg.mask {
|
||||
if r, ok := src.(Reg); ok && r.isVec() {
|
||||
ll = r.vecLenBit()
|
||||
}
|
||||
} else if r, ok := src.(Reg); ok && r.isVec() {
|
||||
ll = r.vecLenBit()
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
@@ -1034,6 +1131,135 @@ func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte,
|
||||
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
||||
}
|
||||
|
||||
// gatherSpec describes a gather/scatter family member: all live in
|
||||
// 66.0F38; the opcode and W select the index and data element widths, and n
|
||||
// is the data element size (the EVEX disp8×N multiplier).
|
||||
type gatherSpec struct {
|
||||
opcode byte
|
||||
w int
|
||||
n int
|
||||
}
|
||||
|
||||
var gatherTable = map[string]gatherSpec{
|
||||
"VGATHERDPS": {0x92, 0, 4},
|
||||
"VGATHERDPD": {0x92, 1, 8},
|
||||
"VGATHERQPS": {0x93, 0, 4},
|
||||
"VGATHERQPD": {0x93, 1, 8},
|
||||
"VPGATHERDD": {0x90, 0, 4},
|
||||
"VPGATHERDQ": {0x90, 1, 8},
|
||||
"VPGATHERQD": {0x91, 0, 4},
|
||||
"VPGATHERQQ": {0x91, 1, 8},
|
||||
}
|
||||
|
||||
var scatterTable = map[string]gatherSpec{
|
||||
"VSCATTERDPS": {0xA2, 0, 4},
|
||||
"VSCATTERDPD": {0xA2, 1, 8},
|
||||
"VSCATTERQPS": {0xA3, 0, 4},
|
||||
"VSCATTERQPD": {0xA3, 1, 8},
|
||||
"VPSCATTERDD": {0xA0, 0, 4},
|
||||
"VPSCATTERDQ": {0xA0, 1, 8},
|
||||
"VPSCATTERQD": {0xA1, 0, 4},
|
||||
"VPSCATTERQQ": {0xA1, 1, 8},
|
||||
}
|
||||
|
||||
// isGather reports whether the mnemonic is a gather instruction.
|
||||
func isGather(upper string) bool {
|
||||
_, ok := gatherTable[upper]
|
||||
return ok
|
||||
}
|
||||
|
||||
// isScatter reports whether the mnemonic is a scatter instruction.
|
||||
func isScatter(upper string) bool {
|
||||
_, ok := scatterTable[upper]
|
||||
return ok
|
||||
}
|
||||
|
||||
// vsibLen validates a VSIB memory operand (the index must be a vector
|
||||
// register) and returns it with the vector length the index selects — the
|
||||
// EVEX L'L field follows the index register, not the data register.
|
||||
func vsibLen(op Operand, what string) (Mem, int, error) {
|
||||
m, ok := op.(Mem)
|
||||
if !ok || !m.HasIndex || !m.Index.isVec() {
|
||||
return Mem{}, 0, fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", what)
|
||||
}
|
||||
return m, m.Index.vecLenBit(), nil
|
||||
}
|
||||
|
||||
// encodeGather encodes a gather. The VEX spelling carries the mask in a
|
||||
// vector register (OP mask, vsib, dst: vvvv = mask, rm = vsib, reg = dst,
|
||||
// L follows the data register); the EVEX spelling carries it in aaa (OP
|
||||
// vsib, K, dst: rm = vsib, reg = dst, L follows the VSIB index).
|
||||
func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexSuffix) error {
|
||||
rest, mask, err := splitMask(ops)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if mask != 0 || sfx.any() {
|
||||
// EVEX form: OP vsib, K, dst.
|
||||
if len(rest) != 2 {
|
||||
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
||||
}
|
||||
vsib, ll, err := vsibLen(rest[0], upper)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dst, ok := rest[1].(Reg)
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||
}
|
||||
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
||||
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
||||
}
|
||||
// VEX form: OP mask, vsib, dst.
|
||||
if len(rest) != 3 {
|
||||
return fmt.Errorf("%s expects 3 operands (mask, vsib, dst), got %d", upper, len(rest))
|
||||
}
|
||||
maskReg, ok := rest[0].(Reg)
|
||||
if !ok || !maskReg.isVec() {
|
||||
return fmt.Errorf("%s: mask must be a vector register", upper)
|
||||
}
|
||||
vsib, _, err := vsibLen(rest[1], upper)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dst, ok := rest[2].(Reg)
|
||||
if !ok || !dst.isVec() {
|
||||
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||
}
|
||||
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
|
||||
rBit := 0
|
||||
if dst.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
|
||||
}
|
||||
|
||||
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src,
|
||||
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
|
||||
// index.
|
||||
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
|
||||
rest, mask, err := splitMask(ops)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if mask == 0 {
|
||||
return fmt.Errorf("%s requires a K mask register", upper)
|
||||
}
|
||||
if len(rest) != 2 {
|
||||
return fmt.Errorf("%s expects 3 operands (src, K, vsib), got %d", upper, len(ops))
|
||||
}
|
||||
src, ok := rest[0].(Reg)
|
||||
if !ok || !src.isVec() {
|
||||
return fmt.Errorf("%s: source must be a vector register", upper)
|
||||
}
|
||||
vsib, ll, err := vsibLen(rest[1], upper)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
||||
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||||
}
|
||||
|
||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
||||
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
||||
|
||||
Reference in New Issue
Block a user