feat(arch): add the amd64 fp16 packed conversion family
Assisted-by: GLM 5.3
This commit is contained in:
1 parent
fb6d01a7d0
commit
8de1b371da
4 files changed
+659
-34
No files matched your search
+193
-28
@@ -10,14 +10,17 @@
|
||||
//
|
||||
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
|
||||
// latter's scalar core with its imm8-control group, its packed 512-bit and
|
||||
// VL arithmetic and the embedded rounding of its FP operations, in their
|
||||
// EVEX register forms. The encodings are transcribed from the SDM
|
||||
// instruction entries and cross-checked against binutils-gdb's assembler
|
||||
// testsuite; the golden vectors in amd64_ext_test.go pin the bytes.
|
||||
// VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, no
|
||||
// longer belong here: the Go toolchain's assembler knows them today, they
|
||||
// live in the generated table and the EVEX encoder, and a mnemonic the
|
||||
// toolchain has is not an extension.
|
||||
// VL arithmetic, the embedded rounding of its FP operations and its fourteen
|
||||
// packed conversion directions, in their EVEX register forms. The encodings
|
||||
// are transcribed from the SDM instruction entries and cross-checked against
|
||||
// binutils-gdb's assembler testsuite; the golden vectors in amd64_ext_test.go
|
||||
// pin the bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19
|
||||
// survey, no longer belong here: the Go toolchain's assembler knows them
|
||||
// today, they live in the generated table and the EVEX encoder, and a
|
||||
// mnemonic the toolchain has is not an extension. VCVTPS2PH, VCVTUDQ2PS
|
||||
// and the rest of the classic conversion set follow the same rule, which is
|
||||
// why the conversions here are the FP16 directions the toolchain has never
|
||||
// emitted.
|
||||
//
|
||||
// The forms encode the unmasked shapes: register forms throughout, and the
|
||||
// memory forms beside them, the scalar ones the manual spells m16, m32 and
|
||||
@@ -462,7 +465,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
||||
switch in.Form {
|
||||
case ExtFormAmdVec3:
|
||||
return in.encodeAmdVec3(ops)
|
||||
case ExtFormAmdVec2, ExtFormAmdVec2Half:
|
||||
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter,
|
||||
ExtFormAmdVec2ToQuarter:
|
||||
return in.encodeAmdVec2(ops)
|
||||
case ExtFormAmdMask2:
|
||||
return in.encodeAmdMask2(ops)
|
||||
@@ -560,18 +564,30 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
|
||||
|
||||
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
|
||||
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
|
||||
// of destination, and at 128 bits the companion stays the class itself. An
|
||||
// entry with Mem set takes the memory shape of the source position too: the
|
||||
// compares and the packed square root read their source from memory, the
|
||||
// packed square root's source carrying the {1toN} broadcast as EVEX.b, and
|
||||
// the narrow BF16 convert reads its full-width source there. An entry with
|
||||
// Er or Sae set takes the rounding decoration on its register form alone;
|
||||
// the memory shape refuses one.
|
||||
// of destination, and at 128 bits the companion stays the class itself. The
|
||||
// wide form narrows the source instead, the widening FP16 conversions, where
|
||||
// L'L names the destination: VCVTPH2DQ converts 256 bits of source into 512
|
||||
// bits of destination. The two quarter forms hold one operand in the XMM
|
||||
// class at every length: the source under the widening VCVTPH2QQ, the
|
||||
// destination under the narrowing VCVTQQ2PH. An entry with Mem set takes
|
||||
// the memory shape of the source position too: the compares and the packed
|
||||
// square root read their source from memory, the packed square root's source
|
||||
// carrying the {1toN} broadcast as EVEX.b, and the narrow BF16 convert reads
|
||||
// its full-width source there. An entry with Er or Sae set takes the
|
||||
// rounding decoration on its register form alone; the memory shape refuses
|
||||
// one.
|
||||
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
destClass := class
|
||||
if in.Form == ExtFormAmdVec2Half {
|
||||
destClass, srcClass := class, class
|
||||
switch in.Form {
|
||||
case ExtFormAmdVec2Half:
|
||||
destClass = amd64HalfClass(class)
|
||||
case ExtFormAmdVec2Wide:
|
||||
srcClass = amd64HalfClass(class)
|
||||
case ExtFormAmdVec2Quarter:
|
||||
srcClass = ExtXMM
|
||||
case ExtFormAmdVec2ToQuarter:
|
||||
destClass = ExtXMM
|
||||
}
|
||||
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[1], 2)
|
||||
if err != nil {
|
||||
@@ -597,7 +613,7 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return out, nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
if err := in.amd64Vector(ops[0], srcClass, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(dest, destClass, 2); err != nil {
|
||||
@@ -878,13 +894,21 @@ func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
|
||||
// (the move into a vector register and the integer conversions) and vec, gpr
|
||||
// (the move out of one). In both orders the second operand is the
|
||||
// destination in the reg field and the first the r/m source; the vector
|
||||
// changes position with the form.
|
||||
// changes position with the form. An entry with Mem set takes the memory
|
||||
// shape of the vector source, the manual's m16 beside the register: the
|
||||
// value converts straight out of memory.
|
||||
func (in ExtInstr) encodeAmdGprPair(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
vecPos := 1
|
||||
if in.Form == ExtFormAmdVecGpr {
|
||||
vecPos = 0
|
||||
}
|
||||
if in.Mem == 1 && in.Form == ExtFormAmdVecGpr && ops[0].Kind == ExtMem {
|
||||
if err := in.amd64Gpr(ops[1], 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1)
|
||||
}
|
||||
if err := in.amd64Vector(ops[vecPos], class, vecPos+1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -1022,16 +1046,16 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"},
|
||||
{Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"},
|
||||
{Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value",
|
||||
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2SS (EVEX.NDS.LIG.MAP6.W0 13 /r)"},
|
||||
{Name: "VCVTSH2SD", Summary: "Convert a low FP16 value to an FP64 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"},
|
||||
{Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"},
|
||||
{Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
@@ -1046,16 +1070,16 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"},
|
||||
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W0 2D /r)"},
|
||||
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 64-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W1 2D /r)"},
|
||||
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 32-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W0 79 /r)"},
|
||||
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 64-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W1 79 /r)"},
|
||||
|
||||
// AVX512-FP16 packed arithmetic: the full ZMM lanes the scalar core
|
||||
@@ -1148,4 +1172,145 @@ var amd64Extensions = []ExtInstr{
|
||||
{Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"},
|
||||
|
||||
// AVX512-FP16 packed conversions: the fourteen directions between the
|
||||
// FP16 lanes and the integer and double-precision companions, three
|
||||
// register widths each. L'L names the governing operand, whose class the
|
||||
// form spells: the source on the narrowing converts (VCVTDQ2PH narrows
|
||||
// its dword source to the half-width destination, VCVTQQ2PH and VCVTPD2PH
|
||||
// to the quarter-width XMM destination), the destination on the widening
|
||||
// ones (VCVTPH2DQ widens its half-width source, VCVTPH2QQ and VCVTPH2PD
|
||||
// their quarter-width XMM source). The FP16-to-integer directions round
|
||||
// and take Er on their 512-bit register forms; VCVTPH2PD widens exactly
|
||||
// and takes none. The integer-to-FP16 sources read their elements from
|
||||
// memory under the {1toN} broadcast, the FP16 sources their full-width
|
||||
// vectors; every destination is write-masked. The encodings are
|
||||
// transcribed from the SDM entries and pinned byte for byte against the
|
||||
// local GNU assembler, whose 2.46 table matches the manual row for row.
|
||||
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.512.66.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.256.66.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.128.66.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.512.NP.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.256.NP.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.128.NP.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.512.F3.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.256.F3.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.128.F3.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.512.F2.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.256.F2.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.128.F2.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.512.66.MAP5.W0 5B /r, half-width source)"},
|
||||
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.256.66.MAP5.W0 5B /r, half-width source)"},
|
||||
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.128.66.MAP5.W0 5B /r, half-width source)"},
|
||||
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.512.NP.MAP5.W0 79 /r, half-width source)"},
|
||||
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.256.NP.MAP5.W0 79 /r, half-width source)"},
|
||||
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.128.NP.MAP5.W0 79 /r, half-width source)"},
|
||||
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.512.NP.MAP5.W0 5B /r, YMM destination)"},
|
||||
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.256.NP.MAP5.W0 5B /r, XMM destination)"},
|
||||
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.128.NP.MAP5.W0 5B /r, XMM destination)"},
|
||||
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.512.F2.MAP5.W0 7A /r, YMM destination)"},
|
||||
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.256.F2.MAP5.W0 7A /r, XMM destination)"},
|
||||
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.128.F2.MAP5.W0 7A /r, XMM destination)"},
|
||||
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.512.66.MAP5.W0 7B /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.256.66.MAP5.W0 7B /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.128.66.MAP5.W0 7B /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.512.66.MAP5.W0 79 /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.256.66.MAP5.W0 79 /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.128.66.MAP5.W0 79 /r, quarter-width source)"},
|
||||
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x84, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.512.NP.MAP5.W1 5B /r, XMM destination)"},
|
||||
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x84, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.256.NP.MAP5.W1 5B /r, XMM destination)"},
|
||||
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x84, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.128.NP.MAP5.W1 5B /r, XMM destination)"},
|
||||
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.512.F2.MAP5.W1 7A /r, XMM destination)"},
|
||||
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.256.F2.MAP5.W1 7A /r, XMM destination)"},
|
||||
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.128.F2.MAP5.W1 7A /r, XMM destination)"},
|
||||
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.512.NP.MAP5.W0 5A /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.256.NP.MAP5.W0 5A /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.128.NP.MAP5.W0 5A /r, quarter-width source)"},
|
||||
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x85, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.512.66.MAP5.W1 5A /r, XMM destination)"},
|
||||
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x85, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.256.66.MAP5.W1 5A /r, XMM destination)"},
|
||||
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x85, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.128.66.MAP5.W1 5A /r, XMM destination)"},
|
||||
}
|
||||
Reference in new issue
Block a user