feat(arch): add the amd64 fp16 packed conversion family
Assisted-by: GLM 5.3
This commit is contained in:
1 parent
fb6d01a7d0
commit
8de1b371da
4 files changed
+659
-34
No files matched your search
+193
-28
@@ -10,14 +10,17 @@
|
||||
//
|
||||
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
|
||||
// latter's scalar core with its imm8-control group, its packed 512-bit and
|
||||
// VL arithmetic and the embedded rounding of its FP operations, in their
|
||||
// EVEX register forms. The encodings are transcribed from the SDM
|
||||
// instruction entries and cross-checked against binutils-gdb's assembler
|
||||
// testsuite; the golden vectors in amd64_ext_test.go pin the bytes.
|
||||
// VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, no
|
||||
// longer belong here: the Go toolchain's assembler knows them today, they
|
||||
// live in the generated table and the EVEX encoder, and a mnemonic the
|
||||
// toolchain has is not an extension.
|
||||
// VL arithmetic, the embedded rounding of its FP operations and its fourteen
|
||||
// packed conversion directions, in their EVEX register forms. The encodings
|
||||
// are transcribed from the SDM instruction entries and cross-checked against
|
||||
// binutils-gdb's assembler testsuite; the golden vectors in amd64_ext_test.go
|
||||
// pin the bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19
|
||||
// survey, no longer belong here: the Go toolchain's assembler knows them
|
||||
// today, they live in the generated table and the EVEX encoder, and a
|
||||
// mnemonic the toolchain has is not an extension. VCVTPS2PH, VCVTUDQ2PS
|
||||
// and the rest of the classic conversion set follow the same rule, which is
|
||||
// why the conversions here are the FP16 directions the toolchain has never
|
||||
// emitted.
|
||||
//
|
||||
// The forms encode the unmasked shapes: register forms throughout, and the
|
||||
// memory forms beside them, the scalar ones the manual spells m16, m32 and
|
||||
@@ -462,7 +465,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
||||
switch in.Form {
|
||||
case ExtFormAmdVec3:
|
||||
return in.encodeAmdVec3(ops)
|
||||
case ExtFormAmdVec2, ExtFormAmdVec2Half:
|
||||
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter,
|
||||
ExtFormAmdVec2ToQuarter:
|
||||
return in.encodeAmdVec2(ops)
|
||||
case ExtFormAmdMask2:
|
||||
return in.encodeAmdMask2(ops)
|
||||
@@ -560,18 +564,30 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
|
||||
|
||||
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
|
||||
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
|
||||
// of destination, and at 128 bits the companion stays the class itself. An
|
||||
// entry with Mem set takes the memory shape of the source position too: the
|
||||
// compares and the packed square root read their source from memory, the
|
||||
// packed square root's source carrying the {1toN} broadcast as EVEX.b, and
|
||||
// the narrow BF16 convert reads its full-width source there. An entry with
|
||||
// Er or Sae set takes the rounding decoration on its register form alone;
|
||||
// the memory shape refuses one.
|
||||
// of destination, and at 128 bits the companion stays the class itself. The
|
||||
// wide form narrows the source instead, the widening FP16 conversions, where
|
||||
// L'L names the destination: VCVTPH2DQ converts 256 bits of source into 512
|
||||
// bits of destination. The two quarter forms hold one operand in the XMM
|
||||
// class at every length: the source under the widening VCVTPH2QQ, the
|
||||
// destination under the narrowing VCVTQQ2PH. An entry with Mem set takes
|
||||
// the memory shape of the source position too: the compares and the packed
|
||||
// square root read their source from memory, the packed square root's source
|
||||
// carrying the {1toN} broadcast as EVEX.b, and the narrow BF16 convert reads
|
||||
// its full-width source there. An entry with Er or Sae set takes the
|
||||
// rounding decoration on its register form alone; the memory shape refuses
|
||||
// one.
|
||||
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
destClass := class
|
||||
if in.Form == ExtFormAmdVec2Half {
|
||||
destClass, srcClass := class, class
|
||||
switch in.Form {
|
||||
case ExtFormAmdVec2Half:
|
||||
destClass = amd64HalfClass(class)
|
||||
case ExtFormAmdVec2Wide:
|
||||
srcClass = amd64HalfClass(class)
|
||||
case ExtFormAmdVec2Quarter:
|
||||
srcClass = ExtXMM
|
||||
case ExtFormAmdVec2ToQuarter:
|
||||
destClass = ExtXMM
|
||||
}
|
||||
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[1], 2)
|
||||
if err != nil {
|
||||
@@ -597,7 +613,7 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return out, nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
if err := in.amd64Vector(ops[0], srcClass, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(dest, destClass, 2); err != nil {
|
||||
@@ -878,13 +894,21 @@ func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
|
||||
// (the move into a vector register and the integer conversions) and vec, gpr
|
||||
// (the move out of one). In both orders the second operand is the
|
||||
// destination in the reg field and the first the r/m source; the vector
|
||||
// changes position with the form.
|
||||
// changes position with the form. An entry with Mem set takes the memory
|
||||
// shape of the vector source, the manual's m16 beside the register: the
|
||||
// value converts straight out of memory.
|
||||
func (in ExtInstr) encodeAmdGprPair(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
vecPos := 1
|
||||
if in.Form == ExtFormAmdVecGpr {
|
||||
vecPos = 0
|
||||
}
|
||||
if in.Mem == 1 && in.Form == ExtFormAmdVecGpr && ops[0].Kind == ExtMem {
|
||||
if err := in.amd64Gpr(ops[1], 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1)
|
||||
}
|
||||
if err := in.amd64Vector(ops[vecPos], class, vecPos+1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -1022,16 +1046,16 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"},
|
||||
{Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"},
|
||||
{Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value",
|
||||
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2SS (EVEX.NDS.LIG.MAP6.W0 13 /r)"},
|
||||
{Name: "VCVTSH2SD", Summary: "Convert a low FP16 value to an FP64 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"},
|
||||
{Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"},
|
||||
{Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
@@ -1046,16 +1070,16 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"},
|
||||
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W0 2D /r)"},
|
||||
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 64-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W1 2D /r)"},
|
||||
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 32-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W0 79 /r)"},
|
||||
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 64-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W1 79 /r)"},
|
||||
|
||||
// AVX512-FP16 packed arithmetic: the full ZMM lanes the scalar core
|
||||
@@ -1148,4 +1172,145 @@ var amd64Extensions = []ExtInstr{
|
||||
{Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"},
|
||||
|
||||
// AVX512-FP16 packed conversions: the fourteen directions between the
|
||||
// FP16 lanes and the integer and double-precision companions, three
|
||||
// register widths each. L'L names the governing operand, whose class the
|
||||
// form spells: the source on the narrowing converts (VCVTDQ2PH narrows
|
||||
// its dword source to the half-width destination, VCVTQQ2PH and VCVTPD2PH
|
||||
// to the quarter-width XMM destination), the destination on the widening
|
||||
// ones (VCVTPH2DQ widens its half-width source, VCVTPH2QQ and VCVTPH2PD
|
||||
// their quarter-width XMM source). The FP16-to-integer directions round
|
||||
// and take Er on their 512-bit register forms; VCVTPH2PD widens exactly
|
||||
// and takes none. The integer-to-FP16 sources read their elements from
|
||||
// memory under the {1toN} broadcast, the FP16 sources their full-width
|
||||
// vectors; every destination is write-masked. The encodings are
|
||||
// transcribed from the SDM entries and pinned byte for byte against the
|
||||
// local GNU assembler, whose 2.46 table matches the manual row for row.
|
||||
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.512.66.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.256.66.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.128.66.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.512.NP.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.256.NP.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.128.NP.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.512.F3.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.256.F3.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.128.F3.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.512.F2.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.256.F2.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.128.F2.MAP5.W0 7D /r)"},
|
||||
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.512.66.MAP5.W0 5B /r, half-width source)"},
|
||||
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.256.66.MAP5.W0 5B /r, half-width source)"},
|
||||
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.128.66.MAP5.W0 5B /r, half-width source)"},
|
||||
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.512.NP.MAP5.W0 79 /r, half-width source)"},
|
||||
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.256.NP.MAP5.W0 79 /r, half-width source)"},
|
||||
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.128.NP.MAP5.W0 79 /r, half-width source)"},
|
||||
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.512.NP.MAP5.W0 5B /r, YMM destination)"},
|
||||
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.256.NP.MAP5.W0 5B /r, XMM destination)"},
|
||||
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.128.NP.MAP5.W0 5B /r, XMM destination)"},
|
||||
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.512.F2.MAP5.W0 7A /r, YMM destination)"},
|
||||
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.256.F2.MAP5.W0 7A /r, XMM destination)"},
|
||||
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.128.F2.MAP5.W0 7A /r, XMM destination)"},
|
||||
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.512.66.MAP5.W0 7B /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.256.66.MAP5.W0 7B /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.128.66.MAP5.W0 7B /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.512.66.MAP5.W0 79 /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.256.66.MAP5.W0 79 /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.128.66.MAP5.W0 79 /r, quarter-width source)"},
|
||||
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x84, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.512.NP.MAP5.W1 5B /r, XMM destination)"},
|
||||
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x84, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.256.NP.MAP5.W1 5B /r, XMM destination)"},
|
||||
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x84, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.128.NP.MAP5.W1 5B /r, XMM destination)"},
|
||||
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.512.F2.MAP5.W1 7A /r, XMM destination)"},
|
||||
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.256.F2.MAP5.W1 7A /r, XMM destination)"},
|
||||
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.128.F2.MAP5.W1 7A /r, XMM destination)"},
|
||||
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.512.NP.MAP5.W0 5A /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.256.NP.MAP5.W0 5A /r, quarter-width source)"},
|
||||
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.128.NP.MAP5.W0 5A /r, quarter-width source)"},
|
||||
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x85, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.512.66.MAP5.W1 5A /r, XMM destination)"},
|
||||
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x85, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.256.66.MAP5.W1 5A /r, XMM destination)"},
|
||||
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x85, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.128.66.MAP5.W1 5A /r, XMM destination)"},
|
||||
}
|
||||
+413
-2
@@ -660,6 +660,396 @@ var amd64GoldenRows = []amd64GoldenRow{
|
||||
{"vcvtusi2sh rd-sae, 64-bit, high registers", "VCVTUSI2SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtGpr64(12), ExtRounded(ExtXmm(30), ExtRoundDown)},
|
||||
"624596307bf4", "62 45 96 30 7b f4 vcvtusi2sh %r12,{rd-sae},%xmm29,%xmm30"},
|
||||
|
||||
// The FP16 packed conversions, the fourteen directions between the FP16
|
||||
// lanes and the integer and double-precision companions. The register
|
||||
// rows and the rounding rows are quoted from the local GNU assembler's
|
||||
// output; the memory rows follow the listing's full-width and broadcast
|
||||
// spellings, whose source displacements are N times the plain SDM
|
||||
// displacement the disp8 bytes carry under the assembler's compressed
|
||||
// displacement scaling, and the rows with no GNU line are derived: the
|
||||
// plain full-width load of a narrowing convert the AT&T syntax cannot
|
||||
// spell without a size hint, laid out by the broadcast row minus EVEX.b.
|
||||
{"vcvtph2w zmm", "VCVTPH2W",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
|
||||
"62f57d487df5", "62 f5 7d 48 7d f5 vcvtph2w %zmm5,%zmm6"},
|
||||
{"vcvtph2w ymm", "VCVTPH2W",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
|
||||
"62f57d287df5", "62 f5 7d 28 7d f5 vcvtph2w %ymm5,%ymm6"},
|
||||
{"vcvtph2w xmm", "VCVTPH2W",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57d087df5", "62 f5 7d 08 7d f5 vcvtph2w %xmm5,%xmm6"},
|
||||
{"vcvtph2uw zmm", "VCVTPH2UW",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
|
||||
"62f57c487df5", "62 f5 7c 48 7d f5 vcvtph2uw %zmm5,%zmm6"},
|
||||
{"vcvtph2uw ymm", "VCVTPH2UW",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
|
||||
"62f57c287df5", "62 f5 7c 28 7d f5 vcvtph2uw %ymm5,%ymm6"},
|
||||
{"vcvtph2uw xmm", "VCVTPH2UW",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57c087df5", "62 f5 7c 08 7d f5 vcvtph2uw %xmm5,%xmm6"},
|
||||
{"vcvtw2ph zmm", "VCVTW2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
|
||||
"62f57e487df5", "62 f5 7e 48 7d f5 vcvtw2ph %zmm5,%zmm6"},
|
||||
{"vcvtw2ph ymm", "VCVTW2PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
|
||||
"62f57e287df5", "62 f5 7e 28 7d f5 vcvtw2ph %ymm5,%ymm6"},
|
||||
{"vcvtw2ph xmm", "VCVTW2PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57e087df5", "62 f5 7e 08 7d f5 vcvtw2ph %xmm5,%xmm6"},
|
||||
{"vcvtuw2ph zmm", "VCVTUW2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
|
||||
"62f57f487df5", "62 f5 7f 48 7d f5 vcvtuw2ph %zmm5,%zmm6"},
|
||||
{"vcvtuw2ph ymm", "VCVTUW2PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
|
||||
"62f57f287df5", "62 f5 7f 28 7d f5 vcvtuw2ph %ymm5,%ymm6"},
|
||||
{"vcvtuw2ph xmm", "VCVTUW2PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57f087df5", "62 f5 7f 08 7d f5 vcvtuw2ph %xmm5,%xmm6"},
|
||||
{"vcvtph2dq zmm", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtYmm(5), ExtZmm(6)},
|
||||
"62f57d485bf5", "62 f5 7d 48 5b f5 vcvtph2dq %ymm5,%zmm6"},
|
||||
{"vcvtph2dq ymm", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
|
||||
"62f57d285bf5", "62 f5 7d 28 5b f5 vcvtph2dq %xmm5,%ymm6"},
|
||||
{"vcvtph2dq xmm", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57d085bf5", "62 f5 7d 08 5b f5 vcvtph2dq %xmm5,%xmm6"},
|
||||
{"vcvtph2udq zmm", "VCVTPH2UDQ",
|
||||
[]ExtOperand{ExtYmm(5), ExtZmm(6)},
|
||||
"62f57c4879f5", "62 f5 7c 48 79 f5 vcvtph2udq %ymm5,%zmm6"},
|
||||
{"vcvtph2udq ymm", "VCVTPH2UDQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
|
||||
"62f57c2879f5", "62 f5 7c 28 79 f5 vcvtph2udq %xmm5,%ymm6"},
|
||||
{"vcvtph2udq xmm", "VCVTPH2UDQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57c0879f5", "62 f5 7c 08 79 f5 vcvtph2udq %xmm5,%xmm6"},
|
||||
{"vcvtdq2ph zmm to ymm", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtYmm(6)},
|
||||
"62f57c485bf5", "62 f5 7c 48 5b f5 vcvtdq2ph %zmm5,%ymm6"},
|
||||
{"vcvtdq2ph ymm to xmm", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
|
||||
"62f57c285bf5", "62 f5 7c 28 5b f5 vcvtdq2ph %ymm5,%xmm6"},
|
||||
{"vcvtdq2ph xmm to xmm", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57c085bf5", "62 f5 7c 08 5b f5 vcvtdq2ph %xmm5,%xmm6"},
|
||||
{"vcvtudq2ph zmm to ymm", "VCVTUDQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtYmm(6)},
|
||||
"62f57f487af5", "62 f5 7f 48 7a f5 vcvtudq2ph %zmm5,%ymm6"},
|
||||
{"vcvtudq2ph ymm to xmm", "VCVTUDQ2PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
|
||||
"62f57f287af5", "62 f5 7f 28 7a f5 vcvtudq2ph %ymm5,%xmm6"},
|
||||
{"vcvtudq2ph xmm to xmm", "VCVTUDQ2PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57f087af5", "62 f5 7f 08 7a f5 vcvtudq2ph %xmm5,%xmm6"},
|
||||
{"vcvtph2qq zmm", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtZmm(6)},
|
||||
"62f57d487bf5", "62 f5 7d 48 7b f5 vcvtph2qq %xmm5,%zmm6"},
|
||||
{"vcvtph2qq ymm", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
|
||||
"62f57d287bf5", "62 f5 7d 28 7b f5 vcvtph2qq %xmm5,%ymm6"},
|
||||
{"vcvtph2qq xmm", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57d087bf5", "62 f5 7d 08 7b f5 vcvtph2qq %xmm5,%xmm6"},
|
||||
{"vcvtph2uqq zmm", "VCVTPH2UQQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtZmm(6)},
|
||||
"62f57d4879f5", "62 f5 7d 48 79 f5 vcvtph2uqq %xmm5,%zmm6"},
|
||||
{"vcvtph2uqq ymm", "VCVTPH2UQQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
|
||||
"62f57d2879f5", "62 f5 7d 28 79 f5 vcvtph2uqq %xmm5,%ymm6"},
|
||||
{"vcvtph2uqq xmm", "VCVTPH2UQQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57d0879f5", "62 f5 7d 08 79 f5 vcvtph2uqq %xmm5,%xmm6"},
|
||||
{"vcvtqq2ph zmm to xmm", "VCVTQQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtXmm(6)},
|
||||
"62f5fc485bf5", "62 f5 fc 48 5b f5 vcvtqq2ph %zmm5,%xmm6"},
|
||||
{"vcvtqq2ph ymm to xmm", "VCVTQQ2PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
|
||||
"62f5fc285bf5", "62 f5 fc 28 5b f5 vcvtqq2ph %ymm5,%xmm6"},
|
||||
{"vcvtqq2ph xmm to xmm", "VCVTQQ2PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f5fc085bf5", "62 f5 fc 08 5b f5 vcvtqq2ph %xmm5,%xmm6"},
|
||||
{"vcvtuqq2ph zmm to xmm", "VCVTUQQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtXmm(6)},
|
||||
"62f5ff487af5", "62 f5 ff 48 7a f5 vcvtuqq2ph %zmm5,%xmm6"},
|
||||
{"vcvtuqq2ph ymm to xmm", "VCVTUQQ2PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
|
||||
"62f5ff287af5", "62 f5 ff 28 7a f5 vcvtuqq2ph %ymm5,%xmm6"},
|
||||
{"vcvtuqq2ph xmm to xmm", "VCVTUQQ2PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f5ff087af5", "62 f5 ff 08 7a f5 vcvtuqq2ph %xmm5,%xmm6"},
|
||||
{"vcvtph2pd zmm", "VCVTPH2PD",
|
||||
[]ExtOperand{ExtXmm(5), ExtZmm(6)},
|
||||
"62f57c485af5", "62 f5 7c 48 5a f5 vcvtph2pd %xmm5,%zmm6"},
|
||||
{"vcvtph2pd ymm", "VCVTPH2PD",
|
||||
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
|
||||
"62f57c285af5", "62 f5 7c 28 5a f5 vcvtph2pd %xmm5,%ymm6"},
|
||||
{"vcvtph2pd xmm", "VCVTPH2PD",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f57c085af5", "62 f5 7c 08 5a f5 vcvtph2pd %xmm5,%xmm6"},
|
||||
{"vcvtpd2ph zmm to xmm", "VCVTPD2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtXmm(6)},
|
||||
"62f5fd485af5", "62 f5 fd 48 5a f5 vcvtpd2ph %zmm5,%xmm6"},
|
||||
{"vcvtpd2ph ymm to xmm", "VCVTPD2PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
|
||||
"62f5fd285af5", "62 f5 fd 28 5a f5 vcvtpd2ph %ymm5,%xmm6"},
|
||||
{"vcvtpd2ph xmm to xmm", "VCVTPD2PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
||||
"62f5fd085af5", "62 f5 fd 08 5a f5 vcvtpd2ph %xmm5,%xmm6"},
|
||||
|
||||
// The rounding rows of the converting directions: the FP16-to-integer
|
||||
// and the integer-to-FP16 conversions round and take EVEX.RC on their
|
||||
// 512-bit register forms, exactly as the arithmetic does, and VCVTPH2PD,
|
||||
// which widens exactly, takes none.
|
||||
{"vcvtph2w rn-sae", "VCVTPH2W",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
|
||||
"62f57d187df5", "62 f5 7d 18 7d f5 vcvtph2w {rn-sae},%zmm5,%zmm6"},
|
||||
{"vcvtph2uw rz-sae", "VCVTPH2UW",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)},
|
||||
"62f57c787df5", "62 f5 7c 78 7d f5 vcvtph2uw {rz-sae},%zmm5,%zmm6"},
|
||||
{"vcvtph2dq rn-sae", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtYmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
|
||||
"62f57d185bf5", "62 f5 7d 18 5b f5 vcvtph2dq {rn-sae},%ymm5,%zmm6"},
|
||||
{"vcvtph2udq rn-sae", "VCVTPH2UDQ",
|
||||
[]ExtOperand{ExtYmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
|
||||
"62f57c1879f5", "62 f5 7c 18 79 f5 vcvtph2udq {rn-sae},%ymm5,%zmm6"},
|
||||
{"vcvtph2qq rz-sae", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)},
|
||||
"62f57d787bf5", "62 f5 7d 78 7b f5 vcvtph2qq {rz-sae},%xmm5,%zmm6"},
|
||||
{"vcvtph2uqq rz-sae", "VCVTPH2UQQ",
|
||||
[]ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)},
|
||||
"62f57d7879f5", "62 f5 7d 78 79 f5 vcvtph2uqq {rz-sae},%xmm5,%zmm6"},
|
||||
{"vcvtw2ph rz-sae", "VCVTW2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)},
|
||||
"62f57e787df5", "62 f5 7e 78 7d f5 vcvtw2ph {rz-sae},%zmm5,%zmm6"},
|
||||
{"vcvtuw2ph rn-sae", "VCVTUW2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
|
||||
"62f57f187df5", "62 f5 7f 18 7d f5 vcvtuw2ph {rn-sae},%zmm5,%zmm6"},
|
||||
{"vcvtdq2ph rn-sae", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtYmm(6), ExtRoundNearest)},
|
||||
"62f57c185bf5", "62 f5 7c 18 5b f5 vcvtdq2ph {rn-sae},%zmm5,%ymm6"},
|
||||
{"vcvtudq2ph ru-sae", "VCVTUDQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtYmm(6), ExtRoundUp)},
|
||||
"62f57f587af5", "62 f5 7f 58 7a f5 vcvtudq2ph {ru-sae},%zmm5,%ymm6"},
|
||||
{"vcvtqq2ph rz-sae", "VCVTQQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundTruncate)},
|
||||
"62f5fc785bf5", "62 f5 fc 78 5b f5 vcvtqq2ph {rz-sae},%zmm5,%xmm6"},
|
||||
{"vcvtuqq2ph rd-sae", "VCVTUQQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundDown)},
|
||||
"62f5ff387af5", "62 f5 ff 38 7a f5 vcvtuqq2ph {rd-sae},%zmm5,%xmm6"},
|
||||
{"vcvtpd2ph rz-sae", "VCVTPD2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundTruncate)},
|
||||
"62f5fd785af5", "62 f5 fd 78 5a f5 vcvtpd2ph {rz-sae},%zmm5,%xmm6"},
|
||||
|
||||
// High registers across the conversion shapes, exercising every EVEX
|
||||
// extension bit: the source above 15 clears B bar and X bar, the
|
||||
// destination above 15 R bar.
|
||||
{"vcvtph2w, high registers", "VCVTPH2W",
|
||||
[]ExtOperand{ExtZmm(28), ExtZmm(29)},
|
||||
"62057d487dec", "62 05 7d 48 7d ec vcvtph2w %zmm28,%zmm29"},
|
||||
{"vcvtph2dq, high registers", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtYmm(28), ExtZmm(29)},
|
||||
"62057d485bec", "62 05 7d 48 5b ec vcvtph2dq %ymm28,%zmm29"},
|
||||
{"vcvtdq2ph, high registers", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtYmm(30)},
|
||||
"62057c485bf5", "62 05 7c 48 5b f5 vcvtdq2ph %zmm29,%ymm30"},
|
||||
{"vcvtph2qq, high registers", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtXmm(28), ExtZmm(29)},
|
||||
"62057d487bec", "62 05 7d 48 7b ec vcvtph2qq %xmm28,%zmm29"},
|
||||
{"vcvtpd2ph, high registers", "VCVTPD2PH",
|
||||
[]ExtOperand{ExtZmm(28), ExtXmm(29)},
|
||||
"6205fd485aec", "62 05 fd 48 5a ec vcvtpd2ph %zmm28,%xmm29"},
|
||||
|
||||
// The memory forms of the conversions: the FP16 sources read full-width
|
||||
// vectors, the integer sources their elements under the broadcast. The
|
||||
// disp8 rows quote the listing's compressed spellings, whose source is N
|
||||
// times the plain displacement the bytes carry.
|
||||
{"vcvtph2w memory source", "VCVTPH2W",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62457d487d31", "62 45 7d 48 7d 31 vcvtph2w (%r9),%zmm30"},
|
||||
{"vcvtph2w memory source disp8", "VCVTPH2W",
|
||||
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
|
||||
"62657d487d717f", "62 65 7d 48 7d 71 7f vcvtph2w 0x1fc0(%rcx),%zmm30 (Disp8(7f))"},
|
||||
{"vcvtph2w ymm memory source", "VCVTPH2W",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
|
||||
"62457d287d31", "62 45 7d 28 7d 31 vcvtph2w (%r9),%ymm30"},
|
||||
{"vcvtph2w xmm memory source", "VCVTPH2W",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
|
||||
"62457d087d31", "62 45 7d 08 7d 31 vcvtph2w (%r9),%xmm30"},
|
||||
{"vcvtph2uw memory source", "VCVTPH2UW",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62457c487d31", "62 45 7c 48 7d 31 vcvtph2uw (%r9),%zmm30"},
|
||||
{"vcvtw2ph memory source", "VCVTW2PH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62457e487d31", "62 45 7e 48 7d 31 vcvtw2ph (%r9),%zmm30"},
|
||||
{"vcvtw2ph ymm memory source", "VCVTW2PH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
|
||||
"62457e287d31", "62 45 7e 28 7d 31 vcvtw2ph (%r9),%ymm30"},
|
||||
{"vcvtph2dq memory source", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62457d485b31", "62 45 7d 48 5b 31 vcvtph2dq (%r9),%zmm30"},
|
||||
{"vcvtph2dq memory source disp8", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
|
||||
"62657d485b717f", "62 65 7d 48 5b 71 7f vcvtph2dq 0xfe0(%rcx),%zmm30 (Disp8(7f))"},
|
||||
{"vcvtph2dq ymm memory source", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
|
||||
"62457d285b31", "62 45 7d 28 5b 31 vcvtph2dq (%r9),%ymm30"},
|
||||
{"vcvtph2udq memory source", "VCVTPH2UDQ",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62457c487931", "62 45 7c 48 79 31 vcvtph2udq (%r9),%zmm30"},
|
||||
{"vcvtdq2ph memory source", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
|
||||
"62457c485b31", "62 45 7c 48 5b 31 vcvtdq2ph (%r9),%ymm30"},
|
||||
{"vcvtph2qq memory source", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62457d487b31", "62 45 7d 48 7b 31 vcvtph2qq (%r9),%zmm30"},
|
||||
{"vcvtph2qq memory source disp8", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
|
||||
"62657d487b717f", "62 65 7d 48 7b 71 7f vcvtph2qq 0x7f0(%rcx),%zmm30 (Disp8(7f))"},
|
||||
{"vcvtph2qq ymm memory source", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
|
||||
"62457d287b31", "62 45 7d 28 7b 31 vcvtph2qq (%r9),%ymm30"},
|
||||
{"vcvtph2uqq memory source", "VCVTPH2UQQ",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62457d487931", "62 45 7d 48 79 31 vcvtph2uqq (%r9),%zmm30"},
|
||||
{"vcvtph2pd memory source", "VCVTPH2PD",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62457c485a31", "62 45 7c 48 5a 31 vcvtph2pd (%r9),%zmm30"},
|
||||
{"vcvtph2pd memory source disp8", "VCVTPH2PD",
|
||||
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
|
||||
"62657c485a717f", "62 65 7c 48 5a 71 7f vcvtph2pd 0x7f0(%rcx),%zmm30 (Disp8(7f))"},
|
||||
{"vcvtph2pd ymm memory source", "VCVTPH2PD",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
|
||||
"62457c285a31", "62 45 7c 28 5a 31 vcvtph2pd (%r9),%ymm30"},
|
||||
{"vcvtqq2ph plain memory source", "VCVTQQ2PH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
|
||||
"6245fc485b31", ""},
|
||||
{"vcvtuqq2ph plain memory source", "VCVTUQQ2PH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
|
||||
"6245ff487a31", ""},
|
||||
{"vcvtpd2ph plain memory source", "VCVTPD2PH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
|
||||
"6245fd485a31", ""},
|
||||
|
||||
// The broadcast forms of the converting sources: a single element the
|
||||
// hardware splats across the lanes, one word, dword or qword as the
|
||||
// direction's input element spells it.
|
||||
{"vcvtw2ph broadcast source", "VCVTW2PH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"62457e587d31", "62 45 7e 58 7d 31 vcvtw2ph (%r9){1to32},%zmm30"},
|
||||
{"vcvtw2ph broadcast source disp8", "VCVTW2PH",
|
||||
[]ExtOperand{ExtBroadcast(1, 127), ExtZmm(30)},
|
||||
"62657e587d717f", "62 65 7e 58 7d 71 7f vcvtw2ph 0xfe(%rcx){1to32},%zmm30 (Disp8(7f))"},
|
||||
{"vcvtuw2ph broadcast source", "VCVTUW2PH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"62457f587d31", "62 45 7f 58 7d 31 vcvtuw2ph (%r9){1to32},%zmm30"},
|
||||
{"vcvtuw2ph broadcast source disp8", "VCVTUW2PH",
|
||||
[]ExtOperand{ExtBroadcast(1, 127), ExtZmm(30)},
|
||||
"62657f587d717f", "62 65 7f 58 7d 71 7f vcvtuw2ph 0xfe(%rcx){1to32},%zmm30 (Disp8(7f))"},
|
||||
{"vcvtdq2ph broadcast source", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtYmm(30)},
|
||||
"62457c585b31", "62 45 7c 58 5b 31 vcvtdq2ph (%r9){1to16},%ymm30"},
|
||||
{"vcvtdq2ph broadcast source disp8", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtBroadcast(1, 127), ExtYmm(30)},
|
||||
"62657c585b717f", "62 65 7c 58 5b 71 7f vcvtdq2ph 0x1fc(%rcx){1to16},%ymm30 (Disp8(7f))"},
|
||||
{"vcvtudq2ph broadcast source", "VCVTUDQ2PH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtYmm(30)},
|
||||
"62457f587a31", "62 45 7f 58 7a 31 vcvtudq2ph (%r9){1to16},%ymm30"},
|
||||
{"vcvtudq2ph broadcast source disp8", "VCVTUDQ2PH",
|
||||
[]ExtOperand{ExtBroadcast(1, 127), ExtYmm(30)},
|
||||
"62657f587a717f", "62 65 7f 58 7a 71 7f vcvtudq2ph 0x1fc(%rcx){1to16},%ymm30 (Disp8(7f))"},
|
||||
{"vcvtqq2ph broadcast source", "VCVTQQ2PH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
|
||||
"6245fc585b31", "62 45 fc 58 5b 31 vcvtqq2ph (%r9){1to8},%xmm30"},
|
||||
{"vcvtqq2ph broadcast source disp8", "VCVTQQ2PH",
|
||||
[]ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)},
|
||||
"6265fc585b717f", "62 65 fc 58 5b 71 7f vcvtqq2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"},
|
||||
// The 256-bit broadcast row shares the 512-bit row's operand list, both
|
||||
// destinations being XMM, so the resolver cannot tell them apart; the
|
||||
// register row above pins the 256-bit template alone.
|
||||
{"vcvtuqq2ph broadcast source", "VCVTUQQ2PH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
|
||||
"6245ff587a31", "62 45 ff 58 7a 31 vcvtuqq2ph (%r9){1to8},%xmm30"},
|
||||
{"vcvtuqq2ph broadcast source disp8", "VCVTUQQ2PH",
|
||||
[]ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)},
|
||||
"6265ff587a717f", "62 65 ff 58 7a 71 7f vcvtuqq2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"},
|
||||
{"vcvtpd2ph broadcast source", "VCVTPD2PH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
|
||||
"6245fd585a31", "62 45 fd 58 5a 31 vcvtpd2ph (%r9){1to8},%xmm30"},
|
||||
{"vcvtpd2ph broadcast source disp8", "VCVTPD2PH",
|
||||
[]ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)},
|
||||
"6265fd585a717f", "62 65 fd 58 5a 71 7f vcvtpd2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"},
|
||||
// The 256-bit broadcast row shares the 512-bit row's operand list, both
|
||||
// destinations being XMM, so the resolver cannot tell them apart; the
|
||||
// register row above pins the 256-bit template alone.
|
||||
|
||||
// The write mask over the converting destinations, and one composed
|
||||
// row: the broadcast, the mask and the zeroing on the one word.
|
||||
{"vcvtph2w k7 zeroing", "VCVTPH2W",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)},
|
||||
"62457dcf7d31", "62 45 7d cf 7d 31 vcvtph2w (%r9),%zmm30{%k7}{z}"},
|
||||
{"vcvtph2dq k7 zeroing, memory source", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)},
|
||||
"62457dcf5b31", "62 45 7d cf 5b 31 vcvtph2dq (%r9),%zmm30{%k7}{z}"},
|
||||
{"vcvtph2qq k7, memory source", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, false)},
|
||||
"62457d4f7b31", "62 45 7d 4f 7b 31 vcvtph2qq (%r9),%zmm30{%k7}"},
|
||||
{"vcvtph2pd k7, memory source", "VCVTPH2PD",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, false)},
|
||||
"62457c4f5a31", "62 45 7c 4f 5a 31 vcvtph2pd (%r9),%zmm30{%k7}"},
|
||||
{"vcvtw2ph k7 zeroing", "VCVTW2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtWriteMasked(ExtZmm(6), 7, true)},
|
||||
"62f57ecf7df5", "62 f5 7e cf 7d f5 vcvtw2ph %zmm5,%zmm6{%k7}{z}"},
|
||||
{"vcvtdq2ph k7 zeroing", "VCVTDQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtWriteMasked(ExtYmm(6), 7, true)},
|
||||
"62f57ccf5bf5", "62 f5 7c cf 5b f5 vcvtdq2ph %zmm5,%ymm6{%k7}{z}"},
|
||||
{"vcvtw2ph broadcast under k7 zeroing", "VCVTW2PH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)},
|
||||
"62457edf7d31", "62 45 7e df 7d 31 vcvtw2ph (%r9){1to32},%zmm30{%k7}{z}"},
|
||||
|
||||
// The memory forms of the scalar conversions, the m16, m32 and m64
|
||||
// sources the manual spells beside each register form. The disp8 rows
|
||||
// quote the listing's compressed spellings, whose source is N times the
|
||||
// plain displacement, N the element size.
|
||||
{"vcvtsh2ss memory source", "VCVTSH2SS",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)},
|
||||
"62d654081331", "62 d6 54 08 13 31 vcvtsh2ss (%r9),%xmm5,%xmm6"},
|
||||
{"vcvtsh2ss memory source disp8", "VCVTSH2SS",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)},
|
||||
"62f65408137101", "62 f6 54 08 13 71 01 vcvtsh2ss 0x2(%rcx),%xmm5,%xmm6 (Disp8(01))"},
|
||||
{"vcvtsh2sd memory source", "VCVTSH2SD",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)},
|
||||
"62d556085a31", "62 d5 56 08 5a 31 vcvtsh2sd (%r9),%xmm5,%xmm6"},
|
||||
{"vcvtsh2sd memory source disp8", "VCVTSH2SD",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)},
|
||||
"62f556085a7101", "62 f5 56 08 5a 71 01 vcvtsh2sd 0x2(%rcx),%xmm5,%xmm6 (Disp8(01))"},
|
||||
{"vcvtss2sh memory source", "VCVTSS2SH",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)},
|
||||
"62d554081d31", "62 d5 54 08 1d 31 vcvtss2sh (%r9),%xmm5,%xmm6"},
|
||||
{"vcvtss2sh memory source disp8", "VCVTSS2SH",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)},
|
||||
"62f554081d7101", "62 f5 54 08 1d 71 01 vcvtss2sh 0x4(%rcx),%xmm5,%xmm6 (Disp8(01))"},
|
||||
{"vcvtsd2sh memory source", "VCVTSD2SH",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)},
|
||||
"62d5d7085a31", "62 d5 d7 08 5a 31 vcvtsd2sh (%r9),%xmm5,%xmm6"},
|
||||
{"vcvtsd2sh memory source disp8", "VCVTSD2SH",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)},
|
||||
"62f5d7085a7101", "62 f5 d7 08 5a 71 01 vcvtsd2sh 0x8(%rcx),%xmm5,%xmm6 (Disp8(01))"},
|
||||
{"vcvtsh2si memory source", "VCVTSH2SI",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtGpr32(0)},
|
||||
"62d57e082d01", "62 d5 7e 08 2d 01 vcvtsh2si (%r9),%eax"},
|
||||
{"vcvtsh2si 64-bit memory source", "VCVTSH2SI",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtGpr64(12)},
|
||||
"6255fe082d21", "62 55 fe 08 2d 21 vcvtsh2si (%r9),%r12"},
|
||||
{"vcvtsh2si memory source disp8", "VCVTSH2SI",
|
||||
[]ExtOperand{ExtMemory(1, 1), ExtGpr32(0)},
|
||||
"62f57e082d4101", "62 f5 7e 08 2d 41 01 vcvtsh2si 0x2(%rcx),%eax (Disp8(01))"},
|
||||
{"vcvtsh2usi memory source", "VCVTSH2USI",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtGpr32(0)},
|
||||
"62d57e087901", "62 d5 7e 08 79 01 vcvtsh2usi (%r9),%eax"},
|
||||
{"vcvtsh2usi 64-bit memory source disp8", "VCVTSH2USI",
|
||||
[]ExtOperand{ExtMemory(1, 1), ExtGpr64(12)},
|
||||
"6275fe08796101", "62 75 fe 08 79 61 01 vcvtsh2usi 0x2(%rcx),%r12 (Disp8(01))"},
|
||||
}
|
||||
|
||||
// amd64ResolveEntry finds the table entry a golden row exercises: the entry
|
||||
@@ -951,6 +1341,12 @@ func TestAmd64ExtRejects(t *testing.T) {
|
||||
{"rounding over a memory source", "VADDSH",
|
||||
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtRounded(ExtXmm(30), ExtRoundNearest)},
|
||||
"the memory form takes no rounding control"},
|
||||
{"rounding over the scalar convert's memory source", "VCVTSS2SH",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtRounded(ExtXmm(6), ExtRoundNearest)},
|
||||
"the memory form takes no rounding control"},
|
||||
{"rounding beside the converting memory source", "VCVTW2PH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtRounded(ExtZmm(30), ExtRoundNearest)},
|
||||
"the memory form takes no rounding control"},
|
||||
{"rounding on a source position", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(4), ExtRoundUp), ExtZmm(6)},
|
||||
"the position takes none"},
|
||||
@@ -963,6 +1359,21 @@ func TestAmd64ExtRejects(t *testing.T) {
|
||||
{"rounding on an entry without the capability", "VMOVSH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundNearest)},
|
||||
"the entry's destination takes none"},
|
||||
{"rounding on the exactly-widening convert", "VCVTPH2PD",
|
||||
[]ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
|
||||
"the entry's destination takes none"},
|
||||
{"wrong source class on the widening convert", "VCVTPH2DQ",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
|
||||
"wants a YMM register"},
|
||||
{"wrong source class on the quarter convert", "VCVTPH2QQ",
|
||||
[]ExtOperand{ExtYmm(5), ExtZmm(6)},
|
||||
"wants an XMM register"},
|
||||
{"wrong destination on the quarter-destination convert", "VCVTQQ2PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtYmm(6)},
|
||||
"wants an XMM register"},
|
||||
{"broadcast on the full-width FP16 source", "VCVTPH2W",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"the entry's memory operand takes none"},
|
||||
} {
|
||||
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
|
||||
_, err := in.Encode(tt.ops)
|
||||
@@ -1107,7 +1518,7 @@ func TestAmd64ExtArchBinding(t *testing.T) {
|
||||
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
|
||||
}
|
||||
}
|
||||
if got := Extensions(AMD64); len(got) != 70 {
|
||||
t.Errorf("the amd64 layer registers %d instructions, want 70", len(got))
|
||||
if got := Extensions(AMD64); len(got) != 112 {
|
||||
t.Errorf("the amd64 layer registers %d instructions, want 112", len(got))
|
||||
}
|
||||
}
|
||||
+22
-2
@@ -444,6 +444,18 @@ const (
|
||||
// half-width companion of the source, equal at 128 bits: VCVTNEPS2BF16
|
||||
// Y6, Z5. Operands: src, dest.
|
||||
ExtFormAmdVec2Half
|
||||
// ExtFormAmdVec2Wide is the two-vector form whose source is the half-width
|
||||
// companion of the destination, equal at 128 bits: VCVTPH2DQ Z6, Y5, where
|
||||
// L'L names the destination. Operands: src, dest.
|
||||
ExtFormAmdVec2Wide
|
||||
// ExtFormAmdVec2Quarter is the two-vector form whose source is the
|
||||
// quarter-width companion of the destination, the XMM class at every
|
||||
// length: VCVTPH2QQ Z6, X5. Operands: src, dest.
|
||||
ExtFormAmdVec2Quarter
|
||||
// ExtFormAmdVec2ToQuarter is the two-vector form whose destination is the
|
||||
// quarter-width companion of the source, the XMM class at every length:
|
||||
// VCVTQQ2PH X6, Z5. Operands: src, dest.
|
||||
ExtFormAmdVec2ToQuarter
|
||||
// ExtFormAmdMask2 is the form with an opmask destination and two vector
|
||||
// sources: VP2INTERSECTD K0, Z2, Z1. Operands: src1, src2, dest.
|
||||
ExtFormAmdMask2
|
||||
@@ -668,7 +680,8 @@ func (f ExtForm) Arity() int {
|
||||
return 2
|
||||
case ExtFormAmdVec3, ExtFormAmdMask2, ExtFormAmdVecGprVec:
|
||||
return 3
|
||||
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdGprVec, ExtFormAmdVecGpr:
|
||||
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter,
|
||||
ExtFormAmdVec2ToQuarter, ExtFormAmdGprVec, ExtFormAmdVecGpr:
|
||||
return 2
|
||||
case ExtFormAmdVec3Imm, ExtFormAmdMask2Imm:
|
||||
return 4
|
||||
@@ -777,6 +790,12 @@ func (f ExtForm) String() string {
|
||||
return "two vectors"
|
||||
case ExtFormAmdVec2Half:
|
||||
return "two vectors, half-width destination"
|
||||
case ExtFormAmdVec2Wide:
|
||||
return "two vectors, half-width source"
|
||||
case ExtFormAmdVec2Quarter:
|
||||
return "two vectors, quarter-width source"
|
||||
case ExtFormAmdVec2ToQuarter:
|
||||
return "two vectors, quarter-width destination"
|
||||
case ExtFormAmdMask2:
|
||||
return "two vectors into an opmask"
|
||||
case ExtFormAmdVecGprVec:
|
||||
@@ -1037,7 +1056,8 @@ func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) {
|
||||
return in.encodeSignedImmediate(ops)
|
||||
case ExtFormAmdVec3, ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdMask2,
|
||||
ExtFormAmdVecGprVec, ExtFormAmdGprVec, ExtFormAmdVecGpr,
|
||||
ExtFormAmdVec3Imm, ExtFormAmdMask2Imm, ExtFormAmdMemVec, ExtFormAmdVecMem:
|
||||
ExtFormAmdVec3Imm, ExtFormAmdMask2Imm, ExtFormAmdMemVec, ExtFormAmdVecMem,
|
||||
ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter, ExtFormAmdVec2ToQuarter:
|
||||
return in.encodeAmd64(ops)
|
||||
case ExtFormPredicateLogical, ExtFormPredicateSelect:
|
||||
return in.encodePredicateLogical(ops)
|
||||
|
||||
@@ -41,6 +41,20 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
|
||||
{"VGETMANTSH", 1},
|
||||
{"VREDUCESH", 1},
|
||||
{"VRNDSCALESH", 1},
|
||||
{"VCVTPH2W", 3},
|
||||
{"VCVTPH2UW", 3},
|
||||
{"VCVTW2PH", 3},
|
||||
{"VCVTUW2PH", 3},
|
||||
{"VCVTPH2DQ", 3},
|
||||
{"VCVTPH2UDQ", 3},
|
||||
{"VCVTDQ2PH", 3},
|
||||
{"VCVTUDQ2PH", 3},
|
||||
{"VCVTPH2QQ", 3},
|
||||
{"VCVTPH2UQQ", 3},
|
||||
{"VCVTQQ2PH", 3},
|
||||
{"VCVTUQQ2PH", 3},
|
||||
{"VCVTPH2PD", 3},
|
||||
{"VCVTPD2PH", 3},
|
||||
} {
|
||||
cands, ok := LookupExtension(arch.AMD64, tt.mnem)
|
||||
if !ok {
|
||||
@@ -54,8 +68,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
|
||||
t.Errorf("the %s lookup is not case-insensitive", tt.mnem)
|
||||
}
|
||||
}
|
||||
if got := arch.Extensions(arch.AMD64); len(got) != 70 {
|
||||
t.Errorf("the amd64 layer registers %d instructions, want 70", len(got))
|
||||
if got := arch.Extensions(arch.AMD64); len(got) != 112 {
|
||||
t.Errorf("the amd64 layer registers %d instructions, want 112", len(got))
|
||||
}
|
||||
if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok {
|
||||
t.Error("a non-extended mnemonic resolved")
|
||||
@@ -131,6 +145,21 @@ func TestEncodeExtensionAmd64(t *testing.T) {
|
||||
{"scalar fp16 minimum with exceptions suppressed", "VMINSH",
|
||||
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundSAE)},
|
||||
"62f556185df4"},
|
||||
{"fp16 to signed words", "VCVTPH2W",
|
||||
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(6)},
|
||||
"62f57d487df5"},
|
||||
{"dwords to fp16 over a broadcast source", "VCVTDQ2PH",
|
||||
[]arch.ExtOperand{arch.ExtBroadcast(9, 0), arch.ExtYmm(30)},
|
||||
"62457c585b31"},
|
||||
{"fp16 to signed qwords, rounded", "VCVTPH2QQ",
|
||||
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundTruncate)},
|
||||
"62f57d787bf5"},
|
||||
{"fp16 scalar out of memory into an integer", "VCVTSH2SI",
|
||||
[]arch.ExtOperand{arch.ExtMemory(9, 0), arch.ExtGpr64(12)},
|
||||
"6255fe082d21"},
|
||||
{"fp16 to double-precision, widened", "VCVTPH2PD",
|
||||
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtZmm(6)},
|
||||
"62f57c485af5"},
|
||||
} {
|
||||
got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...)
|
||||
if err != nil {
|
||||
|
||||
Reference in new issue
Block a user