feat(arch): add the amd64 fp16 packed conversion family

Assisted-by: GLM 5.3
This commit is contained in:
petrbalvin committed 2026-10-07 13:51:48 +02:00
1 parent fb6d01a7d0
commit 8de1b371da
4 files changed
+659 -34

No files matched your search

+193 -28
View File
@@ -10,14 +10,17 @@
//
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
// latter's scalar core with its imm8-control group, its packed 512-bit and
// VL arithmetic and the embedded rounding of its FP operations, in their
// EVEX register forms. The encodings are transcribed from the SDM
// instruction entries and cross-checked against binutils-gdb's assembler
// testsuite; the golden vectors in amd64_ext_test.go pin the bytes.
// VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, no
// longer belong here: the Go toolchain's assembler knows them today, they
// live in the generated table and the EVEX encoder, and a mnemonic the
// toolchain has is not an extension.
// VL arithmetic, the embedded rounding of its FP operations and its fourteen
// packed conversion directions, in their EVEX register forms. The encodings
// are transcribed from the SDM instruction entries and cross-checked against
// binutils-gdb's assembler testsuite; the golden vectors in amd64_ext_test.go
// pin the bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19
// survey, no longer belong here: the Go toolchain's assembler knows them
// today, they live in the generated table and the EVEX encoder, and a
// mnemonic the toolchain has is not an extension. VCVTPS2PH, VCVTUDQ2PS
// and the rest of the classic conversion set follow the same rule, which is
// why the conversions here are the FP16 directions the toolchain has never
// emitted.
//
// The forms encode the unmasked shapes: register forms throughout, and the
// memory forms beside them, the scalar ones the manual spells m16, m32 and
@@ -462,7 +465,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
switch in.Form {
case ExtFormAmdVec3:
return in.encodeAmdVec3(ops)
case ExtFormAmdVec2, ExtFormAmdVec2Half:
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter,
ExtFormAmdVec2ToQuarter:
return in.encodeAmdVec2(ops)
case ExtFormAmdMask2:
return in.encodeAmdMask2(ops)
@@ -560,18 +564,30 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
// of destination, and at 128 bits the companion stays the class itself. An
// entry with Mem set takes the memory shape of the source position too: the
// compares and the packed square root read their source from memory, the
// packed square root's source carrying the {1toN} broadcast as EVEX.b, and
// the narrow BF16 convert reads its full-width source there. An entry with
// Er or Sae set takes the rounding decoration on its register form alone;
// the memory shape refuses one.
// of destination, and at 128 bits the companion stays the class itself. The
// wide form narrows the source instead, the widening FP16 conversions, where
// L'L names the destination: VCVTPH2DQ converts 256 bits of source into 512
// bits of destination. The two quarter forms hold one operand in the XMM
// class at every length: the source under the widening VCVTPH2QQ, the
// destination under the narrowing VCVTQQ2PH. An entry with Mem set takes
// the memory shape of the source position too: the compares and the packed
// square root read their source from memory, the packed square root's source
// carrying the {1toN} broadcast as EVEX.b, and the narrow BF16 convert reads
// its full-width source there. An entry with Er or Sae set takes the
// rounding decoration on its register form alone; the memory shape refuses
// one.
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
destClass := class
if in.Form == ExtFormAmdVec2Half {
destClass, srcClass := class, class
switch in.Form {
case ExtFormAmdVec2Half:
destClass = amd64HalfClass(class)
case ExtFormAmdVec2Wide:
srcClass = amd64HalfClass(class)
case ExtFormAmdVec2Quarter:
srcClass = ExtXMM
case ExtFormAmdVec2ToQuarter:
destClass = ExtXMM
}
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[1], 2)
if err != nil {
@@ -597,7 +613,7 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
amd64ApplyMask(out, mask, zeroing)
return out, nil
}
if err := in.amd64Vector(ops[0], class, 1); err != nil {
if err := in.amd64Vector(ops[0], srcClass, 1); err != nil {
return nil, err
}
if err := in.amd64Vector(dest, destClass, 2); err != nil {
@@ -878,13 +894,21 @@ func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
// (the move into a vector register and the integer conversions) and vec, gpr
// (the move out of one). In both orders the second operand is the
// destination in the reg field and the first the r/m source; the vector
// changes position with the form.
// changes position with the form. An entry with Mem set takes the memory
// shape of the vector source, the manual's m16 beside the register: the
// value converts straight out of memory.
func (in ExtInstr) encodeAmdGprPair(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
vecPos := 1
if in.Form == ExtFormAmdVecGpr {
vecPos = 0
}
if in.Mem == 1 && in.Form == ExtFormAmdVecGpr && ops[0].Kind == ExtMem {
if err := in.amd64Gpr(ops[1], 2); err != nil {
return nil, err
}
return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1)
}
if err := in.amd64Vector(ops[vecPos], class, vecPos+1); err != nil {
return nil, err
}
@@ -1022,16 +1046,16 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"},
{Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"},
{Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value",
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SS (EVEX.NDS.LIG.MAP6.W0 13 /r)"},
{Name: "VCVTSH2SD", Summary: "Convert a low FP16 value to an FP64 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"},
{Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"},
{Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
@@ -1046,16 +1070,16 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"},
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W0 2D /r)"},
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 64-bit integer",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W1 2D /r)"},
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 32-bit integer",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W0 79 /r)"},
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 64-bit integer",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W1 79 /r)"},
// AVX512-FP16 packed arithmetic: the full ZMM lanes the scalar core
@@ -1148,4 +1172,145 @@ var amd64Extensions = []ExtInstr{
{Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"},
// AVX512-FP16 packed conversions: the fourteen directions between the
// FP16 lanes and the integer and double-precision companions, three
// register widths each. L'L names the governing operand, whose class the
// form spells: the source on the narrowing converts (VCVTDQ2PH narrows
// its dword source to the half-width destination, VCVTQQ2PH and VCVTPD2PH
// to the quarter-width XMM destination), the destination on the widening
// ones (VCVTPH2DQ widens its half-width source, VCVTPH2QQ and VCVTPH2PD
// their quarter-width XMM source). The FP16-to-integer directions round
// and take Er on their 512-bit register forms; VCVTPH2PD widens exactly
// and takes none. The integer-to-FP16 sources read their elements from
// memory under the {1toN} broadcast, the FP16 sources their full-width
// vectors; every destination is write-masked. The encodings are
// transcribed from the SDM entries and pinned byte for byte against the
// local GNU assembler, whose 2.46 table matches the manual row for row.
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.512.66.MAP5.W0 7D /r)"},
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.256.66.MAP5.W0 7D /r)"},
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.128.66.MAP5.W0 7D /r)"},
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.512.NP.MAP5.W0 7D /r)"},
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.256.NP.MAP5.W0 7D /r)"},
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.128.NP.MAP5.W0 7D /r)"},
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.512.F3.MAP5.W0 7D /r)"},
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.256.F3.MAP5.W0 7D /r)"},
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.128.F3.MAP5.W0 7D /r)"},
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.512.F2.MAP5.W0 7D /r)"},
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.256.F2.MAP5.W0 7D /r)"},
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.128.F2.MAP5.W0 7D /r)"},
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.512.66.MAP5.W0 5B /r, half-width source)"},
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.256.66.MAP5.W0 5B /r, half-width source)"},
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.128.66.MAP5.W0 5B /r, half-width source)"},
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.512.NP.MAP5.W0 79 /r, half-width source)"},
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.256.NP.MAP5.W0 79 /r, half-width source)"},
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.128.NP.MAP5.W0 79 /r, half-width source)"},
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.512.NP.MAP5.W0 5B /r, YMM destination)"},
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.256.NP.MAP5.W0 5B /r, XMM destination)"},
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.128.NP.MAP5.W0 5B /r, XMM destination)"},
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.512.F2.MAP5.W0 7A /r, YMM destination)"},
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.256.F2.MAP5.W0 7A /r, XMM destination)"},
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.128.F2.MAP5.W0 7A /r, XMM destination)"},
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.512.66.MAP5.W0 7B /r, quarter-width source)"},
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.256.66.MAP5.W0 7B /r, quarter-width source)"},
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.128.66.MAP5.W0 7B /r, quarter-width source)"},
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.512.66.MAP5.W0 79 /r, quarter-width source)"},
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.256.66.MAP5.W0 79 /r, quarter-width source)"},
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.128.66.MAP5.W0 79 /r, quarter-width source)"},
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x84, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.512.NP.MAP5.W1 5B /r, XMM destination)"},
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x84, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.256.NP.MAP5.W1 5B /r, XMM destination)"},
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x84, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.128.NP.MAP5.W1 5B /r, XMM destination)"},
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x87, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.512.F2.MAP5.W1 7A /r, XMM destination)"},
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x87, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.256.F2.MAP5.W1 7A /r, XMM destination)"},
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.128.F2.MAP5.W1 7A /r, XMM destination)"},
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.512.NP.MAP5.W0 5A /r, quarter-width source)"},
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.256.NP.MAP5.W0 5A /r, quarter-width source)"},
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.128.NP.MAP5.W0 5A /r, quarter-width source)"},
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x85, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.512.66.MAP5.W1 5A /r, XMM destination)"},
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x85, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.256.66.MAP5.W1 5A /r, XMM destination)"},
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x85, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.128.66.MAP5.W1 5A /r, XMM destination)"},
}
+413 -2
View File
@@ -660,6 +660,396 @@ var amd64GoldenRows = []amd64GoldenRow{
{"vcvtusi2sh rd-sae, 64-bit, high registers", "VCVTUSI2SH",
[]ExtOperand{ExtXmm(29), ExtGpr64(12), ExtRounded(ExtXmm(30), ExtRoundDown)},
"624596307bf4", "62 45 96 30 7b f4 vcvtusi2sh %r12,{rd-sae},%xmm29,%xmm30"},
// The FP16 packed conversions, the fourteen directions between the FP16
// lanes and the integer and double-precision companions. The register
// rows and the rounding rows are quoted from the local GNU assembler's
// output; the memory rows follow the listing's full-width and broadcast
// spellings, whose source displacements are N times the plain SDM
// displacement the disp8 bytes carry under the assembler's compressed
// displacement scaling, and the rows with no GNU line are derived: the
// plain full-width load of a narrowing convert the AT&T syntax cannot
// spell without a size hint, laid out by the broadcast row minus EVEX.b.
{"vcvtph2w zmm", "VCVTPH2W",
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
"62f57d487df5", "62 f5 7d 48 7d f5 vcvtph2w %zmm5,%zmm6"},
{"vcvtph2w ymm", "VCVTPH2W",
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
"62f57d287df5", "62 f5 7d 28 7d f5 vcvtph2w %ymm5,%ymm6"},
{"vcvtph2w xmm", "VCVTPH2W",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57d087df5", "62 f5 7d 08 7d f5 vcvtph2w %xmm5,%xmm6"},
{"vcvtph2uw zmm", "VCVTPH2UW",
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
"62f57c487df5", "62 f5 7c 48 7d f5 vcvtph2uw %zmm5,%zmm6"},
{"vcvtph2uw ymm", "VCVTPH2UW",
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
"62f57c287df5", "62 f5 7c 28 7d f5 vcvtph2uw %ymm5,%ymm6"},
{"vcvtph2uw xmm", "VCVTPH2UW",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57c087df5", "62 f5 7c 08 7d f5 vcvtph2uw %xmm5,%xmm6"},
{"vcvtw2ph zmm", "VCVTW2PH",
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
"62f57e487df5", "62 f5 7e 48 7d f5 vcvtw2ph %zmm5,%zmm6"},
{"vcvtw2ph ymm", "VCVTW2PH",
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
"62f57e287df5", "62 f5 7e 28 7d f5 vcvtw2ph %ymm5,%ymm6"},
{"vcvtw2ph xmm", "VCVTW2PH",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57e087df5", "62 f5 7e 08 7d f5 vcvtw2ph %xmm5,%xmm6"},
{"vcvtuw2ph zmm", "VCVTUW2PH",
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
"62f57f487df5", "62 f5 7f 48 7d f5 vcvtuw2ph %zmm5,%zmm6"},
{"vcvtuw2ph ymm", "VCVTUW2PH",
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
"62f57f287df5", "62 f5 7f 28 7d f5 vcvtuw2ph %ymm5,%ymm6"},
{"vcvtuw2ph xmm", "VCVTUW2PH",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57f087df5", "62 f5 7f 08 7d f5 vcvtuw2ph %xmm5,%xmm6"},
{"vcvtph2dq zmm", "VCVTPH2DQ",
[]ExtOperand{ExtYmm(5), ExtZmm(6)},
"62f57d485bf5", "62 f5 7d 48 5b f5 vcvtph2dq %ymm5,%zmm6"},
{"vcvtph2dq ymm", "VCVTPH2DQ",
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
"62f57d285bf5", "62 f5 7d 28 5b f5 vcvtph2dq %xmm5,%ymm6"},
{"vcvtph2dq xmm", "VCVTPH2DQ",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57d085bf5", "62 f5 7d 08 5b f5 vcvtph2dq %xmm5,%xmm6"},
{"vcvtph2udq zmm", "VCVTPH2UDQ",
[]ExtOperand{ExtYmm(5), ExtZmm(6)},
"62f57c4879f5", "62 f5 7c 48 79 f5 vcvtph2udq %ymm5,%zmm6"},
{"vcvtph2udq ymm", "VCVTPH2UDQ",
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
"62f57c2879f5", "62 f5 7c 28 79 f5 vcvtph2udq %xmm5,%ymm6"},
{"vcvtph2udq xmm", "VCVTPH2UDQ",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57c0879f5", "62 f5 7c 08 79 f5 vcvtph2udq %xmm5,%xmm6"},
{"vcvtdq2ph zmm to ymm", "VCVTDQ2PH",
[]ExtOperand{ExtZmm(5), ExtYmm(6)},
"62f57c485bf5", "62 f5 7c 48 5b f5 vcvtdq2ph %zmm5,%ymm6"},
{"vcvtdq2ph ymm to xmm", "VCVTDQ2PH",
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
"62f57c285bf5", "62 f5 7c 28 5b f5 vcvtdq2ph %ymm5,%xmm6"},
{"vcvtdq2ph xmm to xmm", "VCVTDQ2PH",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57c085bf5", "62 f5 7c 08 5b f5 vcvtdq2ph %xmm5,%xmm6"},
{"vcvtudq2ph zmm to ymm", "VCVTUDQ2PH",
[]ExtOperand{ExtZmm(5), ExtYmm(6)},
"62f57f487af5", "62 f5 7f 48 7a f5 vcvtudq2ph %zmm5,%ymm6"},
{"vcvtudq2ph ymm to xmm", "VCVTUDQ2PH",
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
"62f57f287af5", "62 f5 7f 28 7a f5 vcvtudq2ph %ymm5,%xmm6"},
{"vcvtudq2ph xmm to xmm", "VCVTUDQ2PH",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57f087af5", "62 f5 7f 08 7a f5 vcvtudq2ph %xmm5,%xmm6"},
{"vcvtph2qq zmm", "VCVTPH2QQ",
[]ExtOperand{ExtXmm(5), ExtZmm(6)},
"62f57d487bf5", "62 f5 7d 48 7b f5 vcvtph2qq %xmm5,%zmm6"},
{"vcvtph2qq ymm", "VCVTPH2QQ",
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
"62f57d287bf5", "62 f5 7d 28 7b f5 vcvtph2qq %xmm5,%ymm6"},
{"vcvtph2qq xmm", "VCVTPH2QQ",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57d087bf5", "62 f5 7d 08 7b f5 vcvtph2qq %xmm5,%xmm6"},
{"vcvtph2uqq zmm", "VCVTPH2UQQ",
[]ExtOperand{ExtXmm(5), ExtZmm(6)},
"62f57d4879f5", "62 f5 7d 48 79 f5 vcvtph2uqq %xmm5,%zmm6"},
{"vcvtph2uqq ymm", "VCVTPH2UQQ",
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
"62f57d2879f5", "62 f5 7d 28 79 f5 vcvtph2uqq %xmm5,%ymm6"},
{"vcvtph2uqq xmm", "VCVTPH2UQQ",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57d0879f5", "62 f5 7d 08 79 f5 vcvtph2uqq %xmm5,%xmm6"},
{"vcvtqq2ph zmm to xmm", "VCVTQQ2PH",
[]ExtOperand{ExtZmm(5), ExtXmm(6)},
"62f5fc485bf5", "62 f5 fc 48 5b f5 vcvtqq2ph %zmm5,%xmm6"},
{"vcvtqq2ph ymm to xmm", "VCVTQQ2PH",
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
"62f5fc285bf5", "62 f5 fc 28 5b f5 vcvtqq2ph %ymm5,%xmm6"},
{"vcvtqq2ph xmm to xmm", "VCVTQQ2PH",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f5fc085bf5", "62 f5 fc 08 5b f5 vcvtqq2ph %xmm5,%xmm6"},
{"vcvtuqq2ph zmm to xmm", "VCVTUQQ2PH",
[]ExtOperand{ExtZmm(5), ExtXmm(6)},
"62f5ff487af5", "62 f5 ff 48 7a f5 vcvtuqq2ph %zmm5,%xmm6"},
{"vcvtuqq2ph ymm to xmm", "VCVTUQQ2PH",
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
"62f5ff287af5", "62 f5 ff 28 7a f5 vcvtuqq2ph %ymm5,%xmm6"},
{"vcvtuqq2ph xmm to xmm", "VCVTUQQ2PH",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f5ff087af5", "62 f5 ff 08 7a f5 vcvtuqq2ph %xmm5,%xmm6"},
{"vcvtph2pd zmm", "VCVTPH2PD",
[]ExtOperand{ExtXmm(5), ExtZmm(6)},
"62f57c485af5", "62 f5 7c 48 5a f5 vcvtph2pd %xmm5,%zmm6"},
{"vcvtph2pd ymm", "VCVTPH2PD",
[]ExtOperand{ExtXmm(5), ExtYmm(6)},
"62f57c285af5", "62 f5 7c 28 5a f5 vcvtph2pd %xmm5,%ymm6"},
{"vcvtph2pd xmm", "VCVTPH2PD",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57c085af5", "62 f5 7c 08 5a f5 vcvtph2pd %xmm5,%xmm6"},
{"vcvtpd2ph zmm to xmm", "VCVTPD2PH",
[]ExtOperand{ExtZmm(5), ExtXmm(6)},
"62f5fd485af5", "62 f5 fd 48 5a f5 vcvtpd2ph %zmm5,%xmm6"},
{"vcvtpd2ph ymm to xmm", "VCVTPD2PH",
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
"62f5fd285af5", "62 f5 fd 28 5a f5 vcvtpd2ph %ymm5,%xmm6"},
{"vcvtpd2ph xmm to xmm", "VCVTPD2PH",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f5fd085af5", "62 f5 fd 08 5a f5 vcvtpd2ph %xmm5,%xmm6"},
// The rounding rows of the converting directions: the FP16-to-integer
// and the integer-to-FP16 conversions round and take EVEX.RC on their
// 512-bit register forms, exactly as the arithmetic does, and VCVTPH2PD,
// which widens exactly, takes none.
{"vcvtph2w rn-sae", "VCVTPH2W",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
"62f57d187df5", "62 f5 7d 18 7d f5 vcvtph2w {rn-sae},%zmm5,%zmm6"},
{"vcvtph2uw rz-sae", "VCVTPH2UW",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)},
"62f57c787df5", "62 f5 7c 78 7d f5 vcvtph2uw {rz-sae},%zmm5,%zmm6"},
{"vcvtph2dq rn-sae", "VCVTPH2DQ",
[]ExtOperand{ExtYmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
"62f57d185bf5", "62 f5 7d 18 5b f5 vcvtph2dq {rn-sae},%ymm5,%zmm6"},
{"vcvtph2udq rn-sae", "VCVTPH2UDQ",
[]ExtOperand{ExtYmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
"62f57c1879f5", "62 f5 7c 18 79 f5 vcvtph2udq {rn-sae},%ymm5,%zmm6"},
{"vcvtph2qq rz-sae", "VCVTPH2QQ",
[]ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)},
"62f57d787bf5", "62 f5 7d 78 7b f5 vcvtph2qq {rz-sae},%xmm5,%zmm6"},
{"vcvtph2uqq rz-sae", "VCVTPH2UQQ",
[]ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)},
"62f57d7879f5", "62 f5 7d 78 79 f5 vcvtph2uqq {rz-sae},%xmm5,%zmm6"},
{"vcvtw2ph rz-sae", "VCVTW2PH",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)},
"62f57e787df5", "62 f5 7e 78 7d f5 vcvtw2ph {rz-sae},%zmm5,%zmm6"},
{"vcvtuw2ph rn-sae", "VCVTUW2PH",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
"62f57f187df5", "62 f5 7f 18 7d f5 vcvtuw2ph {rn-sae},%zmm5,%zmm6"},
{"vcvtdq2ph rn-sae", "VCVTDQ2PH",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtYmm(6), ExtRoundNearest)},
"62f57c185bf5", "62 f5 7c 18 5b f5 vcvtdq2ph {rn-sae},%zmm5,%ymm6"},
{"vcvtudq2ph ru-sae", "VCVTUDQ2PH",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtYmm(6), ExtRoundUp)},
"62f57f587af5", "62 f5 7f 58 7a f5 vcvtudq2ph {ru-sae},%zmm5,%ymm6"},
{"vcvtqq2ph rz-sae", "VCVTQQ2PH",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundTruncate)},
"62f5fc785bf5", "62 f5 fc 78 5b f5 vcvtqq2ph {rz-sae},%zmm5,%xmm6"},
{"vcvtuqq2ph rd-sae", "VCVTUQQ2PH",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundDown)},
"62f5ff387af5", "62 f5 ff 38 7a f5 vcvtuqq2ph {rd-sae},%zmm5,%xmm6"},
{"vcvtpd2ph rz-sae", "VCVTPD2PH",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundTruncate)},
"62f5fd785af5", "62 f5 fd 78 5a f5 vcvtpd2ph {rz-sae},%zmm5,%xmm6"},
// High registers across the conversion shapes, exercising every EVEX
// extension bit: the source above 15 clears B bar and X bar, the
// destination above 15 R bar.
{"vcvtph2w, high registers", "VCVTPH2W",
[]ExtOperand{ExtZmm(28), ExtZmm(29)},
"62057d487dec", "62 05 7d 48 7d ec vcvtph2w %zmm28,%zmm29"},
{"vcvtph2dq, high registers", "VCVTPH2DQ",
[]ExtOperand{ExtYmm(28), ExtZmm(29)},
"62057d485bec", "62 05 7d 48 5b ec vcvtph2dq %ymm28,%zmm29"},
{"vcvtdq2ph, high registers", "VCVTDQ2PH",
[]ExtOperand{ExtZmm(29), ExtYmm(30)},
"62057c485bf5", "62 05 7c 48 5b f5 vcvtdq2ph %zmm29,%ymm30"},
{"vcvtph2qq, high registers", "VCVTPH2QQ",
[]ExtOperand{ExtXmm(28), ExtZmm(29)},
"62057d487bec", "62 05 7d 48 7b ec vcvtph2qq %xmm28,%zmm29"},
{"vcvtpd2ph, high registers", "VCVTPD2PH",
[]ExtOperand{ExtZmm(28), ExtXmm(29)},
"6205fd485aec", "62 05 fd 48 5a ec vcvtpd2ph %zmm28,%xmm29"},
// The memory forms of the conversions: the FP16 sources read full-width
// vectors, the integer sources their elements under the broadcast. The
// disp8 rows quote the listing's compressed spellings, whose source is N
// times the plain displacement the bytes carry.
{"vcvtph2w memory source", "VCVTPH2W",
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
"62457d487d31", "62 45 7d 48 7d 31 vcvtph2w (%r9),%zmm30"},
{"vcvtph2w memory source disp8", "VCVTPH2W",
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
"62657d487d717f", "62 65 7d 48 7d 71 7f vcvtph2w 0x1fc0(%rcx),%zmm30 (Disp8(7f))"},
{"vcvtph2w ymm memory source", "VCVTPH2W",
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
"62457d287d31", "62 45 7d 28 7d 31 vcvtph2w (%r9),%ymm30"},
{"vcvtph2w xmm memory source", "VCVTPH2W",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
"62457d087d31", "62 45 7d 08 7d 31 vcvtph2w (%r9),%xmm30"},
{"vcvtph2uw memory source", "VCVTPH2UW",
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
"62457c487d31", "62 45 7c 48 7d 31 vcvtph2uw (%r9),%zmm30"},
{"vcvtw2ph memory source", "VCVTW2PH",
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
"62457e487d31", "62 45 7e 48 7d 31 vcvtw2ph (%r9),%zmm30"},
{"vcvtw2ph ymm memory source", "VCVTW2PH",
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
"62457e287d31", "62 45 7e 28 7d 31 vcvtw2ph (%r9),%ymm30"},
{"vcvtph2dq memory source", "VCVTPH2DQ",
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
"62457d485b31", "62 45 7d 48 5b 31 vcvtph2dq (%r9),%zmm30"},
{"vcvtph2dq memory source disp8", "VCVTPH2DQ",
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
"62657d485b717f", "62 65 7d 48 5b 71 7f vcvtph2dq 0xfe0(%rcx),%zmm30 (Disp8(7f))"},
{"vcvtph2dq ymm memory source", "VCVTPH2DQ",
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
"62457d285b31", "62 45 7d 28 5b 31 vcvtph2dq (%r9),%ymm30"},
{"vcvtph2udq memory source", "VCVTPH2UDQ",
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
"62457c487931", "62 45 7c 48 79 31 vcvtph2udq (%r9),%zmm30"},
{"vcvtdq2ph memory source", "VCVTDQ2PH",
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
"62457c485b31", "62 45 7c 48 5b 31 vcvtdq2ph (%r9),%ymm30"},
{"vcvtph2qq memory source", "VCVTPH2QQ",
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
"62457d487b31", "62 45 7d 48 7b 31 vcvtph2qq (%r9),%zmm30"},
{"vcvtph2qq memory source disp8", "VCVTPH2QQ",
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
"62657d487b717f", "62 65 7d 48 7b 71 7f vcvtph2qq 0x7f0(%rcx),%zmm30 (Disp8(7f))"},
{"vcvtph2qq ymm memory source", "VCVTPH2QQ",
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
"62457d287b31", "62 45 7d 28 7b 31 vcvtph2qq (%r9),%ymm30"},
{"vcvtph2uqq memory source", "VCVTPH2UQQ",
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
"62457d487931", "62 45 7d 48 79 31 vcvtph2uqq (%r9),%zmm30"},
{"vcvtph2pd memory source", "VCVTPH2PD",
[]ExtOperand{ExtMemory(9, 0), ExtZmm(30)},
"62457c485a31", "62 45 7c 48 5a 31 vcvtph2pd (%r9),%zmm30"},
{"vcvtph2pd memory source disp8", "VCVTPH2PD",
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
"62657c485a717f", "62 65 7c 48 5a 71 7f vcvtph2pd 0x7f0(%rcx),%zmm30 (Disp8(7f))"},
{"vcvtph2pd ymm memory source", "VCVTPH2PD",
[]ExtOperand{ExtMemory(9, 0), ExtYmm(30)},
"62457c285a31", "62 45 7c 28 5a 31 vcvtph2pd (%r9),%ymm30"},
{"vcvtqq2ph plain memory source", "VCVTQQ2PH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
"6245fc485b31", ""},
{"vcvtuqq2ph plain memory source", "VCVTUQQ2PH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
"6245ff487a31", ""},
{"vcvtpd2ph plain memory source", "VCVTPD2PH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
"6245fd485a31", ""},
// The broadcast forms of the converting sources: a single element the
// hardware splats across the lanes, one word, dword or qword as the
// direction's input element spells it.
{"vcvtw2ph broadcast source", "VCVTW2PH",
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
"62457e587d31", "62 45 7e 58 7d 31 vcvtw2ph (%r9){1to32},%zmm30"},
{"vcvtw2ph broadcast source disp8", "VCVTW2PH",
[]ExtOperand{ExtBroadcast(1, 127), ExtZmm(30)},
"62657e587d717f", "62 65 7e 58 7d 71 7f vcvtw2ph 0xfe(%rcx){1to32},%zmm30 (Disp8(7f))"},
{"vcvtuw2ph broadcast source", "VCVTUW2PH",
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
"62457f587d31", "62 45 7f 58 7d 31 vcvtuw2ph (%r9){1to32},%zmm30"},
{"vcvtuw2ph broadcast source disp8", "VCVTUW2PH",
[]ExtOperand{ExtBroadcast(1, 127), ExtZmm(30)},
"62657f587d717f", "62 65 7f 58 7d 71 7f vcvtuw2ph 0xfe(%rcx){1to32},%zmm30 (Disp8(7f))"},
{"vcvtdq2ph broadcast source", "VCVTDQ2PH",
[]ExtOperand{ExtBroadcast(9, 0), ExtYmm(30)},
"62457c585b31", "62 45 7c 58 5b 31 vcvtdq2ph (%r9){1to16},%ymm30"},
{"vcvtdq2ph broadcast source disp8", "VCVTDQ2PH",
[]ExtOperand{ExtBroadcast(1, 127), ExtYmm(30)},
"62657c585b717f", "62 65 7c 58 5b 71 7f vcvtdq2ph 0x1fc(%rcx){1to16},%ymm30 (Disp8(7f))"},
{"vcvtudq2ph broadcast source", "VCVTUDQ2PH",
[]ExtOperand{ExtBroadcast(9, 0), ExtYmm(30)},
"62457f587a31", "62 45 7f 58 7a 31 vcvtudq2ph (%r9){1to16},%ymm30"},
{"vcvtudq2ph broadcast source disp8", "VCVTUDQ2PH",
[]ExtOperand{ExtBroadcast(1, 127), ExtYmm(30)},
"62657f587a717f", "62 65 7f 58 7a 71 7f vcvtudq2ph 0x1fc(%rcx){1to16},%ymm30 (Disp8(7f))"},
{"vcvtqq2ph broadcast source", "VCVTQQ2PH",
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
"6245fc585b31", "62 45 fc 58 5b 31 vcvtqq2ph (%r9){1to8},%xmm30"},
{"vcvtqq2ph broadcast source disp8", "VCVTQQ2PH",
[]ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)},
"6265fc585b717f", "62 65 fc 58 5b 71 7f vcvtqq2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"},
// The 256-bit broadcast row shares the 512-bit row's operand list, both
// destinations being XMM, so the resolver cannot tell them apart; the
// register row above pins the 256-bit template alone.
{"vcvtuqq2ph broadcast source", "VCVTUQQ2PH",
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
"6245ff587a31", "62 45 ff 58 7a 31 vcvtuqq2ph (%r9){1to8},%xmm30"},
{"vcvtuqq2ph broadcast source disp8", "VCVTUQQ2PH",
[]ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)},
"6265ff587a717f", "62 65 ff 58 7a 71 7f vcvtuqq2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"},
{"vcvtpd2ph broadcast source", "VCVTPD2PH",
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
"6245fd585a31", "62 45 fd 58 5a 31 vcvtpd2ph (%r9){1to8},%xmm30"},
{"vcvtpd2ph broadcast source disp8", "VCVTPD2PH",
[]ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)},
"6265fd585a717f", "62 65 fd 58 5a 71 7f vcvtpd2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"},
// The 256-bit broadcast row shares the 512-bit row's operand list, both
// destinations being XMM, so the resolver cannot tell them apart; the
// register row above pins the 256-bit template alone.
// The write mask over the converting destinations, and one composed
// row: the broadcast, the mask and the zeroing on the one word.
{"vcvtph2w k7 zeroing", "VCVTPH2W",
[]ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)},
"62457dcf7d31", "62 45 7d cf 7d 31 vcvtph2w (%r9),%zmm30{%k7}{z}"},
{"vcvtph2dq k7 zeroing, memory source", "VCVTPH2DQ",
[]ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)},
"62457dcf5b31", "62 45 7d cf 5b 31 vcvtph2dq (%r9),%zmm30{%k7}{z}"},
{"vcvtph2qq k7, memory source", "VCVTPH2QQ",
[]ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, false)},
"62457d4f7b31", "62 45 7d 4f 7b 31 vcvtph2qq (%r9),%zmm30{%k7}"},
{"vcvtph2pd k7, memory source", "VCVTPH2PD",
[]ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, false)},
"62457c4f5a31", "62 45 7c 4f 5a 31 vcvtph2pd (%r9),%zmm30{%k7}"},
{"vcvtw2ph k7 zeroing", "VCVTW2PH",
[]ExtOperand{ExtZmm(5), ExtWriteMasked(ExtZmm(6), 7, true)},
"62f57ecf7df5", "62 f5 7e cf 7d f5 vcvtw2ph %zmm5,%zmm6{%k7}{z}"},
{"vcvtdq2ph k7 zeroing", "VCVTDQ2PH",
[]ExtOperand{ExtZmm(5), ExtWriteMasked(ExtYmm(6), 7, true)},
"62f57ccf5bf5", "62 f5 7c cf 5b f5 vcvtdq2ph %zmm5,%ymm6{%k7}{z}"},
{"vcvtw2ph broadcast under k7 zeroing", "VCVTW2PH",
[]ExtOperand{ExtBroadcast(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)},
"62457edf7d31", "62 45 7e df 7d 31 vcvtw2ph (%r9){1to32},%zmm30{%k7}{z}"},
// The memory forms of the scalar conversions, the m16, m32 and m64
// sources the manual spells beside each register form. The disp8 rows
// quote the listing's compressed spellings, whose source is N times the
// plain displacement, N the element size.
{"vcvtsh2ss memory source", "VCVTSH2SS",
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)},
"62d654081331", "62 d6 54 08 13 31 vcvtsh2ss (%r9),%xmm5,%xmm6"},
{"vcvtsh2ss memory source disp8", "VCVTSH2SS",
[]ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)},
"62f65408137101", "62 f6 54 08 13 71 01 vcvtsh2ss 0x2(%rcx),%xmm5,%xmm6 (Disp8(01))"},
{"vcvtsh2sd memory source", "VCVTSH2SD",
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)},
"62d556085a31", "62 d5 56 08 5a 31 vcvtsh2sd (%r9),%xmm5,%xmm6"},
{"vcvtsh2sd memory source disp8", "VCVTSH2SD",
[]ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)},
"62f556085a7101", "62 f5 56 08 5a 71 01 vcvtsh2sd 0x2(%rcx),%xmm5,%xmm6 (Disp8(01))"},
{"vcvtss2sh memory source", "VCVTSS2SH",
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)},
"62d554081d31", "62 d5 54 08 1d 31 vcvtss2sh (%r9),%xmm5,%xmm6"},
{"vcvtss2sh memory source disp8", "VCVTSS2SH",
[]ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)},
"62f554081d7101", "62 f5 54 08 1d 71 01 vcvtss2sh 0x4(%rcx),%xmm5,%xmm6 (Disp8(01))"},
{"vcvtsd2sh memory source", "VCVTSD2SH",
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)},
"62d5d7085a31", "62 d5 d7 08 5a 31 vcvtsd2sh (%r9),%xmm5,%xmm6"},
{"vcvtsd2sh memory source disp8", "VCVTSD2SH",
[]ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)},
"62f5d7085a7101", "62 f5 d7 08 5a 71 01 vcvtsd2sh 0x8(%rcx),%xmm5,%xmm6 (Disp8(01))"},
{"vcvtsh2si memory source", "VCVTSH2SI",
[]ExtOperand{ExtMemory(9, 0), ExtGpr32(0)},
"62d57e082d01", "62 d5 7e 08 2d 01 vcvtsh2si (%r9),%eax"},
{"vcvtsh2si 64-bit memory source", "VCVTSH2SI",
[]ExtOperand{ExtMemory(9, 0), ExtGpr64(12)},
"6255fe082d21", "62 55 fe 08 2d 21 vcvtsh2si (%r9),%r12"},
{"vcvtsh2si memory source disp8", "VCVTSH2SI",
[]ExtOperand{ExtMemory(1, 1), ExtGpr32(0)},
"62f57e082d4101", "62 f5 7e 08 2d 41 01 vcvtsh2si 0x2(%rcx),%eax (Disp8(01))"},
{"vcvtsh2usi memory source", "VCVTSH2USI",
[]ExtOperand{ExtMemory(9, 0), ExtGpr32(0)},
"62d57e087901", "62 d5 7e 08 79 01 vcvtsh2usi (%r9),%eax"},
{"vcvtsh2usi 64-bit memory source disp8", "VCVTSH2USI",
[]ExtOperand{ExtMemory(1, 1), ExtGpr64(12)},
"6275fe08796101", "62 75 fe 08 79 61 01 vcvtsh2usi 0x2(%rcx),%r12 (Disp8(01))"},
}
// amd64ResolveEntry finds the table entry a golden row exercises: the entry
@@ -951,6 +1341,12 @@ func TestAmd64ExtRejects(t *testing.T) {
{"rounding over a memory source", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtRounded(ExtXmm(30), ExtRoundNearest)},
"the memory form takes no rounding control"},
{"rounding over the scalar convert's memory source", "VCVTSS2SH",
[]ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtRounded(ExtXmm(6), ExtRoundNearest)},
"the memory form takes no rounding control"},
{"rounding beside the converting memory source", "VCVTW2PH",
[]ExtOperand{ExtMemory(9, 0), ExtRounded(ExtZmm(30), ExtRoundNearest)},
"the memory form takes no rounding control"},
{"rounding on a source position", "VADDPH",
[]ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(4), ExtRoundUp), ExtZmm(6)},
"the position takes none"},
@@ -963,6 +1359,21 @@ func TestAmd64ExtRejects(t *testing.T) {
{"rounding on an entry without the capability", "VMOVSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundNearest)},
"the entry's destination takes none"},
{"rounding on the exactly-widening convert", "VCVTPH2PD",
[]ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)},
"the entry's destination takes none"},
{"wrong source class on the widening convert", "VCVTPH2DQ",
[]ExtOperand{ExtZmm(5), ExtZmm(6)},
"wants a YMM register"},
{"wrong source class on the quarter convert", "VCVTPH2QQ",
[]ExtOperand{ExtYmm(5), ExtZmm(6)},
"wants an XMM register"},
{"wrong destination on the quarter-destination convert", "VCVTQQ2PH",
[]ExtOperand{ExtZmm(5), ExtYmm(6)},
"wants an XMM register"},
{"broadcast on the full-width FP16 source", "VCVTPH2W",
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
"the entry's memory operand takes none"},
} {
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
_, err := in.Encode(tt.ops)
@@ -1107,7 +1518,7 @@ func TestAmd64ExtArchBinding(t *testing.T) {
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
}
}
if got := Extensions(AMD64); len(got) != 70 {
t.Errorf("the amd64 layer registers %d instructions, want 70", len(got))
if got := Extensions(AMD64); len(got) != 112 {
t.Errorf("the amd64 layer registers %d instructions, want 112", len(got))
}
}
+22 -2
View File
@@ -444,6 +444,18 @@ const (
// half-width companion of the source, equal at 128 bits: VCVTNEPS2BF16
// Y6, Z5. Operands: src, dest.
ExtFormAmdVec2Half
// ExtFormAmdVec2Wide is the two-vector form whose source is the half-width
// companion of the destination, equal at 128 bits: VCVTPH2DQ Z6, Y5, where
// L'L names the destination. Operands: src, dest.
ExtFormAmdVec2Wide
// ExtFormAmdVec2Quarter is the two-vector form whose source is the
// quarter-width companion of the destination, the XMM class at every
// length: VCVTPH2QQ Z6, X5. Operands: src, dest.
ExtFormAmdVec2Quarter
// ExtFormAmdVec2ToQuarter is the two-vector form whose destination is the
// quarter-width companion of the source, the XMM class at every length:
// VCVTQQ2PH X6, Z5. Operands: src, dest.
ExtFormAmdVec2ToQuarter
// ExtFormAmdMask2 is the form with an opmask destination and two vector
// sources: VP2INTERSECTD K0, Z2, Z1. Operands: src1, src2, dest.
ExtFormAmdMask2
@@ -668,7 +680,8 @@ func (f ExtForm) Arity() int {
return 2
case ExtFormAmdVec3, ExtFormAmdMask2, ExtFormAmdVecGprVec:
return 3
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdGprVec, ExtFormAmdVecGpr:
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter,
ExtFormAmdVec2ToQuarter, ExtFormAmdGprVec, ExtFormAmdVecGpr:
return 2
case ExtFormAmdVec3Imm, ExtFormAmdMask2Imm:
return 4
@@ -777,6 +790,12 @@ func (f ExtForm) String() string {
return "two vectors"
case ExtFormAmdVec2Half:
return "two vectors, half-width destination"
case ExtFormAmdVec2Wide:
return "two vectors, half-width source"
case ExtFormAmdVec2Quarter:
return "two vectors, quarter-width source"
case ExtFormAmdVec2ToQuarter:
return "two vectors, quarter-width destination"
case ExtFormAmdMask2:
return "two vectors into an opmask"
case ExtFormAmdVecGprVec:
@@ -1037,7 +1056,8 @@ func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) {
return in.encodeSignedImmediate(ops)
case ExtFormAmdVec3, ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdMask2,
ExtFormAmdVecGprVec, ExtFormAmdGprVec, ExtFormAmdVecGpr,
ExtFormAmdVec3Imm, ExtFormAmdMask2Imm, ExtFormAmdMemVec, ExtFormAmdVecMem:
ExtFormAmdVec3Imm, ExtFormAmdMask2Imm, ExtFormAmdMemVec, ExtFormAmdVecMem,
ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter, ExtFormAmdVec2ToQuarter:
return in.encodeAmd64(ops)
case ExtFormPredicateLogical, ExtFormPredicateSelect:
return in.encodePredicateLogical(ops)
+31 -2
View File
@@ -41,6 +41,20 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
{"VGETMANTSH", 1},
{"VREDUCESH", 1},
{"VRNDSCALESH", 1},
{"VCVTPH2W", 3},
{"VCVTPH2UW", 3},
{"VCVTW2PH", 3},
{"VCVTUW2PH", 3},
{"VCVTPH2DQ", 3},
{"VCVTPH2UDQ", 3},
{"VCVTDQ2PH", 3},
{"VCVTUDQ2PH", 3},
{"VCVTPH2QQ", 3},
{"VCVTPH2UQQ", 3},
{"VCVTQQ2PH", 3},
{"VCVTUQQ2PH", 3},
{"VCVTPH2PD", 3},
{"VCVTPD2PH", 3},
} {
cands, ok := LookupExtension(arch.AMD64, tt.mnem)
if !ok {
@@ -54,8 +68,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
t.Errorf("the %s lookup is not case-insensitive", tt.mnem)
}
}
if got := arch.Extensions(arch.AMD64); len(got) != 70 {
t.Errorf("the amd64 layer registers %d instructions, want 70", len(got))
if got := arch.Extensions(arch.AMD64); len(got) != 112 {
t.Errorf("the amd64 layer registers %d instructions, want 112", len(got))
}
if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok {
t.Error("a non-extended mnemonic resolved")
@@ -131,6 +145,21 @@ func TestEncodeExtensionAmd64(t *testing.T) {
{"scalar fp16 minimum with exceptions suppressed", "VMINSH",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundSAE)},
"62f556185df4"},
{"fp16 to signed words", "VCVTPH2W",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(6)},
"62f57d487df5"},
{"dwords to fp16 over a broadcast source", "VCVTDQ2PH",
[]arch.ExtOperand{arch.ExtBroadcast(9, 0), arch.ExtYmm(30)},
"62457c585b31"},
{"fp16 to signed qwords, rounded", "VCVTPH2QQ",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundTruncate)},
"62f57d787bf5"},
{"fp16 scalar out of memory into an integer", "VCVTSH2SI",
[]arch.ExtOperand{arch.ExtMemory(9, 0), arch.ExtGpr64(12)},
"6255fe082d21"},
{"fp16 to double-precision, widened", "VCVTPH2PD",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtZmm(6)},
"62f57c485af5"},
} {
got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...)
if err != nil {