From 8de1b371da3644b52331b9ce5c3080d624d061a9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 13:08:54 +0200 Subject: [PATCH] feat(arch): add the amd64 fp16 packed conversion family Assisted-by: GLM 5.3 --- arch/amd64_ext.go | 221 ++++++++++++++++--- arch/amd64_ext_test.go | 415 +++++++++++++++++++++++++++++++++++- arch/arm64_ext.go | 24 ++- asm/extension_amd64_test.go | 33 ++- 4 files changed, 659 insertions(+), 34 deletions(-) diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index 7336e69..c85e3bf 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -10,14 +10,17 @@ // // The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the // latter's scalar core with its imm8-control group, its packed 512-bit and -// VL arithmetic and the embedded rounding of its FP operations, in their -// EVEX register forms. The encodings are transcribed from the SDM -// instruction entries and cross-checked against binutils-gdb's assembler -// testsuite; the golden vectors in amd64_ext_test.go pin the bytes. -// VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, no -// longer belong here: the Go toolchain's assembler knows them today, they -// live in the generated table and the EVEX encoder, and a mnemonic the -// toolchain has is not an extension. +// VL arithmetic, the embedded rounding of its FP operations and its fourteen +// packed conversion directions, in their EVEX register forms. The encodings +// are transcribed from the SDM instruction entries and cross-checked against +// binutils-gdb's assembler testsuite; the golden vectors in amd64_ext_test.go +// pin the bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 +// survey, no longer belong here: the Go toolchain's assembler knows them +// today, they live in the generated table and the EVEX encoder, and a +// mnemonic the toolchain has is not an extension. VCVTPS2PH, VCVTUDQ2PS +// and the rest of the classic conversion set follow the same rule, which is +// why the conversions here are the FP16 directions the toolchain has never +// emitted. // // The forms encode the unmasked shapes: register forms throughout, and the // memory forms beside them, the scalar ones the manual spells m16, m32 and @@ -462,7 +465,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) { switch in.Form { case ExtFormAmdVec3: return in.encodeAmdVec3(ops) - case ExtFormAmdVec2, ExtFormAmdVec2Half: + case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter, + ExtFormAmdVec2ToQuarter: return in.encodeAmdVec2(ops) case ExtFormAmdMask2: return in.encodeAmdMask2(ops) @@ -560,18 +564,30 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) { // encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows // the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits -// of destination, and at 128 bits the companion stays the class itself. An -// entry with Mem set takes the memory shape of the source position too: the -// compares and the packed square root read their source from memory, the -// packed square root's source carrying the {1toN} broadcast as EVEX.b, and -// the narrow BF16 convert reads its full-width source there. An entry with -// Er or Sae set takes the rounding decoration on its register form alone; -// the memory shape refuses one. +// of destination, and at 128 bits the companion stays the class itself. The +// wide form narrows the source instead, the widening FP16 conversions, where +// L'L names the destination: VCVTPH2DQ converts 256 bits of source into 512 +// bits of destination. The two quarter forms hold one operand in the XMM +// class at every length: the source under the widening VCVTPH2QQ, the +// destination under the narrowing VCVTQQ2PH. An entry with Mem set takes +// the memory shape of the source position too: the compares and the packed +// square root read their source from memory, the packed square root's source +// carrying the {1toN} broadcast as EVEX.b, and the narrow BF16 convert reads +// its full-width source there. An entry with Er or Sae set takes the +// rounding decoration on its register form alone; the memory shape refuses +// one. func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) - destClass := class - if in.Form == ExtFormAmdVec2Half { + destClass, srcClass := class, class + switch in.Form { + case ExtFormAmdVec2Half: destClass = amd64HalfClass(class) + case ExtFormAmdVec2Wide: + srcClass = amd64HalfClass(class) + case ExtFormAmdVec2Quarter: + srcClass = ExtXMM + case ExtFormAmdVec2ToQuarter: + destClass = ExtXMM } dest, mask, zeroing, round, err := in.amd64WriteMask(ops[1], 2) if err != nil { @@ -597,7 +613,7 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { amd64ApplyMask(out, mask, zeroing) return out, nil } - if err := in.amd64Vector(ops[0], class, 1); err != nil { + if err := in.amd64Vector(ops[0], srcClass, 1); err != nil { return nil, err } if err := in.amd64Vector(dest, destClass, 2); err != nil { @@ -878,13 +894,21 @@ func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) { // (the move into a vector register and the integer conversions) and vec, gpr // (the move out of one). In both orders the second operand is the // destination in the reg field and the first the r/m source; the vector -// changes position with the form. +// changes position with the form. An entry with Mem set takes the memory +// shape of the vector source, the manual's m16 beside the register: the +// value converts straight out of memory. func (in ExtInstr) encodeAmdGprPair(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) vecPos := 1 if in.Form == ExtFormAmdVecGpr { vecPos = 0 } + if in.Mem == 1 && in.Form == ExtFormAmdVecGpr && ops[0].Kind == ExtMem { + if err := in.amd64Gpr(ops[1], 2); err != nil { + return nil, err + } + return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1) + } if err := in.amd64Vector(ops[vecPos], class, vecPos+1); err != nil { return nil, err } @@ -1022,16 +1046,16 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"}, {Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"}, {Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value", - Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSH2SS (EVEX.NDS.LIG.MAP6.W0 13 /r)"}, {Name: "VCVTSH2SD", Summary: "Convert a low FP16 value to an FP64 value", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"}, {Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"}, {Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value", Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16, @@ -1046,16 +1070,16 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"}, {Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W0 2D /r)"}, {Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 64-bit integer", - Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W1 2D /r)"}, {Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 32-bit integer", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W0 79 /r)"}, {Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 64-bit integer", - Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W1 79 /r)"}, // AVX512-FP16 packed arithmetic: the full ZMM lanes the scalar core @@ -1148,4 +1172,145 @@ var amd64Extensions = []ExtInstr{ {Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control", Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"}, + + // AVX512-FP16 packed conversions: the fourteen directions between the + // FP16 lanes and the integer and double-precision companions, three + // register widths each. L'L names the governing operand, whose class the + // form spells: the source on the narrowing converts (VCVTDQ2PH narrows + // its dword source to the half-width destination, VCVTQQ2PH and VCVTPD2PH + // to the quarter-width XMM destination), the destination on the widening + // ones (VCVTPH2DQ widens its half-width source, VCVTPH2QQ and VCVTPH2PD + // their quarter-width XMM source). The FP16-to-integer directions round + // and take Er on their 512-bit register forms; VCVTPH2PD widens exactly + // and takes none. The integer-to-FP16 sources read their elements from + // memory under the {1toN} broadcast, the FP16 sources their full-width + // vectors; every destination is write-masked. The encodings are + // transcribed from the SDM entries and pinned byte for byte against the + // local GNU assembler, whose 2.46 table matches the manual row for row. + {Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers", + Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.512.66.MAP5.W0 7D /r)"}, + {Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers", + Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.256.66.MAP5.W0 7D /r)"}, + {Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers", + Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.128.66.MAP5.W0 7D /r)"}, + {Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.512.NP.MAP5.W0 7D /r)"}, + {Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers", + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.256.NP.MAP5.W0 7D /r)"}, + {Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers", + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.128.NP.MAP5.W0 7D /r)"}, + {Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.512.F3.MAP5.W0 7D /r)"}, + {Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.256.F3.MAP5.W0 7D /r)"}, + {Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.128.F3.MAP5.W0 7D /r)"}, + {Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.512.F2.MAP5.W0 7D /r)"}, + {Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.256.F2.MAP5.W0 7D /r)"}, + {Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.128.F2.MAP5.W0 7D /r)"}, + {Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.512.66.MAP5.W0 5B /r, half-width source)"}, + {Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.256.66.MAP5.W0 5B /r, half-width source)"}, + {Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.128.66.MAP5.W0 5B /r, half-width source)"}, + {Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.512.NP.MAP5.W0 79 /r, half-width source)"}, + {Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.256.NP.MAP5.W0 79 /r, half-width source)"}, + {Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.128.NP.MAP5.W0 79 /r, half-width source)"}, + {Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.512.NP.MAP5.W0 5B /r, YMM destination)"}, + {Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination", + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.256.NP.MAP5.W0 5B /r, XMM destination)"}, + {Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination", + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.128.NP.MAP5.W0 5B /r, XMM destination)"}, + {Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination", + Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.512.F2.MAP5.W0 7A /r, YMM destination)"}, + {Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination", + Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.256.F2.MAP5.W0 7A /r, XMM destination)"}, + {Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination", + Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.128.F2.MAP5.W0 7A /r, XMM destination)"}, + {Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.512.66.MAP5.W0 7B /r, quarter-width source)"}, + {Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.256.66.MAP5.W0 7B /r, quarter-width source)"}, + {Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.128.66.MAP5.W0 7B /r, quarter-width source)"}, + {Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.512.66.MAP5.W0 79 /r, quarter-width source)"}, + {Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.256.66.MAP5.W0 79 /r, quarter-width source)"}, + {Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened", + Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.128.66.MAP5.W0 79 /r, quarter-width source)"}, + {Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x84, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.512.NP.MAP5.W1 5B /r, XMM destination)"}, + {Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x84, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.256.NP.MAP5.W1 5B /r, XMM destination)"}, + {Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x84, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.128.NP.MAP5.W1 5B /r, XMM destination)"}, + {Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x87, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.512.F2.MAP5.W1 7A /r, XMM destination)"}, + {Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x87, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.256.F2.MAP5.W1 7A /r, XMM destination)"}, + {Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.128.F2.MAP5.W1 7A /r, XMM destination)"}, + {Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.512.NP.MAP5.W0 5A /r, quarter-width source)"}, + {Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly", + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.256.NP.MAP5.W0 5A /r, quarter-width source)"}, + {Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly", + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.128.NP.MAP5.W0 5A /r, quarter-width source)"}, + {Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x85, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.512.66.MAP5.W1 5A /r, XMM destination)"}, + {Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x85, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.256.66.MAP5.W1 5A /r, XMM destination)"}, + {Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination", + Bytes: []byte{0x62, 0x05, 0x85, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.128.66.MAP5.W1 5A /r, XMM destination)"}, } diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index a6122dc..f2ba36d 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -660,6 +660,396 @@ var amd64GoldenRows = []amd64GoldenRow{ {"vcvtusi2sh rd-sae, 64-bit, high registers", "VCVTUSI2SH", []ExtOperand{ExtXmm(29), ExtGpr64(12), ExtRounded(ExtXmm(30), ExtRoundDown)}, "624596307bf4", "62 45 96 30 7b f4 vcvtusi2sh %r12,{rd-sae},%xmm29,%xmm30"}, + + // The FP16 packed conversions, the fourteen directions between the FP16 + // lanes and the integer and double-precision companions. The register + // rows and the rounding rows are quoted from the local GNU assembler's + // output; the memory rows follow the listing's full-width and broadcast + // spellings, whose source displacements are N times the plain SDM + // displacement the disp8 bytes carry under the assembler's compressed + // displacement scaling, and the rows with no GNU line are derived: the + // plain full-width load of a narrowing convert the AT&T syntax cannot + // spell without a size hint, laid out by the broadcast row minus EVEX.b. + {"vcvtph2w zmm", "VCVTPH2W", + []ExtOperand{ExtZmm(5), ExtZmm(6)}, + "62f57d487df5", "62 f5 7d 48 7d f5 vcvtph2w %zmm5,%zmm6"}, + {"vcvtph2w ymm", "VCVTPH2W", + []ExtOperand{ExtYmm(5), ExtYmm(6)}, + "62f57d287df5", "62 f5 7d 28 7d f5 vcvtph2w %ymm5,%ymm6"}, + {"vcvtph2w xmm", "VCVTPH2W", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57d087df5", "62 f5 7d 08 7d f5 vcvtph2w %xmm5,%xmm6"}, + {"vcvtph2uw zmm", "VCVTPH2UW", + []ExtOperand{ExtZmm(5), ExtZmm(6)}, + "62f57c487df5", "62 f5 7c 48 7d f5 vcvtph2uw %zmm5,%zmm6"}, + {"vcvtph2uw ymm", "VCVTPH2UW", + []ExtOperand{ExtYmm(5), ExtYmm(6)}, + "62f57c287df5", "62 f5 7c 28 7d f5 vcvtph2uw %ymm5,%ymm6"}, + {"vcvtph2uw xmm", "VCVTPH2UW", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57c087df5", "62 f5 7c 08 7d f5 vcvtph2uw %xmm5,%xmm6"}, + {"vcvtw2ph zmm", "VCVTW2PH", + []ExtOperand{ExtZmm(5), ExtZmm(6)}, + "62f57e487df5", "62 f5 7e 48 7d f5 vcvtw2ph %zmm5,%zmm6"}, + {"vcvtw2ph ymm", "VCVTW2PH", + []ExtOperand{ExtYmm(5), ExtYmm(6)}, + "62f57e287df5", "62 f5 7e 28 7d f5 vcvtw2ph %ymm5,%ymm6"}, + {"vcvtw2ph xmm", "VCVTW2PH", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57e087df5", "62 f5 7e 08 7d f5 vcvtw2ph %xmm5,%xmm6"}, + {"vcvtuw2ph zmm", "VCVTUW2PH", + []ExtOperand{ExtZmm(5), ExtZmm(6)}, + "62f57f487df5", "62 f5 7f 48 7d f5 vcvtuw2ph %zmm5,%zmm6"}, + {"vcvtuw2ph ymm", "VCVTUW2PH", + []ExtOperand{ExtYmm(5), ExtYmm(6)}, + "62f57f287df5", "62 f5 7f 28 7d f5 vcvtuw2ph %ymm5,%ymm6"}, + {"vcvtuw2ph xmm", "VCVTUW2PH", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57f087df5", "62 f5 7f 08 7d f5 vcvtuw2ph %xmm5,%xmm6"}, + {"vcvtph2dq zmm", "VCVTPH2DQ", + []ExtOperand{ExtYmm(5), ExtZmm(6)}, + "62f57d485bf5", "62 f5 7d 48 5b f5 vcvtph2dq %ymm5,%zmm6"}, + {"vcvtph2dq ymm", "VCVTPH2DQ", + []ExtOperand{ExtXmm(5), ExtYmm(6)}, + "62f57d285bf5", "62 f5 7d 28 5b f5 vcvtph2dq %xmm5,%ymm6"}, + {"vcvtph2dq xmm", "VCVTPH2DQ", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57d085bf5", "62 f5 7d 08 5b f5 vcvtph2dq %xmm5,%xmm6"}, + {"vcvtph2udq zmm", "VCVTPH2UDQ", + []ExtOperand{ExtYmm(5), ExtZmm(6)}, + "62f57c4879f5", "62 f5 7c 48 79 f5 vcvtph2udq %ymm5,%zmm6"}, + {"vcvtph2udq ymm", "VCVTPH2UDQ", + []ExtOperand{ExtXmm(5), ExtYmm(6)}, + "62f57c2879f5", "62 f5 7c 28 79 f5 vcvtph2udq %xmm5,%ymm6"}, + {"vcvtph2udq xmm", "VCVTPH2UDQ", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57c0879f5", "62 f5 7c 08 79 f5 vcvtph2udq %xmm5,%xmm6"}, + {"vcvtdq2ph zmm to ymm", "VCVTDQ2PH", + []ExtOperand{ExtZmm(5), ExtYmm(6)}, + "62f57c485bf5", "62 f5 7c 48 5b f5 vcvtdq2ph %zmm5,%ymm6"}, + {"vcvtdq2ph ymm to xmm", "VCVTDQ2PH", + []ExtOperand{ExtYmm(5), ExtXmm(6)}, + "62f57c285bf5", "62 f5 7c 28 5b f5 vcvtdq2ph %ymm5,%xmm6"}, + {"vcvtdq2ph xmm to xmm", "VCVTDQ2PH", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57c085bf5", "62 f5 7c 08 5b f5 vcvtdq2ph %xmm5,%xmm6"}, + {"vcvtudq2ph zmm to ymm", "VCVTUDQ2PH", + []ExtOperand{ExtZmm(5), ExtYmm(6)}, + "62f57f487af5", "62 f5 7f 48 7a f5 vcvtudq2ph %zmm5,%ymm6"}, + {"vcvtudq2ph ymm to xmm", "VCVTUDQ2PH", + []ExtOperand{ExtYmm(5), ExtXmm(6)}, + "62f57f287af5", "62 f5 7f 28 7a f5 vcvtudq2ph %ymm5,%xmm6"}, + {"vcvtudq2ph xmm to xmm", "VCVTUDQ2PH", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57f087af5", "62 f5 7f 08 7a f5 vcvtudq2ph %xmm5,%xmm6"}, + {"vcvtph2qq zmm", "VCVTPH2QQ", + []ExtOperand{ExtXmm(5), ExtZmm(6)}, + "62f57d487bf5", "62 f5 7d 48 7b f5 vcvtph2qq %xmm5,%zmm6"}, + {"vcvtph2qq ymm", "VCVTPH2QQ", + []ExtOperand{ExtXmm(5), ExtYmm(6)}, + "62f57d287bf5", "62 f5 7d 28 7b f5 vcvtph2qq %xmm5,%ymm6"}, + {"vcvtph2qq xmm", "VCVTPH2QQ", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57d087bf5", "62 f5 7d 08 7b f5 vcvtph2qq %xmm5,%xmm6"}, + {"vcvtph2uqq zmm", "VCVTPH2UQQ", + []ExtOperand{ExtXmm(5), ExtZmm(6)}, + "62f57d4879f5", "62 f5 7d 48 79 f5 vcvtph2uqq %xmm5,%zmm6"}, + {"vcvtph2uqq ymm", "VCVTPH2UQQ", + []ExtOperand{ExtXmm(5), ExtYmm(6)}, + "62f57d2879f5", "62 f5 7d 28 79 f5 vcvtph2uqq %xmm5,%ymm6"}, + {"vcvtph2uqq xmm", "VCVTPH2UQQ", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57d0879f5", "62 f5 7d 08 79 f5 vcvtph2uqq %xmm5,%xmm6"}, + {"vcvtqq2ph zmm to xmm", "VCVTQQ2PH", + []ExtOperand{ExtZmm(5), ExtXmm(6)}, + "62f5fc485bf5", "62 f5 fc 48 5b f5 vcvtqq2ph %zmm5,%xmm6"}, + {"vcvtqq2ph ymm to xmm", "VCVTQQ2PH", + []ExtOperand{ExtYmm(5), ExtXmm(6)}, + "62f5fc285bf5", "62 f5 fc 28 5b f5 vcvtqq2ph %ymm5,%xmm6"}, + {"vcvtqq2ph xmm to xmm", "VCVTQQ2PH", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f5fc085bf5", "62 f5 fc 08 5b f5 vcvtqq2ph %xmm5,%xmm6"}, + {"vcvtuqq2ph zmm to xmm", "VCVTUQQ2PH", + []ExtOperand{ExtZmm(5), ExtXmm(6)}, + "62f5ff487af5", "62 f5 ff 48 7a f5 vcvtuqq2ph %zmm5,%xmm6"}, + {"vcvtuqq2ph ymm to xmm", "VCVTUQQ2PH", + []ExtOperand{ExtYmm(5), ExtXmm(6)}, + "62f5ff287af5", "62 f5 ff 28 7a f5 vcvtuqq2ph %ymm5,%xmm6"}, + {"vcvtuqq2ph xmm to xmm", "VCVTUQQ2PH", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f5ff087af5", "62 f5 ff 08 7a f5 vcvtuqq2ph %xmm5,%xmm6"}, + {"vcvtph2pd zmm", "VCVTPH2PD", + []ExtOperand{ExtXmm(5), ExtZmm(6)}, + "62f57c485af5", "62 f5 7c 48 5a f5 vcvtph2pd %xmm5,%zmm6"}, + {"vcvtph2pd ymm", "VCVTPH2PD", + []ExtOperand{ExtXmm(5), ExtYmm(6)}, + "62f57c285af5", "62 f5 7c 28 5a f5 vcvtph2pd %xmm5,%ymm6"}, + {"vcvtph2pd xmm", "VCVTPH2PD", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f57c085af5", "62 f5 7c 08 5a f5 vcvtph2pd %xmm5,%xmm6"}, + {"vcvtpd2ph zmm to xmm", "VCVTPD2PH", + []ExtOperand{ExtZmm(5), ExtXmm(6)}, + "62f5fd485af5", "62 f5 fd 48 5a f5 vcvtpd2ph %zmm5,%xmm6"}, + {"vcvtpd2ph ymm to xmm", "VCVTPD2PH", + []ExtOperand{ExtYmm(5), ExtXmm(6)}, + "62f5fd285af5", "62 f5 fd 28 5a f5 vcvtpd2ph %ymm5,%xmm6"}, + {"vcvtpd2ph xmm to xmm", "VCVTPD2PH", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f5fd085af5", "62 f5 fd 08 5a f5 vcvtpd2ph %xmm5,%xmm6"}, + + // The rounding rows of the converting directions: the FP16-to-integer + // and the integer-to-FP16 conversions round and take EVEX.RC on their + // 512-bit register forms, exactly as the arithmetic does, and VCVTPH2PD, + // which widens exactly, takes none. + {"vcvtph2w rn-sae", "VCVTPH2W", + []ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f57d187df5", "62 f5 7d 18 7d f5 vcvtph2w {rn-sae},%zmm5,%zmm6"}, + {"vcvtph2uw rz-sae", "VCVTPH2UW", + []ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)}, + "62f57c787df5", "62 f5 7c 78 7d f5 vcvtph2uw {rz-sae},%zmm5,%zmm6"}, + {"vcvtph2dq rn-sae", "VCVTPH2DQ", + []ExtOperand{ExtYmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f57d185bf5", "62 f5 7d 18 5b f5 vcvtph2dq {rn-sae},%ymm5,%zmm6"}, + {"vcvtph2udq rn-sae", "VCVTPH2UDQ", + []ExtOperand{ExtYmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f57c1879f5", "62 f5 7c 18 79 f5 vcvtph2udq {rn-sae},%ymm5,%zmm6"}, + {"vcvtph2qq rz-sae", "VCVTPH2QQ", + []ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)}, + "62f57d787bf5", "62 f5 7d 78 7b f5 vcvtph2qq {rz-sae},%xmm5,%zmm6"}, + {"vcvtph2uqq rz-sae", "VCVTPH2UQQ", + []ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)}, + "62f57d7879f5", "62 f5 7d 78 79 f5 vcvtph2uqq {rz-sae},%xmm5,%zmm6"}, + {"vcvtw2ph rz-sae", "VCVTW2PH", + []ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundTruncate)}, + "62f57e787df5", "62 f5 7e 78 7d f5 vcvtw2ph {rz-sae},%zmm5,%zmm6"}, + {"vcvtuw2ph rn-sae", "VCVTUW2PH", + []ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f57f187df5", "62 f5 7f 18 7d f5 vcvtuw2ph {rn-sae},%zmm5,%zmm6"}, + {"vcvtdq2ph rn-sae", "VCVTDQ2PH", + []ExtOperand{ExtZmm(5), ExtRounded(ExtYmm(6), ExtRoundNearest)}, + "62f57c185bf5", "62 f5 7c 18 5b f5 vcvtdq2ph {rn-sae},%zmm5,%ymm6"}, + {"vcvtudq2ph ru-sae", "VCVTUDQ2PH", + []ExtOperand{ExtZmm(5), ExtRounded(ExtYmm(6), ExtRoundUp)}, + "62f57f587af5", "62 f5 7f 58 7a f5 vcvtudq2ph {ru-sae},%zmm5,%ymm6"}, + {"vcvtqq2ph rz-sae", "VCVTQQ2PH", + []ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundTruncate)}, + "62f5fc785bf5", "62 f5 fc 78 5b f5 vcvtqq2ph {rz-sae},%zmm5,%xmm6"}, + {"vcvtuqq2ph rd-sae", "VCVTUQQ2PH", + []ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundDown)}, + "62f5ff387af5", "62 f5 ff 38 7a f5 vcvtuqq2ph {rd-sae},%zmm5,%xmm6"}, + {"vcvtpd2ph rz-sae", "VCVTPD2PH", + []ExtOperand{ExtZmm(5), ExtRounded(ExtXmm(6), ExtRoundTruncate)}, + "62f5fd785af5", "62 f5 fd 78 5a f5 vcvtpd2ph {rz-sae},%zmm5,%xmm6"}, + + // High registers across the conversion shapes, exercising every EVEX + // extension bit: the source above 15 clears B bar and X bar, the + // destination above 15 R bar. + {"vcvtph2w, high registers", "VCVTPH2W", + []ExtOperand{ExtZmm(28), ExtZmm(29)}, + "62057d487dec", "62 05 7d 48 7d ec vcvtph2w %zmm28,%zmm29"}, + {"vcvtph2dq, high registers", "VCVTPH2DQ", + []ExtOperand{ExtYmm(28), ExtZmm(29)}, + "62057d485bec", "62 05 7d 48 5b ec vcvtph2dq %ymm28,%zmm29"}, + {"vcvtdq2ph, high registers", "VCVTDQ2PH", + []ExtOperand{ExtZmm(29), ExtYmm(30)}, + "62057c485bf5", "62 05 7c 48 5b f5 vcvtdq2ph %zmm29,%ymm30"}, + {"vcvtph2qq, high registers", "VCVTPH2QQ", + []ExtOperand{ExtXmm(28), ExtZmm(29)}, + "62057d487bec", "62 05 7d 48 7b ec vcvtph2qq %xmm28,%zmm29"}, + {"vcvtpd2ph, high registers", "VCVTPD2PH", + []ExtOperand{ExtZmm(28), ExtXmm(29)}, + "6205fd485aec", "62 05 fd 48 5a ec vcvtpd2ph %zmm28,%xmm29"}, + + // The memory forms of the conversions: the FP16 sources read full-width + // vectors, the integer sources their elements under the broadcast. The + // disp8 rows quote the listing's compressed spellings, whose source is N + // times the plain displacement the bytes carry. + {"vcvtph2w memory source", "VCVTPH2W", + []ExtOperand{ExtMemory(9, 0), ExtZmm(30)}, + "62457d487d31", "62 45 7d 48 7d 31 vcvtph2w (%r9),%zmm30"}, + {"vcvtph2w memory source disp8", "VCVTPH2W", + []ExtOperand{ExtMemory(1, 127), ExtZmm(30)}, + "62657d487d717f", "62 65 7d 48 7d 71 7f vcvtph2w 0x1fc0(%rcx),%zmm30 (Disp8(7f))"}, + {"vcvtph2w ymm memory source", "VCVTPH2W", + []ExtOperand{ExtMemory(9, 0), ExtYmm(30)}, + "62457d287d31", "62 45 7d 28 7d 31 vcvtph2w (%r9),%ymm30"}, + {"vcvtph2w xmm memory source", "VCVTPH2W", + []ExtOperand{ExtMemory(9, 0), ExtXmm(30)}, + "62457d087d31", "62 45 7d 08 7d 31 vcvtph2w (%r9),%xmm30"}, + {"vcvtph2uw memory source", "VCVTPH2UW", + []ExtOperand{ExtMemory(9, 0), ExtZmm(30)}, + "62457c487d31", "62 45 7c 48 7d 31 vcvtph2uw (%r9),%zmm30"}, + {"vcvtw2ph memory source", "VCVTW2PH", + []ExtOperand{ExtMemory(9, 0), ExtZmm(30)}, + "62457e487d31", "62 45 7e 48 7d 31 vcvtw2ph (%r9),%zmm30"}, + {"vcvtw2ph ymm memory source", "VCVTW2PH", + []ExtOperand{ExtMemory(9, 0), ExtYmm(30)}, + "62457e287d31", "62 45 7e 28 7d 31 vcvtw2ph (%r9),%ymm30"}, + {"vcvtph2dq memory source", "VCVTPH2DQ", + []ExtOperand{ExtMemory(9, 0), ExtZmm(30)}, + "62457d485b31", "62 45 7d 48 5b 31 vcvtph2dq (%r9),%zmm30"}, + {"vcvtph2dq memory source disp8", "VCVTPH2DQ", + []ExtOperand{ExtMemory(1, 127), ExtZmm(30)}, + "62657d485b717f", "62 65 7d 48 5b 71 7f vcvtph2dq 0xfe0(%rcx),%zmm30 (Disp8(7f))"}, + {"vcvtph2dq ymm memory source", "VCVTPH2DQ", + []ExtOperand{ExtMemory(9, 0), ExtYmm(30)}, + "62457d285b31", "62 45 7d 28 5b 31 vcvtph2dq (%r9),%ymm30"}, + {"vcvtph2udq memory source", "VCVTPH2UDQ", + []ExtOperand{ExtMemory(9, 0), ExtZmm(30)}, + "62457c487931", "62 45 7c 48 79 31 vcvtph2udq (%r9),%zmm30"}, + {"vcvtdq2ph memory source", "VCVTDQ2PH", + []ExtOperand{ExtMemory(9, 0), ExtYmm(30)}, + "62457c485b31", "62 45 7c 48 5b 31 vcvtdq2ph (%r9),%ymm30"}, + {"vcvtph2qq memory source", "VCVTPH2QQ", + []ExtOperand{ExtMemory(9, 0), ExtZmm(30)}, + "62457d487b31", "62 45 7d 48 7b 31 vcvtph2qq (%r9),%zmm30"}, + {"vcvtph2qq memory source disp8", "VCVTPH2QQ", + []ExtOperand{ExtMemory(1, 127), ExtZmm(30)}, + "62657d487b717f", "62 65 7d 48 7b 71 7f vcvtph2qq 0x7f0(%rcx),%zmm30 (Disp8(7f))"}, + {"vcvtph2qq ymm memory source", "VCVTPH2QQ", + []ExtOperand{ExtMemory(9, 0), ExtYmm(30)}, + "62457d287b31", "62 45 7d 28 7b 31 vcvtph2qq (%r9),%ymm30"}, + {"vcvtph2uqq memory source", "VCVTPH2UQQ", + []ExtOperand{ExtMemory(9, 0), ExtZmm(30)}, + "62457d487931", "62 45 7d 48 79 31 vcvtph2uqq (%r9),%zmm30"}, + {"vcvtph2pd memory source", "VCVTPH2PD", + []ExtOperand{ExtMemory(9, 0), ExtZmm(30)}, + "62457c485a31", "62 45 7c 48 5a 31 vcvtph2pd (%r9),%zmm30"}, + {"vcvtph2pd memory source disp8", "VCVTPH2PD", + []ExtOperand{ExtMemory(1, 127), ExtZmm(30)}, + "62657c485a717f", "62 65 7c 48 5a 71 7f vcvtph2pd 0x7f0(%rcx),%zmm30 (Disp8(7f))"}, + {"vcvtph2pd ymm memory source", "VCVTPH2PD", + []ExtOperand{ExtMemory(9, 0), ExtYmm(30)}, + "62457c285a31", "62 45 7c 28 5a 31 vcvtph2pd (%r9),%ymm30"}, + {"vcvtqq2ph plain memory source", "VCVTQQ2PH", + []ExtOperand{ExtMemory(9, 0), ExtXmm(30)}, + "6245fc485b31", ""}, + {"vcvtuqq2ph plain memory source", "VCVTUQQ2PH", + []ExtOperand{ExtMemory(9, 0), ExtXmm(30)}, + "6245ff487a31", ""}, + {"vcvtpd2ph plain memory source", "VCVTPD2PH", + []ExtOperand{ExtMemory(9, 0), ExtXmm(30)}, + "6245fd485a31", ""}, + + // The broadcast forms of the converting sources: a single element the + // hardware splats across the lanes, one word, dword or qword as the + // direction's input element spells it. + {"vcvtw2ph broadcast source", "VCVTW2PH", + []ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)}, + "62457e587d31", "62 45 7e 58 7d 31 vcvtw2ph (%r9){1to32},%zmm30"}, + {"vcvtw2ph broadcast source disp8", "VCVTW2PH", + []ExtOperand{ExtBroadcast(1, 127), ExtZmm(30)}, + "62657e587d717f", "62 65 7e 58 7d 71 7f vcvtw2ph 0xfe(%rcx){1to32},%zmm30 (Disp8(7f))"}, + {"vcvtuw2ph broadcast source", "VCVTUW2PH", + []ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)}, + "62457f587d31", "62 45 7f 58 7d 31 vcvtuw2ph (%r9){1to32},%zmm30"}, + {"vcvtuw2ph broadcast source disp8", "VCVTUW2PH", + []ExtOperand{ExtBroadcast(1, 127), ExtZmm(30)}, + "62657f587d717f", "62 65 7f 58 7d 71 7f vcvtuw2ph 0xfe(%rcx){1to32},%zmm30 (Disp8(7f))"}, + {"vcvtdq2ph broadcast source", "VCVTDQ2PH", + []ExtOperand{ExtBroadcast(9, 0), ExtYmm(30)}, + "62457c585b31", "62 45 7c 58 5b 31 vcvtdq2ph (%r9){1to16},%ymm30"}, + {"vcvtdq2ph broadcast source disp8", "VCVTDQ2PH", + []ExtOperand{ExtBroadcast(1, 127), ExtYmm(30)}, + "62657c585b717f", "62 65 7c 58 5b 71 7f vcvtdq2ph 0x1fc(%rcx){1to16},%ymm30 (Disp8(7f))"}, + {"vcvtudq2ph broadcast source", "VCVTUDQ2PH", + []ExtOperand{ExtBroadcast(9, 0), ExtYmm(30)}, + "62457f587a31", "62 45 7f 58 7a 31 vcvtudq2ph (%r9){1to16},%ymm30"}, + {"vcvtudq2ph broadcast source disp8", "VCVTUDQ2PH", + []ExtOperand{ExtBroadcast(1, 127), ExtYmm(30)}, + "62657f587a717f", "62 65 7f 58 7a 71 7f vcvtudq2ph 0x1fc(%rcx){1to16},%ymm30 (Disp8(7f))"}, + {"vcvtqq2ph broadcast source", "VCVTQQ2PH", + []ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)}, + "6245fc585b31", "62 45 fc 58 5b 31 vcvtqq2ph (%r9){1to8},%xmm30"}, + {"vcvtqq2ph broadcast source disp8", "VCVTQQ2PH", + []ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)}, + "6265fc585b717f", "62 65 fc 58 5b 71 7f vcvtqq2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"}, + // The 256-bit broadcast row shares the 512-bit row's operand list, both + // destinations being XMM, so the resolver cannot tell them apart; the + // register row above pins the 256-bit template alone. + {"vcvtuqq2ph broadcast source", "VCVTUQQ2PH", + []ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)}, + "6245ff587a31", "62 45 ff 58 7a 31 vcvtuqq2ph (%r9){1to8},%xmm30"}, + {"vcvtuqq2ph broadcast source disp8", "VCVTUQQ2PH", + []ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)}, + "6265ff587a717f", "62 65 ff 58 7a 71 7f vcvtuqq2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"}, + {"vcvtpd2ph broadcast source", "VCVTPD2PH", + []ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)}, + "6245fd585a31", "62 45 fd 58 5a 31 vcvtpd2ph (%r9){1to8},%xmm30"}, + {"vcvtpd2ph broadcast source disp8", "VCVTPD2PH", + []ExtOperand{ExtBroadcast(1, 127), ExtXmm(30)}, + "6265fd585a717f", "62 65 fd 58 5a 71 7f vcvtpd2ph 0x3f8(%rcx){1to8},%xmm30 (Disp8(7f))"}, + // The 256-bit broadcast row shares the 512-bit row's operand list, both + // destinations being XMM, so the resolver cannot tell them apart; the + // register row above pins the 256-bit template alone. + + // The write mask over the converting destinations, and one composed + // row: the broadcast, the mask and the zeroing on the one word. + {"vcvtph2w k7 zeroing", "VCVTPH2W", + []ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)}, + "62457dcf7d31", "62 45 7d cf 7d 31 vcvtph2w (%r9),%zmm30{%k7}{z}"}, + {"vcvtph2dq k7 zeroing, memory source", "VCVTPH2DQ", + []ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)}, + "62457dcf5b31", "62 45 7d cf 5b 31 vcvtph2dq (%r9),%zmm30{%k7}{z}"}, + {"vcvtph2qq k7, memory source", "VCVTPH2QQ", + []ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, false)}, + "62457d4f7b31", "62 45 7d 4f 7b 31 vcvtph2qq (%r9),%zmm30{%k7}"}, + {"vcvtph2pd k7, memory source", "VCVTPH2PD", + []ExtOperand{ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, false)}, + "62457c4f5a31", "62 45 7c 4f 5a 31 vcvtph2pd (%r9),%zmm30{%k7}"}, + {"vcvtw2ph k7 zeroing", "VCVTW2PH", + []ExtOperand{ExtZmm(5), ExtWriteMasked(ExtZmm(6), 7, true)}, + "62f57ecf7df5", "62 f5 7e cf 7d f5 vcvtw2ph %zmm5,%zmm6{%k7}{z}"}, + {"vcvtdq2ph k7 zeroing", "VCVTDQ2PH", + []ExtOperand{ExtZmm(5), ExtWriteMasked(ExtYmm(6), 7, true)}, + "62f57ccf5bf5", "62 f5 7c cf 5b f5 vcvtdq2ph %zmm5,%ymm6{%k7}{z}"}, + {"vcvtw2ph broadcast under k7 zeroing", "VCVTW2PH", + []ExtOperand{ExtBroadcast(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)}, + "62457edf7d31", "62 45 7e df 7d 31 vcvtw2ph (%r9){1to32},%zmm30{%k7}{z}"}, + + // The memory forms of the scalar conversions, the m16, m32 and m64 + // sources the manual spells beside each register form. The disp8 rows + // quote the listing's compressed spellings, whose source is N times the + // plain displacement, N the element size. + {"vcvtsh2ss memory source", "VCVTSH2SS", + []ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)}, + "62d654081331", "62 d6 54 08 13 31 vcvtsh2ss (%r9),%xmm5,%xmm6"}, + {"vcvtsh2ss memory source disp8", "VCVTSH2SS", + []ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)}, + "62f65408137101", "62 f6 54 08 13 71 01 vcvtsh2ss 0x2(%rcx),%xmm5,%xmm6 (Disp8(01))"}, + {"vcvtsh2sd memory source", "VCVTSH2SD", + []ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)}, + "62d556085a31", "62 d5 56 08 5a 31 vcvtsh2sd (%r9),%xmm5,%xmm6"}, + {"vcvtsh2sd memory source disp8", "VCVTSH2SD", + []ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)}, + "62f556085a7101", "62 f5 56 08 5a 71 01 vcvtsh2sd 0x2(%rcx),%xmm5,%xmm6 (Disp8(01))"}, + {"vcvtss2sh memory source", "VCVTSS2SH", + []ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)}, + "62d554081d31", "62 d5 54 08 1d 31 vcvtss2sh (%r9),%xmm5,%xmm6"}, + {"vcvtss2sh memory source disp8", "VCVTSS2SH", + []ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)}, + "62f554081d7101", "62 f5 54 08 1d 71 01 vcvtss2sh 0x4(%rcx),%xmm5,%xmm6 (Disp8(01))"}, + {"vcvtsd2sh memory source", "VCVTSD2SH", + []ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtXmm(6)}, + "62d5d7085a31", "62 d5 d7 08 5a 31 vcvtsd2sh (%r9),%xmm5,%xmm6"}, + {"vcvtsd2sh memory source disp8", "VCVTSD2SH", + []ExtOperand{ExtXmm(5), ExtMemory(1, 1), ExtXmm(6)}, + "62f5d7085a7101", "62 f5 d7 08 5a 71 01 vcvtsd2sh 0x8(%rcx),%xmm5,%xmm6 (Disp8(01))"}, + {"vcvtsh2si memory source", "VCVTSH2SI", + []ExtOperand{ExtMemory(9, 0), ExtGpr32(0)}, + "62d57e082d01", "62 d5 7e 08 2d 01 vcvtsh2si (%r9),%eax"}, + {"vcvtsh2si 64-bit memory source", "VCVTSH2SI", + []ExtOperand{ExtMemory(9, 0), ExtGpr64(12)}, + "6255fe082d21", "62 55 fe 08 2d 21 vcvtsh2si (%r9),%r12"}, + {"vcvtsh2si memory source disp8", "VCVTSH2SI", + []ExtOperand{ExtMemory(1, 1), ExtGpr32(0)}, + "62f57e082d4101", "62 f5 7e 08 2d 41 01 vcvtsh2si 0x2(%rcx),%eax (Disp8(01))"}, + {"vcvtsh2usi memory source", "VCVTSH2USI", + []ExtOperand{ExtMemory(9, 0), ExtGpr32(0)}, + "62d57e087901", "62 d5 7e 08 79 01 vcvtsh2usi (%r9),%eax"}, + {"vcvtsh2usi 64-bit memory source disp8", "VCVTSH2USI", + []ExtOperand{ExtMemory(1, 1), ExtGpr64(12)}, + "6275fe08796101", "62 75 fe 08 79 61 01 vcvtsh2usi 0x2(%rcx),%r12 (Disp8(01))"}, } // amd64ResolveEntry finds the table entry a golden row exercises: the entry @@ -951,6 +1341,12 @@ func TestAmd64ExtRejects(t *testing.T) { {"rounding over a memory source", "VADDSH", []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtRounded(ExtXmm(30), ExtRoundNearest)}, "the memory form takes no rounding control"}, + {"rounding over the scalar convert's memory source", "VCVTSS2SH", + []ExtOperand{ExtXmm(5), ExtMemory(9, 0), ExtRounded(ExtXmm(6), ExtRoundNearest)}, + "the memory form takes no rounding control"}, + {"rounding beside the converting memory source", "VCVTW2PH", + []ExtOperand{ExtMemory(9, 0), ExtRounded(ExtZmm(30), ExtRoundNearest)}, + "the memory form takes no rounding control"}, {"rounding on a source position", "VADDPH", []ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(4), ExtRoundUp), ExtZmm(6)}, "the position takes none"}, @@ -963,6 +1359,21 @@ func TestAmd64ExtRejects(t *testing.T) { {"rounding on an entry without the capability", "VMOVSH", []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundNearest)}, "the entry's destination takes none"}, + {"rounding on the exactly-widening convert", "VCVTPH2PD", + []ExtOperand{ExtXmm(5), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "the entry's destination takes none"}, + {"wrong source class on the widening convert", "VCVTPH2DQ", + []ExtOperand{ExtZmm(5), ExtZmm(6)}, + "wants a YMM register"}, + {"wrong source class on the quarter convert", "VCVTPH2QQ", + []ExtOperand{ExtYmm(5), ExtZmm(6)}, + "wants an XMM register"}, + {"wrong destination on the quarter-destination convert", "VCVTQQ2PH", + []ExtOperand{ExtZmm(5), ExtYmm(6)}, + "wants an XMM register"}, + {"broadcast on the full-width FP16 source", "VCVTPH2W", + []ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)}, + "the entry's memory operand takes none"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) @@ -1107,7 +1518,7 @@ func TestAmd64ExtArchBinding(t *testing.T) { t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) } } - if got := Extensions(AMD64); len(got) != 70 { - t.Errorf("the amd64 layer registers %d instructions, want 70", len(got)) + if got := Extensions(AMD64); len(got) != 112 { + t.Errorf("the amd64 layer registers %d instructions, want 112", len(got)) } } diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index 1ae6e8d..2ef8b5e 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -444,6 +444,18 @@ const ( // half-width companion of the source, equal at 128 bits: VCVTNEPS2BF16 // Y6, Z5. Operands: src, dest. ExtFormAmdVec2Half + // ExtFormAmdVec2Wide is the two-vector form whose source is the half-width + // companion of the destination, equal at 128 bits: VCVTPH2DQ Z6, Y5, where + // L'L names the destination. Operands: src, dest. + ExtFormAmdVec2Wide + // ExtFormAmdVec2Quarter is the two-vector form whose source is the + // quarter-width companion of the destination, the XMM class at every + // length: VCVTPH2QQ Z6, X5. Operands: src, dest. + ExtFormAmdVec2Quarter + // ExtFormAmdVec2ToQuarter is the two-vector form whose destination is the + // quarter-width companion of the source, the XMM class at every length: + // VCVTQQ2PH X6, Z5. Operands: src, dest. + ExtFormAmdVec2ToQuarter // ExtFormAmdMask2 is the form with an opmask destination and two vector // sources: VP2INTERSECTD K0, Z2, Z1. Operands: src1, src2, dest. ExtFormAmdMask2 @@ -668,7 +680,8 @@ func (f ExtForm) Arity() int { return 2 case ExtFormAmdVec3, ExtFormAmdMask2, ExtFormAmdVecGprVec: return 3 - case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdGprVec, ExtFormAmdVecGpr: + case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter, + ExtFormAmdVec2ToQuarter, ExtFormAmdGprVec, ExtFormAmdVecGpr: return 2 case ExtFormAmdVec3Imm, ExtFormAmdMask2Imm: return 4 @@ -777,6 +790,12 @@ func (f ExtForm) String() string { return "two vectors" case ExtFormAmdVec2Half: return "two vectors, half-width destination" + case ExtFormAmdVec2Wide: + return "two vectors, half-width source" + case ExtFormAmdVec2Quarter: + return "two vectors, quarter-width source" + case ExtFormAmdVec2ToQuarter: + return "two vectors, quarter-width destination" case ExtFormAmdMask2: return "two vectors into an opmask" case ExtFormAmdVecGprVec: @@ -1037,7 +1056,8 @@ func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) { return in.encodeSignedImmediate(ops) case ExtFormAmdVec3, ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdMask2, ExtFormAmdVecGprVec, ExtFormAmdGprVec, ExtFormAmdVecGpr, - ExtFormAmdVec3Imm, ExtFormAmdMask2Imm, ExtFormAmdMemVec, ExtFormAmdVecMem: + ExtFormAmdVec3Imm, ExtFormAmdMask2Imm, ExtFormAmdMemVec, ExtFormAmdVecMem, + ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter, ExtFormAmdVec2ToQuarter: return in.encodeAmd64(ops) case ExtFormPredicateLogical, ExtFormPredicateSelect: return in.encodePredicateLogical(ops) diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index 57340be..6e6801f 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -41,6 +41,20 @@ func TestAmd64ExtensionRegistry(t *testing.T) { {"VGETMANTSH", 1}, {"VREDUCESH", 1}, {"VRNDSCALESH", 1}, + {"VCVTPH2W", 3}, + {"VCVTPH2UW", 3}, + {"VCVTW2PH", 3}, + {"VCVTUW2PH", 3}, + {"VCVTPH2DQ", 3}, + {"VCVTPH2UDQ", 3}, + {"VCVTDQ2PH", 3}, + {"VCVTUDQ2PH", 3}, + {"VCVTPH2QQ", 3}, + {"VCVTPH2UQQ", 3}, + {"VCVTQQ2PH", 3}, + {"VCVTUQQ2PH", 3}, + {"VCVTPH2PD", 3}, + {"VCVTPD2PH", 3}, } { cands, ok := LookupExtension(arch.AMD64, tt.mnem) if !ok { @@ -54,8 +68,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { t.Errorf("the %s lookup is not case-insensitive", tt.mnem) } } - if got := arch.Extensions(arch.AMD64); len(got) != 70 { - t.Errorf("the amd64 layer registers %d instructions, want 70", len(got)) + if got := arch.Extensions(arch.AMD64); len(got) != 112 { + t.Errorf("the amd64 layer registers %d instructions, want 112", len(got)) } if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { t.Error("a non-extended mnemonic resolved") @@ -131,6 +145,21 @@ func TestEncodeExtensionAmd64(t *testing.T) { {"scalar fp16 minimum with exceptions suppressed", "VMINSH", []arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundSAE)}, "62f556185df4"}, + {"fp16 to signed words", "VCVTPH2W", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(6)}, + "62f57d487df5"}, + {"dwords to fp16 over a broadcast source", "VCVTDQ2PH", + []arch.ExtOperand{arch.ExtBroadcast(9, 0), arch.ExtYmm(30)}, + "62457c585b31"}, + {"fp16 to signed qwords, rounded", "VCVTPH2QQ", + []arch.ExtOperand{arch.ExtXmm(5), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundTruncate)}, + "62f57d787bf5"}, + {"fp16 scalar out of memory into an integer", "VCVTSH2SI", + []arch.ExtOperand{arch.ExtMemory(9, 0), arch.ExtGpr64(12)}, + "6255fe082d21"}, + {"fp16 to double-precision, widened", "VCVTPH2PD", + []arch.ExtOperand{arch.ExtXmm(5), arch.ExtZmm(6)}, + "62f57c485af5"}, } { got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...) if err != nil {