feat(arch): add the FP16 complex multiply and minimum-maximum to the extension layer

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 18:59:09 +02:00
1 parent 8080e0acef
commit 509afbb6c9
3 files changed
+145 -4

No files matched your search

+49
View File
@@ -1545,4 +1545,53 @@ var amd64Extensions = []ExtInstr{
{Name: "VFMSUB231SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB231SH (EVEX.NDS.LIG.66.MAP6.W0 BB /r)"},
// AVX512-FP16 complex multiply: the packed pair over the FP16 complex
// pairs, F3 prefixing the multiply by the conjugate of the second
// source and F2 the multiply of the conjugate of the first source, and
// their scalar mirrors at D7. A complex value needs both of its halves
// beside one another, so the memory shape reads its full-width vector
// plain and takes no broadcast, unlike the packed FMA; the 512-bit
// register forms take the embedded rounding. The scalar forms read
// their m16 plain and take the rounding beside it.
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x40, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.512.F3.MAP6.W0 D6 /r)"},
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x20, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.256.F3.MAP6.W0 D6 /r)"},
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.128.F3.MAP6.W0 D6 /r)"},
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x40, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.512.F2.MAP6.W0 D6 /r)"},
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x20, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.256.F2.MAP6.W0 D6 /r)"},
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.128.F2.MAP6.W0 D6 /r)"},
{Name: "VFMULCSH", Summary: "Multiply scalar complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMULCSH (EVEX.NDS.LIG.F3.MAP6.W0 D7 /r)"},
{Name: "VFCMULCSH", Summary: "Multiply scalar complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCSH (EVEX.NDS.LIG.F2.MAP6.W0 D7 /r)"},
// AVX512-FP16 minimum or maximum: VMINMAXPH, the per-lane selection
// under the imm8 control the AVX512DQ double- and single-precision pair
// carries into the half-precision set, one control byte over the lanes.
// EVEX.NDS.0F3A.W0 52 /r /ib, the control byte riding last as the
// immediate operand it is. The destinations take the write mask and the
// memory shape of the second source reads its full-width vector plain.
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.512.0F3A.W0 52 /r /ib)"},
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.256.0F3A.W0 52 /r /ib)"},
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.128.0F3A.W0 52 /r /ib)"},
}