feat(arch): add the FP16 complex fused multiply-add to the extension layer
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
86cb785e58
commit
03f9ef0ac6
3 files changed
+128
-4
No files matched your search
@@ -1579,6 +1579,39 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFCMULCSH (EVEX.NDS.LIG.F2.MAP6.W0 D7 /r)"},
|
||||
|
||||
// AVX512-FP16 complex fused multiply-add: the packed pair over the FP16
|
||||
// complex pairs, F3 prefixing the accumulate with the second source
|
||||
// conjugated and F2 the accumulate with the first source conjugated,
|
||||
// and their scalar mirrors at 57. The word sums three complex values,
|
||||
// so the memory shape reads its full-width vector plain and takes no
|
||||
// broadcast, like the complex multiply above; the 512-bit register
|
||||
// forms take the embedded rounding and the scalar forms read their m16
|
||||
// plain with the rounding beside them.
|
||||
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
|
||||
Bytes: []byte{0x62, 0x06, 0x06, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.512.F3.MAP6.W0 56 /r)"},
|
||||
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
|
||||
Bytes: []byte{0x62, 0x06, 0x06, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.256.F3.MAP6.W0 56 /r)"},
|
||||
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
|
||||
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.128.F3.MAP6.W0 56 /r)"},
|
||||
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
|
||||
Bytes: []byte{0x62, 0x06, 0x07, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.512.F2.MAP6.W0 56 /r)"},
|
||||
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
|
||||
Bytes: []byte{0x62, 0x06, 0x07, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.256.F2.MAP6.W0 56 /r)"},
|
||||
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
|
||||
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.128.F2.MAP6.W0 56 /r)"},
|
||||
{Name: "VFMADDCSH", Summary: "Multiply-add scalar complex FP16 values, conjugating the second source",
|
||||
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDCSH (EVEX.NDS.LIG.F3.MAP6.W0 57 /r)"},
|
||||
{Name: "VFCMADDCSH", Summary: "Multiply-add scalar complex FP16 values, conjugating the first source",
|
||||
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFCMADDCSH (EVEX.NDS.LIG.F2.MAP6.W0 57 /r)"},
|
||||
|
||||
// AVX512-FP16 minimum or maximum: VMINMAXPH, the per-lane selection
|
||||
// under the imm8 control the AVX512DQ double- and single-precision pair
|
||||
// carries into the half-precision set, one control byte over the lanes.
|
||||
|
||||
Reference in new issue
Block a user