feat(arch): add the FP16 complex fused multiply-add to the extension layer

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 19:49:15 +02:00
1 parent 86cb785e58
commit 03f9ef0ac6
3 files changed
+128 -4

No files matched your search

+33
View File
@@ -1579,6 +1579,39 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCSH (EVEX.NDS.LIG.F2.MAP6.W0 D7 /r)"},
// AVX512-FP16 complex fused multiply-add: the packed pair over the FP16
// complex pairs, F3 prefixing the accumulate with the second source
// conjugated and F2 the accumulate with the first source conjugated,
// and their scalar mirrors at 57. The word sums three complex values,
// so the memory shape reads its full-width vector plain and takes no
// broadcast, like the complex multiply above; the 512-bit register
// forms take the embedded rounding and the scalar forms read their m16
// plain with the rounding beside them.
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.512.F3.MAP6.W0 56 /r)"},
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.256.F3.MAP6.W0 56 /r)"},
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.128.F3.MAP6.W0 56 /r)"},
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.512.F2.MAP6.W0 56 /r)"},
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.256.F2.MAP6.W0 56 /r)"},
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.128.F2.MAP6.W0 56 /r)"},
{Name: "VFMADDCSH", Summary: "Multiply-add scalar complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDCSH (EVEX.NDS.LIG.F3.MAP6.W0 57 /r)"},
{Name: "VFCMADDCSH", Summary: "Multiply-add scalar complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMADDCSH (EVEX.NDS.LIG.F2.MAP6.W0 57 /r)"},
// AVX512-FP16 minimum or maximum: VMINMAXPH, the per-lane selection
// under the imm8 control the AVX512DQ double- and single-precision pair
// carries into the half-precision set, one control byte over the lanes.
+80 -2
View File
@@ -1370,6 +1370,75 @@ var amd64GoldenRows = []amd64GoldenRow{
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)},
"62461745d631", "62 46 17 45 d6 31 vfcmulcph (%r9),%zmm29,%zmm30{%k5}"},
// The complex fused multiply-add, EVEX.NDS.MAP6.W0 56 and 57: F3
// conjugating the second source and F2 the first, the scalar mirrors
// sharing the opcode one step up. The memory shape reads its
// full-width vector plain, a complex value needing both of its halves
// beside one another, so no broadcast spelling exists to encode.
{"vfmaddcph", "VFMADDCPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
"6206164056f4", "62 06 16 40 56 f4 vfmaddcph %zmm28,%zmm29,%zmm30"},
{"vfmaddcph ymm", "VFMADDCPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f6562856f4", "62 f6 56 28 56 f4 vfmaddcph %ymm4,%ymm5,%ymm6"},
{"vfmaddcph xmm", "VFMADDCPH",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f6560856f4", "62 f6 56 08 56 f4 vfmaddcph %xmm4,%xmm5,%xmm6"},
{"vfcmaddcph", "VFCMADDCPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
"6206174056f4", "62 06 17 40 56 f4 vfcmaddcph %zmm28,%zmm29,%zmm30"},
{"vfcmaddcph ymm", "VFCMADDCPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f6572856f4", "62 f6 57 28 56 f4 vfcmaddcph %ymm4,%ymm5,%ymm6"},
{"vfcmaddcph xmm", "VFCMADDCPH",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f6570856f4", "62 f6 57 08 56 f4 vfcmaddcph %xmm4,%xmm5,%xmm6"},
{"vfmaddcsh", "VFMADDCSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6206160057f4", "62 06 16 00 57 f4 vfmaddcsh %xmm28,%xmm29,%xmm30"},
{"vfcmaddcsh", "VFCMADDCSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6206170057f4", "62 06 17 00 57 f4 vfcmaddcsh %xmm28,%xmm29,%xmm30"},
{"vfmaddcph memory source", "VFMADDCPH",
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
"6266164056717f", "62 66 16 40 56 71 7f vfmaddcph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
{"vfcmaddcph memory source", "VFCMADDCPH",
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)},
"624617405631", "62 46 17 40 56 31 vfcmaddcph (%r9),%zmm29,%zmm30"},
{"vfmaddcsh memory source", "VFMADDCSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624616005731", "62 46 16 00 57 31 vfmaddcsh (%r9),%xmm29,%xmm30"},
{"vfcmaddcsh memory source", "VFCMADDCSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624617005731", "62 46 17 00 57 31 vfcmaddcsh (%r9),%xmm29,%xmm30"},
{"vfmaddcsh memory source disp32", "VFMADDCSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 8128), ExtXmm(30)},
"6266160057b1c01f0000", "62 66 16 00 57 b1 c0 1f 00 00 vfmaddcsh 0x1fc0(%rcx),%xmm29,%xmm30"},
// The scalar complex shapes leave the small displacements alone: the
// local assembler never compresses them to disp8, so the disp8 row is
// derived from the proven zero-displacement row above.
{"vfmaddcsh memory source disp8", "VFMADDCSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
"6266160057717f", ""},
{"vfmaddcph rz-sae", "VFMADDCPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtRounded(ExtZmm(30), ExtRoundTruncate)},
"6206167056f4", "62 06 16 70 56 f4 vfmaddcph {rz-sae},%zmm28,%zmm29,%zmm30"},
{"vfcmaddcph rn-sae", "VFCMADDCPH",
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)},
"62f6571856f4", "62 f6 57 18 56 f4 vfcmaddcph {rn-sae},%zmm4,%zmm5,%zmm6"},
{"vfmaddcsh rz-sae", "VFMADDCSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundTruncate)},
"6206167057f4", "62 06 16 70 57 f4 vfmaddcsh {rz-sae},%xmm28,%xmm29,%xmm30"},
{"vfcmaddcsh rn-sae, high registers", "VFCMADDCSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundNearest)},
"6206171057f4", "62 06 17 10 57 f4 vfcmaddcsh {rn-sae},%xmm28,%xmm29,%xmm30"},
{"vfmaddcph k7 zeroing", "VFMADDCPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)},
"620616c756f4", "62 06 16 c7 56 f4 vfmaddcph %zmm28,%zmm29,%zmm30{%k7}{z}"},
{"vfcmaddcph k5 merging, memory source", "VFCMADDCPH",
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)},
"624617455631", ""},
// VMINMAXPH, EVEX.NDS.0F3A.W0 52 /r /ib: the control byte leads the
// operand list and rides last in the word, the order the reference
// listings write both in.
@@ -1727,6 +1796,15 @@ func TestAmd64ExtRejects(t *testing.T) {
{"broadcast on the complex multiply", "VFMULCPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"the entry's memory operand takes none"},
{"broadcast on the complex multiply-add", "VFMADDCPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"the entry's memory operand takes none"},
{"rounding on the 256-bit complex multiply-add", "VFMADDCPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)},
"the entry's destination takes none"},
{"rounding over the complex multiply-add's memory source", "VFMADDCPH",
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtRounded(ExtZmm(30), ExtRoundNearest)},
"the memory form takes no rounding control"},
{"rounding on the 256-bit multiply-add", "VFMADD132PH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)},
"the entry's destination takes none"},
@@ -1883,7 +1961,7 @@ func TestAmd64ExtArchBinding(t *testing.T) {
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
}
}
if got := Extensions(AMD64); len(got) != 174 {
t.Errorf("the amd64 layer registers %d instructions, want 174", len(got))
if got := Extensions(AMD64); len(got) != 182 {
t.Errorf("the amd64 layer registers %d instructions, want 182", len(got))
}
}
+15 -2
View File
@@ -80,6 +80,10 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
{"VFCMULCPH", 3},
{"VFMULCSH", 1},
{"VFCMULCSH", 1},
{"VFMADDCPH", 3},
{"VFCMADDCPH", 3},
{"VFMADDCSH", 1},
{"VFCMADDCSH", 1},
{"VMINMAXPH", 3},
} {
cands, ok := LookupExtension(arch.AMD64, tt.mnem)
@@ -94,8 +98,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
t.Errorf("the %s lookup is not case-insensitive", tt.mnem)
}
}
if got := arch.Extensions(arch.AMD64); len(got) != 174 {
t.Errorf("the amd64 layer registers %d instructions, want 174", len(got))
if got := arch.Extensions(arch.AMD64); len(got) != 182 {
t.Errorf("the amd64 layer registers %d instructions, want 182", len(got))
}
if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok {
t.Error("a non-extended mnemonic resolved")
@@ -201,6 +205,15 @@ func TestEncodeExtensionAmd64(t *testing.T) {
{"complex multiply, conjugating the first source", "VFCMULCPH",
[]arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)},
"62061740d6f4"},
{"complex multiply-add, conjugating the second source", "VFMADDCPH",
[]arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)},
"6206164056f4"},
{"complex multiply-add, conjugated first source, rounded", "VFCMADDCPH",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundNearest)},
"62f6571856f4"},
{"scalar complex multiply-add out of memory", "VFMADDCSH",
[]arch.ExtOperand{arch.ExtXmm(29), arch.ExtMemory(9, 0), arch.ExtXmm(30)},
"624616005731"},
{"minimum or maximum under a control byte", "VMINMAXPH",
[]arch.ExtOperand{arch.ExtImmediate(0x88), arch.ExtZmm(29), arch.ExtMemory(9, 0), arch.ExtZmm(30)},
"62431440523188"},