From 03f9ef0ac6411410e9ccb6803e5f762f98b27c8f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 19:32:23 +0200 Subject: [PATCH] feat(arch): add the FP16 complex fused multiply-add to the extension layer Assisted-by: GLM 5.3 Flash --- arch/amd64_ext.go | 33 +++++++++++++++ arch/amd64_ext_test.go | 82 ++++++++++++++++++++++++++++++++++++- asm/extension_amd64_test.go | 17 +++++++- 3 files changed, 128 insertions(+), 4 deletions(-) diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index 9fea0b3..f29f94e 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -1579,6 +1579,39 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VFCMULCSH (EVEX.NDS.LIG.F2.MAP6.W0 D7 /r)"}, + // AVX512-FP16 complex fused multiply-add: the packed pair over the FP16 + // complex pairs, F3 prefixing the accumulate with the second source + // conjugated and F2 the accumulate with the first source conjugated, + // and their scalar mirrors at 57. The word sums three complex values, + // so the memory shape reads its full-width vector plain and takes no + // broadcast, like the complex multiply above; the 512-bit register + // forms take the embedded rounding and the scalar forms read their m16 + // plain with the rounding beside them. + {Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source", + Bytes: []byte{0x62, 0x06, 0x06, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.512.F3.MAP6.W0 56 /r)"}, + {Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source", + Bytes: []byte{0x62, 0x06, 0x06, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.256.F3.MAP6.W0 56 /r)"}, + {Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source", + Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.128.F3.MAP6.W0 56 /r)"}, + {Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source", + Bytes: []byte{0x62, 0x06, 0x07, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.512.F2.MAP6.W0 56 /r)"}, + {Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source", + Bytes: []byte{0x62, 0x06, 0x07, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.256.F2.MAP6.W0 56 /r)"}, + {Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source", + Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.128.F2.MAP6.W0 56 /r)"}, + {Name: "VFMADDCSH", Summary: "Multiply-add scalar complex FP16 values, conjugating the second source", + Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDCSH (EVEX.NDS.LIG.F3.MAP6.W0 57 /r)"}, + {Name: "VFCMADDCSH", Summary: "Multiply-add scalar complex FP16 values, conjugating the first source", + Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFCMADDCSH (EVEX.NDS.LIG.F2.MAP6.W0 57 /r)"}, + // AVX512-FP16 minimum or maximum: VMINMAXPH, the per-lane selection // under the imm8 control the AVX512DQ double- and single-precision pair // carries into the half-precision set, one control byte over the lanes. diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 6e2fa74..d2ac54a 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -1370,6 +1370,75 @@ var amd64GoldenRows = []amd64GoldenRow{ []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)}, "62461745d631", "62 46 17 45 d6 31 vfcmulcph (%r9),%zmm29,%zmm30{%k5}"}, + // The complex fused multiply-add, EVEX.NDS.MAP6.W0 56 and 57: F3 + // conjugating the second source and F2 the first, the scalar mirrors + // sharing the opcode one step up. The memory shape reads its + // full-width vector plain, a complex value needing both of its halves + // beside one another, so no broadcast spelling exists to encode. + {"vfmaddcph", "VFMADDCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "6206164056f4", "62 06 16 40 56 f4 vfmaddcph %zmm28,%zmm29,%zmm30"}, + {"vfmaddcph ymm", "VFMADDCPH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f6562856f4", "62 f6 56 28 56 f4 vfmaddcph %ymm4,%ymm5,%ymm6"}, + {"vfmaddcph xmm", "VFMADDCPH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f6560856f4", "62 f6 56 08 56 f4 vfmaddcph %xmm4,%xmm5,%xmm6"}, + {"vfcmaddcph", "VFCMADDCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "6206174056f4", "62 06 17 40 56 f4 vfcmaddcph %zmm28,%zmm29,%zmm30"}, + {"vfcmaddcph ymm", "VFCMADDCPH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f6572856f4", "62 f6 57 28 56 f4 vfcmaddcph %ymm4,%ymm5,%ymm6"}, + {"vfcmaddcph xmm", "VFCMADDCPH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f6570856f4", "62 f6 57 08 56 f4 vfcmaddcph %xmm4,%xmm5,%xmm6"}, + {"vfmaddcsh", "VFMADDCSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6206160057f4", "62 06 16 00 57 f4 vfmaddcsh %xmm28,%xmm29,%xmm30"}, + {"vfcmaddcsh", "VFCMADDCSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6206170057f4", "62 06 17 00 57 f4 vfcmaddcsh %xmm28,%xmm29,%xmm30"}, + {"vfmaddcph memory source", "VFMADDCPH", + []ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)}, + "6266164056717f", "62 66 16 40 56 71 7f vfmaddcph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"}, + {"vfcmaddcph memory source", "VFCMADDCPH", + []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)}, + "624617405631", "62 46 17 40 56 31 vfcmaddcph (%r9),%zmm29,%zmm30"}, + {"vfmaddcsh memory source", "VFMADDCSH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624616005731", "62 46 16 00 57 31 vfmaddcsh (%r9),%xmm29,%xmm30"}, + {"vfcmaddcsh memory source", "VFCMADDCSH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624617005731", "62 46 17 00 57 31 vfcmaddcsh (%r9),%xmm29,%xmm30"}, + {"vfmaddcsh memory source disp32", "VFMADDCSH", + []ExtOperand{ExtXmm(29), ExtMemory(1, 8128), ExtXmm(30)}, + "6266160057b1c01f0000", "62 66 16 00 57 b1 c0 1f 00 00 vfmaddcsh 0x1fc0(%rcx),%xmm29,%xmm30"}, + // The scalar complex shapes leave the small displacements alone: the + // local assembler never compresses them to disp8, so the disp8 row is + // derived from the proven zero-displacement row above. + {"vfmaddcsh memory source disp8", "VFMADDCSH", + []ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)}, + "6266160057717f", ""}, + {"vfmaddcph rz-sae", "VFMADDCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtRounded(ExtZmm(30), ExtRoundTruncate)}, + "6206167056f4", "62 06 16 70 56 f4 vfmaddcph {rz-sae},%zmm28,%zmm29,%zmm30"}, + {"vfcmaddcph rn-sae", "VFCMADDCPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f6571856f4", "62 f6 57 18 56 f4 vfcmaddcph {rn-sae},%zmm4,%zmm5,%zmm6"}, + {"vfmaddcsh rz-sae", "VFMADDCSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundTruncate)}, + "6206167057f4", "62 06 16 70 57 f4 vfmaddcsh {rz-sae},%xmm28,%xmm29,%xmm30"}, + {"vfcmaddcsh rn-sae, high registers", "VFCMADDCSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundNearest)}, + "6206171057f4", "62 06 17 10 57 f4 vfcmaddcsh {rn-sae},%xmm28,%xmm29,%xmm30"}, + {"vfmaddcph k7 zeroing", "VFMADDCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)}, + "620616c756f4", "62 06 16 c7 56 f4 vfmaddcph %zmm28,%zmm29,%zmm30{%k7}{z}"}, + {"vfcmaddcph k5 merging, memory source", "VFCMADDCPH", + []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)}, + "624617455631", ""}, + // VMINMAXPH, EVEX.NDS.0F3A.W0 52 /r /ib: the control byte leads the // operand list and rides last in the word, the order the reference // listings write both in. @@ -1727,6 +1796,15 @@ func TestAmd64ExtRejects(t *testing.T) { {"broadcast on the complex multiply", "VFMULCPH", []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, "the entry's memory operand takes none"}, + {"broadcast on the complex multiply-add", "VFMADDCPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "the entry's memory operand takes none"}, + {"rounding on the 256-bit complex multiply-add", "VFMADDCPH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)}, + "the entry's destination takes none"}, + {"rounding over the complex multiply-add's memory source", "VFMADDCPH", + []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtRounded(ExtZmm(30), ExtRoundNearest)}, + "the memory form takes no rounding control"}, {"rounding on the 256-bit multiply-add", "VFMADD132PH", []ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)}, "the entry's destination takes none"}, @@ -1883,7 +1961,7 @@ func TestAmd64ExtArchBinding(t *testing.T) { t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) } } - if got := Extensions(AMD64); len(got) != 174 { - t.Errorf("the amd64 layer registers %d instructions, want 174", len(got)) + if got := Extensions(AMD64); len(got) != 182 { + t.Errorf("the amd64 layer registers %d instructions, want 182", len(got)) } } diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index 3ed7b46..81a8861 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -80,6 +80,10 @@ func TestAmd64ExtensionRegistry(t *testing.T) { {"VFCMULCPH", 3}, {"VFMULCSH", 1}, {"VFCMULCSH", 1}, + {"VFMADDCPH", 3}, + {"VFCMADDCPH", 3}, + {"VFMADDCSH", 1}, + {"VFCMADDCSH", 1}, {"VMINMAXPH", 3}, } { cands, ok := LookupExtension(arch.AMD64, tt.mnem) @@ -94,8 +98,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { t.Errorf("the %s lookup is not case-insensitive", tt.mnem) } } - if got := arch.Extensions(arch.AMD64); len(got) != 174 { - t.Errorf("the amd64 layer registers %d instructions, want 174", len(got)) + if got := arch.Extensions(arch.AMD64); len(got) != 182 { + t.Errorf("the amd64 layer registers %d instructions, want 182", len(got)) } if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { t.Error("a non-extended mnemonic resolved") @@ -201,6 +205,15 @@ func TestEncodeExtensionAmd64(t *testing.T) { {"complex multiply, conjugating the first source", "VFCMULCPH", []arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)}, "62061740d6f4"}, + {"complex multiply-add, conjugating the second source", "VFMADDCPH", + []arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)}, + "6206164056f4"}, + {"complex multiply-add, conjugated first source, rounded", "VFCMADDCPH", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundNearest)}, + "62f6571856f4"}, + {"scalar complex multiply-add out of memory", "VFMADDCSH", + []arch.ExtOperand{arch.ExtXmm(29), arch.ExtMemory(9, 0), arch.ExtXmm(30)}, + "624616005731"}, {"minimum or maximum under a control byte", "VMINMAXPH", []arch.ExtOperand{arch.ExtImmediate(0x88), arch.ExtZmm(29), arch.ExtMemory(9, 0), arch.ExtZmm(30)}, "62431440523188"},