From 509afbb6c9dfdad28fcb6db7dfae992df2e18380 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 18:59:09 +0200 Subject: [PATCH] feat(arch): add the FP16 complex multiply and minimum-maximum to the extension layer Assisted-by: GLM 5.3 Flash --- arch/amd64_ext.go | 49 ++++++++++++++++++++ arch/amd64_ext_test.go | 91 ++++++++++++++++++++++++++++++++++++- asm/extension_amd64_test.go | 9 +++- 3 files changed, 145 insertions(+), 4 deletions(-) diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index d4990ca..9fea0b3 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -1545,4 +1545,53 @@ var amd64Extensions = []ExtInstr{ {Name: "VFMSUB231SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying the subtracted term", Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VFMSUB231SH (EVEX.NDS.LIG.66.MAP6.W0 BB /r)"}, + + // AVX512-FP16 complex multiply: the packed pair over the FP16 complex + // pairs, F3 prefixing the multiply by the conjugate of the second + // source and F2 the multiply of the conjugate of the first source, and + // their scalar mirrors at D7. A complex value needs both of its halves + // beside one another, so the memory shape reads its full-width vector + // plain and takes no broadcast, unlike the packed FMA; the 512-bit + // register forms take the embedded rounding. The scalar forms read + // their m16 plain and take the rounding beside it. + {Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source", + Bytes: []byte{0x62, 0x06, 0x06, 0x40, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.512.F3.MAP6.W0 D6 /r)"}, + {Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source", + Bytes: []byte{0x62, 0x06, 0x06, 0x20, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.256.F3.MAP6.W0 D6 /r)"}, + {Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source", + Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.128.F3.MAP6.W0 D6 /r)"}, + {Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source", + Bytes: []byte{0x62, 0x06, 0x07, 0x40, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.512.F2.MAP6.W0 D6 /r)"}, + {Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source", + Bytes: []byte{0x62, 0x06, 0x07, 0x20, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.256.F2.MAP6.W0 D6 /r)"}, + {Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source", + Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.128.F2.MAP6.W0 D6 /r)"}, + {Name: "VFMULCSH", Summary: "Multiply scalar complex FP16 values, conjugating the second source", + Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMULCSH (EVEX.NDS.LIG.F3.MAP6.W0 D7 /r)"}, + {Name: "VFCMULCSH", Summary: "Multiply scalar complex FP16 values, conjugating the first source", + Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFCMULCSH (EVEX.NDS.LIG.F2.MAP6.W0 D7 /r)"}, + + // AVX512-FP16 minimum or maximum: VMINMAXPH, the per-lane selection + // under the imm8 control the AVX512DQ double- and single-precision pair + // carries into the half-precision set, one control byte over the lanes. + // EVEX.NDS.0F3A.W0 52 /r /ib, the control byte riding last as the + // immediate operand it is. The destinations take the write mask and the + // memory shape of the second source reads its full-width vector plain. + {Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control", + Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.512.0F3A.W0 52 /r /ib)"}, + {Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control", + Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.256.0F3A.W0 52 /r /ib)"}, + {Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control", + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.128.0F3A.W0 52 /r /ib)"}, } diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index fda31e2..6e2fa74 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -1313,6 +1313,84 @@ var amd64GoldenRows = []amd64GoldenRow{ {"vfmaddsub132ph k5 merging, memory source", "VFMADDSUB132PH", []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)}, "624615459631", "62 46 15 45 96 31 vfmaddsub132ph (%r9),%zmm29,%zmm30{%k5}"}, + + // The complex multiply, EVEX.NDS.F3.MAP6.W0 D6 and EVEX.NDS.F2.MAP6.W0 + // D6, the scalar pair at D7. No broadcast: a complex value needs both + // of its halves beside one another, so the memory shape reads its + // full-width vector plain. + {"vfmulcph", "VFMULCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061640d6f4", "62 06 16 40 d6 f4 vfmulcph %zmm28,%zmm29,%zmm30"}, + {"vfmulcph ymm", "VFMULCPH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65628d6f4", "62 f6 56 28 d6 f4 vfmulcph %ymm4,%ymm5,%ymm6"}, + {"vfmulcph xmm", "VFMULCPH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65608d6f4", "62 f6 56 08 d6 f4 vfmulcph %xmm4,%xmm5,%xmm6"}, + {"vfcmulcph", "VFCMULCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061740d6f4", "62 06 17 40 d6 f4 vfcmulcph %zmm28,%zmm29,%zmm30"}, + {"vfcmulcph ymm", "VFCMULCPH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65728d6f4", "62 f6 57 28 d6 f4 vfcmulcph %ymm4,%ymm5,%ymm6"}, + {"vfcmulcph xmm", "VFCMULCPH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65708d6f4", "62 f6 57 08 d6 f4 vfcmulcph %xmm4,%xmm5,%xmm6"}, + {"vfmulcsh", "VFMULCSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "62061600d7f4", "62 06 16 00 d7 f4 vfmulcsh %xmm28,%xmm29,%xmm30"}, + {"vfcmulcsh", "VFCMULCSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "62061700d7f4", "62 06 17 00 d7 f4 vfcmulcsh %xmm28,%xmm29,%xmm30"}, + {"vfmulcph memory source", "VFMULCPH", + []ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)}, + "62661640d6717f", "62 66 16 40 d6 71 7f vfmulcph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"}, + {"vfcmulcph memory source", "VFCMULCPH", + []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)}, + "62461740d631", "62 46 17 40 d6 31 vfcmulcph (%r9),%zmm29,%zmm30"}, + {"vfmulcsh memory source", "VFMULCSH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "62461600d731", "62 46 16 00 d7 31 vfmulcsh (%r9),%xmm29,%xmm30"}, + {"vfmulcph rz-sae", "VFMULCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtRounded(ExtZmm(30), ExtRoundTruncate)}, + "62061670d6f4", "62 06 16 70 d6 f4 vfmulcph {rz-sae},%zmm28,%zmm29,%zmm30"}, + {"vfcmulcph rn-sae", "VFCMULCPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f65718d6f4", "62 f6 57 18 d6 f4 vfcmulcph {rn-sae},%zmm4,%zmm5,%zmm6"}, + {"vfmulcsh rn-sae", "VFMULCSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundNearest)}, + "62f65618d7f4", "62 f6 56 18 d7 f4 vfmulcsh {rn-sae},%xmm4,%xmm5,%xmm6"}, + {"vfcmulcsh ru-sae, high registers", "VFCMULCSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundUp)}, + "62061750d7f4", "62 06 17 50 d7 f4 vfcmulcsh {ru-sae},%xmm28,%xmm29,%xmm30"}, + {"vfmulcph k7 zeroing", "VFMULCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)}, + "620616c7d6f4", "62 06 16 c7 d6 f4 vfmulcph %zmm28,%zmm29,%zmm30{%k7}{z}"}, + {"vfcmulcph k5 merging, memory source", "VFCMULCPH", + []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)}, + "62461745d631", "62 46 17 45 d6 31 vfcmulcph (%r9),%zmm29,%zmm30{%k5}"}, + + // VMINMAXPH, EVEX.NDS.0F3A.W0 52 /r /ib: the control byte leads the + // operand list and rides last in the word, the order the reference + // listings write both in. + {"vminmaxph", "VMINMAXPH", + []ExtOperand{ExtImmediate(0x88), ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "6203144052f488", "62 03 14 40 52 f4 88 vminmaxph $0x88,%zmm28,%zmm29,%zmm30"}, + {"vminmaxph ymm", "VMINMAXPH", + []ExtOperand{ExtImmediate(0x88), ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f3542852f488", "62 f3 54 28 52 f4 88 vminmaxph $0x88,%ymm4,%ymm5,%ymm6"}, + {"vminmaxph xmm", "VMINMAXPH", + []ExtOperand{ExtImmediate(0x88), ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f3540852f488", "62 f3 54 08 52 f4 88 vminmaxph $0x88,%xmm4,%xmm5,%xmm6"}, + {"vminmaxph memory source", "VMINMAXPH", + []ExtOperand{ExtImmediate(0x88), ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)}, + "6263144052717f88", "62 63 14 40 52 71 7f 88 vminmaxph $0x88,0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"}, + {"vminmaxph memory source plain", "VMINMAXPH", + []ExtOperand{ExtImmediate(0x88), ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)}, + "62431440523188", "62 43 14 40 52 31 88 vminmaxph $0x88,(%r9),%zmm29,%zmm30"}, + {"vminmaxph k7 zeroing", "VMINMAXPH", + []ExtOperand{ExtImmediate(0x88), ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)}, + "620314c752f488", "62 03 14 c7 52 f4 88 vminmaxph $0x88,%zmm28,%zmm29,%zmm30{%k7}{z}"}, } // amd64ResolveEntry finds the table entry a golden row exercises: the entry @@ -1646,12 +1724,21 @@ func TestAmd64ExtRejects(t *testing.T) { {"broadcast on the scalar multiply-add", "VFMADD132SH", []ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)}, "the entry's memory operand takes none"}, + {"broadcast on the complex multiply", "VFMULCPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "the entry's memory operand takes none"}, {"rounding on the 256-bit multiply-add", "VFMADD132PH", []ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)}, "the entry's destination takes none"}, + {"rounding on the minimum-maximum", "VMINMAXPH", + []ExtOperand{ExtImmediate(0x88), ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "the entry's destination takes none"}, {"a register where the multiply-add reads memory", "VFMSUB231PH", []ExtOperand{ExtZmm(29), ExtYmm(4), ExtZmm(30)}, "wants a ZMM register"}, + {"the complex multiply's scalar destination", "VFMULCPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtXmm(30)}, + "wants a ZMM register"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) @@ -1796,7 +1883,7 @@ func TestAmd64ExtArchBinding(t *testing.T) { t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) } } - if got := Extensions(AMD64); len(got) != 163 { - t.Errorf("the amd64 layer registers %d instructions, want 163", len(got)) + if got := Extensions(AMD64); len(got) != 174 { + t.Errorf("the amd64 layer registers %d instructions, want 174", len(got)) } } diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index b93d9b6..a058b46 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -76,6 +76,11 @@ func TestAmd64ExtensionRegistry(t *testing.T) { {"VFMSUB132SH", 1}, {"VFMSUB213SH", 1}, {"VFMSUB231SH", 1}, + {"VFMULCPH", 3}, + {"VFCMULCPH", 3}, + {"VFMULCSH", 1}, + {"VFCMULCSH", 1}, + {"VMINMAXPH", 3}, } { cands, ok := LookupExtension(arch.AMD64, tt.mnem) if !ok { @@ -89,8 +94,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { t.Errorf("the %s lookup is not case-insensitive", tt.mnem) } } - if got := arch.Extensions(arch.AMD64); len(got) != 163 { - t.Errorf("the amd64 layer registers %d instructions, want 163", len(got)) + if got := arch.Extensions(arch.AMD64); len(got) != 174 { + t.Errorf("the amd64 layer registers %d instructions, want 174", len(got)) } if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { t.Error("a non-extended mnemonic resolved")