feat(arch): add the FP16 complex multiply and minimum-maximum to the extension layer
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
8080e0acef
commit
509afbb6c9
3 files changed
+145
-4
No files matched your search
@@ -1545,4 +1545,53 @@ var amd64Extensions = []ExtInstr{
|
|||||||
{Name: "VFMSUB231SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying the subtracted term",
|
{Name: "VFMSUB231SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||||
Ref: "Intel SDM Vol. 2C, VFMSUB231SH (EVEX.NDS.LIG.66.MAP6.W0 BB /r)"},
|
Ref: "Intel SDM Vol. 2C, VFMSUB231SH (EVEX.NDS.LIG.66.MAP6.W0 BB /r)"},
|
||||||
|
|
||||||
|
// AVX512-FP16 complex multiply: the packed pair over the FP16 complex
|
||||||
|
// pairs, F3 prefixing the multiply by the conjugate of the second
|
||||||
|
// source and F2 the multiply of the conjugate of the first source, and
|
||||||
|
// their scalar mirrors at D7. A complex value needs both of its halves
|
||||||
|
// beside one another, so the memory shape reads its full-width vector
|
||||||
|
// plain and takes no broadcast, unlike the packed FMA; the 512-bit
|
||||||
|
// register forms take the embedded rounding. The scalar forms read
|
||||||
|
// their m16 plain and take the rounding beside it.
|
||||||
|
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
|
||||||
|
Bytes: []byte{0x62, 0x06, 0x06, 0x40, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.512.F3.MAP6.W0 D6 /r)"},
|
||||||
|
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
|
||||||
|
Bytes: []byte{0x62, 0x06, 0x06, 0x20, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.256.F3.MAP6.W0 D6 /r)"},
|
||||||
|
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
|
||||||
|
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.128.F3.MAP6.W0 D6 /r)"},
|
||||||
|
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
|
||||||
|
Bytes: []byte{0x62, 0x06, 0x07, 0x40, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.512.F2.MAP6.W0 D6 /r)"},
|
||||||
|
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
|
||||||
|
Bytes: []byte{0x62, 0x06, 0x07, 0x20, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.256.F2.MAP6.W0 D6 /r)"},
|
||||||
|
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
|
||||||
|
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.128.F2.MAP6.W0 D6 /r)"},
|
||||||
|
{Name: "VFMULCSH", Summary: "Multiply scalar complex FP16 values, conjugating the second source",
|
||||||
|
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VFMULCSH (EVEX.NDS.LIG.F3.MAP6.W0 D7 /r)"},
|
||||||
|
{Name: "VFCMULCSH", Summary: "Multiply scalar complex FP16 values, conjugating the first source",
|
||||||
|
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VFCMULCSH (EVEX.NDS.LIG.F2.MAP6.W0 D7 /r)"},
|
||||||
|
|
||||||
|
// AVX512-FP16 minimum or maximum: VMINMAXPH, the per-lane selection
|
||||||
|
// under the imm8 control the AVX512DQ double- and single-precision pair
|
||||||
|
// carries into the half-precision set, one control byte over the lanes.
|
||||||
|
// EVEX.NDS.0F3A.W0 52 /r /ib, the control byte riding last as the
|
||||||
|
// immediate operand it is. The destinations take the write mask and the
|
||||||
|
// memory shape of the second source reads its full-width vector plain.
|
||||||
|
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
|
||||||
|
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.512.0F3A.W0 52 /r /ib)"},
|
||||||
|
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
|
||||||
|
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.256.0F3A.W0 52 /r /ib)"},
|
||||||
|
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
|
||||||
|
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
|
||||||
|
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.128.0F3A.W0 52 /r /ib)"},
|
||||||
}
|
}
|
||||||
+89
-2
@@ -1313,6 +1313,84 @@ var amd64GoldenRows = []amd64GoldenRow{
|
|||||||
{"vfmaddsub132ph k5 merging, memory source", "VFMADDSUB132PH",
|
{"vfmaddsub132ph k5 merging, memory source", "VFMADDSUB132PH",
|
||||||
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)},
|
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)},
|
||||||
"624615459631", "62 46 15 45 96 31 vfmaddsub132ph (%r9),%zmm29,%zmm30{%k5}"},
|
"624615459631", "62 46 15 45 96 31 vfmaddsub132ph (%r9),%zmm29,%zmm30{%k5}"},
|
||||||
|
|
||||||
|
// The complex multiply, EVEX.NDS.F3.MAP6.W0 D6 and EVEX.NDS.F2.MAP6.W0
|
||||||
|
// D6, the scalar pair at D7. No broadcast: a complex value needs both
|
||||||
|
// of its halves beside one another, so the memory shape reads its
|
||||||
|
// full-width vector plain.
|
||||||
|
{"vfmulcph", "VFMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||||
|
"62061640d6f4", "62 06 16 40 d6 f4 vfmulcph %zmm28,%zmm29,%zmm30"},
|
||||||
|
{"vfmulcph ymm", "VFMULCPH",
|
||||||
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||||
|
"62f65628d6f4", "62 f6 56 28 d6 f4 vfmulcph %ymm4,%ymm5,%ymm6"},
|
||||||
|
{"vfmulcph xmm", "VFMULCPH",
|
||||||
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||||
|
"62f65608d6f4", "62 f6 56 08 d6 f4 vfmulcph %xmm4,%xmm5,%xmm6"},
|
||||||
|
{"vfcmulcph", "VFCMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||||
|
"62061740d6f4", "62 06 17 40 d6 f4 vfcmulcph %zmm28,%zmm29,%zmm30"},
|
||||||
|
{"vfcmulcph ymm", "VFCMULCPH",
|
||||||
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||||
|
"62f65728d6f4", "62 f6 57 28 d6 f4 vfcmulcph %ymm4,%ymm5,%ymm6"},
|
||||||
|
{"vfcmulcph xmm", "VFCMULCPH",
|
||||||
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||||
|
"62f65708d6f4", "62 f6 57 08 d6 f4 vfcmulcph %xmm4,%xmm5,%xmm6"},
|
||||||
|
{"vfmulcsh", "VFMULCSH",
|
||||||
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
||||||
|
"62061600d7f4", "62 06 16 00 d7 f4 vfmulcsh %xmm28,%xmm29,%xmm30"},
|
||||||
|
{"vfcmulcsh", "VFCMULCSH",
|
||||||
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
||||||
|
"62061700d7f4", "62 06 17 00 d7 f4 vfcmulcsh %xmm28,%xmm29,%xmm30"},
|
||||||
|
{"vfmulcph memory source", "VFMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||||
|
"62661640d6717f", "62 66 16 40 d6 71 7f vfmulcph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||||
|
{"vfcmulcph memory source", "VFCMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)},
|
||||||
|
"62461740d631", "62 46 17 40 d6 31 vfcmulcph (%r9),%zmm29,%zmm30"},
|
||||||
|
{"vfmulcsh memory source", "VFMULCSH",
|
||||||
|
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
||||||
|
"62461600d731", "62 46 16 00 d7 31 vfmulcsh (%r9),%xmm29,%xmm30"},
|
||||||
|
{"vfmulcph rz-sae", "VFMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtRounded(ExtZmm(30), ExtRoundTruncate)},
|
||||||
|
"62061670d6f4", "62 06 16 70 d6 f4 vfmulcph {rz-sae},%zmm28,%zmm29,%zmm30"},
|
||||||
|
{"vfcmulcph rn-sae", "VFCMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)},
|
||||||
|
"62f65718d6f4", "62 f6 57 18 d6 f4 vfcmulcph {rn-sae},%zmm4,%zmm5,%zmm6"},
|
||||||
|
{"vfmulcsh rn-sae", "VFMULCSH",
|
||||||
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundNearest)},
|
||||||
|
"62f65618d7f4", "62 f6 56 18 d7 f4 vfmulcsh {rn-sae},%xmm4,%xmm5,%xmm6"},
|
||||||
|
{"vfcmulcsh ru-sae, high registers", "VFCMULCSH",
|
||||||
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundUp)},
|
||||||
|
"62061750d7f4", "62 06 17 50 d7 f4 vfcmulcsh {ru-sae},%xmm28,%xmm29,%xmm30"},
|
||||||
|
{"vfmulcph k7 zeroing", "VFMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)},
|
||||||
|
"620616c7d6f4", "62 06 16 c7 d6 f4 vfmulcph %zmm28,%zmm29,%zmm30{%k7}{z}"},
|
||||||
|
{"vfcmulcph k5 merging, memory source", "VFCMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)},
|
||||||
|
"62461745d631", "62 46 17 45 d6 31 vfcmulcph (%r9),%zmm29,%zmm30{%k5}"},
|
||||||
|
|
||||||
|
// VMINMAXPH, EVEX.NDS.0F3A.W0 52 /r /ib: the control byte leads the
|
||||||
|
// operand list and rides last in the word, the order the reference
|
||||||
|
// listings write both in.
|
||||||
|
{"vminmaxph", "VMINMAXPH",
|
||||||
|
[]ExtOperand{ExtImmediate(0x88), ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||||
|
"6203144052f488", "62 03 14 40 52 f4 88 vminmaxph $0x88,%zmm28,%zmm29,%zmm30"},
|
||||||
|
{"vminmaxph ymm", "VMINMAXPH",
|
||||||
|
[]ExtOperand{ExtImmediate(0x88), ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||||
|
"62f3542852f488", "62 f3 54 28 52 f4 88 vminmaxph $0x88,%ymm4,%ymm5,%ymm6"},
|
||||||
|
{"vminmaxph xmm", "VMINMAXPH",
|
||||||
|
[]ExtOperand{ExtImmediate(0x88), ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||||
|
"62f3540852f488", "62 f3 54 08 52 f4 88 vminmaxph $0x88,%xmm4,%xmm5,%xmm6"},
|
||||||
|
{"vminmaxph memory source", "VMINMAXPH",
|
||||||
|
[]ExtOperand{ExtImmediate(0x88), ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||||
|
"6263144052717f88", "62 63 14 40 52 71 7f 88 vminmaxph $0x88,0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||||
|
{"vminmaxph memory source plain", "VMINMAXPH",
|
||||||
|
[]ExtOperand{ExtImmediate(0x88), ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)},
|
||||||
|
"62431440523188", "62 43 14 40 52 31 88 vminmaxph $0x88,(%r9),%zmm29,%zmm30"},
|
||||||
|
{"vminmaxph k7 zeroing", "VMINMAXPH",
|
||||||
|
[]ExtOperand{ExtImmediate(0x88), ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)},
|
||||||
|
"620314c752f488", "62 03 14 c7 52 f4 88 vminmaxph $0x88,%zmm28,%zmm29,%zmm30{%k7}{z}"},
|
||||||
}
|
}
|
||||||
|
|
||||||
// amd64ResolveEntry finds the table entry a golden row exercises: the entry
|
// amd64ResolveEntry finds the table entry a golden row exercises: the entry
|
||||||
@@ -1646,12 +1724,21 @@ func TestAmd64ExtRejects(t *testing.T) {
|
|||||||
{"broadcast on the scalar multiply-add", "VFMADD132SH",
|
{"broadcast on the scalar multiply-add", "VFMADD132SH",
|
||||||
[]ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)},
|
[]ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)},
|
||||||
"the entry's memory operand takes none"},
|
"the entry's memory operand takes none"},
|
||||||
|
{"broadcast on the complex multiply", "VFMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||||
|
"the entry's memory operand takes none"},
|
||||||
{"rounding on the 256-bit multiply-add", "VFMADD132PH",
|
{"rounding on the 256-bit multiply-add", "VFMADD132PH",
|
||||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)},
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)},
|
||||||
"the entry's destination takes none"},
|
"the entry's destination takes none"},
|
||||||
|
{"rounding on the minimum-maximum", "VMINMAXPH",
|
||||||
|
[]ExtOperand{ExtImmediate(0x88), ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)},
|
||||||
|
"the entry's destination takes none"},
|
||||||
{"a register where the multiply-add reads memory", "VFMSUB231PH",
|
{"a register where the multiply-add reads memory", "VFMSUB231PH",
|
||||||
[]ExtOperand{ExtZmm(29), ExtYmm(4), ExtZmm(30)},
|
[]ExtOperand{ExtZmm(29), ExtYmm(4), ExtZmm(30)},
|
||||||
"wants a ZMM register"},
|
"wants a ZMM register"},
|
||||||
|
{"the complex multiply's scalar destination", "VFMULCPH",
|
||||||
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtXmm(30)},
|
||||||
|
"wants a ZMM register"},
|
||||||
} {
|
} {
|
||||||
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
|
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
|
||||||
_, err := in.Encode(tt.ops)
|
_, err := in.Encode(tt.ops)
|
||||||
@@ -1796,7 +1883,7 @@ func TestAmd64ExtArchBinding(t *testing.T) {
|
|||||||
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
|
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if got := Extensions(AMD64); len(got) != 163 {
|
if got := Extensions(AMD64); len(got) != 174 {
|
||||||
t.Errorf("the amd64 layer registers %d instructions, want 163", len(got))
|
t.Errorf("the amd64 layer registers %d instructions, want 174", len(got))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -76,6 +76,11 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
|
|||||||
{"VFMSUB132SH", 1},
|
{"VFMSUB132SH", 1},
|
||||||
{"VFMSUB213SH", 1},
|
{"VFMSUB213SH", 1},
|
||||||
{"VFMSUB231SH", 1},
|
{"VFMSUB231SH", 1},
|
||||||
|
{"VFMULCPH", 3},
|
||||||
|
{"VFCMULCPH", 3},
|
||||||
|
{"VFMULCSH", 1},
|
||||||
|
{"VFCMULCSH", 1},
|
||||||
|
{"VMINMAXPH", 3},
|
||||||
} {
|
} {
|
||||||
cands, ok := LookupExtension(arch.AMD64, tt.mnem)
|
cands, ok := LookupExtension(arch.AMD64, tt.mnem)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -89,8 +94,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
|
|||||||
t.Errorf("the %s lookup is not case-insensitive", tt.mnem)
|
t.Errorf("the %s lookup is not case-insensitive", tt.mnem)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if got := arch.Extensions(arch.AMD64); len(got) != 163 {
|
if got := arch.Extensions(arch.AMD64); len(got) != 174 {
|
||||||
t.Errorf("the amd64 layer registers %d instructions, want 163", len(got))
|
t.Errorf("the amd64 layer registers %d instructions, want 174", len(got))
|
||||||
}
|
}
|
||||||
if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok {
|
if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok {
|
||||||
t.Error("a non-extended mnemonic resolved")
|
t.Error("a non-extended mnemonic resolved")
|
||||||
|
|||||||
Reference in new issue
Block a user