feat(arch): add the AVX-VNNI-INT16 dot products to the extension layer

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 19:49:15 +02:00
1 parent 96000dd64d
commit 4632ac1bb9
3 files changed
+118 -4

No files matched your search

+34
View File
@@ -48,6 +48,7 @@ const (
ExtFeatureBF16 ExtFeature = "avx512bf16"
ExtFeatureVP2INTERSECT ExtFeature = "avx512vp2intersect"
ExtFeatureFP16 ExtFeature = "avx512fp16"
ExtFeatureVnniInt16 ExtFeature = "avxvnniint16"
)
// ExtXmm, ExtYmm and ExtZmm build vector operands of the three EVEX register
@@ -1798,4 +1799,37 @@ var amd64Extensions = []ExtInstr{
{Name: "VMINMAXSH", Summary: "Return the minimum or maximum of scalar FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x53, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINMAXSH (EVEX.NDS.LIG.0F3A.W0 53 /r /ib)"},
// The AVX-VNNI-INT16 dot products: VPDPWSUD and VPDPWSUDS, the
// unsigned by signed word pairs whose accumulation saturates under the
// S suffix, and VPDPWUSD and VPDPWUSDS, the signed by unsigned ones.
// VEX.NDS.F3.0F38.W0 D2 and D3, and the 66-prefixed pair beside them,
// the quartet the VEX mechanism carries: no AVX-512 form exists, so no
// register above 15, no write mask, no broadcast and no rounding ever
// applies, and the memory shape reads its second source plain, the
// m128 and m256 the manual spells.
{Name: "VPDPWSUD", Summary: "Dot product of unsigned and signed 16-bit integers with 32-bit accumulation",
Bytes: []byte{0xC4, 0x02, 0x02, 0xD2, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Vex: true, Feature: ExtFeatureVnniInt16,
Ref: "Intel SDM Vol. 2C, VPDPWSUD (VEX.NDS.128.F3.0F38.W0 D2 /r)"},
{Name: "VPDPWSUD", Summary: "Dot product of unsigned and signed 16-bit integers with 32-bit accumulation",
Bytes: []byte{0xC4, 0x02, 0x06, 0xD2, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Vex: true, Feature: ExtFeatureVnniInt16,
Ref: "Intel SDM Vol. 2C, VPDPWSUD (VEX.NDS.256.F3.0F38.W0 D2 /r)"},
{Name: "VPDPWSUDS", Summary: "Dot product of unsigned and signed 16-bit integers, the accumulation saturating",
Bytes: []byte{0xC4, 0x02, 0x02, 0xD3, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Vex: true, Feature: ExtFeatureVnniInt16,
Ref: "Intel SDM Vol. 2C, VPDPWSUDS (VEX.NDS.128.F3.0F38.W0 D3 /r)"},
{Name: "VPDPWSUDS", Summary: "Dot product of unsigned and signed 16-bit integers, the accumulation saturating",
Bytes: []byte{0xC4, 0x02, 0x06, 0xD3, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Vex: true, Feature: ExtFeatureVnniInt16,
Ref: "Intel SDM Vol. 2C, VPDPWSUDS (VEX.NDS.256.F3.0F38.W0 D3 /r)"},
{Name: "VPDPWUSD", Summary: "Dot product of signed and unsigned 16-bit integers with 32-bit accumulation",
Bytes: []byte{0xC4, 0x02, 0x01, 0xD2, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Vex: true, Feature: ExtFeatureVnniInt16,
Ref: "Intel SDM Vol. 2C, VPDPWUSD (VEX.NDS.128.66.0F38.W0 D2 /r)"},
{Name: "VPDPWUSD", Summary: "Dot product of signed and unsigned 16-bit integers with 32-bit accumulation",
Bytes: []byte{0xC4, 0x02, 0x05, 0xD2, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Vex: true, Feature: ExtFeatureVnniInt16,
Ref: "Intel SDM Vol. 2C, VPDPWUSD (VEX.NDS.256.66.0F38.W0 D2 /r)"},
{Name: "VPDPWUSDS", Summary: "Dot product of signed and unsigned 16-bit integers, the accumulation saturating",
Bytes: []byte{0xC4, 0x02, 0x01, 0xD3, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Vex: true, Feature: ExtFeatureVnniInt16,
Ref: "Intel SDM Vol. 2C, VPDPWUSDS (VEX.NDS.128.66.0F38.W0 D3 /r)"},
{Name: "VPDPWUSDS", Summary: "Dot product of signed and unsigned 16-bit integers, the accumulation saturating",
Bytes: []byte{0xC4, 0x02, 0x05, 0xD3, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Vex: true, Feature: ExtFeatureVnniInt16,
Ref: "Intel SDM Vol. 2C, VPDPWUSDS (VEX.NDS.256.66.0F38.W0 D3 /r)"},
}