From 22055b9bf3f45dace741116de405d01189a2bbd2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 00:02:27 +0200 Subject: [PATCH] feat(arch): add the packed FP16 arithmetic to the amd64 extension layer Assisted-by: GLM 5.3 Flash --- arch/amd64_ext.go | 25 +++++++++++++++++++++++++ arch/amd64_ext_test.go | 28 ++++++++++++++++++++++++++-- asm/extension_amd64_test.go | 6 ++++-- 3 files changed, 55 insertions(+), 4 deletions(-) diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index 456c778..6b68e02 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -406,4 +406,29 @@ var amd64Extensions = []ExtInstr{ {Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 64-bit integer", Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W1 79 /r)"}, + + // AVX512-FP16 packed, the 512-bit arithmetic the scalar core mirrors: + // full ZMM lanes, EVEX.NDS.MAP5 with no mandatory prefix, rounding + // control left to MXCSR. + {Name: "VADDPH", Summary: "Add packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"}, + {Name: "VSUBPH", Summary: "Subtract packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"}, + {Name: "VMULPH", Summary: "Multiply packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"}, + {Name: "VDIVPH", Summary: "Divide packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"}, + {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"}, + {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"}, + {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"}, } diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 5ba4578..ec542df 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -168,6 +168,30 @@ var amd64GoldenRows = []amd64GoldenRow{ []ExtOperand{ExtXmm(30), ExtGpr32(12)}, "62157d087ee6", ""}, + // AVX512-FP16 packed arithmetic, EVEX.NDS.512.MAP5.W0 with no + // mandatory prefix. The GNU vector again sits on high registers. + {"vaddph", "VADDPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "6205144058f4", "62 05 14 40 58 f4 vaddph %zmm28,%zmm29,%zmm30"}, + {"vsubph", "VSUBPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "620514405cf4", "62 05 14 40 5c f4 vsubph %zmm28,%zmm29,%zmm30"}, + {"vmulph", "VMULPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "6205144059f4", "62 05 14 40 59 f4 vmulph %zmm28,%zmm29,%zmm30"}, + {"vdivph", "VDIVPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "620514405ef4", "62 05 14 40 5e f4 vdivph %zmm28,%zmm29,%zmm30"}, + {"vminph", "VMINPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "620514405df4", "62 05 14 40 5d f4 vminph %zmm28,%zmm29,%zmm30"}, + {"vmaxph", "VMAXPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "620514405ff4", "62 05 14 40 5f f4 vmaxph %zmm28,%zmm29,%zmm30"}, + {"vsqrtph", "VSQRTPH", + []ExtOperand{ExtZmm(29), ExtZmm(30)}, + "62057c4851f5", "62 05 7c 48 51 f5 vsqrtph %zmm29,%zmm30"}, + // High registers in a 512-bit form exercise the EVEX extension bits: // with both sources above 15 the B bar and X bar bits clear, while the // destination zmm23 keeps R bar set in byte one (derived from the @@ -392,7 +416,7 @@ func TestAmd64ExtArchBinding(t *testing.T) { t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) } } - if got := Extensions(AMD64); len(got) != 39 { - t.Errorf("the amd64 layer registers %d instructions, want 39", len(got)) + if got := Extensions(AMD64); len(got) != 46 { + t.Errorf("the amd64 layer registers %d instructions, want 46", len(got)) } } diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index 33515fd..f15ba1e 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -33,6 +33,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { {"VCVTSH2SS", 1}, {"VCVTSI2SH", 2}, {"VCVTSH2SI", 2}, + {"VADDPH", 1}, + {"VSQRTPH", 1}, } { cands, ok := LookupExtension(arch.AMD64, tt.mnem) if !ok { @@ -46,8 +48,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { t.Errorf("the %s lookup is not case-insensitive", tt.mnem) } } - if got := arch.Extensions(arch.AMD64); len(got) != 39 { - t.Errorf("the amd64 layer registers %d instructions, want 39", len(got)) + if got := arch.Extensions(arch.AMD64); len(got) != 46 { + t.Errorf("the amd64 layer registers %d instructions, want 46", len(got)) } if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { t.Error("a non-extended mnemonic resolved")