From d03de62c07b13289e0ae5a0c684b6c52913b3959 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 13:15:08 +0200 Subject: [PATCH] feat(arch): add the amd64 fp16 packed imm8-control group Assisted-by: GLM 5.3 --- arch/amd64_ext.go | 86 ++++++++++++++++++++++++++++++++++--- arch/amd64_ext_test.go | 60 +++++++++++++++++++++++++- arch/arm64_ext.go | 12 +++++- asm/extension_amd64_test.go | 10 ++++- 4 files changed, 158 insertions(+), 10 deletions(-) diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index c85e3bf..65c5135 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -10,11 +10,12 @@ // // The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the // latter's scalar core with its imm8-control group, its packed 512-bit and -// VL arithmetic, the embedded rounding of its FP operations and its fourteen -// packed conversion directions, in their EVEX register forms. The encodings -// are transcribed from the SDM instruction entries and cross-checked against -// binutils-gdb's assembler testsuite; the golden vectors in amd64_ext_test.go -// pin the bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 +// VL arithmetic, the packed mirror of the imm8-control group, the embedded +// rounding of its FP operations and its fourteen packed conversion +// directions, in their EVEX register forms. The encodings are transcribed +// from the SDM instruction entries and cross-checked against binutils-gdb's +// assembler testsuite; the golden vectors in amd64_ext_test.go pin the +// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 // survey, no longer belong here: the Go toolchain's assembler knows them // today, they live in the generated table and the EVEX encoder, and a // mnemonic the toolchain has is not an extension. VCVTPS2PH, VCVTUDQ2PS @@ -482,6 +483,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) { return in.encodeAmdVec3Imm(ops) case ExtFormAmdMask2Imm: return in.encodeAmdMask2Imm(ops) + case ExtFormAmdVec2Imm: + return in.encodeAmdVec2Imm(ops) default: return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form) } @@ -711,6 +714,43 @@ func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) { return append(out, imm), nil } +// encodeAmdVec2Imm fills the two-vector form with a control immediate: imm, +// src, dest, the packed imm8-control group. An entry with Mem set takes the +// memory shape of the source, zmm2/m512 in the manual; the control byte +// rides after the ModR/M and its displacement bytes, the last byte of the +// word. +func (in ExtInstr) encodeAmdVec2Imm(ops []ExtOperand) ([]byte, error) { + class := amd64LengthClass(in.Bytes) + imm, err := in.amd64Imm8(ops[0], 1) + if err != nil { + return nil, err + } + dest, mask, zeroing, _, err := in.amd64WriteMask(ops[2], 3) + if err != nil { + return nil, err + } + if in.Mem == 2 && ops[1].Kind == ExtMem { + if err := in.amd64Vector(dest, class, 3); err != nil { + return nil, err + } + out, err := in.amd64MemBytes(in.Bytes, dest.Reg, -1, ops[1], 2) + if err != nil { + return nil, err + } + amd64ApplyMask(out, mask, zeroing) + return append(out, imm), nil + } + if err := in.amd64Vector(ops[1], class, 2); err != nil { + return nil, err + } + if err := in.amd64Vector(dest, class, 3); err != nil { + return nil, err + } + out := amd64Encode(in.Bytes, dest.Reg, -1, ops[1].Reg) + amd64ApplyMask(out, mask, zeroing) + return append(out, imm), nil +} + // encodeAmdMask2Imm fills the opmask-destination form with a control // immediate: imm, src1, src2, dest. An entry with Mem set takes the memory // shape of the second source. @@ -1313,4 +1353,40 @@ var amd64Extensions = []ExtInstr{ {Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination", Bytes: []byte{0x62, 0x05, 0x85, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.128.66.MAP5.W1 5A /r, XMM destination)"}, + + // AVX512-FP16 packed, the imm8-control group: the packed mirror of the + // scalar core's mantissa extraction, reduction and rounding to fraction + // bits, one control byte over every lane of the vector. The controls + // share the immediate layouts and the tables the scalar entries carry, + // ExtImm8ScaleRound and ExtImm8GetMant, the reserved upper nibble of the + // mantissa control refused rather than encoded. The sources read from + // memory full-width, no broadcast: the control governs the lanes, not a + // splatted element. + {Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control", + Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.512.NP.0F3A.W0 08 /r /ib)"}, + {Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control", + Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.256.NP.0F3A.W0 08 /r /ib)"}, + {Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control", + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.128.NP.0F3A.W0 08 /r /ib)"}, + {Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control", + Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.512.NP.0F3A.W0 56 /r /ib)"}, + {Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control", + Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.256.NP.0F3A.W0 56 /r /ib)"}, + {Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control", + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.128.NP.0F3A.W0 56 /r /ib)"}, + {Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control", + Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.512.NP.0F3A.W0 26 /r /ib)"}, + {Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control", + Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.256.NP.0F3A.W0 26 /r /ib)"}, + {Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control", + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.128.NP.0F3A.W0 26 /r /ib)"}, } diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index f2ba36d..57b7d88 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -1050,6 +1050,56 @@ var amd64GoldenRows = []amd64GoldenRow{ {"vcvtsh2usi 64-bit memory source disp8", "VCVTSH2USI", []ExtOperand{ExtMemory(1, 1), ExtGpr64(12)}, "6275fe08796101", "62 75 fe 08 79 61 01 vcvtsh2usi 0x2(%rcx),%r12 (Disp8(01))"}, + + // The packed imm8-control group, the packed mirror of the scalar core's + // mantissa extraction, reduction and rounding. The control byte leads, + // the sources read from memory full-width, and the immediate layouts are + // the ones the scalar rows share: every row takes the $0x7b the suite + // drives through the fraction-bit forms, save VGETMANTPH, whose reserved + // upper nibble the layer refuses and whose $0x0b encodes the same opcode + // row the suite's $0x7b spells. + {"vrndscaleph zmm", "VRNDSCALEPH", + []ExtOperand{ExtImmediate(0x7b), ExtZmm(5), ExtZmm(6)}, + "62f37c4808f57b", "62 f3 7c 48 08 f5 7b vrndscaleph $0x7b,%zmm5,%zmm6"}, + {"vrndscaleph ymm", "VRNDSCALEPH", + []ExtOperand{ExtImmediate(0x7b), ExtYmm(5), ExtYmm(6)}, + "62f37c2808f57b", "62 f3 7c 28 08 f5 7b vrndscaleph $0x7b,%ymm5,%ymm6"}, + {"vrndscaleph xmm", "VRNDSCALEPH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(5), ExtXmm(6)}, + "62f37c0808f57b", "62 f3 7c 08 08 f5 7b vrndscaleph $0x7b,%xmm5,%xmm6"}, + {"vreduceph zmm", "VREDUCEPH", + []ExtOperand{ExtImmediate(0x7b), ExtZmm(5), ExtZmm(6)}, + "62f37c4856f57b", "62 f3 7c 48 56 f5 7b vreduceph $0x7b,%zmm5,%zmm6"}, + {"vreduceph ymm", "VREDUCEPH", + []ExtOperand{ExtImmediate(0x7b), ExtYmm(5), ExtYmm(6)}, + "62f37c2856f57b", "62 f3 7c 28 56 f5 7b vreduceph $0x7b,%ymm5,%ymm6"}, + {"vreduceph xmm", "VREDUCEPH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(5), ExtXmm(6)}, + "62f37c0856f57b", "62 f3 7c 08 56 f5 7b vreduceph $0x7b,%xmm5,%xmm6"}, + {"vgetmantph zmm", "VGETMANTPH", + []ExtOperand{ExtImmediate(0x0b), ExtZmm(5), ExtZmm(6)}, + "62f37c4826f50b", "62 f3 7c 48 26 f5 0b vgetmantph $0xb,%zmm5,%zmm6"}, + {"vgetmantph ymm", "VGETMANTPH", + []ExtOperand{ExtImmediate(0x0b), ExtYmm(5), ExtYmm(6)}, + "62f37c2826f50b", "62 f3 7c 28 26 f5 0b vgetmantph $0xb,%ymm5,%ymm6"}, + {"vgetmantph xmm", "VGETMANTPH", + []ExtOperand{ExtImmediate(0x0b), ExtXmm(5), ExtXmm(6)}, + "62f37c0826f50b", "62 f3 7c 08 26 f5 0b vgetmantph $0xb,%xmm5,%xmm6"}, + {"vrndscaleph memory source", "VRNDSCALEPH", + []ExtOperand{ExtImmediate(0x7b), ExtMemory(9, 0), ExtZmm(30)}, + "62437c4808317b", "62 43 7c 48 08 31 7b vrndscaleph $0x7b,(%r9),%zmm30"}, + {"vrndscaleph memory source disp8", "VRNDSCALEPH", + []ExtOperand{ExtImmediate(0x7b), ExtMemory(1, 127), ExtZmm(6)}, + "62f37c4808717f7b", "62 f3 7c 48 08 71 7f 7b vrndscaleph $0x7b,0x1fc0(%rcx),%zmm6 (Disp8(7f))"}, + {"vreduceph memory source", "VREDUCEPH", + []ExtOperand{ExtImmediate(0x7b), ExtMemory(9, 0), ExtZmm(30)}, + "62437c4856317b", "62 43 7c 48 56 31 7b vreduceph $0x7b,(%r9),%zmm30"}, + {"vgetmantph memory source", "VGETMANTPH", + []ExtOperand{ExtImmediate(0x0b), ExtMemory(9, 0), ExtZmm(30)}, + "62437c4826310b", "62 43 7c 48 26 31 0b vgetmantph $0xb,(%r9),%zmm30"}, + {"vrndscaleph k7 zeroing", "VRNDSCALEPH", + []ExtOperand{ExtImmediate(0x7b), ExtZmm(5), ExtWriteMasked(ExtZmm(6), 7, true)}, + "62f37ccf08f57b", "62 f3 7c cf 08 f5 7b vrndscaleph $0x7b,%zmm5,%zmm6{%k7}{z}"}, } // amd64ResolveEntry finds the table entry a golden row exercises: the entry @@ -1374,6 +1424,12 @@ func TestAmd64ExtRejects(t *testing.T) { {"broadcast on the full-width FP16 source", "VCVTPH2W", []ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)}, "the entry's memory operand takes none"}, + {"broadcast on the packed control form's source", "VRNDSCALEPH", + []ExtOperand{ExtImmediate(0x7b), ExtBroadcast(9, 0), ExtZmm(30)}, + "the entry's memory operand takes none"}, + {"reserved upper nibble on the packed mantissa control", "VGETMANTPH", + []ExtOperand{ExtImmediate(0x7b), ExtZmm(5), ExtZmm(6)}, + "reserved and must be zero"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) @@ -1518,7 +1574,7 @@ func TestAmd64ExtArchBinding(t *testing.T) { t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) } } - if got := Extensions(AMD64); len(got) != 112 { - t.Errorf("the amd64 layer registers %d instructions, want 112", len(got)) + if got := Extensions(AMD64); len(got) != 121 { + t.Errorf("the amd64 layer registers %d instructions, want 121", len(got)) } } diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index 2ef8b5e..825496f 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -480,6 +480,11 @@ const ( // where dest is an opmask register and the immediate's layout is named // by the entry's Imm8 kind. ExtFormAmdMask2Imm + // ExtFormAmdVec2Imm is the two-vector form with a control immediate, + // VRNDSCALEPH $0x7b, Z5, Z6 style: the immediate leads, the order the + // reference listings write it in. Operands: imm, src, dest, and the + // immediate's layout is named by the entry's Imm8 kind. + ExtFormAmdVec2Imm // ExtFormAmdMemVec is the memory-load form: VMOVSH X30, 4660(R8) shape, // the manual's xmm1, m16 lines that stand beside the register form. // Operands: mem, dest. @@ -685,6 +690,8 @@ func (f ExtForm) Arity() int { return 2 case ExtFormAmdVec3Imm, ExtFormAmdMask2Imm: return 4 + case ExtFormAmdVec2Imm: + return 3 case ExtFormAmdMemVec, ExtFormAmdVecMem: return 2 case ExtFormPredicateOne, ExtFormPredicateWrite, ExtFormPredicateCounter: @@ -808,6 +815,8 @@ func (f ExtForm) String() string { return "immediate, three vectors" case ExtFormAmdMask2Imm: return "immediate, two vectors into an opmask" + case ExtFormAmdVec2Imm: + return "immediate and two vectors" case ExtFormAmdMemVec: return "memory into a vector" case ExtFormAmdVecMem: @@ -1057,7 +1066,8 @@ func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) { case ExtFormAmdVec3, ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdMask2, ExtFormAmdVecGprVec, ExtFormAmdGprVec, ExtFormAmdVecGpr, ExtFormAmdVec3Imm, ExtFormAmdMask2Imm, ExtFormAmdMemVec, ExtFormAmdVecMem, - ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter, ExtFormAmdVec2ToQuarter: + ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter, ExtFormAmdVec2ToQuarter, + ExtFormAmdVec2Imm: return in.encodeAmd64(ops) case ExtFormPredicateLogical, ExtFormPredicateSelect: return in.encodePredicateLogical(ops) diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index 6e6801f..325fd22 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -55,6 +55,9 @@ func TestAmd64ExtensionRegistry(t *testing.T) { {"VCVTUQQ2PH", 3}, {"VCVTPH2PD", 3}, {"VCVTPD2PH", 3}, + {"VRNDSCALEPH", 3}, + {"VREDUCEPH", 3}, + {"VGETMANTPH", 3}, } { cands, ok := LookupExtension(arch.AMD64, tt.mnem) if !ok { @@ -68,8 +71,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { t.Errorf("the %s lookup is not case-insensitive", tt.mnem) } } - if got := arch.Extensions(arch.AMD64); len(got) != 112 { - t.Errorf("the amd64 layer registers %d instructions, want 112", len(got)) + if got := arch.Extensions(arch.AMD64); len(got) != 121 { + t.Errorf("the amd64 layer registers %d instructions, want 121", len(got)) } if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { t.Error("a non-extended mnemonic resolved") @@ -160,6 +163,9 @@ func TestEncodeExtensionAmd64(t *testing.T) { {"fp16 to double-precision, widened", "VCVTPH2PD", []arch.ExtOperand{arch.ExtXmm(5), arch.ExtZmm(6)}, "62f57c485af5"}, + {"packed fp16 rounding to fraction bits", "VRNDSCALEPH", + []arch.ExtOperand{arch.ExtImmediate(0x7b), arch.ExtZmm(5), arch.ExtZmm(6)}, + "62f37c4808f57b"}, } { got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...) if err != nil {