feat(arch): add the amd64 fp16 packed imm8-control group

Assisted-by: GLM 5.3
This commit is contained in:
petrbalvin committed 2026-10-07 13:51:48 +02:00
1 parent 8de1b371da
commit d03de62c07
4 files changed
+158 -10

No files matched your search

+81 -5
View File
@@ -10,11 +10,12 @@
//
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
// latter's scalar core with its imm8-control group, its packed 512-bit and
// VL arithmetic, the embedded rounding of its FP operations and its fourteen
// packed conversion directions, in their EVEX register forms. The encodings
// are transcribed from the SDM instruction entries and cross-checked against
// binutils-gdb's assembler testsuite; the golden vectors in amd64_ext_test.go
// pin the bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19
// VL arithmetic, the packed mirror of the imm8-control group, the embedded
// rounding of its FP operations and its fourteen packed conversion
// directions, in their EVEX register forms. The encodings are transcribed
// from the SDM instruction entries and cross-checked against binutils-gdb's
// assembler testsuite; the golden vectors in amd64_ext_test.go pin the
// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19
// survey, no longer belong here: the Go toolchain's assembler knows them
// today, they live in the generated table and the EVEX encoder, and a
// mnemonic the toolchain has is not an extension. VCVTPS2PH, VCVTUDQ2PS
@@ -482,6 +483,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
return in.encodeAmdVec3Imm(ops)
case ExtFormAmdMask2Imm:
return in.encodeAmdMask2Imm(ops)
case ExtFormAmdVec2Imm:
return in.encodeAmdVec2Imm(ops)
default:
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
}
@@ -711,6 +714,43 @@ func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
return append(out, imm), nil
}
// encodeAmdVec2Imm fills the two-vector form with a control immediate: imm,
// src, dest, the packed imm8-control group. An entry with Mem set takes the
// memory shape of the source, zmm2/m512 in the manual; the control byte
// rides after the ModR/M and its displacement bytes, the last byte of the
// word.
func (in ExtInstr) encodeAmdVec2Imm(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
imm, err := in.amd64Imm8(ops[0], 1)
if err != nil {
return nil, err
}
dest, mask, zeroing, _, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, -1, ops[1], 2)
if err != nil {
return nil, err
}
amd64ApplyMask(out, mask, zeroing)
return append(out, imm), nil
}
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, dest.Reg, -1, ops[1].Reg)
amd64ApplyMask(out, mask, zeroing)
return append(out, imm), nil
}
// encodeAmdMask2Imm fills the opmask-destination form with a control
// immediate: imm, src1, src2, dest. An entry with Mem set takes the memory
// shape of the second source.
@@ -1313,4 +1353,40 @@ var amd64Extensions = []ExtInstr{
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x85, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.128.66.MAP5.W1 5A /r, XMM destination)"},
// AVX512-FP16 packed, the imm8-control group: the packed mirror of the
// scalar core's mantissa extraction, reduction and rounding to fraction
// bits, one control byte over every lane of the vector. The controls
// share the immediate layouts and the tables the scalar entries carry,
// ExtImm8ScaleRound and ExtImm8GetMant, the reserved upper nibble of the
// mantissa control refused rather than encoded. The sources read from
// memory full-width, no broadcast: the control governs the lanes, not a
// splatted element.
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.512.NP.0F3A.W0 08 /r /ib)"},
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.256.NP.0F3A.W0 08 /r /ib)"},
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.128.NP.0F3A.W0 08 /r /ib)"},
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.512.NP.0F3A.W0 56 /r /ib)"},
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.256.NP.0F3A.W0 56 /r /ib)"},
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.128.NP.0F3A.W0 56 /r /ib)"},
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.512.NP.0F3A.W0 26 /r /ib)"},
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.256.NP.0F3A.W0 26 /r /ib)"},
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.128.NP.0F3A.W0 26 /r /ib)"},
}