feat(arch): add the amd64 fp16 packed imm8-control group
Assisted-by: GLM 5.3
This commit is contained in:
1 parent
8de1b371da
commit
d03de62c07
4 files changed
+158
-10
No files matched your search
+81
-5
@@ -10,11 +10,12 @@
|
||||
//
|
||||
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
|
||||
// latter's scalar core with its imm8-control group, its packed 512-bit and
|
||||
// VL arithmetic, the embedded rounding of its FP operations and its fourteen
|
||||
// packed conversion directions, in their EVEX register forms. The encodings
|
||||
// are transcribed from the SDM instruction entries and cross-checked against
|
||||
// binutils-gdb's assembler testsuite; the golden vectors in amd64_ext_test.go
|
||||
// pin the bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19
|
||||
// VL arithmetic, the packed mirror of the imm8-control group, the embedded
|
||||
// rounding of its FP operations and its fourteen packed conversion
|
||||
// directions, in their EVEX register forms. The encodings are transcribed
|
||||
// from the SDM instruction entries and cross-checked against binutils-gdb's
|
||||
// assembler testsuite; the golden vectors in amd64_ext_test.go pin the
|
||||
// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19
|
||||
// survey, no longer belong here: the Go toolchain's assembler knows them
|
||||
// today, they live in the generated table and the EVEX encoder, and a
|
||||
// mnemonic the toolchain has is not an extension. VCVTPS2PH, VCVTUDQ2PS
|
||||
@@ -482,6 +483,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
||||
return in.encodeAmdVec3Imm(ops)
|
||||
case ExtFormAmdMask2Imm:
|
||||
return in.encodeAmdMask2Imm(ops)
|
||||
case ExtFormAmdVec2Imm:
|
||||
return in.encodeAmdVec2Imm(ops)
|
||||
default:
|
||||
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
|
||||
}
|
||||
@@ -711,6 +714,43 @@ func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
|
||||
return append(out, imm), nil
|
||||
}
|
||||
|
||||
// encodeAmdVec2Imm fills the two-vector form with a control immediate: imm,
|
||||
// src, dest, the packed imm8-control group. An entry with Mem set takes the
|
||||
// memory shape of the source, zmm2/m512 in the manual; the control byte
|
||||
// rides after the ModR/M and its displacement bytes, the last byte of the
|
||||
// word.
|
||||
func (in ExtInstr) encodeAmdVec2Imm(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
imm, err := in.amd64Imm8(ops[0], 1)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
dest, mask, zeroing, _, err := in.amd64WriteMask(ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
||||
if err := in.amd64Vector(dest, class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, -1, ops[1], 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(dest, class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := amd64Encode(in.Bytes, dest.Reg, -1, ops[1].Reg)
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
|
||||
// encodeAmdMask2Imm fills the opmask-destination form with a control
|
||||
// immediate: imm, src1, src2, dest. An entry with Mem set takes the memory
|
||||
// shape of the second source.
|
||||
@@ -1313,4 +1353,40 @@ var amd64Extensions = []ExtInstr{
|
||||
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
|
||||
Bytes: []byte{0x62, 0x05, 0x85, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.128.66.MAP5.W1 5A /r, XMM destination)"},
|
||||
|
||||
// AVX512-FP16 packed, the imm8-control group: the packed mirror of the
|
||||
// scalar core's mantissa extraction, reduction and rounding to fraction
|
||||
// bits, one control byte over every lane of the vector. The controls
|
||||
// share the immediate layouts and the tables the scalar entries carry,
|
||||
// ExtImm8ScaleRound and ExtImm8GetMant, the reserved upper nibble of the
|
||||
// mantissa control refused rather than encoded. The sources read from
|
||||
// memory full-width, no broadcast: the control governs the lanes, not a
|
||||
// splatted element.
|
||||
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.512.NP.0F3A.W0 08 /r /ib)"},
|
||||
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.256.NP.0F3A.W0 08 /r /ib)"},
|
||||
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.128.NP.0F3A.W0 08 /r /ib)"},
|
||||
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.512.NP.0F3A.W0 56 /r /ib)"},
|
||||
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.256.NP.0F3A.W0 56 /r /ib)"},
|
||||
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.128.NP.0F3A.W0 56 /r /ib)"},
|
||||
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.512.NP.0F3A.W0 26 /r /ib)"},
|
||||
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.256.NP.0F3A.W0 26 /r /ib)"},
|
||||
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.128.NP.0F3A.W0 26 /r /ib)"},
|
||||
}
|
||||
Reference in new issue
Block a user