feat(arch): add the imm8 scalar FP16 controls to the amd64 extension layer

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 00:33:30 +02:00
1 parent 0354a1f4c1
commit aa9c7ca030
4 files changed
+288 -8

No files matched your search

+125 -2
View File
@@ -8,8 +8,9 @@
// arch/amd64_gen.go stays untouched, and asm.Encodable keeps answering false
// for every mnemonic here, so the layer stays out of the main encoders.
//
// The families are AVX512-BF16, AVX512-VP2INTERSECT and the scalar core of
// AVX512-FP16, in their EVEX register forms. The encodings are transcribed
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
// latter's scalar core with its imm8-control group and its packed 512-bit
// arithmetic, in their EVEX register forms. The encodings are transcribed
// from the SDM instruction entries and cross-checked against binutils-gdb's
// assembler testsuite; the golden vectors in amd64_ext_test.go pin the
// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey,
@@ -171,6 +172,10 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
return in.encodeAmdVecGprVec(ops)
case ExtFormAmdGprVec, ExtFormAmdVecGpr:
return in.encodeAmdGprPair(ops)
case ExtFormAmdVec3Imm:
return in.encodeAmdVec3Imm(ops)
case ExtFormAmdMask2Imm:
return in.encodeAmdMask2Imm(ops)
default:
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
}
@@ -225,6 +230,102 @@ func (in ExtInstr) encodeAmdMask2(ops []ExtOperand) ([]byte, error) {
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
}
// amd64Imm8 validates the leading immediate operand of an imm8-control form:
// an ExtImm with no shift, inside the unsigned byte range, and free of the
// bits the entry's control layout reserves. The reserved upper nibble of
// the VGETMANTSH control must encode as zero; the SDM marks every other
// layout here fully defined, and the VCMPSH hardware masks its predicate to
// five bits.
func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
if op.Kind != ExtImm {
return 0, fmt.Errorf("%s: operand %d wants an immediate control byte, got %s", in.Name, pos, op.Kind)
}
if op.Arr != ExtArrNone {
return 0, fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
}
if op.HasShift {
return 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 imm8 forms take none", in.Name, pos)
}
if op.Imm < 0 || op.Imm > 255 {
return 0, fmt.Errorf("%s: operand %d is immediate %d, outside the unsigned byte range 0-255", in.Name, pos, op.Imm)
}
if in.Imm8 == ExtImm8GetMant && op.Imm > 15 {
return 0, fmt.Errorf("%s: operand %d is immediate %d, the upper nibble of the mantissa control is reserved and must be zero", in.Name, pos, op.Imm)
}
return byte(op.Imm), nil
}
// encodeAmdVec3Imm fills the three-vector form with a control immediate:
// imm, src1, src2, dest, the order the reference listings write it in.
func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
imm, err := in.amd64Imm8(ops[0], 1)
if err != nil {
return nil, err
}
for i, op := range ops[1:] {
if err := in.amd64Vector(op, class, i+2); err != nil {
return nil, err
}
}
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
return append(out, imm), nil
}
// encodeAmdMask2Imm fills the opmask-destination form with a control
// immediate: imm, src1, src2, dest.
func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
imm, err := in.amd64Imm8(ops[0], 1)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
if ops[3].Kind != ExtKReg {
return nil, fmt.Errorf("%s: operand 4 wants an opmask register, got %s", in.Name, ops[3].Kind)
}
if err := in.amd64PlainReg(ops[3], 7, 4); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
return append(out, imm), nil
}
// ExtFP16RoundingModes names the two-bit rounding mode the round control of
// VRNDSCALESH and VREDUCESH carries, indexed by imm8[1:0], the SDM's RC
// field encoding.
var ExtFP16RoundingModes = [4]string{
"round to nearest (even)",
"round down (toward -infinity)",
"round up (toward +infinity)",
"round toward zero (truncate)",
}
// ExtFP16GetMantSigns names the sign control imm8[3:2] of the VGETMANTSH
// immediate, indexed by the field: the source's own sign, a forced positive,
// and the two encodings that yield the indefinite NaN on a negative source.
var ExtFP16GetMantSigns = [4]string{
"the sign of the source",
"positive",
"the indefinite NaN when the source is negative",
"the indefinite NaN when the source is negative",
}
// ExtFP16CmpPredicates names the 32 comparison predicates the VCMPSH
// immediate carries in imm8[4:0], in encoding order. The SDM's own
// spellings are the fixed vocabulary of the predicate suffixes.
var ExtFP16CmpPredicates = [32]string{
"EQ_OQ", "LT_OS", "LE_OS", "UNORD_Q", "NEQ_UQ", "NLT_US", "NLE_US", "ORD_Q",
"EQ_UQ", "NGE_US", "NGT_US", "FALSE_OQ", "NEQ_OQ", "GE_OS", "GT_OS", "TRUE_UQ",
"EQ_OS", "LT_OQ", "LE_OQ", "UNORD_S", "NEQ_US", "NLT_UQ", "NLE_UQ", "ORD_S",
"EQ_US", "NGE_UQ", "NGT_UQ", "FALSE_OS", "NEQ_OS", "GE_OQ", "GT_OQ", "TRUE_US",
}
// encodeAmdVecGprVec fills the conversion form with a general-register
// source: src1, gpr, dest. VCVTSI2SH XMM1, XMM2, EAX style.
func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
@@ -437,4 +538,26 @@ var amd64Extensions = []ExtInstr{
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
// AVX512-FP16 scalar, the imm8-control group: mantissa extraction,
// reduction, rounding to fraction bits and the compare into an opmask.
// Each carries its control byte as the leading immediate operand, the
// order the reference listings write it in. The controls live in map
// 0F3A: the compare with the F3 prefix the manual gives the compare
// family, the other three unprefixed. The immediate layouts and their
// tables are ExtFP16RoundingModes, ExtFP16GetMantSigns and
// ExtFP16CmpPredicates above; the reserved upper nibble of the mantissa
// control is refused rather than encoded.
{Name: "VCMPSH", Summary: "Compare scalar FP16 values into an opmask under an imm8 predicate",
Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCMPSH (EVEX.LLIG.F3.0F3A.W0 C2 /r /ib)"},
{Name: "VGETMANTSH", Summary: "Extract the normalised mantissa of a scalar FP16 value under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTSH (EVEX.LLIG.NP.0F3A.W0 27 /r /ib)"},
{Name: "VREDUCESH", Summary: "Reduce a scalar FP16 value by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCESH (EVEX.LLIG.NP.0F3A.W0 57 /r /ib)"},
{Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"},
}