feat(arch): encode the amd64 embedded rounding and SAE decorations

Assisted-by: GLM 5.3
This commit is contained in:
petrbalvin committed 2026-10-07 13:51:48 +02:00
1 parent e56c04e9ee
commit fb6d01a7d0
4 files changed
+381 -81

No files matched your search

+223 -79
View File
@@ -9,26 +9,31 @@
// for every mnemonic here, so the layer stays out of the main encoders.
//
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
// latter's scalar core with its imm8-control group and its packed 512-bit
// arithmetic, in their EVEX register forms. The encodings are transcribed
// from the SDM instruction entries and cross-checked against binutils-gdb's
// assembler testsuite; the golden vectors in amd64_ext_test.go pin the
// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey,
// no longer belong here: the Go toolchain's assembler knows them today, they
// latter's scalar core with its imm8-control group, its packed 512-bit and
// VL arithmetic and the embedded rounding of its FP operations, in their
// EVEX register forms. The encodings are transcribed from the SDM
// instruction entries and cross-checked against binutils-gdb's assembler
// testsuite; the golden vectors in amd64_ext_test.go pin the bytes.
// VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, no
// longer belong here: the Go toolchain's assembler knows them today, they
// live in the generated table and the EVEX encoder, and a mnemonic the
// toolchain has is not an extension.
//
// The forms encode the unmasked shapes: register forms throughout, and the
// memory forms beside them, the scalar ones the manual spells m16 and the
// packed ones with the {1toN} broadcast, base-relative operands with the
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the
// scaled index and the broadcast laying the SIB byte and EVEX.b over the
// same displacement semantics. The packed destinations take the write mask
// beside all of it, the {k1}{z} decorations the SDM spells: EVEX.aaa carries
// the masking register and EVEX.z the zeroing bit, K0 masks nothing, and the
// scalar forms, the compares, the opmask destinations and the load and store
// shapes take no mask at all. Embedded rounding still arrives with a later
// slice.
// memory forms beside them, the scalar ones the manual spells m16, m32 and
// m64 and the packed ones with the {1toN} broadcast, base-relative operands
// with the ModR/M disp8 and disp32 choices and the SIB byte RSP and R12
// demand, the scaled index and the broadcast laying the SIB byte and EVEX.b
// over the same displacement semantics. The packed destinations take the
// write mask beside all of it, the {k1}{z} decorations the SDM spells:
// EVEX.aaa carries the masking register and EVEX.z the zeroing bit, K0 masks
// nothing, and the scalar forms, the compares, the opmask destinations and
// the load and store shapes take no mask at all. The FP destinations take
// the embedded rounding beside that, the {sae} and {rn-sae} through
// {rz-sae} decorations: EVEX.b selects the rounding context and EVEX.RC,
// the L'L bits it replaces, the mode, on the 512-bit and scalar register
// forms alone, exactly the two lengths whose word has no vector length left
// to lose.
package arch
@@ -159,6 +164,9 @@ func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, er
if op.HasShift {
return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos)
}
if op.Round != ExtRoundNone {
return 0, 0, fmt.Errorf("%s: operand %d carries a rounding control, the memory operand takes none", in.Name, pos)
}
if op.Broadcast && !in.Bcast {
return 0, 0, fmt.Errorf("%s: operand %d carries a broadcast, the entry's memory operand takes none", in.Name, pos)
}
@@ -307,27 +315,63 @@ func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos in
return amd64EncodeMemory(b, dest, vvvv, base, disp), nil
}
// amd64WriteMask lifts the write mask spelling out of a destination operand:
// the returned copy carries the register bits alone, the mask register and
// the zeroing flag come back beside it. K0 never masks, zeroing is valid
// only beside a mask, and an entry whose destination takes none refuses the
// spellings outright, so nothing arrives at the encoding half-claimed.
func (in ExtInstr) amd64WriteMask(op ExtOperand, pos int) (ExtOperand, int, bool, error) {
// amd64WriteMask lifts the decorations off a destination operand: the
// returned copy carries the register bits alone, while the mask register,
// the zeroing flag and the rounding control come back beside it. K0 never
// masks, zeroing is valid only beside a mask, and an entry whose destination
// takes neither decoration refuses the spellings outright, so nothing
// arrives at the encoding half-claimed. The rounding decoration is
// validated against the entry's capability: an entry that carries neither
// Er nor Sae takes none, an Er entry demands one of the four modes, and an
// entry that suppresses exceptions alone takes {sae} and refuses every
// mode, the split the manual's two decoration classes draw.
func (in ExtInstr) amd64WriteMask(op ExtOperand, pos int) (ExtOperand, int, bool, ExtRounding, error) {
stripped := op
stripped.Mask, stripped.HasMask, stripped.Zeroing = 0, false, false
if !op.HasMask && !op.Zeroing {
return stripped, 0, false, nil
round := op.Round
stripped.Round = ExtRoundNone
if !op.HasMask && !op.Zeroing && round == ExtRoundNone {
return stripped, 0, false, ExtRoundNone, nil
}
if !in.Mask {
return stripped, 0, false, fmt.Errorf("%s: operand %d carries a write mask, the entry's destination takes none", in.Name, pos)
if op.HasMask || op.Zeroing {
if !in.Mask {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries a write mask, the entry's destination takes none", in.Name, pos)
}
if !op.HasMask {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries zeroing without a write mask", in.Name, pos)
}
if op.Mask < 1 || op.Mask > 7 {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d names write mask k%d, outside the masking registers k1-k7", in.Name, pos, op.Mask)
}
}
if !op.HasMask {
return stripped, 0, false, fmt.Errorf("%s: operand %d carries zeroing without a write mask", in.Name, pos)
if round != ExtRoundNone {
if !in.Er && !in.Sae {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries a rounding control, the entry's destination takes none", in.Name, pos)
}
if in.Sae && round != ExtRoundSAE {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d names the rounding mode %s, the entry suppresses exceptions alone and takes {sae}", in.Name, pos, round)
}
if in.Er && round == ExtRoundSAE {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d spells {sae} without a mode, the embedded rounding wants {rn-sae} through {rz-sae}", in.Name, pos)
}
}
if op.Mask < 1 || op.Mask > 7 {
return stripped, 0, false, fmt.Errorf("%s: operand %d names write mask k%d, outside the masking registers k1-k7", in.Name, pos, op.Mask)
return stripped, op.Mask, op.Zeroing, round, nil
}
// amd64ApplyRounding lays the validated rounding decoration into the encoded
// word: EVEX.b, bit four of byte three, selects the rounding context, and
// EVEX.RC, the L'L bits, names the mode. The L'L bits the template carries
// are cleared, because a word with an embedded rounding mode has no vector
// length of its own to spell; the write mask keeps its bits under the
// decoration, EVEX.aaa and EVEX.z untouched. An undecorated destination
// leaves the word alone: its L'L carries the vector length.
func amd64ApplyRounding(b []byte, mode ExtRounding) {
if mode == ExtRoundNone {
return
}
return stripped, op.Mask, op.Zeroing, nil
b[3] &^= 0x60
b[3] |= 0x10
b[3] |= amd64RoundingBits(mode) << 5
}
// amd64ApplyMask lays the validated write mask into the encoded word:
@@ -361,6 +405,9 @@ func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
if op.Zeroing {
return fmt.Errorf("%s: operand %d carries zeroing, the position takes none", in.Name, pos)
}
if op.Round != ExtRoundNone {
return fmt.Errorf("%s: operand %d carries a rounding control, the position takes none", in.Name, pos)
}
if op.Arr != ExtArrNone {
return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
}
@@ -442,17 +489,23 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
// arithmetic, spelled xmm3/m16 in the manual, may be a base-relative
// operand, which rides the r/m field with its displacement bytes after the
// opcode, and the packed entries lay the source's {1toN} broadcast over the
// same encoding as EVEX.b.
// same encoding as EVEX.b. An entry with Er or Sae set takes the rounding
// decoration on its register form alone: the memory shape refuses one, the
// embedded rounding and the broadcast share EVEX.b and cannot ride the same
// word.
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
dest, mask, zeroing, err := in.amd64WriteMask(ops[2], 3)
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if round != ExtRoundNone {
return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name)
}
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
@@ -468,7 +521,7 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
return nil, err
}
}
dest, mask, zeroing, err := in.amd64WriteMask(ops[2], 3)
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
@@ -477,6 +530,7 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
}
out := amd64Encode(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg)
amd64ApplyMask(out, mask, zeroing)
amd64ApplyRounding(out, round)
return out, nil
}
@@ -507,27 +561,32 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
// of destination, and at 128 bits the companion stays the class itself. An
// entry with Mem set takes the memory shape of that position too: the
// entry with Mem set takes the memory shape of the source position too: the
// compares and the packed square root read their source from memory, the
// packed square root's source carrying the {1toN} broadcast as EVEX.b, and
// the narrow BF16 convert reads its full-width source there.
// the narrow BF16 convert reads its full-width source there. An entry with
// Er or Sae set takes the rounding decoration on its register form alone;
// the memory shape refuses one.
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
destClass := class
if in.Form == ExtFormAmdVec2Half {
destClass = amd64HalfClass(class)
}
dest, mask, zeroing, err := in.amd64WriteMask(ops[1], 2)
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[1], 2)
if err != nil {
return nil, err
}
if in.Mem == 1 && ops[0].Kind == ExtMem {
// The memory spelling is validated before the destination: on the
// half form the destination's narrowed class is the likelier
// narrow form the destination's narrowed class is the likelier
// rejection, but a miswritten source spelling names itself first.
if _, _, err := in.amd64Memory(ops[0], 1); err != nil {
return nil, err
}
if round != ExtRoundNone {
return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name)
}
if err := in.amd64Vector(dest, destClass, 2); err != nil {
return nil, err
}
@@ -538,17 +597,6 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
amd64ApplyMask(out, mask, zeroing)
return out, nil
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
out, err := in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, dest, 2)
if err != nil {
return nil, err
}
amd64ApplyMask(out, mask, zeroing)
return out, nil
}
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
@@ -557,6 +605,7 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
}
out := amd64Encode(in.Bytes, dest.Reg, -1, ops[0].Reg)
amd64ApplyMask(out, mask, zeroing)
amd64ApplyRounding(out, round)
return out, nil
}
@@ -601,6 +650,9 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
if op.Zeroing {
return 0, fmt.Errorf("%s: operand %d carries zeroing, the immediate takes none", in.Name, pos)
}
if op.Round != ExtRoundNone {
return 0, fmt.Errorf("%s: operand %d carries a rounding control, the immediate takes none", in.Name, pos)
}
if op.Imm < 0 || op.Imm > 255 {
return 0, fmt.Errorf("%s: operand %d is immediate %d, outside the unsigned byte range 0-255", in.Name, pos, op.Imm)
}
@@ -676,7 +728,7 @@ func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
return append(out, imm), nil
}
// ExtFP16RoundingModes names the two-bit rounding mode the round control of
// ExtRoundingMode names the two-bit rounding mode the round control of
// VRNDSCALESH and VREDUCESH carries, indexed by imm8[1:0], the SDM's RC
// field encoding.
var ExtFP16RoundingModes = [4]string{
@@ -686,6 +738,87 @@ var ExtFP16RoundingModes = [4]string{
"round toward zero (truncate)",
}
// ExtRounding names the rounding decoration an FP destination of the amd64
// layer carries, the SDM's {sae} and {rn-sae} through {rz-sae} spellings.
// The decoration is the EVEX embedded rounding: EVEX.b, bit four of byte
// three, selects the rounding context, and EVEX.RC, the two bits above it
// that carry L'L in every other word, names the mode, in the very encoding
// the imm8 round control uses: 00 round to nearest even, 01 down toward
// negative infinity, 10 up toward positive infinity, 11 toward zero. The
// L'L bits the template carries are cleared when the decoration applies,
// which is why only the 512-bit and the scalar forms take one: a word with
// an embedded mode has no vector length of its own to spell. ExtRoundSAE
// suppresses the floating-point exceptions alone and leaves EVEX.RC zero;
// the entries that take it (Er) demand a mode, the entries that take SAE
// alone refuse every mode, exactly the split the manual's two decoration
// classes draw.
type ExtRounding uint8
// The rounding decorations, ExtRoundNone first as the zero value every
// undecorated destination carries.
const (
// ExtRoundNone marks a destination without a rounding decoration: the
// MXCSR rules govern the operation.
ExtRoundNone ExtRounding = iota
// ExtRoundSAE is {sae}: suppress all floating-point exceptions, no mode
// named.
ExtRoundSAE
// ExtRoundNearest is {rn-sae}: round to nearest, ties to even, EVEX.RC 00.
ExtRoundNearest
// ExtRoundDown is {rd-sae}: round down toward negative infinity, EVEX.RC 01.
ExtRoundDown
// ExtRoundUp is {ru-sae}: round up toward positive infinity, EVEX.RC 10.
ExtRoundUp
// ExtRoundTruncate is {rz-sae}: round toward zero, EVEX.RC 11.
ExtRoundTruncate
)
// String returns the spelling the decoration carries in the reference
// listings, for diagnostics.
func (r ExtRounding) String() string {
switch r {
case ExtRoundSAE:
return "{sae}"
case ExtRoundNearest:
return "{rn-sae}"
case ExtRoundDown:
return "{rd-sae}"
case ExtRoundUp:
return "{ru-sae}"
case ExtRoundTruncate:
return "{rz-sae}"
default:
return "no rounding control"
}
}
// amd64RoundingBits lays the decoration's EVEX.RC encoding out: the four
// named modes in the SDM's RC field order, and {sae}, which names no mode,
// as the nearest-even bits the exception suppression shares.
func amd64RoundingBits(r ExtRounding) byte {
switch r {
case ExtRoundDown:
return 1
case ExtRoundUp:
return 2
case ExtRoundTruncate:
return 3
default:
return 0
}
}
// ExtRounded builds the rounded spelling of an FP destination, ZMM1{k7}{z},
// {rz-sae} style: the operation rounds by the named mode instead of the
// MXCSR, or suppresses its exceptions under {sae}. The mode rides the
// destination operand beside the write mask, the two decorations the
// reference listings spell on the one line; they compose, EVEX.aaa and
// EVEX.z keeping their bits under EVEX.RC.
func ExtRounded(dest ExtOperand, mode ExtRounding) ExtOperand {
dest.Round = mode
return dest
}
// ExtFP16GetMantSigns names the sign control imm8[3:2] of the VGETMANTSH
// immediate, indexed by the field: the source's own sign, a forced positive,
// and the two encodings that yield the indefinite NaN on a negative source.
@@ -709,25 +842,36 @@ var ExtFP16CmpPredicates = [32]string{
// encodeAmdVecGprVec fills the conversion form with a general-register
// source: src1, gpr, dest. VCVTSI2SH X1, X2, EAX style. An entry with Mem
// set takes the memory shape of the integer source, which the manual spells
// r/m32: the value converts straight out of memory.
// r/m32: the value converts straight out of memory. The register form takes
// the rounding decoration the integer converts carry; the memory shape
// refuses one.
func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
dest, _, _, round, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
if err := in.amd64Vector(ops[2], class, 3); err != nil {
if round != ExtRoundNone {
return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name)
}
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
return in.amd64MemBytes(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1], 2)
return in.amd64MemBytes(in.Bytes, dest.Reg, ops[0].Reg, ops[1], 2)
}
if err := in.amd64Gpr(ops[1], 2); err != nil {
return nil, err
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
out := amd64Encode(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg)
amd64ApplyRounding(out, round)
return out, nil
}
// encodeAmdGprPair fills the two-operand general-register forms: gpr, vec
@@ -845,40 +989,40 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r, m16 destination)"},
{Name: "VADDSH", Summary: "Add scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDSH (EVEX.NDS.LIG.F3.MAP5.W0 58 /r)"},
{Name: "VSUBSH", Summary: "Subtract scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBSH (EVEX.NDS.LIG.F3.MAP5.W0 5C /r)"},
{Name: "VMULSH", Summary: "Multiply scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULSH (EVEX.NDS.LIG.F3.MAP5.W0 59 /r)"},
{Name: "VDIVSH", Summary: "Divide scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVSH (EVEX.NDS.LIG.F3.MAP5.W0 5E /r)"},
{Name: "VMINSH", Summary: "Return the minimum of scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINSH (EVEX.NDS.LIG.F3.MAP5.W0 5D /r)"},
{Name: "VMAXSH", Summary: "Return the maximum of scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXSH (EVEX.NDS.LIG.F3.MAP5.W0 5F /r)"},
{Name: "VSQRTSH", Summary: "Compute the square root of a scalar FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"},
{Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSCALEFSH (EVEX.NDS.LIG.66.MAP6.W0 2D /r)"},
{Name: "VGETEXPSH", Summary: "Convert the exponent of a scalar FP16 value to an FP16 value",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETEXPSH (EVEX.NDS.LIG.66.MAP6.W0 43 /r)"},
{Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCOMISH (EVEX.LIG.MAP5.W0 2F /r)"},
{Name: "VUCOMISH", Summary: "Unordered-compare a scalar FP16 value and set EFLAGS",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"},
{Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"},
{Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value",
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
@@ -887,19 +1031,19 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"},
{Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"},
{Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 2A /r)"},
{Name: "VCVTSI2SH", Summary: "Convert one signed 64-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 2A /r)"},
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 32-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 7B /r)"},
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 64-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"},
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
@@ -920,7 +1064,7 @@ var amd64Extensions = []ExtInstr{
// quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from
// avx512_fp16_vl.d, on the same low registers the suite uses.
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"},
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
@@ -929,7 +1073,7 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
@@ -938,7 +1082,7 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
@@ -947,7 +1091,7 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
@@ -956,7 +1100,7 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Sae: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
@@ -965,7 +1109,7 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Sae: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
@@ -974,7 +1118,7 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,