Files
gasm-sdk/arch/amd64_ext.go
T
2026-10-07 19:49:15 +02:00

1633 lines
101 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// This file carries the amd64 side of the extended-instruction layer:
// instructions the Go toolchain does not know at all, described as data and
// validated against golden vectors from the Intel SDM rather than against the
// toolchain. It sits beside the generated table, never inside it:
// arch/amd64_gen.go stays untouched, and asm.Encodable keeps answering false
// for every mnemonic here, so the layer stays out of the main encoders.
//
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
// latter's scalar core with its imm8-control group, its packed 512-bit and
// VL arithmetic, the packed mirror of the imm8-control group, the embedded
// rounding of its FP operations and its fourteen packed conversion
// directions, in their EVEX register forms. The encodings are transcribed
// from the SDM instruction entries and cross-checked against binutils-gdb's
// assembler testsuite; the golden vectors in amd64_ext_test.go pin the
// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19
// survey, no longer belong here: the Go toolchain's assembler knows them
// today, they live in the generated table and the EVEX encoder, and a
// mnemonic the toolchain has is not an extension. VCVTPS2PH, VCVTUDQ2PS
// and the rest of the classic conversion set follow the same rule, which is
// why the conversions here are the FP16 directions the toolchain has never
// emitted.
//
// The forms encode the unmasked shapes: register forms throughout, and the
// memory forms beside them, the scalar ones the manual spells m16, m32 and
// m64 and the packed ones with the {1toN} broadcast, base-relative operands
// with the ModR/M disp8 and disp32 choices and the SIB byte RSP and R12
// demand, the scaled index and the broadcast laying the SIB byte and EVEX.b
// over the same displacement semantics. The packed destinations take the
// write mask beside all of it, the {k1}{z} decorations the SDM spells:
// EVEX.aaa carries the masking register and EVEX.z the zeroing bit, K0 masks
// nothing, and the scalar forms, the compares, the opmask destinations and
// the load and store shapes take no mask at all. The FP destinations take
// the embedded rounding beside that, the {sae} and {rn-sae} through
// {rz-sae} decorations: EVEX.b selects the rounding context and EVEX.RC,
// the L'L bits it replaces, the mode, on the 512-bit and scalar register
// forms alone, exactly the two lengths whose word has no vector length left
// to lose.
package arch
import "fmt"
// The features the amd64 layer covers.
const (
ExtFeatureBF16 ExtFeature = "avx512bf16"
ExtFeatureVP2INTERSECT ExtFeature = "avx512vp2intersect"
ExtFeatureFP16 ExtFeature = "avx512fp16"
)
// ExtXmm, ExtYmm and ExtZmm build vector operands of the three EVEX register
// widths, VADDPS ZMM1, ZMM2, ZMM3 style. The register number runs 0..31,
// XMM16 and above included: EVEX carries five register bits in every
// position, and the golden vectors exercise the high registers on purpose.
func ExtXmm(reg int) ExtOperand { return ExtOperand{Kind: ExtXMM, Reg: reg} }
func ExtYmm(reg int) ExtOperand { return ExtOperand{Kind: ExtYMM, Reg: reg} }
func ExtZmm(reg int) ExtOperand { return ExtOperand{Kind: ExtZMM, Reg: reg} }
// ExtMask builds an opmask operand, VP2INTERSECTD K1, ZMM2, ZMM3 style. The
// register number runs 0..7.
func ExtMask(reg int) ExtOperand { return ExtOperand{Kind: ExtKReg, Reg: reg} }
// ExtWriteMasked builds the write-masked spelling of a packed destination,
// ZMM1{k7}{z} style: only the lanes mask selects take the result, and the
// zeroing flag turns the inactive lanes into zeros instead of keeping the
// destination's. The register runs 1..7, K0 never masks.
func ExtWriteMasked(dest ExtOperand, mask int, zeroing bool) ExtOperand {
dest.Mask = mask
dest.HasMask = true
dest.Zeroing = zeroing
return dest
}
// ExtGpr32 and ExtGpr64 build general-register operands, VCVTSI2SH XMM1,
// XMM2, EAX style. The register number runs 0..15.
func ExtGpr32(reg int) ExtOperand { return ExtOperand{Kind: ExtR32, Reg: reg} }
func ExtGpr64(reg int) ExtOperand { return ExtOperand{Kind: ExtR64, Reg: reg} }
// amd64LengthClass reads the vector length the template encodes out of the
// L'L field of the EVEX byte three and names the register class every vector
// operand of that entry must carry.
func amd64LengthClass(b []byte) ExtOperandKind {
switch (b[3] >> 5) & 3 {
case 0:
return ExtXMM
case 1:
return ExtYMM
default:
return ExtZMM
}
}
// amd64HalfClass names the half-width companion of a vector class, the
// destination class of the narrow conversions. At 128 bits the companion is
// the class itself, which is what the manual gives for the narrowest form.
func amd64HalfClass(k ExtOperandKind) ExtOperandKind {
switch k {
case ExtZMM:
return ExtYMM
case ExtYMM:
return ExtXMM
default:
return ExtXMM
}
}
// amd64Encode returns the template with the register-derived bits filled in:
// dest and rm are register numbers for the ModR/M reg and r/m fields, vvvv is
// the third-operand register or -1 when the form leaves it unused. The EVEX
// plumbing follows the encoder in asm: reg[3] rides R bar and reg[4] R prime
// bar, rm[3] rides B bar, and in a register form rm[4] rides X bar, while
// vvvv[4] rides V prime bar in byte three.
func amd64Encode(b []byte, dest, vvvv, rm int) []byte {
out := make([]byte, len(b))
copy(out, b)
rBar, rPrimeBar := 1, 1
if dest&8 != 0 {
rBar = 0
}
if dest&16 != 0 {
rPrimeBar = 0
}
xBar, bBar := 1, 1
if rm&8 != 0 {
bBar = 0
}
if rm&16 != 0 {
xBar = 0
}
out[1] |= byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4)
vBar, vPrimeBar := 15, 1
if vvvv >= 0 {
vBar = 15 - (vvvv & 15)
if vvvv&16 != 0 {
vPrimeBar = 0
}
}
out[2] |= byte(vBar << 3)
out[3] |= byte(vPrimeBar << 3)
out[5] |= byte((dest&7)<<3 | rm&7)
return out
}
// amd64Memory validates a memory operand of an amd64 entry: no arrangement
// and no qualifier, a base general register inside 0-15, a signed 32-bit
// displacement and no shift. The base number rides the operand's Reg and
// the displacement its Imm. A broadcast spelling is refused unless the
// entry carries Bcast: the scalar forms read a plain m16 and the full-width
// sources a plain vector, and neither splats.
func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, err error) {
if op.Kind != ExtMem {
return 0, 0, fmt.Errorf("%s: operand %d wants a memory operand, got %s", in.Name, pos, op.Kind)
}
if op.Arr != ExtArrNone {
return 0, 0, fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
}
if op.Qual != ExtQualNone {
return 0, 0, fmt.Errorf("%s: operand %d carries a predicate qualifier, the amd64 layer takes none", in.Name, pos)
}
if op.HasMask {
return 0, 0, fmt.Errorf("%s: operand %d carries a write mask, the memory operand takes none", in.Name, pos)
}
if op.Zeroing {
return 0, 0, fmt.Errorf("%s: operand %d carries zeroing, the memory operand takes none", in.Name, pos)
}
if op.HasShift {
return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos)
}
if op.Round != ExtRoundNone {
return 0, 0, fmt.Errorf("%s: operand %d carries a rounding control, the memory operand takes none", in.Name, pos)
}
if op.Broadcast && !in.Bcast {
return 0, 0, fmt.Errorf("%s: operand %d carries a broadcast, the entry's memory operand takes none", in.Name, pos)
}
if op.Reg < 0 || op.Reg > 15 {
return 0, 0, fmt.Errorf("%s: operand %d names base register %d, outside 0-15", in.Name, pos, op.Reg)
}
if op.Imm < -1<<31 || op.Imm >= 1<<31 {
return 0, 0, fmt.Errorf("%s: operand %d carries displacement %d, outside the signed 32-bit range", in.Name, pos, op.Imm)
}
return op.Reg, op.Imm, nil
}
// amd64DispTail lays the displacement bytes and their ModR/M mod field: no
// bytes at displacement zero, a disp8 when the value fits a signed byte and
// a disp32 otherwise. The force flag makes a zero displacement encode
// anyway, the mod-01 shape the RBP and R13 bases demand, whose bare r/m 101
// would otherwise address RIP-relative.
func amd64DispTail(disp int64, force bool) (mod byte, tail []byte) {
if !force && disp == 0 {
return 0, nil
}
if disp >= -128 && disp <= 127 {
return 1, []byte{byte(disp)}
}
return 2, []byte{byte(disp), byte(disp >> 8), byte(disp >> 16), byte(disp >> 24)}
}
// amd64Sib builds the SIB byte the scaled forms carry: the scale field over
// the index and the base, whose number rides the r/m field.
func amd64Sib(base, index, scale int) byte {
sib := byte(base & 7)
sib |= byte(index&7) << 3
switch scale {
case 2:
sib |= 1 << 6
case 4:
sib |= 2 << 6
case 8:
sib |= 3 << 6
}
return sib
}
// amd64EncodeMemory returns the register-form template with a base-relative
// memory operand filled in: dest and vvvv keep their register meanings, the
// ModR/M r/m field carries the base, and the high base bit rides EVEX.B as
// amd64Encode lays it. The ModR/M mod bits and the trailing SIB and
// displacement bytes follow the canonical choices the GNU assembler makes
// for the plain, unscaled SDM displacements, and the operand must have
// passed amd64Memory first.
func amd64EncodeMemory(b []byte, dest, vvvv, base int, disp int64) []byte {
out := amd64Encode(b, dest, vvvv, base)
rm := base & 7
mod, tail := amd64DispTail(disp, rm == 5)
if rm == 4 {
// RSP and R12 need the SIB byte: no index, base 100.
tail = append([]byte{0x24}, tail...)
}
out[5] = out[5]&0x3f | mod<<6
return append(out, tail...)
}
// amd64EncodeBroadcast returns the memory encoding with EVEX.b set: the
// {1toN} broadcast, whose single element the hardware splats across every
// lane of the destination. EVEX.b is bit 4 of byte three, and the ModR/M,
// SIB and displacement bytes keep the plain semantics amd64EncodeMemory
// chooses; only the prefix bit changes. The operand must have passed
// amd64Memory on an entry that carries Bcast.
func amd64EncodeBroadcast(b []byte, dest, vvvv, base int, disp int64) []byte {
out := amd64EncodeMemory(b, dest, vvvv, base, disp)
out[3] |= 0x10
return out
}
// amd64EncodeScaledMemory returns the register-form template with a
// base-plus-scaled-index memory operand filled in: the SIB byte follows the
// ModR/M and carries the scale field, the index and the base, whose number
// rides the r/m field as 100. In a SIB form EVEX.B keeps carrying base bit
// three, as amd64Encode laid it from the base, and EVEX.X changes meaning
// from the register's bit four to the index's bit three, so it clears when
// the index sits above 7. The ModR/M and displacement choices stay the
// canonical ones amd64EncodeMemory makes, with the RBP and R13 bases
// keeping their forced displacement: with a SIB byte present, mod 00 with
// base 101 addresses baseless disp32, never through the base.
func amd64EncodeScaledMemory(b []byte, dest, vvvv, base, index, scale int, disp int64) []byte {
out := amd64Encode(b, dest, vvvv, base)
if index&8 != 0 {
out[1] &^= 0x40
}
rm := base & 7
mod, tail := amd64DispTail(disp, rm == 5)
tail = append([]byte{amd64Sib(rm, index, scale)}, tail...)
out[5] = out[5]&0x38 | mod<<6 | 4
return append(out, tail...)
}
// amd64Index validates the scaled index of a memory operand: a general
// register inside 0-15 and never RSP, whose SIB encoding 100 means no index,
// and a scale the byte multipliers carry.
func (in ExtInstr) amd64Index(op ExtOperand, pos int) (index, scale int, err error) {
if op.Index < 0 || op.Index > 15 {
return 0, 0, fmt.Errorf("%s: operand %d names index register %d, outside 0-15", in.Name, pos, op.Index)
}
if op.Index == 4 {
return 0, 0, fmt.Errorf("%s: operand %d names RSP as the index, which the SIB byte cannot encode", in.Name, pos)
}
switch op.Scale {
case 1, 2, 4, 8:
default:
return 0, 0, fmt.Errorf("%s: operand %d carries a scale of %d, outside the byte multipliers 1, 2, 4 and 8", in.Name, pos, op.Scale)
}
return op.Index, op.Scale, nil
}
// amd64MemBytes encodes one validated memory position: the plain
// base-plus-displacement form, the scaled index over it, and the broadcast
// bit over either, each an additive layer on the same displacement
// semantics. The operand must have passed amd64Memory's kind gate, which
// the encode paths reach only at the entry's Mem position.
func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos int) ([]byte, error) {
base, disp, err := in.amd64Memory(op, pos)
if err != nil {
return nil, err
}
if op.HasIndex {
index, scale, err := in.amd64Index(op, pos)
if err != nil {
return nil, err
}
out := amd64EncodeScaledMemory(b, dest, vvvv, base, index, scale, disp)
if op.Broadcast {
out[3] |= 0x10
}
return out, nil
}
if op.Broadcast {
return amd64EncodeBroadcast(b, dest, vvvv, base, disp), nil
}
return amd64EncodeMemory(b, dest, vvvv, base, disp), nil
}
// amd64WriteMask lifts the decorations off a destination operand: the
// returned copy carries the register bits alone, while the mask register,
// the zeroing flag and the rounding control come back beside it. K0 never
// masks, zeroing is valid only beside a mask, and an entry whose destination
// takes neither decoration refuses the spellings outright, so nothing
// arrives at the encoding half-claimed. The rounding decoration is
// validated against the entry's capability: an entry that carries neither
// Er nor Sae takes none, an Er entry demands one of the four modes, and an
// entry that suppresses exceptions alone takes {sae} and refuses every
// mode, the split the manual's two decoration classes draw.
func (in ExtInstr) amd64WriteMask(op ExtOperand, pos int) (ExtOperand, int, bool, ExtRounding, error) {
stripped := op
stripped.Mask, stripped.HasMask, stripped.Zeroing = 0, false, false
round := op.Round
stripped.Round = ExtRoundNone
if !op.HasMask && !op.Zeroing && round == ExtRoundNone {
return stripped, 0, false, ExtRoundNone, nil
}
if op.HasMask || op.Zeroing {
if !in.Mask {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries a write mask, the entry's destination takes none", in.Name, pos)
}
if !op.HasMask {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries zeroing without a write mask", in.Name, pos)
}
if op.Mask < 1 || op.Mask > 7 {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d names write mask k%d, outside the masking registers k1-k7", in.Name, pos, op.Mask)
}
}
if round != ExtRoundNone {
if !in.Er && !in.Sae {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries a rounding control, the entry's destination takes none", in.Name, pos)
}
if in.Sae && round != ExtRoundSAE {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d names the rounding mode %s, the entry suppresses exceptions alone and takes {sae}", in.Name, pos, round)
}
if in.Er && round == ExtRoundSAE {
return stripped, 0, false, round, fmt.Errorf("%s: operand %d spells {sae} without a mode, the embedded rounding wants {rn-sae} through {rz-sae}", in.Name, pos)
}
}
return stripped, op.Mask, op.Zeroing, round, nil
}
// amd64ApplyRounding lays the validated rounding decoration into the encoded
// word: EVEX.b, bit four of byte three, selects the rounding context, and
// EVEX.RC, the L'L bits, names the mode. The L'L bits the template carries
// are cleared, because a word with an embedded rounding mode has no vector
// length of its own to spell; the write mask keeps its bits under the
// decoration, EVEX.aaa and EVEX.z untouched. An undecorated destination
// leaves the word alone: its L'L carries the vector length.
func amd64ApplyRounding(b []byte, mode ExtRounding) {
if mode == ExtRoundNone {
return
}
b[3] &^= 0x60
b[3] |= 0x10
b[3] |= amd64RoundingBits(mode) << 5
}
// amd64ApplyMask lays the validated write mask into the encoded word:
// EVEX.aaa, bits two through nought of byte three, carries the masking
// register and EVEX.z, bit seven, the zeroing flag. An unmasked destination
// leaves the bits the template carries, which the template integrity keeps
// zero.
func amd64ApplyMask(b []byte, mask int, zeroing bool) {
if mask > 0 {
b[3] |= byte(mask)
}
if zeroing {
b[3] |= 0x80
}
}
// amd64PlainReg checks the invariants every amd64 register operand carries:
// no arm64 arrangement, no predicate qualifier, and a register number inside
// the class the instruction encodes. A broadcast spelling names a memory
// location, so a register position refuses it outright, and a write mask or
// zeroing decoration belongs to a destination alone, which its encoder lifts
// before the position checks see the operand: any spelling that survives to
// here sits on a position that takes none.
func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
if op.Broadcast {
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
}
if op.HasMask {
return fmt.Errorf("%s: operand %d carries a write mask, the position takes none", in.Name, pos)
}
if op.Zeroing {
return fmt.Errorf("%s: operand %d carries zeroing, the position takes none", in.Name, pos)
}
if op.Round != ExtRoundNone {
return fmt.Errorf("%s: operand %d carries a rounding control, the position takes none", in.Name, pos)
}
if op.Arr != ExtArrNone {
return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
}
if op.Qual != ExtQualNone {
return fmt.Errorf("%s: operand %d carries a predicate qualifier, the amd64 layer takes none", in.Name, pos)
}
if op.Reg < 0 || op.Reg > max {
return fmt.Errorf("%s: operand %d is register %d, outside 0-%d", in.Name, pos, op.Reg, max)
}
return nil
}
// amd64Vector checks one vector operand against the class the entry encodes.
// The broadcast spelling is named before the kind, so the diagnostic says
// what the operand carries rather than what the position wanted.
func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error {
if op.Broadcast {
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
}
if op.Kind != class {
article := "a"
if class == ExtXMM {
article = "an"
}
return fmt.Errorf("%s: operand %d wants %s %s, got %s", in.Name, pos, article, class, op.Kind)
}
return in.amd64PlainReg(op, 31, pos)
}
// amd64Gpr checks the general-register operand against the width the entry
// encodes: the W bit picks 32-bit or 64-bit, unless the entry ignores W, and
// the general registers run 0..15.
func (in ExtInstr) amd64Gpr(op ExtOperand, pos int) error {
want := ExtR32
if in.Bytes[2]&0x80 != 0 {
want = ExtR64
}
if in.Wig {
if op.Kind != ExtR32 && op.Kind != ExtR64 {
return fmt.Errorf("%s: operand %d wants a 32-bit or 64-bit general register, got %s", in.Name, pos, op.Kind)
}
} else if op.Kind != want {
return fmt.Errorf("%s: operand %d wants a %s, got %s", in.Name, pos, want, op.Kind)
}
return in.amd64PlainReg(op, 15, pos)
}
// encodeAmd64 encodes the amd64 forms: it validates the operand list against
// the class the template encodes and fills the register bits. An operand the
// form cannot carry is an error, never a silent mis-encoding.
func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
switch in.Form {
case ExtFormAmdVec3:
return in.encodeAmdVec3(ops)
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter,
ExtFormAmdVec2ToQuarter:
return in.encodeAmdVec2(ops)
case ExtFormAmdMask2:
return in.encodeAmdMask2(ops)
case ExtFormAmdVecGprVec:
return in.encodeAmdVecGprVec(ops)
case ExtFormAmdGprVec, ExtFormAmdVecGpr:
return in.encodeAmdGprPair(ops)
case ExtFormAmdMemVec:
return in.encodeAmdMemVec(ops)
case ExtFormAmdVecMem:
return in.encodeAmdVecMem(ops)
case ExtFormAmdVec3Imm:
return in.encodeAmdVec3Imm(ops)
case ExtFormAmdMask2Imm:
return in.encodeAmdMask2Imm(ops)
case ExtFormAmdVec2Imm:
return in.encodeAmdVec2Imm(ops)
default:
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
}
}
// encodeAmdVec3 fills the non-destructive three-vector form: src1, src2,
// dest, all under one register class. An entry with Mem set takes the
// memory shape of that position too: the second source of the scalar
// arithmetic, spelled xmm3/m16 in the manual, may be a base-relative
// operand, which rides the r/m field with its displacement bytes after the
// opcode, and the packed entries lay the source's {1toN} broadcast over the
// same encoding as EVEX.b. An entry with Er or Sae set takes the rounding
// decoration on its register form alone: the memory shape refuses one, the
// embedded rounding and the broadcast share EVEX.b and cannot ride the same
// word.
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if round != ExtRoundNone {
return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name)
}
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, ops[0].Reg, ops[1], 2)
if err != nil {
return nil, err
}
amd64ApplyMask(out, mask, zeroing)
return out, nil
}
for i, op := range ops[1:2] {
if err := in.amd64Vector(op, class, i+2); err != nil {
return nil, err
}
}
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg)
amd64ApplyMask(out, mask, zeroing)
amd64ApplyRounding(out, round)
return out, nil
}
// encodeAmdMemVec fills the memory-load form: mem, dest. VMOVSH X30,
// 4660(R8) shape, the manual's xmm1, m16 lines beside the register form.
// The form reads one value from memory, so the third register slot stays
// unused, which the encoding spells as vvvv 1111.
func (in ExtInstr) encodeAmdMemVec(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1)
}
// encodeAmdVecMem fills the memory-store form: src, mem. VMOVSH 4660(R9),
// X29 shape, the manual's m16, xmm1 lines. The register source sits in the
// ModR/M reg field and the memory destination in r/m, and vvvv stays
// unused.
func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
return in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, ops[1], 2)
}
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
// of destination, and at 128 bits the companion stays the class itself. The
// wide form narrows the source instead, the widening FP16 conversions, where
// L'L names the destination: VCVTPH2DQ converts 256 bits of source into 512
// bits of destination. The two quarter forms hold one operand in the XMM
// class at every length: the source under the widening VCVTPH2QQ, the
// destination under the narrowing VCVTQQ2PH. An entry with Mem set takes
// the memory shape of the source position too: the compares and the packed
// square root read their source from memory, the packed square root's source
// carrying the {1toN} broadcast as EVEX.b, and the narrow BF16 convert reads
// its full-width source there. An entry with Er or Sae set takes the
// rounding decoration on its register form alone; the memory shape refuses
// one.
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
destClass, srcClass := class, class
switch in.Form {
case ExtFormAmdVec2Half:
destClass = amd64HalfClass(class)
case ExtFormAmdVec2Wide:
srcClass = amd64HalfClass(class)
case ExtFormAmdVec2Quarter:
srcClass = ExtXMM
case ExtFormAmdVec2ToQuarter:
destClass = ExtXMM
}
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[1], 2)
if err != nil {
return nil, err
}
if in.Mem == 1 && ops[0].Kind == ExtMem {
// The memory spelling is validated before the destination: on the
// narrow form the destination's narrowed class is the likelier
// rejection, but a miswritten source spelling names itself first.
if _, _, err := in.amd64Memory(ops[0], 1); err != nil {
return nil, err
}
if round != ExtRoundNone {
return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name)
}
if err := in.amd64Vector(dest, destClass, 2); err != nil {
return nil, err
}
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, -1, ops[0], 1)
if err != nil {
return nil, err
}
amd64ApplyMask(out, mask, zeroing)
return out, nil
}
if err := in.amd64Vector(ops[0], srcClass, 1); err != nil {
return nil, err
}
if err := in.amd64Vector(dest, destClass, 2); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, dest.Reg, -1, ops[0].Reg)
amd64ApplyMask(out, mask, zeroing)
amd64ApplyRounding(out, round)
return out, nil
}
// encodeAmdMask2 fills the mask-destination form: src1, src2, dest, where the
// destination is an opmask register and both sources share the class.
func (in ExtInstr) encodeAmdMask2(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
if ops[2].Kind != ExtKReg {
return nil, fmt.Errorf("%s: operand 3 wants an opmask register, got %s", in.Name, ops[2].Kind)
}
if err := in.amd64PlainReg(ops[2], 7, 3); err != nil {
return nil, err
}
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
}
// amd64Imm8 validates the leading immediate operand of an imm8-control form:
// an ExtImm with no shift, inside the unsigned byte range, and free of the
// bits the entry's control layout reserves. The reserved upper nibble of
// the VGETMANTSH control must encode as zero; the SDM marks every other
// layout here fully defined, and the VCMPSH hardware masks its predicate to
// five bits.
func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
if op.Kind != ExtImm {
return 0, fmt.Errorf("%s: operand %d wants an immediate control byte, got %s", in.Name, pos, op.Kind)
}
if op.Arr != ExtArrNone {
return 0, fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
}
if op.HasShift {
return 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 imm8 forms take none", in.Name, pos)
}
if op.HasMask {
return 0, fmt.Errorf("%s: operand %d carries a write mask, the immediate takes none", in.Name, pos)
}
if op.Zeroing {
return 0, fmt.Errorf("%s: operand %d carries zeroing, the immediate takes none", in.Name, pos)
}
if op.Round != ExtRoundNone {
return 0, fmt.Errorf("%s: operand %d carries a rounding control, the immediate takes none", in.Name, pos)
}
if op.Imm < 0 || op.Imm > 255 {
return 0, fmt.Errorf("%s: operand %d is immediate %d, outside the unsigned byte range 0-255", in.Name, pos, op.Imm)
}
if in.Imm8 == ExtImm8GetMant && op.Imm > 15 {
return 0, fmt.Errorf("%s: operand %d is immediate %d, the upper nibble of the mantissa control is reserved and must be zero", in.Name, pos, op.Imm)
}
return byte(op.Imm), nil
}
// encodeAmdVec3Imm fills the three-vector form with a control immediate:
// imm, src1, src2, dest, the order the reference listings write it in. An
// entry with Mem set takes the memory shape of the second source, xmm3/m16
// in the manual. The destination takes the write mask where the entry
// declares one, the packed imm8-control forms the manual masks; the scalar
// forms take none, and the entry's flag refuses the spelling for them.
func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
imm, err := in.amd64Imm8(ops[0], 1)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
dest, mask, zeroing, _, err := in.amd64WriteMask(ops[3], 4)
if err != nil {
return nil, err
}
if in.Mem == 3 && ops[2].Kind == ExtMem {
if err := in.amd64Vector(dest, class, 4); err != nil {
return nil, err
}
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, ops[1].Reg, ops[2], 3)
if err != nil {
return nil, err
}
amd64ApplyMask(out, mask, zeroing)
return append(out, imm), nil
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
if err := in.amd64Vector(dest, class, 4); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, dest.Reg, ops[1].Reg, ops[2].Reg)
amd64ApplyMask(out, mask, zeroing)
return append(out, imm), nil
}
// encodeAmdVec2Imm fills the two-vector form with a control immediate: imm,
// src, dest, the packed imm8-control group. An entry with Mem set takes the
// memory shape of the source, zmm2/m512 in the manual; the control byte
// rides after the ModR/M and its displacement bytes, the last byte of the
// word.
func (in ExtInstr) encodeAmdVec2Imm(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
imm, err := in.amd64Imm8(ops[0], 1)
if err != nil {
return nil, err
}
dest, mask, zeroing, _, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, -1, ops[1], 2)
if err != nil {
return nil, err
}
amd64ApplyMask(out, mask, zeroing)
return append(out, imm), nil
}
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, dest.Reg, -1, ops[1].Reg)
amd64ApplyMask(out, mask, zeroing)
return append(out, imm), nil
}
// encodeAmdMask2Imm fills the opmask-destination form with a control
// immediate: imm, src1, src2, dest. An entry with Mem set takes the memory
// shape of the second source.
func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
imm, err := in.amd64Imm8(ops[0], 1)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
if ops[3].Kind != ExtKReg {
return nil, fmt.Errorf("%s: operand 4 wants an opmask register, got %s", in.Name, ops[3].Kind)
}
if err := in.amd64PlainReg(ops[3], 7, 4); err != nil {
return nil, err
}
var out []byte
if in.Mem == 3 && ops[2].Kind == ExtMem {
out, err := in.amd64MemBytes(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2], 3)
if err != nil {
return nil, err
}
return append(out, imm), nil
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
out = amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
return append(out, imm), nil
}
// ExtRoundingMode names the two-bit rounding mode the round control of
// VRNDSCALESH and VREDUCESH carries, indexed by imm8[1:0], the SDM's RC
// field encoding.
var ExtFP16RoundingModes = [4]string{
"round to nearest (even)",
"round down (toward -infinity)",
"round up (toward +infinity)",
"round toward zero (truncate)",
}
// ExtRounding names the rounding decoration an FP destination of the amd64
// layer carries, the SDM's {sae} and {rn-sae} through {rz-sae} spellings.
// The decoration is the EVEX embedded rounding: EVEX.b, bit four of byte
// three, selects the rounding context, and EVEX.RC, the two bits above it
// that carry L'L in every other word, names the mode, in the very encoding
// the imm8 round control uses: 00 round to nearest even, 01 down toward
// negative infinity, 10 up toward positive infinity, 11 toward zero. The
// L'L bits the template carries are cleared when the decoration applies,
// which is why only the 512-bit and the scalar forms take one: a word with
// an embedded mode has no vector length of its own to spell. ExtRoundSAE
// suppresses the floating-point exceptions alone and leaves EVEX.RC zero;
// the entries that take it (Er) demand a mode, the entries that take SAE
// alone refuse every mode, exactly the split the manual's two decoration
// classes draw.
type ExtRounding uint8
// The rounding decorations, ExtRoundNone first as the zero value every
// undecorated destination carries.
const (
// ExtRoundNone marks a destination without a rounding decoration: the
// MXCSR rules govern the operation.
ExtRoundNone ExtRounding = iota
// ExtRoundSAE is {sae}: suppress all floating-point exceptions, no mode
// named.
ExtRoundSAE
// ExtRoundNearest is {rn-sae}: round to nearest, ties to even, EVEX.RC 00.
ExtRoundNearest
// ExtRoundDown is {rd-sae}: round down toward negative infinity, EVEX.RC 01.
ExtRoundDown
// ExtRoundUp is {ru-sae}: round up toward positive infinity, EVEX.RC 10.
ExtRoundUp
// ExtRoundTruncate is {rz-sae}: round toward zero, EVEX.RC 11.
ExtRoundTruncate
)
// String returns the spelling the decoration carries in the reference
// listings, for diagnostics.
func (r ExtRounding) String() string {
switch r {
case ExtRoundSAE:
return "{sae}"
case ExtRoundNearest:
return "{rn-sae}"
case ExtRoundDown:
return "{rd-sae}"
case ExtRoundUp:
return "{ru-sae}"
case ExtRoundTruncate:
return "{rz-sae}"
default:
return "no rounding control"
}
}
// amd64RoundingBits lays the decoration's EVEX.RC encoding out: the four
// named modes in the SDM's RC field order, and {sae}, which names no mode,
// as the nearest-even bits the exception suppression shares.
func amd64RoundingBits(r ExtRounding) byte {
switch r {
case ExtRoundDown:
return 1
case ExtRoundUp:
return 2
case ExtRoundTruncate:
return 3
default:
return 0
}
}
// ExtRounded builds the rounded spelling of an FP destination, ZMM1{k7}{z},
// {rz-sae} style: the operation rounds by the named mode instead of the
// MXCSR, or suppresses its exceptions under {sae}. The mode rides the
// destination operand beside the write mask, the two decorations the
// reference listings spell on the one line; they compose, EVEX.aaa and
// EVEX.z keeping their bits under EVEX.RC.
func ExtRounded(dest ExtOperand, mode ExtRounding) ExtOperand {
dest.Round = mode
return dest
}
// ExtFP16GetMantSigns names the sign control imm8[3:2] of the VGETMANTSH
// immediate, indexed by the field: the source's own sign, a forced positive,
// and the two encodings that yield the indefinite NaN on a negative source.
var ExtFP16GetMantSigns = [4]string{
"the sign of the source",
"positive",
"the indefinite NaN when the source is negative",
"the indefinite NaN when the source is negative",
}
// ExtFP16CmpPredicates names the 32 comparison predicates the VCMPSH
// immediate carries in imm8[4:0], in encoding order. The SDM's own
// spellings are the fixed vocabulary of the predicate suffixes.
var ExtFP16CmpPredicates = [32]string{
"EQ_OQ", "LT_OS", "LE_OS", "UNORD_Q", "NEQ_UQ", "NLT_US", "NLE_US", "ORD_Q",
"EQ_UQ", "NGE_US", "NGT_US", "FALSE_OQ", "NEQ_OQ", "GE_OS", "GT_OS", "TRUE_UQ",
"EQ_OS", "LT_OQ", "LE_OQ", "UNORD_S", "NEQ_US", "NLT_UQ", "NLE_UQ", "ORD_S",
"EQ_US", "NGE_UQ", "NGT_UQ", "FALSE_OS", "NEQ_OS", "GE_OQ", "GT_OQ", "TRUE_US",
}
// encodeAmdVecGprVec fills the conversion form with a general-register
// source: src1, gpr, dest. VCVTSI2SH X1, X2, EAX style. An entry with Mem
// set takes the memory shape of the integer source, which the manual spells
// r/m32: the value converts straight out of memory. The register form takes
// the rounding decoration the integer converts carry; the memory shape
// refuses one.
func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
dest, _, _, round, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
if round != ExtRoundNone {
return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name)
}
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
return in.amd64MemBytes(in.Bytes, dest.Reg, ops[0].Reg, ops[1], 2)
}
if err := in.amd64Gpr(ops[1], 2); err != nil {
return nil, err
}
if err := in.amd64Vector(dest, class, 3); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg)
amd64ApplyRounding(out, round)
return out, nil
}
// encodeAmdGprPair fills the two-operand general-register forms: gpr, vec
// (the move into a vector register and the integer conversions) and vec, gpr
// (the move out of one). In both orders the second operand is the
// destination in the reg field and the first the r/m source; the vector
// changes position with the form. An entry with Mem set takes the memory
// shape of the vector source, the manual's m16 beside the register: the
// value converts straight out of memory.
func (in ExtInstr) encodeAmdGprPair(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
vecPos := 1
if in.Form == ExtFormAmdVecGpr {
vecPos = 0
}
if in.Mem == 1 && in.Form == ExtFormAmdVecGpr && ops[0].Kind == ExtMem {
if err := in.amd64Gpr(ops[1], 2); err != nil {
return nil, err
}
return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1)
}
if err := in.amd64Vector(ops[vecPos], class, vecPos+1); err != nil {
return nil, err
}
if err := in.amd64Gpr(ops[1-vecPos], 2-vecPos); err != nil {
return nil, err
}
return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil
}
// --- the amd64 AVX512-BF16 and VP2INTERSECT table -----------------------------
// amd64Extensions is the extended-instruction layer of amd64. The encodings
// are transcribed from the Intel SDM instruction entries and cross-checked
// against binutils-gdb's assembler testsuite (gas/testsuite/gas/i386/
// avx512_bf16.d, avx512_bf16_vl.d and x86-64-vp2intersect.d), whose register
// forms the golden vectors in amd64_ext_test.go quote byte for byte. Each
// template carries the fixed bits of one encoding with every register-derived
// bit zero: the map selection in byte one, the W bit, the mandatory prefix
// and the reserved one-bit in byte two, the vector length in byte three, and
// the ModR/M mod bits.
var amd64Extensions = []ExtInstr{
// AVX512-BF16: the two-way packed single to BF16 conversion and the
// dot product accumulate. The prefixes differ inside the family, the
// three-register convert carries F2 while the narrow convert and the dot
// product carry F3, which the golden vectors pin byte for byte.
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.512.F2.0F38.W0 72 /r)"},
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.256.F2.0F38.W0 72 /r)"},
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.128.F2.0F38.W0 72 /r)"},
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.512.F3.0F38.W0 72 /r, YMM destination)"},
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.256.F3.0F38.W0 72 /r, XMM destination)"},
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"},
// AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask
// destination and two vector sources, EVEX.NDS.66.0F38. The instruction
// takes no write mask of its own.
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.512.F2.0F38.W0 68 /r)"},
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.256.F2.0F38.W0 68 /r)"},
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W0 68 /r)"},
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
Bytes: []byte{0x62, 0x02, 0x87, 0x40, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.512.F2.0F38.W1 68 /r)"},
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
Bytes: []byte{0x62, 0x02, 0x87, 0x20, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.256.F2.0F38.W1 68 /r)"},
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
Bytes: []byte{0x62, 0x02, 0x87, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W1 68 /r)"},
// AVX512-FP16, the scalar core: move, arithmetic, compare and
// conversion on one half-precision value in the low XMM lane, the
// register forms of the manual's scalar entries. The family lives in
// the EVEX maps five and six the toolchain has never emitted, with the
// mandatory prefixes the manual gives each entry; the golden vectors
// pin every prefix byte for byte. LIG encodes as L'L = 00, the XMM
// class alone.
{Name: "VMOVSH", Summary: "Move a scalar FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.NDS.LIG.F3.MAP5.W0 10 /r)"},
{Name: "VMOVSH", Summary: "Move a scalar FP16 value from memory into an XMM register",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdMemVec, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.LIG.F3.MAP5.W0 10 /r, m16 source)"},
{Name: "VMOVSH", Summary: "Move a scalar FP16 value from an XMM register to memory",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x11, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.LIG.F3.MAP5.W0 11 /r, m16 destination)"},
{Name: "VMOVW", Summary: "Move a word between a general register and an XMM register",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdGprVec, Feature: ExtFeatureFP16, Wig: true,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r)"},
{Name: "VMOVW", Summary: "Move a word between an XMM register and a general register",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, Wig: true,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r)"},
{Name: "VMOVW", Summary: "Move a word from memory into an XMM register",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdMemVec, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r, m16 source)"},
{Name: "VMOVW", Summary: "Move a word from an XMM register to memory",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r, m16 destination)"},
{Name: "VADDSH", Summary: "Add scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDSH (EVEX.NDS.LIG.F3.MAP5.W0 58 /r)"},
{Name: "VSUBSH", Summary: "Subtract scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBSH (EVEX.NDS.LIG.F3.MAP5.W0 5C /r)"},
{Name: "VMULSH", Summary: "Multiply scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULSH (EVEX.NDS.LIG.F3.MAP5.W0 59 /r)"},
{Name: "VDIVSH", Summary: "Divide scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVSH (EVEX.NDS.LIG.F3.MAP5.W0 5E /r)"},
{Name: "VMINSH", Summary: "Return the minimum of scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINSH (EVEX.NDS.LIG.F3.MAP5.W0 5D /r)"},
{Name: "VMAXSH", Summary: "Return the maximum of scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXSH (EVEX.NDS.LIG.F3.MAP5.W0 5F /r)"},
{Name: "VSQRTSH", Summary: "Compute the square root of a scalar FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"},
{Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSCALEFSH (EVEX.NDS.LIG.66.MAP6.W0 2D /r)"},
{Name: "VGETEXPSH", Summary: "Convert the exponent of a scalar FP16 value to an FP16 value",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETEXPSH (EVEX.NDS.LIG.66.MAP6.W0 43 /r)"},
{Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCOMISH (EVEX.LIG.MAP5.W0 2F /r)"},
{Name: "VUCOMISH", Summary: "Unordered-compare a scalar FP16 value and set EFLAGS",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"},
{Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"},
{Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value",
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SS (EVEX.NDS.LIG.MAP6.W0 13 /r)"},
{Name: "VCVTSH2SD", Summary: "Convert a low FP16 value to an FP64 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"},
{Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"},
{Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 2A /r)"},
{Name: "VCVTSI2SH", Summary: "Convert one signed 64-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 2A /r)"},
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 32-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 7B /r)"},
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 64-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"},
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W0 2D /r)"},
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 64-bit integer",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W1 2D /r)"},
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 32-bit integer",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W0 79 /r)"},
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 64-bit integer",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Mem: 1, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W1 79 /r)"},
// AVX512-FP16 packed arithmetic: the full ZMM lanes the scalar core
// mirrors plus the VL forms, EVEX.NDS.MAP5 with no mandatory prefix,
// rounding control left to MXCSR. The 512-bit register forms are
// quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from
// avx512_fp16_vl.d, on the same low registers the suite uses.
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"},
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"},
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Sae: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Sae: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"},
// AVX512-FP16 scalar, the imm8-control group: mantissa extraction,
// reduction, rounding to fraction bits and the compare into an opmask.
// Each carries its control byte as the leading immediate operand, the
// order the reference listings write it in. The controls live in map
// 0F3A: the compare with the F3 prefix the manual gives the compare
// family, the other three unprefixed. The immediate layouts and their
// tables are ExtFP16RoundingModes, ExtFP16GetMantSigns and
// ExtFP16CmpPredicates above; the reserved upper nibble of the mantissa
// control is refused rather than encoded.
{Name: "VCMPSH", Summary: "Compare scalar FP16 values into an opmask under an imm8 predicate",
Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Mem: 3, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCMPSH (EVEX.LLIG.F3.0F3A.W0 C2 /r /ib)"},
{Name: "VGETMANTSH", Summary: "Extract the normalised mantissa of a scalar FP16 value under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTSH (EVEX.LLIG.NP.0F3A.W0 27 /r /ib)"},
{Name: "VREDUCESH", Summary: "Reduce a scalar FP16 value by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCESH (EVEX.LLIG.NP.0F3A.W0 57 /r /ib)"},
{Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"},
// AVX512-FP16 packed conversions: the fourteen directions between the
// FP16 lanes and the integer and double-precision companions, three
// register widths each. L'L names the governing operand, whose class the
// form spells: the source on the narrowing converts (VCVTDQ2PH narrows
// its dword source to the half-width destination, VCVTQQ2PH and VCVTPD2PH
// to the quarter-width XMM destination), the destination on the widening
// ones (VCVTPH2DQ widens its half-width source, VCVTPH2QQ and VCVTPH2PD
// their quarter-width XMM source). The FP16-to-integer directions round
// and take Er on their 512-bit register forms; VCVTPH2PD widens exactly
// and takes none. The integer-to-FP16 sources read their elements from
// memory under the {1toN} broadcast, the FP16 sources their full-width
// vectors; every destination is write-masked. The encodings are
// transcribed from the SDM entries and pinned byte for byte against the
// local GNU assembler, whose 2.46 table matches the manual row for row.
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.512.66.MAP5.W0 7D /r)"},
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.256.66.MAP5.W0 7D /r)"},
{Name: "VCVTPH2W", Summary: "Convert packed FP16 values to packed signed 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2W (EVEX.128.66.MAP5.W0 7D /r)"},
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.512.NP.MAP5.W0 7D /r)"},
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.256.NP.MAP5.W0 7D /r)"},
{Name: "VCVTPH2UW", Summary: "Convert packed FP16 values to packed unsigned 16-bit integers",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UW (EVEX.128.NP.MAP5.W0 7D /r)"},
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.512.F3.MAP5.W0 7D /r)"},
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.256.F3.MAP5.W0 7D /r)"},
{Name: "VCVTW2PH", Summary: "Convert packed signed 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTW2PH (EVEX.128.F3.MAP5.W0 7D /r)"},
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.512.F2.MAP5.W0 7D /r)"},
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.256.F2.MAP5.W0 7D /r)"},
{Name: "VCVTUW2PH", Summary: "Convert packed unsigned 16-bit integers to packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7D, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUW2PH (EVEX.128.F2.MAP5.W0 7D /r)"},
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.512.66.MAP5.W0 5B /r, half-width source)"},
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.256.66.MAP5.W0 5B /r, half-width source)"},
{Name: "VCVTPH2DQ", Summary: "Convert packed FP16 values to packed signed 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2DQ (EVEX.128.66.MAP5.W0 5B /r, half-width source)"},
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.512.NP.MAP5.W0 79 /r, half-width source)"},
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.256.NP.MAP5.W0 79 /r, half-width source)"},
{Name: "VCVTPH2UDQ", Summary: "Convert packed FP16 values to packed unsigned 32-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Wide, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UDQ (EVEX.128.NP.MAP5.W0 79 /r, half-width source)"},
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.512.NP.MAP5.W0 5B /r, YMM destination)"},
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.256.NP.MAP5.W0 5B /r, XMM destination)"},
{Name: "VCVTDQ2PH", Summary: "Convert packed signed 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTDQ2PH (EVEX.128.NP.MAP5.W0 5B /r, XMM destination)"},
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x07, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.512.F2.MAP5.W0 7A /r, YMM destination)"},
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x07, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.256.F2.MAP5.W0 7A /r, XMM destination)"},
{Name: "VCVTUDQ2PH", Summary: "Convert packed unsigned 32-bit integers to packed FP16 values, half-width destination",
Bytes: []byte{0x62, 0x05, 0x07, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUDQ2PH (EVEX.128.F2.MAP5.W0 7A /r, XMM destination)"},
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.512.66.MAP5.W0 7B /r, quarter-width source)"},
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.256.66.MAP5.W0 7B /r, quarter-width source)"},
{Name: "VCVTPH2QQ", Summary: "Convert packed FP16 values to packed signed 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2QQ (EVEX.128.66.MAP5.W0 7B /r, quarter-width source)"},
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x40, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.512.66.MAP5.W0 79 /r, quarter-width source)"},
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x20, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.256.66.MAP5.W0 79 /r, quarter-width source)"},
{Name: "VCVTPH2UQQ", Summary: "Convert packed FP16 values to packed unsigned 64-bit integers, widened",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2UQQ (EVEX.128.66.MAP5.W0 79 /r, quarter-width source)"},
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x84, 0x40, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.512.NP.MAP5.W1 5B /r, XMM destination)"},
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x84, 0x20, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.256.NP.MAP5.W1 5B /r, XMM destination)"},
{Name: "VCVTQQ2PH", Summary: "Convert packed signed 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x84, 0x00, 0x5B, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTQQ2PH (EVEX.128.NP.MAP5.W1 5B /r, XMM destination)"},
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x87, 0x40, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.512.F2.MAP5.W1 7A /r, XMM destination)"},
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x87, 0x20, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.256.F2.MAP5.W1 7A /r, XMM destination)"},
{Name: "VCVTUQQ2PH", Summary: "Convert packed unsigned 64-bit integers to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x7A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUQQ2PH (EVEX.128.F2.MAP5.W1 7A /r, XMM destination)"},
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.512.NP.MAP5.W0 5A /r, quarter-width source)"},
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.256.NP.MAP5.W0 5A /r, quarter-width source)"},
{Name: "VCVTPH2PD", Summary: "Convert packed FP16 values to packed double-precision values, widened exactly",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2Quarter, Mem: 1, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPH2PD (EVEX.128.NP.MAP5.W0 5A /r, quarter-width source)"},
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x85, 0x40, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.512.66.MAP5.W1 5A /r, XMM destination)"},
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x85, 0x20, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.256.66.MAP5.W1 5A /r, XMM destination)"},
{Name: "VCVTPD2PH", Summary: "Convert packed double-precision values to packed FP16 values, quarter-width destination",
Bytes: []byte{0x62, 0x05, 0x85, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec2ToQuarter, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTPD2PH (EVEX.128.66.MAP5.W1 5A /r, XMM destination)"},
// AVX512-FP16 packed, the imm8-control group: the packed mirror of the
// scalar core's mantissa extraction, reduction and rounding to fraction
// bits, one control byte over every lane of the vector. The controls
// share the immediate layouts and the tables the scalar entries carry,
// ExtImm8ScaleRound and ExtImm8GetMant, the reserved upper nibble of the
// mantissa control refused rather than encoded. The sources read from
// memory full-width, no broadcast: the control governs the lanes, not a
// splatted element.
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.512.NP.0F3A.W0 08 /r /ib)"},
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.256.NP.0F3A.W0 08 /r /ib)"},
{Name: "VRNDSCALEPH", Summary: "Round packed FP16 values to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x08, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALEPH (EVEX.128.NP.0F3A.W0 08 /r /ib)"},
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.512.NP.0F3A.W0 56 /r /ib)"},
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.256.NP.0F3A.W0 56 /r /ib)"},
{Name: "VREDUCEPH", Summary: "Reduce packed FP16 values by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCEPH (EVEX.128.NP.0F3A.W0 56 /r /ib)"},
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.512.NP.0F3A.W0 26 /r /ib)"},
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.256.NP.0F3A.W0 26 /r /ib)"},
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.128.NP.0F3A.W0 26 /r /ib)"},
// AVX512-FP16 packed fused multiply-add: the twelve packed FMA
// mnemonics of the FMA group, 132, 213 and 231 under the multiply-add,
// multiply-subtract, add-subtract and subtract-add pairings, three
// register widths each. EVEX.NDS.66.MAP6.W0 throughout, the opcode low
// byte the AVX512F single-precision FMA group carries one map over: 98
// the add, 9A the subtract, 96 the add-subtract and 97 the
// subtract-add, the 213 and 231 forms ten and twenty above. The 512-bit
// register forms take the embedded rounding, the VL forms none; the
// destinations take the write mask and the memory shape of the second
// source the {1toN} broadcast. The golden vectors are quoted from the
// local GNU assembler, whose FP16 table matches the SDM entries row for
// row; the add-subtract pairing adds on the odd lanes and subtracts on
// the even ones, the subtract-add pairing the reverse.
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.512.66.MAP6.W0 98 /r)"},
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.256.66.MAP6.W0 98 /r)"},
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.128.66.MAP6.W0 98 /r)"},
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.512.66.MAP6.W0 A8 /r)"},
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.256.66.MAP6.W0 A8 /r)"},
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.128.66.MAP6.W0 A8 /r)"},
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.512.66.MAP6.W0 B8 /r)"},
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.256.66.MAP6.W0 B8 /r)"},
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.128.66.MAP6.W0 B8 /r)"},
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.512.66.MAP6.W0 9A /r)"},
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.256.66.MAP6.W0 9A /r)"},
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.128.66.MAP6.W0 9A /r)"},
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.512.66.MAP6.W0 AA /r)"},
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.256.66.MAP6.W0 AA /r)"},
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.128.66.MAP6.W0 AA /r)"},
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.512.66.MAP6.W0 BA /r)"},
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.256.66.MAP6.W0 BA /r)"},
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.128.66.MAP6.W0 BA /r)"},
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.512.66.MAP6.W0 96 /r)"},
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.256.66.MAP6.W0 96 /r)"},
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.128.66.MAP6.W0 96 /r)"},
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.512.66.MAP6.W0 A6 /r)"},
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.256.66.MAP6.W0 A6 /r)"},
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.128.66.MAP6.W0 A6 /r)"},
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.512.66.MAP6.W0 B6 /r)"},
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.256.66.MAP6.W0 B6 /r)"},
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.128.66.MAP6.W0 B6 /r)"},
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.512.66.MAP6.W0 97 /r)"},
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.256.66.MAP6.W0 97 /r)"},
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.128.66.MAP6.W0 97 /r)"},
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.512.66.MAP6.W0 A7 /r)"},
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.256.66.MAP6.W0 A7 /r)"},
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.128.66.MAP6.W0 A7 /r)"},
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.512.66.MAP6.W0 B7 /r)"},
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.256.66.MAP6.W0 B7 /r)"},
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.128.66.MAP6.W0 B7 /r)"},
// AVX512-FP16 scalar fused multiply-add: the scalar mirrors of the
// packed FMA group, one half-precision value per lane, the 132, 213 and
// 231 pairings of the multiply-add and the multiply-subtract. The
// scalar opcodes sit one above the packed ones, 99 the add and 9B the
// subtract, in the LIG shape the rest of the scalar core carries. The
// register forms take the embedded rounding, the memory shape of the
// second source reads its m16 plain.
{Name: "VFMADD132SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x99, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD132SH (EVEX.NDS.LIG.66.MAP6.W0 99 /r)"},
{Name: "VFMADD213SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying a factor and the added term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA9, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD213SH (EVEX.NDS.LIG.66.MAP6.W0 A9 /r)"},
{Name: "VFMADD231SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying the added term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB9, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADD231SH (EVEX.NDS.LIG.66.MAP6.W0 B9 /r)"},
{Name: "VFMSUB132SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying a factor",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x9B, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB132SH (EVEX.NDS.LIG.66.MAP6.W0 9B /r)"},
{Name: "VFMSUB213SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xAB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB213SH (EVEX.NDS.LIG.66.MAP6.W0 AB /r)"},
{Name: "VFMSUB231SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying the subtracted term",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMSUB231SH (EVEX.NDS.LIG.66.MAP6.W0 BB /r)"},
// AVX512-FP16 complex multiply: the packed pair over the FP16 complex
// pairs, F3 prefixing the multiply by the conjugate of the second
// source and F2 the multiply of the conjugate of the first source, and
// their scalar mirrors at D7. A complex value needs both of its halves
// beside one another, so the memory shape reads its full-width vector
// plain and takes no broadcast, unlike the packed FMA; the 512-bit
// register forms take the embedded rounding. The scalar forms read
// their m16 plain and take the rounding beside it.
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x40, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.512.F3.MAP6.W0 D6 /r)"},
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x20, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.256.F3.MAP6.W0 D6 /r)"},
{Name: "VFMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMULCPH (EVEX.NDS.128.F3.MAP6.W0 D6 /r)"},
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x40, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.512.F2.MAP6.W0 D6 /r)"},
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x20, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.256.F2.MAP6.W0 D6 /r)"},
{Name: "VFCMULCPH", Summary: "Multiply packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCPH (EVEX.NDS.128.F2.MAP6.W0 D6 /r)"},
{Name: "VFMULCSH", Summary: "Multiply scalar complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMULCSH (EVEX.NDS.LIG.F3.MAP6.W0 D7 /r)"},
{Name: "VFCMULCSH", Summary: "Multiply scalar complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0xD7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMULCSH (EVEX.NDS.LIG.F2.MAP6.W0 D7 /r)"},
// AVX512-FP16 complex fused multiply-add: the packed pair over the FP16
// complex pairs, F3 prefixing the accumulate with the second source
// conjugated and F2 the accumulate with the first source conjugated,
// and their scalar mirrors at 57. The word sums three complex values,
// so the memory shape reads its full-width vector plain and takes no
// broadcast, like the complex multiply above; the 512-bit register
// forms take the embedded rounding and the scalar forms read their m16
// plain with the rounding beside them.
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.512.F3.MAP6.W0 56 /r)"},
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.256.F3.MAP6.W0 56 /r)"},
{Name: "VFMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDCPH (EVEX.NDS.128.F3.MAP6.W0 56 /r)"},
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x40, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.512.F2.MAP6.W0 56 /r)"},
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x20, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.256.F2.MAP6.W0 56 /r)"},
{Name: "VFCMADDCPH", Summary: "Multiply-add packed complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0x56, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMADDCPH (EVEX.NDS.128.F2.MAP6.W0 56 /r)"},
{Name: "VFMADDCSH", Summary: "Multiply-add scalar complex FP16 values, conjugating the second source",
Bytes: []byte{0x62, 0x06, 0x06, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFMADDCSH (EVEX.NDS.LIG.F3.MAP6.W0 57 /r)"},
{Name: "VFCMADDCSH", Summary: "Multiply-add scalar complex FP16 values, conjugating the first source",
Bytes: []byte{0x62, 0x06, 0x07, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VFCMADDCSH (EVEX.NDS.LIG.F2.MAP6.W0 57 /r)"},
// AVX512-FP16 minimum or maximum: VMINMAXPH, the per-lane selection
// under the imm8 control the AVX512DQ double- and single-precision pair
// carries into the half-precision set, one control byte over the lanes.
// EVEX.NDS.0F3A.W0 52 /r /ib, the control byte riding last as the
// immediate operand it is. The destinations take the write mask and the
// memory shape of the second source reads its full-width vector plain.
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.512.0F3A.W0 52 /r /ib)"},
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.256.0F3A.W0 52 /r /ib)"},
{Name: "VMINMAXPH", Summary: "Return the per-lane minimum or maximum of packed FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Mask: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINMAXPH (EVEX.NDS.128.0F3A.W0 52 /r /ib)"},
// The scalar mirror: VMINMAXSH picks between one pair of FP16 values
// under the same imm8 control, EVEX.NDS.LIG.0F3A.W0 53 /r /ib, the
// control byte riding last as the packed form's does. A scalar
// destination takes no write mask and the m16 memory source no
// broadcast, like the rest of the scalar core.
{Name: "VMINMAXSH", Summary: "Return the minimum or maximum of scalar FP16 values under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x53, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINMAXSH (EVEX.NDS.LIG.0F3A.W0 53 /r /ib)"},
}