2026-10-06 23:43:23 +02:00
|
|
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
|
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
|
|
|
|
|
|
// This file carries the amd64 side of the extended-instruction layer:
|
|
|
|
|
// instructions the Go toolchain does not know at all, described as data and
|
|
|
|
|
// validated against golden vectors from the Intel SDM rather than against the
|
|
|
|
|
// toolchain. It sits beside the generated table, never inside it:
|
|
|
|
|
// arch/amd64_gen.go stays untouched, and asm.Encodable keeps answering false
|
|
|
|
|
// for every mnemonic here, so the layer stays out of the main encoders.
|
|
|
|
|
//
|
2026-10-07 00:33:30 +02:00
|
|
|
// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the
|
|
|
|
|
// latter's scalar core with its imm8-control group and its packed 512-bit
|
|
|
|
|
// arithmetic, in their EVEX register forms. The encodings are transcribed
|
2026-10-06 23:51:58 +02:00
|
|
|
// from the SDM instruction entries and cross-checked against binutils-gdb's
|
|
|
|
|
// assembler testsuite; the golden vectors in amd64_ext_test.go pin the
|
|
|
|
|
// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey,
|
|
|
|
|
// no longer belong here: the Go toolchain's assembler knows them today, they
|
|
|
|
|
// live in the generated table and the EVEX encoder, and a mnemonic the
|
|
|
|
|
// toolchain has is not an extension.
|
2026-10-06 23:43:23 +02:00
|
|
|
//
|
2026-10-07 01:19:06 +02:00
|
|
|
// The forms encode the unmasked shapes: register forms throughout, and the
|
|
|
|
|
// scalar FP16 memory forms beside them, base-relative operands with the
|
|
|
|
|
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand. A
|
|
|
|
|
// scaled index, write masking ({k1}{z}) and embedded rounding still arrive
|
|
|
|
|
// with a later slice.
|
2026-10-06 23:43:23 +02:00
|
|
|
|
|
|
|
|
package arch
|
|
|
|
|
|
|
|
|
|
import "fmt"
|
|
|
|
|
|
|
|
|
|
// The features the amd64 layer covers.
|
|
|
|
|
const (
|
|
|
|
|
ExtFeatureBF16 ExtFeature = "avx512bf16"
|
|
|
|
|
ExtFeatureVP2INTERSECT ExtFeature = "avx512vp2intersect"
|
|
|
|
|
ExtFeatureFP16 ExtFeature = "avx512fp16"
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
// ExtXmm, ExtYmm and ExtZmm build vector operands of the three EVEX register
|
|
|
|
|
// widths, VADDPS ZMM1, ZMM2, ZMM3 style. The register number runs 0..31,
|
|
|
|
|
// XMM16 and above included: EVEX carries five register bits in every
|
|
|
|
|
// position, and the golden vectors exercise the high registers on purpose.
|
|
|
|
|
func ExtXmm(reg int) ExtOperand { return ExtOperand{Kind: ExtXMM, Reg: reg} }
|
|
|
|
|
func ExtYmm(reg int) ExtOperand { return ExtOperand{Kind: ExtYMM, Reg: reg} }
|
|
|
|
|
func ExtZmm(reg int) ExtOperand { return ExtOperand{Kind: ExtZMM, Reg: reg} }
|
|
|
|
|
|
|
|
|
|
// ExtMask builds an opmask operand, VP2INTERSECTD K1, ZMM2, ZMM3 style. The
|
|
|
|
|
// register number runs 0..7.
|
|
|
|
|
func ExtMask(reg int) ExtOperand { return ExtOperand{Kind: ExtKReg, Reg: reg} }
|
|
|
|
|
|
|
|
|
|
// ExtGpr32 and ExtGpr64 build general-register operands, VCVTSI2SH XMM1,
|
|
|
|
|
// XMM2, EAX style. The register number runs 0..15.
|
|
|
|
|
func ExtGpr32(reg int) ExtOperand { return ExtOperand{Kind: ExtR32, Reg: reg} }
|
|
|
|
|
func ExtGpr64(reg int) ExtOperand { return ExtOperand{Kind: ExtR64, Reg: reg} }
|
|
|
|
|
|
|
|
|
|
// amd64LengthClass reads the vector length the template encodes out of the
|
|
|
|
|
// L'L field of the EVEX byte three and names the register class every vector
|
|
|
|
|
// operand of that entry must carry.
|
|
|
|
|
func amd64LengthClass(b []byte) ExtOperandKind {
|
|
|
|
|
switch (b[3] >> 5) & 3 {
|
|
|
|
|
case 0:
|
|
|
|
|
return ExtXMM
|
|
|
|
|
case 1:
|
|
|
|
|
return ExtYMM
|
|
|
|
|
default:
|
|
|
|
|
return ExtZMM
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// amd64HalfClass names the half-width companion of a vector class, the
|
|
|
|
|
// destination class of the narrow conversions. At 128 bits the companion is
|
|
|
|
|
// the class itself, which is what the manual gives for the narrowest form.
|
|
|
|
|
func amd64HalfClass(k ExtOperandKind) ExtOperandKind {
|
|
|
|
|
switch k {
|
|
|
|
|
case ExtZMM:
|
|
|
|
|
return ExtYMM
|
|
|
|
|
case ExtYMM:
|
|
|
|
|
return ExtXMM
|
|
|
|
|
default:
|
|
|
|
|
return ExtXMM
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// amd64Encode returns the template with the register-derived bits filled in:
|
|
|
|
|
// dest and rm are register numbers for the ModR/M reg and r/m fields, vvvv is
|
|
|
|
|
// the third-operand register or -1 when the form leaves it unused. The EVEX
|
|
|
|
|
// plumbing follows the encoder in asm: reg[3] rides R bar and reg[4] R prime
|
|
|
|
|
// bar, rm[3] rides B bar, and in a register form rm[4] rides X bar, while
|
|
|
|
|
// vvvv[4] rides V prime bar in byte three.
|
|
|
|
|
func amd64Encode(b []byte, dest, vvvv, rm int) []byte {
|
|
|
|
|
out := make([]byte, len(b))
|
|
|
|
|
copy(out, b)
|
|
|
|
|
rBar, rPrimeBar := 1, 1
|
|
|
|
|
if dest&8 != 0 {
|
|
|
|
|
rBar = 0
|
|
|
|
|
}
|
|
|
|
|
if dest&16 != 0 {
|
|
|
|
|
rPrimeBar = 0
|
|
|
|
|
}
|
|
|
|
|
xBar, bBar := 1, 1
|
|
|
|
|
if rm&8 != 0 {
|
|
|
|
|
bBar = 0
|
|
|
|
|
}
|
|
|
|
|
if rm&16 != 0 {
|
|
|
|
|
xBar = 0
|
|
|
|
|
}
|
|
|
|
|
out[1] |= byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4)
|
|
|
|
|
vBar, vPrimeBar := 15, 1
|
|
|
|
|
if vvvv >= 0 {
|
|
|
|
|
vBar = 15 - (vvvv & 15)
|
|
|
|
|
if vvvv&16 != 0 {
|
|
|
|
|
vPrimeBar = 0
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
out[2] |= byte(vBar << 3)
|
|
|
|
|
out[3] |= byte(vPrimeBar << 3)
|
|
|
|
|
out[5] |= byte((dest&7)<<3 | rm&7)
|
|
|
|
|
return out
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-07 01:12:38 +02:00
|
|
|
// amd64Memory validates a memory operand of an amd64 entry: no arrangement
|
|
|
|
|
// and no qualifier, a base general register inside 0-15, a signed 32-bit
|
|
|
|
|
// displacement and no shift. The base number rides the operand's Reg and
|
|
|
|
|
// the displacement its Imm.
|
|
|
|
|
func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, err error) {
|
|
|
|
|
if op.Kind != ExtMem {
|
|
|
|
|
return 0, 0, fmt.Errorf("%s: operand %d wants a memory operand, got %s", in.Name, pos, op.Kind)
|
|
|
|
|
}
|
|
|
|
|
if op.Arr != ExtArrNone {
|
|
|
|
|
return 0, 0, fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
|
|
|
|
|
}
|
|
|
|
|
if op.Qual != ExtQualNone {
|
|
|
|
|
return 0, 0, fmt.Errorf("%s: operand %d carries a predicate qualifier, the amd64 layer takes none", in.Name, pos)
|
|
|
|
|
}
|
|
|
|
|
if op.HasShift {
|
|
|
|
|
return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos)
|
|
|
|
|
}
|
|
|
|
|
if op.Reg < 0 || op.Reg > 15 {
|
|
|
|
|
return 0, 0, fmt.Errorf("%s: operand %d names base register %d, outside 0-15", in.Name, pos, op.Reg)
|
|
|
|
|
}
|
|
|
|
|
if op.Imm < -1<<31 || op.Imm >= 1<<31 {
|
|
|
|
|
return 0, 0, fmt.Errorf("%s: operand %d carries displacement %d, outside the signed 32-bit range", in.Name, pos, op.Imm)
|
|
|
|
|
}
|
|
|
|
|
return op.Reg, op.Imm, nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// amd64EncodeMemory returns the register-form template with a base-relative
|
|
|
|
|
// memory operand filled in: dest and vvvv keep their register meanings, the
|
|
|
|
|
// ModR/M r/m field carries the base, and the high base bit rides EVEX.B as
|
|
|
|
|
// amd64Encode lays it. The ModR/M mod bits and the trailing SIB and
|
|
|
|
|
// displacement bytes follow the canonical choices the GNU assembler makes
|
|
|
|
|
// for the plain, unscaled SDM displacements: no displacement bytes at
|
|
|
|
|
// displacement zero, a disp8 when the value fits a signed byte and a disp32
|
|
|
|
|
// otherwise, the SIB byte 0x24 when the base is RSP or R12, whose r/m
|
|
|
|
|
// encoding 100 demands it, and a forced displacement on RBP and R13, whose
|
|
|
|
|
// mod-00 r/m encoding 101 means RIP-relative. The operand must have passed
|
|
|
|
|
// amd64Memory first.
|
|
|
|
|
func amd64EncodeMemory(b []byte, dest, vvvv, base int, disp int64) []byte {
|
|
|
|
|
out := amd64Encode(b, dest, vvvv, base)
|
|
|
|
|
rm := base & 7
|
|
|
|
|
var tail []byte
|
|
|
|
|
mod := byte(0)
|
|
|
|
|
switch {
|
|
|
|
|
case rm == 5 || disp != 0:
|
|
|
|
|
// RBP and R13 cannot drop the displacement: mod 00 with r/m 101
|
|
|
|
|
// addresses RIP-relative, not through the base.
|
|
|
|
|
if disp >= -128 && disp <= 127 {
|
|
|
|
|
mod = 1
|
|
|
|
|
tail = []byte{byte(disp)}
|
|
|
|
|
} else {
|
|
|
|
|
mod = 2
|
|
|
|
|
tail = []byte{byte(disp), byte(disp >> 8), byte(disp >> 16), byte(disp >> 24)}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if rm == 4 {
|
|
|
|
|
// RSP and R12 need the SIB byte: no index, base 100.
|
|
|
|
|
tail = append([]byte{0x24}, tail...)
|
|
|
|
|
}
|
|
|
|
|
out[5] = out[5]&0x3f | mod<<6
|
|
|
|
|
return append(out, tail...)
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-06 23:43:23 +02:00
|
|
|
// amd64PlainReg checks the invariants every amd64 register operand carries:
|
|
|
|
|
// no arm64 arrangement, no predicate qualifier, and a register number inside
|
|
|
|
|
// the class the instruction encodes.
|
|
|
|
|
func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
|
|
|
|
|
if op.Arr != ExtArrNone {
|
|
|
|
|
return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
|
|
|
|
|
}
|
|
|
|
|
if op.Qual != ExtQualNone {
|
|
|
|
|
return fmt.Errorf("%s: operand %d carries a predicate qualifier, the amd64 layer takes none", in.Name, pos)
|
|
|
|
|
}
|
|
|
|
|
if op.Reg < 0 || op.Reg > max {
|
|
|
|
|
return fmt.Errorf("%s: operand %d is register %d, outside 0-%d", in.Name, pos, op.Reg, max)
|
|
|
|
|
}
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// amd64Vector checks one vector operand against the class the entry encodes.
|
|
|
|
|
func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error {
|
|
|
|
|
if op.Kind != class {
|
2026-10-07 01:19:06 +02:00
|
|
|
article := "a"
|
|
|
|
|
if class == ExtXMM {
|
|
|
|
|
article = "an"
|
|
|
|
|
}
|
|
|
|
|
return fmt.Errorf("%s: operand %d wants %s %s, got %s", in.Name, pos, article, class, op.Kind)
|
2026-10-06 23:43:23 +02:00
|
|
|
}
|
|
|
|
|
return in.amd64PlainReg(op, 31, pos)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// amd64Gpr checks the general-register operand against the width the entry
|
|
|
|
|
// encodes: the W bit picks 32-bit or 64-bit, unless the entry ignores W, and
|
|
|
|
|
// the general registers run 0..15.
|
|
|
|
|
func (in ExtInstr) amd64Gpr(op ExtOperand, pos int) error {
|
|
|
|
|
want := ExtR32
|
|
|
|
|
if in.Bytes[2]&0x80 != 0 {
|
|
|
|
|
want = ExtR64
|
|
|
|
|
}
|
|
|
|
|
if in.Wig {
|
|
|
|
|
if op.Kind != ExtR32 && op.Kind != ExtR64 {
|
|
|
|
|
return fmt.Errorf("%s: operand %d wants a 32-bit or 64-bit general register, got %s", in.Name, pos, op.Kind)
|
|
|
|
|
}
|
|
|
|
|
} else if op.Kind != want {
|
|
|
|
|
return fmt.Errorf("%s: operand %d wants a %s, got %s", in.Name, pos, want, op.Kind)
|
|
|
|
|
}
|
|
|
|
|
return in.amd64PlainReg(op, 15, pos)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// encodeAmd64 encodes the amd64 forms: it validates the operand list against
|
|
|
|
|
// the class the template encodes and fills the register bits. An operand the
|
|
|
|
|
// form cannot carry is an error, never a silent mis-encoding.
|
|
|
|
|
func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
switch in.Form {
|
|
|
|
|
case ExtFormAmdVec3:
|
|
|
|
|
return in.encodeAmdVec3(ops)
|
|
|
|
|
case ExtFormAmdVec2, ExtFormAmdVec2Half:
|
|
|
|
|
return in.encodeAmdVec2(ops)
|
|
|
|
|
case ExtFormAmdMask2:
|
|
|
|
|
return in.encodeAmdMask2(ops)
|
|
|
|
|
case ExtFormAmdVecGprVec:
|
|
|
|
|
return in.encodeAmdVecGprVec(ops)
|
|
|
|
|
case ExtFormAmdGprVec, ExtFormAmdVecGpr:
|
|
|
|
|
return in.encodeAmdGprPair(ops)
|
2026-10-07 01:19:06 +02:00
|
|
|
case ExtFormAmdMemVec:
|
|
|
|
|
return in.encodeAmdMemVec(ops)
|
|
|
|
|
case ExtFormAmdVecMem:
|
|
|
|
|
return in.encodeAmdVecMem(ops)
|
2026-10-07 00:33:30 +02:00
|
|
|
case ExtFormAmdVec3Imm:
|
|
|
|
|
return in.encodeAmdVec3Imm(ops)
|
|
|
|
|
case ExtFormAmdMask2Imm:
|
|
|
|
|
return in.encodeAmdMask2Imm(ops)
|
2026-10-06 23:43:23 +02:00
|
|
|
default:
|
|
|
|
|
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// encodeAmdVec3 fills the non-destructive three-vector form: src1, src2,
|
2026-10-07 01:19:06 +02:00
|
|
|
// dest, all under one register class. An entry with Mem set takes the
|
|
|
|
|
// memory shape of that position too: the second source of the scalar
|
|
|
|
|
// arithmetic, spelled xmm3/m16 in the manual, may be a base-relative
|
|
|
|
|
// operand, which rides the r/m field with its displacement bytes after the
|
|
|
|
|
// opcode.
|
2026-10-06 23:43:23 +02:00
|
|
|
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
2026-10-07 01:19:06 +02:00
|
|
|
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
|
|
|
|
base, disp, err := in.amd64Memory(ops[1], 2)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
|
|
|
|
|
}
|
|
|
|
|
for i, op := range ops[1:] {
|
|
|
|
|
if err := in.amd64Vector(op, class, i+2); err != nil {
|
2026-10-06 23:43:23 +02:00
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-07 01:19:06 +02:00
|
|
|
// encodeAmdMemVec fills the memory-load form: mem, dest. VMOVSH X30,
|
|
|
|
|
// 4660(R8) shape, the manual's xmm1, m16 lines beside the register form.
|
|
|
|
|
// The form reads one value from memory, so the third register slot stays
|
|
|
|
|
// unused, which the encoding spells as vvvv 1111.
|
|
|
|
|
func (in ExtInstr) encodeAmdMemVec(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
|
|
|
|
base, disp, err := in.amd64Memory(ops[0], 1)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// encodeAmdVecMem fills the memory-store form: src, mem. VMOVSH 4660(R9),
|
|
|
|
|
// X29 shape, the manual's m16, xmm1 lines. The register source sits in the
|
|
|
|
|
// ModR/M reg field and the memory destination in r/m, and vvvv stays
|
|
|
|
|
// unused.
|
|
|
|
|
func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
|
|
|
|
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
base, disp, err := in.amd64Memory(ops[1], 2)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-06 23:43:23 +02:00
|
|
|
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
|
|
|
|
|
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
|
2026-10-07 01:28:07 +02:00
|
|
|
// of destination, and at 128 bits the companion stays the class itself. An
|
|
|
|
|
// entry with Mem set takes the memory shape of that position too: the
|
|
|
|
|
// compares and the packed square root read their source from memory, and
|
|
|
|
|
// the narrow BF16 convert reads its full-width source there.
|
2026-10-06 23:43:23 +02:00
|
|
|
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
|
|
|
|
destClass := class
|
|
|
|
|
if in.Form == ExtFormAmdVec2Half {
|
|
|
|
|
destClass = amd64HalfClass(class)
|
|
|
|
|
}
|
2026-10-07 01:28:07 +02:00
|
|
|
if in.Mem == 1 && ops[0].Kind == ExtMem {
|
|
|
|
|
base, disp, err := in.amd64Memory(ops[0], 1)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
|
|
|
|
|
}
|
|
|
|
|
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
|
|
|
|
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
base, disp, err := in.amd64Memory(ops[1], 2)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil
|
|
|
|
|
}
|
2026-10-06 23:43:23 +02:00
|
|
|
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// encodeAmdMask2 fills the mask-destination form: src1, src2, dest, where the
|
|
|
|
|
// destination is an opmask register and both sources share the class.
|
|
|
|
|
func (in ExtInstr) encodeAmdMask2(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
|
|
|
|
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if ops[2].Kind != ExtKReg {
|
|
|
|
|
return nil, fmt.Errorf("%s: operand 3 wants an opmask register, got %s", in.Name, ops[2].Kind)
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64PlainReg(ops[2], 7, 3); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-07 00:33:30 +02:00
|
|
|
// amd64Imm8 validates the leading immediate operand of an imm8-control form:
|
|
|
|
|
// an ExtImm with no shift, inside the unsigned byte range, and free of the
|
|
|
|
|
// bits the entry's control layout reserves. The reserved upper nibble of
|
|
|
|
|
// the VGETMANTSH control must encode as zero; the SDM marks every other
|
|
|
|
|
// layout here fully defined, and the VCMPSH hardware masks its predicate to
|
|
|
|
|
// five bits.
|
|
|
|
|
func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
|
|
|
|
|
if op.Kind != ExtImm {
|
|
|
|
|
return 0, fmt.Errorf("%s: operand %d wants an immediate control byte, got %s", in.Name, pos, op.Kind)
|
|
|
|
|
}
|
|
|
|
|
if op.Arr != ExtArrNone {
|
|
|
|
|
return 0, fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
|
|
|
|
|
}
|
|
|
|
|
if op.HasShift {
|
|
|
|
|
return 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 imm8 forms take none", in.Name, pos)
|
|
|
|
|
}
|
|
|
|
|
if op.Imm < 0 || op.Imm > 255 {
|
|
|
|
|
return 0, fmt.Errorf("%s: operand %d is immediate %d, outside the unsigned byte range 0-255", in.Name, pos, op.Imm)
|
|
|
|
|
}
|
|
|
|
|
if in.Imm8 == ExtImm8GetMant && op.Imm > 15 {
|
|
|
|
|
return 0, fmt.Errorf("%s: operand %d is immediate %d, the upper nibble of the mantissa control is reserved and must be zero", in.Name, pos, op.Imm)
|
|
|
|
|
}
|
|
|
|
|
return byte(op.Imm), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// encodeAmdVec3Imm fills the three-vector form with a control immediate:
|
|
|
|
|
// imm, src1, src2, dest, the order the reference listings write it in.
|
|
|
|
|
func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
|
|
|
|
imm, err := in.amd64Imm8(ops[0], 1)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
for i, op := range ops[1:] {
|
|
|
|
|
if err := in.amd64Vector(op, class, i+2); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
|
|
|
|
|
return append(out, imm), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// encodeAmdMask2Imm fills the opmask-destination form with a control
|
|
|
|
|
// immediate: imm, src1, src2, dest.
|
|
|
|
|
func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
|
|
|
|
imm, err := in.amd64Imm8(ops[0], 1)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if ops[3].Kind != ExtKReg {
|
|
|
|
|
return nil, fmt.Errorf("%s: operand 4 wants an opmask register, got %s", in.Name, ops[3].Kind)
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64PlainReg(ops[3], 7, 4); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
|
|
|
|
|
return append(out, imm), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ExtFP16RoundingModes names the two-bit rounding mode the round control of
|
|
|
|
|
// VRNDSCALESH and VREDUCESH carries, indexed by imm8[1:0], the SDM's RC
|
|
|
|
|
// field encoding.
|
|
|
|
|
var ExtFP16RoundingModes = [4]string{
|
|
|
|
|
"round to nearest (even)",
|
|
|
|
|
"round down (toward -infinity)",
|
|
|
|
|
"round up (toward +infinity)",
|
|
|
|
|
"round toward zero (truncate)",
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ExtFP16GetMantSigns names the sign control imm8[3:2] of the VGETMANTSH
|
|
|
|
|
// immediate, indexed by the field: the source's own sign, a forced positive,
|
|
|
|
|
// and the two encodings that yield the indefinite NaN on a negative source.
|
|
|
|
|
var ExtFP16GetMantSigns = [4]string{
|
|
|
|
|
"the sign of the source",
|
|
|
|
|
"positive",
|
|
|
|
|
"the indefinite NaN when the source is negative",
|
|
|
|
|
"the indefinite NaN when the source is negative",
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ExtFP16CmpPredicates names the 32 comparison predicates the VCMPSH
|
|
|
|
|
// immediate carries in imm8[4:0], in encoding order. The SDM's own
|
|
|
|
|
// spellings are the fixed vocabulary of the predicate suffixes.
|
|
|
|
|
var ExtFP16CmpPredicates = [32]string{
|
|
|
|
|
"EQ_OQ", "LT_OS", "LE_OS", "UNORD_Q", "NEQ_UQ", "NLT_US", "NLE_US", "ORD_Q",
|
|
|
|
|
"EQ_UQ", "NGE_US", "NGT_US", "FALSE_OQ", "NEQ_OQ", "GE_OS", "GT_OS", "TRUE_UQ",
|
|
|
|
|
"EQ_OS", "LT_OQ", "LE_OQ", "UNORD_S", "NEQ_US", "NLT_UQ", "NLE_UQ", "ORD_S",
|
|
|
|
|
"EQ_US", "NGE_UQ", "NGT_UQ", "FALSE_OS", "NEQ_OS", "GE_OQ", "GT_OQ", "TRUE_US",
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-06 23:43:23 +02:00
|
|
|
// encodeAmdVecGprVec fills the conversion form with a general-register
|
|
|
|
|
// source: src1, gpr, dest. VCVTSI2SH XMM1, XMM2, EAX style.
|
|
|
|
|
func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
|
|
|
|
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Gpr(ops[1], 2); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// encodeAmdGprPair fills the two-operand general-register forms: gpr, vec
|
|
|
|
|
// (the move into a vector register and the integer conversions) and vec, gpr
|
|
|
|
|
// (the move out of one). In both orders the second operand is the
|
|
|
|
|
// destination in the reg field and the first the r/m source; the vector
|
|
|
|
|
// changes position with the form.
|
|
|
|
|
func (in ExtInstr) encodeAmdGprPair(ops []ExtOperand) ([]byte, error) {
|
|
|
|
|
class := amd64LengthClass(in.Bytes)
|
|
|
|
|
vecPos := 1
|
|
|
|
|
if in.Form == ExtFormAmdVecGpr {
|
|
|
|
|
vecPos = 0
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Vector(ops[vecPos], class, vecPos+1); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if err := in.amd64Gpr(ops[1-vecPos], 2-vecPos); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// --- the amd64 AVX512-BF16 and VP2INTERSECT table -----------------------------
|
|
|
|
|
|
|
|
|
|
// amd64Extensions is the extended-instruction layer of amd64. The encodings
|
|
|
|
|
// are transcribed from the Intel SDM instruction entries and cross-checked
|
|
|
|
|
// against binutils-gdb's assembler testsuite (gas/testsuite/gas/i386/
|
|
|
|
|
// avx512_bf16.d, avx512_bf16_vl.d and x86-64-vp2intersect.d), whose register
|
|
|
|
|
// forms the golden vectors in amd64_ext_test.go quote byte for byte. Each
|
|
|
|
|
// template carries the fixed bits of one encoding with every register-derived
|
|
|
|
|
// bit zero: the map selection in byte one, the W bit, the mandatory prefix
|
|
|
|
|
// and the reserved one-bit in byte two, the vector length in byte three, and
|
|
|
|
|
// the ModR/M mod bits.
|
|
|
|
|
var amd64Extensions = []ExtInstr{
|
|
|
|
|
// AVX512-BF16: the two-way packed single to BF16 conversion and the
|
|
|
|
|
// dot product accumulate. The prefixes differ inside the family, the
|
|
|
|
|
// three-register convert carries F2 while the narrow convert and the dot
|
|
|
|
|
// product carry F3, which the golden vectors pin byte for byte.
|
|
|
|
|
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.512.F2.0F38.W0 72 /r)"},
|
|
|
|
|
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.256.F2.0F38.W0 72 /r)"},
|
|
|
|
|
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.128.F2.0F38.W0 72 /r)"},
|
|
|
|
|
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
2026-10-06 23:43:23 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.512.F3.0F38.W0 72 /r, YMM destination)"},
|
|
|
|
|
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
2026-10-06 23:43:23 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.256.F3.0F38.W0 72 /r, XMM destination)"},
|
|
|
|
|
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
2026-10-06 23:43:23 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"},
|
|
|
|
|
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
2026-10-06 23:43:23 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"},
|
|
|
|
|
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
2026-10-06 23:43:23 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"},
|
|
|
|
|
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
2026-10-06 23:43:23 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"},
|
|
|
|
|
|
|
|
|
|
// AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask
|
|
|
|
|
// destination and two vector sources, EVEX.NDS.66.0F38. The instruction
|
|
|
|
|
// takes no write mask of its own.
|
|
|
|
|
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.512.F2.0F38.W0 68 /r)"},
|
|
|
|
|
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.256.F2.0F38.W0 68 /r)"},
|
|
|
|
|
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W0 68 /r)"},
|
|
|
|
|
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x87, 0x40, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.512.F2.0F38.W1 68 /r)"},
|
|
|
|
|
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x87, 0x20, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.256.F2.0F38.W1 68 /r)"},
|
|
|
|
|
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
|
|
|
|
|
Bytes: []byte{0x62, 0x02, 0x87, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W1 68 /r)"},
|
2026-10-06 23:51:58 +02:00
|
|
|
|
|
|
|
|
// AVX512-FP16, the scalar core: move, arithmetic, compare and
|
|
|
|
|
// conversion on one half-precision value in the low XMM lane, the
|
|
|
|
|
// register forms of the manual's scalar entries. The family lives in
|
|
|
|
|
// the EVEX maps five and six the toolchain has never emitted, with the
|
|
|
|
|
// mandatory prefixes the manual gives each entry; the golden vectors
|
|
|
|
|
// pin every prefix byte for byte. LIG encodes as L'L = 00, the XMM
|
|
|
|
|
// class alone.
|
|
|
|
|
{Name: "VMOVSH", Summary: "Move a scalar FP16 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.NDS.LIG.F3.MAP5.W0 10 /r)"},
|
2026-10-07 01:19:06 +02:00
|
|
|
{Name: "VMOVSH", Summary: "Move a scalar FP16 value from memory into an XMM register",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdMemVec, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.LIG.F3.MAP5.W0 10 /r, m16 source)"},
|
|
|
|
|
{Name: "VMOVSH", Summary: "Move a scalar FP16 value from an XMM register to memory",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x11, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.LIG.F3.MAP5.W0 11 /r, m16 destination)"},
|
2026-10-06 23:51:58 +02:00
|
|
|
{Name: "VMOVW", Summary: "Move a word between a general register and an XMM register",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdGprVec, Feature: ExtFeatureFP16, Wig: true,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r)"},
|
|
|
|
|
{Name: "VMOVW", Summary: "Move a word between an XMM register and a general register",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, Wig: true,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r)"},
|
2026-10-07 01:19:06 +02:00
|
|
|
{Name: "VMOVW", Summary: "Move a word from memory into an XMM register",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdMemVec, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r, m16 source)"},
|
|
|
|
|
{Name: "VMOVW", Summary: "Move a word from an XMM register to memory",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r, m16 destination)"},
|
2026-10-06 23:51:58 +02:00
|
|
|
{Name: "VADDSH", Summary: "Add scalar FP16 values",
|
2026-10-07 01:19:06 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VADDSH (EVEX.NDS.LIG.F3.MAP5.W0 58 /r)"},
|
|
|
|
|
{Name: "VSUBSH", Summary: "Subtract scalar FP16 values",
|
2026-10-07 01:19:06 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VSUBSH (EVEX.NDS.LIG.F3.MAP5.W0 5C /r)"},
|
|
|
|
|
{Name: "VMULSH", Summary: "Multiply scalar FP16 values",
|
2026-10-07 01:19:06 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMULSH (EVEX.NDS.LIG.F3.MAP5.W0 59 /r)"},
|
|
|
|
|
{Name: "VDIVSH", Summary: "Divide scalar FP16 values",
|
2026-10-07 01:19:06 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VDIVSH (EVEX.NDS.LIG.F3.MAP5.W0 5E /r)"},
|
|
|
|
|
{Name: "VMINSH", Summary: "Return the minimum of scalar FP16 values",
|
2026-10-07 01:19:06 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMINSH (EVEX.NDS.LIG.F3.MAP5.W0 5D /r)"},
|
|
|
|
|
{Name: "VMAXSH", Summary: "Return the maximum of scalar FP16 values",
|
2026-10-07 01:19:06 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMAXSH (EVEX.NDS.LIG.F3.MAP5.W0 5F /r)"},
|
|
|
|
|
{Name: "VSQRTSH", Summary: "Compute the square root of a scalar FP16 value",
|
2026-10-07 01:19:06 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"},
|
2026-10-07 00:05:48 +02:00
|
|
|
{Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others",
|
|
|
|
|
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VSCALEFSH (EVEX.NDS.LIG.66.MAP6.W0 2D /r)"},
|
|
|
|
|
{Name: "VGETEXPSH", Summary: "Convert the exponent of a scalar FP16 value to an FP16 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VGETEXPSH (EVEX.NDS.LIG.66.MAP6.W0 43 /r)"},
|
2026-10-06 23:51:58 +02:00
|
|
|
{Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VCOMISH (EVEX.LIG.MAP5.W0 2F /r)"},
|
|
|
|
|
{Name: "VUCOMISH", Summary: "Unordered-compare a scalar FP16 value and set EFLAGS",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
2026-10-06 23:51:58 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"},
|
|
|
|
|
{Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"},
|
|
|
|
|
{Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSH2SS (EVEX.NDS.LIG.MAP6.W0 13 /r)"},
|
|
|
|
|
{Name: "VCVTSH2SD", Summary: "Convert a low FP16 value to an FP64 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"},
|
|
|
|
|
{Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"},
|
|
|
|
|
{Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 2A /r)"},
|
|
|
|
|
{Name: "VCVTSI2SH", Summary: "Convert one signed 64-bit integer to one FP16 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 2A /r)"},
|
|
|
|
|
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 32-bit integer to one FP16 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 7B /r)"},
|
|
|
|
|
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 64-bit integer to one FP16 value",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"},
|
|
|
|
|
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W0 2D /r)"},
|
|
|
|
|
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 64-bit integer",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W1 2D /r)"},
|
|
|
|
|
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 32-bit integer",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W0 79 /r)"},
|
|
|
|
|
{Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 64-bit integer",
|
|
|
|
|
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W1 79 /r)"},
|
2026-10-07 00:02:27 +02:00
|
|
|
|
2026-10-07 00:37:41 +02:00
|
|
|
// AVX512-FP16 packed arithmetic: the full ZMM lanes the scalar core
|
|
|
|
|
// mirrors plus the VL forms, EVEX.NDS.MAP5 with no mandatory prefix,
|
|
|
|
|
// rounding control left to MXCSR. The 512-bit register forms are
|
|
|
|
|
// quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from
|
|
|
|
|
// avx512_fp16_vl.d, on the same low registers the suite uses.
|
2026-10-07 00:02:27 +02:00
|
|
|
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:02:27 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"},
|
2026-10-07 00:37:41 +02:00
|
|
|
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"},
|
|
|
|
|
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"},
|
2026-10-07 00:02:27 +02:00
|
|
|
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:02:27 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"},
|
2026-10-07 00:37:41 +02:00
|
|
|
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"},
|
|
|
|
|
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"},
|
2026-10-07 00:02:27 +02:00
|
|
|
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:02:27 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"},
|
2026-10-07 00:37:41 +02:00
|
|
|
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"},
|
|
|
|
|
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"},
|
2026-10-07 00:02:27 +02:00
|
|
|
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:02:27 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"},
|
2026-10-07 00:37:41 +02:00
|
|
|
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"},
|
|
|
|
|
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"},
|
2026-10-07 00:02:27 +02:00
|
|
|
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:02:27 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"},
|
2026-10-07 00:37:41 +02:00
|
|
|
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"},
|
|
|
|
|
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"},
|
2026-10-07 00:02:27 +02:00
|
|
|
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:02:27 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"},
|
2026-10-07 00:37:41 +02:00
|
|
|
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"},
|
|
|
|
|
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"},
|
2026-10-07 00:02:27 +02:00
|
|
|
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
2026-10-07 00:02:27 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
|
2026-10-07 00:37:41 +02:00
|
|
|
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"},
|
|
|
|
|
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
2026-10-07 01:28:07 +02:00
|
|
|
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
2026-10-07 00:37:41 +02:00
|
|
|
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"},
|
2026-10-07 00:33:30 +02:00
|
|
|
|
|
|
|
|
// AVX512-FP16 scalar, the imm8-control group: mantissa extraction,
|
|
|
|
|
// reduction, rounding to fraction bits and the compare into an opmask.
|
|
|
|
|
// Each carries its control byte as the leading immediate operand, the
|
|
|
|
|
// order the reference listings write it in. The controls live in map
|
|
|
|
|
// 0F3A: the compare with the F3 prefix the manual gives the compare
|
|
|
|
|
// family, the other three unprefixed. The immediate layouts and their
|
|
|
|
|
// tables are ExtFP16RoundingModes, ExtFP16GetMantSigns and
|
|
|
|
|
// ExtFP16CmpPredicates above; the reserved upper nibble of the mantissa
|
|
|
|
|
// control is refused rather than encoded.
|
|
|
|
|
{Name: "VCMPSH", Summary: "Compare scalar FP16 values into an opmask under an imm8 predicate",
|
|
|
|
|
Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VCMPSH (EVEX.LLIG.F3.0F3A.W0 C2 /r /ib)"},
|
|
|
|
|
{Name: "VGETMANTSH", Summary: "Extract the normalised mantissa of a scalar FP16 value under an imm8 control",
|
|
|
|
|
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VGETMANTSH (EVEX.LLIG.NP.0F3A.W0 27 /r /ib)"},
|
|
|
|
|
{Name: "VREDUCESH", Summary: "Reduce a scalar FP16 value by imm8 fraction bits under an imm8 round control",
|
|
|
|
|
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VREDUCESH (EVEX.LLIG.NP.0F3A.W0 57 /r /ib)"},
|
|
|
|
|
{Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control",
|
|
|
|
|
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
|
|
|
|
Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"},
|
2026-10-06 23:43:23 +02:00
|
|
|
}
|