feat(arch): add the packed FP16 and BF16 memory forms to the extension layer
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
d275dee3ae
commit
454a21f5b7
2 files changed
+137
-30
No files matched your search
+53
-30
@@ -316,13 +316,36 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
|
||||
|
||||
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
|
||||
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
|
||||
// of destination, and at 128 bits the companion stays the class itself.
|
||||
// of destination, and at 128 bits the companion stays the class itself. An
|
||||
// entry with Mem set takes the memory shape of that position too: the
|
||||
// compares and the packed square root read their source from memory, and
|
||||
// the narrow BF16 convert reads its full-width source there.
|
||||
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
destClass := class
|
||||
if in.Form == ExtFormAmdVec2Half {
|
||||
destClass = amd64HalfClass(class)
|
||||
}
|
||||
if in.Mem == 1 && ops[0].Kind == ExtMem {
|
||||
base, disp, err := in.amd64Memory(ops[0], 1)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
|
||||
}
|
||||
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
base, disp, err := in.amd64Memory(ops[1], 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -509,22 +532,22 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.128.F2.0F38.W0 72 /r)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.512.F3.0F38.W0 72 /r, YMM destination)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.256.F3.0F38.W0 72 /r, XMM destination)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"},
|
||||
|
||||
// AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask
|
||||
@@ -605,10 +628,10 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VGETEXPSH (EVEX.NDS.LIG.66.MAP6.W0 43 /r)"},
|
||||
{Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCOMISH (EVEX.LIG.MAP5.W0 2F /r)"},
|
||||
{Name: "VUCOMISH", Summary: "Unordered-compare a scalar FP16 value and set EFLAGS",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"},
|
||||
{Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
@@ -653,67 +676,67 @@ var amd64Extensions = []ExtInstr{
|
||||
// quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from
|
||||
// avx512_fp16_vl.d, on the same low registers the suite uses.
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"},
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"},
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"},
|
||||
|
||||
// AVX512-FP16 scalar, the imm8-control group: mantissa extraction,
|
||||
|
||||
@@ -333,6 +333,81 @@ var amd64GoldenRows = []amd64GoldenRow{
|
||||
[]ExtOperand{ExtXmm(29), ExtMemory(2, -128), ExtXmm(30)},
|
||||
"62651600517280", "62 65 16 87 51 72 80 vsqrtsh -0x100(%rdx),%xmm29,%xmm30 (Disp8(80); the GNU row adds {k7}{z})"},
|
||||
|
||||
// The packed memory forms: the arithmetic reads its second source, the
|
||||
// square root its source and the compares their operand from memory.
|
||||
// The disp8 rows quote the listing's plain-base rows, with the plain
|
||||
// displacement the bytes encode; the R12 rows are derived and exercise
|
||||
// the SIB byte.
|
||||
{"vaddph memory source", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||
"6265144058717f", "62 65 14 40 58 71 7f vaddph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||
{"vaddph ymm memory source", "VADDPH",
|
||||
[]ExtOperand{ExtYmm(5), ExtMemory(1, 127), ExtYmm(6)},
|
||||
"62f5542858717f", "62 f5 54 28 58 71 7f vaddph 0xfe0(%ecx),%ymm5,%ymm6 (Disp8(7f))"},
|
||||
{"vaddph xmm memory source", "VADDPH",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(1, 127), ExtXmm(6)},
|
||||
"62f5540858717f", "62 f5 54 08 58 71 7f vaddph 0x7f0(%ecx),%xmm5,%xmm6 (Disp8(7f))"},
|
||||
{"vaddph memory source over an R12 base", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(12, 0), ExtZmm(30)},
|
||||
"62451440583424", ""},
|
||||
{"vsubph memory source", "VSUBPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||
"626514405c717f", "62 65 14 40 5c 71 7f vsubph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||
{"vmulph memory source", "VMULPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||
"6265144059717f", "62 65 14 40 59 71 7f vmulph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||
{"vdivph memory source", "VDIVPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||
"626514405e717f", "62 65 14 40 5e 71 7f vdivph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||
{"vminph memory source", "VMINPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||
"626514405d717f", "62 65 14 40 5d 71 7f vminph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||
{"vmaxph memory source", "VMAXPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||
"626514405f717f", "62 65 14 40 5f 71 7f vmaxph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||
{"vsqrtph memory source", "VSQRTPH",
|
||||
[]ExtOperand{ExtMemory(1, 127), ExtZmm(30)},
|
||||
"62657c4851717f", "62 65 7c 48 51 71 7f vsqrtph 0x1fc0(%rcx),%zmm30 (Disp8(7f))"},
|
||||
{"vsqrtph ymm memory source over an R12 base", "VSQRTPH",
|
||||
[]ExtOperand{ExtMemory(12, 0), ExtYmm(6)},
|
||||
"62d57c28513424", ""},
|
||||
{"vcomish memory source", "VCOMISH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
|
||||
"62457c082f31", "62 45 7c 08 2f 31 vcomish (%r9),%xmm30"},
|
||||
{"vcomish memory source disp8", "VCOMISH",
|
||||
[]ExtOperand{ExtMemory(1, 127), ExtXmm(30)},
|
||||
"62657c082f717f", "62 65 7c 08 2f 71 7f vcomish 0xfe(%rcx),%xmm30 (Disp8(7f))"},
|
||||
{"vcomish memory source negative disp8", "VCOMISH",
|
||||
[]ExtOperand{ExtMemory(2, -128), ExtXmm(30)},
|
||||
"62657c082f7280", "62 65 7c 08 2f 72 80 vcomish -0x100(%rdx),%xmm30 (Disp8(80))"},
|
||||
{"vucomish memory source", "VUCOMISH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
|
||||
"62457c082e31", "62 45 7c 08 2e 31 vucomish (%r9),%xmm30"},
|
||||
{"vucomish memory source negative disp8", "VUCOMISH",
|
||||
[]ExtOperand{ExtMemory(2, -128), ExtXmm(30)},
|
||||
"62657c082e7280", "62 65 7c 08 2e 72 80 vucomish -0x100(%rdx),%xmm30 (Disp8(80))"},
|
||||
|
||||
// The BF16 memory forms: the dot product reads its second source and
|
||||
// the narrow convert its full-width source from memory.
|
||||
{"vdpbf16ps memory source", "VDPBF16PS",
|
||||
[]ExtOperand{ExtZmm(5), ExtMemory(1, 127), ExtZmm(6)},
|
||||
"62f2564852717f", "62 f2 56 48 52 71 7f vdpbf16ps 0x1fc0(%ecx),%zmm5,%zmm6 (Disp8(7f))"},
|
||||
{"vdpbf16ps ymm memory source", "VDPBF16PS",
|
||||
[]ExtOperand{ExtYmm(5), ExtMemory(1, 127), ExtYmm(6)},
|
||||
"62f2562852717f", "62 f2 56 28 52 71 7f vdpbf16ps 0xfe0(%ecx),%ymm5,%ymm6 (Disp8(7f))"},
|
||||
{"vdpbf16ps xmm memory source", "VDPBF16PS",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(1, 127), ExtXmm(6)},
|
||||
"62f2560852717f", "62 f2 56 08 52 71 7f vdpbf16ps 0x7f0(%ecx),%xmm5,%xmm6 (Disp8(7f))"},
|
||||
{"vcvtneps2bf16 memory source", "VCVTNEPS2BF16",
|
||||
[]ExtOperand{ExtMemory(1, 127), ExtYmm(6)},
|
||||
"62f27e4872717f", "62 f2 7e 48 72 71 7f vcvtneps2bf16 0x1fc0(%ecx),%ymm6 (Disp8(7f))"},
|
||||
{"vcvtneps2bf16 ymm memory source", "VCVTNEPS2BF16",
|
||||
[]ExtOperand{ExtMemory(1, 127), ExtXmm(6)},
|
||||
"62f27e2872717f", "62 f2 7e 28 72 71 7f vcvtneps2bf16y 0xfe0(%ecx),%xmm6 (Disp8(7f))"},
|
||||
// The 128-bit convert's memory row shares the 256-bit row's operand
|
||||
// list, both destinations being XMM, so the resolver cannot tell them
|
||||
// apart and the register row above pins the 128-bit template alone.
|
||||
|
||||
// High registers in a 512-bit form exercise the EVEX extension bits:
|
||||
// with both sources above 15 the B bar and X bar bits clear, while the
|
||||
// destination zmm23 keeps R bar set in byte one (derived from the
|
||||
@@ -559,6 +634,15 @@ func TestAmd64ExtRejects(t *testing.T) {
|
||||
{"memory as the control byte", "VGETMANTSH",
|
||||
[]ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
|
||||
"wants an immediate control byte"},
|
||||
{"memory in the compare's destination", "VCOMISH",
|
||||
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0)},
|
||||
"wants an XMM register"},
|
||||
{"memory as the packed square root's destination", "VSQRTPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0)},
|
||||
"wants a ZMM register"},
|
||||
{"memory as the narrow convert's destination", "VCVTNEPS2BF16",
|
||||
[]ExtOperand{ExtZmm(5), ExtMemory(9, 0)},
|
||||
"wants a YMM register"},
|
||||
} {
|
||||
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
|
||||
_, err := in.Encode(tt.ops)
|
||||
|
||||
Reference in new issue
Block a user