feat(arch): add the AVX512-FP16 FMA families to the extension layer
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
87493d9391
commit
8080e0acef
3 files changed
+406
-10
No files matched your search
+161
-5
@@ -684,7 +684,9 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
|
||||
// encodeAmdVec3Imm fills the three-vector form with a control immediate:
|
||||
// imm, src1, src2, dest, the order the reference listings write it in. An
|
||||
// entry with Mem set takes the memory shape of the second source, xmm3/m16
|
||||
// in the manual.
|
||||
// in the manual. The destination takes the write mask where the entry
|
||||
// declares one, the packed imm8-control forms the manual masks; the scalar
|
||||
// forms take none, and the entry's flag refuses the spelling for them.
|
||||
func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
imm, err := in.amd64Imm8(ops[0], 1)
|
||||
@@ -694,23 +696,29 @@ func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
|
||||
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
dest, mask, zeroing, _, err := in.amd64WriteMask(ops[3], 4)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 3 && ops[2].Kind == ExtMem {
|
||||
if err := in.amd64Vector(ops[3], class, 4); err != nil {
|
||||
if err := in.amd64Vector(dest, class, 4); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out, err := in.amd64MemBytes(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2], 3)
|
||||
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, ops[1].Reg, ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[3], class, 4); err != nil {
|
||||
if err := in.amd64Vector(dest, class, 4); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
|
||||
out := amd64Encode(in.Bytes, dest.Reg, ops[1].Reg, ops[2].Reg)
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
|
||||
@@ -1389,4 +1397,152 @@ var amd64Extensions = []ExtInstr{
|
||||
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.128.NP.0F3A.W0 26 /r /ib)"},
|
||||
|
||||
// AVX512-FP16 packed fused multiply-add: the twelve packed FMA
|
||||
// mnemonics of the FMA group, 132, 213 and 231 under the multiply-add,
|
||||
// multiply-subtract, add-subtract and subtract-add pairings, three
|
||||
// register widths each. EVEX.NDS.66.MAP6.W0 throughout, the opcode low
|
||||
// byte the AVX512F single-precision FMA group carries one map over: 98
|
||||
// the add, 9A the subtract, 96 the add-subtract and 97 the
|
||||
// subtract-add, the 213 and 231 forms ten and twenty above. The 512-bit
|
||||
// register forms take the embedded rounding, the VL forms none; the
|
||||
// destinations take the write mask and the memory shape of the second
|
||||
// source the {1toN} broadcast. The golden vectors are quoted from the
|
||||
// local GNU assembler, whose FP16 table matches the SDM entries row for
|
||||
// row; the add-subtract pairing adds on the odd lanes and subtracts on
|
||||
// the even ones, the subtract-add pairing the reverse.
|
||||
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.512.66.MAP6.W0 98 /r)"},
|
||||
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.256.66.MAP6.W0 98 /r)"},
|
||||
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.128.66.MAP6.W0 98 /r)"},
|
||||
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.512.66.MAP6.W0 A8 /r)"},
|
||||
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.256.66.MAP6.W0 A8 /r)"},
|
||||
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.128.66.MAP6.W0 A8 /r)"},
|
||||
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.512.66.MAP6.W0 B8 /r)"},
|
||||
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.256.66.MAP6.W0 B8 /r)"},
|
||||
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.128.66.MAP6.W0 B8 /r)"},
|
||||
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.512.66.MAP6.W0 9A /r)"},
|
||||
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.256.66.MAP6.W0 9A /r)"},
|
||||
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.128.66.MAP6.W0 9A /r)"},
|
||||
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.512.66.MAP6.W0 AA /r)"},
|
||||
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.256.66.MAP6.W0 AA /r)"},
|
||||
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.128.66.MAP6.W0 AA /r)"},
|
||||
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.512.66.MAP6.W0 BA /r)"},
|
||||
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.256.66.MAP6.W0 BA /r)"},
|
||||
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.128.66.MAP6.W0 BA /r)"},
|
||||
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.512.66.MAP6.W0 96 /r)"},
|
||||
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.256.66.MAP6.W0 96 /r)"},
|
||||
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.128.66.MAP6.W0 96 /r)"},
|
||||
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.512.66.MAP6.W0 A6 /r)"},
|
||||
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.256.66.MAP6.W0 A6 /r)"},
|
||||
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.128.66.MAP6.W0 A6 /r)"},
|
||||
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.512.66.MAP6.W0 B6 /r)"},
|
||||
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.256.66.MAP6.W0 B6 /r)"},
|
||||
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.128.66.MAP6.W0 B6 /r)"},
|
||||
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.512.66.MAP6.W0 97 /r)"},
|
||||
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.256.66.MAP6.W0 97 /r)"},
|
||||
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.128.66.MAP6.W0 97 /r)"},
|
||||
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.512.66.MAP6.W0 A7 /r)"},
|
||||
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.256.66.MAP6.W0 A7 /r)"},
|
||||
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.128.66.MAP6.W0 A7 /r)"},
|
||||
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.512.66.MAP6.W0 B7 /r)"},
|
||||
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.256.66.MAP6.W0 B7 /r)"},
|
||||
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.128.66.MAP6.W0 B7 /r)"},
|
||||
|
||||
// AVX512-FP16 scalar fused multiply-add: the scalar mirrors of the
|
||||
// packed FMA group, one half-precision value per lane, the 132, 213 and
|
||||
// 231 pairings of the multiply-add and the multiply-subtract. The
|
||||
// scalar opcodes sit one above the packed ones, 99 the add and 9B the
|
||||
// subtract, in the LIG shape the rest of the scalar core carries. The
|
||||
// register forms take the embedded rounding, the memory shape of the
|
||||
// second source reads its m16 plain.
|
||||
{Name: "VFMADD132SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x99, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD132SH (EVEX.NDS.LIG.66.MAP6.W0 99 /r)"},
|
||||
{Name: "VFMADD213SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying a factor and the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA9, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD213SH (EVEX.NDS.LIG.66.MAP6.W0 A9 /r)"},
|
||||
{Name: "VFMADD231SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB9, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD231SH (EVEX.NDS.LIG.66.MAP6.W0 B9 /r)"},
|
||||
{Name: "VFMSUB132SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x9B, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB132SH (EVEX.NDS.LIG.66.MAP6.W0 9B /r)"},
|
||||
{Name: "VFMSUB213SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xAB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB213SH (EVEX.NDS.LIG.66.MAP6.W0 AB /r)"},
|
||||
{Name: "VFMSUB231SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB231SH (EVEX.NDS.LIG.66.MAP6.W0 BB /r)"},
|
||||
}
|
||||
+225
-3
@@ -1100,6 +1100,219 @@ var amd64GoldenRows = []amd64GoldenRow{
|
||||
{"vrndscaleph k7 zeroing", "VRNDSCALEPH",
|
||||
[]ExtOperand{ExtImmediate(0x7b), ExtZmm(5), ExtWriteMasked(ExtZmm(6), 7, true)},
|
||||
"62f37ccf08f57b", "62 f3 7c cf 08 f5 7b vrndscaleph $0x7b,%zmm5,%zmm6{%k7}{z}"},
|
||||
|
||||
// The AVX512-FP16 packed fused multiply-add, twelve mnemonics by three
|
||||
// lengths: the 512-bit rows on the suite's high registers, the VL rows
|
||||
// on the low ones, every register form quoted from the local GNU
|
||||
// assembler's output (EVEX.NDS.66.MAP6.W0 throughout).
|
||||
{"vfmadd132ph", "VFMADD132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"6206154098f4", "62 06 15 40 98 f4 vfmadd132ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmadd213ph", "VFMADD213PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"62061540a8f4", "62 06 15 40 a8 f4 vfmadd213ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmadd231ph", "VFMADD231PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"62061540b8f4", "62 06 15 40 b8 f4 vfmadd231ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmsub132ph", "VFMSUB132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"620615409af4", "62 06 15 40 9a f4 vfmsub132ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmsub213ph", "VFMSUB213PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"62061540aaf4", "62 06 15 40 aa f4 vfmsub213ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmsub231ph", "VFMSUB231PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"62061540baf4", "62 06 15 40 ba f4 vfmsub231ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmaddsub132ph", "VFMADDSUB132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"6206154096f4", "62 06 15 40 96 f4 vfmaddsub132ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmaddsub213ph", "VFMADDSUB213PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"62061540a6f4", "62 06 15 40 a6 f4 vfmaddsub213ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmaddsub231ph", "VFMADDSUB231PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"62061540b6f4", "62 06 15 40 b6 f4 vfmaddsub231ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmsubadd132ph", "VFMSUBADD132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"6206154097f4", "62 06 15 40 97 f4 vfmsubadd132ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmsubadd213ph", "VFMSUBADD213PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"62061540a7f4", "62 06 15 40 a7 f4 vfmsubadd213ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmsubadd231ph", "VFMSUBADD231PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
||||
"62061540b7f4", "62 06 15 40 b7 f4 vfmsubadd231ph %zmm28,%zmm29,%zmm30"},
|
||||
{"vfmadd132ph ymm", "VFMADD132PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f6552898f4", "62 f6 55 28 98 f4 vfmadd132ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmadd132ph xmm", "VFMADD132PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f6550898f4", "62 f6 55 08 98 f4 vfmadd132ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmadd213ph ymm", "VFMADD213PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f65528a8f4", "62 f6 55 28 a8 f4 vfmadd213ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmadd213ph xmm", "VFMADD213PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f65508a8f4", "62 f6 55 08 a8 f4 vfmadd213ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmadd231ph ymm", "VFMADD231PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f65528b8f4", "62 f6 55 28 b8 f4 vfmadd231ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmadd231ph xmm", "VFMADD231PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f65508b8f4", "62 f6 55 08 b8 f4 vfmadd231ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmsub132ph ymm", "VFMSUB132PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f655289af4", "62 f6 55 28 9a f4 vfmsub132ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmsub132ph xmm", "VFMSUB132PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f655089af4", "62 f6 55 08 9a f4 vfmsub132ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmsub213ph ymm", "VFMSUB213PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f65528aaf4", "62 f6 55 28 aa f4 vfmsub213ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmsub213ph xmm", "VFMSUB213PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f65508aaf4", "62 f6 55 08 aa f4 vfmsub213ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmsub231ph ymm", "VFMSUB231PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f65528baf4", "62 f6 55 28 ba f4 vfmsub231ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmsub231ph xmm", "VFMSUB231PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f65508baf4", "62 f6 55 08 ba f4 vfmsub231ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmaddsub132ph ymm", "VFMADDSUB132PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f6552896f4", "62 f6 55 28 96 f4 vfmaddsub132ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmaddsub132ph xmm", "VFMADDSUB132PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f6550896f4", "62 f6 55 08 96 f4 vfmaddsub132ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmaddsub213ph ymm", "VFMADDSUB213PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f65528a6f4", "62 f6 55 28 a6 f4 vfmaddsub213ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmaddsub213ph xmm", "VFMADDSUB213PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f65508a6f4", "62 f6 55 08 a6 f4 vfmaddsub213ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmaddsub231ph ymm", "VFMADDSUB231PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f65528b6f4", "62 f6 55 28 b6 f4 vfmaddsub231ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmaddsub231ph xmm", "VFMADDSUB231PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f65508b6f4", "62 f6 55 08 b6 f4 vfmaddsub231ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmsubadd132ph ymm", "VFMSUBADD132PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f6552897f4", "62 f6 55 28 97 f4 vfmsubadd132ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmsubadd132ph xmm", "VFMSUBADD132PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f6550897f4", "62 f6 55 08 97 f4 vfmsubadd132ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmsubadd213ph ymm", "VFMSUBADD213PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f65528a7f4", "62 f6 55 28 a7 f4 vfmsubadd213ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmsubadd213ph xmm", "VFMSUBADD213PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f65508a7f4", "62 f6 55 08 a7 f4 vfmsubadd213ph %xmm4,%xmm5,%xmm6"},
|
||||
{"vfmsubadd231ph ymm", "VFMSUBADD231PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
||||
"62f65528b7f4", "62 f6 55 28 b7 f4 vfmsubadd231ph %ymm4,%ymm5,%ymm6"},
|
||||
{"vfmsubadd231ph xmm", "VFMSUBADD231PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
||||
"62f65508b7f4", "62 f6 55 08 b7 f4 vfmsubadd231ph %xmm4,%xmm5,%xmm6"},
|
||||
|
||||
// The scalar fused multiply-add, EVEX.NDS.LIG.66.MAP6.W0: the add
|
||||
// opcodes one above the packed ones, the subtract two above them.
|
||||
{"vfmadd132sh", "VFMADD132SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
||||
"6206150099f4", "62 06 15 00 99 f4 vfmadd132sh %xmm28,%xmm29,%xmm30"},
|
||||
{"vfmadd213sh", "VFMADD213SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
||||
"62061500a9f4", "62 06 15 00 a9 f4 vfmadd213sh %xmm28,%xmm29,%xmm30"},
|
||||
{"vfmadd231sh", "VFMADD231SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
||||
"62061500b9f4", "62 06 15 00 b9 f4 vfmadd231sh %xmm28,%xmm29,%xmm30"},
|
||||
{"vfmsub132sh", "VFMSUB132SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
||||
"620615009bf4", "62 06 15 00 9b f4 vfmsub132sh %xmm28,%xmm29,%xmm30"},
|
||||
{"vfmsub213sh", "VFMSUB213SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
||||
"62061500abf4", "62 06 15 00 ab f4 vfmsub213sh %xmm28,%xmm29,%xmm30"},
|
||||
{"vfmsub231sh", "VFMSUB231SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
||||
"62061500bbf4", "62 06 15 00 bb f4 vfmsub231sh %xmm28,%xmm29,%xmm30"},
|
||||
|
||||
// The FMA memory forms: the second source reads m512, m256, m128 or the
|
||||
// scalar m16 from memory. The disp8 rows quote the listing's compressed
|
||||
// spellings, whose source is N times the plain displacement the bytes
|
||||
// carry; the R12 row exercises the SIB byte.
|
||||
{"vfmadd132ph memory source", "VFMADD132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)},
|
||||
"6266154098717f", "62 66 15 40 98 71 7f vfmadd132ph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"},
|
||||
{"vfmadd132ph ymm memory source", "VFMADD132PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtMemory(1, 127), ExtYmm(6)},
|
||||
"62f6552898717f", "62 f6 55 28 98 71 7f vfmadd132ph 0xfe0(%ecx),%ymm5,%ymm6 (Disp8(7f))"},
|
||||
{"vfmadd132ph xmm memory source", "VFMADD132PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtMemory(1, 127), ExtXmm(6)},
|
||||
"62f6550898717f", "62 f6 55 08 98 71 7f vfmadd132ph 0x7f0(%ecx),%xmm5,%xmm6 (Disp8(7f))"},
|
||||
{"vfmadd132ph memory source", "VFMADD132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)},
|
||||
"624615409831", "62 46 15 40 98 31 vfmadd132ph (%r9),%zmm29,%zmm30"},
|
||||
{"vfmsub231ph memory source", "VFMSUB231PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)},
|
||||
"62461540ba31", "62 46 15 40 ba 31 vfmsub231ph (%r9),%zmm29,%zmm30"},
|
||||
{"vfmaddsub132ph memory source over an R12 base", "VFMADDSUB132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(12, 0), ExtZmm(30)},
|
||||
"62461540963424", ""},
|
||||
{"vfmadd132sh memory source", "VFMADD132SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
||||
"624615009931", "62 46 15 00 99 31 vfmadd132sh (%r9),%xmm29,%xmm30"},
|
||||
{"vfmadd132sh memory source disp8", "VFMADD132SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtMemory(1, 1), ExtXmm(30)},
|
||||
"62661500997101", "62 66 15 00 99 71 01 vfmadd132sh 0x2(%rcx),%xmm29,%xmm30 (Disp8(01))"},
|
||||
{"vfmsub132sh memory source", "VFMSUB132SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
||||
"624615009b31", "62 46 15 00 9b 31 vfmsub132sh (%r9),%xmm29,%xmm30"},
|
||||
|
||||
// The FMA broadcast forms: one half-precision element the hardware
|
||||
// splats across the lanes, {1to32}, {1to16} and {1to8}.
|
||||
{"vfmadd132ph broadcast source", "VFMADD132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"624615509831", "62 46 15 50 98 31 vfmadd132ph (%r9){1to32},%zmm29,%zmm30"},
|
||||
{"vfmadd132ph ymm broadcast source", "VFMADD132PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)},
|
||||
"62f655389831", "62 f6 55 38 98 31 vfmadd132ph (%ecx){1to16},%ymm5,%ymm6"},
|
||||
{"vfmadd132ph xmm broadcast source", "VFMADD132PH",
|
||||
[]ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)},
|
||||
"62f655189831", "62 f6 55 18 98 31 vfmadd132ph (%ecx){1to8},%xmm5,%xmm6"},
|
||||
{"vfmaddsub132ph broadcast source", "VFMADDSUB132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"624615509631", "62 46 15 50 96 31 vfmaddsub132ph (%r9){1to32},%zmm29,%zmm30"},
|
||||
{"vfmsubadd231ph broadcast source", "VFMSUBADD231PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"62461550b731", "62 46 15 50 b7 31 vfmsubadd231ph (%r9){1to32},%zmm29,%zmm30"},
|
||||
|
||||
// The FMA rounding rows: the 512-bit register forms take EVEX.RC, the
|
||||
// scalar register forms beside them.
|
||||
{"vfmadd132ph rz-sae", "VFMADD132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtRounded(ExtZmm(30), ExtRoundTruncate)},
|
||||
"6206157098f4", "62 06 15 70 98 f4 vfmadd132ph {rz-sae},%zmm28,%zmm29,%zmm30"},
|
||||
{"vfmadd213ph rn-sae", "VFMADD213PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)},
|
||||
"62f65518a8f4", "62 f6 55 18 a8 f4 vfmadd213ph {rn-sae},%zmm4,%zmm5,%zmm6"},
|
||||
{"vfmsub132ph ru-sae", "VFMSUB132PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundUp)},
|
||||
"62f655589af4", "62 f6 55 58 9a f4 vfmsub132ph {ru-sae},%zmm4,%zmm5,%zmm6"},
|
||||
{"vfmaddsub231ph rd-sae", "VFMADDSUB231PH",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundDown)},
|
||||
"62f65538b6f4", "62 f6 55 38 b6 f4 vfmaddsub231ph {rd-sae},%zmm4,%zmm5,%zmm6"},
|
||||
{"vfmadd132sh rn-sae", "VFMADD132SH",
|
||||
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundNearest)},
|
||||
"62f6551899f4", "62 f6 55 18 99 f4 vfmadd132sh {rn-sae},%xmm4,%xmm5,%xmm6"},
|
||||
{"vfmadd231sh rz-sae, high registers", "VFMADD231SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundTruncate)},
|
||||
"62061570b9f4", "62 06 15 70 b9 f4 vfmadd231sh {rz-sae},%xmm28,%xmm29,%xmm30"},
|
||||
|
||||
// The FMA write masks over the packed destinations.
|
||||
{"vfmadd132ph k7 zeroing", "VFMADD132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)},
|
||||
"620615c798f4", "62 06 15 c7 98 f4 vfmadd132ph %zmm28,%zmm29,%zmm30{%k7}{z}"},
|
||||
{"vfmaddsub132ph k5 merging, memory source", "VFMADDSUB132PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)},
|
||||
"624615459631", "62 46 15 45 96 31 vfmaddsub132ph (%r9),%zmm29,%zmm30{%k5}"},
|
||||
}
|
||||
|
||||
// amd64ResolveEntry finds the table entry a golden row exercises: the entry
|
||||
@@ -1375,7 +1588,7 @@ func TestAmd64ExtRejects(t *testing.T) {
|
||||
"the position takes none"},
|
||||
{"write mask on the control form's destination", "VGETMANTSH",
|
||||
[]ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtXmm(28), ExtWriteMasked(ExtXmm(30), 7, true)},
|
||||
"the position takes none"},
|
||||
"the entry's destination takes none"},
|
||||
{"write mask where the destination is memory", "VCOMISH",
|
||||
[]ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, HasMask: true, Mask: 2}},
|
||||
"the entry's destination takes none"},
|
||||
@@ -1430,6 +1643,15 @@ func TestAmd64ExtRejects(t *testing.T) {
|
||||
{"reserved upper nibble on the packed mantissa control", "VGETMANTPH",
|
||||
[]ExtOperand{ExtImmediate(0x7b), ExtZmm(5), ExtZmm(6)},
|
||||
"reserved and must be zero"},
|
||||
{"broadcast on the scalar multiply-add", "VFMADD132SH",
|
||||
[]ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)},
|
||||
"the entry's memory operand takes none"},
|
||||
{"rounding on the 256-bit multiply-add", "VFMADD132PH",
|
||||
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)},
|
||||
"the entry's destination takes none"},
|
||||
{"a register where the multiply-add reads memory", "VFMSUB231PH",
|
||||
[]ExtOperand{ExtZmm(29), ExtYmm(4), ExtZmm(30)},
|
||||
"wants a ZMM register"},
|
||||
} {
|
||||
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
|
||||
_, err := in.Encode(tt.ops)
|
||||
@@ -1574,7 +1796,7 @@ func TestAmd64ExtArchBinding(t *testing.T) {
|
||||
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
|
||||
}
|
||||
}
|
||||
if got := Extensions(AMD64); len(got) != 121 {
|
||||
t.Errorf("the amd64 layer registers %d instructions, want 121", len(got))
|
||||
if got := Extensions(AMD64); len(got) != 163 {
|
||||
t.Errorf("the amd64 layer registers %d instructions, want 163", len(got))
|
||||
}
|
||||
}
|
||||
@@ -58,6 +58,24 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
|
||||
{"VRNDSCALEPH", 3},
|
||||
{"VREDUCEPH", 3},
|
||||
{"VGETMANTPH", 3},
|
||||
{"VFMADD132PH", 3},
|
||||
{"VFMADD213PH", 3},
|
||||
{"VFMADD231PH", 3},
|
||||
{"VFMSUB132PH", 3},
|
||||
{"VFMSUB213PH", 3},
|
||||
{"VFMSUB231PH", 3},
|
||||
{"VFMADDSUB132PH", 3},
|
||||
{"VFMADDSUB213PH", 3},
|
||||
{"VFMADDSUB231PH", 3},
|
||||
{"VFMSUBADD132PH", 3},
|
||||
{"VFMSUBADD213PH", 3},
|
||||
{"VFMSUBADD231PH", 3},
|
||||
{"VFMADD132SH", 1},
|
||||
{"VFMADD213SH", 1},
|
||||
{"VFMADD231SH", 1},
|
||||
{"VFMSUB132SH", 1},
|
||||
{"VFMSUB213SH", 1},
|
||||
{"VFMSUB231SH", 1},
|
||||
} {
|
||||
cands, ok := LookupExtension(arch.AMD64, tt.mnem)
|
||||
if !ok {
|
||||
@@ -71,8 +89,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
|
||||
t.Errorf("the %s lookup is not case-insensitive", tt.mnem)
|
||||
}
|
||||
}
|
||||
if got := arch.Extensions(arch.AMD64); len(got) != 121 {
|
||||
t.Errorf("the amd64 layer registers %d instructions, want 121", len(got))
|
||||
if got := arch.Extensions(arch.AMD64); len(got) != 163 {
|
||||
t.Errorf("the amd64 layer registers %d instructions, want 163", len(got))
|
||||
}
|
||||
if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok {
|
||||
t.Error("a non-extended mnemonic resolved")
|
||||
|
||||
Reference in new issue
Block a user