feat(arch): add the AVX512-FP16 FMA families to the extension layer
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
87493d9391
commit
8080e0acef
3 files changed
+406
-10
No files matched your search
+161
-5
@@ -684,7 +684,9 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
|
||||
// encodeAmdVec3Imm fills the three-vector form with a control immediate:
|
||||
// imm, src1, src2, dest, the order the reference listings write it in. An
|
||||
// entry with Mem set takes the memory shape of the second source, xmm3/m16
|
||||
// in the manual.
|
||||
// in the manual. The destination takes the write mask where the entry
|
||||
// declares one, the packed imm8-control forms the manual masks; the scalar
|
||||
// forms take none, and the entry's flag refuses the spelling for them.
|
||||
func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
imm, err := in.amd64Imm8(ops[0], 1)
|
||||
@@ -694,23 +696,29 @@ func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
|
||||
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
dest, mask, zeroing, _, err := in.amd64WriteMask(ops[3], 4)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 3 && ops[2].Kind == ExtMem {
|
||||
if err := in.amd64Vector(ops[3], class, 4); err != nil {
|
||||
if err := in.amd64Vector(dest, class, 4); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out, err := in.amd64MemBytes(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2], 3)
|
||||
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, ops[1].Reg, ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[3], class, 4); err != nil {
|
||||
if err := in.amd64Vector(dest, class, 4); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
|
||||
out := amd64Encode(in.Bytes, dest.Reg, ops[1].Reg, ops[2].Reg)
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
|
||||
@@ -1389,4 +1397,152 @@ var amd64Extensions = []ExtInstr{
|
||||
{Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.128.NP.0F3A.W0 26 /r /ib)"},
|
||||
|
||||
// AVX512-FP16 packed fused multiply-add: the twelve packed FMA
|
||||
// mnemonics of the FMA group, 132, 213 and 231 under the multiply-add,
|
||||
// multiply-subtract, add-subtract and subtract-add pairings, three
|
||||
// register widths each. EVEX.NDS.66.MAP6.W0 throughout, the opcode low
|
||||
// byte the AVX512F single-precision FMA group carries one map over: 98
|
||||
// the add, 9A the subtract, 96 the add-subtract and 97 the
|
||||
// subtract-add, the 213 and 231 forms ten and twenty above. The 512-bit
|
||||
// register forms take the embedded rounding, the VL forms none; the
|
||||
// destinations take the write mask and the memory shape of the second
|
||||
// source the {1toN} broadcast. The golden vectors are quoted from the
|
||||
// local GNU assembler, whose FP16 table matches the SDM entries row for
|
||||
// row; the add-subtract pairing adds on the odd lanes and subtracts on
|
||||
// the even ones, the subtract-add pairing the reverse.
|
||||
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.512.66.MAP6.W0 98 /r)"},
|
||||
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.256.66.MAP6.W0 98 /r)"},
|
||||
{Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.128.66.MAP6.W0 98 /r)"},
|
||||
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.512.66.MAP6.W0 A8 /r)"},
|
||||
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.256.66.MAP6.W0 A8 /r)"},
|
||||
{Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.128.66.MAP6.W0 A8 /r)"},
|
||||
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.512.66.MAP6.W0 B8 /r)"},
|
||||
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.256.66.MAP6.W0 B8 /r)"},
|
||||
{Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.128.66.MAP6.W0 B8 /r)"},
|
||||
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.512.66.MAP6.W0 9A /r)"},
|
||||
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.256.66.MAP6.W0 9A /r)"},
|
||||
{Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.128.66.MAP6.W0 9A /r)"},
|
||||
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.512.66.MAP6.W0 AA /r)"},
|
||||
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.256.66.MAP6.W0 AA /r)"},
|
||||
{Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.128.66.MAP6.W0 AA /r)"},
|
||||
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.512.66.MAP6.W0 BA /r)"},
|
||||
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.256.66.MAP6.W0 BA /r)"},
|
||||
{Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.128.66.MAP6.W0 BA /r)"},
|
||||
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.512.66.MAP6.W0 96 /r)"},
|
||||
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.256.66.MAP6.W0 96 /r)"},
|
||||
{Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.128.66.MAP6.W0 96 /r)"},
|
||||
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.512.66.MAP6.W0 A6 /r)"},
|
||||
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.256.66.MAP6.W0 A6 /r)"},
|
||||
{Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.128.66.MAP6.W0 A6 /r)"},
|
||||
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.512.66.MAP6.W0 B6 /r)"},
|
||||
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.256.66.MAP6.W0 B6 /r)"},
|
||||
{Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.128.66.MAP6.W0 B6 /r)"},
|
||||
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.512.66.MAP6.W0 97 /r)"},
|
||||
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.256.66.MAP6.W0 97 /r)"},
|
||||
{Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.128.66.MAP6.W0 97 /r)"},
|
||||
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.512.66.MAP6.W0 A7 /r)"},
|
||||
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.256.66.MAP6.W0 A7 /r)"},
|
||||
{Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.128.66.MAP6.W0 A7 /r)"},
|
||||
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.512.66.MAP6.W0 B7 /r)"},
|
||||
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.256.66.MAP6.W0 B7 /r)"},
|
||||
{Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.128.66.MAP6.W0 B7 /r)"},
|
||||
|
||||
// AVX512-FP16 scalar fused multiply-add: the scalar mirrors of the
|
||||
// packed FMA group, one half-precision value per lane, the 132, 213 and
|
||||
// 231 pairings of the multiply-add and the multiply-subtract. The
|
||||
// scalar opcodes sit one above the packed ones, 99 the add and 9B the
|
||||
// subtract, in the LIG shape the rest of the scalar core carries. The
|
||||
// register forms take the embedded rounding, the memory shape of the
|
||||
// second source reads its m16 plain.
|
||||
{Name: "VFMADD132SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x99, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD132SH (EVEX.NDS.LIG.66.MAP6.W0 99 /r)"},
|
||||
{Name: "VFMADD213SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying a factor and the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA9, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD213SH (EVEX.NDS.LIG.66.MAP6.W0 A9 /r)"},
|
||||
{Name: "VFMADD231SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying the added term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB9, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMADD231SH (EVEX.NDS.LIG.66.MAP6.W0 B9 /r)"},
|
||||
{Name: "VFMSUB132SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying a factor",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x9B, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB132SH (EVEX.NDS.LIG.66.MAP6.W0 9B /r)"},
|
||||
{Name: "VFMSUB213SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying a factor and the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xAB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB213SH (EVEX.NDS.LIG.66.MAP6.W0 AB /r)"},
|
||||
{Name: "VFMSUB231SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying the subtracted term",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VFMSUB231SH (EVEX.NDS.LIG.66.MAP6.W0 BB /r)"},
|
||||
}
|
||||
Reference in new issue
Block a user