diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index 65c5135..d4990ca 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -684,7 +684,9 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) { // encodeAmdVec3Imm fills the three-vector form with a control immediate: // imm, src1, src2, dest, the order the reference listings write it in. An // entry with Mem set takes the memory shape of the second source, xmm3/m16 -// in the manual. +// in the manual. The destination takes the write mask where the entry +// declares one, the packed imm8-control forms the manual masks; the scalar +// forms take none, and the entry's flag refuses the spelling for them. func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) imm, err := in.amd64Imm8(ops[0], 1) @@ -694,23 +696,29 @@ func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) { if err := in.amd64Vector(ops[1], class, 2); err != nil { return nil, err } + dest, mask, zeroing, _, err := in.amd64WriteMask(ops[3], 4) + if err != nil { + return nil, err + } if in.Mem == 3 && ops[2].Kind == ExtMem { - if err := in.amd64Vector(ops[3], class, 4); err != nil { + if err := in.amd64Vector(dest, class, 4); err != nil { return nil, err } - out, err := in.amd64MemBytes(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2], 3) + out, err := in.amd64MemBytes(in.Bytes, dest.Reg, ops[1].Reg, ops[2], 3) if err != nil { return nil, err } + amd64ApplyMask(out, mask, zeroing) return append(out, imm), nil } if err := in.amd64Vector(ops[2], class, 3); err != nil { return nil, err } - if err := in.amd64Vector(ops[3], class, 4); err != nil { + if err := in.amd64Vector(dest, class, 4); err != nil { return nil, err } - out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg) + out := amd64Encode(in.Bytes, dest.Reg, ops[1].Reg, ops[2].Reg) + amd64ApplyMask(out, mask, zeroing) return append(out, imm), nil } @@ -1389,4 +1397,152 @@ var amd64Extensions = []ExtInstr{ {Name: "VGETMANTPH", Summary: "Extract the normalised mantissas of packed FP16 values under an imm8 control", Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x26, 0xC0}, Form: ExtFormAmdVec2Imm, Mem: 2, Mask: true, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VGETMANTPH (EVEX.128.NP.0F3A.W0 26 /r /ib)"}, + + // AVX512-FP16 packed fused multiply-add: the twelve packed FMA + // mnemonics of the FMA group, 132, 213 and 231 under the multiply-add, + // multiply-subtract, add-subtract and subtract-add pairings, three + // register widths each. EVEX.NDS.66.MAP6.W0 throughout, the opcode low + // byte the AVX512F single-precision FMA group carries one map over: 98 + // the add, 9A the subtract, 96 the add-subtract and 97 the + // subtract-add, the 213 and 231 forms ten and twenty above. The 512-bit + // register forms take the embedded rounding, the VL forms none; the + // destinations take the write mask and the memory shape of the second + // source the {1toN} broadcast. The golden vectors are quoted from the + // local GNU assembler, whose FP16 table matches the SDM entries row for + // row; the add-subtract pairing adds on the odd lanes and subtracts on + // the even ones, the subtract-add pairing the reverse. + {Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.512.66.MAP6.W0 98 /r)"}, + {Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.256.66.MAP6.W0 98 /r)"}, + {Name: "VFMADD132PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x98, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD132PH (EVEX.NDS.128.66.MAP6.W0 98 /r)"}, + {Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.512.66.MAP6.W0 A8 /r)"}, + {Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.256.66.MAP6.W0 A8 /r)"}, + {Name: "VFMADD213PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying a factor and the added term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD213PH (EVEX.NDS.128.66.MAP6.W0 A8 /r)"}, + {Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.512.66.MAP6.W0 B8 /r)"}, + {Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.256.66.MAP6.W0 B8 /r)"}, + {Name: "VFMADD231PH", Summary: "Multiply packed FP16 values and add the product, the destination supplying the added term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB8, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD231PH (EVEX.NDS.128.66.MAP6.W0 B8 /r)"}, + {Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.512.66.MAP6.W0 9A /r)"}, + {Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.256.66.MAP6.W0 9A /r)"}, + {Name: "VFMSUB132PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x9A, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB132PH (EVEX.NDS.128.66.MAP6.W0 9A /r)"}, + {Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.512.66.MAP6.W0 AA /r)"}, + {Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.256.66.MAP6.W0 AA /r)"}, + {Name: "VFMSUB213PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying a factor and the subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xAA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB213PH (EVEX.NDS.128.66.MAP6.W0 AA /r)"}, + {Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.512.66.MAP6.W0 BA /r)"}, + {Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.256.66.MAP6.W0 BA /r)"}, + {Name: "VFMSUB231PH", Summary: "Multiply packed FP16 values and subtract the product, the destination supplying the subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBA, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB231PH (EVEX.NDS.128.66.MAP6.W0 BA /r)"}, + {Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.512.66.MAP6.W0 96 /r)"}, + {Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.256.66.MAP6.W0 96 /r)"}, + {Name: "VFMADDSUB132PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x96, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB132PH (EVEX.NDS.128.66.MAP6.W0 96 /r)"}, + {Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.512.66.MAP6.W0 A6 /r)"}, + {Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.256.66.MAP6.W0 A6 /r)"}, + {Name: "VFMADDSUB213PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying a factor and the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB213PH (EVEX.NDS.128.66.MAP6.W0 A6 /r)"}, + {Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.512.66.MAP6.W0 B6 /r)"}, + {Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.256.66.MAP6.W0 B6 /r)"}, + {Name: "VFMADDSUB231PH", Summary: "Multiply packed FP16 values, adding the product on the odd lanes and subtracting it on the even ones, the destination supplying the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB6, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADDSUB231PH (EVEX.NDS.128.66.MAP6.W0 B6 /r)"}, + {Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.512.66.MAP6.W0 97 /r)"}, + {Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.256.66.MAP6.W0 97 /r)"}, + {Name: "VFMSUBADD132PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x97, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD132PH (EVEX.NDS.128.66.MAP6.W0 97 /r)"}, + {Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.512.66.MAP6.W0 A7 /r)"}, + {Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.256.66.MAP6.W0 A7 /r)"}, + {Name: "VFMSUBADD213PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying a factor and the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD213PH (EVEX.NDS.128.66.MAP6.W0 A7 /r)"}, + {Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x40, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.512.66.MAP6.W0 B7 /r)"}, + {Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x20, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.256.66.MAP6.W0 B7 /r)"}, + {Name: "VFMSUBADD231PH", Summary: "Multiply packed FP16 values, subtracting the product on the odd lanes and adding it on the even ones, the destination supplying the added or subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB7, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUBADD231PH (EVEX.NDS.128.66.MAP6.W0 B7 /r)"}, + + // AVX512-FP16 scalar fused multiply-add: the scalar mirrors of the + // packed FMA group, one half-precision value per lane, the 132, 213 and + // 231 pairings of the multiply-add and the multiply-subtract. The + // scalar opcodes sit one above the packed ones, 99 the add and 9B the + // subtract, in the LIG shape the rest of the scalar core carries. The + // register forms take the embedded rounding, the memory shape of the + // second source reads its m16 plain. + {Name: "VFMADD132SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x99, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD132SH (EVEX.NDS.LIG.66.MAP6.W0 99 /r)"}, + {Name: "VFMADD213SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying a factor and the added term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xA9, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD213SH (EVEX.NDS.LIG.66.MAP6.W0 A9 /r)"}, + {Name: "VFMADD231SH", Summary: "Multiply scalar FP16 values and add the product, the destination supplying the added term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xB9, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMADD231SH (EVEX.NDS.LIG.66.MAP6.W0 B9 /r)"}, + {Name: "VFMSUB132SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying a factor", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x9B, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB132SH (EVEX.NDS.LIG.66.MAP6.W0 9B /r)"}, + {Name: "VFMSUB213SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying a factor and the subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xAB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB213SH (EVEX.NDS.LIG.66.MAP6.W0 AB /r)"}, + {Name: "VFMSUB231SH", Summary: "Multiply scalar FP16 values and subtract the product, the destination supplying the subtracted term", + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0xBB, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VFMSUB231SH (EVEX.NDS.LIG.66.MAP6.W0 BB /r)"}, } diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 57b7d88..fda31e2 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -1100,6 +1100,219 @@ var amd64GoldenRows = []amd64GoldenRow{ {"vrndscaleph k7 zeroing", "VRNDSCALEPH", []ExtOperand{ExtImmediate(0x7b), ExtZmm(5), ExtWriteMasked(ExtZmm(6), 7, true)}, "62f37ccf08f57b", "62 f3 7c cf 08 f5 7b vrndscaleph $0x7b,%zmm5,%zmm6{%k7}{z}"}, + + // The AVX512-FP16 packed fused multiply-add, twelve mnemonics by three + // lengths: the 512-bit rows on the suite's high registers, the VL rows + // on the low ones, every register form quoted from the local GNU + // assembler's output (EVEX.NDS.66.MAP6.W0 throughout). + {"vfmadd132ph", "VFMADD132PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "6206154098f4", "62 06 15 40 98 f4 vfmadd132ph %zmm28,%zmm29,%zmm30"}, + {"vfmadd213ph", "VFMADD213PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061540a8f4", "62 06 15 40 a8 f4 vfmadd213ph %zmm28,%zmm29,%zmm30"}, + {"vfmadd231ph", "VFMADD231PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061540b8f4", "62 06 15 40 b8 f4 vfmadd231ph %zmm28,%zmm29,%zmm30"}, + {"vfmsub132ph", "VFMSUB132PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "620615409af4", "62 06 15 40 9a f4 vfmsub132ph %zmm28,%zmm29,%zmm30"}, + {"vfmsub213ph", "VFMSUB213PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061540aaf4", "62 06 15 40 aa f4 vfmsub213ph %zmm28,%zmm29,%zmm30"}, + {"vfmsub231ph", "VFMSUB231PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061540baf4", "62 06 15 40 ba f4 vfmsub231ph %zmm28,%zmm29,%zmm30"}, + {"vfmaddsub132ph", "VFMADDSUB132PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "6206154096f4", "62 06 15 40 96 f4 vfmaddsub132ph %zmm28,%zmm29,%zmm30"}, + {"vfmaddsub213ph", "VFMADDSUB213PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061540a6f4", "62 06 15 40 a6 f4 vfmaddsub213ph %zmm28,%zmm29,%zmm30"}, + {"vfmaddsub231ph", "VFMADDSUB231PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061540b6f4", "62 06 15 40 b6 f4 vfmaddsub231ph %zmm28,%zmm29,%zmm30"}, + {"vfmsubadd132ph", "VFMSUBADD132PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "6206154097f4", "62 06 15 40 97 f4 vfmsubadd132ph %zmm28,%zmm29,%zmm30"}, + {"vfmsubadd213ph", "VFMSUBADD213PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061540a7f4", "62 06 15 40 a7 f4 vfmsubadd213ph %zmm28,%zmm29,%zmm30"}, + {"vfmsubadd231ph", "VFMSUBADD231PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)}, + "62061540b7f4", "62 06 15 40 b7 f4 vfmsubadd231ph %zmm28,%zmm29,%zmm30"}, + {"vfmadd132ph ymm", "VFMADD132PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f6552898f4", "62 f6 55 28 98 f4 vfmadd132ph %ymm4,%ymm5,%ymm6"}, + {"vfmadd132ph xmm", "VFMADD132PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f6550898f4", "62 f6 55 08 98 f4 vfmadd132ph %xmm4,%xmm5,%xmm6"}, + {"vfmadd213ph ymm", "VFMADD213PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65528a8f4", "62 f6 55 28 a8 f4 vfmadd213ph %ymm4,%ymm5,%ymm6"}, + {"vfmadd213ph xmm", "VFMADD213PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65508a8f4", "62 f6 55 08 a8 f4 vfmadd213ph %xmm4,%xmm5,%xmm6"}, + {"vfmadd231ph ymm", "VFMADD231PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65528b8f4", "62 f6 55 28 b8 f4 vfmadd231ph %ymm4,%ymm5,%ymm6"}, + {"vfmadd231ph xmm", "VFMADD231PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65508b8f4", "62 f6 55 08 b8 f4 vfmadd231ph %xmm4,%xmm5,%xmm6"}, + {"vfmsub132ph ymm", "VFMSUB132PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f655289af4", "62 f6 55 28 9a f4 vfmsub132ph %ymm4,%ymm5,%ymm6"}, + {"vfmsub132ph xmm", "VFMSUB132PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f655089af4", "62 f6 55 08 9a f4 vfmsub132ph %xmm4,%xmm5,%xmm6"}, + {"vfmsub213ph ymm", "VFMSUB213PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65528aaf4", "62 f6 55 28 aa f4 vfmsub213ph %ymm4,%ymm5,%ymm6"}, + {"vfmsub213ph xmm", "VFMSUB213PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65508aaf4", "62 f6 55 08 aa f4 vfmsub213ph %xmm4,%xmm5,%xmm6"}, + {"vfmsub231ph ymm", "VFMSUB231PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65528baf4", "62 f6 55 28 ba f4 vfmsub231ph %ymm4,%ymm5,%ymm6"}, + {"vfmsub231ph xmm", "VFMSUB231PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65508baf4", "62 f6 55 08 ba f4 vfmsub231ph %xmm4,%xmm5,%xmm6"}, + {"vfmaddsub132ph ymm", "VFMADDSUB132PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f6552896f4", "62 f6 55 28 96 f4 vfmaddsub132ph %ymm4,%ymm5,%ymm6"}, + {"vfmaddsub132ph xmm", "VFMADDSUB132PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f6550896f4", "62 f6 55 08 96 f4 vfmaddsub132ph %xmm4,%xmm5,%xmm6"}, + {"vfmaddsub213ph ymm", "VFMADDSUB213PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65528a6f4", "62 f6 55 28 a6 f4 vfmaddsub213ph %ymm4,%ymm5,%ymm6"}, + {"vfmaddsub213ph xmm", "VFMADDSUB213PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65508a6f4", "62 f6 55 08 a6 f4 vfmaddsub213ph %xmm4,%xmm5,%xmm6"}, + {"vfmaddsub231ph ymm", "VFMADDSUB231PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65528b6f4", "62 f6 55 28 b6 f4 vfmaddsub231ph %ymm4,%ymm5,%ymm6"}, + {"vfmaddsub231ph xmm", "VFMADDSUB231PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65508b6f4", "62 f6 55 08 b6 f4 vfmaddsub231ph %xmm4,%xmm5,%xmm6"}, + {"vfmsubadd132ph ymm", "VFMSUBADD132PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f6552897f4", "62 f6 55 28 97 f4 vfmsubadd132ph %ymm4,%ymm5,%ymm6"}, + {"vfmsubadd132ph xmm", "VFMSUBADD132PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f6550897f4", "62 f6 55 08 97 f4 vfmsubadd132ph %xmm4,%xmm5,%xmm6"}, + {"vfmsubadd213ph ymm", "VFMSUBADD213PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65528a7f4", "62 f6 55 28 a7 f4 vfmsubadd213ph %ymm4,%ymm5,%ymm6"}, + {"vfmsubadd213ph xmm", "VFMSUBADD213PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65508a7f4", "62 f6 55 08 a7 f4 vfmsubadd213ph %xmm4,%xmm5,%xmm6"}, + {"vfmsubadd231ph ymm", "VFMSUBADD231PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f65528b7f4", "62 f6 55 28 b7 f4 vfmsubadd231ph %ymm4,%ymm5,%ymm6"}, + {"vfmsubadd231ph xmm", "VFMSUBADD231PH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f65508b7f4", "62 f6 55 08 b7 f4 vfmsubadd231ph %xmm4,%xmm5,%xmm6"}, + + // The scalar fused multiply-add, EVEX.NDS.LIG.66.MAP6.W0: the add + // opcodes one above the packed ones, the subtract two above them. + {"vfmadd132sh", "VFMADD132SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6206150099f4", "62 06 15 00 99 f4 vfmadd132sh %xmm28,%xmm29,%xmm30"}, + {"vfmadd213sh", "VFMADD213SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "62061500a9f4", "62 06 15 00 a9 f4 vfmadd213sh %xmm28,%xmm29,%xmm30"}, + {"vfmadd231sh", "VFMADD231SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "62061500b9f4", "62 06 15 00 b9 f4 vfmadd231sh %xmm28,%xmm29,%xmm30"}, + {"vfmsub132sh", "VFMSUB132SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620615009bf4", "62 06 15 00 9b f4 vfmsub132sh %xmm28,%xmm29,%xmm30"}, + {"vfmsub213sh", "VFMSUB213SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "62061500abf4", "62 06 15 00 ab f4 vfmsub213sh %xmm28,%xmm29,%xmm30"}, + {"vfmsub231sh", "VFMSUB231SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "62061500bbf4", "62 06 15 00 bb f4 vfmsub231sh %xmm28,%xmm29,%xmm30"}, + + // The FMA memory forms: the second source reads m512, m256, m128 or the + // scalar m16 from memory. The disp8 rows quote the listing's compressed + // spellings, whose source is N times the plain displacement the bytes + // carry; the R12 row exercises the SIB byte. + {"vfmadd132ph memory source", "VFMADD132PH", + []ExtOperand{ExtZmm(29), ExtMemory(1, 127), ExtZmm(30)}, + "6266154098717f", "62 66 15 40 98 71 7f vfmadd132ph 0x1fc0(%rcx),%zmm29,%zmm30 (Disp8(7f))"}, + {"vfmadd132ph ymm memory source", "VFMADD132PH", + []ExtOperand{ExtYmm(5), ExtMemory(1, 127), ExtYmm(6)}, + "62f6552898717f", "62 f6 55 28 98 71 7f vfmadd132ph 0xfe0(%ecx),%ymm5,%ymm6 (Disp8(7f))"}, + {"vfmadd132ph xmm memory source", "VFMADD132PH", + []ExtOperand{ExtXmm(5), ExtMemory(1, 127), ExtXmm(6)}, + "62f6550898717f", "62 f6 55 08 98 71 7f vfmadd132ph 0x7f0(%ecx),%xmm5,%xmm6 (Disp8(7f))"}, + {"vfmadd132ph memory source", "VFMADD132PH", + []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)}, + "624615409831", "62 46 15 40 98 31 vfmadd132ph (%r9),%zmm29,%zmm30"}, + {"vfmsub231ph memory source", "VFMSUB231PH", + []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtZmm(30)}, + "62461540ba31", "62 46 15 40 ba 31 vfmsub231ph (%r9),%zmm29,%zmm30"}, + {"vfmaddsub132ph memory source over an R12 base", "VFMADDSUB132PH", + []ExtOperand{ExtZmm(29), ExtMemory(12, 0), ExtZmm(30)}, + "62461540963424", ""}, + {"vfmadd132sh memory source", "VFMADD132SH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624615009931", "62 46 15 00 99 31 vfmadd132sh (%r9),%xmm29,%xmm30"}, + {"vfmadd132sh memory source disp8", "VFMADD132SH", + []ExtOperand{ExtXmm(29), ExtMemory(1, 1), ExtXmm(30)}, + "62661500997101", "62 66 15 00 99 71 01 vfmadd132sh 0x2(%rcx),%xmm29,%xmm30 (Disp8(01))"}, + {"vfmsub132sh memory source", "VFMSUB132SH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624615009b31", "62 46 15 00 9b 31 vfmsub132sh (%r9),%xmm29,%xmm30"}, + + // The FMA broadcast forms: one half-precision element the hardware + // splats across the lanes, {1to32}, {1to16} and {1to8}. + {"vfmadd132ph broadcast source", "VFMADD132PH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "624615509831", "62 46 15 50 98 31 vfmadd132ph (%r9){1to32},%zmm29,%zmm30"}, + {"vfmadd132ph ymm broadcast source", "VFMADD132PH", + []ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)}, + "62f655389831", "62 f6 55 38 98 31 vfmadd132ph (%ecx){1to16},%ymm5,%ymm6"}, + {"vfmadd132ph xmm broadcast source", "VFMADD132PH", + []ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)}, + "62f655189831", "62 f6 55 18 98 31 vfmadd132ph (%ecx){1to8},%xmm5,%xmm6"}, + {"vfmaddsub132ph broadcast source", "VFMADDSUB132PH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "624615509631", "62 46 15 50 96 31 vfmaddsub132ph (%r9){1to32},%zmm29,%zmm30"}, + {"vfmsubadd231ph broadcast source", "VFMSUBADD231PH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "62461550b731", "62 46 15 50 b7 31 vfmsubadd231ph (%r9){1to32},%zmm29,%zmm30"}, + + // The FMA rounding rows: the 512-bit register forms take EVEX.RC, the + // scalar register forms beside them. + {"vfmadd132ph rz-sae", "VFMADD132PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtRounded(ExtZmm(30), ExtRoundTruncate)}, + "6206157098f4", "62 06 15 70 98 f4 vfmadd132ph {rz-sae},%zmm28,%zmm29,%zmm30"}, + {"vfmadd213ph rn-sae", "VFMADD213PH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f65518a8f4", "62 f6 55 18 a8 f4 vfmadd213ph {rn-sae},%zmm4,%zmm5,%zmm6"}, + {"vfmsub132ph ru-sae", "VFMSUB132PH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundUp)}, + "62f655589af4", "62 f6 55 58 9a f4 vfmsub132ph {ru-sae},%zmm4,%zmm5,%zmm6"}, + {"vfmaddsub231ph rd-sae", "VFMADDSUB231PH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundDown)}, + "62f65538b6f4", "62 f6 55 38 b6 f4 vfmaddsub231ph {rd-sae},%zmm4,%zmm5,%zmm6"}, + {"vfmadd132sh rn-sae", "VFMADD132SH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundNearest)}, + "62f6551899f4", "62 f6 55 18 99 f4 vfmadd132sh {rn-sae},%xmm4,%xmm5,%xmm6"}, + {"vfmadd231sh rz-sae, high registers", "VFMADD231SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundTruncate)}, + "62061570b9f4", "62 06 15 70 b9 f4 vfmadd231sh {rz-sae},%xmm28,%xmm29,%xmm30"}, + + // The FMA write masks over the packed destinations. + {"vfmadd132ph k7 zeroing", "VFMADD132PH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)}, + "620615c798f4", "62 06 15 c7 98 f4 vfmadd132ph %zmm28,%zmm29,%zmm30{%k7}{z}"}, + {"vfmaddsub132ph k5 merging, memory source", "VFMADDSUB132PH", + []ExtOperand{ExtZmm(29), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 5, false)}, + "624615459631", "62 46 15 45 96 31 vfmaddsub132ph (%r9),%zmm29,%zmm30{%k5}"}, } // amd64ResolveEntry finds the table entry a golden row exercises: the entry @@ -1375,7 +1588,7 @@ func TestAmd64ExtRejects(t *testing.T) { "the position takes none"}, {"write mask on the control form's destination", "VGETMANTSH", []ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtXmm(28), ExtWriteMasked(ExtXmm(30), 7, true)}, - "the position takes none"}, + "the entry's destination takes none"}, {"write mask where the destination is memory", "VCOMISH", []ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, HasMask: true, Mask: 2}}, "the entry's destination takes none"}, @@ -1430,6 +1643,15 @@ func TestAmd64ExtRejects(t *testing.T) { {"reserved upper nibble on the packed mantissa control", "VGETMANTPH", []ExtOperand{ExtImmediate(0x7b), ExtZmm(5), ExtZmm(6)}, "reserved and must be zero"}, + {"broadcast on the scalar multiply-add", "VFMADD132SH", + []ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)}, + "the entry's memory operand takes none"}, + {"rounding on the 256-bit multiply-add", "VFMADD132PH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)}, + "the entry's destination takes none"}, + {"a register where the multiply-add reads memory", "VFMSUB231PH", + []ExtOperand{ExtZmm(29), ExtYmm(4), ExtZmm(30)}, + "wants a ZMM register"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) @@ -1574,7 +1796,7 @@ func TestAmd64ExtArchBinding(t *testing.T) { t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) } } - if got := Extensions(AMD64); len(got) != 121 { - t.Errorf("the amd64 layer registers %d instructions, want 121", len(got)) + if got := Extensions(AMD64); len(got) != 163 { + t.Errorf("the amd64 layer registers %d instructions, want 163", len(got)) } } diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index 325fd22..b93d9b6 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -58,6 +58,24 @@ func TestAmd64ExtensionRegistry(t *testing.T) { {"VRNDSCALEPH", 3}, {"VREDUCEPH", 3}, {"VGETMANTPH", 3}, + {"VFMADD132PH", 3}, + {"VFMADD213PH", 3}, + {"VFMADD231PH", 3}, + {"VFMSUB132PH", 3}, + {"VFMSUB213PH", 3}, + {"VFMSUB231PH", 3}, + {"VFMADDSUB132PH", 3}, + {"VFMADDSUB213PH", 3}, + {"VFMADDSUB231PH", 3}, + {"VFMSUBADD132PH", 3}, + {"VFMSUBADD213PH", 3}, + {"VFMSUBADD231PH", 3}, + {"VFMADD132SH", 1}, + {"VFMADD213SH", 1}, + {"VFMADD231SH", 1}, + {"VFMSUB132SH", 1}, + {"VFMSUB213SH", 1}, + {"VFMSUB231SH", 1}, } { cands, ok := LookupExtension(arch.AMD64, tt.mnem) if !ok { @@ -71,8 +89,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { t.Errorf("the %s lookup is not case-insensitive", tt.mnem) } } - if got := arch.Extensions(arch.AMD64); len(got) != 121 { - t.Errorf("the amd64 layer registers %d instructions, want 121", len(got)) + if got := arch.Extensions(arch.AMD64); len(got) != 163 { + t.Errorf("the amd64 layer registers %d instructions, want 163", len(got)) } if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { t.Error("a non-extended mnemonic resolved")