diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index 90598cb..a1004a5 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -400,24 +400,42 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) { } // encodeAmdVec3Imm fills the three-vector form with a control immediate: -// imm, src1, src2, dest, the order the reference listings write it in. +// imm, src1, src2, dest, the order the reference listings write it in. An +// entry with Mem set takes the memory shape of the second source, xmm3/m16 +// in the manual. func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) imm, err := in.amd64Imm8(ops[0], 1) if err != nil { return nil, err } - for i, op := range ops[1:] { - if err := in.amd64Vector(op, class, i+2); err != nil { + if err := in.amd64Vector(ops[1], class, 2); err != nil { + return nil, err + } + if in.Mem == 3 && ops[2].Kind == ExtMem { + base, disp, err := in.amd64Memory(ops[2], 3) + if err != nil { return nil, err } + if err := in.amd64Vector(ops[3], class, 4); err != nil { + return nil, err + } + out := amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp) + return append(out, imm), nil + } + if err := in.amd64Vector(ops[2], class, 3); err != nil { + return nil, err + } + if err := in.amd64Vector(ops[3], class, 4); err != nil { + return nil, err } out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg) return append(out, imm), nil } // encodeAmdMask2Imm fills the opmask-destination form with a control -// immediate: imm, src1, src2, dest. +// immediate: imm, src1, src2, dest. An entry with Mem set takes the memory +// shape of the second source. func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) imm, err := in.amd64Imm8(ops[0], 1) @@ -427,16 +445,25 @@ func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) { if err := in.amd64Vector(ops[1], class, 2); err != nil { return nil, err } - if err := in.amd64Vector(ops[2], class, 3); err != nil { - return nil, err - } if ops[3].Kind != ExtKReg { return nil, fmt.Errorf("%s: operand 4 wants an opmask register, got %s", in.Name, ops[3].Kind) } if err := in.amd64PlainReg(ops[3], 7, 4); err != nil { return nil, err } - out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg) + var out []byte + if in.Mem == 3 && ops[2].Kind == ExtMem { + base, disp, err := in.amd64Memory(ops[2], 3) + if err != nil { + return nil, err + } + out = amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp) + return append(out, imm), nil + } + if err := in.amd64Vector(ops[2], class, 3); err != nil { + return nil, err + } + out = amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg) return append(out, imm), nil } @@ -471,12 +498,24 @@ var ExtFP16CmpPredicates = [32]string{ } // encodeAmdVecGprVec fills the conversion form with a general-register -// source: src1, gpr, dest. VCVTSI2SH XMM1, XMM2, EAX style. +// source: src1, gpr, dest. VCVTSI2SH X1, X2, EAX style. An entry with Mem +// set takes the memory shape of the integer source, which the manual spells +// r/m32: the value converts straight out of memory. func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err } + if in.Mem == 2 && ops[1].Kind == ExtMem { + base, disp, err := in.amd64Memory(ops[1], 2) + if err != nil { + return nil, err + } + if err := in.amd64Vector(ops[2], class, 3); err != nil { + return nil, err + } + return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil + } if err := in.amd64Gpr(ops[1], 2); err != nil { return nil, err } @@ -622,10 +661,10 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"}, {Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others", - Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSCALEFSH (EVEX.NDS.LIG.66.MAP6.W0 2D /r)"}, {Name: "VGETEXPSH", Summary: "Convert the exponent of a scalar FP16 value to an FP16 value", - Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VGETEXPSH (EVEX.NDS.LIG.66.MAP6.W0 43 /r)"}, {Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS", Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16, @@ -646,16 +685,16 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"}, {Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 2A /r)"}, {Name: "VCVTSI2SH", Summary: "Convert one signed 64-bit integer to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 2A /r)"}, {Name: "VCVTUSI2SH", Summary: "Convert one unsigned 32-bit integer to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 7B /r)"}, {Name: "VCVTUSI2SH", Summary: "Convert one unsigned 64-bit integer to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"}, {Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer", Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, @@ -749,15 +788,15 @@ var amd64Extensions = []ExtInstr{ // ExtFP16CmpPredicates above; the reserved upper nibble of the mantissa // control is refused rather than encoded. {Name: "VCMPSH", Summary: "Compare scalar FP16 values into an opmask under an imm8 predicate", - Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Mem: 3, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCMPSH (EVEX.LLIG.F3.0F3A.W0 C2 /r /ib)"}, {Name: "VGETMANTSH", Summary: "Extract the normalised mantissa of a scalar FP16 value under an imm8 control", - Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VGETMANTSH (EVEX.LLIG.NP.0F3A.W0 27 /r /ib)"}, {Name: "VREDUCESH", Summary: "Reduce a scalar FP16 value by imm8 fraction bits under an imm8 round control", - Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VREDUCESH (EVEX.LLIG.NP.0F3A.W0 57 /r /ib)"}, {Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control", - Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"}, } diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 0c5a27f..49b5a54 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -408,6 +408,43 @@ var amd64GoldenRows = []amd64GoldenRow{ // list, both destinations being XMM, so the resolver cannot tell them // apart and the register row above pins the 128-bit template alone. + // The remaining scalar memory forms: the scale and exponent extracts, + // the imm8-control group, and the integer converts, whose second + // source the manual spells r/m32. The W1 integer converts take the + // same operand list as the W0 ones, memory carrying no width to pick + // between them, so the memory rows pin the W0 templates and the W1 + // entries rest on their register rows. + {"vscalefsh memory source", "VSCALEFSH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624615002d31", "62 46 15 00 2d 31 vscalefsh (%r9),%xmm29,%xmm30"}, + {"vgetexpsh memory source", "VGETEXPSH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624615004331", "62 46 15 00 43 31 vgetexpsh (%r9),%xmm29,%xmm30"}, + {"vgetexpsh memory source disp8", "VGETEXPSH", + []ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)}, + "6266150043717f", "62 66 15 00 43 71 7f vgetexpsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"}, + {"vcmpsh memory source", "VCMPSH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtMemory(9, 0), ExtMask(5)}, + "62d31600c2297b", "62 d3 16 00 c2 29 7b vcmpsh $0x7b,(%r9),%xmm29,%k5"}, + {"vcmpsh memory source disp8", "VCMPSH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtMemory(1, 127), ExtMask(5)}, + "62f31600c2697f7b", "62 f3 16 00 c2 69 7f 7b vcmpsh $0x7b,0xfe(%rcx),%xmm29,%k5 (Disp8(7f))"}, + {"vgetmantsh memory source", "VGETMANTSH", + []ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "6243140027310b", "62 43 14 00 27 31 7b vgetmantsh $0x7b,(%r9),%xmm29,%xmm30 (opcode row only)"}, + {"vreducesh memory source", "VREDUCESH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "6243140057317b", "62 43 14 00 57 31 7b vreducesh $0x7b,(%r9),%xmm29,%xmm30"}, + {"vrndscalesh memory source", "VRNDSCALESH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624314000a317b", "62 43 14 00 0a 31 7b vrndscalesh $0x7b,(%r9),%xmm29,%xmm30"}, + {"vcvtsi2sh memory source", "VCVTSI2SH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624516002a31", "62 45 16 00 2a 31 vcvtsi2shl (%r9),%xmm29,%xmm30"}, + {"vcvtusi2sh memory source", "VCVTUSI2SH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, + "624516007b31", "62 45 16 00 7b 31 vcvtusi2shl (%r9),%xmm29,%xmm30"}, + // High registers in a 512-bit form exercise the EVEX extension bits: // with both sources above 15 the B bar and X bar bits clear, while the // destination zmm23 keeps R bar set in byte one (derived from the @@ -619,11 +656,14 @@ func TestAmd64ExtRejects(t *testing.T) { {"memory in the arithmetic's destination", "VADDSH", []ExtOperand{ExtXmm(28), ExtXmm(29), ExtMemory(9, 0)}, "wants an XMM register"}, - {"memory where the general register belongs", "VCVTSI2SH", - []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)}, - "wants a 32-bit general register"}, - {"memory as the compare's second source", "VCMPSH", - []ExtOperand{ExtImmediate(7), ExtXmm(28), ExtMemory(9, 0), ExtMask(5)}, + {"memory as the compare's mask destination", "VCMPSH", + []ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtMemory(9, 0)}, + "wants an opmask register"}, + {"memory as the convert's first source", "VCVTSI2SH", + []ExtOperand{ExtMemory(9, 0), ExtGpr32(2), ExtXmm(30)}, + "wants an XMM register"}, + {"memory as the mantissa control's first source", "VGETMANTSH", + []ExtOperand{ExtImmediate(0x0b), ExtMemory(9, 0), ExtXmm(29), ExtXmm(30)}, "wants an XMM register"}, {"memory as the intersect source", "VP2INTERSECTD", []ExtOperand{ExtZmm(2), ExtMemory(9, 0), ExtMask(0)},