feat(asm): complete the EVEX conversions, narrowing and mask-vector moves
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
+107
-4
@@ -308,6 +308,75 @@ var evexTable = map[string]evexSpec{
|
|||||||
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
|
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
|
||||||
// rm=dst, imm8 — the extract layout).
|
// rm=dst, imm8 — the extract layout).
|
||||||
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
|
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// EVEX — unsigned and truncating conversions. The PD sources are the
|
||||||
|
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
|
||||||
|
// the length); the PS/UQQ destinations are wide and follow the
|
||||||
|
// destination.
|
||||||
|
"VCVTPD2PS": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2PSX": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTPD2PSY": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTPD2UDQ": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2UDQX": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTPD2UDQY": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTTPD2UDQ": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTTPD2UDQX": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTTPD2UDQY": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTTPD2UQQ": {1, 0x78, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2UDQ": {1, 0x79, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTTPS2UDQ": {1, 0x78, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2UQQ": {1, 0x79, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTTPS2UQQ": {1, 0x78, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTTPD2QQ": {1, 0x7A, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTTPS2QQ": {1, 0x7A, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTUQQ2PD": {1, 0x7A, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTUQQ2PS": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTUQQ2PSX": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTUQQ2PSY": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
|
||||||
|
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow
|
||||||
|
// source; disp8×N follows its size).
|
||||||
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg,
|
||||||
|
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
|
||||||
|
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVSDW": {2, 0x23, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSQW": {2, 0x24, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSQD": {2, 0x25, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSWB": {2, 0x20, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSWB": {2, 0x10, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSDB": {2, 0x11, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVUSQB": {2, 0x12, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVUSDW": {2, 0x13, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSQW": {2, 0x14, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVUSQD": {2, 0x15, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
|
||||||
|
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register
|
||||||
|
// into a vector (rm = K source, reg = vector destination), *2M does the
|
||||||
|
// reverse (reg = K destination, rm = vector source, the length follows
|
||||||
|
// the vector).
|
||||||
|
"VPMOVM2B": {2, 0x28, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2W": {2, 0x28, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2D": {2, 0x38, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2Q": {2, 0x38, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVB2M": {2, 0x29, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||||||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||||||
// the xmm/ymm/zmm destination lengths).
|
// the xmm/ymm/zmm destination lengths).
|
||||||
@@ -543,6 +612,9 @@ var evexRound = map[string]bool{
|
|||||||
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
|
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
|
||||||
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
|
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
|
||||||
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
|
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
|
||||||
|
"VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true,
|
||||||
|
"VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true,
|
||||||
|
"VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true,
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
||||||
@@ -562,6 +634,9 @@ var evexBcstN = map[string]int{
|
|||||||
"VRANGEPD": 8, "VRANGEPS": 4,
|
"VRANGEPD": 8, "VRANGEPS": 4,
|
||||||
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
|
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
|
||||||
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
|
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
|
||||||
|
"VCVTPD2PS": 8, "VCVTPD2UDQ": 8, "VCVTTPD2UDQ": 8, "VCVTTPD2UQQ": 8,
|
||||||
|
"VCVTPS2UDQ": 4, "VCVTTPS2UDQ": 4, "VCVTPS2UQQ": 4, "VCVTTPS2UQQ": 4,
|
||||||
|
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
|
||||||
}
|
}
|
||||||
|
|
||||||
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
||||||
@@ -591,6 +666,18 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
|
|||||||
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
||||||
// broadcast.
|
// broadcast.
|
||||||
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
||||||
|
// The mask/vector conversions take the K register as a genuine operand
|
||||||
|
// (source or destination), not as a mask, and accept no suffixes.
|
||||||
|
if evexKOperand[mnemUpper] {
|
||||||
|
if sfx.any() {
|
||||||
|
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
|
||||||
|
}
|
||||||
|
spec, ok := evexTable[mnemUpper]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("unsupported instruction %q", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexRM(spec, ops, 0, sfx)
|
||||||
|
}
|
||||||
spec, inTable := evexTable[mnemUpper]
|
spec, inTable := evexTable[mnemUpper]
|
||||||
if inTable {
|
if inTable {
|
||||||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||||||
@@ -713,17 +800,25 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuf
|
|||||||
}
|
}
|
||||||
|
|
||||||
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
||||||
// no vvvv), e.g. VCVTQQ2PD.
|
// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the
|
||||||
|
// *2M mask conversions), in which case the vector length comes from the
|
||||||
|
// source.
|
||||||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||||||
}
|
}
|
||||||
src, dst := ops[0], ops[1]
|
src, dst := ops[0], ops[1]
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||||||
return fmt.Errorf("EVEX destination must be a vector register")
|
return fmt.Errorf("EVEX destination must be a vector or mask register")
|
||||||
}
|
}
|
||||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
|
ll := dstReg.vecLenBit()
|
||||||
|
if dstReg.mask {
|
||||||
|
if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
|
ll = r.vecLenBit()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
||||||
@@ -1260,6 +1355,14 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
|
|||||||
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// evexKOperand lists the instructions whose K register is a genuine operand
|
||||||
|
// (the source or destination of a mask/vector conversion) rather than a
|
||||||
|
// mask modifier — the M2 and 2M conversions. They take no masking.
|
||||||
|
var evexKOperand = map[string]bool{
|
||||||
|
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
||||||
|
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
||||||
|
}
|
||||||
|
|
||||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||||
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
||||||
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
||||||
|
|||||||
@@ -467,6 +467,96 @@ func TestEvexHelperGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
|
||||||
|
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
|
||||||
|
// narrowing stores and the mask/vector conversions, byte for byte against
|
||||||
|
// the Go assembler.
|
||||||
|
func TestEvexConversionGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// Unsigned and truncating conversions.
|
||||||
|
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
|
||||||
|
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
|
||||||
|
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
|
||||||
|
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
|
||||||
|
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
|
||||||
|
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
|
||||||
|
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
|
||||||
|
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
|
||||||
|
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
|
||||||
|
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
|
||||||
|
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
|
||||||
|
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
|
||||||
|
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
|
||||||
|
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
|
||||||
|
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
|
||||||
|
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
|
||||||
|
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
|
||||||
|
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
|
||||||
|
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
|
||||||
|
// The remaining sign/zero-extending moves.
|
||||||
|
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
|
||||||
|
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
|
||||||
|
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
|
||||||
|
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
|
||||||
|
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
|
||||||
|
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
|
||||||
|
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
|
||||||
|
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
|
||||||
|
// Signed narrowing stores.
|
||||||
|
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
|
||||||
|
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
|
||||||
|
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
|
||||||
|
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
|
||||||
|
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
|
||||||
|
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
|
||||||
|
// Unsigned narrowing stores.
|
||||||
|
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
|
||||||
|
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
|
||||||
|
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
|
||||||
|
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
|
||||||
|
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
|
||||||
|
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
|
||||||
|
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
|
||||||
|
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
|
||||||
|
// Mask/vector conversions (the K register is an operand, not a
|
||||||
|
// mask).
|
||||||
|
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
|
||||||
|
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
|
||||||
|
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
|
||||||
|
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
|
||||||
|
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
|
||||||
|
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
|
||||||
|
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
|
||||||
|
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestEvexErrors checks the EVEX-specific error paths.
|
// TestEvexErrors checks the EVEX-specific error paths.
|
||||||
func TestEvexErrors(t *testing.T) {
|
func TestEvexErrors(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
|
|||||||
+14
-1
@@ -130,9 +130,15 @@ var vexTable = map[string]vexSpec{
|
|||||||
// no vvvv).
|
// no vvvv).
|
||||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||||
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM},
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
||||||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
||||||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
||||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||||
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
||||||
@@ -198,6 +204,11 @@ var vexTable = map[string]vexSpec{
|
|||||||
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
||||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
||||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
||||||
|
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
|
||||||
|
// spellings: the destination is always XMM and the spelling fixes the
|
||||||
|
// source length (X = 128, Y = 256).
|
||||||
|
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||||
|
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||||
|
|
||||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||||
@@ -222,6 +233,8 @@ var vexSrcLen = map[string]int{
|
|||||||
"VCVTPD2DQY": 1,
|
"VCVTPD2DQY": 1,
|
||||||
"VCVTTPD2DQX": 0,
|
"VCVTTPD2DQX": 0,
|
||||||
"VCVTTPD2DQY": 1,
|
"VCVTTPD2DQY": 1,
|
||||||
|
"VCVTPD2PSX": 0,
|
||||||
|
"VCVTPD2PSY": 1,
|
||||||
}
|
}
|
||||||
|
|
||||||
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
||||||
|
|||||||
+1
-1
@@ -28,7 +28,7 @@ import (
|
|||||||
|
|
||||||
// version is the release version, stamped at build time via
|
// version is the release version, stamped at build time via
|
||||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
// -ldflags "-X main.version=…" (defaulting to the current release).
|
||||||
var version = "0.14.0"
|
var version = "0.15.0"
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if len(os.Args) < 2 {
|
if len(os.Args) < 2 {
|
||||||
|
|||||||
@@ -240,8 +240,9 @@ expand/compress family, the broadcasts, the opmask-register instructions
|
|||||||
moves and the remaining extending/narrowing moves, the floating-point
|
moves and the remaining extending/narrowing moves, the floating-point
|
||||||
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
|
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
|
||||||
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
|
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
|
||||||
opmask destination, and the remaining VCVT* conversions including the
|
opmask destination, and the VCVT* conversions — signed, unsigned and
|
||||||
half-precision pair), and gather/scatter with VSIB addressing — both the
|
truncating, including the length-suffixed X/Y spellings and the
|
||||||
|
mask/vector conversions VPMOVM2*/VPMOV*2M), and gather/scatter with VSIB addressing — both the
|
||||||
VEX spelling with a vector mask register and the EVEX spelling with an
|
VEX spelling with a vector mask register and the EVEX spelling with an
|
||||||
explicit K mask, where the EVEX length follows the VSIB index register,
|
explicit K mask, where the EVEX length follows the VSIB index register,
|
||||||
not the data register. The EVEX mnemonic
|
not the data register. The EVEX mnemonic
|
||||||
|
|||||||
Reference in New Issue
Block a user