Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5af12e15ac | ||
|
|
db8e3fc160 | ||
|
|
1312122a99 | ||
|
|
11f962fbcc |
+7
-1
@@ -57,7 +57,7 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || base == "KMOVW" || base == "KMOVQ" {
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
||||||
return e.encodeVec(base, ops, sfx)
|
return e.encodeVec(base, ops, sfx)
|
||||||
}
|
}
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
@@ -130,6 +130,12 @@ func splitSize(upper string) (base string, size int) {
|
|||||||
// takes EVEX when an operand demands it (a ZMM or K register, or an
|
// takes EVEX when an operand demands it (a ZMM or K register, or an
|
||||||
// EVEX-only mnemonic) and VEX otherwise.
|
// EVEX-only mnemonic) and VEX otherwise.
|
||||||
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
||||||
|
if gs, ok := gatherTable[upper]; ok {
|
||||||
|
return e.encodeGather(upper, gs, ops, sfx)
|
||||||
|
}
|
||||||
|
if ss, ok := scatterTable[upper]; ok {
|
||||||
|
return e.encodeScatter(upper, ss, ops, sfx)
|
||||||
|
}
|
||||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
if upper == "KMOVW" || upper == "KMOVQ" {
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||||
|
|||||||
+378
-8
@@ -234,6 +234,180 @@ var evexTable = map[string]evexSpec{
|
|||||||
// EVEX W1 qword shifts.
|
// EVEX W1 qword shifts.
|
||||||
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src).
|
||||||
|
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is
|
||||||
|
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
|
||||||
|
// forms, these take the 66 prefix; W selects double/single.
|
||||||
|
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VRCP14SS": {2, 0x4D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VRSQRT14SD": {2, 0x4F, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
// EVEX.66.0F38 — scale by a power of two (NDS form).
|
||||||
|
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
|
||||||
|
// EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst,
|
||||||
|
// rm=src, imm8).
|
||||||
|
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VGETMANTPD": {3, 0x26, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS +
|
||||||
|
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
|
||||||
|
// prefix; W selects double/single.
|
||||||
|
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VRNDSCALESS": {3, 0x0A, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
"VGETMANTSD": {3, 0x27, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VGETMANTSS": {3, 0x27, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
"VREDUCESD": {3, 0x57, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VREDUCESS": {3, 0x57, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
"VFIXUPIMMPD": {3, 0x54, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VFIXUPIMMPS": {3, 0x54, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VFIXUPIMMSD": {3, 0x55, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VFIXUPIMMSS": {3, 0x55, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
"VRANGEPD": {3, 0x50, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VRANGEPS": {3, 0x50, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||||
|
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||||
|
|
||||||
|
// EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the
|
||||||
|
// reg field carries the opmask destination. The packed forms carry an
|
||||||
|
// explicit length in the mnemonic (X/Y/Z).
|
||||||
|
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
|
||||||
|
"VFPCLASSPDY": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 32, 0}},
|
||||||
|
"VFPCLASSPDZ": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 0, 64}},
|
||||||
|
"VFPCLASSPSX": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{16, 0, 0}},
|
||||||
|
"VFPCLASSPSY": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 32, 0}},
|
||||||
|
"VFPCLASSPSZ": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 0, 64}},
|
||||||
|
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
|
||||||
|
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
|
||||||
|
|
||||||
|
// EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit
|
||||||
|
// source sets the length); the rest follow the destination.
|
||||||
|
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
// EVEX.66.0F38 — half-precision convert (half-width source).
|
||||||
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
|
||||||
|
// rm=dst, imm8 — the extract layout).
|
||||||
|
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// EVEX — unsigned and truncating conversions. The PD sources are the
|
||||||
|
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
|
||||||
|
// the length); the PS/UQQ destinations are wide and follow the
|
||||||
|
// destination.
|
||||||
|
"VCVTPD2PS": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2PSX": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTPD2PSY": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTPD2UDQ": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTPD2UDQX": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTPD2UDQY": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTTPD2UDQ": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTTPD2UDQX": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTTPD2UDQY": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTTPD2UQQ": {1, 0x78, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2UDQ": {1, 0x79, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTTPS2UDQ": {1, 0x78, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTPS2UQQ": {1, 0x79, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTTPS2UQQ": {1, 0x78, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTTPD2QQ": {1, 0x7A, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTTPS2QQ": {1, 0x7A, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VCVTUQQ2PD": {1, 0x7A, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCVTUQQ2PS": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}},
|
||||||
|
"VCVTUQQ2PSX": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTUQQ2PSY": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||||
|
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||||
|
|
||||||
|
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow
|
||||||
|
// source; disp8×N follows its size).
|
||||||
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg,
|
||||||
|
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
|
||||||
|
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVSDW": {2, 0x23, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSQW": {2, 0x24, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVSQD": {2, 0x25, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVSWB": {2, 0x20, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSWB": {2, 0x10, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSDB": {2, 0x11, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVUSQB": {2, 0x12, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||||
|
"VPMOVUSDW": {2, 0x13, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVUSQW": {2, 0x14, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVUSQD": {2, 0x15, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
|
||||||
|
|
||||||
|
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register
|
||||||
|
// into a vector (rm = K source, reg = vector destination), *2M does the
|
||||||
|
// reverse (reg = K destination, rm = vector source, the length follows
|
||||||
|
// the vector).
|
||||||
|
"VPMOVM2B": {2, 0x28, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2W": {2, 0x28, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2D": {2, 0x38, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVM2Q": {2, 0x38, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVB2M": {2, 0x29, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX — scalar conversions between vector and general-purpose
|
||||||
|
// registers. Vector to GPR (two operands: vec/mem source, GPR
|
||||||
|
// destination, vvvv unused): the signed and truncated pair, and the
|
||||||
|
// unsigned forms (EVEX only).
|
||||||
|
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSD2USIL": {1, 0x79, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSD2USIQ": {1, 0x79, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSS2USIL": {1, 0x79, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSS2USIQ": {1, 0x79, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSD2USIL": {1, 0x78, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSD2USIQ": {1, 0x78, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VCVTTSS2USIL": {1, 0x78, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
"VCVTTSS2USIQ": {1, 0x78, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||||
|
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||||
|
// vector source in vvvv, vector destination in reg).
|
||||||
|
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTUSI2SDL": {1, 0x7B, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
|
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||||
|
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||||||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||||||
// the xmm/ymm/zmm destination lengths).
|
// the xmm/ymm/zmm destination lengths).
|
||||||
@@ -466,6 +640,18 @@ var evexRound = map[string]bool{
|
|||||||
"VMINSD": true, "VMAXSD": true,
|
"VMINSD": true, "VMAXSD": true,
|
||||||
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
|
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
|
||||||
"VMINSS": true, "VMAXSS": true,
|
"VMINSS": true, "VMAXSS": true,
|
||||||
|
"VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true,
|
||||||
|
"VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true,
|
||||||
|
"VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true,
|
||||||
|
"VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true,
|
||||||
|
"VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true,
|
||||||
|
"VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true,
|
||||||
|
"VCVTSD2SI": true, "VCVTSD2SIQ": true, "VCVTSS2SI": true, "VCVTSS2SIQ": true,
|
||||||
|
"VCVTSD2USIL": true, "VCVTSD2USIQ": true, "VCVTSS2USIL": true, "VCVTSS2USIQ": true,
|
||||||
|
"VCVTTSD2SI": true, "VCVTTSD2SIQ": true, "VCVTTSS2SI": true, "VCVTTSS2SIQ": true,
|
||||||
|
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
|
||||||
|
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
|
||||||
|
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
||||||
@@ -475,6 +661,19 @@ var evexBcstN = map[string]int{
|
|||||||
"VMINPD": 8, "VMAXPD": 8,
|
"VMINPD": 8, "VMAXPD": 8,
|
||||||
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
|
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
|
||||||
"VMINPS": 4, "VMAXPS": 4,
|
"VMINPS": 4, "VMAXPS": 4,
|
||||||
|
"VRCP14PD": 8, "VRCP14PS": 4, "VRSQRT14PD": 8, "VRSQRT14PS": 4,
|
||||||
|
"VGETEXPPD": 8, "VGETEXPPS": 4,
|
||||||
|
"VSCALEFPD": 8, "VSCALEFPS": 4,
|
||||||
|
"VRNDSCALEPD": 8, "VRNDSCALEPS": 4,
|
||||||
|
"VGETMANTPD": 8, "VGETMANTPS": 4,
|
||||||
|
"VREDUCEPD": 8, "VREDUCEPS": 4,
|
||||||
|
"VFIXUPIMMPD": 8, "VFIXUPIMMPS": 4,
|
||||||
|
"VRANGEPD": 8, "VRANGEPS": 4,
|
||||||
|
"VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8,
|
||||||
|
"VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4,
|
||||||
|
"VCVTPD2PS": 8, "VCVTPD2UDQ": 8, "VCVTTPD2UDQ": 8, "VCVTTPD2UQQ": 8,
|
||||||
|
"VCVTPS2UDQ": 4, "VCVTTPS2UDQ": 4, "VCVTPS2UQQ": 4, "VCVTTPS2UQQ": 4,
|
||||||
|
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
|
||||||
}
|
}
|
||||||
|
|
||||||
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
||||||
@@ -504,6 +703,18 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
|
|||||||
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
||||||
// broadcast.
|
// broadcast.
|
||||||
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
||||||
|
// The mask/vector conversions take the K register as a genuine operand
|
||||||
|
// (source or destination), not as a mask, and accept no suffixes.
|
||||||
|
if evexKOperand[mnemUpper] {
|
||||||
|
if sfx.any() {
|
||||||
|
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
|
||||||
|
}
|
||||||
|
spec, ok := evexTable[mnemUpper]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("unsupported instruction %q", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexRM(spec, ops, 0, sfx)
|
||||||
|
}
|
||||||
spec, inTable := evexTable[mnemUpper]
|
spec, inTable := evexTable[mnemUpper]
|
||||||
if inTable {
|
if inTable {
|
||||||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||||||
@@ -547,6 +758,10 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||||
return kdst(e.encodeEvexNDS3Imm)
|
return kdst(e.encodeEvexNDS3Imm)
|
||||||
}
|
}
|
||||||
|
case vexImmRM:
|
||||||
|
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||||
|
return kdst(e.encodeEvexImmRM)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -622,21 +837,35 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuf
|
|||||||
}
|
}
|
||||||
|
|
||||||
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
||||||
// no vvvv), e.g. VCVTQQ2PD.
|
// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the
|
||||||
|
// *2M mask conversions) or a general-purpose register (the scalar
|
||||||
|
// vector-to-GPR conversions); in both cases the vector length comes from
|
||||||
|
// the source.
|
||||||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||||||
}
|
}
|
||||||
src, dst := ops[0], ops[1]
|
src, dst := ops[0], ops[1]
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok {
|
||||||
return fmt.Errorf("EVEX destination must be a vector register")
|
return fmt.Errorf("EVEX destination must be a register")
|
||||||
}
|
}
|
||||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
|
ll := dstReg.vecLenBit()
|
||||||
|
if !dstReg.isVec() {
|
||||||
|
// Mask or GPR destination: the length follows the vector source
|
||||||
|
// (128 for a memory source).
|
||||||
|
ll = 0
|
||||||
|
if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
|
ll = r.vecLenBit()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
||||||
// (reg = dst, rm = src, imm8), e.g. VPSHUFD.
|
// (reg = dst, rm = src, imm8), e.g. VPSHUFD. The destination may be an
|
||||||
|
// opmask register (VFPCLASS*), in which case the vector length comes from
|
||||||
|
// the source.
|
||||||
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
if len(ops) != 3 {
|
if len(ops) != 3 {
|
||||||
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
@@ -647,11 +876,15 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
|
|||||||
return fmt.Errorf("shuffle control must be an immediate")
|
return fmt.Errorf("shuffle control must be an immediate")
|
||||||
}
|
}
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||||||
return fmt.Errorf("shuffle destination must be a vector register")
|
return fmt.Errorf("shuffle destination must be a vector or mask register")
|
||||||
}
|
}
|
||||||
ll := dstReg.vecLenBit()
|
ll := dstReg.vecLenBit()
|
||||||
if r, ok := src.(Reg); ok && r.isVec() {
|
if dstReg.mask {
|
||||||
|
if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
|
ll = r.vecLenBit()
|
||||||
|
}
|
||||||
|
} else if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
ll = r.vecLenBit()
|
ll = r.vecLenBit()
|
||||||
}
|
}
|
||||||
immByte, err := imm8(int64(immVal))
|
immByte, err := imm8(int64(immVal))
|
||||||
@@ -1034,6 +1267,143 @@ func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte,
|
|||||||
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// gatherSpec describes a gather/scatter family member: all live in
|
||||||
|
// 66.0F38; the opcode and W select the index and data element widths, and n
|
||||||
|
// is the data element size (the EVEX disp8×N multiplier).
|
||||||
|
type gatherSpec struct {
|
||||||
|
opcode byte
|
||||||
|
w int
|
||||||
|
n int
|
||||||
|
}
|
||||||
|
|
||||||
|
var gatherTable = map[string]gatherSpec{
|
||||||
|
"VGATHERDPS": {0x92, 0, 4},
|
||||||
|
"VGATHERDPD": {0x92, 1, 8},
|
||||||
|
"VGATHERQPS": {0x93, 0, 4},
|
||||||
|
"VGATHERQPD": {0x93, 1, 8},
|
||||||
|
"VPGATHERDD": {0x90, 0, 4},
|
||||||
|
"VPGATHERDQ": {0x90, 1, 8},
|
||||||
|
"VPGATHERQD": {0x91, 0, 4},
|
||||||
|
"VPGATHERQQ": {0x91, 1, 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
var scatterTable = map[string]gatherSpec{
|
||||||
|
"VSCATTERDPS": {0xA2, 0, 4},
|
||||||
|
"VSCATTERDPD": {0xA2, 1, 8},
|
||||||
|
"VSCATTERQPS": {0xA3, 0, 4},
|
||||||
|
"VSCATTERQPD": {0xA3, 1, 8},
|
||||||
|
"VPSCATTERDD": {0xA0, 0, 4},
|
||||||
|
"VPSCATTERDQ": {0xA0, 1, 8},
|
||||||
|
"VPSCATTERQD": {0xA1, 0, 4},
|
||||||
|
"VPSCATTERQQ": {0xA1, 1, 8},
|
||||||
|
}
|
||||||
|
|
||||||
|
// isGather reports whether the mnemonic is a gather instruction.
|
||||||
|
func isGather(upper string) bool {
|
||||||
|
_, ok := gatherTable[upper]
|
||||||
|
return ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// isScatter reports whether the mnemonic is a scatter instruction.
|
||||||
|
func isScatter(upper string) bool {
|
||||||
|
_, ok := scatterTable[upper]
|
||||||
|
return ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// vsibLen validates a VSIB memory operand (the index must be a vector
|
||||||
|
// register) and returns it with the vector length the index selects — the
|
||||||
|
// EVEX L'L field follows the index register, not the data register.
|
||||||
|
func vsibLen(op Operand, what string) (Mem, int, error) {
|
||||||
|
m, ok := op.(Mem)
|
||||||
|
if !ok || !m.HasIndex || !m.Index.isVec() {
|
||||||
|
return Mem{}, 0, fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", what)
|
||||||
|
}
|
||||||
|
return m, m.Index.vecLenBit(), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeGather encodes a gather. The VEX spelling carries the mask in a
|
||||||
|
// vector register (OP mask, vsib, dst: vvvv = mask, rm = vsib, reg = dst,
|
||||||
|
// L follows the data register); the EVEX spelling carries it in aaa (OP
|
||||||
|
// vsib, K, dst: rm = vsib, reg = dst, L follows the VSIB index).
|
||||||
|
func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexSuffix) error {
|
||||||
|
rest, mask, err := splitMask(ops)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if mask != 0 || sfx.any() {
|
||||||
|
// EVEX form: OP vsib, K, dst.
|
||||||
|
if len(rest) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
vsib, ll, err := vsibLen(rest[0], upper)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dst, ok := rest[1].(Reg)
|
||||||
|
if !ok || !dst.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||||
|
}
|
||||||
|
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
||||||
|
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
||||||
|
}
|
||||||
|
// VEX form: OP mask, vsib, dst.
|
||||||
|
if len(rest) != 3 {
|
||||||
|
return fmt.Errorf("%s expects 3 operands (mask, vsib, dst), got %d", upper, len(rest))
|
||||||
|
}
|
||||||
|
maskReg, ok := rest[0].(Reg)
|
||||||
|
if !ok || !maskReg.isVec() {
|
||||||
|
return fmt.Errorf("%s: mask must be a vector register", upper)
|
||||||
|
}
|
||||||
|
vsib, _, err := vsibLen(rest[1], upper)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dst, ok := rest[2].(Reg)
|
||||||
|
if !ok || !dst.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||||
|
}
|
||||||
|
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
|
||||||
|
rBit := 0
|
||||||
|
if dst.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src,
|
||||||
|
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
|
||||||
|
// index.
|
||||||
|
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
|
||||||
|
rest, mask, err := splitMask(ops)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if mask == 0 {
|
||||||
|
return fmt.Errorf("%s requires a K mask register", upper)
|
||||||
|
}
|
||||||
|
if len(rest) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 3 operands (src, K, vsib), got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
src, ok := rest[0].(Reg)
|
||||||
|
if !ok || !src.isVec() {
|
||||||
|
return fmt.Errorf("%s: source must be a vector register", upper)
|
||||||
|
}
|
||||||
|
vsib, ll, err := vsibLen(rest[1], upper)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
||||||
|
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexKOperand lists the instructions whose K register is a genuine operand
|
||||||
|
// (the source or destination of a mask/vector conversion) rather than a
|
||||||
|
// mask modifier — the M2 and 2M conversions. They take no masking.
|
||||||
|
var evexKOperand = map[string]bool{
|
||||||
|
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
||||||
|
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
||||||
|
}
|
||||||
|
|
||||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||||
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
||||||
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
||||||
|
|||||||
@@ -364,6 +364,267 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
|
||||||
|
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
|
||||||
|
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
|
||||||
|
// plus gather/scatter with VSIB addressing, byte for byte against the Go
|
||||||
|
// assembler.
|
||||||
|
func TestEvexHelperGroundTruth(t *testing.T) {
|
||||||
|
vsib := func(base, idx string, scale int) Operand {
|
||||||
|
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// Reciprocals and rsqrt (packed RM, scalar NDS).
|
||||||
|
{"VRCP14PD", "VRCP14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484cd1"},
|
||||||
|
{"VRCP14PS", "VRCP14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484cd1"},
|
||||||
|
{"VRCP14SD", "VRCP14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084dd9"},
|
||||||
|
{"VRCP14SS", "VRCP14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084dd9"},
|
||||||
|
{"VRSQRT14PD", "VRSQRT14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484ed1"},
|
||||||
|
{"VRSQRT14PS", "VRSQRT14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484ed1"},
|
||||||
|
{"VRSQRT14SD", "VRSQRT14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084fd9"},
|
||||||
|
{"VRSQRT14SS", "VRSQRT14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084fd9"},
|
||||||
|
// Getexp (packed RM, scalar NDS).
|
||||||
|
{"VGETEXPPD", "VGETEXPPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd4842d1"},
|
||||||
|
{"VGETEXPPS", "VGETEXPPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d4842d1"},
|
||||||
|
{"VGETEXPSD", "VGETEXPSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed0843d9"},
|
||||||
|
{"VGETEXPSS", "VGETEXPSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d0843d9"},
|
||||||
|
// Scalef (NDS).
|
||||||
|
{"VSCALEFPD", "VSCALEFPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed482cd9"},
|
||||||
|
{"VSCALEFPS", "VSCALEFPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482cd9"},
|
||||||
|
{"VSCALEFSD", "VSCALEFSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed082dd9"},
|
||||||
|
{"VSCALEFSS", "VSCALEFSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d082dd9"},
|
||||||
|
// Rndscale / getmant / reduce (packed $imm,src,dst; scalar NDS+imm).
|
||||||
|
{"VRNDSCALEPD", "VRNDSCALEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4809d104"},
|
||||||
|
{"VRNDSCALEPS", "VRNDSCALEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4808d104"},
|
||||||
|
{"VRNDSCALESD", "VRNDSCALESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed080bd904"},
|
||||||
|
{"VRNDSCALESS", "VRNDSCALESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d080ad904"},
|
||||||
|
{"VGETMANTPD", "VGETMANTPD", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4826d103"},
|
||||||
|
{"VGETMANTPS", "VGETMANTPS", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4826d103"},
|
||||||
|
{"VGETMANTSD", "VGETMANTSD", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0827d903"},
|
||||||
|
{"VGETMANTSS", "VGETMANTSS", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0827d903"},
|
||||||
|
{"VREDUCEPD", "VREDUCEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4856d104"},
|
||||||
|
{"VREDUCEPS", "VREDUCEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4856d104"},
|
||||||
|
{"VREDUCESD", "VREDUCESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0857d904"},
|
||||||
|
{"VREDUCESS", "VREDUCESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0857d904"},
|
||||||
|
// Fixupimm / range (NDS + imm8).
|
||||||
|
{"VFIXUPIMMPD", "VFIXUPIMMPD", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4854d902"},
|
||||||
|
{"VFIXUPIMMPS", "VFIXUPIMMPS", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4854d902"},
|
||||||
|
{"VFIXUPIMMSD", "VFIXUPIMMSD", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0855d902"},
|
||||||
|
{"VFIXUPIMMSS", "VFIXUPIMMSS", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0855d902"},
|
||||||
|
{"VRANGEPD", "VRANGEPD", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4850d901"},
|
||||||
|
{"VRANGEPS", "VRANGEPS", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4850d901"},
|
||||||
|
{"VRANGESD", "VRANGESD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0851d901"},
|
||||||
|
{"VRANGESS", "VRANGESS", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0851d901"},
|
||||||
|
// FP class test ($imm, src, kdst; packed forms carry the length in
|
||||||
|
// the X/Y/Z mnemonic suffix the decoder drops).
|
||||||
|
{"VFPCLASSPDZ", "VFPCLASSPDZ", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2")}, "62f3fd4866d104"},
|
||||||
|
{"VFPCLASSPSY", "VFPCLASSPSY", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "K2")}, "62f37d2866d104"},
|
||||||
|
{"VFPCLASSSD", "VFPCLASSSD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f3fd0867d104"},
|
||||||
|
{"VFPCLASSSS", "VFPCLASSSS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f37d0867d104"},
|
||||||
|
// Gather: VEX spelling (mask register, VSIB, destination) and EVEX
|
||||||
|
// spelling (VSIB, K mask, destination; L'L follows the VSIB index).
|
||||||
|
{"VGATHERDPS vex", "VGATHERDPS", []Operand{vreg(t, "X2"), vsib("SI", "X1", 4), vreg(t, "X3")}, "c4e269921c8e"},
|
||||||
|
{"VPGATHERDD vex", "VPGATHERDD", []Operand{vreg(t, "Y2"), vsib("SI", "Y1", 4), vreg(t, "Y3")}, "c4e26d901c8e"},
|
||||||
|
{"VGATHERDPS evex", "VGATHERDPS", []Operand{vsib("SI", "X1", 4), vreg(t, "K2"), vreg(t, "X3")}, "62f27d0a921c8e"},
|
||||||
|
{"VPGATHERQD evex", "VPGATHERQD", []Operand{vsib("SI", "Z1", 8), vreg(t, "K2"), vreg(t, "Y3")}, "62f27d4a911cce"},
|
||||||
|
// Scatter (EVEX only: source, K mask, VSIB).
|
||||||
|
{"VSCATTERDPS", "VSCATTERDPS", []Operand{vreg(t, "X3"), vreg(t, "K1"), vsib("SI", "X1", 4)}, "62f27d09a21c8e"},
|
||||||
|
{"VSCATTERQPD", "VSCATTERQPD", []Operand{vreg(t, "Z3"), vreg(t, "K1"), vsib("SI", "Z1", 8)}, "62f2fd49a31cce"},
|
||||||
|
// The remaining conversions.
|
||||||
|
{"VCVTDQ2PS", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"},
|
||||||
|
{"VCVTQQ2PS", "VCVTQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc485bd1"},
|
||||||
|
{"VCVTPD2QQ", "VCVTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487bd1"},
|
||||||
|
{"VCVTPS2QQ", "VCVTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487bd1"},
|
||||||
|
{"VCVTUDQ2PD", "VCVTUDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "62f17e287ad1"},
|
||||||
|
{"VCVTPH2PS", "VCVTPH2PS", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f27d4813d1"},
|
||||||
|
{"VCVTPS2PH", "VCVTPS2PH", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "X2")}, "c4e37d1dca04"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||||
|
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
||||||
|
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||||
|
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
||||||
|
// for byte against the Go assembler, including memory sources and extended
|
||||||
|
// GPRs.
|
||||||
|
func TestEvexGprGroundTruth(t *testing.T) {
|
||||||
|
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
|
||||||
|
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
|
||||||
|
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
|
||||||
|
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
|
||||||
|
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
|
||||||
|
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
|
||||||
|
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
|
||||||
|
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
|
||||||
|
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
|
||||||
|
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
|
||||||
|
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
|
||||||
|
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
|
||||||
|
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
|
||||||
|
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
|
||||||
|
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
|
||||||
|
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
|
||||||
|
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
|
||||||
|
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
|
||||||
|
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
|
||||||
|
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
|
||||||
|
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
|
||||||
|
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
|
||||||
|
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
|
||||||
|
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
|
||||||
|
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
|
||||||
|
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
|
||||||
|
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
|
||||||
|
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// The decoder does not distinguish the Plan 9 SIQ spelling (the
|
||||||
|
// 64-bit GPR destination) from the base name; the W bit carries it.
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
|
||||||
|
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
|
||||||
|
// narrowing stores and the mask/vector conversions, byte for byte against
|
||||||
|
// the Go assembler.
|
||||||
|
func TestEvexConversionGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// Unsigned and truncating conversions.
|
||||||
|
{"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"},
|
||||||
|
{"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"},
|
||||||
|
{"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"},
|
||||||
|
{"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"},
|
||||||
|
{"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"},
|
||||||
|
{"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"},
|
||||||
|
{"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"},
|
||||||
|
{"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"},
|
||||||
|
{"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"},
|
||||||
|
{"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"},
|
||||||
|
{"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"},
|
||||||
|
{"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"},
|
||||||
|
{"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"},
|
||||||
|
{"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"},
|
||||||
|
{"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"},
|
||||||
|
{"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"},
|
||||||
|
{"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"},
|
||||||
|
{"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"},
|
||||||
|
{"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"},
|
||||||
|
// The remaining sign/zero-extending moves.
|
||||||
|
{"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"},
|
||||||
|
{"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"},
|
||||||
|
{"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"},
|
||||||
|
{"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"},
|
||||||
|
{"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"},
|
||||||
|
{"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"},
|
||||||
|
{"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"},
|
||||||
|
{"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"},
|
||||||
|
// Signed narrowing stores.
|
||||||
|
{"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"},
|
||||||
|
{"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"},
|
||||||
|
{"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"},
|
||||||
|
{"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"},
|
||||||
|
{"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"},
|
||||||
|
{"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"},
|
||||||
|
// Unsigned narrowing stores.
|
||||||
|
{"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"},
|
||||||
|
{"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"},
|
||||||
|
{"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"},
|
||||||
|
{"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"},
|
||||||
|
{"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"},
|
||||||
|
{"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"},
|
||||||
|
{"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"},
|
||||||
|
{"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"},
|
||||||
|
// Mask/vector conversions (the K register is an operand, not a
|
||||||
|
// mask).
|
||||||
|
{"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"},
|
||||||
|
{"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"},
|
||||||
|
{"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"},
|
||||||
|
{"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"},
|
||||||
|
{"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"},
|
||||||
|
{"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"},
|
||||||
|
{"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"},
|
||||||
|
{"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
inst, err := x86asm.Decode(code, 64)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
want := c.mnem
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestEvexErrors checks the EVEX-specific error paths.
|
// TestEvexErrors checks the EVEX-specific error paths.
|
||||||
func TestEvexErrors(t *testing.T) {
|
func TestEvexErrors(t *testing.T) {
|
||||||
cases := []struct {
|
cases := []struct {
|
||||||
|
|||||||
+38
-1
@@ -130,9 +130,15 @@ var vexTable = map[string]vexSpec{
|
|||||||
// no vvvv).
|
// no vvvv).
|
||||||
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
|
||||||
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
|
||||||
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM},
|
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM},
|
||||||
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM},
|
||||||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM},
|
||||||
|
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
|
||||||
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
|
||||||
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
|
||||||
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
|
||||||
@@ -178,6 +184,9 @@ var vexTable = map[string]vexSpec{
|
|||||||
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
|
||||||
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
|
||||||
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
|
||||||
|
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
|
||||||
|
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
|
||||||
|
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
|
||||||
|
|
||||||
// VEX.128.0F.W0 — no operands.
|
// VEX.128.0F.W0 — no operands.
|
||||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||||
@@ -189,9 +198,35 @@ var vexTable = map[string]vexSpec{
|
|||||||
// rm=scalar memory; SD is 256-bit only).
|
// rm=scalar memory; SD is 256-bit only).
|
||||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||||
|
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
|
||||||
|
// source).
|
||||||
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||||
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
|
||||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
|
||||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
|
||||||
|
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
|
||||||
|
// spellings: the destination is always XMM and the spelling fixes the
|
||||||
|
// source length (X = 128, Y = 256).
|
||||||
|
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||||
|
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||||
|
|
||||||
|
// VEX scalar conversions between vector and general-purpose registers.
|
||||||
|
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
||||||
|
// unused; the length follows the source).
|
||||||
|
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
||||||
|
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
||||||
|
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
||||||
|
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
||||||
|
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
||||||
|
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
||||||
|
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
||||||
|
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
||||||
|
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||||
|
// vector source in vvvv, vector destination in reg).
|
||||||
|
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
||||||
|
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
||||||
|
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
||||||
|
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
||||||
|
|
||||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||||
@@ -216,6 +251,8 @@ var vexSrcLen = map[string]int{
|
|||||||
"VCVTPD2DQY": 1,
|
"VCVTPD2DQY": 1,
|
||||||
"VCVTTPD2DQX": 0,
|
"VCVTTPD2DQX": 0,
|
||||||
"VCVTTPD2DQY": 1,
|
"VCVTTPD2DQY": 1,
|
||||||
|
"VCVTPD2PSX": 0,
|
||||||
|
"VCVTPD2PSY": 1,
|
||||||
}
|
}
|
||||||
|
|
||||||
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
|
||||||
|
|||||||
+6
-3
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
|
|||||||
}
|
}
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
|
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if inst.Op.String() != mnem {
|
// The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
|
||||||
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
|
// SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
|
||||||
|
got := inst.Op.String()
|
||||||
|
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
|
||||||
|
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -28,7 +28,7 @@ import (
|
|||||||
|
|
||||||
// version is the release version, stamped at build time via
|
// version is the release version, stamped at build time via
|
||||||
// -ldflags "-X main.version=…" (defaulting to the current release).
|
// -ldflags "-X main.version=…" (defaulting to the current release).
|
||||||
var version = "0.13.0"
|
var version = "0.16.0"
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if len(os.Args) < 2 {
|
if len(os.Args) < 2 {
|
||||||
|
|||||||
+11
-1
@@ -237,7 +237,17 @@ operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts
|
|||||||
and extracts, compares with an opmask destination, the permutes, the
|
and extracts, compares with an opmask destination, the permutes, the
|
||||||
expand/compress family, the broadcasts, the opmask-register instructions
|
expand/compress family, the broadcasts, the opmask-register instructions
|
||||||
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
|
(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned
|
||||||
moves and the remaining extending/narrowing moves. The EVEX mnemonic
|
moves and the remaining extending/narrowing moves, the floating-point
|
||||||
|
helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*,
|
||||||
|
VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an
|
||||||
|
opmask destination, and the VCVT* conversions — signed, unsigned and
|
||||||
|
truncating, including the length-suffixed X/Y spellings and the
|
||||||
|
mask/vector conversions VPMOVM2*/VPMOV*2M, and the scalar conversions
|
||||||
|
between vector and general-purpose registers (VCVT{,T}S{D,S}2SI{,Q} and
|
||||||
|
the unsigned forms, VCVTSI2*/VCVTUSI2*), and gather/scatter with VSIB addressing — both the
|
||||||
|
VEX spelling with a vector mask register and the EVEX spelling with an
|
||||||
|
explicit K mask, where the EVEX length follows the VSIB index register,
|
||||||
|
not the data register. The EVEX mnemonic
|
||||||
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
|
suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE),
|
||||||
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
|
suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX
|
||||||
b bit and the L'L rounding-control field (broadcast keeps the vector length
|
b bit and the L'L rounding-control field (broadcast keeps the vector length
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
# Deferred decisions
|
||||||
|
|
||||||
|
Design decisions deliberately postponed, with enough context to pick them up
|
||||||
|
again without re-deriving the analysis. Each entry records what is deferred,
|
||||||
|
why, the options on the table, and the trigger that should reopen it.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## GOOBJ external (cross-package) symbol references
|
||||||
|
|
||||||
|
**Status:** deferred (v0.15.0, 2026-08-02). The GOOBJ emitter resolves only
|
||||||
|
symbols defined in the file being assembled; a reference to any other symbol
|
||||||
|
is rejected.
|
||||||
|
|
||||||
|
**Why it is deferred.** GOOBJ symbol references are *positional*: a
|
||||||
|
reference is a `{PkgIdx, SymIdx}` pair, where `SymIdx` is the index of the
|
||||||
|
symbol in the *referenced package's* symbol-definition table. That ordering
|
||||||
|
is not derivable from the reference site — it lives in the referenced
|
||||||
|
package's gc export data (the iexport binary format, which evolves with the
|
||||||
|
toolchain). `cmd/asm` reads it with `cmd/internal` readers gasm cannot
|
||||||
|
import, so emitting external references means either parsing export data
|
||||||
|
ourselves or taking a dependency that does.
|
||||||
|
|
||||||
|
**What works today.** Single-package objects: every symbol the file defines
|
||||||
|
(as `TEXT` or `GLOBL`, static or exported) and every reference to them.
|
||||||
|
This covers the production use case — the go-flac / go-lz4 kernels carry no
|
||||||
|
`FUNCDATA`/`PCDATA`, hence no references into `runtime`, and the Go side
|
||||||
|
references the assembly symbols, never the reverse. Such a package builds
|
||||||
|
with its assembly object replaced by a gasm-emitted one.
|
||||||
|
|
||||||
|
**The options, when we return.**
|
||||||
|
|
||||||
|
1. **`golang.org/x/tools/go/gcexportdata` as a production dependency.**
|
||||||
|
The straightforward path: read each imported package's export file
|
||||||
|
(paths from `-importcfg` or `go list -export`), assign symbol indices in
|
||||||
|
its symbol order, write `PkgIndex`/`Autolib` entries (fingerprints from
|
||||||
|
the export files' build IDs) and positional references. Robust across
|
||||||
|
toolchain versions — `x/tools` tracks the format. **Cost:** the first
|
||||||
|
production dependency beyond the standard library, an explicit deviation
|
||||||
|
from the "production code depends only on the standard library"
|
||||||
|
principle in the README. Requires the user's explicit agreement.
|
||||||
|
2. **A minimal iexport parser of our own.** Preserves self-containment.
|
||||||
|
Substantial effort and inherently fragile: the format is an internal
|
||||||
|
contract that changes with Go releases, so the parser needs a
|
||||||
|
version-gated fallback and regression tests against several toolchains.
|
||||||
|
3. **Shell out to the toolchain for symbol metadata.** Consistent with the
|
||||||
|
existing GOOBJ preamble probe (which already runs `go tool asm`), but no
|
||||||
|
toolchain command exposes a package's symbols *in definition-index
|
||||||
|
order* — `go tool nm` sorts differently — so this does not solve the
|
||||||
|
core problem on its own; it would only feed option 1 or 2.
|
||||||
|
|
||||||
|
**Trigger to reopen.** An assembly file that needs a cross-package
|
||||||
|
reference — in practice `FUNCDATA $…, runtime·…(SB)` (stack maps / GC
|
||||||
|
metadata written in assembly), or any kernel that calls into another
|
||||||
|
package directly. Until then, option 3's limitation is moot and the
|
||||||
|
single-package emitter suffices.
|
||||||
Reference in New Issue
Block a user