feat(asm): add the GPR-interchanging conversions, completing the amd64 EVEX set
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
+46
-5
@@ -377,6 +377,37 @@ var evexTable = map[string]evexSpec{
|
||||
"VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX — scalar conversions between vector and general-purpose
|
||||
// registers. Vector to GPR (two operands: vec/mem source, GPR
|
||||
// destination, vvvv unused): the signed and truncated pair, and the
|
||||
// unsigned forms (EVEX only).
|
||||
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VCVTSD2USIL": {1, 0x79, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VCVTSD2USIQ": {1, 0x79, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VCVTSS2USIL": {1, 0x79, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VCVTSS2USIQ": {1, 0x79, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VCVTTSD2USIL": {1, 0x78, 0, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VCVTTSD2USIQ": {1, 0x78, 1, 3, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VCVTTSS2USIL": {1, 0x78, 0, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VCVTTSS2USIQ": {1, 0x78, 1, 2, -1, vexRM, [3]int{4, 4, 4}},
|
||||
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||
// vector source in vvvv, vector destination in reg).
|
||||
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VCVTUSI2SDL": {1, 0x7B, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
|
||||
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
|
||||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||||
// the xmm/ymm/zmm destination lengths).
|
||||
@@ -615,6 +646,12 @@ var evexRound = map[string]bool{
|
||||
"VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true,
|
||||
"VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true,
|
||||
"VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true,
|
||||
"VCVTSD2SI": true, "VCVTSD2SIQ": true, "VCVTSS2SI": true, "VCVTSS2SIQ": true,
|
||||
"VCVTSD2USIL": true, "VCVTSD2USIQ": true, "VCVTSS2USIL": true, "VCVTSS2USIQ": true,
|
||||
"VCVTTSD2SI": true, "VCVTTSD2SIQ": true, "VCVTTSS2SI": true, "VCVTTSS2SIQ": true,
|
||||
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
|
||||
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
|
||||
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
|
||||
}
|
||||
|
||||
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
||||
@@ -801,19 +838,23 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuf
|
||||
|
||||
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
||||
// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the
|
||||
// *2M mask conversions), in which case the vector length comes from the
|
||||
// source.
|
||||
// *2M mask conversions) or a general-purpose register (the scalar
|
||||
// vector-to-GPR conversions); in both cases the vector length comes from
|
||||
// the source.
|
||||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||||
return fmt.Errorf("EVEX destination must be a vector or mask register")
|
||||
if !ok {
|
||||
return fmt.Errorf("EVEX destination must be a register")
|
||||
}
|
||||
ll := dstReg.vecLenBit()
|
||||
if dstReg.mask {
|
||||
if !dstReg.isVec() {
|
||||
// Mask or GPR destination: the length follows the vector source
|
||||
// (128 for a memory source).
|
||||
ll = 0
|
||||
if r, ok := src.(Reg); ok && r.isVec() {
|
||||
ll = r.vecLenBit()
|
||||
}
|
||||
|
||||
@@ -467,6 +467,74 @@ func TestEvexHelperGroundTruth(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexGprGroundTruth covers the scalar conversions between vector and
|
||||
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
|
||||
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
|
||||
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
|
||||
// for byte against the Go assembler, including memory sources and extended
|
||||
// GPRs.
|
||||
func TestEvexGprGroundTruth(t *testing.T) {
|
||||
mem := func(b Reg) Operand { return Ptr(b, 0, 8) }
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"},
|
||||
{"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"},
|
||||
{"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"},
|
||||
{"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"},
|
||||
{"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"},
|
||||
{"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"},
|
||||
{"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"},
|
||||
{"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"},
|
||||
{"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"},
|
||||
{"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"},
|
||||
{"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"},
|
||||
{"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"},
|
||||
{"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"},
|
||||
{"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"},
|
||||
{"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"},
|
||||
{"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"},
|
||||
{"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"},
|
||||
{"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"},
|
||||
{"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"},
|
||||
{"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"},
|
||||
{"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"},
|
||||
{"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"},
|
||||
{"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"},
|
||||
{"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"},
|
||||
{"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"},
|
||||
{"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"},
|
||||
{"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"},
|
||||
{"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
// The decoder does not distinguish the Plan 9 SIQ spelling (the
|
||||
// 64-bit GPR destination) from the base name; the W bit carries it.
|
||||
want := c.mnem
|
||||
got := inst.Op.String()
|
||||
if got != want && !(len(want) > len(got) && want[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s", c.name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT*
|
||||
// conversions, the remaining sign/zero-extending moves, the signed/unsigned
|
||||
// narrowing stores and the mask/vector conversions, byte for byte against
|
||||
|
||||
+18
@@ -210,6 +210,24 @@ var vexTable = map[string]vexSpec{
|
||||
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
|
||||
|
||||
// VEX scalar conversions between vector and general-purpose registers.
|
||||
// Vector to GPR (two operands: vec/mem source, GPR destination, vvvv
|
||||
// unused; the length follows the source).
|
||||
"VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM},
|
||||
"VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM},
|
||||
"VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM},
|
||||
"VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM},
|
||||
"VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM},
|
||||
"VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM},
|
||||
"VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM},
|
||||
"VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM},
|
||||
// GPR to vector (three operands: GPR/mem source in r/m, the preserved
|
||||
// vector source in vvvv, vector destination in reg).
|
||||
"VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3},
|
||||
"VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3},
|
||||
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
|
||||
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
|
||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
|
||||
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
|
||||
|
||||
+6
-3
@@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) {
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, code, err)
|
||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != mnem {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code)
|
||||
// The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/
|
||||
// SDQ, SSL/SSQ) onto the base name; the W bit carries the width.
|
||||
got := inst.Op.String()
|
||||
if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) {
|
||||
t.Errorf("%s: decoded as %s (% x)", mnem, got, code)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user