feat(asm): add vpcmp compare, full opmask set, legacy sse integers and bswap
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m35s
Test / build (push) Successful in 41s

Assisted-by: GLM 5.3
This commit is contained in:
2026-08-27 22:41:03 +02:00
parent 163480e283
commit 19a26e049b
5 changed files with 170 additions and 32 deletions
+8 -2
View File
@@ -82,8 +82,12 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if m, ok := sseShufTable[upper]; ok { if m, ok := sseShufTable[upper]; ok {
return e.encodeSSEShuf(m, ops) return e.encodeSSEShuf(m, ops)
} }
// Legacy SSE packed binary ops and imm8 shuffles (ADDPS/MULPS/ // Legacy SSE packed binaries dispatch on the full name: the packed
// SHUFPS/PSHUFD/...): no size suffix, dispatch on the full name. // integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
// which the size split must not eat.
if m, ok := sseBinTable[upper]; ok {
return e.encodeSSEBin(m, ops)
}
if m, ok := sseBinTable[base]; ok { if m, ok := sseBinTable[base]; ok {
return e.encodeSSEBin(m, ops) return e.encodeSSEBin(m, ops)
} }
@@ -108,6 +112,8 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodePushPop(ops, false) return e.encodePushPop(ops, false)
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT": case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
return e.encodeCount(base, ops, size) return e.encodeCount(base, ops, size)
case "BSWAP":
return e.encodeBswap(ops, size)
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX": case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
return e.encodeMovExtend(base, ops) return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD": case "CVTSL2SD", "CVTSQ2SD":
+52 -3
View File
@@ -166,6 +166,19 @@ var evexTable = map[string]evexSpec{
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}}, "VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}}, "VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — integer compares with an opmask destination, the same
// NDS3Imm-with-k-reg shape as the floating-point compares; W selects the
// operand width (byte/word vs dword/qword), the opcode the signedness.
// The memory form takes a full vector, so disp8×N is 16/32/64.
"VPCMPB": {3, 0x3F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUB": {3, 0x3E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPW": {3, 0x3F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUW": {3, 0x3E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPD": {3, 0x1F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUD": {3, 0x1E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — permutes (NDS form). // EVEX.66.0F38 — permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -1471,24 +1484,60 @@ type kOpSpec struct {
var kOpsTable = map[string]kOpSpec{ var kOpsTable = map[string]kOpSpec{
// k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask // k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask
// registers). // registers). Byte/word widths share W0 and differ by the 66 prefix;
// dword/qword share the W bit selection the Go assembler emits.
"KANDB": {1, 0x41, 0, 1, 1, vexNDS3}, "KANDB": {1, 0x41, 0, 1, 1, vexNDS3},
"KANDW": {1, 0x41, 0, 0, 1, vexNDS3}, "KANDW": {1, 0x41, 0, 0, 1, vexNDS3},
"KANDD": {1, 0x41, 1, 1, 1, vexNDS3},
"KANDQ": {1, 0x41, 1, 0, 1, vexNDS3}, "KANDQ": {1, 0x41, 1, 0, 1, vexNDS3},
"KANDNB": {1, 0x42, 0, 1, 1, vexNDS3},
"KANDNW": {1, 0x42, 0, 0, 1, vexNDS3},
"KANDND": {1, 0x42, 1, 1, 1, vexNDS3},
"KANDNQ": {1, 0x42, 1, 0, 1, vexNDS3},
"KORB": {1, 0x45, 0, 1, 1, vexNDS3}, "KORB": {1, 0x45, 0, 1, 1, vexNDS3},
"KORW": {1, 0x45, 0, 0, 1, vexNDS3},
"KORD": {1, 0x45, 1, 1, 1, vexNDS3}, "KORD": {1, 0x45, 1, 1, 1, vexNDS3},
"KORQ": {1, 0x45, 1, 0, 1, vexNDS3},
"KXNORB": {1, 0x46, 0, 1, 1, vexNDS3},
"KXNORW": {1, 0x46, 0, 0, 1, vexNDS3}, "KXNORW": {1, 0x46, 0, 0, 1, vexNDS3},
"KXNORD": {1, 0x46, 1, 1, 1, vexNDS3},
"KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3}, "KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3},
"KXORB": {1, 0x47, 0, 1, 1, vexNDS3},
"KXORW": {1, 0x47, 0, 0, 1, vexNDS3},
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3}, "KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3}, "KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3}, "KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3}, "KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
"KADDD": {1, 0x4A, 1, 1, 1, vexNDS3},
"KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3}, "KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3},
// k ← OP k (KNOT) and flags ← k OP k (KORTEST): reg = dst, rm = src. // k ← OP k (KNOT), k ← k AND~ k (KTEST-style RM) and flags ← k OP k
// (KORTEST): reg = dst, rm = src.
"KNOTB": {1, 0x44, 0, 1, 0, vexRM}, "KNOTB": {1, 0x44, 0, 1, 0, vexRM},
"KNOTW": {1, 0x44, 0, 0, 0, vexRM},
"KNOTD": {1, 0x44, 1, 1, 0, vexRM},
"KNOTQ": {1, 0x44, 1, 0, 0, vexRM},
"KORTESTB": {1, 0x98, 0, 1, 0, vexRM},
"KORTESTW": {1, 0x98, 0, 0, 0, vexRM},
"KORTESTD": {1, 0x98, 1, 1, 0, vexRM}, "KORTESTD": {1, 0x98, 1, 1, 0, vexRM},
// OP $imm, src, dst: reg = dst, rm = src, imm8. "KORTESTQ": {1, 0x98, 1, 0, 0, vexRM},
"KTESTB": {1, 0x99, 0, 1, 0, vexRM},
"KTESTW": {1, 0x99, 0, 0, 0, vexRM},
"KTESTD": {1, 0x99, 1, 1, 0, vexRM},
"KTESTQ": {1, 0x99, 1, 0, 0, vexRM},
// OP $imm, src, dst: reg = dst, rm = src, imm8. The opcodes split by
// direction (0x32/0x33 left, 0x30/0x31 right) and within each by
// element half (0x32 byte/word, 0x33 dword/qword); W picks byte/dword
// (W0) against word/qword (W1).
"KSHIFTLB": {3, 0x32, 0, 1, 0, vexImmRM},
"KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM}, "KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM},
"KSHIFTLD": {3, 0x33, 0, 1, 0, vexImmRM},
"KSHIFTLQ": {3, 0x33, 1, 1, 0, vexImmRM},
"KSHIFTRB": {3, 0x30, 0, 1, 0, vexImmRM},
"KSHIFTRW": {3, 0x30, 1, 1, 0, vexImmRM},
"KSHIFTRD": {3, 0x31, 0, 1, 0, vexImmRM},
"KSHIFTRQ": {3, 0x31, 1, 1, 0, vexImmRM},
} }
// isKOp reports whether the mnemonic is an opmask-register instruction. // isKOp reports whether the mnemonic is an opmask-register instruction.
+37
View File
@@ -319,6 +319,43 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"}, {"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"}, {"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"}, {"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
// Completed opmask families (ANDN, NOT, OR/XOR word+qword, TEST,
// word-width shifts; byte-exact against go tool asm).
{"KANDNW", "KANDNW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec42d9"},
{"KANDNB", "KANDNB", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c5d542f4"},
{"KANDND", "KANDND", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed42d9"},
{"KANDNQ", "KANDNQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d442f4"},
{"KANDD", "KANDD", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed41d9"},
{"KADDD", "KADDD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d54af4"},
{"KNOTW", "KNOTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f844d1"},
{"KNOTD", "KNOTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f944e3"},
{"KNOTQ", "KNOTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f844f5"},
{"KORW", "KORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec45d9"},
{"KORQ", "KORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d445f4"},
{"KXNORB", "KXNORB", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed46d9"},
{"KXORW", "KXORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec47d9"},
{"KXORQ", "KXORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d447f4"},
{"KORTESTW", "KORTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f898d1"},
{"KORTESTB", "KORTESTB", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c5f998e3"},
{"KORTESTQ", "KORTESTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f898f5"},
{"KTESTW", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f899d1"},
{"KTESTD", "KTESTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f999e3"},
{"KSHIFTLB", "KSHIFTLB", []Operand{Imm(1), vreg(t, "K1"), vreg(t, "K2")}, "c4e37932d101"},
{"KSHIFTLD", "KSHIFTLD", []Operand{Imm(2), vreg(t, "K3"), vreg(t, "K4")}, "c4e37933e302"},
{"KSHIFTLQ", "KSHIFTLQ", []Operand{Imm(3), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f933f503"},
{"KSHIFTRB", "KSHIFTRB", []Operand{Imm(4), vreg(t, "K1"), vreg(t, "K2")}, "c4e37930d104"},
{"KSHIFTRW", "KSHIFTRW", []Operand{Imm(5), vreg(t, "K3"), vreg(t, "K4")}, "c4e3f930e305"},
{"KSHIFTRQ", "KSHIFTRQ", []Operand{Imm(6), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f931f506"},
// Integer compares with an opmask destination (0F3A map, the
// go-bzip2 partition kernel's classify instructions).
{"VPCMPUB", "VPCMPUB", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X0"), vreg(t, "K1")}, "62f37d083ec901"},
{"VPCMPB", "VPCMPB", []Operand{Imm(2), vreg(t, "Y2"), vreg(t, "Y3"), vreg(t, "K2")}, "62f365283fd202"},
{"VPCMPUW", "VPCMPUW", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f3ed483ed905"},
{"VPCMPW", "VPCMPW", []Operand{Imm(6), vreg(t, "X3"), vreg(t, "X4"), vreg(t, "K4")}, "62f3dd083fe306"},
{"VPCMPD", "VPCMPD", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K1")}, "62f36d281fc900"},
{"VPCMPUD", "VPCMPUD", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "Z3"), vreg(t, "K2")}, "62f365481ed201"},
{"VPCMPQ", "VPCMPQ", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3")}, "62f3ed081fd902"},
{"VPCMPUQ", "VPCMPUQ", []Operand{Imm(3), vreg(t, "Y3"), vreg(t, "Y4"), vreg(t, "K4")}, "62f3dd281ee303"},
// Lane extract / insert. // Lane extract / insert.
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"}, {"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"}, {"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
+71 -27
View File
@@ -48,15 +48,21 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
} }
src, dst := ops[0], ops[1] src, dst := ops[0], ops[1]
// MOVQ with an XMM operand is the SSE2 packed-quadword move, NOT a // Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2
// GPR move: mem→xmm and xmm↔xmm encode as F3 0F 7E (reg = dst), // packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E
// xmm→mem as 66 0F D6 (rm = xmm). A GPR-move fallback would // (reg = dst, no REX.W — the Go assembler's form), xmm→mem as
// silently emit REX.W 8B with the wrong operand meaning. // 66 0F D6 (rm = xmm). MOVL is the packed-dword move instead:
// 66 0F 6E load, 66 0F 7E store. A GPR-move fallback would silently
// emit REX.W 8B with the wrong operand meaning.
_, srcVec := vecReg(src) _, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst) dstReg, dstVec := vecReg(dst)
if srcVec || dstVec { if srcVec || dstVec {
if dstVec { if dstVec {
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1} i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1}
if size == 4 {
i.prefix = 0x66
i.opcode = []byte{0x0F, 0x6E}
}
if err := setRM(i, dstReg, src, 8); err != nil { if err := setRM(i, dstReg, src, 8); err != nil {
return err return err
} }
@@ -64,9 +70,12 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
} }
srcXMM, srcIsXMM := src.(Reg) srcXMM, srcIsXMM := src.(Reg)
if !srcIsXMM || !srcXMM.isVec() { if !srcIsXMM || !srcXMM.isVec() {
return fmt.Errorf("MOVQ: store needs an XMM source") return fmt.Errorf("MOV: store needs an XMM source")
} }
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1} i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if size == 4 {
i.opcode = []byte{0x0F, 0x7E}
}
if err := setRM(i, srcXMM, dst, 8); err != nil { if err := setRM(i, srcXMM, dst, 8); err != nil {
return err return err
} }
@@ -682,6 +691,21 @@ func (e *enc) encodeCount(base string, ops []Operand, size int) error {
return e.emit(i) return e.emit(i)
} }
// encodeBswap encodes BSWAP: the single register operand is encoded in the
// opcode byte (0F C8+r), with REX.B for R8-R15 and REX.W for the quad form.
func (e *enc) encodeBswap(ops []Operand, size int) error {
if len(ops) != 1 {
return fmt.Errorf("BSWAP expects 1 operand, got %d", len(ops))
}
reg, ok := ops[0].(Reg)
if !ok {
return fmt.Errorf("BSWAP operand must be a register")
}
i := newInstr(size, []byte{0x0F, 0xC8 + byte(reg.idx&7)})
i.rexB = reg.idx >= 8
return e.emit(i)
}
// --- mixed-width sign/zero-extending moves ----------------------------------- // --- mixed-width sign/zero-extending moves -----------------------------------
// movExtendOp maps Go's mixed-width move names to their opcode and destination // movExtendOp maps Go's mixed-width move names to their opcode and destination
@@ -784,33 +808,49 @@ func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
// --- legacy SSE packed binary and shuffles ----------------------------------- // --- legacy SSE packed binary and shuffles -----------------------------------
// sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an // sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an
// optional mandatory prefix plus the 0F-prefixed opcode. Plan 9 asm // optional mandatory prefix plus the 0F-prefixed opcode (0F38 for the
// lists the source operand first, so MULPS X0, X1 computes X1 = X1 * X0. // SSSE3 integer shuffles). Plan 9 asm lists the source operand first, so
// MULPS X0, X1 computes X1 = X1 * X0.
type sseBin struct { type sseBin struct {
prefix byte // 0, 0x66, 0xF2 or 0xF3 prefix byte // 0, 0x66, 0xF2 or 0xF3
op byte op byte
map38 bool // opcode lives under 0F38 instead of 0F
} }
var sseBinTable = map[string]sseBin{ var sseBinTable = map[string]sseBin{
"ADDPS": {0, 0x58}, "ADDPD": {0x66, 0x58}, "ADDPS": {0, 0x58, false}, "ADDPD": {0x66, 0x58, false},
"MULPS": {0, 0x59}, "MULPD": {0x66, 0x59}, "MULPS": {0, 0x59, false}, "MULPD": {0x66, 0x59, false},
"SUBPS": {0, 0x5C}, "SUBPD": {0x66, 0x5C}, "SUBPS": {0, 0x5C, false}, "SUBPD": {0x66, 0x5C, false},
"DIVPS": {0, 0x5E}, "DIVPD": {0x66, 0x5E}, "DIVPS": {0, 0x5E, false}, "DIVPD": {0x66, 0x5E, false},
"ANDPS": {0, 0x54}, "ANDPD": {0x66, 0x54}, "ANDPS": {0, 0x54, false}, "ANDPD": {0x66, 0x54, false},
"ORPS": {0, 0x56}, "ORPD": {0x66, 0x56}, "ORPS": {0, 0x56, false}, "ORPD": {0x66, 0x56, false},
"XORPS": {0, 0x57}, "XORPD": {0x66, 0x57}, "XORPS": {0, 0x57, false}, "XORPD": {0x66, 0x57, false},
"MINPS": {0, 0x5D}, "MINPD": {0x66, 0x5D}, "MINPS": {0, 0x5D, false}, "MINPD": {0x66, 0x5D, false},
"MAXPS": {0, 0x5F}, "MAXPD": {0x66, 0x5F}, "MAXPS": {0, 0x5F, false}, "MAXPD": {0x66, 0x5F, false},
"ADDSS": {0xF3, 0x58}, "ADDSD": {0xF2, 0x58}, "ADDSS": {0xF3, 0x58, false}, "ADDSD": {0xF2, 0x58, false},
"MULSS": {0xF3, 0x59}, "MULSD": {0xF2, 0x59}, "MULSS": {0xF3, 0x59, false}, "MULSD": {0xF2, 0x59, false},
"SUBSS": {0xF3, 0x5C}, "SUBSD": {0xF2, 0x5C}, "SUBSS": {0xF3, 0x5C, false}, "SUBSD": {0xF2, 0x5C, false},
"DIVSS": {0xF3, 0x5E}, "DIVSD": {0xF2, 0x5E}, "DIVSS": {0xF3, 0x5E, false}, "DIVSD": {0xF2, 0x5E, false},
"MINSS": {0xF3, 0x5D}, "MINSD": {0xF2, 0x5D}, "MINSS": {0xF3, 0x5D, false}, "MINSD": {0xF2, 0x5D, false},
"MAXSS": {0xF3, 0x5F}, "MAXSD": {0xF2, 0x5F}, "MAXSS": {0xF3, 0x5F, false}, "MAXSD": {0xF2, 0x5F, false},
"UNPCKLPS": {0, 0x14}, "UNPCKHPS": {0, 0x15}, "UNPCKLPS": {0, 0x14, false}, "UNPCKHPS": {0, 0x15, false},
"UNPCKLPD": {0x66, 0x14}, "UNPCKHPD": {0x66, 0x15}, "UNPCKLPD": {0x66, 0x14, false}, "UNPCKHPD": {0x66, 0x15, false},
"CVTSS2SD": {0xF3, 0x5A}, "CVTSD2SS": {0xF2, 0x5A}, "CVTSS2SD": {0xF3, 0x5A, false}, "CVTSD2SS": {0xF2, 0x5A, false},
"CVTPS2PD": {0, 0x5A}, "CVTPD2PS": {0x66, 0x5A}, "CVTPS2PD": {0, 0x5A, false}, "CVTPD2PS": {0x66, 0x5A, false},
// SSE2 packed integers (reg = reg op rm) and the SSSE3 byte shuffle.
"PXOR": {0x66, 0xEF, false},
"POR": {0x66, 0xEB, false},
"PAND": {0x66, 0xDB, false},
"PANDN": {0x66, 0xDF, false},
"PADDB": {0x66, 0xFC, false}, "PADDW": {0x66, 0xFD, false},
"PADDD": {0x66, 0xFE, false}, "PADDQ": {0x66, 0xD4, false},
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
"PCMPEQD": {0x66, 0x76, false},
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
"PCMPGTD": {0x66, 0x66, false},
"PSHUFB": {0x66, 0x00, true},
} }
// sseShuf describes a legacy SSE shuffle taking a trailing imm8 // sseShuf describes a legacy SSE shuffle taking a trailing imm8
@@ -835,7 +875,11 @@ func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
if !ok || !dstReg.isVec() { if !ok || !dstReg.isVec() {
return fmt.Errorf("SSE binary destination must be a vector register") return fmt.Errorf("SSE binary destination must be a vector register")
} }
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} opcode := []byte{0x0F, m.op}
if m.map38 {
opcode = []byte{0x0F, 0x38, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil { if err := setRM(i, dstReg, src, 8); err != nil {
return err return err
} }
+2
View File
@@ -141,6 +141,8 @@ var vexTable = map[string]vexSpec{
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM}, "VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion // VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination). // (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM}, "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},