From 19a26e049bb3845308d9958ad75aeec9dfd7330f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Thu, 27 Aug 2026 22:41:03 +0200 Subject: [PATCH] feat(asm): add vpcmp compare, full opmask set, legacy sse integers and bswap Assisted-by: GLM 5.3 --- asm/encode.go | 10 ++++- asm/evex.go | 55 +++++++++++++++++++++++++-- asm/evex_test.go | 37 ++++++++++++++++++ asm/instrs.go | 98 +++++++++++++++++++++++++++++++++++------------- asm/vex.go | 2 + 5 files changed, 170 insertions(+), 32 deletions(-) diff --git a/asm/encode.go b/asm/encode.go index 82bd9c8..1e89640 100644 --- a/asm/encode.go +++ b/asm/encode.go @@ -82,8 +82,12 @@ func (e *enc) encode(mnem string, ops []Operand) error { if m, ok := sseShufTable[upper]; ok { return e.encodeSSEShuf(m, ops) } - // Legacy SSE packed binary ops and imm8 shuffles (ADDPS/MULPS/ - // SHUFPS/PSHUFD/...): no size suffix, dispatch on the full name. + // Legacy SSE packed binaries dispatch on the full name: the packed + // integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...), + // which the size split must not eat. + if m, ok := sseBinTable[upper]; ok { + return e.encodeSSEBin(m, ops) + } if m, ok := sseBinTable[base]; ok { return e.encodeSSEBin(m, ops) } @@ -108,6 +112,8 @@ func (e *enc) encode(mnem string, ops []Operand) error { return e.encodePushPop(ops, false) case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT": return e.encodeCount(base, ops, size) + case "BSWAP": + return e.encodeBswap(ops, size) case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX": return e.encodeMovExtend(base, ops) case "CVTSL2SD", "CVTSQ2SD": diff --git a/asm/evex.go b/asm/evex.go index 9aed61b..555f82d 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -166,6 +166,19 @@ var evexTable = map[string]evexSpec{ "VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}}, "VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}}, + // EVEX.66.0F3A — integer compares with an opmask destination, the same + // NDS3Imm-with-k-reg shape as the floating-point compares; W selects the + // operand width (byte/word vs dword/qword), the opcode the signedness. + // The memory form takes a full vector, so disp8×N is 16/32/64. + "VPCMPB": {3, 0x3F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPCMPUB": {3, 0x3E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPCMPW": {3, 0x3F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPCMPUW": {3, 0x3E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPCMPD": {3, 0x1F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPCMPUD": {3, 0x1E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + // EVEX.66.0F38 — permutes (NDS form). "VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, @@ -1471,24 +1484,60 @@ type kOpSpec struct { var kOpsTable = map[string]kOpSpec{ // k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask - // registers). + // registers). Byte/word widths share W0 and differ by the 66 prefix; + // dword/qword share the W bit selection the Go assembler emits. "KANDB": {1, 0x41, 0, 1, 1, vexNDS3}, "KANDW": {1, 0x41, 0, 0, 1, vexNDS3}, + "KANDD": {1, 0x41, 1, 1, 1, vexNDS3}, "KANDQ": {1, 0x41, 1, 0, 1, vexNDS3}, + "KANDNB": {1, 0x42, 0, 1, 1, vexNDS3}, + "KANDNW": {1, 0x42, 0, 0, 1, vexNDS3}, + "KANDND": {1, 0x42, 1, 1, 1, vexNDS3}, + "KANDNQ": {1, 0x42, 1, 0, 1, vexNDS3}, "KORB": {1, 0x45, 0, 1, 1, vexNDS3}, + "KORW": {1, 0x45, 0, 0, 1, vexNDS3}, "KORD": {1, 0x45, 1, 1, 1, vexNDS3}, + "KORQ": {1, 0x45, 1, 0, 1, vexNDS3}, + "KXNORB": {1, 0x46, 0, 1, 1, vexNDS3}, "KXNORW": {1, 0x46, 0, 0, 1, vexNDS3}, + "KXNORD": {1, 0x46, 1, 1, 1, vexNDS3}, "KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3}, + "KXORB": {1, 0x47, 0, 1, 1, vexNDS3}, + "KXORW": {1, 0x47, 0, 0, 1, vexNDS3}, + "KXORD": {1, 0x47, 1, 1, 1, vexNDS3}, + "KXORQ": {1, 0x47, 1, 0, 1, vexNDS3}, "KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3}, "KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3}, "KADDB": {1, 0x4A, 0, 1, 1, vexNDS3}, "KADDW": {1, 0x4A, 0, 0, 1, vexNDS3}, + "KADDD": {1, 0x4A, 1, 1, 1, vexNDS3}, "KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3}, - // k ← OP k (KNOT) and flags ← k OP k (KORTEST): reg = dst, rm = src. + // k ← OP k (KNOT), k ← k AND~ k (KTEST-style RM) and flags ← k OP k + // (KORTEST): reg = dst, rm = src. "KNOTB": {1, 0x44, 0, 1, 0, vexRM}, + "KNOTW": {1, 0x44, 0, 0, 0, vexRM}, + "KNOTD": {1, 0x44, 1, 1, 0, vexRM}, + "KNOTQ": {1, 0x44, 1, 0, 0, vexRM}, + "KORTESTB": {1, 0x98, 0, 1, 0, vexRM}, + "KORTESTW": {1, 0x98, 0, 0, 0, vexRM}, "KORTESTD": {1, 0x98, 1, 1, 0, vexRM}, - // OP $imm, src, dst: reg = dst, rm = src, imm8. + "KORTESTQ": {1, 0x98, 1, 0, 0, vexRM}, + "KTESTB": {1, 0x99, 0, 1, 0, vexRM}, + "KTESTW": {1, 0x99, 0, 0, 0, vexRM}, + "KTESTD": {1, 0x99, 1, 1, 0, vexRM}, + "KTESTQ": {1, 0x99, 1, 0, 0, vexRM}, + // OP $imm, src, dst: reg = dst, rm = src, imm8. The opcodes split by + // direction (0x32/0x33 left, 0x30/0x31 right) and within each by + // element half (0x32 byte/word, 0x33 dword/qword); W picks byte/dword + // (W0) against word/qword (W1). + "KSHIFTLB": {3, 0x32, 0, 1, 0, vexImmRM}, "KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM}, + "KSHIFTLD": {3, 0x33, 0, 1, 0, vexImmRM}, + "KSHIFTLQ": {3, 0x33, 1, 1, 0, vexImmRM}, + "KSHIFTRB": {3, 0x30, 0, 1, 0, vexImmRM}, + "KSHIFTRW": {3, 0x30, 1, 1, 0, vexImmRM}, + "KSHIFTRD": {3, 0x31, 0, 1, 0, vexImmRM}, + "KSHIFTRQ": {3, 0x31, 1, 1, 0, vexImmRM}, } // isKOp reports whether the mnemonic is an opmask-register instruction. diff --git a/asm/evex_test.go b/asm/evex_test.go index 0c4f316..09f5334 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -319,6 +319,43 @@ func TestEvexExtendedGroundTruth(t *testing.T) { {"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"}, {"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"}, {"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"}, + // Completed opmask families (ANDN, NOT, OR/XOR word+qword, TEST, + // word-width shifts; byte-exact against go tool asm). + {"KANDNW", "KANDNW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec42d9"}, + {"KANDNB", "KANDNB", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c5d542f4"}, + {"KANDND", "KANDND", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed42d9"}, + {"KANDNQ", "KANDNQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d442f4"}, + {"KANDD", "KANDD", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed41d9"}, + {"KADDD", "KADDD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d54af4"}, + {"KNOTW", "KNOTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f844d1"}, + {"KNOTD", "KNOTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f944e3"}, + {"KNOTQ", "KNOTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f844f5"}, + {"KORW", "KORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec45d9"}, + {"KORQ", "KORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d445f4"}, + {"KXNORB", "KXNORB", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed46d9"}, + {"KXORW", "KXORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec47d9"}, + {"KXORQ", "KXORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d447f4"}, + {"KORTESTW", "KORTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f898d1"}, + {"KORTESTB", "KORTESTB", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c5f998e3"}, + {"KORTESTQ", "KORTESTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f898f5"}, + {"KTESTW", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f899d1"}, + {"KTESTD", "KTESTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f999e3"}, + {"KSHIFTLB", "KSHIFTLB", []Operand{Imm(1), vreg(t, "K1"), vreg(t, "K2")}, "c4e37932d101"}, + {"KSHIFTLD", "KSHIFTLD", []Operand{Imm(2), vreg(t, "K3"), vreg(t, "K4")}, "c4e37933e302"}, + {"KSHIFTLQ", "KSHIFTLQ", []Operand{Imm(3), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f933f503"}, + {"KSHIFTRB", "KSHIFTRB", []Operand{Imm(4), vreg(t, "K1"), vreg(t, "K2")}, "c4e37930d104"}, + {"KSHIFTRW", "KSHIFTRW", []Operand{Imm(5), vreg(t, "K3"), vreg(t, "K4")}, "c4e3f930e305"}, + {"KSHIFTRQ", "KSHIFTRQ", []Operand{Imm(6), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f931f506"}, + // Integer compares with an opmask destination (0F3A map, the + // go-bzip2 partition kernel's classify instructions). + {"VPCMPUB", "VPCMPUB", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X0"), vreg(t, "K1")}, "62f37d083ec901"}, + {"VPCMPB", "VPCMPB", []Operand{Imm(2), vreg(t, "Y2"), vreg(t, "Y3"), vreg(t, "K2")}, "62f365283fd202"}, + {"VPCMPUW", "VPCMPUW", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f3ed483ed905"}, + {"VPCMPW", "VPCMPW", []Operand{Imm(6), vreg(t, "X3"), vreg(t, "X4"), vreg(t, "K4")}, "62f3dd083fe306"}, + {"VPCMPD", "VPCMPD", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K1")}, "62f36d281fc900"}, + {"VPCMPUD", "VPCMPUD", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "Z3"), vreg(t, "K2")}, "62f365481ed201"}, + {"VPCMPQ", "VPCMPQ", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3")}, "62f3ed081fd902"}, + {"VPCMPUQ", "VPCMPUQ", []Operand{Imm(3), vreg(t, "Y3"), vreg(t, "Y4"), vreg(t, "K4")}, "62f3dd281ee303"}, // Lane extract / insert. {"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"}, {"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"}, diff --git a/asm/instrs.go b/asm/instrs.go index 5ea32c6..26256a9 100644 --- a/asm/instrs.go +++ b/asm/instrs.go @@ -48,15 +48,21 @@ func (e *enc) encodeMov(ops []Operand, size int) error { } src, dst := ops[0], ops[1] - // MOVQ with an XMM operand is the SSE2 packed-quadword move, NOT a - // GPR move: mem→xmm and xmm↔xmm encode as F3 0F 7E (reg = dst), - // xmm→mem as 66 0F D6 (rm = xmm). A GPR-move fallback would - // silently emit REX.W 8B with the wrong operand meaning. + // Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2 + // packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E + // (reg = dst, no REX.W — the Go assembler's form), xmm→mem as + // 66 0F D6 (rm = xmm). MOVL is the packed-dword move instead: + // 66 0F 6E load, 66 0F 7E store. A GPR-move fallback would silently + // emit REX.W 8B with the wrong operand meaning. _, srcVec := vecReg(src) dstReg, dstVec := vecReg(dst) if srcVec || dstVec { if dstVec { i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1} + if size == 4 { + i.prefix = 0x66 + i.opcode = []byte{0x0F, 0x6E} + } if err := setRM(i, dstReg, src, 8); err != nil { return err } @@ -64,9 +70,12 @@ func (e *enc) encodeMov(ops []Operand, size int) error { } srcXMM, srcIsXMM := src.(Reg) if !srcIsXMM || !srcXMM.isVec() { - return fmt.Errorf("MOVQ: store needs an XMM source") + return fmt.Errorf("MOV: store needs an XMM source") } i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1} + if size == 4 { + i.opcode = []byte{0x0F, 0x7E} + } if err := setRM(i, srcXMM, dst, 8); err != nil { return err } @@ -682,6 +691,21 @@ func (e *enc) encodeCount(base string, ops []Operand, size int) error { return e.emit(i) } +// encodeBswap encodes BSWAP: the single register operand is encoded in the +// opcode byte (0F C8+r), with REX.B for R8-R15 and REX.W for the quad form. +func (e *enc) encodeBswap(ops []Operand, size int) error { + if len(ops) != 1 { + return fmt.Errorf("BSWAP expects 1 operand, got %d", len(ops)) + } + reg, ok := ops[0].(Reg) + if !ok { + return fmt.Errorf("BSWAP operand must be a register") + } + i := newInstr(size, []byte{0x0F, 0xC8 + byte(reg.idx&7)}) + i.rexB = reg.idx >= 8 + return e.emit(i) +} + // --- mixed-width sign/zero-extending moves ----------------------------------- // movExtendOp maps Go's mixed-width move names to their opcode and destination @@ -784,33 +808,49 @@ func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error { // --- legacy SSE packed binary and shuffles ----------------------------------- // sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an -// optional mandatory prefix plus the 0F-prefixed opcode. Plan 9 asm -// lists the source operand first, so MULPS X0, X1 computes X1 = X1 * X0. +// optional mandatory prefix plus the 0F-prefixed opcode (0F38 for the +// SSSE3 integer shuffles). Plan 9 asm lists the source operand first, so +// MULPS X0, X1 computes X1 = X1 * X0. type sseBin struct { prefix byte // 0, 0x66, 0xF2 or 0xF3 op byte + map38 bool // opcode lives under 0F38 instead of 0F } var sseBinTable = map[string]sseBin{ - "ADDPS": {0, 0x58}, "ADDPD": {0x66, 0x58}, - "MULPS": {0, 0x59}, "MULPD": {0x66, 0x59}, - "SUBPS": {0, 0x5C}, "SUBPD": {0x66, 0x5C}, - "DIVPS": {0, 0x5E}, "DIVPD": {0x66, 0x5E}, - "ANDPS": {0, 0x54}, "ANDPD": {0x66, 0x54}, - "ORPS": {0, 0x56}, "ORPD": {0x66, 0x56}, - "XORPS": {0, 0x57}, "XORPD": {0x66, 0x57}, - "MINPS": {0, 0x5D}, "MINPD": {0x66, 0x5D}, - "MAXPS": {0, 0x5F}, "MAXPD": {0x66, 0x5F}, - "ADDSS": {0xF3, 0x58}, "ADDSD": {0xF2, 0x58}, - "MULSS": {0xF3, 0x59}, "MULSD": {0xF2, 0x59}, - "SUBSS": {0xF3, 0x5C}, "SUBSD": {0xF2, 0x5C}, - "DIVSS": {0xF3, 0x5E}, "DIVSD": {0xF2, 0x5E}, - "MINSS": {0xF3, 0x5D}, "MINSD": {0xF2, 0x5D}, - "MAXSS": {0xF3, 0x5F}, "MAXSD": {0xF2, 0x5F}, - "UNPCKLPS": {0, 0x14}, "UNPCKHPS": {0, 0x15}, - "UNPCKLPD": {0x66, 0x14}, "UNPCKHPD": {0x66, 0x15}, - "CVTSS2SD": {0xF3, 0x5A}, "CVTSD2SS": {0xF2, 0x5A}, - "CVTPS2PD": {0, 0x5A}, "CVTPD2PS": {0x66, 0x5A}, + "ADDPS": {0, 0x58, false}, "ADDPD": {0x66, 0x58, false}, + "MULPS": {0, 0x59, false}, "MULPD": {0x66, 0x59, false}, + "SUBPS": {0, 0x5C, false}, "SUBPD": {0x66, 0x5C, false}, + "DIVPS": {0, 0x5E, false}, "DIVPD": {0x66, 0x5E, false}, + "ANDPS": {0, 0x54, false}, "ANDPD": {0x66, 0x54, false}, + "ORPS": {0, 0x56, false}, "ORPD": {0x66, 0x56, false}, + "XORPS": {0, 0x57, false}, "XORPD": {0x66, 0x57, false}, + "MINPS": {0, 0x5D, false}, "MINPD": {0x66, 0x5D, false}, + "MAXPS": {0, 0x5F, false}, "MAXPD": {0x66, 0x5F, false}, + "ADDSS": {0xF3, 0x58, false}, "ADDSD": {0xF2, 0x58, false}, + "MULSS": {0xF3, 0x59, false}, "MULSD": {0xF2, 0x59, false}, + "SUBSS": {0xF3, 0x5C, false}, "SUBSD": {0xF2, 0x5C, false}, + "DIVSS": {0xF3, 0x5E, false}, "DIVSD": {0xF2, 0x5E, false}, + "MINSS": {0xF3, 0x5D, false}, "MINSD": {0xF2, 0x5D, false}, + "MAXSS": {0xF3, 0x5F, false}, "MAXSD": {0xF2, 0x5F, false}, + "UNPCKLPS": {0, 0x14, false}, "UNPCKHPS": {0, 0x15, false}, + "UNPCKLPD": {0x66, 0x14, false}, "UNPCKHPD": {0x66, 0x15, false}, + "CVTSS2SD": {0xF3, 0x5A, false}, "CVTSD2SS": {0xF2, 0x5A, false}, + "CVTPS2PD": {0, 0x5A, false}, "CVTPD2PS": {0x66, 0x5A, false}, + // SSE2 packed integers (reg = reg op rm) and the SSSE3 byte shuffle. + "PXOR": {0x66, 0xEF, false}, + "POR": {0x66, 0xEB, false}, + "PAND": {0x66, 0xDB, false}, + "PANDN": {0x66, 0xDF, false}, + "PADDB": {0x66, 0xFC, false}, "PADDW": {0x66, 0xFD, false}, + "PADDD": {0x66, 0xFE, false}, "PADDQ": {0x66, 0xD4, false}, + "PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false}, + "PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false}, + "PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false}, + "PCMPEQD": {0x66, 0x76, false}, + "PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false}, + "PCMPGTD": {0x66, 0x66, false}, + "PSHUFB": {0x66, 0x00, true}, } // sseShuf describes a legacy SSE shuffle taking a trailing imm8 @@ -835,7 +875,11 @@ func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error { if !ok || !dstReg.isVec() { return fmt.Errorf("SSE binary destination must be a vector register") } - i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} + opcode := []byte{0x0F, m.op} + if m.map38 { + opcode = []byte{0x0F, 0x38, m.op} + } + i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1} if err := setRM(i, dstReg, src, 8); err != nil { return err } diff --git a/asm/vex.go b/asm/vex.go index 283d423..009f99c 100644 --- a/asm/vex.go +++ b/asm/vex.go @@ -141,6 +141,8 @@ var vexTable = map[string]vexSpec{ "VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, + "VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM}, + "VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM}, // VEX.128/256.F3.0F.WIG — signed dword to packed double conversion // (reg=dst, rm=src, no vvvv; the length follows the destination). "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},