566 lines
19 KiB
Go
566 lines
19 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|||
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
||
|
|
|
||
|
|
package asm
|
||
|
|
|
||
|
|
import "fmt"
|
||
|
|
|
||
|
|
// This file implements the amd64 SIMD pieces the main tables lack: the MMX
|
||
|
|
// bank glue (EMMS, the masked stores, the MMX shifts and shuffle, the
|
||
|
|
// byte-mask extract), the MOVQ bank crossings' odd spellings and the leaf
|
||
|
|
// aliases. Every encoding here is pinned byte for byte against go tool asm
|
||
|
|
// through the corpus lines in amd64_sse_test.go.
|
||
|
|
|
||
|
|
// sseMaskmov maps the masked cache-line stores to their prefix and opcode:
|
||
|
|
// OP src, dst with the second operand in the reg field, no memory operand.
|
||
|
|
var sseMaskmov = map[string]sseBin{
|
||
|
|
"MASKMOVQ": {0, 0xF7, false},
|
||
|
|
"MASKMOVOU": {0x66, 0xF7, false},
|
||
|
|
}
|
||
|
|
|
||
|
|
// sseMoreBin maps the packed and scalar legacy binaries the main table
|
||
|
|
// lacks, reg = destination and r/m = source: the float comparisons, square
|
||
|
|
// roots and reciprocal estimates, the SSE3 horizontal arithmetic, the SSE4.1
|
||
|
|
// packed integers and the SSSE3 sign and horizontal ops. The MMX twins of
|
||
|
|
// the 0x66-prefixed members drop the prefix in the shared encoder.
|
||
|
|
var sseMoreBin = map[string]sseBin{
|
||
|
|
"COMISS": {0, 0x2F, false},
|
||
|
|
"UCOMISS": {0, 0x2E, false},
|
||
|
|
"UCOMISD": {0x66, 0x2E, false},
|
||
|
|
"SQRTPS": {0, 0x51, false},
|
||
|
|
"SQRTPD": {0x66, 0x51, false},
|
||
|
|
"SQRTSS": {0xF3, 0x51, false},
|
||
|
|
"RCPPS": {0, 0x53, false},
|
||
|
|
"RCPSS": {0xF3, 0x53, false},
|
||
|
|
"RSQRTPS": {0, 0x52, false},
|
||
|
|
"RSQRTSS": {0xF3, 0x52, false},
|
||
|
|
"ADDSUBPD": {0x66, 0xD0, false},
|
||
|
|
"ADDSUBPS": {0xF2, 0xD0, false},
|
||
|
|
"HADDPD": {0x66, 0x7C, false},
|
||
|
|
"HADDPS": {0xF2, 0x7C, false},
|
||
|
|
"HSUBPD": {0x66, 0x7D, false},
|
||
|
|
"HSUBPS": {0xF2, 0x7D, false},
|
||
|
|
"MOVDDUP": {0xF2, 0x12, false},
|
||
|
|
"MOVSHDUP": {0xF3, 0x16, false},
|
||
|
|
"MOVSLDUP": {0xF3, 0x12, false},
|
||
|
|
"LDDQU": {0xF2, 0xF0, false},
|
||
|
|
"MOVNTDQA": {0x66, 0x2A, true},
|
||
|
|
"PTEST": {0x66, 0x17, true},
|
||
|
|
"PABSB": {0x66, 0x1C, true},
|
||
|
|
"PABSW": {0x66, 0x1D, true},
|
||
|
|
"PABSD": {0x66, 0x1E, true},
|
||
|
|
"PACKSSWB": {0x66, 0x63, false},
|
||
|
|
"PACKUSWB": {0x66, 0x67, false},
|
||
|
|
"PACKSSLW": {0x66, 0x6B, false},
|
||
|
|
"PACKUSDW": {0x66, 0x2B, true},
|
||
|
|
"PADDSB": {0x66, 0xEC, false},
|
||
|
|
"PADDSW": {0x66, 0xED, false},
|
||
|
|
"PADDUSB": {0x66, 0xDC, false},
|
||
|
|
"PADDUSW": {0x66, 0xDD, false},
|
||
|
|
"PAVGB": {0x66, 0xE0, false},
|
||
|
|
"PAVGW": {0x66, 0xE3, false},
|
||
|
|
"PCMPEQQ": {0x66, 0x29, true},
|
||
|
|
"PCMPGTQ": {0x66, 0x37, true},
|
||
|
|
"PHADDW": {0x66, 0x01, true},
|
||
|
|
"PHADDD": {0x66, 0x02, true},
|
||
|
|
"PHADDSW": {0x66, 0x03, true},
|
||
|
|
"PHSUBW": {0x66, 0x05, true},
|
||
|
|
"PHSUBD": {0x66, 0x06, true},
|
||
|
|
"PHSUBSW": {0x66, 0x07, true},
|
||
|
|
"PHMINPOSUW": {0x66, 0x41, true},
|
||
|
|
"PMADDUBSW": {0x66, 0x04, true},
|
||
|
|
"PMADDWL": {0x66, 0xF5, false},
|
||
|
|
"PMAXSB": {0x66, 0x3C, true},
|
||
|
|
"PMAXSD": {0x66, 0x3D, true},
|
||
|
|
"PMAXSW": {0x66, 0xEE, false},
|
||
|
|
"PMAXUB": {0x66, 0xDE, false},
|
||
|
|
"PMAXUD": {0x66, 0x3F, true},
|
||
|
|
"PMAXUW": {0x66, 0x3E, true},
|
||
|
|
"PMINSB": {0x66, 0x38, true},
|
||
|
|
"PMINSD": {0x66, 0x39, true},
|
||
|
|
"PMINSW": {0x66, 0xEA, false},
|
||
|
|
"PMINUB": {0x66, 0xDA, false},
|
||
|
|
"PMINUD": {0x66, 0x3B, true},
|
||
|
|
"PMINUW": {0x66, 0x3A, true},
|
||
|
|
"PMULDQ": {0x66, 0x28, true},
|
||
|
|
"PMULLD": {0x66, 0x40, true},
|
||
|
|
"PMULHRSW": {0x66, 0x0B, true},
|
||
|
|
"PMULHUW": {0x66, 0xE4, false},
|
||
|
|
"PMULHW": {0x66, 0xE5, false},
|
||
|
|
"PMULLW": {0x66, 0xD5, false},
|
||
|
|
"PMULULQ": {0x66, 0xF4, false},
|
||
|
|
"PCMPGTL": {0x66, 0x66, false},
|
||
|
|
"PMOVSXBD": {0x66, 0x21, true},
|
||
|
|
"PMOVSXBQ": {0x66, 0x22, true},
|
||
|
|
"PMOVSXBW": {0x66, 0x20, true},
|
||
|
|
"PMOVSXDQ": {0x66, 0x25, true},
|
||
|
|
"PMOVSXWD": {0x66, 0x23, true},
|
||
|
|
"PMOVSXWQ": {0x66, 0x24, true},
|
||
|
|
"PMOVZXBD": {0x66, 0x31, true},
|
||
|
|
"PMOVZXBQ": {0x66, 0x32, true},
|
||
|
|
"PMOVZXBW": {0x66, 0x30, true},
|
||
|
|
"PMOVZXDQ": {0x66, 0x35, true},
|
||
|
|
"PMOVZXWD": {0x66, 0x33, true},
|
||
|
|
"PMOVZXWQ": {0x66, 0x34, true},
|
||
|
|
// The conversion aliases the Plan 9 table spells with an L: the dword
|
||
|
|
// sources and destinations of the packed integer/float converts.
|
||
|
|
"CVTPL2PD": {0xF3, 0xE6, false},
|
||
|
|
"CVTPL2PS": {0, 0x5B, false},
|
||
|
|
"CVTPD2PL": {0xF2, 0xE6, false},
|
||
|
|
"CVTPS2PL": {0x66, 0x5B, false},
|
||
|
|
"CVTTPD2PL": {0x66, 0xE6, false},
|
||
|
|
"CVTTPS2PL": {0xF3, 0x5B, false},
|
||
|
|
"PSADBW": {0x66, 0xF6, false},
|
||
|
|
"PSUBSB": {0x66, 0xE8, false},
|
||
|
|
"PSUBSW": {0x66, 0xE9, false},
|
||
|
|
"PSUBUSB": {0x66, 0xD8, false},
|
||
|
|
"PSUBUSW": {0x66, 0xD9, false},
|
||
|
|
"PSIGNB": {0x66, 0x08, true},
|
||
|
|
"PSIGNW": {0x66, 0x09, true},
|
||
|
|
"PSIGND": {0x66, 0x0A, true},
|
||
|
|
"PUNPCKHBW": {0x66, 0x68, false},
|
||
|
|
"PUNPCKHLQ": {0x66, 0x6A, false},
|
||
|
|
"PUNPCKHQDQ": {0x66, 0x6D, false},
|
||
|
|
"PUNPCKHWL": {0x66, 0x69, false},
|
||
|
|
"PUNPCKLLQ": {0x66, 0x62, false},
|
||
|
|
"PUNPCKLQDQ": {0x66, 0x6C, false},
|
||
|
|
"PUNPCKLWL": {0x66, 0x61, false},
|
||
|
|
}
|
||
|
|
|
||
|
|
// sseMoreImm3 maps the imm8-controlled three-operand instructions the main
|
||
|
|
// table lacks: OP $imm, src, dst.
|
||
|
|
var sseMoreImm3 = map[string]sseImm3{
|
||
|
|
"ROUNDPS": {0x66, 0x08, true},
|
||
|
|
"ROUNDPD": {0x66, 0x09, true},
|
||
|
|
"ROUNDSS": {0x66, 0x0A, true},
|
||
|
|
"ROUNDSD": {0x66, 0x0B, true},
|
||
|
|
"DPPS": {0x66, 0x40, true},
|
||
|
|
"DPPD": {0x66, 0x41, true},
|
||
|
|
"BLENDPS": {0x66, 0x0C, true},
|
||
|
|
"BLENDPD": {0x66, 0x0D, true},
|
||
|
|
"INSERTPS": {0x66, 0x21, true},
|
||
|
|
"MPSADBW": {0x66, 0x42, true},
|
||
|
|
"PCMPESTRM": {0x66, 0x60, true},
|
||
|
|
"PCMPESTRI": {0x66, 0x61, true},
|
||
|
|
"PCMPISTRM": {0x66, 0x62, true},
|
||
|
|
"PCMPISTRI": {0x66, 0x63, true},
|
||
|
|
}
|
||
|
|
|
||
|
|
// sseBlendv maps the variable blends whose implicit mask is X0: the first
|
||
|
|
// operand must be the literal X0, the register the hardware reads.
|
||
|
|
var sseBlendv = map[string]sseBin{
|
||
|
|
"BLENDVPS": {0x66, 0x14, true},
|
||
|
|
"BLENDVPD": {0x66, 0x15, true},
|
||
|
|
"PBLENDVB": {0x66, 0x10, true},
|
||
|
|
}
|
||
|
|
|
||
|
|
// sseHighLow maps the high/low half moves to their load/store opcode pair.
|
||
|
|
// A memory source loads (reg = destination), a memory destination stores
|
||
|
|
// (reg = the register source).
|
||
|
|
var sseHighLow = map[string]sseMove{
|
||
|
|
"MOVHPD": {0x66, 0x16, 0x17},
|
||
|
|
"MOVHPS": {0, 0x16, 0x17},
|
||
|
|
"MOVLPD": {0x66, 0x12, 0x13},
|
||
|
|
"MOVLPS": {0, 0x12, 0x13},
|
||
|
|
}
|
||
|
|
|
||
|
|
// sseRegReg maps the register-to-register half moves, register destination
|
||
|
|
// and register source alone: MOVHLPS and MOVLHPS.
|
||
|
|
var sseRegReg = map[string]sseBin{
|
||
|
|
"MOVHLPS": {0, 0x12, false},
|
||
|
|
"MOVLHPS": {0, 0x16, false},
|
||
|
|
}
|
||
|
|
|
||
|
|
// sseMovmsk maps the sign-mask extractions to a GPR: OP vec, gpr.
|
||
|
|
var sseMovmsk = map[string]sseBin{
|
||
|
|
"MOVMSKPS": {0, 0x50, false},
|
||
|
|
"MOVMSKPD": {0x66, 0x50, false},
|
||
|
|
}
|
||
|
|
|
||
|
|
// sseMovnt maps the non-temporal stores, OP reg, mem, plus MOVNTDQA's load
|
||
|
|
// (which rides sseMoreBin).
|
||
|
|
var sseMovnt = map[string]sseBin{
|
||
|
|
"MOVNTPS": {0, 0x2B, false},
|
||
|
|
"MOVNTPD": {0x66, 0x2B, false},
|
||
|
|
"MOVNTQ": {0, 0xE7, false},
|
||
|
|
"MOVNTO": {0x66, 0xE7, false},
|
||
|
|
"MOVNTIL": {0, 0xC3, false},
|
||
|
|
"MOVNTIQ": {0, 0xC3, false},
|
||
|
|
}
|
||
|
|
|
||
|
|
// mmxShiftImm lists the packed integer shifts whose immediate form the MMX
|
||
|
|
// bank spells without the 0x66 prefix; the digit rides the 0F 71/72/73
|
||
|
|
// group, the same /digits the XMM forms carry.
|
||
|
|
var mmxShiftImm = map[string]sseShift{
|
||
|
|
"PSLLW": {0x71, 6},
|
||
|
|
"PSRLW": {0x71, 2},
|
||
|
|
"PSRAW": {0x71, 4},
|
||
|
|
"PSLLL": {0x72, 6},
|
||
|
|
"PSRLL": {0x72, 2},
|
||
|
|
"PSRAL": {0x72, 4},
|
||
|
|
"PSLLQ": {0x73, 6},
|
||
|
|
"PSRLQ": {0x73, 2},
|
||
|
|
}
|
||
|
|
|
||
|
|
// mmxShiftVar lists the variable-count forms over the MMX bank, the same
|
||
|
|
// opcodes the XMM variable shifts ride, prefix dropped.
|
||
|
|
var mmxShiftVar = map[string]byte{
|
||
|
|
"PSLLW": 0xF1,
|
||
|
|
"PSRLW": 0xD1,
|
||
|
|
"PSRAW": 0xE1,
|
||
|
|
"PSLLL": 0xF2,
|
||
|
|
"PSRLL": 0xD2,
|
||
|
|
"PSRAL": 0xE2,
|
||
|
|
"PSLLQ": 0xF3,
|
||
|
|
"PSRLQ": 0xD3,
|
||
|
|
}
|
||
|
|
|
||
|
|
// sseOctaShift lists the octa byte shifts' extra spellings: PSLLO and PSRLO
|
||
|
|
// are the Plan 9 names of PSLLDQ/PSRLDQ, XMM only.
|
||
|
|
var sseOctaShift = map[string]sseShift{
|
||
|
|
"PSLLO": {0x73, 7},
|
||
|
|
"PSRLO": {0x73, 3},
|
||
|
|
}
|
||
|
|
|
||
|
|
// isMmx reports whether the operand is an MMX register.
|
||
|
|
func isMmx(op Operand) bool {
|
||
|
|
r, ok := op.(Reg)
|
||
|
|
return ok && r.mmx
|
||
|
|
}
|
||
|
|
|
||
|
|
// encodeSSEMore encodes the MMX glue and the leaf SIMD spellings. It
|
||
|
|
// reports whether the mnemonic belongs to the layer; a false result hands
|
||
|
|
// the mnemonic back to the caller.
|
||
|
|
func (e *enc) encodeSSEMore(upper string, ops []Operand) (bool, error) {
|
||
|
|
if upper == "EMMS" {
|
||
|
|
if len(ops) != 0 {
|
||
|
|
return true, fmt.Errorf("EMMS takes no operands, got %d", len(ops))
|
||
|
|
}
|
||
|
|
return true, e.emit(&instr{opcode: []byte{0x0F, 0x77}, modrm: -1, sib: -1})
|
||
|
|
}
|
||
|
|
if m, ok := sseMaskmov[upper]; ok {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
srcReg, ok1 := ops[0].(Reg)
|
||
|
|
dstReg, ok2 := ops[1].(Reg)
|
||
|
|
if !ok1 || !ok2 || !srcReg.isVec() && !srcReg.mmx || !dstReg.isVec() && !dstReg.mmx {
|
||
|
|
return true, fmt.Errorf("%s takes two vector or MMX registers", upper)
|
||
|
|
}
|
||
|
|
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||
|
|
if err := setRM(i, dstReg, srcReg, 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
// The packed shifts over the MMX bank drop the 0x66 prefix the XMM forms
|
||
|
|
// carry; only a shift name enters the MMX path, the XMM spellings of the
|
||
|
|
// shifts and every other packed binary fall through to the main tables.
|
||
|
|
if _, isShift := mmxShiftImm[upper]; isShift {
|
||
|
|
if e.encodeMmxShiftGate(ops) {
|
||
|
|
return e.encodeMmxShift(upper, ops)
|
||
|
|
}
|
||
|
|
} else if _, isVar := mmxShiftVar[upper]; isVar {
|
||
|
|
if e.encodeMmxShiftGate(ops) {
|
||
|
|
return e.encodeMmxShift(upper, ops)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// PSHUFW is the MMX word shuffle, 0F 70 with no prefix: the XMM twins
|
||
|
|
// (PSHUFD and friends) dispatch through the main shuffle table.
|
||
|
|
if upper == "PSHUFW" {
|
||
|
|
if len(ops) != 3 {
|
||
|
|
return true, fmt.Errorf("PSHUFW expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||
|
|
}
|
||
|
|
imm, ok := ops[0].(Imm)
|
||
|
|
if !ok {
|
||
|
|
return true, fmt.Errorf("PSHUFW needs an imm8 first operand")
|
||
|
|
}
|
||
|
|
immByte, err := imm8(int64(imm))
|
||
|
|
if err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
dstReg, ok2 := ops[2].(Reg)
|
||
|
|
if !ok2 || !dstReg.mmx {
|
||
|
|
return true, fmt.Errorf("PSHUFW destination must be an MMX register")
|
||
|
|
}
|
||
|
|
i := &instr{opcode: []byte{0x0F, 0x70}, modrm: -1, sib: -1}
|
||
|
|
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
i.imm = []byte{immByte}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
if spec, ok := sseOctaShift[upper]; ok {
|
||
|
|
return e.encodeMmxShiftForm(upper, spec, 0x66, ops)
|
||
|
|
}
|
||
|
|
// The byte-mask extract over the MMX bank rides the same 0F D7 opcode
|
||
|
|
// without the prefix; the XMM spelling falls through.
|
||
|
|
if upper == "PMOVMSKB" && len(ops) == 2 {
|
||
|
|
if srcReg, ok := ops[0].(Reg); ok && srcReg.mmx {
|
||
|
|
dstReg, ok2 := ops[1].(Reg)
|
||
|
|
if !ok2 || dstReg.isVec() || dstReg.mmx {
|
||
|
|
return true, fmt.Errorf("%s destination must be a general register", upper)
|
||
|
|
}
|
||
|
|
i := newInstr(4, []byte{0x0F, 0xD7})
|
||
|
|
if err := setRM(i, dstReg, srcReg, 4); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// MOVQOZX is the octa-to-quad zero-extend load, F3 0F D6: an MMX or
|
||
|
|
// memory source into an XMM destination, the MOVQ2DQ opcode.
|
||
|
|
if upper == "MOVQOZX" {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("MOVQOZX expects 2 operands, got %d", len(ops))
|
||
|
|
}
|
||
|
|
dstReg, ok := ops[1].(Reg)
|
||
|
|
if !ok || !dstReg.isVec() {
|
||
|
|
return true, fmt.Errorf("MOVQOZX destination must be an XMM register")
|
||
|
|
}
|
||
|
|
switch ops[0].(type) {
|
||
|
|
case Reg, Mem, sbMem:
|
||
|
|
default:
|
||
|
|
return true, fmt.Errorf("MOVQOZX source must be an MMX register or memory")
|
||
|
|
}
|
||
|
|
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
|
||
|
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
// MOVZWW and MOVSWW are the word zero/sign-extend moves under aliases:
|
||
|
|
// the MOVWLZX and MOVWLSX opcodes carrying the word width's 0x66 prefix.
|
||
|
|
switch upper {
|
||
|
|
case "MOVZWW", "MOVSWW":
|
||
|
|
op := []byte{0x0F, 0xB7}
|
||
|
|
if upper == "MOVSWW" {
|
||
|
|
op[1] = 0xBF
|
||
|
|
}
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
dstReg, ok := ops[1].(Reg)
|
||
|
|
if !ok {
|
||
|
|
return true, fmt.Errorf("%s destination must be a register", upper)
|
||
|
|
}
|
||
|
|
i := newInstr(2, op)
|
||
|
|
if err := setRM(i, dstReg, ops[0], 2); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
// The packed and scalar binaries: reg = destination, r/m = source, the
|
||
|
|
// shared encoder carrying the MMX prefix drop.
|
||
|
|
if m, ok := sseMoreBin[upper]; ok {
|
||
|
|
return true, e.encodeSSEBin(m, ops)
|
||
|
|
}
|
||
|
|
// The imm8-controlled instructions: OP $imm, src, dst.
|
||
|
|
if m, ok := sseMoreImm3[upper]; ok {
|
||
|
|
return true, e.encodeSSEImm3(m, ops)
|
||
|
|
}
|
||
|
|
// The variable blends with their implicit X0 mask: the first operand is
|
||
|
|
// the literal X0, the register the encoding leaves out.
|
||
|
|
if m, ok := sseBlendv[upper]; ok {
|
||
|
|
if len(ops) != 3 {
|
||
|
|
return true, fmt.Errorf("%s expects 3 operands (X0, src, dst), got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
x0, ok := ops[0].(Reg)
|
||
|
|
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
|
||
|
|
return true, fmt.Errorf("%s first operand must be X0", upper)
|
||
|
|
}
|
||
|
|
return true, e.encodeSSEBin(m, ops[1:])
|
||
|
|
}
|
||
|
|
// The high/low half moves split by direction: a memory source loads, a
|
||
|
|
// memory destination stores, both two-operand.
|
||
|
|
if m, ok := sseHighLow[upper]; ok {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
srcReg, srcVec := vecReg(ops[0])
|
||
|
|
dstReg, dstVec := vecReg(ops[1])
|
||
|
|
var op byte
|
||
|
|
var reg Reg
|
||
|
|
var rm Operand
|
||
|
|
switch {
|
||
|
|
case srcVec && isX86Mem(ops[1]):
|
||
|
|
op, reg, rm = m.store, srcReg, ops[1]
|
||
|
|
case dstVec && isX86Mem(ops[0]):
|
||
|
|
op, reg, rm = m.load, dstReg, ops[0]
|
||
|
|
default:
|
||
|
|
return true, fmt.Errorf("%s takes one vector register and one memory operand", upper)
|
||
|
|
}
|
||
|
|
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||
|
|
if err := setRM(i, reg, rm, 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
// The register-to-register half moves.
|
||
|
|
if m, ok := sseRegReg[upper]; ok {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
srcReg, srcVec := vecReg(ops[0])
|
||
|
|
dstReg, dstVec := vecReg(ops[1])
|
||
|
|
if !srcVec || !dstVec {
|
||
|
|
return true, fmt.Errorf("%s takes two XMM registers", upper)
|
||
|
|
}
|
||
|
|
i := &instr{opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||
|
|
if err := setRM(i, dstReg, srcReg, 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
// The sign-mask extractions: the vector source's sign bits pack into a
|
||
|
|
// general register.
|
||
|
|
if m, ok := sseMovmsk[upper]; ok {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
srcReg, srcVec := vecReg(ops[0])
|
||
|
|
dstReg, ok := ops[1].(Reg)
|
||
|
|
if !srcVec || !ok || dstReg.isVec() || dstReg.mmx {
|
||
|
|
return true, fmt.Errorf("%s takes a vector register and a general register", upper)
|
||
|
|
}
|
||
|
|
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||
|
|
if err := setRM(i, dstReg, srcReg, 4); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
// The non-temporal stores: the register source rides reg, the memory
|
||
|
|
// destination r/m; MOVNTIL/IQ store from a general register and MOVNTIQ
|
||
|
|
// carries REX.W.
|
||
|
|
if m, ok := sseMovnt[upper]; ok {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
var srcReg Reg
|
||
|
|
switch r := ops[0].(type) {
|
||
|
|
case Reg:
|
||
|
|
if upper == "MOVNTIL" || upper == "MOVNTIQ" {
|
||
|
|
if r.isVec() || r.mmx {
|
||
|
|
return true, fmt.Errorf("%s source must be a general register", upper)
|
||
|
|
}
|
||
|
|
srcReg = r
|
||
|
|
} else {
|
||
|
|
if !r.isVec() && !r.mmx {
|
||
|
|
return true, fmt.Errorf("%s source must be a vector or MMX register", upper)
|
||
|
|
}
|
||
|
|
srcReg = r
|
||
|
|
}
|
||
|
|
default:
|
||
|
|
return true, fmt.Errorf("%s source must be a register", upper)
|
||
|
|
}
|
||
|
|
if !isX86Mem(ops[1]) {
|
||
|
|
return true, fmt.Errorf("%s destination must be memory", upper)
|
||
|
|
}
|
||
|
|
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1, rexW: upper == "MOVNTIQ"}
|
||
|
|
if err := setRM(i, srcReg, ops[1], 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
// EXTRACTPS is the lane extract to a GPR or memory, the PEXTR layout.
|
||
|
|
if upper == "EXTRACTPS" {
|
||
|
|
return true, e.encodeSSEExtract(sseExtract{op: []byte{0x0F, 0x3A, 0x17}}, ops)
|
||
|
|
}
|
||
|
|
// CVTSL2SS and CVTSQ2SS are the integer-to-scalar-single converts, the
|
||
|
|
// CVTSL2SD pair's F3 twin: F3 0F 2A with reg = XMM destination, REX.W
|
||
|
|
// on the quad source spelling.
|
||
|
|
if upper == "CVTSL2SS" || upper == "CVTSQ2SS" {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
dstReg, ok := ops[1].(Reg)
|
||
|
|
if !ok || !dstReg.isVec() {
|
||
|
|
return true, fmt.Errorf("%s destination must be a vector register", upper)
|
||
|
|
}
|
||
|
|
i := newInstr(0, []byte{0x0F, 0x2A})
|
||
|
|
i.rexW = upper == "CVTSQ2SS"
|
||
|
|
i.prefix = 0xF3
|
||
|
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
return false, nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// encodeMmxShiftGate reports whether the shift's destination operand is an
|
||
|
|
// MMX register, the case the bank's own prefix-free forms cover.
|
||
|
|
func (e *enc) encodeMmxShiftGate(ops []Operand) bool {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return false
|
||
|
|
}
|
||
|
|
dstReg, ok := ops[1].(Reg)
|
||
|
|
return ok && dstReg.mmx
|
||
|
|
}
|
||
|
|
|
||
|
|
// encodeMmxShift routes the MMX shift between its immediate form
|
||
|
|
// (OP $imm, dst, the 0F 71/72/73 /digit group) and its variable-count form
|
||
|
|
// (OP count, dst, the 0F D1-F3 row).
|
||
|
|
func (e *enc) encodeMmxShift(upper string, ops []Operand) (bool, error) {
|
||
|
|
if spec, ok := mmxShiftImm[upper]; ok {
|
||
|
|
if _, isImm := ops[0].(Imm); isImm {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
immByte, err := imm8(int64(ops[0].(Imm)))
|
||
|
|
if err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
i := &instr{opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||
|
|
if err := setRMDigit(i, spec.digit, ops[1], 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
i.imm = []byte{immByte}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if op, ok := mmxShiftVar[upper]; ok {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
if !vecOrMem(ops[0]) && !isMmx(ops[0]) {
|
||
|
|
return true, fmt.Errorf("%s count must be an immediate, an MMX register or memory", upper)
|
||
|
|
}
|
||
|
|
dstReg, ok := ops[1].(Reg)
|
||
|
|
if !ok || !dstReg.mmx {
|
||
|
|
return true, fmt.Errorf("%s destination must be an MMX register", upper)
|
||
|
|
}
|
||
|
|
i := &instr{opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||
|
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|
||
|
|
return true, fmt.Errorf("unsupported instruction %q", upper)
|
||
|
|
}
|
||
|
|
|
||
|
|
// encodeMmxShiftForm emits one XMM octa shift: OP $imm, dst, the 0x66
|
||
|
|
// prefix carried.
|
||
|
|
func (e *enc) encodeMmxShiftForm(upper string, spec sseShift, prefix byte, ops []Operand) (bool, error) {
|
||
|
|
if len(ops) != 2 {
|
||
|
|
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||
|
|
}
|
||
|
|
imm, ok := ops[0].(Imm)
|
||
|
|
if !ok {
|
||
|
|
return true, fmt.Errorf("%s needs an immediate count", upper)
|
||
|
|
}
|
||
|
|
immByte, err := imm8(int64(imm))
|
||
|
|
if err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
dstReg, ok2 := ops[1].(Reg)
|
||
|
|
if !ok2 || !dstReg.isVec() {
|
||
|
|
return true, fmt.Errorf("%s destination must be the second, vector operand", upper)
|
||
|
|
}
|
||
|
|
i := &instr{prefix: prefix, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||
|
|
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
|
||
|
|
return true, err
|
||
|
|
}
|
||
|
|
i.imm = []byte{immByte}
|
||
|
|
return true, e.emit(i)
|
||
|
|
}
|