// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import "fmt" // This file implements the amd64 SIMD pieces the main tables lack: the MMX // bank glue (EMMS, the masked stores, the MMX shifts and shuffle, the // byte-mask extract), the MOVQ bank crossings' odd spellings and the leaf // aliases. Every encoding here is pinned byte for byte against go tool asm // through the corpus lines in amd64_sse_test.go. // sseMaskmov maps the masked cache-line stores to their prefix and opcode: // OP src, dst with the second operand in the reg field, no memory operand. var sseMaskmov = map[string]sseBin{ "MASKMOVQ": {0, 0xF7, false}, "MASKMOVOU": {0x66, 0xF7, false}, } // sseMoreBin maps the packed and scalar legacy binaries the main table // lacks, reg = destination and r/m = source: the float comparisons, square // roots and reciprocal estimates, the SSE3 horizontal arithmetic, the SSE4.1 // packed integers and the SSSE3 sign and horizontal ops. The MMX twins of // the 0x66-prefixed members drop the prefix in the shared encoder. var sseMoreBin = map[string]sseBin{ "COMISS": {0, 0x2F, false}, "UCOMISS": {0, 0x2E, false}, "UCOMISD": {0x66, 0x2E, false}, "SQRTPS": {0, 0x51, false}, "SQRTPD": {0x66, 0x51, false}, "SQRTSS": {0xF3, 0x51, false}, "RCPPS": {0, 0x53, false}, "RCPSS": {0xF3, 0x53, false}, "RSQRTPS": {0, 0x52, false}, "RSQRTSS": {0xF3, 0x52, false}, "ADDSUBPD": {0x66, 0xD0, false}, "ADDSUBPS": {0xF2, 0xD0, false}, "HADDPD": {0x66, 0x7C, false}, "HADDPS": {0xF2, 0x7C, false}, "HSUBPD": {0x66, 0x7D, false}, "HSUBPS": {0xF2, 0x7D, false}, "MOVDDUP": {0xF2, 0x12, false}, "MOVSHDUP": {0xF3, 0x16, false}, "MOVSLDUP": {0xF3, 0x12, false}, "LDDQU": {0xF2, 0xF0, false}, "MOVNTDQA": {0x66, 0x2A, true}, "PTEST": {0x66, 0x17, true}, "PABSB": {0x66, 0x1C, true}, "PABSW": {0x66, 0x1D, true}, "PABSD": {0x66, 0x1E, true}, "PACKSSWB": {0x66, 0x63, false}, "PACKUSWB": {0x66, 0x67, false}, "PACKSSLW": {0x66, 0x6B, false}, "PACKUSDW": {0x66, 0x2B, true}, "PADDSB": {0x66, 0xEC, false}, "PADDSW": {0x66, 0xED, false}, "PADDUSB": {0x66, 0xDC, false}, "PADDUSW": {0x66, 0xDD, false}, "PAVGB": {0x66, 0xE0, false}, "PAVGW": {0x66, 0xE3, false}, "PCMPEQQ": {0x66, 0x29, true}, "PCMPGTQ": {0x66, 0x37, true}, "PHADDW": {0x66, 0x01, true}, "PHADDD": {0x66, 0x02, true}, "PHADDSW": {0x66, 0x03, true}, "PHSUBW": {0x66, 0x05, true}, "PHSUBD": {0x66, 0x06, true}, "PHSUBSW": {0x66, 0x07, true}, "PHMINPOSUW": {0x66, 0x41, true}, "PMADDUBSW": {0x66, 0x04, true}, "PMADDWL": {0x66, 0xF5, false}, "PMAXSB": {0x66, 0x3C, true}, "PMAXSD": {0x66, 0x3D, true}, "PMAXSW": {0x66, 0xEE, false}, "PMAXUB": {0x66, 0xDE, false}, "PMAXUD": {0x66, 0x3F, true}, "PMAXUW": {0x66, 0x3E, true}, "PMINSB": {0x66, 0x38, true}, "PMINSD": {0x66, 0x39, true}, "PMINSW": {0x66, 0xEA, false}, "PMINUB": {0x66, 0xDA, false}, "PMINUD": {0x66, 0x3B, true}, "PMINUW": {0x66, 0x3A, true}, "PMULDQ": {0x66, 0x28, true}, "PMULLD": {0x66, 0x40, true}, "PMULHRSW": {0x66, 0x0B, true}, "PMULHUW": {0x66, 0xE4, false}, "PMULHW": {0x66, 0xE5, false}, "PMULLW": {0x66, 0xD5, false}, "PMULULQ": {0x66, 0xF4, false}, "PCMPGTL": {0x66, 0x66, false}, "PMOVSXBD": {0x66, 0x21, true}, "PMOVSXBQ": {0x66, 0x22, true}, "PMOVSXBW": {0x66, 0x20, true}, "PMOVSXDQ": {0x66, 0x25, true}, "PMOVSXWD": {0x66, 0x23, true}, "PMOVSXWQ": {0x66, 0x24, true}, "PMOVZXBD": {0x66, 0x31, true}, "PMOVZXBQ": {0x66, 0x32, true}, "PMOVZXBW": {0x66, 0x30, true}, "PMOVZXDQ": {0x66, 0x35, true}, "PMOVZXWD": {0x66, 0x33, true}, "PMOVZXWQ": {0x66, 0x34, true}, // The conversion aliases the Plan 9 table spells with an L: the dword // sources and destinations of the packed integer/float converts. "CVTPL2PD": {0xF3, 0xE6, false}, "CVTPL2PS": {0, 0x5B, false}, "CVTPD2PL": {0xF2, 0xE6, false}, "CVTPS2PL": {0x66, 0x5B, false}, "CVTTPD2PL": {0x66, 0xE6, false}, "CVTTPS2PL": {0xF3, 0x5B, false}, "PSADBW": {0x66, 0xF6, false}, "PSUBSB": {0x66, 0xE8, false}, "PSUBSW": {0x66, 0xE9, false}, "PSUBUSB": {0x66, 0xD8, false}, "PSUBUSW": {0x66, 0xD9, false}, "PSIGNB": {0x66, 0x08, true}, "PSIGNW": {0x66, 0x09, true}, "PSIGND": {0x66, 0x0A, true}, "PUNPCKHBW": {0x66, 0x68, false}, "PUNPCKHLQ": {0x66, 0x6A, false}, "PUNPCKHQDQ": {0x66, 0x6D, false}, "PUNPCKHWL": {0x66, 0x69, false}, "PUNPCKLLQ": {0x66, 0x62, false}, "PUNPCKLQDQ": {0x66, 0x6C, false}, "PUNPCKLWL": {0x66, 0x61, false}, } // sseMoreImm3 maps the imm8-controlled three-operand instructions the main // table lacks: OP $imm, src, dst. var sseMoreImm3 = map[string]sseImm3{ "ROUNDPS": {0x66, 0x08, true}, "ROUNDPD": {0x66, 0x09, true}, "ROUNDSS": {0x66, 0x0A, true}, "ROUNDSD": {0x66, 0x0B, true}, "DPPS": {0x66, 0x40, true}, "DPPD": {0x66, 0x41, true}, "BLENDPS": {0x66, 0x0C, true}, "BLENDPD": {0x66, 0x0D, true}, "INSERTPS": {0x66, 0x21, true}, "MPSADBW": {0x66, 0x42, true}, "PCMPESTRM": {0x66, 0x60, true}, "PCMPESTRI": {0x66, 0x61, true}, "PCMPISTRM": {0x66, 0x62, true}, "PCMPISTRI": {0x66, 0x63, true}, } // sseBlendv maps the variable blends whose implicit mask is X0: the first // operand must be the literal X0, the register the hardware reads. var sseBlendv = map[string]sseBin{ "BLENDVPS": {0x66, 0x14, true}, "BLENDVPD": {0x66, 0x15, true}, "PBLENDVB": {0x66, 0x10, true}, } // sseHighLow maps the high/low half moves to their load/store opcode pair. // A memory source loads (reg = destination), a memory destination stores // (reg = the register source). var sseHighLow = map[string]sseMove{ "MOVHPD": {0x66, 0x16, 0x17}, "MOVHPS": {0, 0x16, 0x17}, "MOVLPD": {0x66, 0x12, 0x13}, "MOVLPS": {0, 0x12, 0x13}, } // sseRegReg maps the register-to-register half moves, register destination // and register source alone: MOVHLPS and MOVLHPS. var sseRegReg = map[string]sseBin{ "MOVHLPS": {0, 0x12, false}, "MOVLHPS": {0, 0x16, false}, } // sseMovmsk maps the sign-mask extractions to a GPR: OP vec, gpr. var sseMovmsk = map[string]sseBin{ "MOVMSKPS": {0, 0x50, false}, "MOVMSKPD": {0x66, 0x50, false}, } // sseMovnt maps the non-temporal stores, OP reg, mem, plus MOVNTDQA's load // (which rides sseMoreBin). var sseMovnt = map[string]sseBin{ "MOVNTPS": {0, 0x2B, false}, "MOVNTPD": {0x66, 0x2B, false}, "MOVNTQ": {0, 0xE7, false}, "MOVNTO": {0x66, 0xE7, false}, "MOVNTIL": {0, 0xC3, false}, "MOVNTIQ": {0, 0xC3, false}, } // mmxShiftImm lists the packed integer shifts whose immediate form the MMX // bank spells without the 0x66 prefix; the digit rides the 0F 71/72/73 // group, the same /digits the XMM forms carry. var mmxShiftImm = map[string]sseShift{ "PSLLW": {0x71, 6}, "PSRLW": {0x71, 2}, "PSRAW": {0x71, 4}, "PSLLL": {0x72, 6}, "PSRLL": {0x72, 2}, "PSRAL": {0x72, 4}, "PSLLQ": {0x73, 6}, "PSRLQ": {0x73, 2}, } // mmxShiftVar lists the variable-count forms over the MMX bank, the same // opcodes the XMM variable shifts ride, prefix dropped. var mmxShiftVar = map[string]byte{ "PSLLW": 0xF1, "PSRLW": 0xD1, "PSRAW": 0xE1, "PSLLL": 0xF2, "PSRLL": 0xD2, "PSRAL": 0xE2, "PSLLQ": 0xF3, "PSRLQ": 0xD3, } // sseOctaShift lists the octa byte shifts' extra spellings: PSLLO and PSRLO // are the Plan 9 names of PSLLDQ/PSRLDQ, XMM only. var sseOctaShift = map[string]sseShift{ "PSLLO": {0x73, 7}, "PSRLO": {0x73, 3}, } // isMmx reports whether the operand is an MMX register. func isMmx(op Operand) bool { r, ok := op.(Reg) return ok && r.mmx } // encodeSSEMore encodes the MMX glue and the leaf SIMD spellings. It // reports whether the mnemonic belongs to the layer; a false result hands // the mnemonic back to the caller. func (e *enc) encodeSSEMore(upper string, ops []Operand) (bool, error) { if upper == "EMMS" { if len(ops) != 0 { return true, fmt.Errorf("EMMS takes no operands, got %d", len(ops)) } return true, e.emit(&instr{opcode: []byte{0x0F, 0x77}, modrm: -1, sib: -1}) } if m, ok := sseMaskmov[upper]; ok { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } srcReg, ok1 := ops[0].(Reg) dstReg, ok2 := ops[1].(Reg) if !ok1 || !ok2 || !srcReg.isVec() && !srcReg.mmx || !dstReg.isVec() && !dstReg.mmx { return true, fmt.Errorf("%s takes two vector or MMX registers", upper) } i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} if err := setRM(i, dstReg, srcReg, 8); err != nil { return true, err } return true, e.emit(i) } // The packed shifts over the MMX bank drop the 0x66 prefix the XMM forms // carry; only a shift name enters the MMX path, the XMM spellings of the // shifts and every other packed binary fall through to the main tables. if _, isShift := mmxShiftImm[upper]; isShift { if e.encodeMmxShiftGate(ops) { return e.encodeMmxShift(upper, ops) } } else if _, isVar := mmxShiftVar[upper]; isVar { if e.encodeMmxShiftGate(ops) { return e.encodeMmxShift(upper, ops) } } // PSHUFW is the MMX word shuffle, 0F 70 with no prefix: the XMM twins // (PSHUFD and friends) dispatch through the main shuffle table. if upper == "PSHUFW" { if len(ops) != 3 { return true, fmt.Errorf("PSHUFW expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return true, fmt.Errorf("PSHUFW needs an imm8 first operand") } immByte, err := imm8(int64(imm)) if err != nil { return true, err } dstReg, ok2 := ops[2].(Reg) if !ok2 || !dstReg.mmx { return true, fmt.Errorf("PSHUFW destination must be an MMX register") } i := &instr{opcode: []byte{0x0F, 0x70}, modrm: -1, sib: -1} if err := setRM(i, dstReg, ops[1], 8); err != nil { return true, err } i.imm = []byte{immByte} return true, e.emit(i) } if spec, ok := sseOctaShift[upper]; ok { return e.encodeMmxShiftForm(upper, spec, 0x66, ops) } // The byte-mask extract over the MMX bank rides the same 0F D7 opcode // without the prefix; the XMM spelling falls through. if upper == "PMOVMSKB" && len(ops) == 2 { if srcReg, ok := ops[0].(Reg); ok && srcReg.mmx { dstReg, ok2 := ops[1].(Reg) if !ok2 || dstReg.isVec() || dstReg.mmx { return true, fmt.Errorf("%s destination must be a general register", upper) } i := newInstr(4, []byte{0x0F, 0xD7}) if err := setRM(i, dstReg, srcReg, 4); err != nil { return true, err } return true, e.emit(i) } } // MOVQOZX is the octa-to-quad zero-extend load, F3 0F D6: an MMX or // memory source into an XMM destination, the MOVQ2DQ opcode. if upper == "MOVQOZX" { if len(ops) != 2 { return true, fmt.Errorf("MOVQOZX expects 2 operands, got %d", len(ops)) } dstReg, ok := ops[1].(Reg) if !ok || !dstReg.isVec() { return true, fmt.Errorf("MOVQOZX destination must be an XMM register") } switch ops[0].(type) { case Reg, Mem, sbMem: default: return true, fmt.Errorf("MOVQOZX source must be an MMX register or memory") } i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1} if err := setRM(i, dstReg, ops[0], 8); err != nil { return true, err } return true, e.emit(i) } // MOVZWW and MOVSWW are the word zero/sign-extend moves under aliases: // the MOVWLZX and MOVWLSX opcodes carrying the word width's 0x66 prefix. switch upper { case "MOVZWW", "MOVSWW": op := []byte{0x0F, 0xB7} if upper == "MOVSWW" { op[1] = 0xBF } if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } dstReg, ok := ops[1].(Reg) if !ok { return true, fmt.Errorf("%s destination must be a register", upper) } i := newInstr(2, op) if err := setRM(i, dstReg, ops[0], 2); err != nil { return true, err } return true, e.emit(i) } // The packed and scalar binaries: reg = destination, r/m = source, the // shared encoder carrying the MMX prefix drop. if m, ok := sseMoreBin[upper]; ok { return true, e.encodeSSEBin(m, ops) } // The imm8-controlled instructions: OP $imm, src, dst. if m, ok := sseMoreImm3[upper]; ok { return true, e.encodeSSEImm3(m, ops) } // The variable blends with their implicit X0 mask: the first operand is // the literal X0, the register the encoding leaves out. if m, ok := sseBlendv[upper]; ok { if len(ops) != 3 { return true, fmt.Errorf("%s expects 3 operands (X0, src, dst), got %d", upper, len(ops)) } x0, ok := ops[0].(Reg) if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 { return true, fmt.Errorf("%s first operand must be X0", upper) } return true, e.encodeSSEBin(m, ops[1:]) } // The high/low half moves split by direction: a memory source loads, a // memory destination stores, both two-operand. if m, ok := sseHighLow[upper]; ok { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } srcReg, srcVec := vecReg(ops[0]) dstReg, dstVec := vecReg(ops[1]) var op byte var reg Reg var rm Operand switch { case srcVec && isX86Mem(ops[1]): op, reg, rm = m.store, srcReg, ops[1] case dstVec && isX86Mem(ops[0]): op, reg, rm = m.load, dstReg, ops[0] default: return true, fmt.Errorf("%s takes one vector register and one memory operand", upper) } i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1} if err := setRM(i, reg, rm, 8); err != nil { return true, err } return true, e.emit(i) } // The register-to-register half moves. if m, ok := sseRegReg[upper]; ok { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } srcReg, srcVec := vecReg(ops[0]) dstReg, dstVec := vecReg(ops[1]) if !srcVec || !dstVec { return true, fmt.Errorf("%s takes two XMM registers", upper) } i := &instr{opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} if err := setRM(i, dstReg, srcReg, 8); err != nil { return true, err } return true, e.emit(i) } // The sign-mask extractions: the vector source's sign bits pack into a // general register. if m, ok := sseMovmsk[upper]; ok { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } srcReg, srcVec := vecReg(ops[0]) dstReg, ok := ops[1].(Reg) if !srcVec || !ok || dstReg.isVec() || dstReg.mmx { return true, fmt.Errorf("%s takes a vector register and a general register", upper) } i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} if err := setRM(i, dstReg, srcReg, 4); err != nil { return true, err } return true, e.emit(i) } // The non-temporal stores: the register source rides reg, the memory // destination r/m; MOVNTIL/IQ store from a general register and MOVNTIQ // carries REX.W. if m, ok := sseMovnt[upper]; ok { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } var srcReg Reg switch r := ops[0].(type) { case Reg: if upper == "MOVNTIL" || upper == "MOVNTIQ" { if r.isVec() || r.mmx { return true, fmt.Errorf("%s source must be a general register", upper) } srcReg = r } else { if !r.isVec() && !r.mmx { return true, fmt.Errorf("%s source must be a vector or MMX register", upper) } srcReg = r } default: return true, fmt.Errorf("%s source must be a register", upper) } if !isX86Mem(ops[1]) { return true, fmt.Errorf("%s destination must be memory", upper) } i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1, rexW: upper == "MOVNTIQ"} if err := setRM(i, srcReg, ops[1], 8); err != nil { return true, err } return true, e.emit(i) } // EXTRACTPS is the lane extract to a GPR or memory, the PEXTR layout. if upper == "EXTRACTPS" { return true, e.encodeSSEExtract(sseExtract{op: []byte{0x0F, 0x3A, 0x17}}, ops) } // CVTSL2SS and CVTSQ2SS are the integer-to-scalar-single converts, the // CVTSL2SD pair's F3 twin: F3 0F 2A with reg = XMM destination, REX.W // on the quad source spelling. if upper == "CVTSL2SS" || upper == "CVTSQ2SS" { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } dstReg, ok := ops[1].(Reg) if !ok || !dstReg.isVec() { return true, fmt.Errorf("%s destination must be a vector register", upper) } i := newInstr(0, []byte{0x0F, 0x2A}) i.rexW = upper == "CVTSQ2SS" i.prefix = 0xF3 if err := setRM(i, dstReg, ops[0], 8); err != nil { return true, err } return true, e.emit(i) } return false, nil } // encodeMmxShiftGate reports whether the shift's destination operand is an // MMX register, the case the bank's own prefix-free forms cover. func (e *enc) encodeMmxShiftGate(ops []Operand) bool { if len(ops) != 2 { return false } dstReg, ok := ops[1].(Reg) return ok && dstReg.mmx } // encodeMmxShift routes the MMX shift between its immediate form // (OP $imm, dst, the 0F 71/72/73 /digit group) and its variable-count form // (OP count, dst, the 0F D1-F3 row). func (e *enc) encodeMmxShift(upper string, ops []Operand) (bool, error) { if spec, ok := mmxShiftImm[upper]; ok { if _, isImm := ops[0].(Imm); isImm { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } immByte, err := imm8(int64(ops[0].(Imm))) if err != nil { return true, err } i := &instr{opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1} if err := setRMDigit(i, spec.digit, ops[1], 8); err != nil { return true, err } i.imm = []byte{immByte} return true, e.emit(i) } } if op, ok := mmxShiftVar[upper]; ok { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } if !vecOrMem(ops[0]) && !isMmx(ops[0]) { return true, fmt.Errorf("%s count must be an immediate, an MMX register or memory", upper) } dstReg, ok := ops[1].(Reg) if !ok || !dstReg.mmx { return true, fmt.Errorf("%s destination must be an MMX register", upper) } i := &instr{opcode: []byte{0x0F, op}, modrm: -1, sib: -1} if err := setRM(i, dstReg, ops[0], 8); err != nil { return true, err } return true, e.emit(i) } return true, fmt.Errorf("unsupported instruction %q", upper) } // encodeMmxShiftForm emits one XMM octa shift: OP $imm, dst, the 0x66 // prefix carried. func (e *enc) encodeMmxShiftForm(upper string, spec sseShift, prefix byte, ops []Operand) (bool, error) { if len(ops) != 2 { return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } imm, ok := ops[0].(Imm) if !ok { return true, fmt.Errorf("%s needs an immediate count", upper) } immByte, err := imm8(int64(imm)) if err != nil { return true, err } dstReg, ok2 := ops[1].(Reg) if !ok2 || !dstReg.isVec() { return true, fmt.Errorf("%s destination must be the second, vector operand", upper) } i := &instr{prefix: prefix, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1} if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil { return true, err } i.imm = []byte{immByte} return true, e.emit(i) }