Files
gasm-sdk/asm/amd64_sse.go
T

566 lines
19 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// This file implements the amd64 SIMD pieces the main tables lack: the MMX
// bank glue (EMMS, the masked stores, the MMX shifts and shuffle, the
// byte-mask extract), the MOVQ bank crossings' odd spellings and the leaf
// aliases. Every encoding here is pinned byte for byte against go tool asm
// through the corpus lines in amd64_sse_test.go.
// sseMaskmov maps the masked cache-line stores to their prefix and opcode:
// OP src, dst with the second operand in the reg field, no memory operand.
var sseMaskmov = map[string]sseBin{
"MASKMOVQ": {0, 0xF7, false},
"MASKMOVOU": {0x66, 0xF7, false},
}
// sseMoreBin maps the packed and scalar legacy binaries the main table
// lacks, reg = destination and r/m = source: the float comparisons, square
// roots and reciprocal estimates, the SSE3 horizontal arithmetic, the SSE4.1
// packed integers and the SSSE3 sign and horizontal ops. The MMX twins of
// the 0x66-prefixed members drop the prefix in the shared encoder.
var sseMoreBin = map[string]sseBin{
"COMISS": {0, 0x2F, false},
"UCOMISS": {0, 0x2E, false},
"UCOMISD": {0x66, 0x2E, false},
"SQRTPS": {0, 0x51, false},
"SQRTPD": {0x66, 0x51, false},
"SQRTSS": {0xF3, 0x51, false},
"RCPPS": {0, 0x53, false},
"RCPSS": {0xF3, 0x53, false},
"RSQRTPS": {0, 0x52, false},
"RSQRTSS": {0xF3, 0x52, false},
"ADDSUBPD": {0x66, 0xD0, false},
"ADDSUBPS": {0xF2, 0xD0, false},
"HADDPD": {0x66, 0x7C, false},
"HADDPS": {0xF2, 0x7C, false},
"HSUBPD": {0x66, 0x7D, false},
"HSUBPS": {0xF2, 0x7D, false},
"MOVDDUP": {0xF2, 0x12, false},
"MOVSHDUP": {0xF3, 0x16, false},
"MOVSLDUP": {0xF3, 0x12, false},
"LDDQU": {0xF2, 0xF0, false},
"MOVNTDQA": {0x66, 0x2A, true},
"PTEST": {0x66, 0x17, true},
"PABSB": {0x66, 0x1C, true},
"PABSW": {0x66, 0x1D, true},
"PABSD": {0x66, 0x1E, true},
"PACKSSWB": {0x66, 0x63, false},
"PACKUSWB": {0x66, 0x67, false},
"PACKSSLW": {0x66, 0x6B, false},
"PACKUSDW": {0x66, 0x2B, true},
"PADDSB": {0x66, 0xEC, false},
"PADDSW": {0x66, 0xED, false},
"PADDUSB": {0x66, 0xDC, false},
"PADDUSW": {0x66, 0xDD, false},
"PAVGB": {0x66, 0xE0, false},
"PAVGW": {0x66, 0xE3, false},
"PCMPEQQ": {0x66, 0x29, true},
"PCMPGTQ": {0x66, 0x37, true},
"PHADDW": {0x66, 0x01, true},
"PHADDD": {0x66, 0x02, true},
"PHADDSW": {0x66, 0x03, true},
"PHSUBW": {0x66, 0x05, true},
"PHSUBD": {0x66, 0x06, true},
"PHSUBSW": {0x66, 0x07, true},
"PHMINPOSUW": {0x66, 0x41, true},
"PMADDUBSW": {0x66, 0x04, true},
"PMADDWL": {0x66, 0xF5, false},
"PMAXSB": {0x66, 0x3C, true},
"PMAXSD": {0x66, 0x3D, true},
"PMAXSW": {0x66, 0xEE, false},
"PMAXUB": {0x66, 0xDE, false},
"PMAXUD": {0x66, 0x3F, true},
"PMAXUW": {0x66, 0x3E, true},
"PMINSB": {0x66, 0x38, true},
"PMINSD": {0x66, 0x39, true},
"PMINSW": {0x66, 0xEA, false},
"PMINUB": {0x66, 0xDA, false},
"PMINUD": {0x66, 0x3B, true},
"PMINUW": {0x66, 0x3A, true},
"PMULDQ": {0x66, 0x28, true},
"PMULLD": {0x66, 0x40, true},
"PMULHRSW": {0x66, 0x0B, true},
"PMULHUW": {0x66, 0xE4, false},
"PMULHW": {0x66, 0xE5, false},
"PMULLW": {0x66, 0xD5, false},
"PMULULQ": {0x66, 0xF4, false},
"PCMPGTL": {0x66, 0x66, false},
"PMOVSXBD": {0x66, 0x21, true},
"PMOVSXBQ": {0x66, 0x22, true},
"PMOVSXBW": {0x66, 0x20, true},
"PMOVSXDQ": {0x66, 0x25, true},
"PMOVSXWD": {0x66, 0x23, true},
"PMOVSXWQ": {0x66, 0x24, true},
"PMOVZXBD": {0x66, 0x31, true},
"PMOVZXBQ": {0x66, 0x32, true},
"PMOVZXBW": {0x66, 0x30, true},
"PMOVZXDQ": {0x66, 0x35, true},
"PMOVZXWD": {0x66, 0x33, true},
"PMOVZXWQ": {0x66, 0x34, true},
// The conversion aliases the Plan 9 table spells with an L: the dword
// sources and destinations of the packed integer/float converts.
"CVTPL2PD": {0xF3, 0xE6, false},
"CVTPL2PS": {0, 0x5B, false},
"CVTPD2PL": {0xF2, 0xE6, false},
"CVTPS2PL": {0x66, 0x5B, false},
"CVTTPD2PL": {0x66, 0xE6, false},
"CVTTPS2PL": {0xF3, 0x5B, false},
"PSADBW": {0x66, 0xF6, false},
"PSUBSB": {0x66, 0xE8, false},
"PSUBSW": {0x66, 0xE9, false},
"PSUBUSB": {0x66, 0xD8, false},
"PSUBUSW": {0x66, 0xD9, false},
"PSIGNB": {0x66, 0x08, true},
"PSIGNW": {0x66, 0x09, true},
"PSIGND": {0x66, 0x0A, true},
"PUNPCKHBW": {0x66, 0x68, false},
"PUNPCKHLQ": {0x66, 0x6A, false},
"PUNPCKHQDQ": {0x66, 0x6D, false},
"PUNPCKHWL": {0x66, 0x69, false},
"PUNPCKLLQ": {0x66, 0x62, false},
"PUNPCKLQDQ": {0x66, 0x6C, false},
"PUNPCKLWL": {0x66, 0x61, false},
}
// sseMoreImm3 maps the imm8-controlled three-operand instructions the main
// table lacks: OP $imm, src, dst.
var sseMoreImm3 = map[string]sseImm3{
"ROUNDPS": {0x66, 0x08, true},
"ROUNDPD": {0x66, 0x09, true},
"ROUNDSS": {0x66, 0x0A, true},
"ROUNDSD": {0x66, 0x0B, true},
"DPPS": {0x66, 0x40, true},
"DPPD": {0x66, 0x41, true},
"BLENDPS": {0x66, 0x0C, true},
"BLENDPD": {0x66, 0x0D, true},
"INSERTPS": {0x66, 0x21, true},
"MPSADBW": {0x66, 0x42, true},
"PCMPESTRM": {0x66, 0x60, true},
"PCMPESTRI": {0x66, 0x61, true},
"PCMPISTRM": {0x66, 0x62, true},
"PCMPISTRI": {0x66, 0x63, true},
}
// sseBlendv maps the variable blends whose implicit mask is X0: the first
// operand must be the literal X0, the register the hardware reads.
var sseBlendv = map[string]sseBin{
"BLENDVPS": {0x66, 0x14, true},
"BLENDVPD": {0x66, 0x15, true},
"PBLENDVB": {0x66, 0x10, true},
}
// sseHighLow maps the high/low half moves to their load/store opcode pair.
// A memory source loads (reg = destination), a memory destination stores
// (reg = the register source).
var sseHighLow = map[string]sseMove{
"MOVHPD": {0x66, 0x16, 0x17},
"MOVHPS": {0, 0x16, 0x17},
"MOVLPD": {0x66, 0x12, 0x13},
"MOVLPS": {0, 0x12, 0x13},
}
// sseRegReg maps the register-to-register half moves, register destination
// and register source alone: MOVHLPS and MOVLHPS.
var sseRegReg = map[string]sseBin{
"MOVHLPS": {0, 0x12, false},
"MOVLHPS": {0, 0x16, false},
}
// sseMovmsk maps the sign-mask extractions to a GPR: OP vec, gpr.
var sseMovmsk = map[string]sseBin{
"MOVMSKPS": {0, 0x50, false},
"MOVMSKPD": {0x66, 0x50, false},
}
// sseMovnt maps the non-temporal stores, OP reg, mem, plus MOVNTDQA's load
// (which rides sseMoreBin).
var sseMovnt = map[string]sseBin{
"MOVNTPS": {0, 0x2B, false},
"MOVNTPD": {0x66, 0x2B, false},
"MOVNTQ": {0, 0xE7, false},
"MOVNTO": {0x66, 0xE7, false},
"MOVNTIL": {0, 0xC3, false},
"MOVNTIQ": {0, 0xC3, false},
}
// mmxShiftImm lists the packed integer shifts whose immediate form the MMX
// bank spells without the 0x66 prefix; the digit rides the 0F 71/72/73
// group, the same /digits the XMM forms carry.
var mmxShiftImm = map[string]sseShift{
"PSLLW": {0x71, 6},
"PSRLW": {0x71, 2},
"PSRAW": {0x71, 4},
"PSLLL": {0x72, 6},
"PSRLL": {0x72, 2},
"PSRAL": {0x72, 4},
"PSLLQ": {0x73, 6},
"PSRLQ": {0x73, 2},
}
// mmxShiftVar lists the variable-count forms over the MMX bank, the same
// opcodes the XMM variable shifts ride, prefix dropped.
var mmxShiftVar = map[string]byte{
"PSLLW": 0xF1,
"PSRLW": 0xD1,
"PSRAW": 0xE1,
"PSLLL": 0xF2,
"PSRLL": 0xD2,
"PSRAL": 0xE2,
"PSLLQ": 0xF3,
"PSRLQ": 0xD3,
}
// sseOctaShift lists the octa byte shifts' extra spellings: PSLLO and PSRLO
// are the Plan 9 names of PSLLDQ/PSRLDQ, XMM only.
var sseOctaShift = map[string]sseShift{
"PSLLO": {0x73, 7},
"PSRLO": {0x73, 3},
}
// isMmx reports whether the operand is an MMX register.
func isMmx(op Operand) bool {
r, ok := op.(Reg)
return ok && r.mmx
}
// encodeSSEMore encodes the MMX glue and the leaf SIMD spellings. It
// reports whether the mnemonic belongs to the layer; a false result hands
// the mnemonic back to the caller.
func (e *enc) encodeSSEMore(upper string, ops []Operand) (bool, error) {
if upper == "EMMS" {
if len(ops) != 0 {
return true, fmt.Errorf("EMMS takes no operands, got %d", len(ops))
}
return true, e.emit(&instr{opcode: []byte{0x0F, 0x77}, modrm: -1, sib: -1})
}
if m, ok := sseMaskmov[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, ok1 := ops[0].(Reg)
dstReg, ok2 := ops[1].(Reg)
if !ok1 || !ok2 || !srcReg.isVec() && !srcReg.mmx || !dstReg.isVec() && !dstReg.mmx {
return true, fmt.Errorf("%s takes two vector or MMX registers", upper)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// The packed shifts over the MMX bank drop the 0x66 prefix the XMM forms
// carry; only a shift name enters the MMX path, the XMM spellings of the
// shifts and every other packed binary fall through to the main tables.
if _, isShift := mmxShiftImm[upper]; isShift {
if e.encodeMmxShiftGate(ops) {
return e.encodeMmxShift(upper, ops)
}
} else if _, isVar := mmxShiftVar[upper]; isVar {
if e.encodeMmxShiftGate(ops) {
return e.encodeMmxShift(upper, ops)
}
}
// PSHUFW is the MMX word shuffle, 0F 70 with no prefix: the XMM twins
// (PSHUFD and friends) dispatch through the main shuffle table.
if upper == "PSHUFW" {
if len(ops) != 3 {
return true, fmt.Errorf("PSHUFW expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return true, fmt.Errorf("PSHUFW needs an imm8 first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return true, err
}
dstReg, ok2 := ops[2].(Reg)
if !ok2 || !dstReg.mmx {
return true, fmt.Errorf("PSHUFW destination must be an MMX register")
}
i := &instr{opcode: []byte{0x0F, 0x70}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[1], 8); err != nil {
return true, err
}
i.imm = []byte{immByte}
return true, e.emit(i)
}
if spec, ok := sseOctaShift[upper]; ok {
return e.encodeMmxShiftForm(upper, spec, 0x66, ops)
}
// The byte-mask extract over the MMX bank rides the same 0F D7 opcode
// without the prefix; the XMM spelling falls through.
if upper == "PMOVMSKB" && len(ops) == 2 {
if srcReg, ok := ops[0].(Reg); ok && srcReg.mmx {
dstReg, ok2 := ops[1].(Reg)
if !ok2 || dstReg.isVec() || dstReg.mmx {
return true, fmt.Errorf("%s destination must be a general register", upper)
}
i := newInstr(4, []byte{0x0F, 0xD7})
if err := setRM(i, dstReg, srcReg, 4); err != nil {
return true, err
}
return true, e.emit(i)
}
}
// MOVQOZX is the octa-to-quad zero-extend load, F3 0F D6: an MMX or
// memory source into an XMM destination, the MOVQ2DQ opcode.
if upper == "MOVQOZX" {
if len(ops) != 2 {
return true, fmt.Errorf("MOVQOZX expects 2 operands, got %d", len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.isVec() {
return true, fmt.Errorf("MOVQOZX destination must be an XMM register")
}
switch ops[0].(type) {
case Reg, Mem, sbMem:
default:
return true, fmt.Errorf("MOVQOZX source must be an MMX register or memory")
}
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// MOVZWW and MOVSWW are the word zero/sign-extend moves under aliases:
// the MOVWLZX and MOVWLSX opcodes carrying the word width's 0x66 prefix.
switch upper {
case "MOVZWW", "MOVSWW":
op := []byte{0x0F, 0xB7}
if upper == "MOVSWW" {
op[1] = 0xBF
}
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok {
return true, fmt.Errorf("%s destination must be a register", upper)
}
i := newInstr(2, op)
if err := setRM(i, dstReg, ops[0], 2); err != nil {
return true, err
}
return true, e.emit(i)
}
// The packed and scalar binaries: reg = destination, r/m = source, the
// shared encoder carrying the MMX prefix drop.
if m, ok := sseMoreBin[upper]; ok {
return true, e.encodeSSEBin(m, ops)
}
// The imm8-controlled instructions: OP $imm, src, dst.
if m, ok := sseMoreImm3[upper]; ok {
return true, e.encodeSSEImm3(m, ops)
}
// The variable blends with their implicit X0 mask: the first operand is
// the literal X0, the register the encoding leaves out.
if m, ok := sseBlendv[upper]; ok {
if len(ops) != 3 {
return true, fmt.Errorf("%s expects 3 operands (X0, src, dst), got %d", upper, len(ops))
}
x0, ok := ops[0].(Reg)
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
return true, fmt.Errorf("%s first operand must be X0", upper)
}
return true, e.encodeSSEBin(m, ops[1:])
}
// The high/low half moves split by direction: a memory source loads, a
// memory destination stores, both two-operand.
if m, ok := sseHighLow[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, srcVec := vecReg(ops[0])
dstReg, dstVec := vecReg(ops[1])
var op byte
var reg Reg
var rm Operand
switch {
case srcVec && isX86Mem(ops[1]):
op, reg, rm = m.store, srcReg, ops[1]
case dstVec && isX86Mem(ops[0]):
op, reg, rm = m.load, dstReg, ops[0]
default:
return true, fmt.Errorf("%s takes one vector register and one memory operand", upper)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, reg, rm, 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// The register-to-register half moves.
if m, ok := sseRegReg[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, srcVec := vecReg(ops[0])
dstReg, dstVec := vecReg(ops[1])
if !srcVec || !dstVec {
return true, fmt.Errorf("%s takes two XMM registers", upper)
}
i := &instr{opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// The sign-mask extractions: the vector source's sign bits pack into a
// general register.
if m, ok := sseMovmsk[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, srcVec := vecReg(ops[0])
dstReg, ok := ops[1].(Reg)
if !srcVec || !ok || dstReg.isVec() || dstReg.mmx {
return true, fmt.Errorf("%s takes a vector register and a general register", upper)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 4); err != nil {
return true, err
}
return true, e.emit(i)
}
// The non-temporal stores: the register source rides reg, the memory
// destination r/m; MOVNTIL/IQ store from a general register and MOVNTIQ
// carries REX.W.
if m, ok := sseMovnt[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
var srcReg Reg
switch r := ops[0].(type) {
case Reg:
if upper == "MOVNTIL" || upper == "MOVNTIQ" {
if r.isVec() || r.mmx {
return true, fmt.Errorf("%s source must be a general register", upper)
}
srcReg = r
} else {
if !r.isVec() && !r.mmx {
return true, fmt.Errorf("%s source must be a vector or MMX register", upper)
}
srcReg = r
}
default:
return true, fmt.Errorf("%s source must be a register", upper)
}
if !isX86Mem(ops[1]) {
return true, fmt.Errorf("%s destination must be memory", upper)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1, rexW: upper == "MOVNTIQ"}
if err := setRM(i, srcReg, ops[1], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// EXTRACTPS is the lane extract to a GPR or memory, the PEXTR layout.
if upper == "EXTRACTPS" {
return true, e.encodeSSEExtract(sseExtract{op: []byte{0x0F, 0x3A, 0x17}}, ops)
}
// CVTSL2SS and CVTSQ2SS are the integer-to-scalar-single converts, the
// CVTSL2SD pair's F3 twin: F3 0F 2A with reg = XMM destination, REX.W
// on the quad source spelling.
if upper == "CVTSL2SS" || upper == "CVTSQ2SS" {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.isVec() {
return true, fmt.Errorf("%s destination must be a vector register", upper)
}
i := newInstr(0, []byte{0x0F, 0x2A})
i.rexW = upper == "CVTSQ2SS"
i.prefix = 0xF3
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
return false, nil
}
// encodeMmxShiftGate reports whether the shift's destination operand is an
// MMX register, the case the bank's own prefix-free forms cover.
func (e *enc) encodeMmxShiftGate(ops []Operand) bool {
if len(ops) != 2 {
return false
}
dstReg, ok := ops[1].(Reg)
return ok && dstReg.mmx
}
// encodeMmxShift routes the MMX shift between its immediate form
// (OP $imm, dst, the 0F 71/72/73 /digit group) and its variable-count form
// (OP count, dst, the 0F D1-F3 row).
func (e *enc) encodeMmxShift(upper string, ops []Operand) (bool, error) {
if spec, ok := mmxShiftImm[upper]; ok {
if _, isImm := ops[0].(Imm); isImm {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
immByte, err := imm8(int64(ops[0].(Imm)))
if err != nil {
return true, err
}
i := &instr{opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
if err := setRMDigit(i, spec.digit, ops[1], 8); err != nil {
return true, err
}
i.imm = []byte{immByte}
return true, e.emit(i)
}
}
if op, ok := mmxShiftVar[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
if !vecOrMem(ops[0]) && !isMmx(ops[0]) {
return true, fmt.Errorf("%s count must be an immediate, an MMX register or memory", upper)
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.mmx {
return true, fmt.Errorf("%s destination must be an MMX register", upper)
}
i := &instr{opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
return true, fmt.Errorf("unsupported instruction %q", upper)
}
// encodeMmxShiftForm emits one XMM octa shift: OP $imm, dst, the 0x66
// prefix carried.
func (e *enc) encodeMmxShiftForm(upper string, spec sseShift, prefix byte, ops []Operand) (bool, error) {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return true, fmt.Errorf("%s needs an immediate count", upper)
}
immByte, err := imm8(int64(imm))
if err != nil {
return true, err
}
dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() {
return true, fmt.Errorf("%s destination must be the second, vector operand", upper)
}
i := &instr{prefix: prefix, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
return true, err
}
i.imm = []byte{immByte}
return true, e.emit(i)
}