feat(asm): encode the legacy amd64 SSE and MMX families
The packed integer and float binaries, the imm8-controlled SSE4.1 forms, the variable blends with their X0 mask, the high/low half moves, the sign-mask extractions, the non-temporal stores, the MOVQ bank crossings and their odd spellings, the MMX shifts and shuffle and the cache-line mask stores, plus the scalar leaves LEAVE, INVPCID and the RTM controls. Every encoding is pinned byte for byte against go tool asm through every corpus line the toolchain's own amd64enc.s carries for the families (1465 lines). Two corpus-wide gaps fell out of the comparison: the 64-bit MOV immediate uses the zero-extending form across the unsigned 32-bit span, and the MMX-to-GPR MOVQ puts the bank register in reg. Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
257feace6e
commit
6af3fd60d5
5 files changed
+2159
-13
No files matched your search
@@ -19,6 +19,9 @@ func (e *enc) encodeAmd64Family(upper string, ops []Operand) (bool, error) {
|
||||
if ok, err := e.encodeXsave(upper, ops); ok {
|
||||
return true, err
|
||||
}
|
||||
if ok, err := e.encodeSSEMore(upper, ops); ok {
|
||||
return true, err
|
||||
}
|
||||
return false, nil
|
||||
}
|
||||
|
||||
@@ -91,6 +94,9 @@ func amd64FamilyEncodable(upper string) bool {
|
||||
if upper == "CMPXCHG8B" || upper == "CMPXCHG16B" {
|
||||
return true
|
||||
}
|
||||
if upper == "INVPCID" || upper == "XABORT" {
|
||||
return true
|
||||
}
|
||||
if _, ok := xsaveTable[upper]; ok {
|
||||
return true
|
||||
}
|
||||
@@ -99,5 +105,18 @@ func amd64FamilyEncodable(upper string) bool {
|
||||
return true
|
||||
}
|
||||
}
|
||||
switch upper {
|
||||
case "EMMS", "PSHUFW", "PSLLO", "PSRLO", "PMOVMSKB", "MOVQOZX", "MOVZWW", "MOVSWW":
|
||||
return true
|
||||
}
|
||||
if _, ok := sseMaskmov[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := mmxShiftImm[upper]; ok {
|
||||
return true
|
||||
}
|
||||
if _, ok := mmxShiftVar[upper]; ok {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,566 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package asm
|
||||
|
||||
import "fmt"
|
||||
|
||||
// This file implements the amd64 SIMD pieces the main tables lack: the MMX
|
||||
// bank glue (EMMS, the masked stores, the MMX shifts and shuffle, the
|
||||
// byte-mask extract), the MOVQ bank crossings' odd spellings and the leaf
|
||||
// aliases. Every encoding here is pinned byte for byte against go tool asm
|
||||
// through the corpus lines in amd64_sse_test.go.
|
||||
|
||||
// sseMaskmov maps the masked cache-line stores to their prefix and opcode:
|
||||
// OP src, dst with the second operand in the reg field, no memory operand.
|
||||
var sseMaskmov = map[string]sseBin{
|
||||
"MASKMOVQ": {0, 0xF7, false},
|
||||
"MASKMOVOU": {0x66, 0xF7, false},
|
||||
}
|
||||
|
||||
// sseMoreBin maps the packed and scalar legacy binaries the main table
|
||||
// lacks, reg = destination and r/m = source: the float comparisons, square
|
||||
// roots and reciprocal estimates, the SSE3 horizontal arithmetic, the SSE4.1
|
||||
// packed integers and the SSSE3 sign and horizontal ops. The MMX twins of
|
||||
// the 0x66-prefixed members drop the prefix in the shared encoder.
|
||||
var sseMoreBin = map[string]sseBin{
|
||||
"COMISS": {0, 0x2F, false},
|
||||
"UCOMISS": {0, 0x2E, false},
|
||||
"UCOMISD": {0x66, 0x2E, false},
|
||||
"SQRTPS": {0, 0x51, false},
|
||||
"SQRTPD": {0x66, 0x51, false},
|
||||
"SQRTSS": {0xF3, 0x51, false},
|
||||
"RCPPS": {0, 0x53, false},
|
||||
"RCPSS": {0xF3, 0x53, false},
|
||||
"RSQRTPS": {0, 0x52, false},
|
||||
"RSQRTSS": {0xF3, 0x52, false},
|
||||
"ADDSUBPD": {0x66, 0xD0, false},
|
||||
"ADDSUBPS": {0xF2, 0xD0, false},
|
||||
"HADDPD": {0x66, 0x7C, false},
|
||||
"HADDPS": {0xF2, 0x7C, false},
|
||||
"HSUBPD": {0x66, 0x7D, false},
|
||||
"HSUBPS": {0xF2, 0x7D, false},
|
||||
"MOVDDUP": {0xF2, 0x12, false},
|
||||
"MOVSHDUP": {0xF3, 0x16, false},
|
||||
"MOVSLDUP": {0xF3, 0x12, false},
|
||||
"LDDQU": {0xF2, 0xF0, false},
|
||||
"MOVNTDQA": {0x66, 0x2A, true},
|
||||
"PTEST": {0x66, 0x17, true},
|
||||
"PABSB": {0x66, 0x1C, true},
|
||||
"PABSW": {0x66, 0x1D, true},
|
||||
"PABSD": {0x66, 0x1E, true},
|
||||
"PACKSSWB": {0x66, 0x63, false},
|
||||
"PACKUSWB": {0x66, 0x67, false},
|
||||
"PACKSSLW": {0x66, 0x6B, false},
|
||||
"PACKUSDW": {0x66, 0x2B, true},
|
||||
"PADDSB": {0x66, 0xEC, false},
|
||||
"PADDSW": {0x66, 0xED, false},
|
||||
"PADDUSB": {0x66, 0xDC, false},
|
||||
"PADDUSW": {0x66, 0xDD, false},
|
||||
"PAVGB": {0x66, 0xE0, false},
|
||||
"PAVGW": {0x66, 0xE3, false},
|
||||
"PCMPEQQ": {0x66, 0x29, true},
|
||||
"PCMPGTQ": {0x66, 0x37, true},
|
||||
"PHADDW": {0x66, 0x01, true},
|
||||
"PHADDD": {0x66, 0x02, true},
|
||||
"PHADDSW": {0x66, 0x03, true},
|
||||
"PHSUBW": {0x66, 0x05, true},
|
||||
"PHSUBD": {0x66, 0x06, true},
|
||||
"PHSUBSW": {0x66, 0x07, true},
|
||||
"PHMINPOSUW": {0x66, 0x41, true},
|
||||
"PMADDUBSW": {0x66, 0x04, true},
|
||||
"PMADDWL": {0x66, 0xF5, false},
|
||||
"PMAXSB": {0x66, 0x3C, true},
|
||||
"PMAXSD": {0x66, 0x3D, true},
|
||||
"PMAXSW": {0x66, 0xEE, false},
|
||||
"PMAXUB": {0x66, 0xDE, false},
|
||||
"PMAXUD": {0x66, 0x3F, true},
|
||||
"PMAXUW": {0x66, 0x3E, true},
|
||||
"PMINSB": {0x66, 0x38, true},
|
||||
"PMINSD": {0x66, 0x39, true},
|
||||
"PMINSW": {0x66, 0xEA, false},
|
||||
"PMINUB": {0x66, 0xDA, false},
|
||||
"PMINUD": {0x66, 0x3B, true},
|
||||
"PMINUW": {0x66, 0x3A, true},
|
||||
"PMULDQ": {0x66, 0x28, true},
|
||||
"PMULLD": {0x66, 0x40, true},
|
||||
"PMULHRSW": {0x66, 0x0B, true},
|
||||
"PMULHUW": {0x66, 0xE4, false},
|
||||
"PMULHW": {0x66, 0xE5, false},
|
||||
"PMULLW": {0x66, 0xD5, false},
|
||||
"PMULULQ": {0x66, 0xF4, false},
|
||||
"PCMPGTL": {0x66, 0x66, false},
|
||||
"PMOVSXBD": {0x66, 0x21, true},
|
||||
"PMOVSXBQ": {0x66, 0x22, true},
|
||||
"PMOVSXBW": {0x66, 0x20, true},
|
||||
"PMOVSXDQ": {0x66, 0x25, true},
|
||||
"PMOVSXWD": {0x66, 0x23, true},
|
||||
"PMOVSXWQ": {0x66, 0x24, true},
|
||||
"PMOVZXBD": {0x66, 0x31, true},
|
||||
"PMOVZXBQ": {0x66, 0x32, true},
|
||||
"PMOVZXBW": {0x66, 0x30, true},
|
||||
"PMOVZXDQ": {0x66, 0x35, true},
|
||||
"PMOVZXWD": {0x66, 0x33, true},
|
||||
"PMOVZXWQ": {0x66, 0x34, true},
|
||||
// The conversion aliases the Plan 9 table spells with an L: the dword
|
||||
// sources and destinations of the packed integer/float converts.
|
||||
"CVTPL2PD": {0xF3, 0xE6, false},
|
||||
"CVTPL2PS": {0, 0x5B, false},
|
||||
"CVTPD2PL": {0xF2, 0xE6, false},
|
||||
"CVTPS2PL": {0x66, 0x5B, false},
|
||||
"CVTTPD2PL": {0x66, 0xE6, false},
|
||||
"CVTTPS2PL": {0xF3, 0x5B, false},
|
||||
"PSADBW": {0x66, 0xF6, false},
|
||||
"PSUBSB": {0x66, 0xE8, false},
|
||||
"PSUBSW": {0x66, 0xE9, false},
|
||||
"PSUBUSB": {0x66, 0xD8, false},
|
||||
"PSUBUSW": {0x66, 0xD9, false},
|
||||
"PSIGNB": {0x66, 0x08, true},
|
||||
"PSIGNW": {0x66, 0x09, true},
|
||||
"PSIGND": {0x66, 0x0A, true},
|
||||
"PUNPCKHBW": {0x66, 0x68, false},
|
||||
"PUNPCKHLQ": {0x66, 0x6A, false},
|
||||
"PUNPCKHQDQ": {0x66, 0x6D, false},
|
||||
"PUNPCKHWL": {0x66, 0x69, false},
|
||||
"PUNPCKLLQ": {0x66, 0x62, false},
|
||||
"PUNPCKLQDQ": {0x66, 0x6C, false},
|
||||
"PUNPCKLWL": {0x66, 0x61, false},
|
||||
}
|
||||
|
||||
// sseMoreImm3 maps the imm8-controlled three-operand instructions the main
|
||||
// table lacks: OP $imm, src, dst.
|
||||
var sseMoreImm3 = map[string]sseImm3{
|
||||
"ROUNDPS": {0x66, 0x08, true},
|
||||
"ROUNDPD": {0x66, 0x09, true},
|
||||
"ROUNDSS": {0x66, 0x0A, true},
|
||||
"ROUNDSD": {0x66, 0x0B, true},
|
||||
"DPPS": {0x66, 0x40, true},
|
||||
"DPPD": {0x66, 0x41, true},
|
||||
"BLENDPS": {0x66, 0x0C, true},
|
||||
"BLENDPD": {0x66, 0x0D, true},
|
||||
"INSERTPS": {0x66, 0x21, true},
|
||||
"MPSADBW": {0x66, 0x42, true},
|
||||
"PCMPESTRM": {0x66, 0x60, true},
|
||||
"PCMPESTRI": {0x66, 0x61, true},
|
||||
"PCMPISTRM": {0x66, 0x62, true},
|
||||
"PCMPISTRI": {0x66, 0x63, true},
|
||||
}
|
||||
|
||||
// sseBlendv maps the variable blends whose implicit mask is X0: the first
|
||||
// operand must be the literal X0, the register the hardware reads.
|
||||
var sseBlendv = map[string]sseBin{
|
||||
"BLENDVPS": {0x66, 0x14, true},
|
||||
"BLENDVPD": {0x66, 0x15, true},
|
||||
"PBLENDVB": {0x66, 0x10, true},
|
||||
}
|
||||
|
||||
// sseHighLow maps the high/low half moves to their load/store opcode pair.
|
||||
// A memory source loads (reg = destination), a memory destination stores
|
||||
// (reg = the register source).
|
||||
var sseHighLow = map[string]sseMove{
|
||||
"MOVHPD": {0x66, 0x16, 0x17},
|
||||
"MOVHPS": {0, 0x16, 0x17},
|
||||
"MOVLPD": {0x66, 0x12, 0x13},
|
||||
"MOVLPS": {0, 0x12, 0x13},
|
||||
}
|
||||
|
||||
// sseRegReg maps the register-to-register half moves, register destination
|
||||
// and register source alone: MOVHLPS and MOVLHPS.
|
||||
var sseRegReg = map[string]sseBin{
|
||||
"MOVHLPS": {0, 0x12, false},
|
||||
"MOVLHPS": {0, 0x16, false},
|
||||
}
|
||||
|
||||
// sseMovmsk maps the sign-mask extractions to a GPR: OP vec, gpr.
|
||||
var sseMovmsk = map[string]sseBin{
|
||||
"MOVMSKPS": {0, 0x50, false},
|
||||
"MOVMSKPD": {0x66, 0x50, false},
|
||||
}
|
||||
|
||||
// sseMovnt maps the non-temporal stores, OP reg, mem, plus MOVNTDQA's load
|
||||
// (which rides sseMoreBin).
|
||||
var sseMovnt = map[string]sseBin{
|
||||
"MOVNTPS": {0, 0x2B, false},
|
||||
"MOVNTPD": {0x66, 0x2B, false},
|
||||
"MOVNTQ": {0, 0xE7, false},
|
||||
"MOVNTO": {0x66, 0xE7, false},
|
||||
"MOVNTIL": {0, 0xC3, false},
|
||||
"MOVNTIQ": {0, 0xC3, false},
|
||||
}
|
||||
|
||||
// mmxShiftImm lists the packed integer shifts whose immediate form the MMX
|
||||
// bank spells without the 0x66 prefix; the digit rides the 0F 71/72/73
|
||||
// group, the same /digits the XMM forms carry.
|
||||
var mmxShiftImm = map[string]sseShift{
|
||||
"PSLLW": {0x71, 6},
|
||||
"PSRLW": {0x71, 2},
|
||||
"PSRAW": {0x71, 4},
|
||||
"PSLLL": {0x72, 6},
|
||||
"PSRLL": {0x72, 2},
|
||||
"PSRAL": {0x72, 4},
|
||||
"PSLLQ": {0x73, 6},
|
||||
"PSRLQ": {0x73, 2},
|
||||
}
|
||||
|
||||
// mmxShiftVar lists the variable-count forms over the MMX bank, the same
|
||||
// opcodes the XMM variable shifts ride, prefix dropped.
|
||||
var mmxShiftVar = map[string]byte{
|
||||
"PSLLW": 0xF1,
|
||||
"PSRLW": 0xD1,
|
||||
"PSRAW": 0xE1,
|
||||
"PSLLL": 0xF2,
|
||||
"PSRLL": 0xD2,
|
||||
"PSRAL": 0xE2,
|
||||
"PSLLQ": 0xF3,
|
||||
"PSRLQ": 0xD3,
|
||||
}
|
||||
|
||||
// sseOctaShift lists the octa byte shifts' extra spellings: PSLLO and PSRLO
|
||||
// are the Plan 9 names of PSLLDQ/PSRLDQ, XMM only.
|
||||
var sseOctaShift = map[string]sseShift{
|
||||
"PSLLO": {0x73, 7},
|
||||
"PSRLO": {0x73, 3},
|
||||
}
|
||||
|
||||
// isMmx reports whether the operand is an MMX register.
|
||||
func isMmx(op Operand) bool {
|
||||
r, ok := op.(Reg)
|
||||
return ok && r.mmx
|
||||
}
|
||||
|
||||
// encodeSSEMore encodes the MMX glue and the leaf SIMD spellings. It
|
||||
// reports whether the mnemonic belongs to the layer; a false result hands
|
||||
// the mnemonic back to the caller.
|
||||
func (e *enc) encodeSSEMore(upper string, ops []Operand) (bool, error) {
|
||||
if upper == "EMMS" {
|
||||
if len(ops) != 0 {
|
||||
return true, fmt.Errorf("EMMS takes no operands, got %d", len(ops))
|
||||
}
|
||||
return true, e.emit(&instr{opcode: []byte{0x0F, 0x77}, modrm: -1, sib: -1})
|
||||
}
|
||||
if m, ok := sseMaskmov[upper]; ok {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
srcReg, ok1 := ops[0].(Reg)
|
||||
dstReg, ok2 := ops[1].(Reg)
|
||||
if !ok1 || !ok2 || !srcReg.isVec() && !srcReg.mmx || !dstReg.isVec() && !dstReg.mmx {
|
||||
return true, fmt.Errorf("%s takes two vector or MMX registers", upper)
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, srcReg, 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// The packed shifts over the MMX bank drop the 0x66 prefix the XMM forms
|
||||
// carry; only a shift name enters the MMX path, the XMM spellings of the
|
||||
// shifts and every other packed binary fall through to the main tables.
|
||||
if _, isShift := mmxShiftImm[upper]; isShift {
|
||||
if e.encodeMmxShiftGate(ops) {
|
||||
return e.encodeMmxShift(upper, ops)
|
||||
}
|
||||
} else if _, isVar := mmxShiftVar[upper]; isVar {
|
||||
if e.encodeMmxShiftGate(ops) {
|
||||
return e.encodeMmxShift(upper, ops)
|
||||
}
|
||||
}
|
||||
// PSHUFW is the MMX word shuffle, 0F 70 with no prefix: the XMM twins
|
||||
// (PSHUFD and friends) dispatch through the main shuffle table.
|
||||
if upper == "PSHUFW" {
|
||||
if len(ops) != 3 {
|
||||
return true, fmt.Errorf("PSHUFW expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return true, fmt.Errorf("PSHUFW needs an imm8 first operand")
|
||||
}
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return true, err
|
||||
}
|
||||
dstReg, ok2 := ops[2].(Reg)
|
||||
if !ok2 || !dstReg.mmx {
|
||||
return true, fmt.Errorf("PSHUFW destination must be an MMX register")
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, 0x70}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
if spec, ok := sseOctaShift[upper]; ok {
|
||||
return e.encodeMmxShiftForm(upper, spec, 0x66, ops)
|
||||
}
|
||||
// The byte-mask extract over the MMX bank rides the same 0F D7 opcode
|
||||
// without the prefix; the XMM spelling falls through.
|
||||
if upper == "PMOVMSKB" && len(ops) == 2 {
|
||||
if srcReg, ok := ops[0].(Reg); ok && srcReg.mmx {
|
||||
dstReg, ok2 := ops[1].(Reg)
|
||||
if !ok2 || dstReg.isVec() || dstReg.mmx {
|
||||
return true, fmt.Errorf("%s destination must be a general register", upper)
|
||||
}
|
||||
i := newInstr(4, []byte{0x0F, 0xD7})
|
||||
if err := setRM(i, dstReg, srcReg, 4); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
}
|
||||
// MOVQOZX is the octa-to-quad zero-extend load, F3 0F D6: an MMX or
|
||||
// memory source into an XMM destination, the MOVQ2DQ opcode.
|
||||
if upper == "MOVQOZX" {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("MOVQOZX expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return true, fmt.Errorf("MOVQOZX destination must be an XMM register")
|
||||
}
|
||||
switch ops[0].(type) {
|
||||
case Reg, Mem, sbMem:
|
||||
default:
|
||||
return true, fmt.Errorf("MOVQOZX source must be an MMX register or memory")
|
||||
}
|
||||
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// MOVZWW and MOVSWW are the word zero/sign-extend moves under aliases:
|
||||
// the MOVWLZX and MOVWLSX opcodes carrying the word width's 0x66 prefix.
|
||||
switch upper {
|
||||
case "MOVZWW", "MOVSWW":
|
||||
op := []byte{0x0F, 0xB7}
|
||||
if upper == "MOVSWW" {
|
||||
op[1] = 0xBF
|
||||
}
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok {
|
||||
return true, fmt.Errorf("%s destination must be a register", upper)
|
||||
}
|
||||
i := newInstr(2, op)
|
||||
if err := setRM(i, dstReg, ops[0], 2); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// The packed and scalar binaries: reg = destination, r/m = source, the
|
||||
// shared encoder carrying the MMX prefix drop.
|
||||
if m, ok := sseMoreBin[upper]; ok {
|
||||
return true, e.encodeSSEBin(m, ops)
|
||||
}
|
||||
// The imm8-controlled instructions: OP $imm, src, dst.
|
||||
if m, ok := sseMoreImm3[upper]; ok {
|
||||
return true, e.encodeSSEImm3(m, ops)
|
||||
}
|
||||
// The variable blends with their implicit X0 mask: the first operand is
|
||||
// the literal X0, the register the encoding leaves out.
|
||||
if m, ok := sseBlendv[upper]; ok {
|
||||
if len(ops) != 3 {
|
||||
return true, fmt.Errorf("%s expects 3 operands (X0, src, dst), got %d", upper, len(ops))
|
||||
}
|
||||
x0, ok := ops[0].(Reg)
|
||||
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
|
||||
return true, fmt.Errorf("%s first operand must be X0", upper)
|
||||
}
|
||||
return true, e.encodeSSEBin(m, ops[1:])
|
||||
}
|
||||
// The high/low half moves split by direction: a memory source loads, a
|
||||
// memory destination stores, both two-operand.
|
||||
if m, ok := sseHighLow[upper]; ok {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
srcReg, srcVec := vecReg(ops[0])
|
||||
dstReg, dstVec := vecReg(ops[1])
|
||||
var op byte
|
||||
var reg Reg
|
||||
var rm Operand
|
||||
switch {
|
||||
case srcVec && isX86Mem(ops[1]):
|
||||
op, reg, rm = m.store, srcReg, ops[1]
|
||||
case dstVec && isX86Mem(ops[0]):
|
||||
op, reg, rm = m.load, dstReg, ops[0]
|
||||
default:
|
||||
return true, fmt.Errorf("%s takes one vector register and one memory operand", upper)
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, reg, rm, 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// The register-to-register half moves.
|
||||
if m, ok := sseRegReg[upper]; ok {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
srcReg, srcVec := vecReg(ops[0])
|
||||
dstReg, dstVec := vecReg(ops[1])
|
||||
if !srcVec || !dstVec {
|
||||
return true, fmt.Errorf("%s takes two XMM registers", upper)
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, srcReg, 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// The sign-mask extractions: the vector source's sign bits pack into a
|
||||
// general register.
|
||||
if m, ok := sseMovmsk[upper]; ok {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
srcReg, srcVec := vecReg(ops[0])
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !srcVec || !ok || dstReg.isVec() || dstReg.mmx {
|
||||
return true, fmt.Errorf("%s takes a vector register and a general register", upper)
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, srcReg, 4); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// The non-temporal stores: the register source rides reg, the memory
|
||||
// destination r/m; MOVNTIL/IQ store from a general register and MOVNTIQ
|
||||
// carries REX.W.
|
||||
if m, ok := sseMovnt[upper]; ok {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
var srcReg Reg
|
||||
switch r := ops[0].(type) {
|
||||
case Reg:
|
||||
if upper == "MOVNTIL" || upper == "MOVNTIQ" {
|
||||
if r.isVec() || r.mmx {
|
||||
return true, fmt.Errorf("%s source must be a general register", upper)
|
||||
}
|
||||
srcReg = r
|
||||
} else {
|
||||
if !r.isVec() && !r.mmx {
|
||||
return true, fmt.Errorf("%s source must be a vector or MMX register", upper)
|
||||
}
|
||||
srcReg = r
|
||||
}
|
||||
default:
|
||||
return true, fmt.Errorf("%s source must be a register", upper)
|
||||
}
|
||||
if !isX86Mem(ops[1]) {
|
||||
return true, fmt.Errorf("%s destination must be memory", upper)
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1, rexW: upper == "MOVNTIQ"}
|
||||
if err := setRM(i, srcReg, ops[1], 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// EXTRACTPS is the lane extract to a GPR or memory, the PEXTR layout.
|
||||
if upper == "EXTRACTPS" {
|
||||
return true, e.encodeSSEExtract(sseExtract{op: []byte{0x0F, 0x3A, 0x17}}, ops)
|
||||
}
|
||||
// CVTSL2SS and CVTSQ2SS are the integer-to-scalar-single converts, the
|
||||
// CVTSL2SD pair's F3 twin: F3 0F 2A with reg = XMM destination, REX.W
|
||||
// on the quad source spelling.
|
||||
if upper == "CVTSL2SS" || upper == "CVTSQ2SS" {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return true, fmt.Errorf("%s destination must be a vector register", upper)
|
||||
}
|
||||
i := newInstr(0, []byte{0x0F, 0x2A})
|
||||
i.rexW = upper == "CVTSQ2SS"
|
||||
i.prefix = 0xF3
|
||||
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
return false, nil
|
||||
}
|
||||
|
||||
// encodeMmxShiftGate reports whether the shift's destination operand is an
|
||||
// MMX register, the case the bank's own prefix-free forms cover.
|
||||
func (e *enc) encodeMmxShiftGate(ops []Operand) bool {
|
||||
if len(ops) != 2 {
|
||||
return false
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
return ok && dstReg.mmx
|
||||
}
|
||||
|
||||
// encodeMmxShift routes the MMX shift between its immediate form
|
||||
// (OP $imm, dst, the 0F 71/72/73 /digit group) and its variable-count form
|
||||
// (OP count, dst, the 0F D1-F3 row).
|
||||
func (e *enc) encodeMmxShift(upper string, ops []Operand) (bool, error) {
|
||||
if spec, ok := mmxShiftImm[upper]; ok {
|
||||
if _, isImm := ops[0].(Imm); isImm {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
immByte, err := imm8(int64(ops[0].(Imm)))
|
||||
if err != nil {
|
||||
return true, err
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||||
if err := setRMDigit(i, spec.digit, ops[1], 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
}
|
||||
if op, ok := mmxShiftVar[upper]; ok {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
if !vecOrMem(ops[0]) && !isMmx(ops[0]) {
|
||||
return true, fmt.Errorf("%s count must be an immediate, an MMX register or memory", upper)
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok || !dstReg.mmx {
|
||||
return true, fmt.Errorf("%s destination must be an MMX register", upper)
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
return true, fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
|
||||
// encodeMmxShiftForm emits one XMM octa shift: OP $imm, dst, the 0x66
|
||||
// prefix carried.
|
||||
func (e *enc) encodeMmxShiftForm(upper string, spec sseShift, prefix byte, ops []Operand) (bool, error) {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return true, fmt.Errorf("%s needs an immediate count", upper)
|
||||
}
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return true, err
|
||||
}
|
||||
dstReg, ok2 := ops[1].(Reg)
|
||||
if !ok2 || !dstReg.isVec() {
|
||||
return true, fmt.Errorf("%s destination must be the second, vector operand", upper)
|
||||
}
|
||||
i := &instr{prefix: prefix, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||||
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
@@ -53,6 +53,10 @@ var systemNoOperand = map[string][]byte{
|
||||
"SYSEXIT": {0x0F, 0x35},
|
||||
"SYSEXIT64": {0x48, 0x0F, 0x35},
|
||||
"SYSRET": {0x0F, 0x07},
|
||||
"LEAVE": {0xC9},
|
||||
"LEAVEQ": {0xC9},
|
||||
"XEND": {0x0F, 0x01, 0xD5},
|
||||
"XTEST": {0x0F, 0x01, 0xD6},
|
||||
"CBW": {0x66, 0x98},
|
||||
"CWDE": {0x98},
|
||||
"CDQE": {0x48, 0x98},
|
||||
@@ -253,6 +257,40 @@ func (e *enc) encodeSystem(upper string, ops []Operand) (bool, error) {
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// INVPCID invalidates a translation-cache entry: 66 0F38 82 with the
|
||||
// type in a general register and the descriptor in memory.
|
||||
if upper == "INVPCID" {
|
||||
if len(ops) != 2 {
|
||||
return true, fmt.Errorf("INVPCID expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
if !isX86Mem(ops[0]) {
|
||||
return true, fmt.Errorf("INVPCID requires a memory descriptor first")
|
||||
}
|
||||
srcReg, ok := ops[1].(Reg)
|
||||
if !ok || srcReg.isVec() {
|
||||
return true, fmt.Errorf("INVPCID: the second operand must be a general register")
|
||||
}
|
||||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x38, 0x82}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, srcReg, ops[0], 8); err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(i)
|
||||
}
|
||||
// XABORT carries its imm8 status byte in the C6 F8 group form.
|
||||
if upper == "XABORT" {
|
||||
if len(ops) != 1 {
|
||||
return true, fmt.Errorf("XABORT expects 1 immediate operand, got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return true, fmt.Errorf("XABORT requires an immediate")
|
||||
}
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return true, err
|
||||
}
|
||||
return true, e.emit(&instr{opcode: []byte{0xC6, 0xF8}, modrm: -1, sib: -1, imm: []byte{immByte}})
|
||||
}
|
||||
return false, nil
|
||||
}
|
||||
|
||||
|
||||
+39
-13
@@ -182,13 +182,24 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
|
||||
// MMX register moves: MOVQ M0, mem and MOVQ mem, M0 are the MMX
|
||||
// load/store pair 0F 6F/0F 7F (no prefix); a register pair takes the
|
||||
// load opcode. The XMM MOVQ forms follow below.
|
||||
// load opcode. The GPR crossings ride the MOVD opcodes with REX.W
|
||||
// (0F 6E into the bank, 0F 7E out), and an XMM source crosses into the
|
||||
// bank through the F2 0F D6 move. The XMM MOVQ forms follow below.
|
||||
if m, ok := src.(Reg); ok && m.mmx {
|
||||
switch d := dst.(type) {
|
||||
case Reg:
|
||||
if !d.mmx {
|
||||
if !d.mmx && (d.isVec() || d.fp) {
|
||||
return fmt.Errorf("MOV: MMX register moves stay inside the M bank")
|
||||
}
|
||||
if !d.mmx {
|
||||
// M → GPR: 0F 7E with REX.W, the bank in reg and the GPR in
|
||||
// r/m, the store layout the toolchain picks.
|
||||
i := &instr{opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: true}
|
||||
if err := setRM(i, m, d, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, d, src, 8); err != nil {
|
||||
return err
|
||||
@@ -204,15 +215,30 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
return fmt.Errorf("MOV: invalid MMX destination")
|
||||
}
|
||||
if m, ok := dst.(Reg); ok && m.mmx {
|
||||
srcM, ok := src.(Mem)
|
||||
if !ok {
|
||||
return fmt.Errorf("MOV: MMX load takes a memory source")
|
||||
switch src.(type) {
|
||||
case Mem, sbMem:
|
||||
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, m, src, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case Reg:
|
||||
if g := src.(Reg); g.isVec() {
|
||||
// X → M: F2 0F D6, the bank in reg, the XMM source in r/m.
|
||||
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, m, src, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
// GPR → M: 0F 6E with REX.W, the bank in reg, the GPR in r/m.
|
||||
i := &instr{opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: true}
|
||||
if err := setRM(i, m, src, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, m, srcM, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
return fmt.Errorf("MOV: MMX load takes a register or memory source")
|
||||
}
|
||||
|
||||
if srcVec || dstVec {
|
||||
@@ -323,14 +349,14 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
case Imm:
|
||||
if dstIsReg {
|
||||
v := int64(src)
|
||||
// The Go assembler compresses 64-bit moves whose immediate fits
|
||||
// a signed int32, choosing per sign:
|
||||
// The Go assembler compresses 64-bit moves whose immediate
|
||||
// fits the zero-extending 32-bit span, choosing per sign:
|
||||
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the
|
||||
// hardware, REX.B still emitted for R8-R15);
|
||||
// v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd
|
||||
// form would zero-extend and corrupt the value).
|
||||
// Out-of-range immediates keep the B8+rd imm64 form.
|
||||
if size == 8 && v >= 0 && v <= (1<<31)-1 {
|
||||
if size == 8 && v >= 0 && v <= (1<<32)-1 {
|
||||
i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)})
|
||||
i.rexB = dstReg.idx >= 8
|
||||
i.imm = le32(v)
|
||||
|
||||
Reference in new issue
Block a user