1838 lines
56 KiB
Go
1838 lines
56 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||
// SPDX-License-Identifier: BSD-3-Clause
|
||
|
||
package asm
|
||
|
||
import "fmt"
|
||
|
||
// aluOp maps an arithmetic/logic mnemonic to its base "r/m, r" opcode (for
|
||
// 16/32/64-bit; the 8-bit form is one less) and its /digit for the immediate
|
||
// forms (0x80/0x81/0x83).
|
||
var aluOp = map[string]struct {
|
||
rr byte
|
||
digit int
|
||
}{
|
||
"ADD": {0x01, 0},
|
||
"OR": {0x09, 1},
|
||
"ADC": {0x11, 2},
|
||
"SBB": {0x19, 3},
|
||
"AND": {0x21, 4},
|
||
"SUB": {0x29, 5},
|
||
"XOR": {0x31, 6},
|
||
"CMP": {0x39, 7},
|
||
}
|
||
|
||
// unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode.
|
||
// INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes
|
||
// in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4,
|
||
// DIV /6, IDIV /7; the accumulator is the implicit other operand).
|
||
var unaryOp = map[string]struct {
|
||
digit int
|
||
op byte
|
||
}{
|
||
"INC": {0, 0xFF},
|
||
"DEC": {1, 0xFF},
|
||
"NOT": {2, 0xF7},
|
||
"NEG": {3, 0xF7},
|
||
"MUL": {4, 0xF7},
|
||
"DIV": {6, 0xF7},
|
||
"IDIV": {7, 0xF7},
|
||
}
|
||
|
||
// shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the
|
||
// 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4).
|
||
var shiftOp = map[string]int{
|
||
"SHL": 4,
|
||
"SAL": 4,
|
||
"SHR": 5,
|
||
"SAR": 7,
|
||
"ROL": 0,
|
||
"ROR": 1,
|
||
"RCL": 2,
|
||
"RCR": 3,
|
||
}
|
||
|
||
// bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form;
|
||
// the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's
|
||
// opcode row.
|
||
var bitTestOp = map[string]int{
|
||
"BT": 4,
|
||
"BTS": 5,
|
||
"BTR": 6,
|
||
"BTC": 7,
|
||
}
|
||
|
||
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
|
||
// fence names carry their opcode inside the 0F AE /digit group spelled out in
|
||
// full (E8/F0/F8), and PAUSE is F3 90.
|
||
//
|
||
// LOCK, REP and REPN are the prefix statements. go tool asm encodes each as
|
||
// a standalone one-byte instruction with a PC of its own (F0, F3 and F2
|
||
// respectively), not as a prefix field merged into the next instruction: the
|
||
// statement that follows is encoded unaware of it, and nothing validates
|
||
// that the pairing is a legal one (LOCK before NOP assembles without
|
||
// complaint, each byte pinned against the toolchain). Because the bytes
|
||
// land in the stream before the following statement anyway, a LOCKed
|
||
// CMPXCHGQ encodes identically to a prefixed form.
|
||
var noOperandTable = map[string][]byte{
|
||
"CPUID": {0x0F, 0xA2},
|
||
"RDTSC": {0x0F, 0x31},
|
||
"RDTSCP": {0x0F, 0x01, 0xF9},
|
||
"SYSCALL": {0x0F, 0x05},
|
||
"XGETBV": {0x0F, 0x01, 0xD0},
|
||
"CLD": {0xFC},
|
||
"STD": {0xFD},
|
||
"PAUSE": {0xF3, 0x90},
|
||
"LFENCE": {0x0F, 0xAE, 0xE8},
|
||
"MFENCE": {0x0F, 0xAE, 0xF0},
|
||
"SFENCE": {0x0F, 0xAE, 0xF8},
|
||
"UNDEF": {0x0F, 0x0B},
|
||
"LOCK": {0xF0},
|
||
"REP": {0xF3},
|
||
"REPN": {0xF2},
|
||
}
|
||
|
||
// --- MOV --------------------------------------------------------------------
|
||
|
||
func (e *enc) encodeMov(ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("MOV expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
|
||
// Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2
|
||
// packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E
|
||
// (reg = dst, no REX.W, the Go assembler's form), xmm→mem as
|
||
// 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD
|
||
// opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and
|
||
// 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m
|
||
// would be undefined forms. MOVL is the packed-dword move:
|
||
// 66 0F 6E load, 66 0F 7E store, no REX.W. A GPR-move fallback would
|
||
// silently emit REX.W 8B with the wrong operand meaning.
|
||
_, srcVec := vecReg(src)
|
||
dstReg, dstVec := vecReg(dst)
|
||
if srcVec || dstVec {
|
||
if dstVec {
|
||
if g, ok := src.(Reg); ok && !g.isVec() {
|
||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: size == 8}
|
||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1}
|
||
if size == 4 {
|
||
i.prefix = 0x66
|
||
i.opcode = []byte{0x0F, 0x6E}
|
||
}
|
||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
srcXMM, srcIsXMM := src.(Reg)
|
||
if !srcIsXMM || !srcXMM.isVec() {
|
||
return fmt.Errorf("MOV: store needs an XMM source")
|
||
}
|
||
if g, ok := dst.(Reg); ok && !g.isVec() {
|
||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: size == 8}
|
||
if err := setRM(i, srcXMM, dst, 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
|
||
if size == 4 {
|
||
i.opcode = []byte{0x0F, 0x7E}
|
||
}
|
||
if err := setRM(i, srcXMM, dst, 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
dstReg, dstIsReg := dst.(Reg)
|
||
switch src := src.(type) {
|
||
case Reg:
|
||
if dstIsReg {
|
||
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst, the form the Go
|
||
// assembler emits for register-to-register moves.
|
||
i := newInstr(size, []byte{movRM(size)})
|
||
if err := setRM(i, src, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst(mem).
|
||
i := newInstr(size, []byte{movRM(size)})
|
||
if err := setRM(i, src, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
|
||
case Mem:
|
||
if !dstIsReg {
|
||
return fmt.Errorf("MOV: two memory operands")
|
||
}
|
||
// MOV r, r/m: reg=dst, rm=src(mem).
|
||
i := newInstr(size, []byte{movRR(size)})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
|
||
case sbMem:
|
||
if !dstIsReg {
|
||
return fmt.Errorf("MOV: two memory operands")
|
||
}
|
||
// MOV r, r/m: reg=dst, rm=src(static symbol).
|
||
i := newInstr(size, []byte{movRR(size)})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
|
||
case TLSMem:
|
||
if !dstIsReg {
|
||
return fmt.Errorf("MOV: two memory operands")
|
||
}
|
||
// MOV r, off(TLS): the segment-prefixed absolute load, reg=dst,
|
||
// rm=src(tlsMem) through the SIB escape; the disp32 is the TLS slot
|
||
// offset with its R_TLSLE patch site.
|
||
i := newInstr(size, []byte{movRR(size)})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
|
||
case SegAbs:
|
||
if !dstIsReg {
|
||
return fmt.Errorf("MOV: two memory operands")
|
||
}
|
||
// MOV r, 0x30(GS): the segment-absolute load.
|
||
i := newInstr(size, []byte{movRR(size)})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
|
||
case Imm:
|
||
if dstIsReg {
|
||
v := int64(src)
|
||
// The Go assembler compresses 64-bit moves whose immediate fits
|
||
// a signed int32, choosing per sign:
|
||
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the
|
||
// hardware, REX.B still emitted for R8-R15);
|
||
// v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd
|
||
// form would zero-extend and corrupt the value).
|
||
// Out-of-range immediates keep the B8+rd imm64 form.
|
||
if size == 8 && v >= 0 && v <= (1<<31)-1 {
|
||
i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)})
|
||
i.rexB = dstReg.idx >= 8
|
||
i.imm = le32(v)
|
||
return e.emit(i)
|
||
}
|
||
if size == 8 && v < 0 && v >= -(1<<31) {
|
||
i := newInstr(8, []byte{0xC7})
|
||
if err := setRMDigit(i, 0, dstReg, 8); err != nil {
|
||
return err
|
||
}
|
||
i.imm = le32(v)
|
||
return e.emit(i)
|
||
}
|
||
opBase := byte(0xB8)
|
||
if size == 1 {
|
||
opBase = 0xB0
|
||
}
|
||
i := newInstr(size, []byte{opBase + byte(dstReg.idx&7)})
|
||
i.rexB = dstReg.idx >= 8
|
||
if dstReg.needsREX(size) {
|
||
i.rexForced = true
|
||
}
|
||
imm, err := immediate(v, size, true)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i.imm = imm
|
||
return e.emit(i)
|
||
}
|
||
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. An immediate in the
|
||
// destination slot is the absolute-address crash-store spelling,
|
||
// MOVL $0xf1, 0xf1: the parser reads the trailing bare constant
|
||
// as an immediate, and the store's disp32 carries the address.
|
||
op := byte(0xC7)
|
||
if size == 1 {
|
||
op = 0xC6
|
||
}
|
||
if d, ok := dst.(Imm); ok {
|
||
i := newInstr(size, []byte{op})
|
||
setSegAbs(i, 0, SegAbs{Disp: int64(d)})
|
||
immBytes, err := immediate(int64(src), size, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i.imm = immBytes
|
||
return e.emit(i)
|
||
}
|
||
i := newInstr(size, []byte{op})
|
||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||
return err
|
||
}
|
||
imm, err := immediate(int64(src), size, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i.imm = imm
|
||
return e.emit(i)
|
||
}
|
||
return fmt.Errorf("MOV: invalid operands")
|
||
}
|
||
|
||
func movRR(size int) byte { // MOV r, r/m
|
||
if size == 1 {
|
||
return 0x8A
|
||
}
|
||
return 0x8B
|
||
}
|
||
|
||
func movRM(size int) byte { // MOV r/m, r
|
||
if size == 1 {
|
||
return 0x88
|
||
}
|
||
return 0x89
|
||
}
|
||
|
||
// --- ALU (ADD/OR/AND/SUB/XOR/CMP) -------------------------------------------
|
||
|
||
func (e *enc) encodeALU(op struct {
|
||
rr byte
|
||
digit int
|
||
}, ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("ALU instruction expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
|
||
// CMP never takes its immediate first: the Go assembler rejects
|
||
// CMPL $0, AX outright (only CMPL AX, $0 is legal, unlike TEST and the
|
||
// writing ALU ops whose immediate is naturally the source).
|
||
if imm, ok := src.(Imm); ok {
|
||
if op.digit == 7 {
|
||
return fmt.Errorf("CMP immediate must be the second operand (reg, $imm)")
|
||
}
|
||
return e.encodeALUImm(op.digit, dst, int64(imm), size)
|
||
}
|
||
|
||
// CMP accepts the immediate in the second position too, CMPL CX, $31 is
|
||
// the form the Go assembler itself accepts, and encodes it identically
|
||
// (CMP r/m, imm sets the flags as first − second). No other ALU op takes
|
||
// an immediate destination.
|
||
if imm, ok := dst.(Imm); ok {
|
||
if op.digit != 7 {
|
||
return fmt.Errorf("immediate must be the source operand")
|
||
}
|
||
return e.encodeALUImm(op.digit, src, int64(imm), size)
|
||
}
|
||
|
||
// CMP records first − second without writing anywhere, so the first
|
||
// operand must land as the minuend; every other ALU op writes its second
|
||
// operand and follows the forms below.
|
||
cmp := op.rr == 0x39
|
||
dstReg, dstIsReg := dst.(Reg)
|
||
srcReg, srcIsReg := src.(Reg)
|
||
switch {
|
||
case cmp && dstIsReg:
|
||
// CMP x, reg: OP r/m, r (0x38/0x39) with rm = first operand, reg =
|
||
// second, matching the Go assembler.
|
||
opc := op.rr
|
||
if size == 1 {
|
||
opc = op.rr - 1
|
||
}
|
||
i := newInstr(size, []byte{opc})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
case cmp && srcIsReg:
|
||
// CMP reg, mem: OP r, r/m (0x3A/0x3B) with reg = first operand, rm =
|
||
// second.
|
||
opc := op.rr + 2
|
||
if size == 1 {
|
||
opc = op.rr + 1
|
||
}
|
||
i := newInstr(size, []byte{opc})
|
||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
case srcIsReg:
|
||
// OP r/m, r: reg=src, rm=dst (dst is a register or memory). This is the
|
||
// form the Go assembler prefers when the source is a register.
|
||
opc := op.rr
|
||
if size == 1 {
|
||
opc = op.rr - 1
|
||
}
|
||
i := newInstr(size, []byte{opc})
|
||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
case dstIsReg:
|
||
// OP r, r/m: reg=dst, rm=src(memory).
|
||
opc := op.rr + 2
|
||
if size == 1 {
|
||
opc = op.rr + 1
|
||
}
|
||
i := newInstr(size, []byte{opc})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
return fmt.Errorf("two memory operands")
|
||
}
|
||
|
||
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||
if size == 1 {
|
||
immBytes, err := immediate(imm, 1, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
|
||
// the destination is AL, the form the Go assembler prefers here.
|
||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
|
||
i.imm = immBytes
|
||
return e.emit(i)
|
||
}
|
||
i := newInstr(1, []byte{0x80})
|
||
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
||
return err
|
||
}
|
||
i.imm = immBytes
|
||
return e.emit(i)
|
||
}
|
||
if fits8(imm) {
|
||
// 0x83 /digit, sign-extended imm8.
|
||
i := newInstr(size, []byte{0x83})
|
||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{byte(int8(imm))}
|
||
return e.emit(i)
|
||
}
|
||
// 0x81 /digit, imm16/imm32, or the Go assembler's accumulator short
|
||
// form (opcode+5, no ModR/M) when the destination is AX/AL, which it
|
||
// prefers over the generic form exactly here.
|
||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
|
||
i := newInstr(size, []byte{accOp})
|
||
immBytes, err := immediate(imm, size, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i.imm = immBytes
|
||
return e.emit(i)
|
||
}
|
||
// 0x81 /digit, imm16/imm32.
|
||
i := newInstr(size, []byte{0x81})
|
||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||
return err
|
||
}
|
||
immBytes, err := immediate(imm, size, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i.imm = immBytes
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- TEST -------------------------------------------------------------------
|
||
|
||
func (e *enc) encodeTest(ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("TEST expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
if imm, ok := src.(Imm); ok {
|
||
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0, but the Go assembler
|
||
// always uses the accumulator forms (A8/A9, no ModR/M) when the
|
||
// register operand is AL/AX, whatever the immediate's width.
|
||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||
op := byte(0xA9)
|
||
if size == 1 {
|
||
op = 0xA8
|
||
}
|
||
i := newInstr(size, []byte{op})
|
||
immBytes, err := immediate(int64(imm), size, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i.imm = immBytes
|
||
return e.emit(i)
|
||
}
|
||
op := byte(0xF7)
|
||
if size == 1 {
|
||
op = 0xF6
|
||
}
|
||
i := newInstr(size, []byte{op})
|
||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||
return err
|
||
}
|
||
immBytes, err := immediate(int64(imm), size, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i.imm = immBytes
|
||
return e.emit(i)
|
||
}
|
||
srcReg, ok := src.(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("TEST: source must be a register or immediate")
|
||
}
|
||
// TEST r/m, r: 0x84 (8-bit) / 0x85.
|
||
op := byte(0x85)
|
||
if size == 1 {
|
||
op = 0x84
|
||
}
|
||
i := newInstr(size, []byte{op})
|
||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- LEA --------------------------------------------------------------------
|
||
|
||
func (e *enc) encodeLea(ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("LEA expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1] // LEAQ addr, reg
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("LEA: destination must be a register")
|
||
}
|
||
switch src.(type) {
|
||
case Mem, sbMem:
|
||
default:
|
||
return fmt.Errorf("LEA: source must be a memory operand")
|
||
}
|
||
i := newInstr(size, []byte{0x8D})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- INC/DEC/NEG/NOT --------------------------------------------------------
|
||
|
||
func (e *enc) encodeUnary(op struct {
|
||
digit int
|
||
op byte
|
||
}, ops []Operand, size int) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("unary instruction expects 1 operand, got %d", len(ops))
|
||
}
|
||
base := op.op
|
||
if size == 1 {
|
||
base-- // 0xFF→0xFE, 0xF7→0xF6
|
||
}
|
||
i := newInstr(size, []byte{base})
|
||
if err := setRMDigit(i, op.digit, ops[0], size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- SHL/SHR/SAR ------------------------------------------------------------
|
||
|
||
// doubleShiftOp maps the two mnemonics whose three-operand form go tool asm
|
||
// accepts to the SHLD/SHRD opcode pair (imm8 form, CL form). SAR, SAL and
|
||
// the rotates have no such form: the oracle rejects SARQ/ROLQ with three
|
||
// operands, and so do we.
|
||
var doubleShiftOp = map[string][2]byte{
|
||
"SHL": {0xA4, 0xA5}, // SHLD
|
||
"SHR": {0xAC, 0xAD}, // SHRD
|
||
}
|
||
|
||
// isShiftCountCL reports whether a count operand is the CL register or its
|
||
// CX spelling: go tool asm accepts both (CX names the same low byte) and
|
||
// rejects ECX/RCX.
|
||
func isShiftCountCL(o Operand) bool {
|
||
reg, ok := o.(Reg)
|
||
return ok && reg.idx == 1 && (reg.size == 1 || reg.size == 2)
|
||
}
|
||
|
||
func (e *enc) encodeShift(base string, ops []Operand, size int) error {
|
||
digit := shiftOp[base]
|
||
if len(ops) == 3 {
|
||
return e.encodeDoubleShift(base, ops, size)
|
||
}
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("shift expects 2 operands, got %d", len(ops))
|
||
}
|
||
count, dst := ops[0], ops[1]
|
||
// Count is $1, CL (or its CX spelling), or an imm8.
|
||
if isShiftCountCL(count) {
|
||
// CL: 0xD2 (8-bit) / 0xD3.
|
||
op := byte(0xD3)
|
||
if size == 1 {
|
||
op = 0xD2
|
||
}
|
||
i := newInstr(size, []byte{op})
|
||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
imm, ok := count.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("shift count must be $1, CL or an immediate")
|
||
}
|
||
if imm == 1 {
|
||
// 0xD0 (8-bit) / 0xD1.
|
||
op := byte(0xD1)
|
||
if size == 1 {
|
||
op = 0xD0
|
||
}
|
||
i := newInstr(size, []byte{op})
|
||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
// 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm
|
||
// rejects negative and ≥256 counts, and the hardware masks the count, so
|
||
// a silent truncation ($300 encoding 44) would shift by a different
|
||
// amount than the source states.
|
||
if imm < 0 || imm > 255 {
|
||
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
|
||
}
|
||
op := byte(0xC1)
|
||
if size == 1 {
|
||
op = 0xC0
|
||
}
|
||
i := newInstr(size, []byte{op})
|
||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{byte(imm)}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeDoubleShift emits the three-operand SHL/SHR form, which the Go
|
||
// assembler spells as a shift but encodes as SHLD/SHRD (0F A4/A5, 0F AC/AD):
|
||
// the first operand is the count ($imm or CL), the second feeds the vacated
|
||
// bits (the reg field) and the third is the shifted value (the r/m field),
|
||
// matching go tool asm byte for byte. The W/L/Q widths exist; the oracle
|
||
// rejects the three-operand B form and every SAR/rotate one.
|
||
func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
|
||
opc, ok := doubleShiftOp[base]
|
||
if !ok || size == 1 {
|
||
return fmt.Errorf("%s: shift expects 2 operands, got %d", base, len(ops))
|
||
}
|
||
count, src, dst := ops[0], ops[1], ops[2]
|
||
srcReg, ok := src.(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("%s: middle operand must be a register, like go tool asm", base)
|
||
}
|
||
i := newInstr(size, []byte{0x0F, opc[0]})
|
||
if isShiftCountCL(count) {
|
||
// CL (or CX) form: 0F A5/AD.
|
||
i.opcode[1] = opc[1]
|
||
} else {
|
||
imm, ok := count.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("shift count must be $1, CL or an immediate")
|
||
}
|
||
// The count is an unsigned imm8: the same range convention as the
|
||
// two-operand shift above.
|
||
if imm < 0 || imm > 255 {
|
||
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
|
||
}
|
||
i.imm = []byte{byte(imm)}
|
||
}
|
||
if err := setRMReg(i, srcReg.idx, srcReg.idx >= 8, false, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- IMUL -------------------------------------------------------------------
|
||
|
||
func (e *enc) encodeImul(ops []Operand, size int) error {
|
||
switch len(ops) {
|
||
case 2:
|
||
// Two shapes. The leading-immediate spelling IMUL $imm, r multiplies
|
||
// r in place (dst = rm = r): the shape GOROOT's clock code writes.
|
||
// Otherwise IMUL r, r/m: 0x0F 0xAF.
|
||
if imm, ok := ops[0].(Imm); ok {
|
||
dstReg, isReg := ops[1].(Reg)
|
||
if !isReg {
|
||
return fmt.Errorf("IMUL: destination must be a register")
|
||
}
|
||
return e.encodeImulImm(imm, dstReg, dstReg, size)
|
||
}
|
||
dstReg, ok := ops[1].(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("IMUL: destination must be a register")
|
||
}
|
||
i := newInstr(size, []byte{0x0F, 0xAF})
|
||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
case 3:
|
||
// IMUL r, r/m, imm: 0x6B (imm8) / 0x69 (imm16/32).
|
||
dstReg, ok := ops[2].(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("IMUL: destination must be a register")
|
||
}
|
||
imm, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("IMUL: immediate operand expected first")
|
||
}
|
||
// Plan 9 order: IMUL $imm, src, dst; the source stays a general
|
||
// r/m operand (setRM takes registers and memory alike).
|
||
return e.encodeImulImm(imm, ops[1], dstReg, size)
|
||
}
|
||
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
||
}
|
||
|
||
// encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8
|
||
// when the value fits, 0x69 with a 32-bit immediate otherwise.
|
||
func (e *enc) encodeImulImm(imm Imm, rm Operand, dst Reg, size int) error {
|
||
if fits8(int64(imm)) {
|
||
i := newInstr(size, []byte{0x6B})
|
||
if err := setRM(i, dst, rm, size); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{byte(int8(imm))}
|
||
return e.emit(i)
|
||
}
|
||
i := newInstr(size, []byte{0x69})
|
||
if err := setRM(i, dst, rm, size); err != nil {
|
||
return err
|
||
}
|
||
immBytes, err := immediate(int64(imm), size, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i.imm = immBytes
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- PUSH / POP -------------------------------------------------------------
|
||
|
||
func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
|
||
}
|
||
// In 64-bit mode go tool asm knows the 64-bit push (the default, with or
|
||
// without the Q suffix) and the 16-bit W form with its 0x66 operand-size
|
||
// prefix, and rejects the B and L spellings outright ("illegal in 64-bit
|
||
// mode"); silently widening those would push a different width than the
|
||
// source states.
|
||
switch size {
|
||
case 0, 8, 2:
|
||
default:
|
||
return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode")
|
||
}
|
||
w16 := size == 2
|
||
switch op := ops[0].(type) {
|
||
case Reg:
|
||
base := byte(0x50) // PUSH r; POP is 0x58
|
||
if !push {
|
||
base = 0x58
|
||
}
|
||
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
|
||
i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
|
||
i.rexB = op.idx >= 8
|
||
return e.emit(i)
|
||
case Mem:
|
||
opc := byte(0xFF) // PUSH r/m: /6
|
||
digit := 6
|
||
if !push {
|
||
opc = 0x8F // POP r/m: /0
|
||
digit = 0
|
||
}
|
||
i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1}
|
||
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
case Imm:
|
||
if !push {
|
||
return fmt.Errorf("POP does not take an immediate")
|
||
}
|
||
if fits8(int64(op)) {
|
||
i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
|
||
return e.emit(i)
|
||
}
|
||
// PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
|
||
// immediate by the same signed/unsigned 32-bit span as every other
|
||
// scalar immediate.
|
||
immBytes, err := immediate(int64(op), 8, false)
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes}
|
||
return e.emit(i)
|
||
}
|
||
return fmt.Errorf("PUSH/POP: invalid operand")
|
||
}
|
||
|
||
// --- RET / JMP / CALL / Jcc -------------------------------------------------
|
||
|
||
func (e *enc) encodeRet() error {
|
||
return e.emit(&instr{opcode: []byte{0xC3}, modrm: -1, sib: -1})
|
||
}
|
||
|
||
// encodeJmpRel encodes JMP/CALL with a relative displacement (the operand is an
|
||
// Imm holding the already-computed rel32 offset).
|
||
func (e *enc) encodeJmpRel(ops []Operand, opcode []byte) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("JMP/CALL expects 1 operand, got %d", len(ops))
|
||
}
|
||
imm, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("JMP/CALL: relative offset must be an immediate (labels are resolved by the assembler)")
|
||
}
|
||
return e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, imm: le32(int64(imm))})
|
||
}
|
||
|
||
// encodeIndirectBranch encodes JMP/CALL through a register or memory operand:
|
||
// FF /4 for JMP, FF /2 for CALL. The operand size is fixed at 64 bits in
|
||
// 64-bit mode, so no REX.W is emitted; a REX appears only for R8-R15 bases.
|
||
func (e *enc) encodeIndirectBranch(mnem string, ops []Operand) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
|
||
}
|
||
digit := 4 // JMP r/m64
|
||
if mnem == "CALL" {
|
||
digit = 2 // CALL r/m64
|
||
}
|
||
i := &instr{opcode: []byte{0xFF}, modrm: -1, sib: -1}
|
||
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// condCode maps a Plan 9 conditional-jump mnemonic to its x86 condition code.
|
||
func condCode(upper string) (int, bool) {
|
||
if len(upper) < 2 || upper[0] != 'J' || upper == "JMP" {
|
||
return 0, false
|
||
}
|
||
cc, ok := jccMap[upper[1:]]
|
||
return cc, ok
|
||
}
|
||
|
||
var jccMap = map[string]int{
|
||
"O": 0x0, "NO": 0x1, "OS": 0x0, "OC": 0x1,
|
||
"B": 0x2, "C": 0x2, "NAE": 0x2, "CS": 0x2,
|
||
"NB": 0x3, "NC": 0x3, "AE": 0x3, "CC": 0x3,
|
||
"E": 0x4, "Z": 0x4, "EQ": 0x4,
|
||
"NE": 0x5, "NZ": 0x5,
|
||
"BE": 0x6, "NA": 0x6, "LS": 0x6,
|
||
"NBE": 0x7, "A": 0x7, "HI": 0x7,
|
||
"S": 0x8, "MI": 0x8,
|
||
"NS": 0x9, "PL": 0x9,
|
||
"P": 0xA, "PE": 0xA, "PS": 0xA,
|
||
"NP": 0xB, "PO": 0xB, "PC": 0xB,
|
||
"L": 0xC, "NGE": 0xC, "LT": 0xC,
|
||
"NL": 0xD, "GE": 0xD,
|
||
"LE": 0xE, "NG": 0xE,
|
||
"NLE": 0xF, "G": 0xF, "GT": 0xF,
|
||
}
|
||
|
||
func (e *enc) encodeJcc(cc int, ops []Operand) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("conditional jump expects 1 operand, got %d", len(ops))
|
||
}
|
||
imm, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("conditional jump: relative offset must be an immediate")
|
||
}
|
||
if fits8(int64(imm)) {
|
||
// Short form: 0x70+cc, rel8.
|
||
return e.emit(&instr{opcode: []byte{0x70 + byte(cc)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
|
||
}
|
||
// Near form: 0x0F 0x80+cc, rel32.
|
||
return e.emit(&instr{opcode: []byte{0x0F, 0x80 + byte(cc)}, modrm: -1, sib: -1, imm: le32(int64(imm))})
|
||
}
|
||
|
||
// immediate encodes an immediate of the given operand size. full64 selects the
|
||
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
|
||
// sign-extended immediate is used for 64-bit operands.
|
||
//
|
||
// The span mirrors go tool asm: every scalar immediate must fit a signed or
|
||
// unsigned 32-bit word, and the narrower fields then take the low bits
|
||
// silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the
|
||
// imm64 form may exceed the span; anything wider elsewhere is an error rather
|
||
// than a truncation the source never asked for.
|
||
func immediate(v int64, size int, full64 bool) ([]byte, error) {
|
||
if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) {
|
||
return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v)
|
||
}
|
||
switch size {
|
||
case 1:
|
||
return []byte{byte(int8(v))}, nil
|
||
case 2:
|
||
return le16(v), nil
|
||
case 4:
|
||
return le32(v), nil
|
||
default: // 8
|
||
if full64 {
|
||
return le64(v), nil
|
||
}
|
||
return le32(v), nil // sign-extended imm32
|
||
}
|
||
}
|
||
|
||
// --- CMOVcc / SETcc ---------------------------------------------------------
|
||
|
||
// encodeCmov encodes a conditional move: CMOV + size (W/L/Q) + condition
|
||
// (CMOVLGT, CMOVQEQ, …). The condition reads exactly like the Jcc spellings;
|
||
// the instruction is 0F 40+cc with reg = dst, rm = src.
|
||
func (e *enc) encodeCmov(upper string, ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("CMOVcc expects 2 operands, got %d", len(ops))
|
||
}
|
||
rest := upper[len("CMOV"):]
|
||
if len(rest) < 2 {
|
||
return fmt.Errorf("unsupported instruction %q", upper)
|
||
}
|
||
var size int
|
||
switch rest[0] {
|
||
case 'W':
|
||
size = 2
|
||
case 'L':
|
||
size = 4
|
||
case 'Q':
|
||
size = 8
|
||
default:
|
||
return fmt.Errorf("unsupported instruction %q", upper)
|
||
}
|
||
cc, ok := jccMap[rest[1:]]
|
||
if !ok {
|
||
return fmt.Errorf("unsupported instruction %q", upper)
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("CMOVcc destination must be a register")
|
||
}
|
||
i := newInstr(size, []byte{0x0F, byte(0x40 + cc)})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …),
|
||
// always a byte write, 0F 90+cc /0 into a register or memory operand.
|
||
func (e *enc) encodeSet(upper string, ops []Operand) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops))
|
||
}
|
||
cond := upper[len("SET"):]
|
||
cc, ok := jccMap[cond]
|
||
if !ok || cond == "" {
|
||
return fmt.Errorf("unsupported instruction %q", upper)
|
||
}
|
||
i := &instr{opcode: []byte{0x0F, byte(0x90 + cc)}, modrm: -1, sib: -1}
|
||
if err := setRMDigit(i, 0, ops[0], 1); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- bit scan / bit count ----------------------------------------------------
|
||
|
||
// countOp maps the bit-scan and bit-count mnemonics to their opcode byte and
|
||
// mandatory prefix. TZCNT/LZCNT/POPCNT are the F3-prefixed forms of the
|
||
// same map as BSF/BSR's 0F BC/BD; POPCNT is F3 0F B8.
|
||
var countOp = map[string]struct {
|
||
op byte
|
||
prefix byte
|
||
}{
|
||
"BSF": {0xBC, 0},
|
||
"BSR": {0xBD, 0},
|
||
"TZCNT": {0xBC, 0xF3},
|
||
"LZCNT": {0xBD, 0xF3},
|
||
"POPCNT": {0xB8, 0xF3},
|
||
}
|
||
|
||
// encodeCount encodes the bit-scan and bit-count family, BSF (0F BC),
|
||
// BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8)
|
||
// with reg = dst and rm = src. The size suffix selects the operand width
|
||
// (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the
|
||
// source is zero (unlike their F3-prefixed counterparts); callers must
|
||
// guard non-zero inputs themselves.
|
||
func (e *enc) encodeCount(base string, ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||
}
|
||
spec := countOp[base]
|
||
dstReg, ok := ops[1].(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("%s destination must be a register", base)
|
||
}
|
||
i := newInstr(size, []byte{0x0F, spec.op})
|
||
i.prefix = spec.prefix
|
||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeBswap encodes BSWAP: the single register operand is encoded in the
|
||
// opcode byte (0F C8+r), with REX.B for R8-R15 and REX.W for the quad form.
|
||
func (e *enc) encodeBswap(ops []Operand, size int) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("BSWAP expects 1 operand, got %d", len(ops))
|
||
}
|
||
reg, ok := ops[0].(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("BSWAP operand must be a register")
|
||
}
|
||
i := newInstr(size, []byte{0x0F, 0xC8 + byte(reg.idx&7)})
|
||
i.rexB = reg.idx >= 8
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- mixed-width sign/zero-extending moves -----------------------------------
|
||
|
||
// movExtendOp maps Go's mixed-width move names to their opcode and destination
|
||
// width. The source is narrower than the destination, so the plain size-suffix
|
||
// convention does not apply to these names.
|
||
var movExtendOp = map[string]struct {
|
||
op []byte
|
||
dstSize int
|
||
}{
|
||
"MOVBLZX": {[]byte{0x0F, 0xB6}, 4}, // byte → long, zero-extend
|
||
"MOVBQZX": {[]byte{0x0F, 0xB6}, 8}, // byte → quad, zero-extend
|
||
"MOVWLZX": {[]byte{0x0F, 0xB7}, 4}, // word → long, zero-extend
|
||
"MOVWQZX": {[]byte{0x0F, 0xB7}, 8}, // word → quad, zero-extend
|
||
"MOVWLSX": {[]byte{0x0F, 0xBF}, 4}, // word → long, sign-extend
|
||
"MOVLQSX": {[]byte{0x63}, 8}, // long → quad, sign-extend (MOVSXD)
|
||
"MOVBWZX": {[]byte{0x0F, 0xB6}, 2}, // byte → word, zero-extend
|
||
"MOVBWSX": {[]byte{0x0F, 0xBE}, 2}, // byte → word, sign-extend
|
||
"MOVBLSX": {[]byte{0x0F, 0xBE}, 4}, // byte → long, sign-extend
|
||
"MOVBQSX": {[]byte{0x0F, 0xBE}, 8}, // byte → quad, sign-extend
|
||
"MOVWQSX": {[]byte{0x0F, 0xBF}, 8}, // word → quad, sign-extend
|
||
// A long → quad zero-extend is a plain 32-bit move: every 32-bit
|
||
// operation zero-extends its result into the full register, so the
|
||
// toolchain lowers MOVLQZX to the plain MOVL encoding.
|
||
"MOVLQZX": {[]byte{0x8B}, 4},
|
||
}
|
||
|
||
// encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider
|
||
// operand), rm = src.
|
||
func (e *enc) encodeMovExtend(base string, ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||
}
|
||
spec := movExtendOp[base]
|
||
dstReg, ok := ops[1].(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("%s destination must be a register", base)
|
||
}
|
||
i := newInstr(spec.dstSize, spec.op)
|
||
if err := setRM(i, dstReg, ops[0], spec.dstSize); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodePmovmskb encodes PMOVMSKB, the legacy SSE2 byte mask extract: the
|
||
// XMM source's sign bytes pack into a GP destination, 66 0F D7 /r.
|
||
func (e *enc) encodePmovmskb(base string, ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||
}
|
||
srcReg, srcVec := vecReg(ops[0])
|
||
if !srcVec {
|
||
return fmt.Errorf("%s source must be an XMM register", base)
|
||
}
|
||
dstReg, ok := ops[1].(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("%s destination must be a register", base)
|
||
}
|
||
i := newInstr(4, []byte{0x0F, 0xD7})
|
||
i.prefix = 0x66
|
||
if err := setRM(i, dstReg, srcReg, 4); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- legacy SSE moves --------------------------------------------------------
|
||
|
||
// sseMove describes a legacy (non-VEX) SSE move: a mandatory prefix plus a
|
||
// load opcode (reg = destination, rm = source) and a store opcode (the
|
||
// reverse). The Plan 9 names MOVOU/MOVO are the integer unaligned/aligned
|
||
// octa moves (MOVDQU/MOVDQA), not the packed-single ones.
|
||
type sseMove struct {
|
||
prefix byte // 0, 0x66, 0xF2 or 0xF3
|
||
load byte
|
||
store byte
|
||
}
|
||
|
||
var sseMoveTable = map[string]sseMove{
|
||
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
|
||
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
|
||
"MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias
|
||
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
||
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
||
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
||
"MOVAPD": {0x66, 0x28, 0x29}, // aligned packed double
|
||
"MOVSD": {0xF2, 0x10, 0x11}, // scalar double
|
||
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
|
||
}
|
||
|
||
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
|
||
// load form (reg = destination), matching the Go assembler.
|
||
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("SSE move expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, srcVec := vecReg(src)
|
||
dstReg, dstVec := vecReg(dst)
|
||
op := m.store
|
||
var reg Reg
|
||
var rm Operand
|
||
switch {
|
||
case srcVec && dstVec:
|
||
op = m.load
|
||
reg, rm = dstReg, src
|
||
case srcVec:
|
||
if !isX86Mem(dst) {
|
||
return fmt.Errorf("SSE move: invalid destination operand")
|
||
}
|
||
reg, rm = srcReg, dst
|
||
case dstVec:
|
||
if !isX86Mem(src) {
|
||
return fmt.Errorf("SSE move: invalid source operand")
|
||
}
|
||
op = m.load
|
||
reg, rm = dstReg, src
|
||
default:
|
||
return fmt.Errorf("SSE move needs a vector register operand")
|
||
}
|
||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||
if err := setRM(i, reg, rm, 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- legacy SSE packed binary and shuffles -----------------------------------
|
||
|
||
// sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an
|
||
// optional mandatory prefix plus the 0F-prefixed opcode (0F38 for the
|
||
// SSSE3 integer shuffles). Plan 9 asm lists the source operand first, so
|
||
// MULPS X0, X1 computes X1 = X1 * X0.
|
||
type sseBin struct {
|
||
prefix byte // 0, 0x66, 0xF2 or 0xF3
|
||
op byte
|
||
map38 bool // opcode lives under 0F38 instead of 0F
|
||
}
|
||
|
||
var sseBinTable = map[string]sseBin{
|
||
"ADDPS": {0, 0x58, false}, "ADDPD": {0x66, 0x58, false},
|
||
"MULPS": {0, 0x59, false}, "MULPD": {0x66, 0x59, false},
|
||
"SUBPS": {0, 0x5C, false}, "SUBPD": {0x66, 0x5C, false},
|
||
"DIVPS": {0, 0x5E, false}, "DIVPD": {0x66, 0x5E, false},
|
||
"ANDPS": {0, 0x54, false}, "ANDPD": {0x66, 0x54, false},
|
||
"ORPS": {0, 0x56, false}, "ORPD": {0x66, 0x56, false},
|
||
"XORPS": {0, 0x57, false}, "XORPD": {0x66, 0x57, false},
|
||
"MINPS": {0, 0x5D, false}, "MINPD": {0x66, 0x5D, false},
|
||
"MAXPS": {0, 0x5F, false}, "MAXPD": {0x66, 0x5F, false},
|
||
"ADDSS": {0xF3, 0x58, false}, "ADDSD": {0xF2, 0x58, false},
|
||
"MULSS": {0xF3, 0x59, false}, "MULSD": {0xF2, 0x59, false},
|
||
"SUBSS": {0xF3, 0x5C, false}, "SUBSD": {0xF2, 0x5C, false},
|
||
"DIVSS": {0xF3, 0x5E, false}, "DIVSD": {0xF2, 0x5E, false},
|
||
"MINSS": {0xF3, 0x5D, false}, "MINSD": {0xF2, 0x5D, false},
|
||
"MAXSS": {0xF3, 0x5F, false}, "MAXSD": {0xF2, 0x5F, false},
|
||
"UNPCKLPS": {0, 0x14, false}, "UNPCKHPS": {0, 0x15, false},
|
||
"UNPCKLPD": {0x66, 0x14, false}, "UNPCKHPD": {0x66, 0x15, false},
|
||
"CVTSS2SD": {0xF3, 0x5A, false}, "CVTSD2SS": {0xF2, 0x5A, false},
|
||
"CVTPS2PD": {0, 0x5A, false}, "CVTPD2PS": {0x66, 0x5A, false},
|
||
// SSE2 packed integers (reg = reg op rm) and the SSSE3 byte shuffle.
|
||
"PXOR": {0x66, 0xEF, false},
|
||
"POR": {0x66, 0xEB, false},
|
||
"PAND": {0x66, 0xDB, false},
|
||
"PANDN": {0x66, 0xDF, false},
|
||
"PADDB": {0x66, 0xFC, false}, "PADDW": {0x66, 0xFD, false},
|
||
"PADDD": {0x66, 0xFE, false}, "PADDQ": {0x66, 0xD4, false},
|
||
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
||
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
||
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
||
"PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false},
|
||
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
||
"PCMPGTD": {0x66, 0x66, false},
|
||
"PSHUFB": {0x66, 0x00, true},
|
||
// Scalar compares and square root, packed adds/subtracts and the byte
|
||
// unpack, the spellings the Plan 9 table uses (COMISD orders the
|
||
// operands like every other two-operand form).
|
||
"ANDNPD": {0x66, 0x55, false},
|
||
"ANDNPS": {0x00, 0x55, false},
|
||
"COMISD": {0x66, 0x2F, false},
|
||
"SQRTSD": {0xF2, 0x51, false},
|
||
"PADDL": {0x66, 0xFE, false},
|
||
"PSUBL": {0x66, 0xFA, false},
|
||
"PUNPCKLBW": {0x66, 0x60, false},
|
||
// AES round functions (66 0F38) and the SHA message schedule helpers
|
||
// (no prefix, 0F38).
|
||
"AESENC": {0x66, 0xDC, true},
|
||
"AESENCLAST": {0x66, 0xDD, true},
|
||
"AESDEC": {0x66, 0xDE, true},
|
||
"AESDECLAST": {0x66, 0xDF, true},
|
||
"AESIMC": {0x66, 0xDB, true},
|
||
"SHA1MSG1": {0x00, 0xC9, true},
|
||
"SHA1MSG2": {0x00, 0xCA, true},
|
||
"SHA1NEXTE": {0x00, 0xC8, true},
|
||
"SHA256MSG1": {0x00, 0xCC, true},
|
||
"SHA256MSG2": {0x00, 0xCD, true},
|
||
}
|
||
|
||
// sseImm3 describes a legacy SSE instruction taking a leading imm8 and two
|
||
// further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and
|
||
// map3A select the opcode map the same way as sseBin's.
|
||
type sseImm3 struct {
|
||
prefix byte
|
||
op byte
|
||
map3A bool // opcode lives under 0F3A instead of 0F38
|
||
}
|
||
|
||
// sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3
|
||
// align/blend shuffles, the string compare, carry-less multiply and the AES
|
||
// key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings.
|
||
var sseImm3Table = map[string]sseImm3{
|
||
"PALIGNR": {0x66, 0x0F, true},
|
||
"PBLENDW": {0x66, 0x0E, true},
|
||
"PCMPESTRI": {0x66, 0x61, true},
|
||
"PCLMULQDQ": {0x66, 0x44, true},
|
||
"AESKEYGENASSIST": {0x66, 0xDF, true},
|
||
"SHA1RNDS4": {0x00, 0xCC, true},
|
||
}
|
||
|
||
// sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM
|
||
// source and rm = the destination (GPR or memory). PEXTRW's GPR destination
|
||
// uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one,
|
||
// so it carries both opcodes.
|
||
type sseExtract struct {
|
||
op []byte
|
||
opMem []byte // used when the destination is memory; nil shares op
|
||
rexW bool // PEXTRQ's REX.W
|
||
}
|
||
|
||
var sseExtractTable = map[string]sseExtract{
|
||
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
|
||
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
|
||
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
|
||
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
|
||
}
|
||
|
||
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
|
||
// destination and rm = the source (GPR or memory).
|
||
type sseInsert struct {
|
||
op []byte
|
||
rexW bool // PINSRQ's REX.W
|
||
}
|
||
|
||
var sseInsertTable = map[string]sseInsert{
|
||
"PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false},
|
||
"PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false},
|
||
"PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true},
|
||
"PINSRW": {[]byte{0x0F, 0xC4}, false},
|
||
}
|
||
|
||
// sseShiftImm maps the legacy packed integer shifts' immediate form:
|
||
// OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L
|
||
// (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ.
|
||
var sseShiftImm = map[string]sseShift{
|
||
"PSLLW": {0x71, 6},
|
||
"PSRLW": {0x71, 2},
|
||
"PSRAW": {0x71, 4},
|
||
"PSLLL": {0x72, 6},
|
||
"PSRLL": {0x72, 2},
|
||
"PSRAL": {0x72, 4},
|
||
"PSLLQ": {0x73, 6},
|
||
"PSRLQ": {0x73, 2},
|
||
"PSLLDQ": {0x73, 7},
|
||
"PSRLDQ": {0x73, 3},
|
||
}
|
||
|
||
// sseShiftVar maps the variable-count forms (the count comes from an XMM
|
||
// register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no
|
||
// variable form.
|
||
var sseShiftVar = map[string]byte{
|
||
"PSLLW": 0xF1,
|
||
"PSRLW": 0xD1,
|
||
"PSRAW": 0xE1,
|
||
"PSLLL": 0xF2,
|
||
"PSRLL": 0xD2,
|
||
"PSRAL": 0xE2,
|
||
"PSLLQ": 0xF3,
|
||
"PSRLQ": 0xD3,
|
||
}
|
||
|
||
// sseShift is one /digit selector in the 0F 71/72/73 immediate group.
|
||
type sseShift struct {
|
||
op byte
|
||
digit int
|
||
}
|
||
|
||
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
||
// (PSHUFD/PSHUFHW/PSHUFLW also carry the packed-int 0x66/F3/F2 prefixes).
|
||
type sseShuf struct {
|
||
prefix byte
|
||
op byte
|
||
}
|
||
|
||
var sseShufTable = map[string]sseShuf{
|
||
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
||
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
||
"PSHUFL": {0x66, 0x70},
|
||
}
|
||
|
||
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
||
func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("SSE binary expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("SSE binary destination must be a vector register")
|
||
}
|
||
opcode := []byte{0x0F, m.op}
|
||
if m.map38 {
|
||
opcode = []byte{0x0F, 0x38, m.op}
|
||
}
|
||
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
|
||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeSSEShuf encodes an imm8 shuffle: SHUFPS $imm, src, dst.
|
||
func (e *enc) encodeSSEShuf(m sseShuf, ops []Operand) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("SSE shuffle expects 3 operands, got %d", len(ops))
|
||
}
|
||
imm, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("SSE shuffle needs an imm8 first operand")
|
||
}
|
||
if imm < -128 || imm > 255 {
|
||
return fmt.Errorf("SSE shuffle imm8 %d out of range", imm)
|
||
}
|
||
src, dst := ops[1], ops[2]
|
||
dstReg, ok2 := dst.(Reg)
|
||
if !ok2 || !dstReg.isVec() {
|
||
return fmt.Errorf("SSE shuffle destination must be a vector register")
|
||
}
|
||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{byte(int8(imm))}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
|
||
|
||
// encodeCvtsi2sd encodes a signed integer to scalar double conversion
|
||
// (CVTSL2SD from a 32-bit, CVTSQ2SD from a 64-bit source): F2 0F 2A with
|
||
// reg = XMM dst, rm = GPR/memory src. The Go assembler emits the legacy SSE
|
||
// encoding here, not the VEX form, so we match it byte for byte.
|
||
func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("CVTSx2SD expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("CVTSx2SD destination must be a vector register")
|
||
}
|
||
size := 4
|
||
if quad {
|
||
size = 8
|
||
}
|
||
i := newInstr(size, []byte{0x0F, 0x2A})
|
||
i.prefix = 0xF2
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- carry, bit test, exchange and accumulate -------------------------------
|
||
|
||
// encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9
|
||
// order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with
|
||
// reg = index, rm = target); an immediate index uses 0F BA /digit with imm8.
|
||
func (e *enc) encodeBitTest(name string, ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||
}
|
||
digit := bitTestOp[name]
|
||
index, target := ops[0], ops[1]
|
||
if reg, ok := index.(Reg); ok {
|
||
// Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC),
|
||
// the /digit base plus eight per step.
|
||
i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3})
|
||
if err := setRM(i, reg, target, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
imm, ok := index.(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("%s index must be a register or an immediate", name)
|
||
}
|
||
immByte, err := imm8(int64(imm))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i := newInstr(size, []byte{0x0F, 0xBA})
|
||
if err := setRMDigit(i, digit, target, size); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{immByte}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeExchange encodes XCHG. A register-to-register exchange where either
|
||
// operand is AX uses the 0x90+r accumulator form (with REX.W for the quad
|
||
// form, as the Go assembler emits it); everything else uses 0x86/0x87 with
|
||
// the register operand in ModRM.reg, the memory (or second register) in r/m.
|
||
func (e *enc) encodeExchange(ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, srcIsReg := src.(Reg)
|
||
dstReg, dstIsReg := dst.(Reg)
|
||
if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) {
|
||
// 0x90+r: r is the non-AX register, whichever side it sits on.
|
||
r := dstReg
|
||
if srcReg.idx == 0 {
|
||
r = dstReg
|
||
} else {
|
||
r = srcReg
|
||
}
|
||
i := newInstr(size, []byte{0x90 + byte(r.idx&7)})
|
||
i.rexB = r.idx >= 8
|
||
return e.emit(i)
|
||
}
|
||
op := byte(0x87)
|
||
if size == 1 {
|
||
op = 0x86
|
||
}
|
||
switch {
|
||
case srcIsReg:
|
||
i := newInstr(size, []byte{op})
|
||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
case dstIsReg:
|
||
i := newInstr(size, []byte{op})
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
return fmt.Errorf("XCHG: at least one operand must be a register")
|
||
}
|
||
|
||
// encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG
|
||
// (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the
|
||
// destination writable (register or memory).
|
||
func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||
}
|
||
srcReg, ok := ops[0].(Reg)
|
||
if !ok {
|
||
return fmt.Errorf("%s source must be a register", name)
|
||
}
|
||
opc := op
|
||
if size == 1 {
|
||
opc = op8
|
||
}
|
||
i := newInstr(size, []byte{0x0F, opc})
|
||
if err := setRM(i, srcReg, ops[1], size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for
|
||
// the rest; the word form carries a 0x66 operand-size prefix (66 F2, the
|
||
// prefix order the Go assembler emits) and the quad form REX.W. reg = GPR
|
||
// accumulator, rm = the data source.
|
||
func (e *enc) encodeCrc32(ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops))
|
||
}
|
||
dstReg, ok := ops[1].(Reg)
|
||
if !ok || dstReg.isVec() {
|
||
return fmt.Errorf("CRC32 destination must be a general register")
|
||
}
|
||
i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1}
|
||
if size > 1 {
|
||
i.opcode[2] = 0xF1
|
||
}
|
||
i.rexW = size == 8
|
||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg =
|
||
// destination, rm = source, the carry/overflow flag as the carry-in.
|
||
func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops))
|
||
}
|
||
dstReg, ok := ops[1].(Reg)
|
||
if !ok || dstReg.isVec() {
|
||
return fmt.Errorf("ADCX/ADOX destination must be a general register")
|
||
}
|
||
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8}
|
||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// --- string primitives, flags and INT ----------------------------------------
|
||
|
||
// encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and
|
||
// STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or
|
||
// REX.W prefix.
|
||
func (e *enc) encodeStringOp(base string, ops []Operand, size int) error {
|
||
if len(ops) != 0 {
|
||
return fmt.Errorf("%s takes no operands, got %d", base, len(ops))
|
||
}
|
||
var op byte
|
||
switch base {
|
||
case "MOVS":
|
||
op = 0xA5
|
||
if size == 1 {
|
||
op = 0xA4
|
||
}
|
||
case "STOS":
|
||
op = 0xAB
|
||
if size == 1 {
|
||
op = 0xAA
|
||
}
|
||
default:
|
||
return fmt.Errorf("unsupported string instruction %q", base)
|
||
}
|
||
return e.emit(newInstr(size, []byte{op}))
|
||
}
|
||
|
||
// encodeInt encodes INT with its single imm8 operand. The field takes the
|
||
// low byte silently inside the 32-bit span, matching the scalar convention
|
||
// (go tool asm encodes INT $256 as CD 00).
|
||
func (e *enc) encodeInt(ops []Operand) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("INT expects 1 operand, got %d", len(ops))
|
||
}
|
||
imm, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("INT operand must be an immediate")
|
||
}
|
||
if imm < -(1<<31) || imm > (1<<32)-1 {
|
||
return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm))
|
||
}
|
||
return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}})
|
||
}
|
||
|
||
// encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a
|
||
// single 32-bit memory operand.
|
||
func (e *enc) encodeMxcsr(digit int, ops []Operand) error {
|
||
if len(ops) != 1 {
|
||
return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops))
|
||
}
|
||
m, ok := ops[0].(Mem)
|
||
if !ok {
|
||
return fmt.Errorf("MXCSR instruction requires a memory operand")
|
||
}
|
||
i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1}
|
||
if err := setMem(i, digit, m); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// cvtIntOp maps the scalar float-to-integer conversions to their mandatory
|
||
// prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating
|
||
// CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width.
|
||
var cvtIntOp = map[string]struct {
|
||
prefix byte
|
||
op byte
|
||
}{
|
||
"CVTSD2S": {0xF2, 0x2D},
|
||
"CVTTSD2S": {0xF2, 0x2C},
|
||
"CVTSS2S": {0xF3, 0x2D},
|
||
"CVTTSS2S": {0xF3, 0x2C},
|
||
}
|
||
|
||
// encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C
|
||
// with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the
|
||
// quad spellings.
|
||
func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||
}
|
||
spec := cvtIntOp[base]
|
||
src, dst := ops[0], ops[1]
|
||
dstReg, ok := dst.(Reg)
|
||
if !ok || dstReg.isVec() {
|
||
return fmt.Errorf("%s destination must be a general register", base)
|
||
}
|
||
i := newInstr(size, []byte{0x0F, spec.op})
|
||
i.prefix = spec.prefix
|
||
if err := setRM(i, dstReg, src, size); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeFmov encodes the x87 double move. The memory forms are DD /0
|
||
// (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register
|
||
// move is DD C0+dst (FLD st(dst)), the form the Go assembler emits.
|
||
func (e *enc) encodeFmov(ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops))
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
srcReg, srcIsF := src.(Reg)
|
||
dstReg, dstIsF := dst.(Reg)
|
||
srcF := srcIsF && srcReg.fp
|
||
dstF := dstIsF && dstReg.fp
|
||
switch {
|
||
case srcF && dstF:
|
||
// The register form is DD /2 with rm = the destination (FST st(dst)).
|
||
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||
if err := setRMDigit(i, 2, dstReg, 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
case dstF:
|
||
m, ok := src.(Mem)
|
||
if !ok {
|
||
return fmt.Errorf("FMOVD: invalid source operand")
|
||
}
|
||
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||
if err := setMem(i, 0, m); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
case srcF:
|
||
m, ok := dst.(Mem)
|
||
if !ok {
|
||
return fmt.Errorf("FMOVD: invalid destination operand")
|
||
}
|
||
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||
if err := setMem(i, 2, m); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
return fmt.Errorf("FMOVD needs an x87 register operand")
|
||
}
|
||
|
||
// --- legacy SSE imm8, extract, insert and packed shift families --------------
|
||
|
||
// encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src,
|
||
// dst with reg = dst, rm = src and the immediate appended last (PALIGNR,
|
||
// PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4).
|
||
func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||
}
|
||
imm, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("SSE imm8 instruction needs an immediate first operand")
|
||
}
|
||
immByte, err := imm8(int64(imm))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
src, dst := ops[1], ops[2]
|
||
dstReg, ok2 := dst.(Reg)
|
||
if !ok2 || !dstReg.isVec() {
|
||
return fmt.Errorf("SSE imm8 instruction destination must be a vector register")
|
||
}
|
||
opcode := []byte{0x0F, 0x38, m.op}
|
||
if m.map3A {
|
||
opcode = []byte{0x0F, 0x3A, m.op}
|
||
}
|
||
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
|
||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{immByte}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the
|
||
// XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and
|
||
// PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the
|
||
// SSE4.1 0F3A 15 one).
|
||
func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||
}
|
||
imm, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("extract needs an immediate first operand")
|
||
}
|
||
immByte, err := imm8(int64(imm))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
srcReg, srcVec := vecReg(ops[1])
|
||
if !srcVec {
|
||
return fmt.Errorf("extract source must be an XMM register")
|
||
}
|
||
opcode := m.op
|
||
if m.opMem != nil && memOperand(ops[2]) {
|
||
opcode = m.opMem
|
||
}
|
||
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
|
||
if err := setRM(i, srcReg, ops[2], 8); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{immByte}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the
|
||
// XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and
|
||
// PINSRW).
|
||
func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||
}
|
||
imm, ok := ops[0].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("insert needs an immediate first operand")
|
||
}
|
||
immByte, err := imm8(int64(imm))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
dstReg, dstVec := vecReg(ops[2])
|
||
if !dstVec {
|
||
return fmt.Errorf("insert destination must be an XMM register")
|
||
}
|
||
i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW}
|
||
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{immByte}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeSSEShift encodes the legacy packed integer shifts. The immediate
|
||
// form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form
|
||
// OP count, dst carries the count in an XMM register (or memory) on the
|
||
// 66 0F D1-F3 opcodes. The destination is always the register written.
|
||
func (e *enc) encodeSSEShift(name string, ops []Operand) error {
|
||
if len(ops) != 2 {
|
||
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||
}
|
||
dstReg, ok := ops[1].(Reg)
|
||
if !ok || !dstReg.isVec() {
|
||
return fmt.Errorf("%s destination must be the second, vector operand", name)
|
||
}
|
||
if imm, isImm := ops[0].(Imm); isImm {
|
||
spec := sseShiftImm[name]
|
||
immByte, err := imm8(int64(imm))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{immByte}
|
||
return e.emit(i)
|
||
}
|
||
if !vecOrMem(ops[0]) {
|
||
return fmt.Errorf("%s count must be an immediate, a vector register or memory", name)
|
||
}
|
||
op, ok := sseShiftVar[name]
|
||
if !ok {
|
||
return fmt.Errorf("%s has no variable-count form", name)
|
||
}
|
||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeCmpsd encodes CMPSD, the scalar double compare with its predicate
|
||
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
|
||
// F2 0F C2 with reg = dst, rm = src.
|
||
func (e *enc) encodeCmpsd(ops []Operand) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
|
||
}
|
||
imm, ok := ops[2].(Imm)
|
||
if !ok {
|
||
return fmt.Errorf("CMPSD predicate must be an immediate")
|
||
}
|
||
immByte, err := imm8(int64(imm))
|
||
if err != nil {
|
||
return err
|
||
}
|
||
dstReg, ok2 := ops[1].(Reg)
|
||
if !ok2 || !dstReg.isVec() {
|
||
return fmt.Errorf("CMPSD destination must be a vector register")
|
||
}
|
||
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
|
||
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||
return err
|
||
}
|
||
i.imm = []byte{immByte}
|
||
return e.emit(i)
|
||
}
|
||
|
||
// encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the
|
||
// literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no
|
||
// prefix, reg = dst, rm = src; X0 is implicit on the wire).
|
||
func (e *enc) encodeSha256rnds2(ops []Operand) error {
|
||
if len(ops) != 3 {
|
||
return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops))
|
||
}
|
||
x0, ok := ops[0].(Reg)
|
||
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
|
||
return fmt.Errorf("SHA256RNDS2 first operand must be X0")
|
||
}
|
||
dstReg, ok2 := ops[2].(Reg)
|
||
if !ok2 || !dstReg.isVec() {
|
||
return fmt.Errorf("SHA256RNDS2 destination must be a vector register")
|
||
}
|
||
i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1}
|
||
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||
return err
|
||
}
|
||
return e.emit(i)
|
||
}
|