// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import "fmt" // aluOp maps an arithmetic/logic mnemonic to its base "r/m, r" opcode (for // 16/32/64-bit; the 8-bit form is one less) and its /digit for the immediate // forms (0x80/0x81/0x83). var aluOp = map[string]struct { rr byte digit int }{ "ADD": {0x01, 0}, "OR": {0x09, 1}, "ADC": {0x11, 2}, "SBB": {0x19, 3}, "AND": {0x21, 4}, "SUB": {0x29, 5}, "XOR": {0x31, 6}, "CMP": {0x39, 7}, } // unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode. // INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes // in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4, // DIV /6, IDIV /7; the accumulator is the implicit other operand). var unaryOp = map[string]struct { digit int op byte }{ "INC": {0, 0xFF}, "DEC": {1, 0xFF}, "NOT": {2, 0xF7}, "NEG": {3, 0xF7}, "MUL": {4, 0xF7}, "DIV": {6, 0xF7}, "IDIV": {7, 0xF7}, } // shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the // 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4). var shiftOp = map[string]int{ "SHL": 4, "SAL": 4, "SHR": 5, "SAR": 7, "ROL": 0, "ROR": 1, "RCL": 2, "RCR": 3, } // bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form; // the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's // opcode row. var bitTestOp = map[string]int{ "BT": 4, "BTS": 5, "BTR": 6, "BTC": 7, } // noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The // fence names carry their opcode inside the 0F AE /digit group spelled out in // full (E8/F0/F8), and PAUSE is F3 90. // // LOCK, REP and REPN are the prefix statements. go tool asm encodes each as // a standalone one-byte instruction with a PC of its own (F0, F3 and F2 // respectively), not as a prefix field merged into the next instruction: the // statement that follows is encoded unaware of it, and nothing validates // that the pairing is a legal one (LOCK before NOP assembles without // complaint, each byte pinned against the toolchain). Because the bytes // land in the stream before the following statement anyway, a LOCKed // CMPXCHGQ encodes identically to a prefixed form. var noOperandTable = map[string][]byte{ "CPUID": {0x0F, 0xA2}, "RDTSC": {0x0F, 0x31}, "RDTSCP": {0x0F, 0x01, 0xF9}, "SYSCALL": {0x0F, 0x05}, "XGETBV": {0x0F, 0x01, 0xD0}, "CLD": {0xFC}, "STD": {0xFD}, "PAUSE": {0xF3, 0x90}, "LFENCE": {0x0F, 0xAE, 0xE8}, "MFENCE": {0x0F, 0xAE, 0xF0}, "SFENCE": {0x0F, 0xAE, 0xF8}, "UNDEF": {0x0F, 0x0B}, "LOCK": {0xF0}, "REP": {0xF3}, "REPN": {0xF2}, } // --- MOV -------------------------------------------------------------------- func (e *enc) encodeMov(ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("MOV expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] // Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2 // packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E // (reg = dst, no REX.W, the Go assembler's form), xmm→mem as // 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD // opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and // 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m // would be undefined forms. MOVL is the packed-dword move: // 66 0F 6E load, 66 0F 7E store, no REX.W. A GPR-move fallback would // silently emit REX.W 8B with the wrong operand meaning. _, srcVec := vecReg(src) dstReg, dstVec := vecReg(dst) if srcVec || dstVec { if dstVec { if g, ok := src.(Reg); ok && !g.isVec() { i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: size == 8} if err := setRM(i, dstReg, src, 8); err != nil { return err } return e.emit(i) } i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1} if size == 4 { i.prefix = 0x66 i.opcode = []byte{0x0F, 0x6E} } if err := setRM(i, dstReg, src, 8); err != nil { return err } return e.emit(i) } srcXMM, srcIsXMM := src.(Reg) if !srcIsXMM || !srcXMM.isVec() { return fmt.Errorf("MOV: store needs an XMM source") } if g, ok := dst.(Reg); ok && !g.isVec() { i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: size == 8} if err := setRM(i, srcXMM, dst, 8); err != nil { return err } return e.emit(i) } i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1} if size == 4 { i.opcode = []byte{0x0F, 0x7E} } if err := setRM(i, srcXMM, dst, 8); err != nil { return err } return e.emit(i) } dstReg, dstIsReg := dst.(Reg) switch src := src.(type) { case Reg: if dstIsReg { // MOV r/m, r: 0x88/0x89, reg=src, rm=dst, the form the Go // assembler emits for register-to-register moves. i := newInstr(size, []byte{movRM(size)}) if err := setRM(i, src, dst, size); err != nil { return err } return e.emit(i) } // MOV r/m, r: 0x88/0x89, reg=src, rm=dst(mem). i := newInstr(size, []byte{movRM(size)}) if err := setRM(i, src, dst, size); err != nil { return err } return e.emit(i) case Mem: if !dstIsReg { return fmt.Errorf("MOV: two memory operands") } // MOV r, r/m: reg=dst, rm=src(mem). i := newInstr(size, []byte{movRR(size)}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) case sbMem: if !dstIsReg { return fmt.Errorf("MOV: two memory operands") } // MOV r, r/m: reg=dst, rm=src(static symbol). i := newInstr(size, []byte{movRR(size)}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) case Imm: if dstIsReg { v := int64(src) // The Go assembler compresses 64-bit moves whose immediate fits // a signed int32, choosing per sign: // v >= 0: B8+rd imm32 without REX.W (zero-extended by the // hardware, REX.B still emitted for R8-R15); // v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd // form would zero-extend and corrupt the value). // Out-of-range immediates keep the B8+rd imm64 form. if size == 8 && v >= 0 && v <= (1<<31)-1 { i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)}) i.rexB = dstReg.idx >= 8 i.imm = le32(v) return e.emit(i) } if size == 8 && v < 0 && v >= -(1<<31) { i := newInstr(8, []byte{0xC7}) if err := setRMDigit(i, 0, dstReg, 8); err != nil { return err } i.imm = le32(v) return e.emit(i) } opBase := byte(0xB8) if size == 1 { opBase = 0xB0 } i := newInstr(size, []byte{opBase + byte(dstReg.idx&7)}) i.rexB = dstReg.idx >= 8 if dstReg.needsREX(size) { i.rexForced = true } imm, err := immediate(v, size, true) if err != nil { return err } i.imm = imm return e.emit(i) } // MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. op := byte(0xC7) if size == 1 { op = 0xC6 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, 0, dst, size); err != nil { return err } imm, err := immediate(int64(src), size, false) if err != nil { return err } i.imm = imm return e.emit(i) } return fmt.Errorf("MOV: invalid operands") } func movRR(size int) byte { // MOV r, r/m if size == 1 { return 0x8A } return 0x8B } func movRM(size int) byte { // MOV r/m, r if size == 1 { return 0x88 } return 0x89 } // --- ALU (ADD/OR/AND/SUB/XOR/CMP) ------------------------------------------- func (e *enc) encodeALU(op struct { rr byte digit int }, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("ALU instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] // CMP never takes its immediate first: the Go assembler rejects // CMPL $0, AX outright (only CMPL AX, $0 is legal, unlike TEST and the // writing ALU ops whose immediate is naturally the source). if imm, ok := src.(Imm); ok { if op.digit == 7 { return fmt.Errorf("CMP immediate must be the second operand (reg, $imm)") } return e.encodeALUImm(op.digit, dst, int64(imm), size) } // CMP accepts the immediate in the second position too, CMPL CX, $31 is // the form the Go assembler itself accepts, and encodes it identically // (CMP r/m, imm sets the flags as first − second). No other ALU op takes // an immediate destination. if imm, ok := dst.(Imm); ok { if op.digit != 7 { return fmt.Errorf("immediate must be the source operand") } return e.encodeALUImm(op.digit, src, int64(imm), size) } // CMP records first − second without writing anywhere, so the first // operand must land as the minuend; every other ALU op writes its second // operand and follows the forms below. cmp := op.rr == 0x39 dstReg, dstIsReg := dst.(Reg) srcReg, srcIsReg := src.(Reg) switch { case cmp && dstIsReg: // CMP x, reg: OP r/m, r (0x38/0x39) with rm = first operand, reg = // second, matching the Go assembler. opc := op.rr if size == 1 { opc = op.rr - 1 } i := newInstr(size, []byte{opc}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) case cmp && srcIsReg: // CMP reg, mem: OP r, r/m (0x3A/0x3B) with reg = first operand, rm = // second. opc := op.rr + 2 if size == 1 { opc = op.rr + 1 } i := newInstr(size, []byte{opc}) if err := setRM(i, srcReg, dst, size); err != nil { return err } return e.emit(i) case srcIsReg: // OP r/m, r: reg=src, rm=dst (dst is a register or memory). This is the // form the Go assembler prefers when the source is a register. opc := op.rr if size == 1 { opc = op.rr - 1 } i := newInstr(size, []byte{opc}) if err := setRM(i, srcReg, dst, size); err != nil { return err } return e.emit(i) case dstIsReg: // OP r, r/m: reg=dst, rm=src(memory). opc := op.rr + 2 if size == 1 { opc = op.rr + 1 } i := newInstr(size, []byte{opc}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } return fmt.Errorf("two memory operands") } func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error { if size == 1 { immBytes, err := immediate(imm, 1, false) if err != nil { return err } // The byte accumulator short form (0x04+digit*8, no ModR/M) when // the destination is AL, the form the Go assembler prefers here. if r, ok := dst.(Reg); ok && r.idx == 0 { i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1} i.imm = immBytes return e.emit(i) } i := newInstr(1, []byte{0x80}) if err := setRMDigit(i, digit, dst, 1); err != nil { return err } i.imm = immBytes return e.emit(i) } if fits8(imm) { // 0x83 /digit, sign-extended imm8. i := newInstr(size, []byte{0x83}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } i.imm = []byte{byte(int8(imm))} return e.emit(i) } // 0x81 /digit, imm16/imm32, or the Go assembler's accumulator short // form (opcode+5, no ModR/M) when the destination is AX/AL, which it // prefers over the generic form exactly here. if r, ok := dst.(Reg); ok && r.idx == 0 { accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit] i := newInstr(size, []byte{accOp}) immBytes, err := immediate(imm, size, false) if err != nil { return err } i.imm = immBytes return e.emit(i) } // 0x81 /digit, imm16/imm32. i := newInstr(size, []byte{0x81}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } immBytes, err := immediate(imm, size, false) if err != nil { return err } i.imm = immBytes return e.emit(i) } // --- TEST ------------------------------------------------------------------- func (e *enc) encodeTest(ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("TEST expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] if imm, ok := src.(Imm); ok { // TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0, but the Go assembler // always uses the accumulator forms (A8/A9, no ModR/M) when the // register operand is AL/AX, whatever the immediate's width. if r, ok := dst.(Reg); ok && r.idx == 0 { op := byte(0xA9) if size == 1 { op = 0xA8 } i := newInstr(size, []byte{op}) immBytes, err := immediate(int64(imm), size, false) if err != nil { return err } i.imm = immBytes return e.emit(i) } op := byte(0xF7) if size == 1 { op = 0xF6 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, 0, dst, size); err != nil { return err } immBytes, err := immediate(int64(imm), size, false) if err != nil { return err } i.imm = immBytes return e.emit(i) } srcReg, ok := src.(Reg) if !ok { return fmt.Errorf("TEST: source must be a register or immediate") } // TEST r/m, r: 0x84 (8-bit) / 0x85. op := byte(0x85) if size == 1 { op = 0x84 } i := newInstr(size, []byte{op}) if err := setRM(i, srcReg, dst, size); err != nil { return err } return e.emit(i) } // --- LEA -------------------------------------------------------------------- func (e *enc) encodeLea(ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("LEA expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] // LEAQ addr, reg dstReg, ok := dst.(Reg) if !ok { return fmt.Errorf("LEA: destination must be a register") } switch src.(type) { case Mem, sbMem: default: return fmt.Errorf("LEA: source must be a memory operand") } i := newInstr(size, []byte{0x8D}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } // --- INC/DEC/NEG/NOT -------------------------------------------------------- func (e *enc) encodeUnary(op struct { digit int op byte }, ops []Operand, size int) error { if len(ops) != 1 { return fmt.Errorf("unary instruction expects 1 operand, got %d", len(ops)) } base := op.op if size == 1 { base-- // 0xFF→0xFE, 0xF7→0xF6 } i := newInstr(size, []byte{base}) if err := setRMDigit(i, op.digit, ops[0], size); err != nil { return err } return e.emit(i) } // --- SHL/SHR/SAR ------------------------------------------------------------ // doubleShiftOp maps the two mnemonics whose three-operand form go tool asm // accepts to the SHLD/SHRD opcode pair (imm8 form, CL form). SAR, SAL and // the rotates have no such form: the oracle rejects SARQ/ROLQ with three // operands, and so do we. var doubleShiftOp = map[string][2]byte{ "SHL": {0xA4, 0xA5}, // SHLD "SHR": {0xAC, 0xAD}, // SHRD } // isShiftCountCL reports whether a count operand is the CL register or its // CX spelling: go tool asm accepts both (CX names the same low byte) and // rejects ECX/RCX. func isShiftCountCL(o Operand) bool { reg, ok := o.(Reg) return ok && reg.idx == 1 && (reg.size == 1 || reg.size == 2) } func (e *enc) encodeShift(base string, ops []Operand, size int) error { digit := shiftOp[base] if len(ops) == 3 { return e.encodeDoubleShift(base, ops, size) } if len(ops) != 2 { return fmt.Errorf("shift expects 2 operands, got %d", len(ops)) } count, dst := ops[0], ops[1] // Count is $1, CL (or its CX spelling), or an imm8. if isShiftCountCL(count) { // CL: 0xD2 (8-bit) / 0xD3. op := byte(0xD3) if size == 1 { op = 0xD2 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } return e.emit(i) } imm, ok := count.(Imm) if !ok { return fmt.Errorf("shift count must be $1, CL or an immediate") } if imm == 1 { // 0xD0 (8-bit) / 0xD1. op := byte(0xD1) if size == 1 { op = 0xD0 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } return e.emit(i) } // 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm // rejects negative and ≥256 counts, and the hardware masks the count, so // a silent truncation ($300 encoding 44) would shift by a different // amount than the source states. if imm < 0 || imm > 255 { return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm)) } op := byte(0xC1) if size == 1 { op = 0xC0 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } i.imm = []byte{byte(imm)} return e.emit(i) } // encodeDoubleShift emits the three-operand SHL/SHR form, which the Go // assembler spells as a shift but encodes as SHLD/SHRD (0F A4/A5, 0F AC/AD): // the first operand is the count ($imm or CL), the second feeds the vacated // bits (the reg field) and the third is the shifted value (the r/m field), // matching go tool asm byte for byte. The W/L/Q widths exist; the oracle // rejects the three-operand B form and every SAR/rotate one. func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error { opc, ok := doubleShiftOp[base] if !ok || size == 1 { return fmt.Errorf("%s: shift expects 2 operands, got %d", base, len(ops)) } count, src, dst := ops[0], ops[1], ops[2] srcReg, ok := src.(Reg) if !ok { return fmt.Errorf("%s: middle operand must be a register, like go tool asm", base) } i := newInstr(size, []byte{0x0F, opc[0]}) if isShiftCountCL(count) { // CL (or CX) form: 0F A5/AD. i.opcode[1] = opc[1] } else { imm, ok := count.(Imm) if !ok { return fmt.Errorf("shift count must be $1, CL or an immediate") } // The count is an unsigned imm8: the same range convention as the // two-operand shift above. if imm < 0 || imm > 255 { return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm)) } i.imm = []byte{byte(imm)} } if err := setRMReg(i, srcReg.idx, srcReg.idx >= 8, false, dst, size); err != nil { return err } return e.emit(i) } // --- IMUL ------------------------------------------------------------------- func (e *enc) encodeImul(ops []Operand, size int) error { switch len(ops) { case 2: // IMUL r, r/m: 0x0F 0xAF. dstReg, ok := ops[1].(Reg) if !ok { return fmt.Errorf("IMUL: destination must be a register") } i := newInstr(size, []byte{0x0F, 0xAF}) if err := setRM(i, dstReg, ops[0], size); err != nil { return err } return e.emit(i) case 3: // IMUL r, r/m, imm: 0x6B (imm8) / 0x69 (imm16/32). dstReg, ok := ops[2].(Reg) if !ok { return fmt.Errorf("IMUL: destination must be a register") } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("IMUL: immediate operand expected first") } // Plan 9 order: IMUL $imm, src, dst. if fits8(int64(imm)) { i := newInstr(size, []byte{0x6B}) if err := setRM(i, dstReg, ops[1], size); err != nil { return err } i.imm = []byte{byte(int8(imm))} return e.emit(i) } i := newInstr(size, []byte{0x69}) if err := setRM(i, dstReg, ops[1], size); err != nil { return err } immBytes, err := immediate(int64(imm), size, false) if err != nil { return err } i.imm = immBytes return e.emit(i) } return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops)) } // --- PUSH / POP ------------------------------------------------------------- func (e *enc) encodePushPop(ops []Operand, size int, push bool) error { if len(ops) != 1 { return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops)) } // In 64-bit mode go tool asm knows the 64-bit push (the default, with or // without the Q suffix) and the 16-bit W form with its 0x66 operand-size // prefix, and rejects the B and L spellings outright ("illegal in 64-bit // mode"); silently widening those would push a different width than the // source states. switch size { case 0, 8, 2: default: return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode") } w16 := size == 2 switch op := ops[0].(type) { case Reg: base := byte(0x50) // PUSH r; POP is 0x58 if !push { base = 0x58 } // PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed. i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1} i.rexB = op.idx >= 8 return e.emit(i) case Mem: opc := byte(0xFF) // PUSH r/m: /6 digit := 6 if !push { opc = 0x8F // POP r/m: /0 digit = 0 } i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1} if err := setRMDigit(i, digit, ops[0], 8); err != nil { return err } return e.emit(i) case Imm: if !push { return fmt.Errorf("POP does not take an immediate") } if fits8(int64(op)) { i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}} return e.emit(i) } // PUSH imm32, sign-extended to 64 bits; go tool asm bounds the // immediate by the same signed/unsigned 32-bit span as every other // scalar immediate. immBytes, err := immediate(int64(op), 8, false) if err != nil { return err } i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes} return e.emit(i) } return fmt.Errorf("PUSH/POP: invalid operand") } // --- RET / JMP / CALL / Jcc ------------------------------------------------- func (e *enc) encodeRet() error { return e.emit(&instr{opcode: []byte{0xC3}, modrm: -1, sib: -1}) } // encodeJmpRel encodes JMP/CALL with a relative displacement (the operand is an // Imm holding the already-computed rel32 offset). func (e *enc) encodeJmpRel(ops []Operand, opcode []byte) error { if len(ops) != 1 { return fmt.Errorf("JMP/CALL expects 1 operand, got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("JMP/CALL: relative offset must be an immediate (labels are resolved by the assembler)") } return e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, imm: le32(int64(imm))}) } // encodeIndirectBranch encodes JMP/CALL through a register or memory operand: // FF /4 for JMP, FF /2 for CALL. The operand size is fixed at 64 bits in // 64-bit mode, so no REX.W is emitted; a REX appears only for R8-R15 bases. func (e *enc) encodeIndirectBranch(mnem string, ops []Operand) error { if len(ops) != 1 { return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops)) } digit := 4 // JMP r/m64 if mnem == "CALL" { digit = 2 // CALL r/m64 } i := &instr{opcode: []byte{0xFF}, modrm: -1, sib: -1} if err := setRMDigit(i, digit, ops[0], 8); err != nil { return err } return e.emit(i) } // condCode maps a Plan 9 conditional-jump mnemonic to its x86 condition code. func condCode(upper string) (int, bool) { if len(upper) < 2 || upper[0] != 'J' || upper == "JMP" { return 0, false } cc, ok := jccMap[upper[1:]] return cc, ok } var jccMap = map[string]int{ "O": 0x0, "NO": 0x1, "OS": 0x0, "OC": 0x1, "B": 0x2, "C": 0x2, "NAE": 0x2, "CS": 0x2, "NB": 0x3, "NC": 0x3, "AE": 0x3, "CC": 0x3, "E": 0x4, "Z": 0x4, "EQ": 0x4, "NE": 0x5, "NZ": 0x5, "BE": 0x6, "NA": 0x6, "LS": 0x6, "NBE": 0x7, "A": 0x7, "HI": 0x7, "S": 0x8, "MI": 0x8, "NS": 0x9, "PL": 0x9, "P": 0xA, "PE": 0xA, "PS": 0xA, "NP": 0xB, "PO": 0xB, "PC": 0xB, "L": 0xC, "NGE": 0xC, "LT": 0xC, "NL": 0xD, "GE": 0xD, "LE": 0xE, "NG": 0xE, "NLE": 0xF, "G": 0xF, "GT": 0xF, } func (e *enc) encodeJcc(cc int, ops []Operand) error { if len(ops) != 1 { return fmt.Errorf("conditional jump expects 1 operand, got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("conditional jump: relative offset must be an immediate") } if fits8(int64(imm)) { // Short form: 0x70+cc, rel8. return e.emit(&instr{opcode: []byte{0x70 + byte(cc)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}}) } // Near form: 0x0F 0x80+cc, rel32. return e.emit(&instr{opcode: []byte{0x0F, 0x80 + byte(cc)}, modrm: -1, sib: -1, imm: le32(int64(imm))}) } // immediate encodes an immediate of the given operand size. full64 selects the // 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit // sign-extended immediate is used for 64-bit operands. // // The span mirrors go tool asm: every scalar immediate must fit a signed or // unsigned 32-bit word, and the narrower fields then take the low bits // silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the // imm64 form may exceed the span; anything wider elsewhere is an error rather // than a truncation the source never asked for. func immediate(v int64, size int, full64 bool) ([]byte, error) { if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) { return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v) } switch size { case 1: return []byte{byte(int8(v))}, nil case 2: return le16(v), nil case 4: return le32(v), nil default: // 8 if full64 { return le64(v), nil } return le32(v), nil // sign-extended imm32 } } // --- CMOVcc / SETcc --------------------------------------------------------- // encodeCmov encodes a conditional move: CMOV + size (W/L/Q) + condition // (CMOVLGT, CMOVQEQ, …). The condition reads exactly like the Jcc spellings; // the instruction is 0F 40+cc with reg = dst, rm = src. func (e *enc) encodeCmov(upper string, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("CMOVcc expects 2 operands, got %d", len(ops)) } rest := upper[len("CMOV"):] if len(rest) < 2 { return fmt.Errorf("unsupported instruction %q", upper) } var size int switch rest[0] { case 'W': size = 2 case 'L': size = 4 case 'Q': size = 8 default: return fmt.Errorf("unsupported instruction %q", upper) } cc, ok := jccMap[rest[1:]] if !ok { return fmt.Errorf("unsupported instruction %q", upper) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok { return fmt.Errorf("CMOVcc destination must be a register") } i := newInstr(size, []byte{0x0F, byte(0x40 + cc)}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } // encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …), // always a byte write, 0F 90+cc /0 into a register or memory operand. func (e *enc) encodeSet(upper string, ops []Operand) error { if len(ops) != 1 { return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops)) } cond := upper[len("SET"):] cc, ok := jccMap[cond] if !ok || cond == "" { return fmt.Errorf("unsupported instruction %q", upper) } i := &instr{opcode: []byte{0x0F, byte(0x90 + cc)}, modrm: -1, sib: -1} if err := setRMDigit(i, 0, ops[0], 1); err != nil { return err } return e.emit(i) } // --- bit scan / bit count ---------------------------------------------------- // countOp maps the bit-scan and bit-count mnemonics to their opcode byte and // mandatory prefix. TZCNT/LZCNT/POPCNT are the F3-prefixed forms of the // same map as BSF/BSR's 0F BC/BD; POPCNT is F3 0F B8. var countOp = map[string]struct { op byte prefix byte }{ "BSF": {0xBC, 0}, "BSR": {0xBD, 0}, "TZCNT": {0xBC, 0xF3}, "LZCNT": {0xBD, 0xF3}, "POPCNT": {0xB8, 0xF3}, } // encodeCount encodes the bit-scan and bit-count family, BSF (0F BC), // BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8) // with reg = dst and rm = src. The size suffix selects the operand width // (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the // source is zero (unlike their F3-prefixed counterparts); callers must // guard non-zero inputs themselves. func (e *enc) encodeCount(base string, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops)) } spec := countOp[base] dstReg, ok := ops[1].(Reg) if !ok { return fmt.Errorf("%s destination must be a register", base) } i := newInstr(size, []byte{0x0F, spec.op}) i.prefix = spec.prefix if err := setRM(i, dstReg, ops[0], size); err != nil { return err } return e.emit(i) } // encodeBswap encodes BSWAP: the single register operand is encoded in the // opcode byte (0F C8+r), with REX.B for R8-R15 and REX.W for the quad form. func (e *enc) encodeBswap(ops []Operand, size int) error { if len(ops) != 1 { return fmt.Errorf("BSWAP expects 1 operand, got %d", len(ops)) } reg, ok := ops[0].(Reg) if !ok { return fmt.Errorf("BSWAP operand must be a register") } i := newInstr(size, []byte{0x0F, 0xC8 + byte(reg.idx&7)}) i.rexB = reg.idx >= 8 return e.emit(i) } // --- mixed-width sign/zero-extending moves ----------------------------------- // movExtendOp maps Go's mixed-width move names to their opcode and destination // width. The source is narrower than the destination, so the plain size-suffix // convention does not apply to these names. var movExtendOp = map[string]struct { op []byte dstSize int }{ "MOVBLZX": {[]byte{0x0F, 0xB6}, 4}, // byte → long, zero-extend "MOVBQZX": {[]byte{0x0F, 0xB6}, 8}, // byte → quad, zero-extend "MOVWLZX": {[]byte{0x0F, 0xB7}, 4}, // word → long, zero-extend "MOVWQZX": {[]byte{0x0F, 0xB7}, 8}, // word → quad, zero-extend "MOVWLSX": {[]byte{0x0F, 0xBF}, 4}, // word → long, sign-extend "MOVLQSX": {[]byte{0x63}, 8}, // long → quad, sign-extend (MOVSXD) "MOVBWZX": {[]byte{0x0F, 0xB6}, 2}, // byte → word, zero-extend "MOVBWSX": {[]byte{0x0F, 0xBE}, 2}, // byte → word, sign-extend "MOVBLSX": {[]byte{0x0F, 0xBE}, 4}, // byte → long, sign-extend "MOVBQSX": {[]byte{0x0F, 0xBE}, 8}, // byte → quad, sign-extend "MOVWQSX": {[]byte{0x0F, 0xBF}, 8}, // word → quad, sign-extend // A long → quad zero-extend is a plain 32-bit move: every 32-bit // operation zero-extends its result into the full register, so the // toolchain lowers MOVLQZX to the plain MOVL encoding. "MOVLQZX": {[]byte{0x8B}, 4}, } // encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider // operand), rm = src. func (e *enc) encodeMovExtend(base string, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops)) } spec := movExtendOp[base] dstReg, ok := ops[1].(Reg) if !ok { return fmt.Errorf("%s destination must be a register", base) } i := newInstr(spec.dstSize, spec.op) if err := setRM(i, dstReg, ops[0], spec.dstSize); err != nil { return err } return e.emit(i) } // encodePmovmskb encodes PMOVMSKB, the legacy SSE2 byte mask extract: the // XMM source's sign bytes pack into a GP destination, 66 0F D7 /r. func (e *enc) encodePmovmskb(base string, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops)) } srcReg, srcVec := vecReg(ops[0]) if !srcVec { return fmt.Errorf("%s source must be an XMM register", base) } dstReg, ok := ops[1].(Reg) if !ok { return fmt.Errorf("%s destination must be a register", base) } i := newInstr(4, []byte{0x0F, 0xD7}) i.prefix = 0x66 if err := setRM(i, dstReg, srcReg, 4); err != nil { return err } return e.emit(i) } // --- legacy SSE moves -------------------------------------------------------- // sseMove describes a legacy (non-VEX) SSE move: a mandatory prefix plus a // load opcode (reg = destination, rm = source) and a store opcode (the // reverse). The Plan 9 names MOVOU/MOVO are the integer unaligned/aligned // octa moves (MOVDQU/MOVDQA), not the packed-single ones. type sseMove struct { prefix byte // 0, 0x66, 0xF2 or 0xF3 load byte store byte } var sseMoveTable = map[string]sseMove{ "MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa "MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa "MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias "MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single "MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single "MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double "MOVAPD": {0x66, 0x28, 0x29}, // aligned packed double "MOVSD": {0xF2, 0x10, 0x11}, // scalar double "MOVSS": {0xF3, 0x10, 0x11}, // scalar single } // encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the // load form (reg = destination), matching the Go assembler. func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("SSE move expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcVec := vecReg(src) dstReg, dstVec := vecReg(dst) op := m.store var reg Reg var rm Operand switch { case srcVec && dstVec: op = m.load reg, rm = dstReg, src case srcVec: if !isX86Mem(dst) { return fmt.Errorf("SSE move: invalid destination operand") } reg, rm = srcReg, dst case dstVec: if !isX86Mem(src) { return fmt.Errorf("SSE move: invalid source operand") } op = m.load reg, rm = dstReg, src default: return fmt.Errorf("SSE move needs a vector register operand") } i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1} if err := setRM(i, reg, rm, 8); err != nil { return err } return e.emit(i) } // --- legacy SSE packed binary and shuffles ----------------------------------- // sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an // optional mandatory prefix plus the 0F-prefixed opcode (0F38 for the // SSSE3 integer shuffles). Plan 9 asm lists the source operand first, so // MULPS X0, X1 computes X1 = X1 * X0. type sseBin struct { prefix byte // 0, 0x66, 0xF2 or 0xF3 op byte map38 bool // opcode lives under 0F38 instead of 0F } var sseBinTable = map[string]sseBin{ "ADDPS": {0, 0x58, false}, "ADDPD": {0x66, 0x58, false}, "MULPS": {0, 0x59, false}, "MULPD": {0x66, 0x59, false}, "SUBPS": {0, 0x5C, false}, "SUBPD": {0x66, 0x5C, false}, "DIVPS": {0, 0x5E, false}, "DIVPD": {0x66, 0x5E, false}, "ANDPS": {0, 0x54, false}, "ANDPD": {0x66, 0x54, false}, "ORPS": {0, 0x56, false}, "ORPD": {0x66, 0x56, false}, "XORPS": {0, 0x57, false}, "XORPD": {0x66, 0x57, false}, "MINPS": {0, 0x5D, false}, "MINPD": {0x66, 0x5D, false}, "MAXPS": {0, 0x5F, false}, "MAXPD": {0x66, 0x5F, false}, "ADDSS": {0xF3, 0x58, false}, "ADDSD": {0xF2, 0x58, false}, "MULSS": {0xF3, 0x59, false}, "MULSD": {0xF2, 0x59, false}, "SUBSS": {0xF3, 0x5C, false}, "SUBSD": {0xF2, 0x5C, false}, "DIVSS": {0xF3, 0x5E, false}, "DIVSD": {0xF2, 0x5E, false}, "MINSS": {0xF3, 0x5D, false}, "MINSD": {0xF2, 0x5D, false}, "MAXSS": {0xF3, 0x5F, false}, "MAXSD": {0xF2, 0x5F, false}, "UNPCKLPS": {0, 0x14, false}, "UNPCKHPS": {0, 0x15, false}, "UNPCKLPD": {0x66, 0x14, false}, "UNPCKHPD": {0x66, 0x15, false}, "CVTSS2SD": {0xF3, 0x5A, false}, "CVTSD2SS": {0xF2, 0x5A, false}, "CVTPS2PD": {0, 0x5A, false}, "CVTPD2PS": {0x66, 0x5A, false}, // SSE2 packed integers (reg = reg op rm) and the SSSE3 byte shuffle. "PXOR": {0x66, 0xEF, false}, "POR": {0x66, 0xEB, false}, "PAND": {0x66, 0xDB, false}, "PANDN": {0x66, 0xDF, false}, "PADDB": {0x66, 0xFC, false}, "PADDW": {0x66, 0xFD, false}, "PADDD": {0x66, 0xFE, false}, "PADDQ": {0x66, 0xD4, false}, "PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false}, "PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false}, "PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false}, "PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false}, "PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false}, "PCMPGTD": {0x66, 0x66, false}, "PSHUFB": {0x66, 0x00, true}, // Scalar compares and square root, packed adds/subtracts and the byte // unpack, the spellings the Plan 9 table uses (COMISD orders the // operands like every other two-operand form). "ANDNPD": {0x66, 0x55, false}, "ANDNPS": {0x00, 0x55, false}, "COMISD": {0x66, 0x2F, false}, "SQRTSD": {0xF2, 0x51, false}, "PADDL": {0x66, 0xFE, false}, "PSUBL": {0x66, 0xFA, false}, "PUNPCKLBW": {0x66, 0x60, false}, // AES round functions (66 0F38) and the SHA message schedule helpers // (no prefix, 0F38). "AESENC": {0x66, 0xDC, true}, "AESENCLAST": {0x66, 0xDD, true}, "AESDEC": {0x66, 0xDE, true}, "AESDECLAST": {0x66, 0xDF, true}, "AESIMC": {0x66, 0xDB, true}, "SHA1MSG1": {0x00, 0xC9, true}, "SHA1MSG2": {0x00, 0xCA, true}, "SHA1NEXTE": {0x00, 0xC8, true}, "SHA256MSG1": {0x00, 0xCC, true}, "SHA256MSG2": {0x00, 0xCD, true}, } // sseImm3 describes a legacy SSE instruction taking a leading imm8 and two // further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and // map3A select the opcode map the same way as sseBin's. type sseImm3 struct { prefix byte op byte map3A bool // opcode lives under 0F3A instead of 0F38 } // sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3 // align/blend shuffles, the string compare, carry-less multiply and the AES // key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings. var sseImm3Table = map[string]sseImm3{ "PALIGNR": {0x66, 0x0F, true}, "PBLENDW": {0x66, 0x0E, true}, "PCMPESTRI": {0x66, 0x61, true}, "PCLMULQDQ": {0x66, 0x44, true}, "AESKEYGENASSIST": {0x66, 0xDF, true}, "SHA1RNDS4": {0x00, 0xCC, true}, } // sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM // source and rm = the destination (GPR or memory). PEXTRW's GPR destination // uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one, // so it carries both opcodes. type sseExtract struct { op []byte opMem []byte // used when the destination is memory; nil shares op rexW bool // PEXTRQ's REX.W } var sseExtractTable = map[string]sseExtract{ "PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false}, "PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false}, "PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true}, "PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false}, } // sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM // destination and rm = the source (GPR or memory). type sseInsert struct { op []byte rexW bool // PINSRQ's REX.W } var sseInsertTable = map[string]sseInsert{ "PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false}, "PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false}, "PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true}, "PINSRW": {[]byte{0x0F, 0xC4}, false}, } // sseShiftImm maps the legacy packed integer shifts' immediate form: // OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L // (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ. var sseShiftImm = map[string]sseShift{ "PSLLW": {0x71, 6}, "PSRLW": {0x71, 2}, "PSRAW": {0x71, 4}, "PSLLL": {0x72, 6}, "PSRLL": {0x72, 2}, "PSRAL": {0x72, 4}, "PSLLQ": {0x73, 6}, "PSRLQ": {0x73, 2}, "PSLLDQ": {0x73, 7}, "PSRLDQ": {0x73, 3}, } // sseShiftVar maps the variable-count forms (the count comes from an XMM // register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no // variable form. var sseShiftVar = map[string]byte{ "PSLLW": 0xF1, "PSRLW": 0xD1, "PSRAW": 0xE1, "PSLLL": 0xF2, "PSRLL": 0xD2, "PSRAL": 0xE2, "PSLLQ": 0xF3, "PSRLQ": 0xD3, } // sseShift is one /digit selector in the 0F 71/72/73 immediate group. type sseShift struct { op byte digit int } // sseShuf describes a legacy SSE shuffle taking a trailing imm8 // (PSHUFD/PSHUFHW/PSHUFLW also carry the packed-int 0x66/F3/F2 prefixes). type sseShuf struct { prefix byte op byte } var sseShufTable = map[string]sseShuf{ "SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6}, "PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70}, "PSHUFL": {0x66, 0x70}, } // encodeSSEBin encodes reg = reg op rm (memory allowed for rm). func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("SSE binary expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("SSE binary destination must be a vector register") } opcode := []byte{0x0F, m.op} if m.map38 { opcode = []byte{0x0F, 0x38, m.op} } i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1} if err := setRM(i, dstReg, src, 8); err != nil { return err } return e.emit(i) } // encodeSSEShuf encodes an imm8 shuffle: SHUFPS $imm, src, dst. func (e *enc) encodeSSEShuf(m sseShuf, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("SSE shuffle expects 3 operands, got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("SSE shuffle needs an imm8 first operand") } if imm < -128 || imm > 255 { return fmt.Errorf("SSE shuffle imm8 %d out of range", imm) } src, dst := ops[1], ops[2] dstReg, ok2 := dst.(Reg) if !ok2 || !dstReg.isVec() { return fmt.Errorf("SSE shuffle destination must be a vector register") } i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} if err := setRM(i, dstReg, src, 8); err != nil { return err } i.imm = []byte{byte(int8(imm))} return e.emit(i) } // --- CVTSL2SD / CVTSQ2SD ----------------------------------------------------- // encodeCvtsi2sd encodes a signed integer to scalar double conversion // (CVTSL2SD from a 32-bit, CVTSQ2SD from a 64-bit source): F2 0F 2A with // reg = XMM dst, rm = GPR/memory src. The Go assembler emits the legacy SSE // encoding here, not the VEX form, so we match it byte for byte. func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("CVTSx2SD expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("CVTSx2SD destination must be a vector register") } size := 4 if quad { size = 8 } i := newInstr(size, []byte{0x0F, 0x2A}) i.prefix = 0xF2 if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } // --- carry, bit test, exchange and accumulate ------------------------------- // encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9 // order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with // reg = index, rm = target); an immediate index uses 0F BA /digit with imm8. func (e *enc) encodeBitTest(name string, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops)) } digit := bitTestOp[name] index, target := ops[0], ops[1] if reg, ok := index.(Reg); ok { // Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC), // the /digit base plus eight per step. i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3}) if err := setRM(i, reg, target, size); err != nil { return err } return e.emit(i) } imm, ok := index.(Imm) if !ok { return fmt.Errorf("%s index must be a register or an immediate", name) } immByte, err := imm8(int64(imm)) if err != nil { return err } i := newInstr(size, []byte{0x0F, 0xBA}) if err := setRMDigit(i, digit, target, size); err != nil { return err } i.imm = []byte{immByte} return e.emit(i) } // encodeExchange encodes XCHG. A register-to-register exchange where either // operand is AX uses the 0x90+r accumulator form (with REX.W for the quad // form, as the Go assembler emits it); everything else uses 0x86/0x87 with // the register operand in ModRM.reg, the memory (or second register) in r/m. func (e *enc) encodeExchange(ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcIsReg := src.(Reg) dstReg, dstIsReg := dst.(Reg) if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) { // 0x90+r: r is the non-AX register, whichever side it sits on. r := dstReg if srcReg.idx == 0 { r = dstReg } else { r = srcReg } i := newInstr(size, []byte{0x90 + byte(r.idx&7)}) i.rexB = r.idx >= 8 return e.emit(i) } op := byte(0x87) if size == 1 { op = 0x86 } switch { case srcIsReg: i := newInstr(size, []byte{op}) if err := setRM(i, srcReg, dst, size); err != nil { return err } return e.emit(i) case dstIsReg: i := newInstr(size, []byte{op}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } return fmt.Errorf("XCHG: at least one operand must be a register") } // encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG // (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the // destination writable (register or memory). func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops)) } srcReg, ok := ops[0].(Reg) if !ok { return fmt.Errorf("%s source must be a register", name) } opc := op if size == 1 { opc = op8 } i := newInstr(size, []byte{0x0F, opc}) if err := setRM(i, srcReg, ops[1], size); err != nil { return err } return e.emit(i) } // encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for // the rest; the word form carries a 0x66 operand-size prefix (66 F2, the // prefix order the Go assembler emits) and the quad form REX.W. reg = GPR // accumulator, rm = the data source. func (e *enc) encodeCrc32(ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops)) } dstReg, ok := ops[1].(Reg) if !ok || dstReg.isVec() { return fmt.Errorf("CRC32 destination must be a general register") } i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1} if size > 1 { i.opcode[2] = 0xF1 } i.rexW = size == 8 if err := setRM(i, dstReg, ops[0], size); err != nil { return err } return e.emit(i) } // encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg = // destination, rm = source, the carry/overflow flag as the carry-in. func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops)) } dstReg, ok := ops[1].(Reg) if !ok || dstReg.isVec() { return fmt.Errorf("ADCX/ADOX destination must be a general register") } i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8} if err := setRM(i, dstReg, ops[0], size); err != nil { return err } return e.emit(i) } // --- string primitives, flags and INT ---------------------------------------- // encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and // STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or // REX.W prefix. func (e *enc) encodeStringOp(base string, ops []Operand, size int) error { if len(ops) != 0 { return fmt.Errorf("%s takes no operands, got %d", base, len(ops)) } var op byte switch base { case "MOVS": op = 0xA5 if size == 1 { op = 0xA4 } case "STOS": op = 0xAB if size == 1 { op = 0xAA } default: return fmt.Errorf("unsupported string instruction %q", base) } return e.emit(newInstr(size, []byte{op})) } // encodeInt encodes INT with its single imm8 operand. The field takes the // low byte silently inside the 32-bit span, matching the scalar convention // (go tool asm encodes INT $256 as CD 00). func (e *enc) encodeInt(ops []Operand) error { if len(ops) != 1 { return fmt.Errorf("INT expects 1 operand, got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("INT operand must be an immediate") } if imm < -(1<<31) || imm > (1<<32)-1 { return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm)) } return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}}) } // encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a // single 32-bit memory operand. func (e *enc) encodeMxcsr(digit int, ops []Operand) error { if len(ops) != 1 { return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops)) } m, ok := ops[0].(Mem) if !ok { return fmt.Errorf("MXCSR instruction requires a memory operand") } i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1} if err := setMem(i, digit, m); err != nil { return err } return e.emit(i) } // cvtIntOp maps the scalar float-to-integer conversions to their mandatory // prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating // CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width. var cvtIntOp = map[string]struct { prefix byte op byte }{ "CVTSD2S": {0xF2, 0x2D}, "CVTTSD2S": {0xF2, 0x2C}, "CVTSS2S": {0xF3, 0x2D}, "CVTTSS2S": {0xF3, 0x2C}, } // encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C // with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the // quad spellings. func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops)) } spec := cvtIntOp[base] src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || dstReg.isVec() { return fmt.Errorf("%s destination must be a general register", base) } i := newInstr(size, []byte{0x0F, spec.op}) i.prefix = spec.prefix if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } // encodeFmov encodes the x87 double move. The memory forms are DD /0 // (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register // move is DD C0+dst (FLD st(dst)), the form the Go assembler emits. func (e *enc) encodeFmov(ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcIsF := src.(Reg) dstReg, dstIsF := dst.(Reg) srcF := srcIsF && srcReg.fp dstF := dstIsF && dstReg.fp switch { case srcF && dstF: // The register form is DD /2 with rm = the destination (FST st(dst)). i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1} if err := setRMDigit(i, 2, dstReg, 8); err != nil { return err } return e.emit(i) case dstF: m, ok := src.(Mem) if !ok { return fmt.Errorf("FMOVD: invalid source operand") } i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1} if err := setMem(i, 0, m); err != nil { return err } return e.emit(i) case srcF: m, ok := dst.(Mem) if !ok { return fmt.Errorf("FMOVD: invalid destination operand") } i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1} if err := setMem(i, 2, m); err != nil { return err } return e.emit(i) } return fmt.Errorf("FMOVD needs an x87 register operand") } // --- legacy SSE imm8, extract, insert and packed shift families -------------- // encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src, // dst with reg = dst, rm = src and the immediate appended last (PALIGNR, // PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4). func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("SSE imm8 instruction needs an immediate first operand") } immByte, err := imm8(int64(imm)) if err != nil { return err } src, dst := ops[1], ops[2] dstReg, ok2 := dst.(Reg) if !ok2 || !dstReg.isVec() { return fmt.Errorf("SSE imm8 instruction destination must be a vector register") } opcode := []byte{0x0F, 0x38, m.op} if m.map3A { opcode = []byte{0x0F, 0x3A, m.op} } i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1} if err := setRM(i, dstReg, src, 8); err != nil { return err } i.imm = []byte{immByte} return e.emit(i) } // encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the // XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and // PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the // SSE4.1 0F3A 15 one). func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("extract needs an immediate first operand") } immByte, err := imm8(int64(imm)) if err != nil { return err } srcReg, srcVec := vecReg(ops[1]) if !srcVec { return fmt.Errorf("extract source must be an XMM register") } opcode := m.op if m.opMem != nil && memOperand(ops[2]) { opcode = m.opMem } i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW} if err := setRM(i, srcReg, ops[2], 8); err != nil { return err } i.imm = []byte{immByte} return e.emit(i) } // encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the // XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and // PINSRW). func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("insert needs an immediate first operand") } immByte, err := imm8(int64(imm)) if err != nil { return err } dstReg, dstVec := vecReg(ops[2]) if !dstVec { return fmt.Errorf("insert destination must be an XMM register") } i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW} if err := setRM(i, dstReg, ops[1], 8); err != nil { return err } i.imm = []byte{immByte} return e.emit(i) } // encodeSSEShift encodes the legacy packed integer shifts. The immediate // form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form // OP count, dst carries the count in an XMM register (or memory) on the // 66 0F D1-F3 opcodes. The destination is always the register written. func (e *enc) encodeSSEShift(name string, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops)) } dstReg, ok := ops[1].(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("%s destination must be the second, vector operand", name) } if imm, isImm := ops[0].(Imm); isImm { spec := sseShiftImm[name] immByte, err := imm8(int64(imm)) if err != nil { return err } i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1} if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil { return err } i.imm = []byte{immByte} return e.emit(i) } if !vecOrMem(ops[0]) { return fmt.Errorf("%s count must be an immediate, a vector register or memory", name) } op, ok := sseShiftVar[name] if !ok { return fmt.Errorf("%s has no variable-count form", name) } i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1} if err := setRM(i, dstReg, ops[0], 8); err != nil { return err } return e.emit(i) } // encodeCmpsd encodes CMPSD, the scalar double compare with its predicate // immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family: // F2 0F C2 with reg = dst, rm = src. func (e *enc) encodeCmpsd(ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops)) } imm, ok := ops[2].(Imm) if !ok { return fmt.Errorf("CMPSD predicate must be an immediate") } immByte, err := imm8(int64(imm)) if err != nil { return err } dstReg, ok2 := ops[1].(Reg) if !ok2 || !dstReg.isVec() { return fmt.Errorf("CMPSD destination must be a vector register") } i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1} if err := setRM(i, dstReg, ops[0], 8); err != nil { return err } i.imm = []byte{immByte} return e.emit(i) } // encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the // literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no // prefix, reg = dst, rm = src; X0 is implicit on the wire). func (e *enc) encodeSha256rnds2(ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops)) } x0, ok := ops[0].(Reg) if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 { return fmt.Errorf("SHA256RNDS2 first operand must be X0") } dstReg, ok2 := ops[2].(Reg) if !ok2 || !dstReg.isVec() { return fmt.Errorf("SHA256RNDS2 destination must be a vector register") } i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1} if err := setRM(i, dstReg, ops[1], 8); err != nil { return err } return e.emit(i) }