// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import "fmt" // aluOp maps an arithmetic/logic mnemonic to its base "r/m, r" opcode (for // 16/32/64-bit; the 8-bit form is one less) and its /digit for the immediate // forms (0x80/0x81/0x83). var aluOp = map[string]struct { rr byte digit int }{ "ADD": {0x01, 0}, "OR": {0x09, 1}, "AND": {0x21, 4}, "SUB": {0x29, 5}, "XOR": {0x31, 6}, "CMP": {0x39, 7}, } // unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use // the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit // mode); NEG/NOT use the 0xF6/0xF7 group. var unaryOp = map[string]struct { digit int op byte }{ "INC": {0, 0xFF}, "DEC": {1, 0xFF}, "NOT": {2, 0xF7}, "NEG": {3, 0xF7}, } // shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group. var shiftOp = map[string]int{ "SHL": 4, "SHR": 5, "SAR": 7, } // --- MOV -------------------------------------------------------------------- func (e *enc) encodeMov(ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("MOV expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] // Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2 // packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E // (reg = dst, no REX.W, the Go assembler's form), xmm→mem as // 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD // opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and // 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m // would be undefined forms. MOVL is the packed-dword move: // 66 0F 6E load, 66 0F 7E store, no REX.W. A GPR-move fallback would // silently emit REX.W 8B with the wrong operand meaning. _, srcVec := vecReg(src) dstReg, dstVec := vecReg(dst) if srcVec || dstVec { if dstVec { if g, ok := src.(Reg); ok && !g.isVec() { i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: size == 8} if err := setRM(i, dstReg, src, 8); err != nil { return err } return e.emit(i) } i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1} if size == 4 { i.prefix = 0x66 i.opcode = []byte{0x0F, 0x6E} } if err := setRM(i, dstReg, src, 8); err != nil { return err } return e.emit(i) } srcXMM, srcIsXMM := src.(Reg) if !srcIsXMM || !srcXMM.isVec() { return fmt.Errorf("MOV: store needs an XMM source") } if g, ok := dst.(Reg); ok && !g.isVec() { i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: size == 8} if err := setRM(i, srcXMM, dst, 8); err != nil { return err } return e.emit(i) } i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1} if size == 4 { i.opcode = []byte{0x0F, 0x7E} } if err := setRM(i, srcXMM, dst, 8); err != nil { return err } return e.emit(i) } dstReg, dstIsReg := dst.(Reg) switch src := src.(type) { case Reg: if dstIsReg { // MOV r/m, r: 0x88/0x89, reg=src, rm=dst, the form the Go // assembler emits for register-to-register moves. i := newInstr(size, []byte{movRM(size)}) if err := setRM(i, src, dst, size); err != nil { return err } return e.emit(i) } // MOV r/m, r: 0x88/0x89, reg=src, rm=dst(mem). i := newInstr(size, []byte{movRM(size)}) if err := setRM(i, src, dst, size); err != nil { return err } return e.emit(i) case Mem: if !dstIsReg { return fmt.Errorf("MOV: two memory operands") } // MOV r, r/m: reg=dst, rm=src(mem). i := newInstr(size, []byte{movRR(size)}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) case sbMem: if !dstIsReg { return fmt.Errorf("MOV: two memory operands") } // MOV r, r/m: reg=dst, rm=src(static symbol). i := newInstr(size, []byte{movRR(size)}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) case Imm: if dstIsReg { v := int64(src) // The Go assembler compresses 64-bit moves whose immediate fits // a signed int32, choosing per sign: // v >= 0: B8+rd imm32 without REX.W (zero-extended by the // hardware, REX.B still emitted for R8-R15); // v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd // form would zero-extend and corrupt the value). // Out-of-range immediates keep the B8+rd imm64 form. if size == 8 && v >= 0 && v <= (1<<31)-1 { i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)}) i.rexB = dstReg.idx >= 8 i.imm = le32(v) return e.emit(i) } if size == 8 && v < 0 && v >= -(1<<31) { i := newInstr(8, []byte{0xC7}) if err := setRMDigit(i, 0, dstReg, 8); err != nil { return err } i.imm = le32(v) return e.emit(i) } opBase := byte(0xB8) if size == 1 { opBase = 0xB0 } i := newInstr(size, []byte{opBase + byte(dstReg.idx&7)}) i.rexB = dstReg.idx >= 8 if dstReg.needsREX(size) { i.rexForced = true } i.imm = immediate(v, size, true) return e.emit(i) } // MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. op := byte(0xC7) if size == 1 { op = 0xC6 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, 0, dst, size); err != nil { return err } i.imm = immediate(int64(src), size, false) return e.emit(i) } return fmt.Errorf("MOV: invalid operands") } func movRR(size int) byte { // MOV r, r/m if size == 1 { return 0x8A } return 0x8B } func movRM(size int) byte { // MOV r/m, r if size == 1 { return 0x88 } return 0x89 } // --- ALU (ADD/OR/AND/SUB/XOR/CMP) ------------------------------------------- func (e *enc) encodeALU(op struct { rr byte digit int }, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("ALU instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] // CMP never takes its immediate first: the Go assembler rejects // CMPL $0, AX outright (only CMPL AX, $0 is legal, unlike TEST and the // writing ALU ops whose immediate is naturally the source). if imm, ok := src.(Imm); ok { if op.digit == 7 { return fmt.Errorf("CMP immediate must be the second operand (reg, $imm)") } return e.encodeALUImm(op.digit, dst, int64(imm), size) } // CMP accepts the immediate in the second position too, CMPL CX, $31 is // the form the Go assembler itself accepts, and encodes it identically // (CMP r/m, imm sets the flags as first − second). No other ALU op takes // an immediate destination. if imm, ok := dst.(Imm); ok { if op.digit != 7 { return fmt.Errorf("immediate must be the source operand") } return e.encodeALUImm(op.digit, src, int64(imm), size) } // CMP records first − second without writing anywhere, so the first // operand must land as the minuend; every other ALU op writes its second // operand and follows the forms below. cmp := op.rr == 0x39 dstReg, dstIsReg := dst.(Reg) srcReg, srcIsReg := src.(Reg) switch { case cmp && dstIsReg: // CMP x, reg: OP r/m, r (0x38/0x39) with rm = first operand, reg = // second, matching the Go assembler. opc := op.rr if size == 1 { opc = op.rr - 1 } i := newInstr(size, []byte{opc}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) case cmp && srcIsReg: // CMP reg, mem: OP r, r/m (0x3A/0x3B) with reg = first operand, rm = // second. opc := op.rr + 2 if size == 1 { opc = op.rr + 1 } i := newInstr(size, []byte{opc}) if err := setRM(i, srcReg, dst, size); err != nil { return err } return e.emit(i) case srcIsReg: // OP r/m, r: reg=src, rm=dst (dst is a register or memory). This is the // form the Go assembler prefers when the source is a register. opc := op.rr if size == 1 { opc = op.rr - 1 } i := newInstr(size, []byte{opc}) if err := setRM(i, srcReg, dst, size); err != nil { return err } return e.emit(i) case dstIsReg: // OP r, r/m: reg=dst, rm=src(memory). opc := op.rr + 2 if size == 1 { opc = op.rr + 1 } i := newInstr(size, []byte{opc}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } return fmt.Errorf("two memory operands") } func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error { if size == 1 { i := newInstr(1, []byte{0x80}) if err := setRMDigit(i, digit, dst, 1); err != nil { return err } i.imm = []byte{byte(int8(imm))} return e.emit(i) } if fits8(imm) { // 0x83 /digit, sign-extended imm8. i := newInstr(size, []byte{0x83}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } i.imm = []byte{byte(int8(imm))} return e.emit(i) } // 0x81 /digit, imm16/imm32, or the Go assembler's accumulator short // form (opcode+5, no ModR/M) when the destination is AX/AL, which it // prefers over the generic form exactly here. if r, ok := dst.(Reg); ok && r.idx == 0 { accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit] i := newInstr(size, []byte{accOp}) i.imm = immediate(imm, size, false) return e.emit(i) } // 0x81 /digit, imm16/imm32. i := newInstr(size, []byte{0x81}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } i.imm = immediate(imm, size, false) return e.emit(i) } // --- TEST ------------------------------------------------------------------- func (e *enc) encodeTest(ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("TEST expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] if imm, ok := src.(Imm); ok { // TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0, but the Go assembler // always uses the accumulator forms (A8/A9, no ModR/M) when the // register operand is AL/AX, whatever the immediate's width. if r, ok := dst.(Reg); ok && r.idx == 0 { op := byte(0xA9) if size == 1 { op = 0xA8 } i := newInstr(size, []byte{op}) i.imm = immediate(int64(imm), size, false) return e.emit(i) } op := byte(0xF7) if size == 1 { op = 0xF6 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, 0, dst, size); err != nil { return err } i.imm = immediate(int64(imm), size, false) return e.emit(i) } srcReg, ok := src.(Reg) if !ok { return fmt.Errorf("TEST: source must be a register or immediate") } // TEST r/m, r: 0x84 (8-bit) / 0x85. op := byte(0x85) if size == 1 { op = 0x84 } i := newInstr(size, []byte{op}) if err := setRM(i, srcReg, dst, size); err != nil { return err } return e.emit(i) } // --- LEA -------------------------------------------------------------------- func (e *enc) encodeLea(ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("LEA expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] // LEAQ addr, reg dstReg, ok := dst.(Reg) if !ok { return fmt.Errorf("LEA: destination must be a register") } switch src.(type) { case Mem, sbMem: default: return fmt.Errorf("LEA: source must be a memory operand") } i := newInstr(size, []byte{0x8D}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } // --- INC/DEC/NEG/NOT -------------------------------------------------------- func (e *enc) encodeUnary(op struct { digit int op byte }, ops []Operand, size int) error { if len(ops) != 1 { return fmt.Errorf("unary instruction expects 1 operand, got %d", len(ops)) } base := op.op if size == 1 { base-- // 0xFF→0xFE, 0xF7→0xF6 } i := newInstr(size, []byte{base}) if err := setRMDigit(i, op.digit, ops[0], size); err != nil { return err } return e.emit(i) } // --- SHL/SHR/SAR ------------------------------------------------------------ func (e *enc) encodeShift(digit int, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("shift expects 2 operands, got %d", len(ops)) } count, dst := ops[0], ops[1] // Count is $1, %CL, or an imm8. if reg, ok := count.(Reg); ok && reg.idx == 1 && reg.size <= 1 { // CL: 0xD2 (8-bit) / 0xD3. op := byte(0xD3) if size == 1 { op = 0xD2 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } return e.emit(i) } imm, ok := count.(Imm) if !ok { return fmt.Errorf("shift count must be $1, CL or an immediate") } if imm == 1 { // 0xD0 (8-bit) / 0xD1. op := byte(0xD1) if size == 1 { op = 0xD0 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } return e.emit(i) } // 0xC0 (8-bit) / 0xC1, imm8. op := byte(0xC1) if size == 1 { op = 0xC0 } i := newInstr(size, []byte{op}) if err := setRMDigit(i, digit, dst, size); err != nil { return err } i.imm = []byte{byte(int8(imm))} return e.emit(i) } // --- IMUL ------------------------------------------------------------------- func (e *enc) encodeImul(ops []Operand, size int) error { switch len(ops) { case 2: // IMUL r, r/m: 0x0F 0xAF. dstReg, ok := ops[1].(Reg) if !ok { return fmt.Errorf("IMUL: destination must be a register") } i := newInstr(size, []byte{0x0F, 0xAF}) if err := setRM(i, dstReg, ops[0], size); err != nil { return err } return e.emit(i) case 3: // IMUL r, r/m, imm: 0x6B (imm8) / 0x69 (imm16/32). dstReg, ok := ops[2].(Reg) if !ok { return fmt.Errorf("IMUL: destination must be a register") } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("IMUL: immediate operand expected first") } // Plan 9 order: IMUL $imm, src, dst. if fits8(int64(imm)) { i := newInstr(size, []byte{0x6B}) if err := setRM(i, dstReg, ops[1], size); err != nil { return err } i.imm = []byte{byte(int8(imm))} return e.emit(i) } i := newInstr(size, []byte{0x69}) if err := setRM(i, dstReg, ops[1], size); err != nil { return err } i.imm = immediate(int64(imm), size, false) return e.emit(i) } return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops)) } // --- PUSH / POP ------------------------------------------------------------- func (e *enc) encodePushPop(ops []Operand, push bool) error { if len(ops) != 1 { return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops)) } switch op := ops[0].(type) { case Reg: base := byte(0x50) // PUSH r; POP is 0x58 if !push { base = 0x58 } // PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed. i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1} i.rexB = op.idx >= 8 return e.emit(i) case Mem: opc := byte(0xFF) // PUSH r/m: /6 digit := 6 if !push { opc = 0x8F // POP r/m: /0 digit = 0 } i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1} if err := setRMDigit(i, digit, ops[0], 8); err != nil { return err } return e.emit(i) case Imm: if !push { return fmt.Errorf("POP does not take an immediate") } if fits8(int64(op)) { i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}} return e.emit(i) } i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))} return e.emit(i) } return fmt.Errorf("PUSH/POP: invalid operand") } // --- RET / JMP / CALL / Jcc ------------------------------------------------- func (e *enc) encodeRet() error { return e.emit(&instr{opcode: []byte{0xC3}, modrm: -1, sib: -1}) } // encodeJmpRel encodes JMP/CALL with a relative displacement (the operand is an // Imm holding the already-computed rel32 offset). func (e *enc) encodeJmpRel(ops []Operand, opcode []byte) error { if len(ops) != 1 { return fmt.Errorf("JMP/CALL expects 1 operand, got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("JMP/CALL: relative offset must be an immediate (labels are resolved by the assembler)") } return e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, imm: le32(int64(imm))}) } // condCode maps a Plan 9 conditional-jump mnemonic to its x86 condition code. func condCode(upper string) (int, bool) { if len(upper) < 2 || upper[0] != 'J' || upper == "JMP" { return 0, false } cc, ok := jccMap[upper[1:]] return cc, ok } var jccMap = map[string]int{ "O": 0x0, "NO": 0x1, "OS": 0x0, "OC": 0x1, "B": 0x2, "C": 0x2, "NAE": 0x2, "CS": 0x2, "NB": 0x3, "NC": 0x3, "AE": 0x3, "CC": 0x3, "E": 0x4, "Z": 0x4, "EQ": 0x4, "NE": 0x5, "NZ": 0x5, "BE": 0x6, "NA": 0x6, "LS": 0x6, "NBE": 0x7, "A": 0x7, "HI": 0x7, "S": 0x8, "MI": 0x8, "NS": 0x9, "PL": 0x9, "P": 0xA, "PE": 0xA, "PS": 0xA, "NP": 0xB, "PO": 0xB, "PC": 0xB, "L": 0xC, "NGE": 0xC, "LT": 0xC, "NL": 0xD, "GE": 0xD, "LE": 0xE, "NG": 0xE, "NLE": 0xF, "G": 0xF, "GT": 0xF, } func (e *enc) encodeJcc(cc int, ops []Operand) error { if len(ops) != 1 { return fmt.Errorf("conditional jump expects 1 operand, got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("conditional jump: relative offset must be an immediate") } if fits8(int64(imm)) { // Short form: 0x70+cc, rel8. return e.emit(&instr{opcode: []byte{0x70 + byte(cc)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}}) } // Near form: 0x0F 0x80+cc, rel32. return e.emit(&instr{opcode: []byte{0x0F, 0x80 + byte(cc)}, modrm: -1, sib: -1, imm: le32(int64(imm))}) } // immediate encodes an immediate of the given operand size. full64 selects the // 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit // sign-extended immediate is used for 64-bit operands. func immediate(v int64, size int, full64 bool) []byte { switch size { case 1: return []byte{byte(int8(v))} case 2: return le16(v) case 4: return le32(v) default: // 8 if full64 { return le64(v) } return le32(v) // sign-extended imm32 } } // --- CMOVcc / SETcc --------------------------------------------------------- // encodeCmov encodes a conditional move: CMOV + size (W/L/Q) + condition // (CMOVLGT, CMOVQEQ, …). The condition reads exactly like the Jcc spellings; // the instruction is 0F 40+cc with reg = dst, rm = src. func (e *enc) encodeCmov(upper string, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("CMOVcc expects 2 operands, got %d", len(ops)) } rest := upper[len("CMOV"):] if len(rest) < 2 { return fmt.Errorf("unsupported instruction %q", upper) } var size int switch rest[0] { case 'W': size = 2 case 'L': size = 4 case 'Q': size = 8 default: return fmt.Errorf("unsupported instruction %q", upper) } cc, ok := jccMap[rest[1:]] if !ok { return fmt.Errorf("unsupported instruction %q", upper) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok { return fmt.Errorf("CMOVcc destination must be a register") } i := newInstr(size, []byte{0x0F, byte(0x40 + cc)}) if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) } // encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …), // always a byte write, 0F 90+cc /0 into a register or memory operand. func (e *enc) encodeSet(upper string, ops []Operand) error { if len(ops) != 1 { return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops)) } cond := upper[len("SET"):] cc, ok := jccMap[cond] if !ok || cond == "" { return fmt.Errorf("unsupported instruction %q", upper) } i := &instr{opcode: []byte{0x0F, byte(0x90 + cc)}, modrm: -1, sib: -1} if err := setRMDigit(i, 0, ops[0], 1); err != nil { return err } return e.emit(i) } // --- bit scan / bit count ---------------------------------------------------- // countOp maps the bit-scan and bit-count mnemonics to their opcode byte and // mandatory prefix. TZCNT/LZCNT/POPCNT are the F3-prefixed forms of the // same map as BSF/BSR's 0F BC/BD; POPCNT is F3 0F B8. var countOp = map[string]struct { op byte prefix byte }{ "BSF": {0xBC, 0}, "BSR": {0xBD, 0}, "TZCNT": {0xBC, 0xF3}, "LZCNT": {0xBD, 0xF3}, "POPCNT": {0xB8, 0xF3}, } // encodeCount encodes the bit-scan and bit-count family, BSF (0F BC), // BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8) // with reg = dst and rm = src. The size suffix selects the operand width // (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the // source is zero (unlike their F3-prefixed counterparts); callers must // guard non-zero inputs themselves. func (e *enc) encodeCount(base string, ops []Operand, size int) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops)) } spec := countOp[base] dstReg, ok := ops[1].(Reg) if !ok { return fmt.Errorf("%s destination must be a register", base) } i := newInstr(size, []byte{0x0F, spec.op}) i.prefix = spec.prefix if err := setRM(i, dstReg, ops[0], size); err != nil { return err } return e.emit(i) } // encodeBswap encodes BSWAP: the single register operand is encoded in the // opcode byte (0F C8+r), with REX.B for R8-R15 and REX.W for the quad form. func (e *enc) encodeBswap(ops []Operand, size int) error { if len(ops) != 1 { return fmt.Errorf("BSWAP expects 1 operand, got %d", len(ops)) } reg, ok := ops[0].(Reg) if !ok { return fmt.Errorf("BSWAP operand must be a register") } i := newInstr(size, []byte{0x0F, 0xC8 + byte(reg.idx&7)}) i.rexB = reg.idx >= 8 return e.emit(i) } // --- mixed-width sign/zero-extending moves ----------------------------------- // movExtendOp maps Go's mixed-width move names to their opcode and destination // width. The source is narrower than the destination, so the plain size-suffix // convention does not apply to these names. var movExtendOp = map[string]struct { op []byte dst64 bool }{ "MOVBLZX": {[]byte{0x0F, 0xB6}, false}, // byte → long, zero-extend "MOVBQZX": {[]byte{0x0F, 0xB6}, true}, // byte → quad, zero-extend "MOVWLZX": {[]byte{0x0F, 0xB7}, false}, // word → long, zero-extend "MOVWQZX": {[]byte{0x0F, 0xB7}, true}, // word → quad, zero-extend "MOVWLSX": {[]byte{0x0F, 0xBF}, false}, // word → long, sign-extend "MOVLQSX": {[]byte{0x63}, true}, // long → quad, sign-extend (MOVSXD) } // encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider // operand), rm = src. func (e *enc) encodeMovExtend(base string, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops)) } spec := movExtendOp[base] dstReg, ok := ops[1].(Reg) if !ok { return fmt.Errorf("%s destination must be a register", base) } size := 4 if spec.dst64 { size = 8 } i := newInstr(size, spec.op) if err := setRM(i, dstReg, ops[0], size); err != nil { return err } return e.emit(i) } // --- legacy SSE moves -------------------------------------------------------- // sseMove describes a legacy (non-VEX) SSE move: a mandatory prefix plus a // load opcode (reg = destination, rm = source) and a store opcode (the // reverse). The Plan 9 names MOVOU/MOVO are the integer unaligned/aligned // octa moves (MOVDQU/MOVDQA), not the packed-single ones. type sseMove struct { prefix byte // 0, 0x66, 0xF2 or 0xF3 load byte store byte } var sseMoveTable = map[string]sseMove{ "MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa "MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa "MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single "MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single "MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double "MOVAPD": {0x66, 0x28, 0x29}, // aligned packed double "MOVSD": {0xF2, 0x10, 0x11}, // scalar double "MOVSS": {0xF3, 0x10, 0x11}, // scalar single } // encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the // load form (reg = destination), matching the Go assembler. func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("SSE move expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcVec := vecReg(src) dstReg, dstVec := vecReg(dst) op := m.store var reg Reg var rm Operand switch { case srcVec && dstVec: op = m.load reg, rm = dstReg, src case srcVec: if _, ok := dst.(Mem); !ok { return fmt.Errorf("SSE move: invalid destination operand") } reg, rm = srcReg, dst case dstVec: if _, ok := src.(Mem); !ok { return fmt.Errorf("SSE move: invalid source operand") } op = m.load reg, rm = dstReg, src default: return fmt.Errorf("SSE move needs a vector register operand") } i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1} if err := setRM(i, reg, rm, 8); err != nil { return err } return e.emit(i) } // --- legacy SSE packed binary and shuffles ----------------------------------- // sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an // optional mandatory prefix plus the 0F-prefixed opcode (0F38 for the // SSSE3 integer shuffles). Plan 9 asm lists the source operand first, so // MULPS X0, X1 computes X1 = X1 * X0. type sseBin struct { prefix byte // 0, 0x66, 0xF2 or 0xF3 op byte map38 bool // opcode lives under 0F38 instead of 0F } var sseBinTable = map[string]sseBin{ "ADDPS": {0, 0x58, false}, "ADDPD": {0x66, 0x58, false}, "MULPS": {0, 0x59, false}, "MULPD": {0x66, 0x59, false}, "SUBPS": {0, 0x5C, false}, "SUBPD": {0x66, 0x5C, false}, "DIVPS": {0, 0x5E, false}, "DIVPD": {0x66, 0x5E, false}, "ANDPS": {0, 0x54, false}, "ANDPD": {0x66, 0x54, false}, "ORPS": {0, 0x56, false}, "ORPD": {0x66, 0x56, false}, "XORPS": {0, 0x57, false}, "XORPD": {0x66, 0x57, false}, "MINPS": {0, 0x5D, false}, "MINPD": {0x66, 0x5D, false}, "MAXPS": {0, 0x5F, false}, "MAXPD": {0x66, 0x5F, false}, "ADDSS": {0xF3, 0x58, false}, "ADDSD": {0xF2, 0x58, false}, "MULSS": {0xF3, 0x59, false}, "MULSD": {0xF2, 0x59, false}, "SUBSS": {0xF3, 0x5C, false}, "SUBSD": {0xF2, 0x5C, false}, "DIVSS": {0xF3, 0x5E, false}, "DIVSD": {0xF2, 0x5E, false}, "MINSS": {0xF3, 0x5D, false}, "MINSD": {0xF2, 0x5D, false}, "MAXSS": {0xF3, 0x5F, false}, "MAXSD": {0xF2, 0x5F, false}, "UNPCKLPS": {0, 0x14, false}, "UNPCKHPS": {0, 0x15, false}, "UNPCKLPD": {0x66, 0x14, false}, "UNPCKHPD": {0x66, 0x15, false}, "CVTSS2SD": {0xF3, 0x5A, false}, "CVTSD2SS": {0xF2, 0x5A, false}, "CVTPS2PD": {0, 0x5A, false}, "CVTPD2PS": {0x66, 0x5A, false}, // SSE2 packed integers (reg = reg op rm) and the SSSE3 byte shuffle. "PXOR": {0x66, 0xEF, false}, "POR": {0x66, 0xEB, false}, "PAND": {0x66, 0xDB, false}, "PANDN": {0x66, 0xDF, false}, "PADDB": {0x66, 0xFC, false}, "PADDW": {0x66, 0xFD, false}, "PADDD": {0x66, 0xFE, false}, "PADDQ": {0x66, 0xD4, false}, "PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false}, "PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false}, "PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false}, "PCMPEQD": {0x66, 0x76, false}, "PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false}, "PCMPGTD": {0x66, 0x66, false}, "PSHUFB": {0x66, 0x00, true}, } // sseShuf describes a legacy SSE shuffle taking a trailing imm8 // (PSHUFD/PSHUFHW/PSHUFLW also carry the packed-int 0x66/F3/F2 prefixes). type sseShuf struct { prefix byte op byte } var sseShufTable = map[string]sseShuf{ "SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6}, "PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70}, } // encodeSSEBin encodes reg = reg op rm (memory allowed for rm). func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("SSE binary expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("SSE binary destination must be a vector register") } opcode := []byte{0x0F, m.op} if m.map38 { opcode = []byte{0x0F, 0x38, m.op} } i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1} if err := setRM(i, dstReg, src, 8); err != nil { return err } return e.emit(i) } // encodeSSEShuf encodes an imm8 shuffle: SHUFPS $imm, src, dst. func (e *enc) encodeSSEShuf(m sseShuf, ops []Operand) error { if len(ops) != 3 { return fmt.Errorf("SSE shuffle expects 3 operands, got %d", len(ops)) } imm, ok := ops[0].(Imm) if !ok { return fmt.Errorf("SSE shuffle needs an imm8 first operand") } if imm < -128 || imm > 255 { return fmt.Errorf("SSE shuffle imm8 %d out of range", imm) } src, dst := ops[1], ops[2] dstReg, ok2 := dst.(Reg) if !ok2 || !dstReg.isVec() { return fmt.Errorf("SSE shuffle destination must be a vector register") } i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1} if err := setRM(i, dstReg, src, 8); err != nil { return err } i.imm = []byte{byte(int8(imm))} return e.emit(i) } // --- CVTSL2SD / CVTSQ2SD ----------------------------------------------------- // encodeCvtsi2sd encodes a signed integer to scalar double conversion // (CVTSL2SD from a 32-bit, CVTSQ2SD from a 64-bit source): F2 0F 2A with // reg = XMM dst, rm = GPR/memory src. The Go assembler emits the legacy SSE // encoding here, not the VEX form, so we match it byte for byte. func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("CVTSx2SD expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("CVTSx2SD destination must be a vector register") } size := 4 if quad { size = 8 } i := newInstr(size, []byte{0x0F, 0x2A}) i.prefix = 0xF2 if err := setRM(i, dstReg, src, size); err != nil { return err } return e.emit(i) }