feat(amd64): encode the GOROOT instruction families
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+630
-9
@@ -14,30 +14,70 @@ var aluOp = map[string]struct {
|
||||
}{
|
||||
"ADD": {0x01, 0},
|
||||
"OR": {0x09, 1},
|
||||
"ADC": {0x11, 2},
|
||||
"SBB": {0x19, 3},
|
||||
"AND": {0x21, 4},
|
||||
"SUB": {0x29, 5},
|
||||
"XOR": {0x31, 6},
|
||||
"CMP": {0x39, 7},
|
||||
}
|
||||
|
||||
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
|
||||
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
|
||||
// mode); NEG/NOT use the 0xF6/0xF7 group.
|
||||
// unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode.
|
||||
// INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes
|
||||
// in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4,
|
||||
// DIV /6, IDIV /7; the accumulator is the implicit other operand).
|
||||
var unaryOp = map[string]struct {
|
||||
digit int
|
||||
op byte
|
||||
}{
|
||||
"INC": {0, 0xFF},
|
||||
"DEC": {1, 0xFF},
|
||||
"NOT": {2, 0xF7},
|
||||
"NEG": {3, 0xF7},
|
||||
"INC": {0, 0xFF},
|
||||
"DEC": {1, 0xFF},
|
||||
"NOT": {2, 0xF7},
|
||||
"NEG": {3, 0xF7},
|
||||
"MUL": {4, 0xF7},
|
||||
"DIV": {6, 0xF7},
|
||||
"IDIV": {7, 0xF7},
|
||||
}
|
||||
|
||||
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
|
||||
// shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the
|
||||
// 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4).
|
||||
var shiftOp = map[string]int{
|
||||
"SHL": 4,
|
||||
"SAL": 4,
|
||||
"SHR": 5,
|
||||
"SAR": 7,
|
||||
"ROL": 0,
|
||||
"ROR": 1,
|
||||
"RCL": 2,
|
||||
"RCR": 3,
|
||||
}
|
||||
|
||||
// bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form;
|
||||
// the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's
|
||||
// opcode row.
|
||||
var bitTestOp = map[string]int{
|
||||
"BT": 4,
|
||||
"BTS": 5,
|
||||
"BTR": 6,
|
||||
"BTC": 7,
|
||||
}
|
||||
|
||||
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
|
||||
// fence names carry their opcode inside the 0F AE /digit group spelled out in
|
||||
// full (E8/F0/F8), and PAUSE is F3 90.
|
||||
var noOperandTable = map[string][]byte{
|
||||
"CPUID": {0x0F, 0xA2},
|
||||
"RDTSC": {0x0F, 0x31},
|
||||
"RDTSCP": {0x0F, 0x01, 0xF9},
|
||||
"SYSCALL": {0x0F, 0x05},
|
||||
"XGETBV": {0x0F, 0x01, 0xD0},
|
||||
"CLD": {0xFC},
|
||||
"STD": {0xFD},
|
||||
"PAUSE": {0xF3, 0x90},
|
||||
"LFENCE": {0x0F, 0xAE, 0xE8},
|
||||
"MFENCE": {0x0F, 0xAE, 0xF0},
|
||||
"SFENCE": {0x0F, 0xAE, 0xF8},
|
||||
"UNDEF": {0x0F, 0x0B},
|
||||
}
|
||||
|
||||
// --- MOV --------------------------------------------------------------------
|
||||
@@ -309,6 +349,13 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
|
||||
// the destination is AL, the form the Go assembler prefers here.
|
||||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
|
||||
i.imm = immBytes
|
||||
return e.emit(i)
|
||||
}
|
||||
i := newInstr(1, []byte{0x80})
|
||||
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
||||
return err
|
||||
@@ -913,6 +960,7 @@ type sseMove struct {
|
||||
var sseMoveTable = map[string]sseMove{
|
||||
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
|
||||
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
|
||||
"MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias
|
||||
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
||||
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
||||
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
||||
@@ -1000,10 +1048,120 @@ var sseBinTable = map[string]sseBin{
|
||||
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
||||
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
||||
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
||||
"PCMPEQD": {0x66, 0x76, false},
|
||||
"PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false},
|
||||
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
||||
"PCMPGTD": {0x66, 0x66, false},
|
||||
"PSHUFB": {0x66, 0x00, true},
|
||||
// Scalar compares and square root, packed adds/subtracts and the byte
|
||||
// unpack, the spellings the Plan 9 table uses (COMISD orders the
|
||||
// operands like every other two-operand form).
|
||||
"ANDNPD": {0x66, 0x55, false},
|
||||
"ANDNPS": {0x00, 0x55, false},
|
||||
"COMISD": {0x66, 0x2F, false},
|
||||
"SQRTSD": {0xF2, 0x51, false},
|
||||
"PADDL": {0x66, 0xFE, false},
|
||||
"PSUBL": {0x66, 0xFA, false},
|
||||
"PUNPCKLBW": {0x66, 0x60, false},
|
||||
// AES round functions (66 0F38) and the SHA message schedule helpers
|
||||
// (no prefix, 0F38).
|
||||
"AESENC": {0x66, 0xDC, true},
|
||||
"AESENCLAST": {0x66, 0xDD, true},
|
||||
"AESDEC": {0x66, 0xDE, true},
|
||||
"AESDECLAST": {0x66, 0xDF, true},
|
||||
"AESIMC": {0x66, 0xDB, true},
|
||||
"SHA1MSG1": {0x00, 0xC9, true},
|
||||
"SHA1MSG2": {0x00, 0xCA, true},
|
||||
"SHA1NEXTE": {0x00, 0xC8, true},
|
||||
"SHA256MSG1": {0x00, 0xCC, true},
|
||||
"SHA256MSG2": {0x00, 0xCD, true},
|
||||
}
|
||||
|
||||
// sseImm3 describes a legacy SSE instruction taking a leading imm8 and two
|
||||
// further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and
|
||||
// map3A select the opcode map the same way as sseBin's.
|
||||
type sseImm3 struct {
|
||||
prefix byte
|
||||
op byte
|
||||
map3A bool // opcode lives under 0F3A instead of 0F38
|
||||
}
|
||||
|
||||
// sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3
|
||||
// align/blend shuffles, the string compare, carry-less multiply and the AES
|
||||
// key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings.
|
||||
var sseImm3Table = map[string]sseImm3{
|
||||
"PALIGNR": {0x66, 0x0F, true},
|
||||
"PBLENDW": {0x66, 0x0E, true},
|
||||
"PCMPESTRI": {0x66, 0x61, true},
|
||||
"PCLMULQDQ": {0x66, 0x44, true},
|
||||
"AESKEYGENASSIST": {0x66, 0xDF, true},
|
||||
"SHA1RNDS4": {0x00, 0xCC, true},
|
||||
}
|
||||
|
||||
// sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM
|
||||
// source and rm = the destination (GPR or memory). PEXTRW's GPR destination
|
||||
// uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one,
|
||||
// so it carries both opcodes.
|
||||
type sseExtract struct {
|
||||
op []byte
|
||||
opMem []byte // used when the destination is memory; nil shares op
|
||||
rexW bool // PEXTRQ's REX.W
|
||||
}
|
||||
|
||||
var sseExtractTable = map[string]sseExtract{
|
||||
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
|
||||
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
|
||||
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
|
||||
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
|
||||
}
|
||||
|
||||
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
|
||||
// destination and rm = the source (GPR or memory).
|
||||
type sseInsert struct {
|
||||
op []byte
|
||||
rexW bool // PINSRQ's REX.W
|
||||
}
|
||||
|
||||
var sseInsertTable = map[string]sseInsert{
|
||||
"PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false},
|
||||
"PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false},
|
||||
"PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true},
|
||||
"PINSRW": {[]byte{0x0F, 0xC4}, false},
|
||||
}
|
||||
|
||||
// sseShiftImm maps the legacy packed integer shifts' immediate form:
|
||||
// OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L
|
||||
// (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ.
|
||||
var sseShiftImm = map[string]sseShift{
|
||||
"PSLLW": {0x71, 6},
|
||||
"PSRLW": {0x71, 2},
|
||||
"PSRAW": {0x71, 4},
|
||||
"PSLLL": {0x72, 6},
|
||||
"PSRLL": {0x72, 2},
|
||||
"PSRAL": {0x72, 4},
|
||||
"PSLLQ": {0x73, 6},
|
||||
"PSRLQ": {0x73, 2},
|
||||
"PSLLDQ": {0x73, 7},
|
||||
"PSRLDQ": {0x73, 3},
|
||||
}
|
||||
|
||||
// sseShiftVar maps the variable-count forms (the count comes from an XMM
|
||||
// register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no
|
||||
// variable form.
|
||||
var sseShiftVar = map[string]byte{
|
||||
"PSLLW": 0xF1,
|
||||
"PSRLW": 0xD1,
|
||||
"PSRAW": 0xE1,
|
||||
"PSLLL": 0xF2,
|
||||
"PSRLL": 0xD2,
|
||||
"PSRAL": 0xE2,
|
||||
"PSLLQ": 0xF3,
|
||||
"PSRLQ": 0xD3,
|
||||
}
|
||||
|
||||
// sseShift is one /digit selector in the 0F 71/72/73 immediate group.
|
||||
type sseShift struct {
|
||||
op byte
|
||||
digit int
|
||||
}
|
||||
|
||||
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
||||
@@ -1016,6 +1174,7 @@ type sseShuf struct {
|
||||
var sseShufTable = map[string]sseShuf{
|
||||
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
||||
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
||||
"PSHUFL": {0x66, 0x70},
|
||||
}
|
||||
|
||||
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
||||
@@ -1090,3 +1249,465 @@ func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- carry, bit test, exchange and accumulate -------------------------------
|
||||
|
||||
// encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9
|
||||
// order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with
|
||||
// reg = index, rm = target); an immediate index uses 0F BA /digit with imm8.
|
||||
func (e *enc) encodeBitTest(name string, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||
}
|
||||
digit := bitTestOp[name]
|
||||
index, target := ops[0], ops[1]
|
||||
if reg, ok := index.(Reg); ok {
|
||||
// Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC),
|
||||
// the /digit base plus eight per step.
|
||||
i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3})
|
||||
if err := setRM(i, reg, target, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
imm, ok := index.(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s index must be a register or an immediate", name)
|
||||
}
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, 0xBA})
|
||||
if err := setRMDigit(i, digit, target, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeExchange encodes XCHG. A register-to-register exchange where either
|
||||
// operand is AX uses the 0x90+r accumulator form (with REX.W for the quad
|
||||
// form, as the Go assembler emits it); everything else uses 0x86/0x87 with
|
||||
// the register operand in ModRM.reg, the memory (or second register) in r/m.
|
||||
func (e *enc) encodeExchange(ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
srcReg, srcIsReg := src.(Reg)
|
||||
dstReg, dstIsReg := dst.(Reg)
|
||||
if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) {
|
||||
// 0x90+r: r is the non-AX register, whichever side it sits on.
|
||||
r := dstReg
|
||||
if srcReg.idx == 0 {
|
||||
r = dstReg
|
||||
} else {
|
||||
r = srcReg
|
||||
}
|
||||
i := newInstr(size, []byte{0x90 + byte(r.idx&7)})
|
||||
i.rexB = r.idx >= 8
|
||||
return e.emit(i)
|
||||
}
|
||||
op := byte(0x87)
|
||||
if size == 1 {
|
||||
op = 0x86
|
||||
}
|
||||
switch {
|
||||
case srcIsReg:
|
||||
i := newInstr(size, []byte{op})
|
||||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case dstIsReg:
|
||||
i := newInstr(size, []byte{op})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("XCHG: at least one operand must be a register")
|
||||
}
|
||||
|
||||
// encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG
|
||||
// (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the
|
||||
// destination writable (register or memory).
|
||||
func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||
}
|
||||
srcReg, ok := ops[0].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s source must be a register", name)
|
||||
}
|
||||
opc := op
|
||||
if size == 1 {
|
||||
opc = op8
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, opc})
|
||||
if err := setRM(i, srcReg, ops[1], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for
|
||||
// the rest; the word form carries a 0x66 operand-size prefix (66 F2, the
|
||||
// prefix order the Go assembler emits) and the quad form REX.W. reg = GPR
|
||||
// accumulator, rm = the data source.
|
||||
func (e *enc) encodeCrc32(ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("CRC32 destination must be a general register")
|
||||
}
|
||||
i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1}
|
||||
if size > 1 {
|
||||
i.opcode[2] = 0xF1
|
||||
}
|
||||
i.rexW = size == 8
|
||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg =
|
||||
// destination, rm = source, the carry/overflow flag as the carry-in.
|
||||
func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("ADCX/ADOX destination must be a general register")
|
||||
}
|
||||
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8}
|
||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- string primitives, flags and INT ----------------------------------------
|
||||
|
||||
// encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and
|
||||
// STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or
|
||||
// REX.W prefix.
|
||||
func (e *enc) encodeStringOp(base string, ops []Operand, size int) error {
|
||||
if len(ops) != 0 {
|
||||
return fmt.Errorf("%s takes no operands, got %d", base, len(ops))
|
||||
}
|
||||
var op byte
|
||||
switch base {
|
||||
case "MOVS":
|
||||
op = 0xA5
|
||||
if size == 1 {
|
||||
op = 0xA4
|
||||
}
|
||||
case "STOS":
|
||||
op = 0xAB
|
||||
if size == 1 {
|
||||
op = 0xAA
|
||||
}
|
||||
default:
|
||||
return fmt.Errorf("unsupported string instruction %q", base)
|
||||
}
|
||||
return e.emit(newInstr(size, []byte{op}))
|
||||
}
|
||||
|
||||
// encodeInt encodes INT with its single imm8 operand. The field takes the
|
||||
// low byte silently inside the 32-bit span, matching the scalar convention
|
||||
// (go tool asm encodes INT $256 as CD 00).
|
||||
func (e *enc) encodeInt(ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("INT expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("INT operand must be an immediate")
|
||||
}
|
||||
if imm < -(1<<31) || imm > (1<<32)-1 {
|
||||
return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm))
|
||||
}
|
||||
return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}})
|
||||
}
|
||||
|
||||
// encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a
|
||||
// single 32-bit memory operand.
|
||||
func (e *enc) encodeMxcsr(digit int, ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
m, ok := ops[0].(Mem)
|
||||
if !ok {
|
||||
return fmt.Errorf("MXCSR instruction requires a memory operand")
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1}
|
||||
if err := setMem(i, digit, m); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// cvtIntOp maps the scalar float-to-integer conversions to their mandatory
|
||||
// prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating
|
||||
// CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width.
|
||||
var cvtIntOp = map[string]struct {
|
||||
prefix byte
|
||||
op byte
|
||||
}{
|
||||
"CVTSD2S": {0xF2, 0x2D},
|
||||
"CVTTSD2S": {0xF2, 0x2C},
|
||||
"CVTSS2S": {0xF3, 0x2D},
|
||||
"CVTTSS2S": {0xF3, 0x2C},
|
||||
}
|
||||
|
||||
// encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C
|
||||
// with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the
|
||||
// quad spellings.
|
||||
func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||
}
|
||||
spec := cvtIntOp[base]
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("%s destination must be a general register", base)
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, spec.op})
|
||||
i.prefix = spec.prefix
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeFmov encodes the x87 double move. The memory forms are DD /0
|
||||
// (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register
|
||||
// move is DD C0+dst (FLD st(dst)), the form the Go assembler emits.
|
||||
func (e *enc) encodeFmov(ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
srcReg, srcIsF := src.(Reg)
|
||||
dstReg, dstIsF := dst.(Reg)
|
||||
srcF := srcIsF && srcReg.fp
|
||||
dstF := dstIsF && dstReg.fp
|
||||
switch {
|
||||
case srcF && dstF:
|
||||
// The register form is DD /2 with rm = the destination (FST st(dst)).
|
||||
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||
if err := setRMDigit(i, 2, dstReg, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case dstF:
|
||||
m, ok := src.(Mem)
|
||||
if !ok {
|
||||
return fmt.Errorf("FMOVD: invalid source operand")
|
||||
}
|
||||
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||
if err := setMem(i, 0, m); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case srcF:
|
||||
m, ok := dst.(Mem)
|
||||
if !ok {
|
||||
return fmt.Errorf("FMOVD: invalid destination operand")
|
||||
}
|
||||
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||
if err := setMem(i, 2, m); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("FMOVD needs an x87 register operand")
|
||||
}
|
||||
|
||||
// --- legacy SSE imm8, extract, insert and packed shift families --------------
|
||||
|
||||
// encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src,
|
||||
// dst with reg = dst, rm = src and the immediate appended last (PALIGNR,
|
||||
// PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4).
|
||||
func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("SSE imm8 instruction needs an immediate first operand")
|
||||
}
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
src, dst := ops[1], ops[2]
|
||||
dstReg, ok2 := dst.(Reg)
|
||||
if !ok2 || !dstReg.isVec() {
|
||||
return fmt.Errorf("SSE imm8 instruction destination must be a vector register")
|
||||
}
|
||||
opcode := []byte{0x0F, 0x38, m.op}
|
||||
if m.map3A {
|
||||
opcode = []byte{0x0F, 0x3A, m.op}
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, src, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the
|
||||
// XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and
|
||||
// PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the
|
||||
// SSE4.1 0F3A 15 one).
|
||||
func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("extract needs an immediate first operand")
|
||||
}
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
srcReg, srcVec := vecReg(ops[1])
|
||||
if !srcVec {
|
||||
return fmt.Errorf("extract source must be an XMM register")
|
||||
}
|
||||
opcode := m.op
|
||||
if m.opMem != nil && memOperand(ops[2]) {
|
||||
opcode = m.opMem
|
||||
}
|
||||
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
|
||||
if err := setRM(i, srcReg, ops[2], 8); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the
|
||||
// XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and
|
||||
// PINSRW).
|
||||
func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("insert needs an immediate first operand")
|
||||
}
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dstReg, dstVec := vecReg(ops[2])
|
||||
if !dstVec {
|
||||
return fmt.Errorf("insert destination must be an XMM register")
|
||||
}
|
||||
i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW}
|
||||
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSSEShift encodes the legacy packed integer shifts. The immediate
|
||||
// form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form
|
||||
// OP count, dst carries the count in an XMM register (or memory) on the
|
||||
// 66 0F D1-F3 opcodes. The destination is always the register written.
|
||||
func (e *enc) encodeSSEShift(name string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("%s destination must be the second, vector operand", name)
|
||||
}
|
||||
if imm, isImm := ops[0].(Imm); isImm {
|
||||
spec := sseShiftImm[name]
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||||
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return e.emit(i)
|
||||
}
|
||||
if !vecOrMem(ops[0]) {
|
||||
return fmt.Errorf("%s count must be an immediate, a vector register or memory", name)
|
||||
}
|
||||
op, ok := sseShiftVar[name]
|
||||
if !ok {
|
||||
return fmt.Errorf("%s has no variable-count form", name)
|
||||
}
|
||||
i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeCmpsd encodes CMPSD, the scalar double compare with its predicate
|
||||
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
|
||||
// F2 0F C2 with reg = dst, rm = src.
|
||||
func (e *enc) encodeCmpsd(ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
|
||||
}
|
||||
imm, ok := ops[2].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("CMPSD predicate must be an immediate")
|
||||
}
|
||||
immByte, err := imm8(int64(imm))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dstReg, ok2 := ops[1].(Reg)
|
||||
if !ok2 || !dstReg.isVec() {
|
||||
return fmt.Errorf("CMPSD destination must be a vector register")
|
||||
}
|
||||
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the
|
||||
// literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no
|
||||
// prefix, reg = dst, rm = src; X0 is implicit on the wire).
|
||||
func (e *enc) encodeSha256rnds2(ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops))
|
||||
}
|
||||
x0, ok := ops[0].(Reg)
|
||||
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
|
||||
return fmt.Errorf("SHA256RNDS2 first operand must be X0")
|
||||
}
|
||||
dstReg, ok2 := ops[2].(Reg)
|
||||
if !ok2 || !dstReg.isVec() {
|
||||
return fmt.Errorf("SHA256RNDS2 destination must be a vector register")
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user