feat(amd64): encode the GOROOT instruction families
Assisted-by: DeepSeek V4.1 Flash
This commit is contained in:
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
|
|||||||
for i := 0; i <= 7; i++ {
|
for i := 0; i <= 7; i++ {
|
||||||
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
||||||
}
|
}
|
||||||
|
// x87 stack registers (FMOVD and the other x87 moves).
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
|
||||||
|
}
|
||||||
return regs
|
return regs
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+32
-8
@@ -18,7 +18,11 @@ func Encodable(mnemonic string) bool {
|
|||||||
|
|
||||||
// Fixed-name instructions (no size suffix).
|
// Fixed-name instructions (no size suffix).
|
||||||
switch upper {
|
switch upper {
|
||||||
case "RET", "NOP", "CALL", "JMP":
|
case "RET", "NOP", "CALL", "JMP",
|
||||||
|
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := noOperandTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := condCode(upper); ok {
|
if _, ok := condCode(upper); ok {
|
||||||
@@ -31,7 +35,7 @@ func Encodable(mnemonic string) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
base == "KMOVW" || base == "KMOVQ" {
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -53,13 +57,28 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Legacy SSE shuffles and packed binaries dispatch on the full name.
|
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
|
||||||
|
// do the imm8-controlled instructions, the lane extracts and inserts and
|
||||||
|
// the packed integer shifts (their trailing width letters belong to the
|
||||||
|
// mnemonic).
|
||||||
if _, ok := sseShufTable[upper]; ok {
|
if _, ok := sseShufTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := sseBinTable[upper]; ok {
|
if _, ok := sseBinTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
if _, ok := sseImm3Table[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseExtractTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseInsertTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseShiftImm[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
// The size-suffix split: retry the tables and the scalar switch on the
|
// The size-suffix split: retry the tables and the scalar switch on the
|
||||||
// base.
|
// base.
|
||||||
@@ -74,12 +93,15 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
switch base2 {
|
switch base2 {
|
||||||
case "MOV",
|
case "MOV", "MOVD",
|
||||||
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
|
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
|
||||||
"TEST",
|
"TEST",
|
||||||
"LEA",
|
"LEA",
|
||||||
"INC", "DEC", "NEG", "NOT",
|
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
|
||||||
"SHL", "SHR", "SAR",
|
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
|
||||||
|
"BT", "BTS", "BTR", "BTC",
|
||||||
|
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
|
||||||
|
"MOVS", "STOS",
|
||||||
"IMUL", "IMUL3",
|
"IMUL", "IMUL3",
|
||||||
"PUSH", "POP",
|
"PUSH", "POP",
|
||||||
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
||||||
@@ -88,7 +110,9 @@ func Encodable(mnemonic string) bool {
|
|||||||
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||||
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
|
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
|
||||||
"CVTSL2SD", "CVTSQ2SD",
|
"CVTSL2SD", "CVTSQ2SD",
|
||||||
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
|
||||||
|
"FMOVD",
|
||||||
|
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
// Full-name dispatches the size split would eat (a trailing width
|
// Full-name dispatches the size split would eat (a trailing width
|
||||||
|
|||||||
+81
-6
@@ -58,6 +58,41 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if cc, ok := condCode(upper); ok {
|
if cc, ok := condCode(upper); ok {
|
||||||
return e.encodeJcc(cc, ops)
|
return e.encodeJcc(cc, ops)
|
||||||
}
|
}
|
||||||
|
// No-operand system and string-control instructions (CPUID, RDTSC,
|
||||||
|
// SYSCALL, the fences, UNDEF, …).
|
||||||
|
if op, ok := noOperandTable[upper]; ok {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
|
||||||
|
}
|
||||||
|
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
|
||||||
|
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
|
||||||
|
switch upper {
|
||||||
|
case "POPFQ":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
|
||||||
|
case "PUSHFQ":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
|
||||||
|
case "INT":
|
||||||
|
return e.encodeInt(ops)
|
||||||
|
case "LDMXCSR":
|
||||||
|
return e.encodeMxcsr(2, ops)
|
||||||
|
case "STMXCSR":
|
||||||
|
return e.encodeMxcsr(3, ops)
|
||||||
|
// CMPSD is the scalar double compare, whose predicate immediate comes
|
||||||
|
// LAST in Plan 9 order (src, dst, $imm).
|
||||||
|
case "CMPSD":
|
||||||
|
return e.encodeCmpsd(ops)
|
||||||
|
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
|
||||||
|
case "SHA256RNDS2":
|
||||||
|
return e.encodeSha256rnds2(ops)
|
||||||
|
}
|
||||||
|
|
||||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||||
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
||||||
@@ -67,7 +102,8 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return e.encodeVec(base, ops, sfx)
|
return e.encodeVec(base, ops, sfx)
|
||||||
}
|
}
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
@@ -101,6 +137,21 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if m, ok := sseBinTable[base]; ok {
|
if m, ok := sseBinTable[base]; ok {
|
||||||
return e.encodeSSEBin(m, ops)
|
return e.encodeSSEBin(m, ops)
|
||||||
}
|
}
|
||||||
|
// The imm8-controlled legacy instructions, the lane extracts and inserts
|
||||||
|
// and the packed integer shifts all dispatch on the full name: a trailing
|
||||||
|
// width letter here belongs to the mnemonic, not to the size split.
|
||||||
|
if m, ok := sseImm3Table[upper]; ok {
|
||||||
|
return e.encodeSSEImm3(m, ops)
|
||||||
|
}
|
||||||
|
if m, ok := sseExtractTable[upper]; ok {
|
||||||
|
return e.encodeSSEExtract(m, ops)
|
||||||
|
}
|
||||||
|
if m, ok := sseInsertTable[upper]; ok {
|
||||||
|
return e.encodeSSEInsert(m, ops)
|
||||||
|
}
|
||||||
|
if _, ok := sseShiftImm[upper]; ok {
|
||||||
|
return e.encodeSSEShift(upper, ops)
|
||||||
|
}
|
||||||
// PMOVMSKB ends in a width letter the size split would eat, so it
|
// PMOVMSKB ends in a width letter the size split would eat, so it
|
||||||
// dispatches on the full name like the packed binaries above.
|
// dispatches on the full name like the packed binaries above.
|
||||||
if upper == "PMOVMSKB" {
|
if upper == "PMOVMSKB" {
|
||||||
@@ -109,16 +160,36 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
switch base {
|
switch base {
|
||||||
case "MOV":
|
case "MOV":
|
||||||
return e.encodeMov(ops, size)
|
return e.encodeMov(ops, size)
|
||||||
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
|
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
|
||||||
|
// REX.W and all.
|
||||||
|
case "MOVD":
|
||||||
|
return e.encodeMov(ops, 8)
|
||||||
|
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
|
||||||
return e.encodeALU(aluOp[base], ops, size)
|
return e.encodeALU(aluOp[base], ops, size)
|
||||||
case "TEST":
|
case "TEST":
|
||||||
return e.encodeTest(ops, size)
|
return e.encodeTest(ops, size)
|
||||||
case "LEA":
|
case "LEA":
|
||||||
return e.encodeLea(ops, size)
|
return e.encodeLea(ops, size)
|
||||||
case "INC", "DEC", "NEG", "NOT":
|
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
|
||||||
return e.encodeUnary(unaryOp[base], ops, size)
|
return e.encodeUnary(unaryOp[base], ops, size)
|
||||||
case "SHL", "SHR", "SAR":
|
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
|
||||||
return e.encodeShift(shiftOp[base], ops, size)
|
return e.encodeShift(shiftOp[base], ops, size)
|
||||||
|
case "BT", "BTS", "BTR", "BTC":
|
||||||
|
return e.encodeBitTest(base, ops, size)
|
||||||
|
case "XCHG":
|
||||||
|
return e.encodeExchange(ops, size)
|
||||||
|
case "CMPXCHG":
|
||||||
|
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
|
||||||
|
case "XADD":
|
||||||
|
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
|
||||||
|
case "CRC32":
|
||||||
|
return e.encodeCrc32(ops, size)
|
||||||
|
case "ADCX":
|
||||||
|
return e.encodeCarryExt(0x66, ops, size)
|
||||||
|
case "ADOX":
|
||||||
|
return e.encodeCarryExt(0xF3, ops, size)
|
||||||
|
case "MOVS", "STOS":
|
||||||
|
return e.encodeStringOp(base, ops, size)
|
||||||
case "IMUL", "IMUL3":
|
case "IMUL", "IMUL3":
|
||||||
return e.encodeImul(ops, size)
|
return e.encodeImul(ops, size)
|
||||||
case "PUSH":
|
case "PUSH":
|
||||||
@@ -136,7 +207,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
return e.encodeMovExtend(base, ops)
|
return e.encodeMovExtend(base, ops)
|
||||||
case "CVTSL2SD", "CVTSQ2SD":
|
case "CVTSL2SD", "CVTSQ2SD":
|
||||||
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
||||||
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
|
||||||
|
return e.encodeCvtInt(base, ops, size)
|
||||||
|
case "FMOVD":
|
||||||
|
return e.encodeFmov(ops)
|
||||||
|
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||||
@@ -195,7 +270,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
|||||||
if ss, ok := scatterTable[upper]; ok {
|
if ss, ok := scatterTable[upper]; ok {
|
||||||
return e.encodeScatter(upper, ss, ops, sfx)
|
return e.encodeScatter(upper, ss, ops, sfx)
|
||||||
}
|
}
|
||||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -546,6 +546,248 @@ func TestEncodableCmovSize(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestCarryShiftMulGroundTruth pins the carry-flag ALU family (ADC/SBB with
|
||||||
|
// their accumulator immediate forms), the rotate family, MUL/DIV/IDIV and the
|
||||||
|
// bit-test family byte for byte against go tool asm (see
|
||||||
|
// testdata/verify/scalar_amd64.s).
|
||||||
|
func TestCarryShiftMulGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"ADCQ AX,BX", "ADCQ", []Operand{AX, BX}, "4811c3"},
|
||||||
|
{"ADCL AX,BX", "ADCL", []Operand{AX, BX}, "11c3"},
|
||||||
|
{"ADCB AL,BL", "ADCB", []Operand{AL, BL}, "10c3"},
|
||||||
|
{"ADCW AX,BX", "ADCW", []Operand{AX, BX}, "6611c3"},
|
||||||
|
{"SBBQ AX,BX", "SBBQ", []Operand{AX, BX}, "4819c3"},
|
||||||
|
{"ADCQ $5,BX", "ADCQ", []Operand{Imm(5), BX}, "4883d305"},
|
||||||
|
{"ADCQ $300,BX", "ADCQ", []Operand{Imm(300), BX}, "4881d32c010000"},
|
||||||
|
{"ADCQ $300,AX", "ADCQ", []Operand{Imm(300), AX}, "48152c010000"},
|
||||||
|
{"ADCB $5,AL", "ADCB", []Operand{Imm(5), AL}, "1405"},
|
||||||
|
{"SBBQ $300,AX", "SBBQ", []Operand{Imm(300), AX}, "481d2c010000"},
|
||||||
|
{"ADCQ AX,(BX)", "ADCQ", []Operand{AX, Ptr(BX, 0, 8)}, "481103"},
|
||||||
|
{"ROLQ $3,AX", "ROLQ", []Operand{Imm(3), AX}, "48c1c003"},
|
||||||
|
{"ROLL CX,BX", "ROLL", []Operand{CL, BX}, "d3c3"},
|
||||||
|
{"RORQ CL,AX", "RORQ", []Operand{CL, AX}, "48d3c8"},
|
||||||
|
{"RCRQ $1,BX", "RCRQ", []Operand{Imm(1), BX}, "48d1db"},
|
||||||
|
{"RCLQ $3,AX", "RCLQ", []Operand{Imm(3), AX}, "48c1d003"},
|
||||||
|
{"RORB CL,BL", "RORB", []Operand{CL, BL}, "d2cb"},
|
||||||
|
{"SALQ $2,AX", "SALQ", []Operand{Imm(2), AX}, "48c1e002"},
|
||||||
|
{"ROLW $1,AX", "ROLW", []Operand{Imm(1), AX}, "66d1c0"},
|
||||||
|
{"MULQ CX", "MULQ", []Operand{CX}, "48f7e1"},
|
||||||
|
{"MULL CX", "MULL", []Operand{CX}, "f7e1"},
|
||||||
|
{"MULB CL", "MULB", []Operand{CL}, "f6e1"},
|
||||||
|
{"DIVL CX", "DIVL", []Operand{CX}, "f7f1"},
|
||||||
|
{"IDIVQ CX", "IDIVQ", []Operand{CX}, "48f7f9"},
|
||||||
|
{"MULW CX", "MULW", []Operand{CX}, "66f7e1"},
|
||||||
|
{"BTQ AX,DX", "BTQ", []Operand{AX, DX}, "480fa3c2"},
|
||||||
|
{"BTL AX,DX", "BTL", []Operand{AX, DX}, "0fa3c2"},
|
||||||
|
{"BTW AX,DX", "BTW", []Operand{AX, DX}, "660fa3c2"},
|
||||||
|
{"BTQ $3,BX", "BTQ", []Operand{Imm(3), BX}, "480fbae303"},
|
||||||
|
{"BTQ $3,(AX)", "BTQ", []Operand{Imm(3), Ptr(AX, 0, 8)}, "480fba2003"},
|
||||||
|
{"BTSQ $5,BX", "BTSQ", []Operand{Imm(5), BX}, "480fbaeb05"},
|
||||||
|
{"BTCQ AX,BX", "BTCQ", []Operand{AX, BX}, "480fbbc3"},
|
||||||
|
{"BTRQ $7,BX", "BTRQ", []Operand{Imm(7), BX}, "480fbaf307"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The bit-test immediate is an unsigned bit index with the negative
|
||||||
|
// spelling accepted, the shuffle convention: BTQ $300 must be rejected.
|
||||||
|
if _, err := Encode("BTQ", Imm(300), AX); err == nil {
|
||||||
|
t.Errorf("BTQ $300: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAtomicSystemGroundTruth pins the exchange/compare-exchange/accumulate
|
||||||
|
// family, the string primitives, the flag and system instructions, the MXCSR
|
||||||
|
// pair, the scalar float-to-int conversions and the x87 FMOVD byte for byte
|
||||||
|
// against go tool asm (see testdata/verify/atomics_amd64.s and
|
||||||
|
// testdata/verify/system_amd64.s).
|
||||||
|
func TestAtomicSystemGroundTruth(t *testing.T) {
|
||||||
|
r8 := Reg{idx: 8, size: 8}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"XCHGQ AX,BX", "XCHGQ", []Operand{AX, BX}, "4893"},
|
||||||
|
{"XCHGQ BX,AX", "XCHGQ", []Operand{BX, AX}, "4893"},
|
||||||
|
{"XCHGL AX,BX", "XCHGL", []Operand{AX, BX}, "93"},
|
||||||
|
{"XCHGB AL,BL", "XCHGB", []Operand{AL, BL}, "86c3"},
|
||||||
|
{"XCHGW AX,BX", "XCHGW", []Operand{AX, BX}, "6693"},
|
||||||
|
{"XCHGQ R8,R9", "XCHGQ", []Operand{r8, Reg{idx: 9, size: 8}}, "4d87c1"},
|
||||||
|
{"XCHGQ BX,(AX)", "XCHGQ", []Operand{BX, Ptr(AX, 0, 8)}, "488718"},
|
||||||
|
{"XCHGQ (AX),BX", "XCHGQ", []Operand{Ptr(AX, 0, 8), BX}, "488718"},
|
||||||
|
{"XCHGQ AX,(BX)", "XCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "488703"},
|
||||||
|
{"CMPXCHGL AX,BX", "CMPXCHGL", []Operand{AX, BX}, "0fb1c3"},
|
||||||
|
{"CMPXCHGQ AX,(BX)", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
|
||||||
|
{"CMPXCHGB AL,(BX)", "CMPXCHGB", []Operand{AL, Ptr(BX, 0, 1)}, "0fb003"},
|
||||||
|
{"CMPXCHGW AX,BX", "CMPXCHGW", []Operand{AX, BX}, "660fb1c3"},
|
||||||
|
{"XADDL AX,BX", "XADDL", []Operand{AX, BX}, "0fc1c3"},
|
||||||
|
{"XADDQ AX,(BX)", "XADDQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fc103"},
|
||||||
|
{"XADDB AL,(BX)", "XADDB", []Operand{AL, Ptr(BX, 0, 1)}, "0fc003"},
|
||||||
|
{"XADDW AX,BX", "XADDW", []Operand{AX, BX}, "660fc1c3"},
|
||||||
|
{"ADCXL AX,CX", "ADCXL", []Operand{AX, CX}, "660f38f6c8"},
|
||||||
|
{"ADCXQ AX,CX", "ADCXQ", []Operand{AX, CX}, "66480f38f6c8"},
|
||||||
|
{"ADOXL AX,CX", "ADOXL", []Operand{AX, CX}, "f30f38f6c8"},
|
||||||
|
{"ADOXQ AX,CX", "ADOXQ", []Operand{AX, CX}, "f3480f38f6c8"},
|
||||||
|
{"CRC32B AX,CX", "CRC32B", []Operand{AX, CX}, "f20f38f0c8"},
|
||||||
|
{"CRC32W AX,CX", "CRC32W", []Operand{AX, CX}, "66f20f38f1c8"},
|
||||||
|
{"CRC32L AX,CX", "CRC32L", []Operand{AX, CX}, "f20f38f1c8"},
|
||||||
|
{"CRC32Q AX,CX", "CRC32Q", []Operand{AX, CX}, "f2480f38f1c8"},
|
||||||
|
{"CRC32L (AX),CX", "CRC32L", []Operand{Ptr(AX, 0, 4), CX}, "f20f38f108"},
|
||||||
|
{"MOVSQ", "MOVSQ", []Operand{}, "48a5"},
|
||||||
|
{"MOVSL", "MOVSL", []Operand{}, "a5"},
|
||||||
|
{"MOVSB", "MOVSB", []Operand{}, "a4"},
|
||||||
|
{"MOVSW", "MOVSW", []Operand{}, "66a5"},
|
||||||
|
{"STOSB", "STOSB", []Operand{}, "aa"},
|
||||||
|
{"STOSQ", "STOSQ", []Operand{}, "48ab"},
|
||||||
|
{"STOSL", "STOSL", []Operand{}, "ab"},
|
||||||
|
{"STOSW", "STOSW", []Operand{}, "66ab"},
|
||||||
|
{"CLD", "CLD", []Operand{}, "fc"},
|
||||||
|
{"STD", "STD", []Operand{}, "fd"},
|
||||||
|
{"POPFQ", "POPFQ", []Operand{}, "9d"},
|
||||||
|
{"PUSHFQ", "PUSHFQ", []Operand{}, "9c"},
|
||||||
|
{"CPUID", "CPUID", []Operand{}, "0fa2"},
|
||||||
|
{"RDTSC", "RDTSC", []Operand{}, "0f31"},
|
||||||
|
{"RDTSCP", "RDTSCP", []Operand{}, "0f01f9"},
|
||||||
|
{"SYSCALL", "SYSCALL", []Operand{}, "0f05"},
|
||||||
|
{"XGETBV", "XGETBV", []Operand{}, "0f01d0"},
|
||||||
|
{"PAUSE", "PAUSE", []Operand{}, "f390"},
|
||||||
|
{"LFENCE", "LFENCE", []Operand{}, "0faee8"},
|
||||||
|
{"MFENCE", "MFENCE", []Operand{}, "0faef0"},
|
||||||
|
{"SFENCE", "SFENCE", []Operand{}, "0faef8"},
|
||||||
|
{"UNDEF", "UNDEF", []Operand{}, "0f0b"},
|
||||||
|
{"INT $3", "INT", []Operand{Imm(3)}, "cd03"},
|
||||||
|
{"LDMXCSR (AX)", "LDMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae10"},
|
||||||
|
{"STMXCSR (AX)", "STMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae18"},
|
||||||
|
{"CVTSD2SL X0,AX", "CVTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2dc0"},
|
||||||
|
{"CVTTSD2SQ X0,AX", "CVTTSD2SQ", []Operand{vreg(t, "X0"), AX}, "f2480f2cc0"},
|
||||||
|
{"CVTTSD2SL X0,AX", "CVTTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2cc0"},
|
||||||
|
{"CVTSS2SQ X0,AX", "CVTSS2SQ", []Operand{vreg(t, "X0"), AX}, "f3480f2dc0"},
|
||||||
|
{"FMOVD (AX),F0", "FMOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "F0")}, "dd00"},
|
||||||
|
{"FMOVD F0,(AX)", "FMOVD", []Operand{vreg(t, "F0"), Ptr(AX, 0, 8)}, "dd10"},
|
||||||
|
{"FMOVD F0,F1", "FMOVD", []Operand{vreg(t, "F0"), vreg(t, "F1")}, "ddd1"},
|
||||||
|
{"MOVD AX,X0", "MOVD", []Operand{AX, vreg(t, "X0")}, "66480f6ec0"},
|
||||||
|
{"MOVD X0,AX", "MOVD", []Operand{vreg(t, "X0"), AX}, "66480f7ec0"},
|
||||||
|
{"MOVD X0,X1", "MOVD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f30f7ec8"},
|
||||||
|
{"MOVD (AX),X0", "MOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "X0")}, "f30f7e00"},
|
||||||
|
{"MOVD X0,(AX)", "MOVD", []Operand{vreg(t, "X0"), Ptr(AX, 0, 8)}, "660fd600"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// LDMXCSR/STMXCSR take a memory operand only.
|
||||||
|
if _, err := Encode("LDMXCSR", AX); err == nil {
|
||||||
|
t.Errorf("LDMXCSR AX: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSSEGapsGroundTruth pins the legacy SSE gap families: the scalar
|
||||||
|
// compare and square root, the Plan 9 packed spellings, the imm8-controlled
|
||||||
|
// shuffles, the lane extracts and inserts, the packed integer shifts and the
|
||||||
|
// AES/SHA round instructions, byte for byte against go tool asm (see
|
||||||
|
// testdata/verify/crypto_amd64.s and testdata/verify/sse_amd64.s).
|
||||||
|
func TestSSEGapsGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"ANDNPD X0,X1", "ANDNPD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f55c8"},
|
||||||
|
{"ANDNPS X0,X1", "ANDNPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f55c8"},
|
||||||
|
{"COMISD X0,X1", "COMISD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f2fc8"},
|
||||||
|
{"SQRTSD X0,X1", "SQRTSD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f20f51c8"},
|
||||||
|
{"PSHUFL $3,X0,X1", "PSHUFL", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f70c803"},
|
||||||
|
{"PALIGNR $2,X0,X1", "PALIGNR", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0fc802"},
|
||||||
|
{"PBLENDW $3,X0,X1", "PBLENDW", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0ec803"},
|
||||||
|
{"PCMPESTRI $1,X0,X1", "PCMPESTRI", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}, "660f3a61c801"},
|
||||||
|
{"PCLMULQDQ $0,X0,X1", "PCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3a44c800"},
|
||||||
|
{"PCLMULQDQ $0,(AX),X1", "PCLMULQDQ", []Operand{Imm(0), Ptr(AX, 0, 16), vreg(t, "X1")}, "660f3a440800"},
|
||||||
|
{"PEXTRB $1,X0,AX", "PEXTRB", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a14c001"},
|
||||||
|
{"PEXTRD $1,X0,AX", "PEXTRD", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a16c001"},
|
||||||
|
{"PEXTRQ $1,X0,AX", "PEXTRQ", []Operand{Imm(1), vreg(t, "X0"), AX}, "66480f3a16c001"},
|
||||||
|
{"PEXTRW $1,X0,AX", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), AX}, "660fc5c001"},
|
||||||
|
{"PEXTRW $1,X0,(AX)", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), Ptr(AX, 0, 2)}, "660f3a150001"},
|
||||||
|
{"PINSRB $1,AX,X0", "PINSRB", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a20c001"},
|
||||||
|
{"PINSRD $1,AX,X0", "PINSRD", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a22c001"},
|
||||||
|
{"PINSRQ $1,AX,X0", "PINSRQ", []Operand{Imm(1), AX, vreg(t, "X0")}, "66480f3a22c001"},
|
||||||
|
{"PINSRW $1,AX,X0", "PINSRW", []Operand{Imm(1), AX, vreg(t, "X0")}, "660fc4c001"},
|
||||||
|
{"PINSRW $1,(AX),X0", "PINSRW", []Operand{Imm(1), Ptr(AX, 0, 2), vreg(t, "X0")}, "660fc40001"},
|
||||||
|
{"PSLLL $2,X0", "PSLLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72f002"},
|
||||||
|
{"PSRAL $2,X0", "PSRAL", []Operand{Imm(2), vreg(t, "X0")}, "660f72e002"},
|
||||||
|
{"PSRLL $2,X0", "PSRLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72d002"},
|
||||||
|
{"PSRLQ $2,X0", "PSRLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d002"},
|
||||||
|
{"PSLLQ $2,X0", "PSLLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f002"},
|
||||||
|
{"PSLLW $2,X0", "PSLLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71f002"},
|
||||||
|
{"PSRLW $2,X0", "PSRLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71d002"},
|
||||||
|
{"PSRAW $2,X0", "PSRAW", []Operand{Imm(2), vreg(t, "X0")}, "660f71e002"},
|
||||||
|
{"PSLLDQ $2,X0", "PSLLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f802"},
|
||||||
|
{"PSRLDQ $2,X0", "PSRLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d802"},
|
||||||
|
{"PSLLL X0,X1", "PSLLL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ff2c8"},
|
||||||
|
{"PSRLQ X0,X1", "PSRLQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660fd3c8"},
|
||||||
|
{"PSLLL (AX),X1", "PSLLL", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660ff208"},
|
||||||
|
{"PSUBL X0,X1", "PSUBL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffac8"},
|
||||||
|
{"PADDL X0,X1", "PADDL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffec8"},
|
||||||
|
{"PCMPEQL X0,X1", "PCMPEQL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f76c8"},
|
||||||
|
{"PUNPCKLBW X0,X1", "PUNPCKLBW", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f60c8"},
|
||||||
|
{"MOVOA X0,X1", "MOVOA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f6fc8"},
|
||||||
|
{"MOVOA (AX),X1", "MOVOA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f6f08"},
|
||||||
|
{"MOVOA X0,(AX)", "MOVOA", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "660f7f00"},
|
||||||
|
{"AESIMC X0,X1", "AESIMC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dbc8"},
|
||||||
|
{"AESIMC (AX),X1", "AESIMC", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f38db08"},
|
||||||
|
{"AESENC X0,X1", "AESENC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dcc8"},
|
||||||
|
{"AESENCLAST X0,X1", "AESENCLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38ddc8"},
|
||||||
|
{"AESDEC X0,X1", "AESDEC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dec8"},
|
||||||
|
{"AESDECLAST X0,X1", "AESDECLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dfc8"},
|
||||||
|
{"AESKEYGENASSIST $0,X0,X1", "AESKEYGENASSIST", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3adfc800"},
|
||||||
|
{"SHA1MSG1 X0,X1", "SHA1MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c9c8"},
|
||||||
|
{"SHA1MSG2 X0,X1", "SHA1MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cac8"},
|
||||||
|
{"SHA1NEXTE X0,X1", "SHA1NEXTE", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c8c8"},
|
||||||
|
{"SHA1RNDS4 $0,X0,X1", "SHA1RNDS4", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "0f3accc800"},
|
||||||
|
{"SHA256MSG1 X0,X1", "SHA256MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38ccc8"},
|
||||||
|
{"SHA256MSG2 X0,X1", "SHA256MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cdc8"},
|
||||||
|
{"SHA256RNDS2 X0,X1,X2", "SHA256RNDS2", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "0f38cbd1"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// SHA256RNDS2's first operand must be the literal X0.
|
||||||
|
if _, err := Encode("SHA256RNDS2", vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")); err == nil {
|
||||||
|
t.Errorf("SHA256RNDS2 X1,...: expected an error, got none")
|
||||||
|
}
|
||||||
|
// PSLLDQ has no variable-count form.
|
||||||
|
if _, err := Encode("PSLLDQ", vreg(t, "X0"), vreg(t, "X1")); err == nil {
|
||||||
|
t.Errorf("PSLLDQ X0,X1: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
||||||
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
||||||
func TestSSEBinGroundTruth(t *testing.T) {
|
func TestSSEBinGroundTruth(t *testing.T) {
|
||||||
|
|||||||
+31
-14
@@ -182,8 +182,17 @@ var evexTable = map[string]evexSpec{
|
|||||||
// EVEX.66.0F38, permutes (NDS form).
|
// EVEX.66.0F38, permutes (NDS form).
|
||||||
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2B": {2, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F38, population count (reg=dst, rm=src; W selects byte/word
|
||||||
|
// against dword/qword).
|
||||||
|
"VPOPCNTB": {2, 0x54, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPOPCNTD": {2, 0x55, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPOPCNTQ": {2, 0x55, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F.W1, the qword spelling of the packed OR (VPORQ has no VEX
|
||||||
|
// form in the Go assembler: it always encodes through EVEX).
|
||||||
|
"VPORQ": {1, 0xEB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
@@ -1435,18 +1444,20 @@ var evexKOperand = map[string]bool{
|
|||||||
}
|
}
|
||||||
|
|
||||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||||
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
// direction, kk (k → k), kmem (k → mem), gprk (GPR/mem → k) and kgpr
|
||||||
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
|
// (k → GPR). Each direction group carries its own mandatory prefix and W:
|
||||||
// prefix and W for the wider widths.
|
// the k-destination/source forms share one pair, the GPR forms another.
|
||||||
type kmovSpec struct {
|
type kmovSpec struct {
|
||||||
kk, kmem, gprk, kgpr byte
|
kk, kmem, gprk, kgpr byte
|
||||||
gprPP int
|
kPP, kW int // prefix and VEX.W for the k forms
|
||||||
w int
|
gprPP, gprW int // prefix and VEX.W for the GPR forms
|
||||||
}
|
}
|
||||||
|
|
||||||
var kmovTable = map[string]kmovSpec{
|
var kmovTable = map[string]kmovSpec{
|
||||||
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
|
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0, 0, 0},
|
||||||
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
|
"KMOVB": {0x90, 0x91, 0x92, 0x93, 1, 0, 1, 0},
|
||||||
|
"KMOVD": {0x90, 0x91, 0x92, 0x93, 1, 1, 3, 0},
|
||||||
|
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 0, 1, 3, 1},
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
|
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
|
||||||
@@ -1460,14 +1471,14 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
|
|||||||
dstReg, dstIsReg := dst.(Reg)
|
dstReg, dstIsReg := dst.(Reg)
|
||||||
srcK := srcIsReg && srcReg.mask
|
srcK := srcIsReg && srcReg.mask
|
||||||
dstK := dstIsReg && dstReg.mask
|
dstK := dstIsReg && dstReg.mask
|
||||||
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
|
|
||||||
switch {
|
switch {
|
||||||
case srcK && dstK:
|
case srcK && dstK:
|
||||||
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
|
// k ← k: reg = dst, rm = src.
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: ks.kk, w: ks.kW, pp: ks.kPP, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||||||
case srcK && dstIsReg:
|
case srcK && dstIsReg:
|
||||||
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
|
// GPR ← k: reg = dst, rm = src.
|
||||||
spec.pp = ks.gprPP
|
spec := vexSpec{mapSel: 1, opcode: ks.kgpr, w: ks.gprW, pp: ks.gprPP, opdigit: -1}
|
||||||
rBit := 0
|
rBit := 0
|
||||||
if dstReg.idx >= 8 {
|
if dstReg.idx >= 8 {
|
||||||
rBit = 1
|
rBit = 1
|
||||||
@@ -1477,11 +1488,17 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
|
|||||||
if _, ok := dst.(Mem); !ok {
|
if _, ok := dst.(Mem); !ok {
|
||||||
return fmt.Errorf("%s: invalid destination operand", upper)
|
return fmt.Errorf("%s: invalid destination operand", upper)
|
||||||
}
|
}
|
||||||
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
|
// mem ← k: reg = src, rm = dst.
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: ks.kmem, w: ks.kW, pp: ks.kPP, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
||||||
case dstK:
|
case dstK:
|
||||||
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
|
// k ← GPR: reg = dst, rm = src. A memory source shares the k ← k
|
||||||
spec.pp = ks.gprPP
|
// opcode and prefix group (the ykmovb layout the Go assembler uses).
|
||||||
|
opcode, w, pp := ks.gprk, ks.gprW, ks.gprPP
|
||||||
|
if memOperand(src) {
|
||||||
|
opcode, w, pp = ks.kk, ks.kW, ks.kPP
|
||||||
|
}
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: opcode, w: w, pp: pp, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("%s requires a K register operand", upper)
|
return fmt.Errorf("%s requires a K register operand", upper)
|
||||||
|
|||||||
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
||||||
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
||||||
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
||||||
|
// The qword OR spelling always encodes through EVEX.
|
||||||
|
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
|
||||||
|
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
|
||||||
|
// Byte permute and population count.
|
||||||
|
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
|
||||||
|
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
|
||||||
|
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
|
||||||
|
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
|
||||||
|
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
|
||||||
// Align (NDS + imm8).
|
// Align (NDS + imm8).
|
||||||
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
||||||
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
||||||
@@ -52,6 +61,16 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
||||||
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
||||||
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
||||||
|
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
|
||||||
|
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
|
||||||
|
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
|
||||||
|
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
|
||||||
|
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
|
||||||
|
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
|
||||||
|
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
|
||||||
|
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
|
||||||
|
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
|
||||||
|
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
|
||||||
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
||||||
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
||||||
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
||||||
|
|||||||
+626
-5
@@ -14,15 +14,18 @@ var aluOp = map[string]struct {
|
|||||||
}{
|
}{
|
||||||
"ADD": {0x01, 0},
|
"ADD": {0x01, 0},
|
||||||
"OR": {0x09, 1},
|
"OR": {0x09, 1},
|
||||||
|
"ADC": {0x11, 2},
|
||||||
|
"SBB": {0x19, 3},
|
||||||
"AND": {0x21, 4},
|
"AND": {0x21, 4},
|
||||||
"SUB": {0x29, 5},
|
"SUB": {0x29, 5},
|
||||||
"XOR": {0x31, 6},
|
"XOR": {0x31, 6},
|
||||||
"CMP": {0x39, 7},
|
"CMP": {0x39, 7},
|
||||||
}
|
}
|
||||||
|
|
||||||
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
|
// unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode.
|
||||||
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
|
// INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes
|
||||||
// mode); NEG/NOT use the 0xF6/0xF7 group.
|
// in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4,
|
||||||
|
// DIV /6, IDIV /7; the accumulator is the implicit other operand).
|
||||||
var unaryOp = map[string]struct {
|
var unaryOp = map[string]struct {
|
||||||
digit int
|
digit int
|
||||||
op byte
|
op byte
|
||||||
@@ -31,13 +34,50 @@ var unaryOp = map[string]struct {
|
|||||||
"DEC": {1, 0xFF},
|
"DEC": {1, 0xFF},
|
||||||
"NOT": {2, 0xF7},
|
"NOT": {2, 0xF7},
|
||||||
"NEG": {3, 0xF7},
|
"NEG": {3, 0xF7},
|
||||||
|
"MUL": {4, 0xF7},
|
||||||
|
"DIV": {6, 0xF7},
|
||||||
|
"IDIV": {7, 0xF7},
|
||||||
}
|
}
|
||||||
|
|
||||||
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
|
// shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the
|
||||||
|
// 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4).
|
||||||
var shiftOp = map[string]int{
|
var shiftOp = map[string]int{
|
||||||
"SHL": 4,
|
"SHL": 4,
|
||||||
|
"SAL": 4,
|
||||||
"SHR": 5,
|
"SHR": 5,
|
||||||
"SAR": 7,
|
"SAR": 7,
|
||||||
|
"ROL": 0,
|
||||||
|
"ROR": 1,
|
||||||
|
"RCL": 2,
|
||||||
|
"RCR": 3,
|
||||||
|
}
|
||||||
|
|
||||||
|
// bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form;
|
||||||
|
// the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's
|
||||||
|
// opcode row.
|
||||||
|
var bitTestOp = map[string]int{
|
||||||
|
"BT": 4,
|
||||||
|
"BTS": 5,
|
||||||
|
"BTR": 6,
|
||||||
|
"BTC": 7,
|
||||||
|
}
|
||||||
|
|
||||||
|
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
|
||||||
|
// fence names carry their opcode inside the 0F AE /digit group spelled out in
|
||||||
|
// full (E8/F0/F8), and PAUSE is F3 90.
|
||||||
|
var noOperandTable = map[string][]byte{
|
||||||
|
"CPUID": {0x0F, 0xA2},
|
||||||
|
"RDTSC": {0x0F, 0x31},
|
||||||
|
"RDTSCP": {0x0F, 0x01, 0xF9},
|
||||||
|
"SYSCALL": {0x0F, 0x05},
|
||||||
|
"XGETBV": {0x0F, 0x01, 0xD0},
|
||||||
|
"CLD": {0xFC},
|
||||||
|
"STD": {0xFD},
|
||||||
|
"PAUSE": {0xF3, 0x90},
|
||||||
|
"LFENCE": {0x0F, 0xAE, 0xE8},
|
||||||
|
"MFENCE": {0x0F, 0xAE, 0xF0},
|
||||||
|
"SFENCE": {0x0F, 0xAE, 0xF8},
|
||||||
|
"UNDEF": {0x0F, 0x0B},
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- MOV --------------------------------------------------------------------
|
// --- MOV --------------------------------------------------------------------
|
||||||
@@ -309,6 +349,13 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
|
||||||
|
// the destination is AL, the form the Go assembler prefers here.
|
||||||
|
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||||
|
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
|
||||||
|
i.imm = immBytes
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
i := newInstr(1, []byte{0x80})
|
i := newInstr(1, []byte{0x80})
|
||||||
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -913,6 +960,7 @@ type sseMove struct {
|
|||||||
var sseMoveTable = map[string]sseMove{
|
var sseMoveTable = map[string]sseMove{
|
||||||
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
|
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
|
||||||
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
|
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
|
||||||
|
"MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias
|
||||||
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
||||||
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
||||||
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
||||||
@@ -1000,10 +1048,120 @@ var sseBinTable = map[string]sseBin{
|
|||||||
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
||||||
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
||||||
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
||||||
"PCMPEQD": {0x66, 0x76, false},
|
"PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false},
|
||||||
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
||||||
"PCMPGTD": {0x66, 0x66, false},
|
"PCMPGTD": {0x66, 0x66, false},
|
||||||
"PSHUFB": {0x66, 0x00, true},
|
"PSHUFB": {0x66, 0x00, true},
|
||||||
|
// Scalar compares and square root, packed adds/subtracts and the byte
|
||||||
|
// unpack, the spellings the Plan 9 table uses (COMISD orders the
|
||||||
|
// operands like every other two-operand form).
|
||||||
|
"ANDNPD": {0x66, 0x55, false},
|
||||||
|
"ANDNPS": {0x00, 0x55, false},
|
||||||
|
"COMISD": {0x66, 0x2F, false},
|
||||||
|
"SQRTSD": {0xF2, 0x51, false},
|
||||||
|
"PADDL": {0x66, 0xFE, false},
|
||||||
|
"PSUBL": {0x66, 0xFA, false},
|
||||||
|
"PUNPCKLBW": {0x66, 0x60, false},
|
||||||
|
// AES round functions (66 0F38) and the SHA message schedule helpers
|
||||||
|
// (no prefix, 0F38).
|
||||||
|
"AESENC": {0x66, 0xDC, true},
|
||||||
|
"AESENCLAST": {0x66, 0xDD, true},
|
||||||
|
"AESDEC": {0x66, 0xDE, true},
|
||||||
|
"AESDECLAST": {0x66, 0xDF, true},
|
||||||
|
"AESIMC": {0x66, 0xDB, true},
|
||||||
|
"SHA1MSG1": {0x00, 0xC9, true},
|
||||||
|
"SHA1MSG2": {0x00, 0xCA, true},
|
||||||
|
"SHA1NEXTE": {0x00, 0xC8, true},
|
||||||
|
"SHA256MSG1": {0x00, 0xCC, true},
|
||||||
|
"SHA256MSG2": {0x00, 0xCD, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseImm3 describes a legacy SSE instruction taking a leading imm8 and two
|
||||||
|
// further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and
|
||||||
|
// map3A select the opcode map the same way as sseBin's.
|
||||||
|
type sseImm3 struct {
|
||||||
|
prefix byte
|
||||||
|
op byte
|
||||||
|
map3A bool // opcode lives under 0F3A instead of 0F38
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3
|
||||||
|
// align/blend shuffles, the string compare, carry-less multiply and the AES
|
||||||
|
// key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings.
|
||||||
|
var sseImm3Table = map[string]sseImm3{
|
||||||
|
"PALIGNR": {0x66, 0x0F, true},
|
||||||
|
"PBLENDW": {0x66, 0x0E, true},
|
||||||
|
"PCMPESTRI": {0x66, 0x61, true},
|
||||||
|
"PCLMULQDQ": {0x66, 0x44, true},
|
||||||
|
"AESKEYGENASSIST": {0x66, 0xDF, true},
|
||||||
|
"SHA1RNDS4": {0x00, 0xCC, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM
|
||||||
|
// source and rm = the destination (GPR or memory). PEXTRW's GPR destination
|
||||||
|
// uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one,
|
||||||
|
// so it carries both opcodes.
|
||||||
|
type sseExtract struct {
|
||||||
|
op []byte
|
||||||
|
opMem []byte // used when the destination is memory; nil shares op
|
||||||
|
rexW bool // PEXTRQ's REX.W
|
||||||
|
}
|
||||||
|
|
||||||
|
var sseExtractTable = map[string]sseExtract{
|
||||||
|
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
|
||||||
|
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
|
||||||
|
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
|
||||||
|
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
|
||||||
|
// destination and rm = the source (GPR or memory).
|
||||||
|
type sseInsert struct {
|
||||||
|
op []byte
|
||||||
|
rexW bool // PINSRQ's REX.W
|
||||||
|
}
|
||||||
|
|
||||||
|
var sseInsertTable = map[string]sseInsert{
|
||||||
|
"PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false},
|
||||||
|
"PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false},
|
||||||
|
"PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true},
|
||||||
|
"PINSRW": {[]byte{0x0F, 0xC4}, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShiftImm maps the legacy packed integer shifts' immediate form:
|
||||||
|
// OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L
|
||||||
|
// (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ.
|
||||||
|
var sseShiftImm = map[string]sseShift{
|
||||||
|
"PSLLW": {0x71, 6},
|
||||||
|
"PSRLW": {0x71, 2},
|
||||||
|
"PSRAW": {0x71, 4},
|
||||||
|
"PSLLL": {0x72, 6},
|
||||||
|
"PSRLL": {0x72, 2},
|
||||||
|
"PSRAL": {0x72, 4},
|
||||||
|
"PSLLQ": {0x73, 6},
|
||||||
|
"PSRLQ": {0x73, 2},
|
||||||
|
"PSLLDQ": {0x73, 7},
|
||||||
|
"PSRLDQ": {0x73, 3},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShiftVar maps the variable-count forms (the count comes from an XMM
|
||||||
|
// register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no
|
||||||
|
// variable form.
|
||||||
|
var sseShiftVar = map[string]byte{
|
||||||
|
"PSLLW": 0xF1,
|
||||||
|
"PSRLW": 0xD1,
|
||||||
|
"PSRAW": 0xE1,
|
||||||
|
"PSLLL": 0xF2,
|
||||||
|
"PSRLL": 0xD2,
|
||||||
|
"PSRAL": 0xE2,
|
||||||
|
"PSLLQ": 0xF3,
|
||||||
|
"PSRLQ": 0xD3,
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShift is one /digit selector in the 0F 71/72/73 immediate group.
|
||||||
|
type sseShift struct {
|
||||||
|
op byte
|
||||||
|
digit int
|
||||||
}
|
}
|
||||||
|
|
||||||
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
||||||
@@ -1016,6 +1174,7 @@ type sseShuf struct {
|
|||||||
var sseShufTable = map[string]sseShuf{
|
var sseShufTable = map[string]sseShuf{
|
||||||
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
||||||
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
||||||
|
"PSHUFL": {0x66, 0x70},
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
||||||
@@ -1090,3 +1249,465 @@ func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
|
|||||||
}
|
}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- carry, bit test, exchange and accumulate -------------------------------
|
||||||
|
|
||||||
|
// encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9
|
||||||
|
// order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with
|
||||||
|
// reg = index, rm = target); an immediate index uses 0F BA /digit with imm8.
|
||||||
|
func (e *enc) encodeBitTest(name string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
digit := bitTestOp[name]
|
||||||
|
index, target := ops[0], ops[1]
|
||||||
|
if reg, ok := index.(Reg); ok {
|
||||||
|
// Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC),
|
||||||
|
// the /digit base plus eight per step.
|
||||||
|
i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3})
|
||||||
|
if err := setRM(i, reg, target, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
imm, ok := index.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s index must be a register or an immediate", name)
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, 0xBA})
|
||||||
|
if err := setRMDigit(i, digit, target, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeExchange encodes XCHG. A register-to-register exchange where either
|
||||||
|
// operand is AX uses the 0x90+r accumulator form (with REX.W for the quad
|
||||||
|
// form, as the Go assembler emits it); everything else uses 0x86/0x87 with
|
||||||
|
// the register operand in ModRM.reg, the memory (or second register) in r/m.
|
||||||
|
func (e *enc) encodeExchange(ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
srcReg, srcIsReg := src.(Reg)
|
||||||
|
dstReg, dstIsReg := dst.(Reg)
|
||||||
|
if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) {
|
||||||
|
// 0x90+r: r is the non-AX register, whichever side it sits on.
|
||||||
|
r := dstReg
|
||||||
|
if srcReg.idx == 0 {
|
||||||
|
r = dstReg
|
||||||
|
} else {
|
||||||
|
r = srcReg
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x90 + byte(r.idx&7)})
|
||||||
|
i.rexB = r.idx >= 8
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
op := byte(0x87)
|
||||||
|
if size == 1 {
|
||||||
|
op = 0x86
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case srcIsReg:
|
||||||
|
i := newInstr(size, []byte{op})
|
||||||
|
if err := setRM(i, srcReg, dst, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case dstIsReg:
|
||||||
|
i := newInstr(size, []byte{op})
|
||||||
|
if err := setRM(i, dstReg, src, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("XCHG: at least one operand must be a register")
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG
|
||||||
|
// (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the
|
||||||
|
// destination writable (register or memory).
|
||||||
|
func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
srcReg, ok := ops[0].(Reg)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s source must be a register", name)
|
||||||
|
}
|
||||||
|
opc := op
|
||||||
|
if size == 1 {
|
||||||
|
opc = op8
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, opc})
|
||||||
|
if err := setRM(i, srcReg, ops[1], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for
|
||||||
|
// the rest; the word form carries a 0x66 operand-size prefix (66 F2, the
|
||||||
|
// prefix order the Go assembler emits) and the quad form REX.W. reg = GPR
|
||||||
|
// accumulator, rm = the data source.
|
||||||
|
func (e *enc) encodeCrc32(ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("CRC32 destination must be a general register")
|
||||||
|
}
|
||||||
|
i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1}
|
||||||
|
if size > 1 {
|
||||||
|
i.opcode[2] = 0xF1
|
||||||
|
}
|
||||||
|
i.rexW = size == 8
|
||||||
|
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg =
|
||||||
|
// destination, rm = source, the carry/overflow flag as the carry-in.
|
||||||
|
func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("ADCX/ADOX destination must be a general register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8}
|
||||||
|
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- string primitives, flags and INT ----------------------------------------
|
||||||
|
|
||||||
|
// encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and
|
||||||
|
// STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or
|
||||||
|
// REX.W prefix.
|
||||||
|
func (e *enc) encodeStringOp(base string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s takes no operands, got %d", base, len(ops))
|
||||||
|
}
|
||||||
|
var op byte
|
||||||
|
switch base {
|
||||||
|
case "MOVS":
|
||||||
|
op = 0xA5
|
||||||
|
if size == 1 {
|
||||||
|
op = 0xA4
|
||||||
|
}
|
||||||
|
case "STOS":
|
||||||
|
op = 0xAB
|
||||||
|
if size == 1 {
|
||||||
|
op = 0xAA
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("unsupported string instruction %q", base)
|
||||||
|
}
|
||||||
|
return e.emit(newInstr(size, []byte{op}))
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeInt encodes INT with its single imm8 operand. The field takes the
|
||||||
|
// low byte silently inside the 32-bit span, matching the scalar convention
|
||||||
|
// (go tool asm encodes INT $256 as CD 00).
|
||||||
|
func (e *enc) encodeInt(ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("INT expects 1 operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("INT operand must be an immediate")
|
||||||
|
}
|
||||||
|
if imm < -(1<<31) || imm > (1<<32)-1 {
|
||||||
|
return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}})
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a
|
||||||
|
// single 32-bit memory operand.
|
||||||
|
func (e *enc) encodeMxcsr(digit int, ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
m, ok := ops[0].(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("MXCSR instruction requires a memory operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, digit, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// cvtIntOp maps the scalar float-to-integer conversions to their mandatory
|
||||||
|
// prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating
|
||||||
|
// CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width.
|
||||||
|
var cvtIntOp = map[string]struct {
|
||||||
|
prefix byte
|
||||||
|
op byte
|
||||||
|
}{
|
||||||
|
"CVTSD2S": {0xF2, 0x2D},
|
||||||
|
"CVTTSD2S": {0xF2, 0x2C},
|
||||||
|
"CVTSS2S": {0xF3, 0x2D},
|
||||||
|
"CVTTSS2S": {0xF3, 0x2C},
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C
|
||||||
|
// with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the
|
||||||
|
// quad spellings.
|
||||||
|
func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||||
|
}
|
||||||
|
spec := cvtIntOp[base]
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s destination must be a general register", base)
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, spec.op})
|
||||||
|
i.prefix = spec.prefix
|
||||||
|
if err := setRM(i, dstReg, src, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeFmov encodes the x87 double move. The memory forms are DD /0
|
||||||
|
// (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register
|
||||||
|
// move is DD C0+dst (FLD st(dst)), the form the Go assembler emits.
|
||||||
|
func (e *enc) encodeFmov(ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
srcReg, srcIsF := src.(Reg)
|
||||||
|
dstReg, dstIsF := dst.(Reg)
|
||||||
|
srcF := srcIsF && srcReg.fp
|
||||||
|
dstF := dstIsF && dstReg.fp
|
||||||
|
switch {
|
||||||
|
case srcF && dstF:
|
||||||
|
// The register form is DD /2 with rm = the destination (FST st(dst)).
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setRMDigit(i, 2, dstReg, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case dstF:
|
||||||
|
m, ok := src.(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("FMOVD: invalid source operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, 0, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case srcF:
|
||||||
|
m, ok := dst.(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("FMOVD: invalid destination operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, 2, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("FMOVD needs an x87 register operand")
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- legacy SSE imm8, extract, insert and packed shift families --------------
|
||||||
|
|
||||||
|
// encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src,
|
||||||
|
// dst with reg = dst, rm = src and the immediate appended last (PALIGNR,
|
||||||
|
// PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4).
|
||||||
|
func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
src, dst := ops[1], ops[2]
|
||||||
|
dstReg, ok2 := dst.(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction destination must be a vector register")
|
||||||
|
}
|
||||||
|
opcode := []byte{0x0F, 0x38, m.op}
|
||||||
|
if m.map3A {
|
||||||
|
opcode = []byte{0x0F, 0x3A, m.op}
|
||||||
|
}
|
||||||
|
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, src, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the
|
||||||
|
// XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and
|
||||||
|
// PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the
|
||||||
|
// SSE4.1 0F3A 15 one).
|
||||||
|
func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("extract needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
srcReg, srcVec := vecReg(ops[1])
|
||||||
|
if !srcVec {
|
||||||
|
return fmt.Errorf("extract source must be an XMM register")
|
||||||
|
}
|
||||||
|
opcode := m.op
|
||||||
|
if m.opMem != nil && memOperand(ops[2]) {
|
||||||
|
opcode = m.opMem
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
|
||||||
|
if err := setRM(i, srcReg, ops[2], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the
|
||||||
|
// XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and
|
||||||
|
// PINSRW).
|
||||||
|
func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("insert needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dstReg, dstVec := vecReg(ops[2])
|
||||||
|
if !dstVec {
|
||||||
|
return fmt.Errorf("insert destination must be an XMM register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW}
|
||||||
|
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEShift encodes the legacy packed integer shifts. The immediate
|
||||||
|
// form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form
|
||||||
|
// OP count, dst carries the count in an XMM register (or memory) on the
|
||||||
|
// 66 0F D1-F3 opcodes. The destination is always the register written.
|
||||||
|
func (e *enc) encodeSSEShift(name string, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s destination must be the second, vector operand", name)
|
||||||
|
}
|
||||||
|
if imm, isImm := ops[0].(Imm); isImm {
|
||||||
|
spec := sseShiftImm[name]
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||||||
|
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
if !vecOrMem(ops[0]) {
|
||||||
|
return fmt.Errorf("%s count must be an immediate, a vector register or memory", name)
|
||||||
|
}
|
||||||
|
op, ok := sseShiftVar[name]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s has no variable-count form", name)
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCmpsd encodes CMPSD, the scalar double compare with its predicate
|
||||||
|
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
|
||||||
|
// F2 0F C2 with reg = dst, rm = src.
|
||||||
|
func (e *enc) encodeCmpsd(ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[2].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("CMPSD predicate must be an immediate")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dstReg, ok2 := ops[1].(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("CMPSD destination must be a vector register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the
|
||||||
|
// literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no
|
||||||
|
// prefix, reg = dst, rm = src; X0 is implicit on the wire).
|
||||||
|
func (e *enc) encodeSha256rnds2(ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
x0, ok := ops[0].(Reg)
|
||||||
|
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 first operand must be X0")
|
||||||
|
}
|
||||||
|
dstReg, ok2 := ops[2].(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 destination must be a vector register")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|||||||
+6
-1
@@ -17,12 +17,13 @@ import "strings"
|
|||||||
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
||||||
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||||
// those indices but require one. The mask flag marks the AVX-512 opmask
|
// those indices but require one. The mask flag marks the AVX-512 opmask
|
||||||
// registers K0-K7.
|
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
|
||||||
type Reg struct {
|
type Reg struct {
|
||||||
idx int
|
idx int
|
||||||
size int // informational width implied by the name; the mnemonic decides
|
size int // informational width implied by the name; the mnemonic decides
|
||||||
high bool // AH/CH/DH/BH
|
high bool // AH/CH/DH/BH
|
||||||
mask bool // K0-K7 opmask register
|
mask bool // K0-K7 opmask register
|
||||||
|
fp bool // F0-F7 x87 stack register
|
||||||
}
|
}
|
||||||
|
|
||||||
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
||||||
@@ -144,6 +145,10 @@ func buildRegByName() map[string]Reg {
|
|||||||
for i := 0; i <= 7; i++ {
|
for i := 0; i <= 7; i++ {
|
||||||
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
||||||
}
|
}
|
||||||
|
// x87 stack: F0..F7.
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
|
||||||
|
}
|
||||||
return m
|
return m
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+144
-1
@@ -41,7 +41,7 @@ const (
|
|||||||
vexExtract
|
vexExtract
|
||||||
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
||||||
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
||||||
// narrowing stores (VPMOVDW, VPMOVQD).
|
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
|
||||||
vexRMRev
|
vexRMRev
|
||||||
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
||||||
// vector length follows the source: the packed-double → dword
|
// vector length follows the source: the packed-double → dword
|
||||||
@@ -52,6 +52,15 @@ const (
|
|||||||
vexRMSrcLen
|
vexRMSrcLen
|
||||||
// vexZero is the no-operand form (VZEROUPPER).
|
// vexZero is the no-operand form (VZEROUPPER).
|
||||||
vexZero
|
vexZero
|
||||||
|
// vexZeroAll is the no-operand form that zeroes the full upper state
|
||||||
|
// (VZEROALL, the L = 1 twin of VZEROUPPER).
|
||||||
|
vexZeroAll
|
||||||
|
// vexNDS3GPR is the three-operand NDS form over general-purpose
|
||||||
|
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
|
||||||
|
vexNDS3GPR
|
||||||
|
// vexImmRMGPR is the immediate form over general-purpose registers
|
||||||
|
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
||||||
|
vexImmRMGPR
|
||||||
)
|
)
|
||||||
|
|
||||||
// vexSpec describes one VEX instruction's encoding parameters.
|
// vexSpec describes one VEX instruction's encoding parameters.
|
||||||
@@ -125,6 +134,12 @@ var vexTable = map[string]vexSpec{
|
|||||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
||||||
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
||||||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
||||||
|
// Scalar fused multiply-add (NDS form). The Go assembler carries the
|
||||||
|
// same 66 prefix as the packed forms on every FMA row, and W1 on the
|
||||||
|
// double-precision spellings, so SD shares PD's prefix/W pair and the
|
||||||
|
// scalar width rides on the W bit.
|
||||||
|
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
|
||||||
|
|
||||||
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
||||||
// no vvvv).
|
// no vvvv).
|
||||||
@@ -192,6 +207,31 @@ var vexTable = map[string]vexSpec{
|
|||||||
|
|
||||||
// VEX.128.0F.W0, no operands.
|
// VEX.128.0F.W0, no operands.
|
||||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||||
|
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
|
||||||
|
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
|
||||||
|
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
|
||||||
|
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
|
||||||
|
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
|
||||||
|
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
|
||||||
|
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
|
||||||
|
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
|
||||||
|
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
|
||||||
|
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
|
||||||
|
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
|
||||||
|
// falls through to the EVEX table.
|
||||||
|
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
|
||||||
|
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
|
||||||
|
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
|
||||||
|
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
|
||||||
|
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
|
||||||
|
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
||||||
|
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
||||||
|
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
||||||
|
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||||
|
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||||
|
|
||||||
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||||
@@ -200,6 +240,14 @@ var vexTable = map[string]vexSpec{
|
|||||||
// rm=scalar memory; SD is 256-bit only).
|
// rm=scalar memory; SD is 256-bit only).
|
||||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||||
|
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
|
||||||
|
// YMM (the encoder rejects an XMM destination, as go tool asm does).
|
||||||
|
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
|
||||||
|
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
|
||||||
|
// memory destination in rm).
|
||||||
|
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
|
||||||
|
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
|
||||||
|
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
|
||||||
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
||||||
// source).
|
// source).
|
||||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||||
@@ -290,6 +338,8 @@ type vexMoveSpec struct {
|
|||||||
var vexMoveTable = map[string]vexMoveSpec{
|
var vexMoveTable = map[string]vexMoveSpec{
|
||||||
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
||||||
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||||
|
// VEX.128/256.66.0F.WIG, aligned integer move.
|
||||||
|
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||||
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
||||||
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
||||||
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
||||||
@@ -324,6 +374,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
|
||||||
|
// only; an XMM destination is rejected exactly as go tool asm does.
|
||||||
|
if mnemUpper == "VBROADCASTI128" {
|
||||||
|
dstReg, ok := ops[len(ops)-1].(Reg)
|
||||||
|
if len(ops) != 2 || !ok || dstReg.size != 32 {
|
||||||
|
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
|
||||||
|
}
|
||||||
|
}
|
||||||
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
||||||
return e.encodeVexMove(mnemUpper, ms, ops)
|
return e.encodeVexMove(mnemUpper, ms, ops)
|
||||||
}
|
}
|
||||||
@@ -356,6 +414,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||||
case vexZero:
|
case vexZero:
|
||||||
return e.encodeVexZero(mnemUpper, spec, ops)
|
return e.encodeVexZero(mnemUpper, spec, ops)
|
||||||
|
case vexZeroAll:
|
||||||
|
return e.encodeVexZeroAll(mnemUpper, spec, ops)
|
||||||
|
case vexNDS3GPR:
|
||||||
|
return e.encodeVexNDS3GPR(spec, ops)
|
||||||
|
case vexImmRMGPR:
|
||||||
|
return e.encodeVexImmRMGPR(spec, ops)
|
||||||
|
case vexRMRev:
|
||||||
|
return e.encodeVexRMRev(spec, ops)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
||||||
}
|
}
|
||||||
@@ -607,6 +673,83 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
|
||||||
|
// twin of VZEROUPPER.
|
||||||
|
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
|
||||||
|
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
|
||||||
|
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
|
||||||
|
// rm = src2 and L = 0.
|
||||||
|
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src2, src1, dst := ops[0], ops[1], ops[2]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
vvvvReg, ok := src1.(Reg)
|
||||||
|
if !ok || vvvvReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
||||||
|
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
|
||||||
|
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, src, dst := ops[0], ops[1], ops[2]
|
||||||
|
immVal, ok := imm.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("shift control must be an immediate")
|
||||||
|
}
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(immVal))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.out = append(e.out, immByte)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
||||||
|
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
||||||
|
// a store with no register-destination form).
|
||||||
|
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
srcReg, ok := ops[0].(Reg)
|
||||||
|
if !ok || !srcReg.isVec() {
|
||||||
|
return fmt.Errorf("store source must be a vector register")
|
||||||
|
}
|
||||||
|
if !memOperand(ops[1]) {
|
||||||
|
return fmt.Errorf("store destination must be memory")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if srcReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
|
||||||
|
}
|
||||||
|
|
||||||
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
||||||
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
||||||
// move uses the store-form layout (reg = source, rm = destination), matching
|
// move uses the store-form layout (reg = source, rm = destination), matching
|
||||||
|
|||||||
@@ -19,6 +19,20 @@ func vreg(t *testing.T, name string) Reg {
|
|||||||
return r
|
return r
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
|
||||||
|
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
|
||||||
|
// go tool asm in the ground-truth tests instead.
|
||||||
|
var x86asmUnrecognised = map[string]bool{
|
||||||
|
"ANDNL": true,
|
||||||
|
"ANDNQ": true,
|
||||||
|
"MULXL": true,
|
||||||
|
"MULXQ": true,
|
||||||
|
"RORXL": true,
|
||||||
|
"RORXQ": true,
|
||||||
|
"VFMADD213SD": true,
|
||||||
|
"VFNMADD231SD": true,
|
||||||
|
}
|
||||||
|
|
||||||
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
||||||
// instruction and verifies it round-trips through the x86 decoder to the same
|
// instruction and verifies it round-trips through the x86 decoder to the same
|
||||||
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
||||||
@@ -37,8 +51,15 @@ func TestVexNDS3(t *testing.T) {
|
|||||||
t.Errorf("%s: Encode: %v", mnem, err)
|
t.Errorf("%s: Encode: %v", mnem, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
// The x86 decoder's table lacks a handful of rows the Go assembler
|
||||||
|
// emits (the scalar 213/231 FMA spellings among them); those are
|
||||||
|
// pinned byte for byte against go tool asm in TestVexGroundTruth
|
||||||
|
// instead of round-tripped here.
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
@@ -184,6 +205,38 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
||||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
||||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
||||||
|
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
|
||||||
|
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
|
||||||
|
// Packed single XOR and byte compare (NDS form).
|
||||||
|
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
|
||||||
|
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
|
||||||
|
// Octa byte shifts (vvvv carries the destination).
|
||||||
|
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
|
||||||
|
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
|
||||||
|
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
|
||||||
|
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
|
||||||
|
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
|
||||||
|
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
|
||||||
|
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
|
||||||
|
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
|
||||||
|
// Two-operand test and the non-temporal and broadcast stores.
|
||||||
|
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
|
||||||
|
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
|
||||||
|
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
|
||||||
|
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
|
||||||
|
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
|
||||||
|
// Aligned integer move and the full zeroing form.
|
||||||
|
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
|
||||||
|
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
|
||||||
|
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
|
||||||
|
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
|
||||||
|
// BMI1/BMI2 general-register VEX forms.
|
||||||
|
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
|
||||||
|
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
|
||||||
|
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
|
||||||
|
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
|
||||||
|
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
|
||||||
|
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
|
||||||
// Two-operand reg/rm form (v̄vvv must be 1111).
|
// Two-operand reg/rm form (v̄vvv must be 1111).
|
||||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
||||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
||||||
@@ -287,6 +340,12 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
// The decoder's AVX/BMI table lacks a few rows the Go
|
||||||
|
// assembler emits (the GPR VEX forms and the scalar FMA
|
||||||
|
// spellings); their bytes are the ground truth here.
|
||||||
|
if x86asmUnrecognised[c.mnem] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|||||||
Vendored
+69
@@ -0,0 +1,69 @@
|
|||||||
|
// Atomics and carry-extending multi-word arithmetic: exchange,
|
||||||
|
// compare-exchange, exchange-add, ADCX/ADOX and the CRC-32 accumulator
|
||||||
|
// family. Every result is folded back so no instruction is dead.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func xchg(p *uint64, v uint64) uint64
|
||||||
|
TEXT ·xchg(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), AX
|
||||||
|
MOVQ v+8(FP), BX
|
||||||
|
XCHGQ BX, (AX)
|
||||||
|
XCHGQ BX, CX
|
||||||
|
XCHGL BX, CX
|
||||||
|
XCHGW BX, CX
|
||||||
|
XCHGB BL, CL
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func cmpxchg(p *uint64, old, new uint64) uint8
|
||||||
|
TEXT ·cmpxchg(SB), NOSPLIT, $0-25
|
||||||
|
MOVQ p+0(FP), AX
|
||||||
|
MOVQ old+8(FP), BX
|
||||||
|
MOVQ new+16(FP), CX
|
||||||
|
CMPXCHGQ CX, (AX)
|
||||||
|
CMPXCHGL CX, BX
|
||||||
|
CMPXCHGW CX, BX
|
||||||
|
CMPXCHGB CL, BL
|
||||||
|
SETEQ AL
|
||||||
|
MOVB AL, ret+24(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func xadd(p *uint64, v uint64) uint64
|
||||||
|
TEXT ·xadd(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), AX
|
||||||
|
MOVQ v+8(FP), BX
|
||||||
|
XADDQ BX, (AX)
|
||||||
|
XADDL BX, CX
|
||||||
|
XADDW BX, CX
|
||||||
|
XADDB BL, CL
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func adcx_adox(lo, hi, x, y uint64) uint64
|
||||||
|
TEXT ·adcx_adox(SB), NOSPLIT, $0-40
|
||||||
|
MOVQ lo+0(FP), AX
|
||||||
|
MOVQ hi+8(FP), DX
|
||||||
|
MOVQ x+16(FP), BX
|
||||||
|
MOVQ y+24(FP), CX
|
||||||
|
ADCXQ BX, AX
|
||||||
|
ADOXQ CX, DX
|
||||||
|
ADCXL BX, AX
|
||||||
|
ADOXL CX, DX
|
||||||
|
XORQ BX, BX
|
||||||
|
ADCXQ BX, AX
|
||||||
|
MOVQ AX, ret+32(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func crc32(crc uint32, p *byte, n int) uint32
|
||||||
|
TEXT ·crc32(SB), NOSPLIT, $0-28
|
||||||
|
MOVL crc+0(FP), AX
|
||||||
|
MOVQ p+8(FP), SI
|
||||||
|
MOVQ n+16(FP), CX
|
||||||
|
CRC32B (SI), AX
|
||||||
|
CRC32Q (SI), CX
|
||||||
|
CRC32L (SI), AX
|
||||||
|
MOVW (SI), DX
|
||||||
|
CRC32W DX, AX
|
||||||
|
MOVL AX, ret+24(FP)
|
||||||
|
RET
|
||||||
Vendored
+102
@@ -0,0 +1,102 @@
|
|||||||
|
// The AVX/AVX-512 gap families: fused scalar multiply-add, carries through
|
||||||
|
// GF(2^8) affine transforms, population counts, non-temporal stores, mask
|
||||||
|
// moves and the KMOV widths. Every result is folded back so no instruction
|
||||||
|
// is dead.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func avxblend(a, b []float64) float64
|
||||||
|
TEXT ·avxblend(SB), NOSPLIT, $0-56
|
||||||
|
MOVQ a_base+0(FP), SI
|
||||||
|
MOVQ b_base+24(FP), DI
|
||||||
|
VMOVUPD (SI), Y0
|
||||||
|
VMOVUPD (DI), Y1
|
||||||
|
VXORPS Y2, Y2, Y2
|
||||||
|
VSHUFPD $5, Y0, Y1, Y3
|
||||||
|
VMOVUPD Y3, (SI)
|
||||||
|
VPBLENDD $3, Y0, Y1, Y4
|
||||||
|
VPERM2F128 $1, Y4, Y0, Y0
|
||||||
|
VEXTRACTF128 $1, Y0, X1
|
||||||
|
VZEROALL
|
||||||
|
VMOVSD X1, ret+48(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avxint(p *byte, n int) uint64
|
||||||
|
TEXT ·avxint(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
VMOVDQU (SI), Y0
|
||||||
|
VPCMPEQB Y0, Y0, Y1
|
||||||
|
VPSLLDQ $2, X0, X0
|
||||||
|
VPSRLDQ $4, Y0, Y0
|
||||||
|
VPALIGNR $3, X0, X1, X1
|
||||||
|
VPCLMULQDQ $0, X0, X1, X2
|
||||||
|
VGF2P8AFFINEQB $7, X2, X0, X3
|
||||||
|
VPOPCNTB X3, X4
|
||||||
|
VPOPCNTD Y0, Y5
|
||||||
|
VPERMI2B X0, X1, X2
|
||||||
|
VPTEST X0, X0
|
||||||
|
VPMOVMSKB X1, AX
|
||||||
|
VZEROUPPER
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avxnt(p *float64)
|
||||||
|
TEXT ·avxnt(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ p+0(FP), DI
|
||||||
|
VMOVUPD (DI), Y0
|
||||||
|
VADDPD Y0, Y0, Y0
|
||||||
|
VMOVNTDQ Y0, (DI)
|
||||||
|
VMOVNTDQ X0, 16(DI)
|
||||||
|
VZEROALL
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avxmas(a, b []float64) float64
|
||||||
|
TEXT ·avxmas(SB), NOSPLIT, $0-56
|
||||||
|
MOVQ a_base+0(FP), SI
|
||||||
|
MOVQ b_base+24(FP), DI
|
||||||
|
VMOVSD (SI), X0
|
||||||
|
VMOVSD (DI), X1
|
||||||
|
VFMADD213SD X1, X0, X0
|
||||||
|
VFNMADD231SD X1, X0, X0
|
||||||
|
VADDSD X1, X0, X0
|
||||||
|
VMOVSD X0, ret+48(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avxgpr(x, y uint64) uint64
|
||||||
|
TEXT ·avxgpr(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
MOVQ y+8(FP), BX
|
||||||
|
ANDNL BX, AX, CX
|
||||||
|
MULXQ BX, DX, SI
|
||||||
|
RORXL $3, AX, CX
|
||||||
|
RORXQ $7, BX, SI
|
||||||
|
MOVQ CX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avxmask(kin uint8, p *byte) uint8
|
||||||
|
TEXT ·avxmask(SB), NOSPLIT, $0-17
|
||||||
|
MOVQ p+8(FP), SI
|
||||||
|
KMOVB kin+0(FP), K1
|
||||||
|
KMOVB K1, K2
|
||||||
|
KMOVW K2, K1
|
||||||
|
KMOVD K1, K3
|
||||||
|
KMOVQ K3, K4
|
||||||
|
KMOVB K4, K1
|
||||||
|
KMOVB K1, AX
|
||||||
|
KMOVD K1, (SI)
|
||||||
|
MOVB AL, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func avx512(p *uint64, n int) uint64
|
||||||
|
TEXT ·avx512(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
VMOVDQU64 (SI), Z0
|
||||||
|
VPORQ Z0, Z0, Z1
|
||||||
|
VPOPCNTQ Z1, Z2
|
||||||
|
VPERMB Z1, Z0, Z2
|
||||||
|
VPXORD Z2, Z1, Z0
|
||||||
|
VMOVDQA64 Z0, (SI)
|
||||||
|
VZEROUPPER
|
||||||
|
XORQ AX, AX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
Vendored
+57
@@ -0,0 +1,57 @@
|
|||||||
|
// The AES-NI, SHA and carry-less multiply round instructions as GOROOT's
|
||||||
|
// crypto kernels spell them. Every result is folded back so no instruction
|
||||||
|
// is dead.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func aesround(blk, rk *byte)
|
||||||
|
TEXT ·aesround(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ blk+0(FP), SI
|
||||||
|
MOVQ rk+8(FP), DI
|
||||||
|
MOVOU (SI), X0
|
||||||
|
MOVOU (DI), X1
|
||||||
|
AESENC X1, X0
|
||||||
|
AESENCLAST X1, X0
|
||||||
|
AESDEC X1, X0
|
||||||
|
AESDECLAST X1, X0
|
||||||
|
AESIMC X1, X2
|
||||||
|
AESKEYGENASSIST $1, X1, X3
|
||||||
|
MOVOU X0, (SI)
|
||||||
|
MOVOU X2, (DI)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func sha1block(p *byte, n int, h *[5]uint32)
|
||||||
|
TEXT ·sha1block(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
MOVQ h+16(FP), DI
|
||||||
|
MOVOU (SI), X0
|
||||||
|
MOVOU 16(SI), X1
|
||||||
|
SHA1RNDS4 $0, X1, X0
|
||||||
|
SHA1NEXTE X1, X0
|
||||||
|
SHA1MSG1 X1, X2
|
||||||
|
SHA1MSG2 X1, X2
|
||||||
|
MOVOU X0, (DI)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func sha256block(p *byte, n int, h *[8]uint32)
|
||||||
|
TEXT ·sha256block(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
MOVQ h+16(FP), DI
|
||||||
|
MOVOU (SI), X0
|
||||||
|
MOVOU 16(SI), X1
|
||||||
|
SHA256RNDS2 X0, X1, X0
|
||||||
|
SHA256MSG1 X1, X2
|
||||||
|
SHA256MSG2 X1, X2
|
||||||
|
MOVOU X0, (DI)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func pclmul(a, b *byte)
|
||||||
|
TEXT ·pclmul(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ a+0(FP), SI
|
||||||
|
MOVQ b+8(FP), DI
|
||||||
|
MOVOU (SI), X0
|
||||||
|
MOVOU (DI), X1
|
||||||
|
PCLMULQDQ $0, X1, X0
|
||||||
|
PCLMULQDQ $17, (DI), X0
|
||||||
|
MOVOU X0, (SI)
|
||||||
|
RET
|
||||||
Vendored
+76
@@ -0,0 +1,76 @@
|
|||||||
|
// Carry arithmetic, rotates, unsigned/signed division and bit tests: the
|
||||||
|
// scalar families GOROOT's big-number and crypto kernels use. Every result
|
||||||
|
// is folded back so no instruction is dead.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func carry(a, b uint64) uint64
|
||||||
|
TEXT ·carry(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
MOVQ b+8(FP), BX
|
||||||
|
ADDQ BX, AX
|
||||||
|
ADCQ $0, AX
|
||||||
|
MOVQ BX, CX
|
||||||
|
SBBQ $1, CX
|
||||||
|
ADCL BX, AX
|
||||||
|
ADCB AL, BL
|
||||||
|
ADCW $7, CX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func borrow(a, b uint64) uint64
|
||||||
|
TEXT ·borrow(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
MOVQ b+8(FP), BX
|
||||||
|
SUBQ BX, AX
|
||||||
|
SBBQ $0, AX
|
||||||
|
SBBQ BX, CX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func rot(x uint64, n uint32) uint64
|
||||||
|
TEXT ·rot(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
MOVL n+8(FP), CX
|
||||||
|
ROLQ CL, AX
|
||||||
|
RORQ $7, AX
|
||||||
|
ROLL $1, AX
|
||||||
|
RORL CL, AX
|
||||||
|
RCLQ $1, AX
|
||||||
|
RCRQ CL, AX
|
||||||
|
ROLW $3, AX
|
||||||
|
SALQ $2, AX
|
||||||
|
SALB $1, AX
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func muldiv(a, b uint64) uint64
|
||||||
|
TEXT ·muldiv(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), AX
|
||||||
|
MOVQ b+8(FP), BX
|
||||||
|
MULQ BX
|
||||||
|
MULQ (BX)
|
||||||
|
MOVL (BX), CX
|
||||||
|
MULL CX
|
||||||
|
DIVQ BX
|
||||||
|
IDIVQ BX
|
||||||
|
MOVL a+0(FP), AX
|
||||||
|
DIVL CX
|
||||||
|
IDIVL CX
|
||||||
|
MOVQ AX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func bitfield(w *uint64) uint64
|
||||||
|
TEXT ·bitfield(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ (DI), AX
|
||||||
|
MOVQ (DI), CX
|
||||||
|
BTQ AX, CX
|
||||||
|
BTQ $3, (DI)
|
||||||
|
BTL AX, CX
|
||||||
|
BTW $1, CX
|
||||||
|
BTSQ $5, AX
|
||||||
|
BTRQ AX, CX
|
||||||
|
BTCQ $7, (DI)
|
||||||
|
SETCS AL
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
Vendored
+77
@@ -0,0 +1,77 @@
|
|||||||
|
// The legacy SSE gap families: scalar compares and square roots, the Plan 9
|
||||||
|
// packed spellings, shuffles, lane extracts and inserts, packed integer
|
||||||
|
// shifts and the octa moves. Every result is folded back so no instruction
|
||||||
|
// is dead.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func cmporder(a, b *float64) int
|
||||||
|
TEXT ·cmporder(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ a+0(FP), SI
|
||||||
|
MOVQ b+8(FP), DI
|
||||||
|
MOVSD (SI), X0
|
||||||
|
MOVSD (DI), X1
|
||||||
|
ANDNPD X0, X2
|
||||||
|
ANDNPS X0, X3
|
||||||
|
COMISD X0, X1
|
||||||
|
SQRTSD X0, X2
|
||||||
|
CMPSD X0, X1, $5
|
||||||
|
MOVL SI, CX
|
||||||
|
SETPL CL
|
||||||
|
MOVL CX, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func packed(w *uint64) uint64
|
||||||
|
TEXT ·packed(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ w+0(FP), SI
|
||||||
|
MOVO (SI), X0
|
||||||
|
MOVOA (SI), X1
|
||||||
|
PADDL X0, X1
|
||||||
|
PSUBL X0, X1
|
||||||
|
PCMPEQL X0, X1
|
||||||
|
PUNPCKLBW X0, X1
|
||||||
|
PSHUFL $27, X0, X2
|
||||||
|
MOVOU X2, (SI)
|
||||||
|
MOVQ (SI), AX
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func lanes(p *byte, buf *byte)
|
||||||
|
TEXT ·lanes(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
MOVQ buf+8(FP), DI
|
||||||
|
MOVO (SI), X0
|
||||||
|
MOVQ SI, AX
|
||||||
|
PINSRB $1, AX, X0
|
||||||
|
PINSRW $2, AX, X0
|
||||||
|
PINSRD $3, AX, X0
|
||||||
|
PINSRQ $1, AX, X0
|
||||||
|
PEXTRB $1, X0, AX
|
||||||
|
PEXTRW $2, X0, AX
|
||||||
|
PEXTRD $3, X0, AX
|
||||||
|
PEXTRQ $1, X0, CX
|
||||||
|
PCMPESTRI $4, X0, X0
|
||||||
|
MOVB AL, (DI)
|
||||||
|
MOVOU X0, (SI)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func shifts(p *uint64)
|
||||||
|
TEXT ·shifts(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ p+0(FP), SI
|
||||||
|
MOVO (SI), X0
|
||||||
|
MOVO X0, X1
|
||||||
|
PSLLW $3, X0
|
||||||
|
PSRLW $1, X1
|
||||||
|
PSRAW $2, X0
|
||||||
|
PSLLL $4, X0
|
||||||
|
PSRLL $5, X1
|
||||||
|
PSRAL $1, X0
|
||||||
|
PSLLQ $7, X0
|
||||||
|
PSRLQ $9, X1
|
||||||
|
PSLLL X1, X0
|
||||||
|
PSRLQ X0, X1
|
||||||
|
PSLLDQ $2, X0
|
||||||
|
PSRLDQ $4, X1
|
||||||
|
MOVOU X0, (SI)
|
||||||
|
MOVOU X1, 16(SI)
|
||||||
|
RET
|
||||||
Vendored
+76
@@ -0,0 +1,76 @@
|
|||||||
|
// System, string-primitive and x87 families: flag register moves, the
|
||||||
|
// serialising instructions, MOVS/STOS, the MXCSR pair, scalar float-to-int
|
||||||
|
// conversions and FMOVD. Every result is folded back so no instruction is
|
||||||
|
// dead.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// func system(x uint64) uint64
|
||||||
|
TEXT ·system(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
PUSHFQ
|
||||||
|
POPFQ
|
||||||
|
CPUID
|
||||||
|
RDTSC
|
||||||
|
RDTSCP
|
||||||
|
SYSCALL
|
||||||
|
XGETBV
|
||||||
|
PAUSE
|
||||||
|
LFENCE
|
||||||
|
MFENCE
|
||||||
|
SFENCE
|
||||||
|
UNDEF
|
||||||
|
XORQ AX, BX
|
||||||
|
MOVQ BX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func stringprim(p *byte, n int) uint64
|
||||||
|
TEXT ·stringprim(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ p+0(FP), DI
|
||||||
|
MOVQ n+8(FP), CX
|
||||||
|
LEAQ buf<>(SB), AX
|
||||||
|
MOVQ AX, SI
|
||||||
|
CLD
|
||||||
|
MOVSB
|
||||||
|
MOVSW
|
||||||
|
MOVSL
|
||||||
|
MOVSQ
|
||||||
|
STOSB
|
||||||
|
STOSQ
|
||||||
|
STOSL
|
||||||
|
STOSW
|
||||||
|
MOVQ DI, ret+16(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
DATA buf<>+0x00(SB)/8, $0
|
||||||
|
|
||||||
|
GLOBL buf<>(SB), NOPTR, $8
|
||||||
|
|
||||||
|
// func intgate(x uint64) uint64
|
||||||
|
TEXT ·intgate(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
INT $3
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func fpmxcsr(x float64, csr *uint32) int64
|
||||||
|
TEXT ·fpmxcsr(SB), NOSPLIT, $0-24
|
||||||
|
MOVQ x+0(FP), X0
|
||||||
|
MOVQ csr+8(FP), AX
|
||||||
|
STMXCSR (AX)
|
||||||
|
LDMXCSR (AX)
|
||||||
|
CVTSD2SL X0, CX
|
||||||
|
CVTTSD2SQ X0, DX
|
||||||
|
MOVL (AX), SI
|
||||||
|
MOVQ SI, ret+8(FP)
|
||||||
|
RET
|
||||||
|
|
||||||
|
// func fmove(p *float64) float64
|
||||||
|
TEXT ·fmove(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ p+0(FP), AX
|
||||||
|
FMOVD (AX), F0
|
||||||
|
FMOVD F0, F1
|
||||||
|
FMOVD F0, (AX)
|
||||||
|
MOVQ (AX), AX
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
Reference in New Issue
Block a user