feat(amd64): encode the GOROOT instruction families

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 06:44:51 +02:00
parent 39d2e80145
commit fc2d92eabd
16 changed files with 1709 additions and 43 deletions
+4
View File
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
}
// x87 stack registers (FMOVD and the other x87 moves).
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
}
return regs
}
+32 -8
View File
@@ -18,7 +18,11 @@ func Encodable(mnemonic string) bool {
// Fixed-name instructions (no size suffix).
switch upper {
case "RET", "NOP", "CALL", "JMP":
case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2":
return true
}
if _, ok := noOperandTable[upper]; ok {
return true
}
if _, ok := condCode(upper); ok {
@@ -31,7 +35,7 @@ func Encodable(mnemonic string) bool {
return false
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" {
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return true
}
@@ -53,13 +57,28 @@ func Encodable(mnemonic string) bool {
}
}
// Legacy SSE shuffles and packed binaries dispatch on the full name.
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
// do the imm8-controlled instructions, the lane extracts and inserts and
// the packed integer shifts (their trailing width letters belong to the
// mnemonic).
if _, ok := sseShufTable[upper]; ok {
return true
}
if _, ok := sseBinTable[upper]; ok {
return true
}
if _, ok := sseImm3Table[upper]; ok {
return true
}
if _, ok := sseExtractTable[upper]; ok {
return true
}
if _, ok := sseInsertTable[upper]; ok {
return true
}
if _, ok := sseShiftImm[upper]; ok {
return true
}
// The size-suffix split: retry the tables and the scalar switch on the
// base.
@@ -74,12 +93,15 @@ func Encodable(mnemonic string) bool {
}
}
switch base2 {
case "MOV",
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
case "MOV", "MOVD",
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
"TEST",
"LEA",
"INC", "DEC", "NEG", "NOT",
"SHL", "SHR", "SAR",
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
"BT", "BTS", "BTR", "BTC",
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
"MOVS", "STOS",
"IMUL", "IMUL3",
"PUSH", "POP",
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
@@ -88,7 +110,9 @@ func Encodable(mnemonic string) bool {
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
"CVTSL2SD", "CVTSQ2SD",
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
"FMOVD",
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return true
}
// Full-name dispatches the size split would eat (a trailing width
+81 -6
View File
@@ -58,6 +58,41 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if cc, ok := condCode(upper); ok {
return e.encodeJcc(cc, ops)
}
// No-operand system and string-control instructions (CPUID, RDTSC,
// SYSCALL, the fences, UNDEF, …).
if op, ok := noOperandTable[upper]; ok {
if len(ops) != 0 {
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
}
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper {
case "POPFQ":
if len(ops) != 0 {
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
}
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
case "PUSHFQ":
if len(ops) != 0 {
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
}
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT":
return e.encodeInt(ops)
case "LDMXCSR":
return e.encodeMxcsr(2, ops)
case "STMXCSR":
return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm).
case "CMPSD":
return e.encodeCmpsd(ops)
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2":
return e.encodeSha256rnds2(ops)
}
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
@@ -67,7 +102,8 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if err != nil {
return err
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return e.encodeVec(base, ops, sfx)
}
if sfx.any() {
@@ -101,6 +137,21 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if m, ok := sseBinTable[base]; ok {
return e.encodeSSEBin(m, ops)
}
// The imm8-controlled legacy instructions, the lane extracts and inserts
// and the packed integer shifts all dispatch on the full name: a trailing
// width letter here belongs to the mnemonic, not to the size split.
if m, ok := sseImm3Table[upper]; ok {
return e.encodeSSEImm3(m, ops)
}
if m, ok := sseExtractTable[upper]; ok {
return e.encodeSSEExtract(m, ops)
}
if m, ok := sseInsertTable[upper]; ok {
return e.encodeSSEInsert(m, ops)
}
if _, ok := sseShiftImm[upper]; ok {
return e.encodeSSEShift(upper, ops)
}
// PMOVMSKB ends in a width letter the size split would eat, so it
// dispatches on the full name like the packed binaries above.
if upper == "PMOVMSKB" {
@@ -109,16 +160,36 @@ func (e *enc) encode(mnem string, ops []Operand) error {
switch base {
case "MOV":
return e.encodeMov(ops, size)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
// REX.W and all.
case "MOVD":
return e.encodeMov(ops, 8)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
return e.encodeALU(aluOp[base], ops, size)
case "TEST":
return e.encodeTest(ops, size)
case "LEA":
return e.encodeLea(ops, size)
case "INC", "DEC", "NEG", "NOT":
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
return e.encodeUnary(unaryOp[base], ops, size)
case "SHL", "SHR", "SAR":
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
return e.encodeShift(shiftOp[base], ops, size)
case "BT", "BTS", "BTR", "BTC":
return e.encodeBitTest(base, ops, size)
case "XCHG":
return e.encodeExchange(ops, size)
case "CMPXCHG":
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
case "XADD":
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
case "CRC32":
return e.encodeCrc32(ops, size)
case "ADCX":
return e.encodeCarryExt(0x66, ops, size)
case "ADOX":
return e.encodeCarryExt(0xF3, ops, size)
case "MOVS", "STOS":
return e.encodeStringOp(base, ops, size)
case "IMUL", "IMUL3":
return e.encodeImul(ops, size)
case "PUSH":
@@ -136,7 +207,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD":
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
return e.encodeCvtInt(base, ops, size)
case "FMOVD":
return e.encodeFmov(ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return e.encodeSSEMove(sseMoveTable[base], ops)
}
return fmt.Errorf("unsupported instruction %q", mnem)
@@ -195,7 +270,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
if ss, ok := scatterTable[upper]; ok {
return e.encodeScatter(upper, ss, ops, sfx)
}
if upper == "KMOVW" || upper == "KMOVQ" {
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
+242
View File
@@ -546,6 +546,248 @@ func TestEncodableCmovSize(t *testing.T) {
}
}
// TestCarryShiftMulGroundTruth pins the carry-flag ALU family (ADC/SBB with
// their accumulator immediate forms), the rotate family, MUL/DIV/IDIV and the
// bit-test family byte for byte against go tool asm (see
// testdata/verify/scalar_amd64.s).
func TestCarryShiftMulGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ADCQ AX,BX", "ADCQ", []Operand{AX, BX}, "4811c3"},
{"ADCL AX,BX", "ADCL", []Operand{AX, BX}, "11c3"},
{"ADCB AL,BL", "ADCB", []Operand{AL, BL}, "10c3"},
{"ADCW AX,BX", "ADCW", []Operand{AX, BX}, "6611c3"},
{"SBBQ AX,BX", "SBBQ", []Operand{AX, BX}, "4819c3"},
{"ADCQ $5,BX", "ADCQ", []Operand{Imm(5), BX}, "4883d305"},
{"ADCQ $300,BX", "ADCQ", []Operand{Imm(300), BX}, "4881d32c010000"},
{"ADCQ $300,AX", "ADCQ", []Operand{Imm(300), AX}, "48152c010000"},
{"ADCB $5,AL", "ADCB", []Operand{Imm(5), AL}, "1405"},
{"SBBQ $300,AX", "SBBQ", []Operand{Imm(300), AX}, "481d2c010000"},
{"ADCQ AX,(BX)", "ADCQ", []Operand{AX, Ptr(BX, 0, 8)}, "481103"},
{"ROLQ $3,AX", "ROLQ", []Operand{Imm(3), AX}, "48c1c003"},
{"ROLL CX,BX", "ROLL", []Operand{CL, BX}, "d3c3"},
{"RORQ CL,AX", "RORQ", []Operand{CL, AX}, "48d3c8"},
{"RCRQ $1,BX", "RCRQ", []Operand{Imm(1), BX}, "48d1db"},
{"RCLQ $3,AX", "RCLQ", []Operand{Imm(3), AX}, "48c1d003"},
{"RORB CL,BL", "RORB", []Operand{CL, BL}, "d2cb"},
{"SALQ $2,AX", "SALQ", []Operand{Imm(2), AX}, "48c1e002"},
{"ROLW $1,AX", "ROLW", []Operand{Imm(1), AX}, "66d1c0"},
{"MULQ CX", "MULQ", []Operand{CX}, "48f7e1"},
{"MULL CX", "MULL", []Operand{CX}, "f7e1"},
{"MULB CL", "MULB", []Operand{CL}, "f6e1"},
{"DIVL CX", "DIVL", []Operand{CX}, "f7f1"},
{"IDIVQ CX", "IDIVQ", []Operand{CX}, "48f7f9"},
{"MULW CX", "MULW", []Operand{CX}, "66f7e1"},
{"BTQ AX,DX", "BTQ", []Operand{AX, DX}, "480fa3c2"},
{"BTL AX,DX", "BTL", []Operand{AX, DX}, "0fa3c2"},
{"BTW AX,DX", "BTW", []Operand{AX, DX}, "660fa3c2"},
{"BTQ $3,BX", "BTQ", []Operand{Imm(3), BX}, "480fbae303"},
{"BTQ $3,(AX)", "BTQ", []Operand{Imm(3), Ptr(AX, 0, 8)}, "480fba2003"},
{"BTSQ $5,BX", "BTSQ", []Operand{Imm(5), BX}, "480fbaeb05"},
{"BTCQ AX,BX", "BTCQ", []Operand{AX, BX}, "480fbbc3"},
{"BTRQ $7,BX", "BTRQ", []Operand{Imm(7), BX}, "480fbaf307"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// The bit-test immediate is an unsigned bit index with the negative
// spelling accepted, the shuffle convention: BTQ $300 must be rejected.
if _, err := Encode("BTQ", Imm(300), AX); err == nil {
t.Errorf("BTQ $300: expected an error, got none")
}
}
// TestAtomicSystemGroundTruth pins the exchange/compare-exchange/accumulate
// family, the string primitives, the flag and system instructions, the MXCSR
// pair, the scalar float-to-int conversions and the x87 FMOVD byte for byte
// against go tool asm (see testdata/verify/atomics_amd64.s and
// testdata/verify/system_amd64.s).
func TestAtomicSystemGroundTruth(t *testing.T) {
r8 := Reg{idx: 8, size: 8}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"XCHGQ AX,BX", "XCHGQ", []Operand{AX, BX}, "4893"},
{"XCHGQ BX,AX", "XCHGQ", []Operand{BX, AX}, "4893"},
{"XCHGL AX,BX", "XCHGL", []Operand{AX, BX}, "93"},
{"XCHGB AL,BL", "XCHGB", []Operand{AL, BL}, "86c3"},
{"XCHGW AX,BX", "XCHGW", []Operand{AX, BX}, "6693"},
{"XCHGQ R8,R9", "XCHGQ", []Operand{r8, Reg{idx: 9, size: 8}}, "4d87c1"},
{"XCHGQ BX,(AX)", "XCHGQ", []Operand{BX, Ptr(AX, 0, 8)}, "488718"},
{"XCHGQ (AX),BX", "XCHGQ", []Operand{Ptr(AX, 0, 8), BX}, "488718"},
{"XCHGQ AX,(BX)", "XCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "488703"},
{"CMPXCHGL AX,BX", "CMPXCHGL", []Operand{AX, BX}, "0fb1c3"},
{"CMPXCHGQ AX,(BX)", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
{"CMPXCHGB AL,(BX)", "CMPXCHGB", []Operand{AL, Ptr(BX, 0, 1)}, "0fb003"},
{"CMPXCHGW AX,BX", "CMPXCHGW", []Operand{AX, BX}, "660fb1c3"},
{"XADDL AX,BX", "XADDL", []Operand{AX, BX}, "0fc1c3"},
{"XADDQ AX,(BX)", "XADDQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fc103"},
{"XADDB AL,(BX)", "XADDB", []Operand{AL, Ptr(BX, 0, 1)}, "0fc003"},
{"XADDW AX,BX", "XADDW", []Operand{AX, BX}, "660fc1c3"},
{"ADCXL AX,CX", "ADCXL", []Operand{AX, CX}, "660f38f6c8"},
{"ADCXQ AX,CX", "ADCXQ", []Operand{AX, CX}, "66480f38f6c8"},
{"ADOXL AX,CX", "ADOXL", []Operand{AX, CX}, "f30f38f6c8"},
{"ADOXQ AX,CX", "ADOXQ", []Operand{AX, CX}, "f3480f38f6c8"},
{"CRC32B AX,CX", "CRC32B", []Operand{AX, CX}, "f20f38f0c8"},
{"CRC32W AX,CX", "CRC32W", []Operand{AX, CX}, "66f20f38f1c8"},
{"CRC32L AX,CX", "CRC32L", []Operand{AX, CX}, "f20f38f1c8"},
{"CRC32Q AX,CX", "CRC32Q", []Operand{AX, CX}, "f2480f38f1c8"},
{"CRC32L (AX),CX", "CRC32L", []Operand{Ptr(AX, 0, 4), CX}, "f20f38f108"},
{"MOVSQ", "MOVSQ", []Operand{}, "48a5"},
{"MOVSL", "MOVSL", []Operand{}, "a5"},
{"MOVSB", "MOVSB", []Operand{}, "a4"},
{"MOVSW", "MOVSW", []Operand{}, "66a5"},
{"STOSB", "STOSB", []Operand{}, "aa"},
{"STOSQ", "STOSQ", []Operand{}, "48ab"},
{"STOSL", "STOSL", []Operand{}, "ab"},
{"STOSW", "STOSW", []Operand{}, "66ab"},
{"CLD", "CLD", []Operand{}, "fc"},
{"STD", "STD", []Operand{}, "fd"},
{"POPFQ", "POPFQ", []Operand{}, "9d"},
{"PUSHFQ", "PUSHFQ", []Operand{}, "9c"},
{"CPUID", "CPUID", []Operand{}, "0fa2"},
{"RDTSC", "RDTSC", []Operand{}, "0f31"},
{"RDTSCP", "RDTSCP", []Operand{}, "0f01f9"},
{"SYSCALL", "SYSCALL", []Operand{}, "0f05"},
{"XGETBV", "XGETBV", []Operand{}, "0f01d0"},
{"PAUSE", "PAUSE", []Operand{}, "f390"},
{"LFENCE", "LFENCE", []Operand{}, "0faee8"},
{"MFENCE", "MFENCE", []Operand{}, "0faef0"},
{"SFENCE", "SFENCE", []Operand{}, "0faef8"},
{"UNDEF", "UNDEF", []Operand{}, "0f0b"},
{"INT $3", "INT", []Operand{Imm(3)}, "cd03"},
{"LDMXCSR (AX)", "LDMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae10"},
{"STMXCSR (AX)", "STMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae18"},
{"CVTSD2SL X0,AX", "CVTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2dc0"},
{"CVTTSD2SQ X0,AX", "CVTTSD2SQ", []Operand{vreg(t, "X0"), AX}, "f2480f2cc0"},
{"CVTTSD2SL X0,AX", "CVTTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2cc0"},
{"CVTSS2SQ X0,AX", "CVTSS2SQ", []Operand{vreg(t, "X0"), AX}, "f3480f2dc0"},
{"FMOVD (AX),F0", "FMOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "F0")}, "dd00"},
{"FMOVD F0,(AX)", "FMOVD", []Operand{vreg(t, "F0"), Ptr(AX, 0, 8)}, "dd10"},
{"FMOVD F0,F1", "FMOVD", []Operand{vreg(t, "F0"), vreg(t, "F1")}, "ddd1"},
{"MOVD AX,X0", "MOVD", []Operand{AX, vreg(t, "X0")}, "66480f6ec0"},
{"MOVD X0,AX", "MOVD", []Operand{vreg(t, "X0"), AX}, "66480f7ec0"},
{"MOVD X0,X1", "MOVD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f30f7ec8"},
{"MOVD (AX),X0", "MOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "X0")}, "f30f7e00"},
{"MOVD X0,(AX)", "MOVD", []Operand{vreg(t, "X0"), Ptr(AX, 0, 8)}, "660fd600"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// LDMXCSR/STMXCSR take a memory operand only.
if _, err := Encode("LDMXCSR", AX); err == nil {
t.Errorf("LDMXCSR AX: expected an error, got none")
}
}
// TestSSEGapsGroundTruth pins the legacy SSE gap families: the scalar
// compare and square root, the Plan 9 packed spellings, the imm8-controlled
// shuffles, the lane extracts and inserts, the packed integer shifts and the
// AES/SHA round instructions, byte for byte against go tool asm (see
// testdata/verify/crypto_amd64.s and testdata/verify/sse_amd64.s).
func TestSSEGapsGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ANDNPD X0,X1", "ANDNPD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f55c8"},
{"ANDNPS X0,X1", "ANDNPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f55c8"},
{"COMISD X0,X1", "COMISD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f2fc8"},
{"SQRTSD X0,X1", "SQRTSD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f20f51c8"},
{"PSHUFL $3,X0,X1", "PSHUFL", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f70c803"},
{"PALIGNR $2,X0,X1", "PALIGNR", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0fc802"},
{"PBLENDW $3,X0,X1", "PBLENDW", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0ec803"},
{"PCMPESTRI $1,X0,X1", "PCMPESTRI", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}, "660f3a61c801"},
{"PCLMULQDQ $0,X0,X1", "PCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3a44c800"},
{"PCLMULQDQ $0,(AX),X1", "PCLMULQDQ", []Operand{Imm(0), Ptr(AX, 0, 16), vreg(t, "X1")}, "660f3a440800"},
{"PEXTRB $1,X0,AX", "PEXTRB", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a14c001"},
{"PEXTRD $1,X0,AX", "PEXTRD", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a16c001"},
{"PEXTRQ $1,X0,AX", "PEXTRQ", []Operand{Imm(1), vreg(t, "X0"), AX}, "66480f3a16c001"},
{"PEXTRW $1,X0,AX", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), AX}, "660fc5c001"},
{"PEXTRW $1,X0,(AX)", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), Ptr(AX, 0, 2)}, "660f3a150001"},
{"PINSRB $1,AX,X0", "PINSRB", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a20c001"},
{"PINSRD $1,AX,X0", "PINSRD", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a22c001"},
{"PINSRQ $1,AX,X0", "PINSRQ", []Operand{Imm(1), AX, vreg(t, "X0")}, "66480f3a22c001"},
{"PINSRW $1,AX,X0", "PINSRW", []Operand{Imm(1), AX, vreg(t, "X0")}, "660fc4c001"},
{"PINSRW $1,(AX),X0", "PINSRW", []Operand{Imm(1), Ptr(AX, 0, 2), vreg(t, "X0")}, "660fc40001"},
{"PSLLL $2,X0", "PSLLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72f002"},
{"PSRAL $2,X0", "PSRAL", []Operand{Imm(2), vreg(t, "X0")}, "660f72e002"},
{"PSRLL $2,X0", "PSRLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72d002"},
{"PSRLQ $2,X0", "PSRLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d002"},
{"PSLLQ $2,X0", "PSLLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f002"},
{"PSLLW $2,X0", "PSLLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71f002"},
{"PSRLW $2,X0", "PSRLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71d002"},
{"PSRAW $2,X0", "PSRAW", []Operand{Imm(2), vreg(t, "X0")}, "660f71e002"},
{"PSLLDQ $2,X0", "PSLLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f802"},
{"PSRLDQ $2,X0", "PSRLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d802"},
{"PSLLL X0,X1", "PSLLL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ff2c8"},
{"PSRLQ X0,X1", "PSRLQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660fd3c8"},
{"PSLLL (AX),X1", "PSLLL", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660ff208"},
{"PSUBL X0,X1", "PSUBL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffac8"},
{"PADDL X0,X1", "PADDL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffec8"},
{"PCMPEQL X0,X1", "PCMPEQL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f76c8"},
{"PUNPCKLBW X0,X1", "PUNPCKLBW", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f60c8"},
{"MOVOA X0,X1", "MOVOA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f6fc8"},
{"MOVOA (AX),X1", "MOVOA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f6f08"},
{"MOVOA X0,(AX)", "MOVOA", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "660f7f00"},
{"AESIMC X0,X1", "AESIMC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dbc8"},
{"AESIMC (AX),X1", "AESIMC", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f38db08"},
{"AESENC X0,X1", "AESENC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dcc8"},
{"AESENCLAST X0,X1", "AESENCLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38ddc8"},
{"AESDEC X0,X1", "AESDEC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dec8"},
{"AESDECLAST X0,X1", "AESDECLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dfc8"},
{"AESKEYGENASSIST $0,X0,X1", "AESKEYGENASSIST", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3adfc800"},
{"SHA1MSG1 X0,X1", "SHA1MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c9c8"},
{"SHA1MSG2 X0,X1", "SHA1MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cac8"},
{"SHA1NEXTE X0,X1", "SHA1NEXTE", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c8c8"},
{"SHA1RNDS4 $0,X0,X1", "SHA1RNDS4", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "0f3accc800"},
{"SHA256MSG1 X0,X1", "SHA256MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38ccc8"},
{"SHA256MSG2 X0,X1", "SHA256MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cdc8"},
{"SHA256RNDS2 X0,X1,X2", "SHA256RNDS2", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "0f38cbd1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// SHA256RNDS2's first operand must be the literal X0.
if _, err := Encode("SHA256RNDS2", vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")); err == nil {
t.Errorf("SHA256RNDS2 X1,...: expected an error, got none")
}
// PSLLDQ has no variable-count form.
if _, err := Encode("PSLLDQ", vreg(t, "X0"), vreg(t, "X1")); err == nil {
t.Errorf("PSLLDQ X0,X1: expected an error, got none")
}
}
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
// byte for byte (no prefix / 66 / F2 / F3 variants).
func TestSSEBinGroundTruth(t *testing.T) {
+31 -14
View File
@@ -182,8 +182,17 @@ var evexTable = map[string]evexSpec{
// EVEX.66.0F38, permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2B": {2, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38, population count (reg=dst, rm=src; W selects byte/word
// against dword/qword).
"VPOPCNTB": {2, 0x54, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPOPCNTD": {2, 0x55, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPOPCNTQ": {2, 0x55, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F.W1, the qword spelling of the packed OR (VPORQ has no VEX
// form in the Go assembler: it always encodes through EVEX).
"VPORQ": {1, 0xEB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -1435,18 +1444,20 @@ var evexKOperand = map[string]bool{
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
// prefix and W for the wider widths.
// direction, kk (k → k), kmem (k → mem), gprk (GPR/mem → k) and kgpr
// (k → GPR). Each direction group carries its own mandatory prefix and W:
// the k-destination/source forms share one pair, the GPR forms another.
type kmovSpec struct {
kk, kmem, gprk, kgpr byte
gprPP int
w int
kPP, kW int // prefix and VEX.W for the k forms
gprPP, gprW int // prefix and VEX.W for the GPR forms
}
var kmovTable = map[string]kmovSpec{
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0, 0, 0},
"KMOVB": {0x90, 0x91, 0x92, 0x93, 1, 0, 1, 0},
"KMOVD": {0x90, 0x91, 0x92, 0x93, 1, 1, 3, 0},
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 0, 1, 3, 1},
}
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
@@ -1460,14 +1471,14 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
dstReg, dstIsReg := dst.(Reg)
srcK := srcIsReg && srcReg.mask
dstK := dstIsReg && dstReg.mask
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
switch {
case srcK && dstK:
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
// k ← k: reg = dst, rm = src.
spec := vexSpec{mapSel: 1, opcode: ks.kk, w: ks.kW, pp: ks.kPP, opdigit: -1}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
case srcK && dstIsReg:
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
spec.pp = ks.gprPP
// GPR ← k: reg = dst, rm = src.
spec := vexSpec{mapSel: 1, opcode: ks.kgpr, w: ks.gprW, pp: ks.gprPP, opdigit: -1}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
@@ -1477,11 +1488,17 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
if _, ok := dst.(Mem); !ok {
return fmt.Errorf("%s: invalid destination operand", upper)
}
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
// mem ← k: reg = src, rm = dst.
spec := vexSpec{mapSel: 1, opcode: ks.kmem, w: ks.kW, pp: ks.kPP, opdigit: -1}
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
case dstK:
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
spec.pp = ks.gprPP
// k ← GPR: reg = dst, rm = src. A memory source shares the k ← k
// opcode and prefix group (the ykmovb layout the Go assembler uses).
opcode, w, pp := ks.gprk, ks.gprW, ks.gprPP
if memOperand(src) {
opcode, w, pp = ks.kk, ks.kW, ks.kPP
}
spec := vexSpec{mapSel: 1, opcode: opcode, w: w, pp: pp, opdigit: -1}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
}
return fmt.Errorf("%s requires a K register operand", upper)
+19
View File
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
// The qword OR spelling always encodes through EVEX.
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
// Byte permute and population count.
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
// Align (NDS + imm8).
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
@@ -52,6 +61,16 @@ func TestEvexGroundTruth(t *testing.T) {
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
// Moves, incl. disp8×N (64 for a 512-bit operand).
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
+626 -5
View File
@@ -14,15 +14,18 @@ var aluOp = map[string]struct {
}{
"ADD": {0x01, 0},
"OR": {0x09, 1},
"ADC": {0x11, 2},
"SBB": {0x19, 3},
"AND": {0x21, 4},
"SUB": {0x29, 5},
"XOR": {0x31, 6},
"CMP": {0x39, 7},
}
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
// mode); NEG/NOT use the 0xF6/0xF7 group.
// unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode.
// INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes
// in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4,
// DIV /6, IDIV /7; the accumulator is the implicit other operand).
var unaryOp = map[string]struct {
digit int
op byte
@@ -31,13 +34,50 @@ var unaryOp = map[string]struct {
"DEC": {1, 0xFF},
"NOT": {2, 0xF7},
"NEG": {3, 0xF7},
"MUL": {4, 0xF7},
"DIV": {6, 0xF7},
"IDIV": {7, 0xF7},
}
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
// shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the
// 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4).
var shiftOp = map[string]int{
"SHL": 4,
"SAL": 4,
"SHR": 5,
"SAR": 7,
"ROL": 0,
"ROR": 1,
"RCL": 2,
"RCR": 3,
}
// bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form;
// the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's
// opcode row.
var bitTestOp = map[string]int{
"BT": 4,
"BTS": 5,
"BTR": 6,
"BTC": 7,
}
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
// fence names carry their opcode inside the 0F AE /digit group spelled out in
// full (E8/F0/F8), and PAUSE is F3 90.
var noOperandTable = map[string][]byte{
"CPUID": {0x0F, 0xA2},
"RDTSC": {0x0F, 0x31},
"RDTSCP": {0x0F, 0x01, 0xF9},
"SYSCALL": {0x0F, 0x05},
"XGETBV": {0x0F, 0x01, 0xD0},
"CLD": {0xFC},
"STD": {0xFD},
"PAUSE": {0xF3, 0x90},
"LFENCE": {0x0F, 0xAE, 0xE8},
"MFENCE": {0x0F, 0xAE, 0xF0},
"SFENCE": {0x0F, 0xAE, 0xF8},
"UNDEF": {0x0F, 0x0B},
}
// --- MOV --------------------------------------------------------------------
@@ -309,6 +349,13 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if err != nil {
return err
}
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
// the destination is AL, the form the Go assembler prefers here.
if r, ok := dst.(Reg); ok && r.idx == 0 {
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
i.imm = immBytes
return e.emit(i)
}
i := newInstr(1, []byte{0x80})
if err := setRMDigit(i, digit, dst, 1); err != nil {
return err
@@ -913,6 +960,7 @@ type sseMove struct {
var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
"MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
@@ -1000,10 +1048,120 @@ var sseBinTable = map[string]sseBin{
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
"PCMPEQD": {0x66, 0x76, false},
"PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false},
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
"PCMPGTD": {0x66, 0x66, false},
"PSHUFB": {0x66, 0x00, true},
// Scalar compares and square root, packed adds/subtracts and the byte
// unpack, the spellings the Plan 9 table uses (COMISD orders the
// operands like every other two-operand form).
"ANDNPD": {0x66, 0x55, false},
"ANDNPS": {0x00, 0x55, false},
"COMISD": {0x66, 0x2F, false},
"SQRTSD": {0xF2, 0x51, false},
"PADDL": {0x66, 0xFE, false},
"PSUBL": {0x66, 0xFA, false},
"PUNPCKLBW": {0x66, 0x60, false},
// AES round functions (66 0F38) and the SHA message schedule helpers
// (no prefix, 0F38).
"AESENC": {0x66, 0xDC, true},
"AESENCLAST": {0x66, 0xDD, true},
"AESDEC": {0x66, 0xDE, true},
"AESDECLAST": {0x66, 0xDF, true},
"AESIMC": {0x66, 0xDB, true},
"SHA1MSG1": {0x00, 0xC9, true},
"SHA1MSG2": {0x00, 0xCA, true},
"SHA1NEXTE": {0x00, 0xC8, true},
"SHA256MSG1": {0x00, 0xCC, true},
"SHA256MSG2": {0x00, 0xCD, true},
}
// sseImm3 describes a legacy SSE instruction taking a leading imm8 and two
// further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and
// map3A select the opcode map the same way as sseBin's.
type sseImm3 struct {
prefix byte
op byte
map3A bool // opcode lives under 0F3A instead of 0F38
}
// sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3
// align/blend shuffles, the string compare, carry-less multiply and the AES
// key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings.
var sseImm3Table = map[string]sseImm3{
"PALIGNR": {0x66, 0x0F, true},
"PBLENDW": {0x66, 0x0E, true},
"PCMPESTRI": {0x66, 0x61, true},
"PCLMULQDQ": {0x66, 0x44, true},
"AESKEYGENASSIST": {0x66, 0xDF, true},
"SHA1RNDS4": {0x00, 0xCC, true},
}
// sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM
// source and rm = the destination (GPR or memory). PEXTRW's GPR destination
// uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one,
// so it carries both opcodes.
type sseExtract struct {
op []byte
opMem []byte // used when the destination is memory; nil shares op
rexW bool // PEXTRQ's REX.W
}
var sseExtractTable = map[string]sseExtract{
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
}
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
// destination and rm = the source (GPR or memory).
type sseInsert struct {
op []byte
rexW bool // PINSRQ's REX.W
}
var sseInsertTable = map[string]sseInsert{
"PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false},
"PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false},
"PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true},
"PINSRW": {[]byte{0x0F, 0xC4}, false},
}
// sseShiftImm maps the legacy packed integer shifts' immediate form:
// OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L
// (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ.
var sseShiftImm = map[string]sseShift{
"PSLLW": {0x71, 6},
"PSRLW": {0x71, 2},
"PSRAW": {0x71, 4},
"PSLLL": {0x72, 6},
"PSRLL": {0x72, 2},
"PSRAL": {0x72, 4},
"PSLLQ": {0x73, 6},
"PSRLQ": {0x73, 2},
"PSLLDQ": {0x73, 7},
"PSRLDQ": {0x73, 3},
}
// sseShiftVar maps the variable-count forms (the count comes from an XMM
// register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no
// variable form.
var sseShiftVar = map[string]byte{
"PSLLW": 0xF1,
"PSRLW": 0xD1,
"PSRAW": 0xE1,
"PSLLL": 0xF2,
"PSRLL": 0xD2,
"PSRAL": 0xE2,
"PSLLQ": 0xF3,
"PSRLQ": 0xD3,
}
// sseShift is one /digit selector in the 0F 71/72/73 immediate group.
type sseShift struct {
op byte
digit int
}
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
@@ -1016,6 +1174,7 @@ type sseShuf struct {
var sseShufTable = map[string]sseShuf{
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
"PSHUFL": {0x66, 0x70},
}
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
@@ -1090,3 +1249,465 @@ func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
}
return e.emit(i)
}
// --- carry, bit test, exchange and accumulate -------------------------------
// encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9
// order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with
// reg = index, rm = target); an immediate index uses 0F BA /digit with imm8.
func (e *enc) encodeBitTest(name string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
digit := bitTestOp[name]
index, target := ops[0], ops[1]
if reg, ok := index.(Reg); ok {
// Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC),
// the /digit base plus eight per step.
i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3})
if err := setRM(i, reg, target, size); err != nil {
return err
}
return e.emit(i)
}
imm, ok := index.(Imm)
if !ok {
return fmt.Errorf("%s index must be a register or an immediate", name)
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
i := newInstr(size, []byte{0x0F, 0xBA})
if err := setRMDigit(i, digit, target, size); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeExchange encodes XCHG. A register-to-register exchange where either
// operand is AX uses the 0x90+r accumulator form (with REX.W for the quad
// form, as the Go assembler emits it); everything else uses 0x86/0x87 with
// the register operand in ModRM.reg, the memory (or second register) in r/m.
func (e *enc) encodeExchange(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsReg := src.(Reg)
dstReg, dstIsReg := dst.(Reg)
if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) {
// 0x90+r: r is the non-AX register, whichever side it sits on.
r := dstReg
if srcReg.idx == 0 {
r = dstReg
} else {
r = srcReg
}
i := newInstr(size, []byte{0x90 + byte(r.idx&7)})
i.rexB = r.idx >= 8
return e.emit(i)
}
op := byte(0x87)
if size == 1 {
op = 0x86
}
switch {
case srcIsReg:
i := newInstr(size, []byte{op})
if err := setRM(i, srcReg, dst, size); err != nil {
return err
}
return e.emit(i)
case dstIsReg:
i := newInstr(size, []byte{op})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("XCHG: at least one operand must be a register")
}
// encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG
// (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the
// destination writable (register or memory).
func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok {
return fmt.Errorf("%s source must be a register", name)
}
opc := op
if size == 1 {
opc = op8
}
i := newInstr(size, []byte{0x0F, opc})
if err := setRM(i, srcReg, ops[1], size); err != nil {
return err
}
return e.emit(i)
}
// encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for
// the rest; the word form carries a 0x66 operand-size prefix (66 F2, the
// prefix order the Go assembler emits) and the quad form REX.W. reg = GPR
// accumulator, rm = the data source.
func (e *enc) encodeCrc32(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("CRC32 destination must be a general register")
}
i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1}
if size > 1 {
i.opcode[2] = 0xF1
}
i.rexW = size == 8
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg =
// destination, rm = source, the carry/overflow flag as the carry-in.
func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("ADCX/ADOX destination must be a general register")
}
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8}
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// --- string primitives, flags and INT ----------------------------------------
// encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and
// STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or
// REX.W prefix.
func (e *enc) encodeStringOp(base string, ops []Operand, size int) error {
if len(ops) != 0 {
return fmt.Errorf("%s takes no operands, got %d", base, len(ops))
}
var op byte
switch base {
case "MOVS":
op = 0xA5
if size == 1 {
op = 0xA4
}
case "STOS":
op = 0xAB
if size == 1 {
op = 0xAA
}
default:
return fmt.Errorf("unsupported string instruction %q", base)
}
return e.emit(newInstr(size, []byte{op}))
}
// encodeInt encodes INT with its single imm8 operand. The field takes the
// low byte silently inside the 32-bit span, matching the scalar convention
// (go tool asm encodes INT $256 as CD 00).
func (e *enc) encodeInt(ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("INT expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("INT operand must be an immediate")
}
if imm < -(1<<31) || imm > (1<<32)-1 {
return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm))
}
return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}})
}
// encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a
// single 32-bit memory operand.
func (e *enc) encodeMxcsr(digit int, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops))
}
m, ok := ops[0].(Mem)
if !ok {
return fmt.Errorf("MXCSR instruction requires a memory operand")
}
i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1}
if err := setMem(i, digit, m); err != nil {
return err
}
return e.emit(i)
}
// cvtIntOp maps the scalar float-to-integer conversions to their mandatory
// prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating
// CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width.
var cvtIntOp = map[string]struct {
prefix byte
op byte
}{
"CVTSD2S": {0xF2, 0x2D},
"CVTTSD2S": {0xF2, 0x2C},
"CVTSS2S": {0xF3, 0x2D},
"CVTTSS2S": {0xF3, 0x2C},
}
// encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C
// with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the
// quad spellings.
func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
}
spec := cvtIntOp[base]
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("%s destination must be a general register", base)
}
i := newInstr(size, []byte{0x0F, spec.op})
i.prefix = spec.prefix
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
// encodeFmov encodes the x87 double move. The memory forms are DD /0
// (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register
// move is DD C0+dst (FLD st(dst)), the form the Go assembler emits.
func (e *enc) encodeFmov(ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsF := src.(Reg)
dstReg, dstIsF := dst.(Reg)
srcF := srcIsF && srcReg.fp
dstF := dstIsF && dstReg.fp
switch {
case srcF && dstF:
// The register form is DD /2 with rm = the destination (FST st(dst)).
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setRMDigit(i, 2, dstReg, 8); err != nil {
return err
}
return e.emit(i)
case dstF:
m, ok := src.(Mem)
if !ok {
return fmt.Errorf("FMOVD: invalid source operand")
}
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setMem(i, 0, m); err != nil {
return err
}
return e.emit(i)
case srcF:
m, ok := dst.(Mem)
if !ok {
return fmt.Errorf("FMOVD: invalid destination operand")
}
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setMem(i, 2, m); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("FMOVD needs an x87 register operand")
}
// --- legacy SSE imm8, extract, insert and packed shift families --------------
// encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src,
// dst with reg = dst, rm = src and the immediate appended last (PALIGNR,
// PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4).
func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("SSE imm8 instruction needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
src, dst := ops[1], ops[2]
dstReg, ok2 := dst.(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("SSE imm8 instruction destination must be a vector register")
}
opcode := []byte{0x0F, 0x38, m.op}
if m.map3A {
opcode = []byte{0x0F, 0x3A, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the
// XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and
// PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the
// SSE4.1 0F3A 15 one).
func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("extract needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
srcReg, srcVec := vecReg(ops[1])
if !srcVec {
return fmt.Errorf("extract source must be an XMM register")
}
opcode := m.op
if m.opMem != nil && memOperand(ops[2]) {
opcode = m.opMem
}
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
if err := setRM(i, srcReg, ops[2], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the
// XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and
// PINSRW).
func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("insert needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
dstReg, dstVec := vecReg(ops[2])
if !dstVec {
return fmt.Errorf("insert destination must be an XMM register")
}
i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW}
if err := setRM(i, dstReg, ops[1], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEShift encodes the legacy packed integer shifts. The immediate
// form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form
// OP count, dst carries the count in an XMM register (or memory) on the
// 66 0F D1-F3 opcodes. The destination is always the register written.
func (e *enc) encodeSSEShift(name string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s destination must be the second, vector operand", name)
}
if imm, isImm := ops[0].(Imm); isImm {
spec := sseShiftImm[name]
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
if !vecOrMem(ops[0]) {
return fmt.Errorf("%s count must be an immediate, a vector register or memory", name)
}
op, ok := sseShiftVar[name]
if !ok {
return fmt.Errorf("%s has no variable-count form", name)
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
return e.emit(i)
}
// encodeCmpsd encodes CMPSD, the scalar double compare with its predicate
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
// F2 0F C2 with reg = dst, rm = src.
func (e *enc) encodeCmpsd(ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
}
imm, ok := ops[2].(Imm)
if !ok {
return fmt.Errorf("CMPSD predicate must be an immediate")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("CMPSD destination must be a vector register")
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the
// literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no
// prefix, reg = dst, rm = src; X0 is implicit on the wire).
func (e *enc) encodeSha256rnds2(ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops))
}
x0, ok := ops[0].(Reg)
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
return fmt.Errorf("SHA256RNDS2 first operand must be X0")
}
dstReg, ok2 := ops[2].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("SHA256RNDS2 destination must be a vector register")
}
i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[1], 8); err != nil {
return err
}
return e.emit(i)
}
+6 -1
View File
@@ -17,12 +17,13 @@ import "strings"
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0-K7.
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register
}
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
@@ -144,6 +145,10 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ {
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
}
// x87 stack: F0..F7.
for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
}
return m
}
+144 -1
View File
@@ -41,7 +41,7 @@ const (
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
@@ -52,6 +52,15 @@ const (
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
// vexZeroAll is the no-operand form that zeroes the full upper state
// (VZEROALL, the L = 1 twin of VZEROUPPER).
vexZeroAll
// vexNDS3GPR is the three-operand NDS form over general-purpose
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
vexNDS3GPR
// vexImmRMGPR is the immediate form over general-purpose registers
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
vexImmRMGPR
)
// vexSpec describes one VEX instruction's encoding parameters.
@@ -125,6 +134,12 @@ var vexTable = map[string]vexSpec{
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// Scalar fused multiply-add (NDS form). The Go assembler carries the
// same 66 prefix as the packed forms on every FMA row, and W1 on the
// double-precision spellings, so SD shares PD's prefix/W pair and the
// scalar width rides on the W bit.
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
@@ -192,6 +207,31 @@ var vexTable = map[string]vexSpec{
// VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
// falls through to the EVEX table.
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
@@ -200,6 +240,14 @@ var vexTable = map[string]vexSpec{
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
// YMM (the encoder rejects an XMM destination, as go tool asm does).
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
// memory destination in rm).
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
@@ -290,6 +338,8 @@ type vexMoveSpec struct {
var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, aligned integer move.
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
@@ -324,6 +374,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
}
}
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
// only; an XMM destination is rejected exactly as go tool asm does.
if mnemUpper == "VBROADCASTI128" {
dstReg, ok := ops[len(ops)-1].(Reg)
if len(ops) != 2 || !ok || dstReg.size != 32 {
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
}
}
if ms, ok := vexMoveTable[mnemUpper]; ok {
return e.encodeVexMove(mnemUpper, ms, ops)
}
@@ -356,6 +414,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
case vexZeroAll:
return e.encodeVexZeroAll(mnemUpper, spec, ops)
case vexNDS3GPR:
return e.encodeVexNDS3GPR(spec, ops)
case vexImmRMGPR:
return e.encodeVexImmRMGPR(spec, ops)
case vexRMRev:
return e.encodeVexRMRev(spec, ops)
}
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
}
@@ -607,6 +673,83 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
return nil
}
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
// twin of VZEROUPPER.
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 0 {
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
}
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
return nil
}
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
// rm = src2 and L = 0.
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
vvvvReg, ok := src1.(Reg)
if !ok || vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
}
// encodeVexImmRMGPR encodes the immediate form over general-purpose
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
// a store with no register-destination form).
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("store source must be a vector register")
}
if !memOperand(ops[1]) {
return fmt.Errorf("store destination must be memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
+59
View File
@@ -19,6 +19,20 @@ func vreg(t *testing.T, name string) Reg {
return r
}
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
// go tool asm in the ground-truth tests instead.
var x86asmUnrecognised = map[string]bool{
"ANDNL": true,
"ANDNQ": true,
"MULXL": true,
"MULXQ": true,
"RORXL": true,
"RORXQ": true,
"VFMADD213SD": true,
"VFNMADD231SD": true,
}
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
// instruction and verifies it round-trips through the x86 decoder to the same
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
@@ -37,8 +51,15 @@ func TestVexNDS3(t *testing.T) {
t.Errorf("%s: Encode: %v", mnem, err)
continue
}
// The x86 decoder's table lacks a handful of rows the Go assembler
// emits (the scalar 213/231 FMA spellings among them); those are
// pinned byte for byte against go tool asm in TestVexGroundTruth
// instead of round-tripped here.
inst, err := x86asm.Decode(code, 64)
if err != nil {
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
continue
}
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
continue
}
@@ -184,6 +205,38 @@ func TestVexGroundTruth(t *testing.T) {
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
// Packed single XOR and byte compare (NDS form).
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
// Octa byte shifts (vvvv carries the destination).
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
// Two-operand test and the non-temporal and broadcast stores.
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
// Aligned integer move and the full zeroing form.
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
// BMI1/BMI2 general-register VEX forms.
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
// Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
@@ -287,6 +340,12 @@ func TestVexGroundTruth(t *testing.T) {
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
// The decoder's AVX/BMI table lacks a few rows the Go
// assembler emits (the GPR VEX forms and the scalar FMA
// spellings); their bytes are the ground truth here.
if x86asmUnrecognised[c.mnem] {
continue
}
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
continue
}
+69
View File
@@ -0,0 +1,69 @@
// Atomics and carry-extending multi-word arithmetic: exchange,
// compare-exchange, exchange-add, ADCX/ADOX and the CRC-32 accumulator
// family. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func xchg(p *uint64, v uint64) uint64
TEXT ·xchg(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
XCHGQ BX, (AX)
XCHGQ BX, CX
XCHGL BX, CX
XCHGW BX, CX
XCHGB BL, CL
MOVQ AX, ret+16(FP)
RET
// func cmpxchg(p *uint64, old, new uint64) uint8
TEXT ·cmpxchg(SB), NOSPLIT, $0-25
MOVQ p+0(FP), AX
MOVQ old+8(FP), BX
MOVQ new+16(FP), CX
CMPXCHGQ CX, (AX)
CMPXCHGL CX, BX
CMPXCHGW CX, BX
CMPXCHGB CL, BL
SETEQ AL
MOVB AL, ret+24(FP)
RET
// func xadd(p *uint64, v uint64) uint64
TEXT ·xadd(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
XADDQ BX, (AX)
XADDL BX, CX
XADDW BX, CX
XADDB BL, CL
MOVQ AX, ret+16(FP)
RET
// func adcx_adox(lo, hi, x, y uint64) uint64
TEXT ·adcx_adox(SB), NOSPLIT, $0-40
MOVQ lo+0(FP), AX
MOVQ hi+8(FP), DX
MOVQ x+16(FP), BX
MOVQ y+24(FP), CX
ADCXQ BX, AX
ADOXQ CX, DX
ADCXL BX, AX
ADOXL CX, DX
XORQ BX, BX
ADCXQ BX, AX
MOVQ AX, ret+32(FP)
RET
// func crc32(crc uint32, p *byte, n int) uint32
TEXT ·crc32(SB), NOSPLIT, $0-28
MOVL crc+0(FP), AX
MOVQ p+8(FP), SI
MOVQ n+16(FP), CX
CRC32B (SI), AX
CRC32Q (SI), CX
CRC32L (SI), AX
MOVW (SI), DX
CRC32W DX, AX
MOVL AX, ret+24(FP)
RET
+102
View File
@@ -0,0 +1,102 @@
// The AVX/AVX-512 gap families: fused scalar multiply-add, carries through
// GF(2^8) affine transforms, population counts, non-temporal stores, mask
// moves and the KMOV widths. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func avxblend(a, b []float64) float64
TEXT ·avxblend(SB), NOSPLIT, $0-56
MOVQ a_base+0(FP), SI
MOVQ b_base+24(FP), DI
VMOVUPD (SI), Y0
VMOVUPD (DI), Y1
VXORPS Y2, Y2, Y2
VSHUFPD $5, Y0, Y1, Y3
VMOVUPD Y3, (SI)
VPBLENDD $3, Y0, Y1, Y4
VPERM2F128 $1, Y4, Y0, Y0
VEXTRACTF128 $1, Y0, X1
VZEROALL
VMOVSD X1, ret+48(FP)
RET
// func avxint(p *byte, n int) uint64
TEXT ·avxint(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
VMOVDQU (SI), Y0
VPCMPEQB Y0, Y0, Y1
VPSLLDQ $2, X0, X0
VPSRLDQ $4, Y0, Y0
VPALIGNR $3, X0, X1, X1
VPCLMULQDQ $0, X0, X1, X2
VGF2P8AFFINEQB $7, X2, X0, X3
VPOPCNTB X3, X4
VPOPCNTD Y0, Y5
VPERMI2B X0, X1, X2
VPTEST X0, X0
VPMOVMSKB X1, AX
VZEROUPPER
MOVQ AX, ret+16(FP)
RET
// func avxnt(p *float64)
TEXT ·avxnt(SB), NOSPLIT, $0-8
MOVQ p+0(FP), DI
VMOVUPD (DI), Y0
VADDPD Y0, Y0, Y0
VMOVNTDQ Y0, (DI)
VMOVNTDQ X0, 16(DI)
VZEROALL
RET
// func avxmas(a, b []float64) float64
TEXT ·avxmas(SB), NOSPLIT, $0-56
MOVQ a_base+0(FP), SI
MOVQ b_base+24(FP), DI
VMOVSD (SI), X0
VMOVSD (DI), X1
VFMADD213SD X1, X0, X0
VFNMADD231SD X1, X0, X0
VADDSD X1, X0, X0
VMOVSD X0, ret+48(FP)
RET
// func avxgpr(x, y uint64) uint64
TEXT ·avxgpr(SB), NOSPLIT, $0-24
MOVQ x+0(FP), AX
MOVQ y+8(FP), BX
ANDNL BX, AX, CX
MULXQ BX, DX, SI
RORXL $3, AX, CX
RORXQ $7, BX, SI
MOVQ CX, ret+16(FP)
RET
// func avxmask(kin uint8, p *byte) uint8
TEXT ·avxmask(SB), NOSPLIT, $0-17
MOVQ p+8(FP), SI
KMOVB kin+0(FP), K1
KMOVB K1, K2
KMOVW K2, K1
KMOVD K1, K3
KMOVQ K3, K4
KMOVB K4, K1
KMOVB K1, AX
KMOVD K1, (SI)
MOVB AL, ret+8(FP)
RET
// func avx512(p *uint64, n int) uint64
TEXT ·avx512(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
VMOVDQU64 (SI), Z0
VPORQ Z0, Z0, Z1
VPOPCNTQ Z1, Z2
VPERMB Z1, Z0, Z2
VPXORD Z2, Z1, Z0
VMOVDQA64 Z0, (SI)
VZEROUPPER
XORQ AX, AX
MOVQ AX, ret+16(FP)
RET
+57
View File
@@ -0,0 +1,57 @@
// The AES-NI, SHA and carry-less multiply round instructions as GOROOT's
// crypto kernels spell them. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func aesround(blk, rk *byte)
TEXT ·aesround(SB), NOSPLIT, $0-16
MOVQ blk+0(FP), SI
MOVQ rk+8(FP), DI
MOVOU (SI), X0
MOVOU (DI), X1
AESENC X1, X0
AESENCLAST X1, X0
AESDEC X1, X0
AESDECLAST X1, X0
AESIMC X1, X2
AESKEYGENASSIST $1, X1, X3
MOVOU X0, (SI)
MOVOU X2, (DI)
RET
// func sha1block(p *byte, n int, h *[5]uint32)
TEXT ·sha1block(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ h+16(FP), DI
MOVOU (SI), X0
MOVOU 16(SI), X1
SHA1RNDS4 $0, X1, X0
SHA1NEXTE X1, X0
SHA1MSG1 X1, X2
SHA1MSG2 X1, X2
MOVOU X0, (DI)
RET
// func sha256block(p *byte, n int, h *[8]uint32)
TEXT ·sha256block(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ h+16(FP), DI
MOVOU (SI), X0
MOVOU 16(SI), X1
SHA256RNDS2 X0, X1, X0
SHA256MSG1 X1, X2
SHA256MSG2 X1, X2
MOVOU X0, (DI)
RET
// func pclmul(a, b *byte)
TEXT ·pclmul(SB), NOSPLIT, $0-16
MOVQ a+0(FP), SI
MOVQ b+8(FP), DI
MOVOU (SI), X0
MOVOU (DI), X1
PCLMULQDQ $0, X1, X0
PCLMULQDQ $17, (DI), X0
MOVOU X0, (SI)
RET
+76
View File
@@ -0,0 +1,76 @@
// Carry arithmetic, rotates, unsigned/signed division and bit tests: the
// scalar families GOROOT's big-number and crypto kernels use. Every result
// is folded back so no instruction is dead.
#include "textflag.h"
// func carry(a, b uint64) uint64
TEXT ·carry(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
ADCQ $0, AX
MOVQ BX, CX
SBBQ $1, CX
ADCL BX, AX
ADCB AL, BL
ADCW $7, CX
MOVQ AX, ret+16(FP)
RET
// func borrow(a, b uint64) uint64
TEXT ·borrow(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
SUBQ BX, AX
SBBQ $0, AX
SBBQ BX, CX
MOVQ AX, ret+16(FP)
RET
// func rot(x uint64, n uint32) uint64
TEXT ·rot(SB), NOSPLIT, $0-24
MOVQ x+0(FP), AX
MOVL n+8(FP), CX
ROLQ CL, AX
RORQ $7, AX
ROLL $1, AX
RORL CL, AX
RCLQ $1, AX
RCRQ CL, AX
ROLW $3, AX
SALQ $2, AX
SALB $1, AX
MOVQ AX, ret+8(FP)
RET
// func muldiv(a, b uint64) uint64
TEXT ·muldiv(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
MULQ BX
MULQ (BX)
MOVL (BX), CX
MULL CX
DIVQ BX
IDIVQ BX
MOVL a+0(FP), AX
DIVL CX
IDIVL CX
MOVQ AX, ret+16(FP)
RET
// func bitfield(w *uint64) uint64
TEXT ·bitfield(SB), NOSPLIT, $0-16
MOVQ (DI), AX
MOVQ (DI), CX
BTQ AX, CX
BTQ $3, (DI)
BTL AX, CX
BTW $1, CX
BTSQ $5, AX
BTRQ AX, CX
BTCQ $7, (DI)
SETCS AL
MOVQ AX, ret+8(FP)
RET
+77
View File
@@ -0,0 +1,77 @@
// The legacy SSE gap families: scalar compares and square roots, the Plan 9
// packed spellings, shuffles, lane extracts and inserts, packed integer
// shifts and the octa moves. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func cmporder(a, b *float64) int
TEXT ·cmporder(SB), NOSPLIT, $0-24
MOVQ a+0(FP), SI
MOVQ b+8(FP), DI
MOVSD (SI), X0
MOVSD (DI), X1
ANDNPD X0, X2
ANDNPS X0, X3
COMISD X0, X1
SQRTSD X0, X2
CMPSD X0, X1, $5
MOVL SI, CX
SETPL CL
MOVL CX, ret+16(FP)
RET
// func packed(w *uint64) uint64
TEXT ·packed(SB), NOSPLIT, $0-16
MOVQ w+0(FP), SI
MOVO (SI), X0
MOVOA (SI), X1
PADDL X0, X1
PSUBL X0, X1
PCMPEQL X0, X1
PUNPCKLBW X0, X1
PSHUFL $27, X0, X2
MOVOU X2, (SI)
MOVQ (SI), AX
MOVQ AX, ret+8(FP)
RET
// func lanes(p *byte, buf *byte)
TEXT ·lanes(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
MOVQ buf+8(FP), DI
MOVO (SI), X0
MOVQ SI, AX
PINSRB $1, AX, X0
PINSRW $2, AX, X0
PINSRD $3, AX, X0
PINSRQ $1, AX, X0
PEXTRB $1, X0, AX
PEXTRW $2, X0, AX
PEXTRD $3, X0, AX
PEXTRQ $1, X0, CX
PCMPESTRI $4, X0, X0
MOVB AL, (DI)
MOVOU X0, (SI)
RET
// func shifts(p *uint64)
TEXT ·shifts(SB), NOSPLIT, $0-8
MOVQ p+0(FP), SI
MOVO (SI), X0
MOVO X0, X1
PSLLW $3, X0
PSRLW $1, X1
PSRAW $2, X0
PSLLL $4, X0
PSRLL $5, X1
PSRAL $1, X0
PSLLQ $7, X0
PSRLQ $9, X1
PSLLL X1, X0
PSRLQ X0, X1
PSLLDQ $2, X0
PSRLDQ $4, X1
MOVOU X0, (SI)
MOVOU X1, 16(SI)
RET
+76
View File
@@ -0,0 +1,76 @@
// System, string-primitive and x87 families: flag register moves, the
// serialising instructions, MOVS/STOS, the MXCSR pair, scalar float-to-int
// conversions and FMOVD. Every result is folded back so no instruction is
// dead.
#include "textflag.h"
// func system(x uint64) uint64
TEXT ·system(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
PUSHFQ
POPFQ
CPUID
RDTSC
RDTSCP
SYSCALL
XGETBV
PAUSE
LFENCE
MFENCE
SFENCE
UNDEF
XORQ AX, BX
MOVQ BX, ret+8(FP)
RET
// func stringprim(p *byte, n int) uint64
TEXT ·stringprim(SB), NOSPLIT, $0-24
MOVQ p+0(FP), DI
MOVQ n+8(FP), CX
LEAQ buf<>(SB), AX
MOVQ AX, SI
CLD
MOVSB
MOVSW
MOVSL
MOVSQ
STOSB
STOSQ
STOSL
STOSW
MOVQ DI, ret+16(FP)
RET
DATA buf<>+0x00(SB)/8, $0
GLOBL buf<>(SB), NOPTR, $8
// func intgate(x uint64) uint64
TEXT ·intgate(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
INT $3
MOVQ AX, ret+8(FP)
RET
// func fpmxcsr(x float64, csr *uint32) int64
TEXT ·fpmxcsr(SB), NOSPLIT, $0-24
MOVQ x+0(FP), X0
MOVQ csr+8(FP), AX
STMXCSR (AX)
LDMXCSR (AX)
CVTSD2SL X0, CX
CVTTSD2SQ X0, DX
MOVL (AX), SI
MOVQ SI, ret+8(FP)
RET
// func fmove(p *float64) float64
TEXT ·fmove(SB), NOSPLIT, $0-16
MOVQ p+0(FP), AX
FMOVD (AX), F0
FMOVD F0, F1
FMOVD F0, (AX)
MOVQ (AX), AX
MOVQ AX, ret+8(FP)
RET