diff --git a/asm/evex.go b/asm/evex.go index b0b099f..ae2ffb0 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -11,9 +11,11 @@ import ( // This file implements EVEX (AVX-512) instruction encoding: the four-byte // EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the // compressed disp8×N displacement, and the operand shapes the go-flac -// AVX-512 kernels use. Masking ({k}) and zeroing ({z}) are not supported — -// the kernels do not use them. K-register operands (mask destinations, -// KMOVW, KTESTW) are. +// AVX-512 kernels use plus the common floating-point and conversion set. +// Masking follows the Go assembler's spelling: an explicit K1–K7 operand +// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for +// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are +// supported too. // evexSpec describes one EVEX instruction's encoding parameters. The form // field reuses the vexForm shapes, which carry over unchanged. @@ -49,6 +51,31 @@ var evexTable = map[string]evexSpec{ // EVEX.128/256/512.66.0F.W1 — packed double arithmetic. "VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + // EVEX.128/256/512.66.0F.W1 — packed double unpack. + "VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + + // EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with + // an F2 pp; the EVEX forms exist for masked and zeroing use). The + // memory operand is a single double, so disp8×N = 8. + "VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, + "VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, + "VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, + "VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, + "VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, + "VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, + + // EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4). + "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, + "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, + "VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, + "VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, + "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, + "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, // EVEX.512.66.0F3A — align (NDS + imm8). "VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, @@ -62,6 +89,36 @@ var evexTable = map[string]evexSpec{ // EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst, // rm=src, no vvvv). "VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, + // EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst, + // rm=src, no vvvv): a 128-bit destination reads a single double from + // memory (disp8×8), the wider ones read the full operand. + "VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}}, + // EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst, + // rm=src, no vvvv, no mandatory prefix — as in the VEX form). + "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}}, + // EVEX.128/256/512.0F.W0 — packed single to packed double: the + // destination is twice the source width and sets the length; disp8×N + // follows the narrow memory source. No F3 prefix: the Go assembler + // emits this instruction with pp = 00 (Intel's maps would call that + // undefined) and gasm reproduces the Go assembler's bytes — its machine + // code is the oracle, not the manual. + "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, + // EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX + // form of the VEX instruction; the destination sets the length, disp8×N + // follows the narrow memory source). + "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, + // EVEX packed double → dword conversions: the source is the wide + // operand and the mnemonic fixes the length — the bare names are + // 512-bit only (ZMM source, XMM destination), the X/Y spellings are + // EVEX-128/256. Exactly one slot of n is valid; it names the vector + // length (and the disp8×N multiplier) a register or memory source + // encodes. + "VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}}, + "VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}}, + "VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}}, + "VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}}, + "VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}}, + "VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}}, // EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // operand is the narrow source, so disp8×N follows its size (8/16/32 for // the xmm/ymm/zmm destination lengths). @@ -292,6 +349,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error { return e.encodeEvexNDS3Imm(spec, ops, mask, zeroing) case vexExtract: return e.encodeEvexExtract(spec, ops, mask, zeroing) + case vexRMSrcLen: + return e.encodeEvexRMSrcLen(spec, ops, mask, zeroing) } return fmt.Errorf("unhandled EVEX form for %s", mnemUpper) } @@ -487,6 +546,46 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, zeroing) } +// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with +// the destination always XMM and the length fixed by the mnemonic — the +// single valid slot of spec.n names the vector length (and the disp8×N +// multiplier) a register or memory source encodes. +func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing bool) error { + if len(ops) != 2 { + return fmt.Errorf("conversion expects 2 operands, got %d", len(ops)) + } + src, dst := ops[0], ops[1] + dstReg, ok := dst.(Reg) + if !ok || !dstReg.isVec() { + return fmt.Errorf("EVEX destination must be a vector register") + } + ll, err := soleLen(spec.n) + if err != nil { + return err + } + return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing) +} + +// soleLen returns the vector-length index of the single valid slot of n — +// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes +// regardless of its operands. +func soleLen(n [3]int) (int, error) { + ll := -1 + for i, v := range n { + if v == 0 { + continue + } + if ll >= 0 { + return 0, fmt.Errorf("ambiguous vector-length table %v", n) + } + ll = i + } + if ll < 0 { + return 0, fmt.Errorf("empty vector-length table") + } + return ll, nil +} + // memOperand reports whether op is a memory reference (including a // static-symbol reference). func memOperand(op Operand) bool { diff --git a/asm/evex_test.go b/asm/evex_test.go index 40cb84b..39327bb 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -101,6 +101,32 @@ func TestEvexGroundTruth(t *testing.T) { {"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"}, // Register indices 16–31 exist only in EVEX encodings. {"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"}, + // Packed double arithmetic / unpack (EVEX forms carry W=1). + {"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"}, + {"VDIVPD Z4,Z5,Z6", "VDIVPD", []Operand{vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "Z6")}, "62f1d5485ef4"}, + {"VMINPD Z7,Z8,Z9", "VMINPD", []Operand{vreg(t, "Z7"), vreg(t, "Z8"), vreg(t, "Z9")}, "6271bd485dcf"}, + {"VMAXPD Z10,Z11,Z12", "VMAXPD", []Operand{vreg(t, "Z10"), vreg(t, "Z11"), vreg(t, "Z12")}, "6251a5485fe2"}, + {"VUNPCKLPD Z1,Z2,Z3", "VUNPCKLPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4814d9"}, + {"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"}, + {"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"}, + {"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"}, + // VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and + // X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄). + {"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"}, + {"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"}, + {"VMOVDDUP X16,X17", "VMOVDDUP", []Operand{vreg(t, "X16"), vreg(t, "X17")}, "62a1ff0812c8"}, + // Conversions: DQ→PS, PS→PD (pp = 00, the Go assembler's choice), + // DQ→PD (the destination sets the length). + {"VCVTDQ2PS Z1,Z2", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"}, + {"VCVTPS2PD Y1,Z2", "VCVTPS2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17c485ad1"}, + {"VCVTPS2PD 32(AX),Z2", "VCVTPS2PD", []Operand{Ptr(AX, 32, 32), vreg(t, "Z2")}, "62f17c485a5001"}, + {"VCVTDQ2PD Y1,Z2", "VCVTDQ2PD", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17e48e6d1"}, + // PD→DQ conversions: the source is the wide operand and fixes the + // length (ZMM source → L'L = 10 even with an XMM destination; a + // memory source takes the length the mnemonic's spelling implies). + {"VCVTPD2DQ Z1,Y2", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff48e6d1"}, + {"VCVTPD2DQ 64(AX),Y2", "VCVTPD2DQ", []Operand{Ptr(AX, 64, 64), vreg(t, "Y2")}, "62f1ff48e65001"}, + {"VCVTTPD2DQ Z3,Y4", "VCVTTPD2DQ", []Operand{vreg(t, "Z3"), vreg(t, "Y4")}, "62f1fd48e6e3"}, } for _, c := range cases { want := strings.ReplaceAll(c.want, " ", "") @@ -157,6 +183,18 @@ func TestEvexMasking(t *testing.T) { {"VMOVDQU32 store", "VMOVDQU32", []Operand{vreg(t, "Z1"), vreg(t, "K4"), Ptr(DI, 0, 64)}, "62f17e4c7f0f"}, // Masked comparison with a K destination: dst K1, mask K2. {"VPCMPEQD k-dst+mask", "VPCMPEQD", []Operand{vreg(t, "Z0"), vreg(t, "Z3"), vreg(t, "K2"), vreg(t, "K1")}, "62f1654a76c8"}, + // Masked floating point: packed double, the scalar SD/SS forms (which + // exist under EVEX only for masked and zeroing use) and conversions. + {"VSUBPD.Z", "VSUBPD.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f1edcb5ce1"}, + {"VADDSD merge", "VADDSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3"), vreg(t, "X4")}, "62f1ef0b58e1"}, + {"VSUBSD.Z", "VSUBSD.Z", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5"), vreg(t, "X3")}, "62f1ef8d5cd9"}, + {"VADDSS merge", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f16e0958d9"}, + {"VCVTPD2DQ merge", "VCVTPD2DQ", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1ff4ae6d9"}, + {"VCVTTPD2DQ.Z", "VCVTTPD2DQ.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Y3")}, "62f1fdcae6d9"}, + {"VCVTDQ2PS.Z", "VCVTDQ2PS.Z", []Operand{vreg(t, "Z1"), vreg(t, "K4"), vreg(t, "Z2")}, "62f17ccc5bd1"}, + {"VCVTDQ2PD merge", "VCVTDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "K2"), vreg(t, "X3")}, "62f17e0ae6d9"}, + {"VCVTDQ2PD.Z", "VCVTDQ2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f17ecae6d1"}, + {"VCVTPS2PD.Z", "VCVTPS2PD.Z", []Operand{vreg(t, "Y1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f17ccb5ad1"}, } for _, c := range cases { code, err := Encode(c.mnem, c.ops...) diff --git a/asm/vex.go b/asm/vex.go index 2cdd9a0..2719756 100644 --- a/asm/vex.go +++ b/asm/vex.go @@ -43,6 +43,13 @@ const ( // in ModRM.reg and the destination in r/m — the layout of the EVEX // narrowing stores (VPMOVDW, VPMOVQD). vexRMRev + // vexRMSrcLen is the two-operand conversion form `OP src, dst` whose + // vector length follows the source: the packed-double → dword + // conversions (VCVTPD2DQ/VCVTTPD2DQ and their X/Y spellings) narrow into + // an XMM destination, so the L bit rides with the wider source. The + // mnemonic's spelling fixes the length (X = 128, Y = 256), which also + // covers a memory source. ModRM.reg = dst, ModRM.rm = src, no vvvv. + vexRMSrcLen // vexZero is the no-operand form (VZEROUPPER). vexZero ) @@ -86,12 +93,29 @@ var vexTable = map[string]vexSpec{ // VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic. "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, + "VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3}, + "VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3}, + "VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3}, + "VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3}, "VXORPD": {1, 0x57, 0, 1, -1, vexNDS3}, "VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3}, + "VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3}, // VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed // opcodes with an F2 pp). "VADDSD": {1, 0x58, 0, 3, -1, vexNDS3}, + "VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3}, "VMULSD": {1, 0x59, 0, 3, -1, vexNDS3}, + "VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3}, + "VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3}, + "VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3}, + // VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed + // opcodes with an F3 pp). + "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3}, + "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3}, + "VMULSS": {1, 0x59, 0, 2, -1, vexNDS3}, + "VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3}, + "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3}, + "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3}, // VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form). "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3}, @@ -105,6 +129,19 @@ var vexTable = map[string]vexSpec{ // VEX.128/256.F3.0F.WIG — signed dword to packed double conversion // (reg=dst, rm=src, no vvvv; the length follows the destination). "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM}, + // VEX.128/256.0F.WIG — signed dword to packed single conversion + // (reg=dst, rm=src, no vvvv, no mandatory prefix). + "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM}, + // VEX.128/256.0F.WIG — packed single to packed double conversion + // (reg=dst, rm=src; the destination is the wide operand and sets the + // length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but + // the Go assembler emits the instruction with pp = 00, and gasm follows + // the Go assembler's bytes — its machine code is the oracle, not the + // manual. + "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM}, + // VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane + // (reg=dst, rm=src, no vvvv; the length follows the destination). + "VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM}, // VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src). "VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM}, "VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD) @@ -138,6 +175,25 @@ var vexTable = map[string]vexSpec{ // VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src). "KTESTW": {1, 0x99, 0, 0, -1, vexRM}, + + // VEX.F2.0F — packed double to packed dword conversions, truncating and + // non-truncating. The destination is always XMM; the X/Y spellings fix + // the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen. + "VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, + "VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen}, + "VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, + "VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen}, +} + +// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of +// the packed-double → dword conversions) to its fixed vector length: +// X = 128 (L = 0), Y = 256 (L = 1). The spelling fixes the length even for +// a memory source, matching the Go assembler's ytab. +var vexSrcLen = map[string]int{ + "VCVTPD2DQX": 0, + "VCVTPD2DQY": 1, + "VCVTTPD2DQX": 0, + "VCVTTPD2DQY": 1, } // vexVarShift maps the shift mnemonics to their variable-count opcode — the @@ -230,6 +286,8 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error { return e.encodeVexNDS3Imm(spec, ops) case vexExtract: return e.encodeVexExtract(spec, ops) + case vexRMSrcLen: + return e.encodeVexRMSrcLen(mnemUpper, spec, ops) case vexZero: return e.encodeVexZero(mnemUpper, spec, ops) } @@ -295,6 +353,32 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error { return e.emitVexFields(spec, l, regField, rBit, 15, src) } +// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with +// the destination always XMM and the VEX.L bit following the source — fixed +// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when +// the source is memory. +func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error { + if len(ops) != 2 { + return fmt.Errorf("conversion expects 2 operands, got %d", len(ops)) + } + src, dst := ops[0], ops[1] + dstReg, ok := dst.(Reg) + if !ok || !dstReg.isVec() { + return fmt.Errorf("VEX destination must be a vector register") + } + ll, ok := vexSrcLen[mnem] + if !ok { + return fmt.Errorf("no fixed vector length for %s", mnem) + } + regField := dstReg.idx & 7 + rBit := 0 + if dstReg.idx >= 8 { + rBit = 1 + } + // An unused vvvv field must be stored as all ones (v̄vvv = 1111). + return e.emitVexFields(spec, ll, regField, rBit, 15, src) +} + // encodeVexShiftImm encodes an immediate-shift instruction: OP $imm, src, dst. // The destination is carried in VEX.vvvv, the source in ModRM.rm, and the // shift kind in the ModRM.reg /digit. diff --git a/asm/vex_test.go b/asm/vex_test.go index 742835e..4085b04 100644 --- a/asm/vex_test.go +++ b/asm/vex_test.go @@ -156,82 +156,121 @@ func TestVexShiftImm(t *testing.T) { // as well as every new operand form. func TestVexGroundTruth(t *testing.T) { cases := []struct { - name string - mnem string - ops []Operand - want string + name string + mnem string + ops []Operand + want string + wantOp string // decoded mnemonic, when it differs from mnem (the X/Y spellings) }{ // Three-operand NDS form. - {"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0"}, - {"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1"}, - {"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff"}, - {"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9"}, - {"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9"}, - {"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec"}, - {"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9"}, + {"VPADDQ Y8,Y9,Y8", "VPADDQ", []Operand{vreg(t, "Y8"), vreg(t, "Y9"), vreg(t, "Y8")}, "c44135d4c0", ""}, + {"VPADDQ X9,X8,X8", "VPADDQ", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c44139d4c1", ""}, + {"VPXOR X7,X7,X7", "VPXOR", []Operand{vreg(t, "X7"), vreg(t, "X7"), vreg(t, "X7")}, "c5c1efff", ""}, + {"VPSHUFB Y1,Y2,Y3", "VPSHUFB", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d00d9", ""}, + {"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""}, + {"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""}, + {"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""}, // Floating point (packed and scalar) and FMA — same NDS form, the pp // bits and map select the operation. - {"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1"}, - {"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9"}, - {"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4"}, - {"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0"}, - {"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8"}, - {"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1"}, - {"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8"}, - {"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6"}, - {"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807"}, + {"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""}, + {"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""}, + {"VMULPD Y12,Y12,Y12", "VMULPD", []Operand{vreg(t, "Y12"), vreg(t, "Y12"), vreg(t, "Y12")}, "c4411d59e4", ""}, + {"VXORPD Y8,Y8,Y8", "VXORPD", []Operand{vreg(t, "Y8"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d57c0", ""}, + {"VUNPCKHPD X8,X8,X9", "VUNPCKHPD", []Operand{vreg(t, "X8"), vreg(t, "X8"), vreg(t, "X9")}, "c4413915c8", ""}, + {"VADDSD X9,X8,X8", "VADDSD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "X8")}, "c4413b58c1", ""}, + {"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""}, + {"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""}, + {"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""}, // Two-operand reg/rm form (v̄vvv must be 1111). - {"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0"}, - {"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306"}, - {"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8"}, - {"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4"}, - {"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626"}, - {"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3"}, - {"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7"}, + {"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""}, + {"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""}, + {"VPBROADCASTD X0,Y15", "VPBROADCASTD", []Operand{vreg(t, "X0"), vreg(t, "Y15")}, "c4627d58f8", ""}, + {"VCVTDQ2PD X12,Y12", "VCVTDQ2PD", []Operand{vreg(t, "X12"), vreg(t, "Y12")}, "c4417ee6e4", ""}, + {"VCVTDQ2PD (SI),Y4", "VCVTDQ2PD", []Operand{Ptr(SI, 0, 16), vreg(t, "Y4")}, "c5fee626", ""}, + {"VPMOVMSKB X11,AX", "VPMOVMSKB", []Operand{vreg(t, "X11"), AX}, "c4c179d7c3", ""}, + {"VMOVMSKPS Y7,AX", "VMOVMSKPS", []Operand{vreg(t, "Y7"), AX}, "c5fc50c7", ""}, // Immediate shifts. - {"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301"}, - {"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502"}, + {"VPSLLD $1,Y3,Y4", "VPSLLD", []Operand{Imm(1), vreg(t, "Y3"), vreg(t, "Y4")}, "c5dd72f301", ""}, + {"VPSRLQ $2,Y5,Y6", "VPSRLQ", []Operand{Imm(2), vreg(t, "Y5"), vreg(t, "Y6")}, "c5cd73d502", ""}, // Variable-count shifts: the count lives in an XMM register or memory // and the instruction takes the NDS form. - {"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0"}, - {"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300"}, - {"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0"}, - {"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0"}, - {"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0"}, - {"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0"}, + {"VPSRLQ X0,Y8,Y8", "VPSRLQ", []Operand{vreg(t, "X0"), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd3c0", ""}, + {"VPSRLQ (AX),Y8,Y8", "VPSRLQ", []Operand{Ptr(AX, 0, 16), vreg(t, "Y8"), vreg(t, "Y8")}, "c53dd300", ""}, + {"VPSLLD X0,Y1,Y2", "VPSLLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f2d0", ""}, + {"VPSRLD X0,Y1,Y2", "VPSRLD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5d2d0", ""}, + {"VPSRAD X0,Y1,Y2", "VPSRAD", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5e2d0", ""}, + {"VPSLLQ X0,Y1,Y2", "VPSLLQ", []Operand{vreg(t, "X0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f5f3d0", ""}, // Immediate shuffle (reg=dst, rm=src, imm8). - {"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee"}, - {"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee"}, - {"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b"}, - {"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b"}, + {"VPSHUFD $0xEE,X8,X9", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "X8"), vreg(t, "X9")}, "c4417970c8ee", ""}, + {"VPSHUFD $0xEE,Y1,Y2", "VPSHUFD", []Operand{Imm(0xEE), vreg(t, "Y1"), vreg(t, "Y2")}, "c5fd70d1ee", ""}, + {"VPERMQ $0x1B,Y1,Y2", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e3fd00d11b", ""}, + {"VPERMQ $0x1B,Y11,Y12", "VPERMQ", []Operand{Imm(0x1B), vreg(t, "Y11"), vreg(t, "Y12")}, "c443fd00e31b", ""}, // Three-operand + immediate (reg=dst, vvvv=src1, rm=src2, imm8). - {"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901"}, - {"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901"}, - {"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931"}, - {"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501"}, + {"VSHUFPD $1,X1,X2,X3", "VSHUFPD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e9c6d901", ""}, + {"VSHUFPD $1,Y1,Y2,Y3", "VSHUFPD", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5edc6d901", ""}, + {"VPERM2I128 $0x31,Y1,Y2,Y3", "VPERM2I128", []Operand{Imm(0x31), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e36d46d931", ""}, + {"VINSERTI128 $1,X5,Y1,Y2", "VINSERTI128", []Operand{Imm(1), vreg(t, "X5"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37538d501", ""}, // Lane extract (reg=YMM source, rm=XMM/memory destination, imm8). - {"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101"}, - {"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701"}, - {"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101"}, + {"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""}, + {"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""}, + {"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""}, // Moves — each direction picks its own opcode and VEX.W. - {"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e"}, - {"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f"}, - {"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca"}, - {"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037"}, - {"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137"}, - {"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca"}, - {"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0"}, - {"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8"}, - {"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07"}, - {"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e"}, - {"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2"}, - {"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0"}, - {"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06"}, - {"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0"}, - {"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006"}, - {"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106"}, + {"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""}, + {"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""}, + {"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""}, + {"VMOVUPD (DI),Y14", "VMOVUPD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y14")}, "c57d1037", ""}, + {"VMOVUPD Y14,(DI)", "VMOVUPD", []Operand{vreg(t, "Y14"), Ptr(DI, 0, 32)}, "c57d1137", ""}, + {"VMOVUPD X1,X2", "VMOVUPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f911ca", ""}, + {"VMOVQ X8,AX", "VMOVQ", []Operand{vreg(t, "X8"), AX}, "c461f97ec0", ""}, + {"VMOVQ AX,X9", "VMOVQ", []Operand{AX, vreg(t, "X9")}, "c461f96ec8", ""}, + {"VMOVQ X8,(DI)", "VMOVQ", []Operand{vreg(t, "X8"), Ptr(DI, 0, 8)}, "c461f97e07", ""}, + {"VMOVQ (SI),X9", "VMOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X9")}, "c461f96e0e", ""}, + {"VMOVQ X8,X2", "VMOVQ", []Operand{vreg(t, "X8"), vreg(t, "X2")}, "c579d6c2", ""}, + {"VMOVQ X2,X8", "VMOVQ", []Operand{vreg(t, "X2"), vreg(t, "X8")}, "c4c179d6d0", ""}, + {"VMOVD X0,(SI)", "VMOVD", []Operand{vreg(t, "X0"), Ptr(SI, 0, 4)}, "c5f97e06", ""}, + {"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""}, + {"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""}, + {"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""}, + // Packed double arithmetic and unpack — the NDS form, the opcode + // selects the operation. + {"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""}, + {"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""}, + {"VMINPD Y1,Y2,Y3", "VMINPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5dd9", ""}, + {"VMAXPD X4,X5,X6", "VMAXPD", []Operand{vreg(t, "X4"), vreg(t, "X5"), vreg(t, "X6")}, "c5d15ff4", ""}, + {"VUNPCKLPD X1,X2,X3", "VUNPCKLPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e914d9", ""}, + {"VUNPCKLPD Y1,Y2,Y3", "VUNPCKLPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed14d9", ""}, + {"VSUBPD (AX),X1,X2", "VSUBPD", []Operand{Ptr(AX, 0, 16), vreg(t, "X1"), vreg(t, "X2")}, "c5f15c10", ""}, + // Scalar double and single arithmetic (F2 / F3 pp, 128-bit only). + {"VSUBSD X1,X2,X3", "VSUBSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5cd9", ""}, + {"VDIVSD X7,X1,X2", "VDIVSD", []Operand{vreg(t, "X7"), vreg(t, "X1"), vreg(t, "X2")}, "c5f35ed7", ""}, + {"VMINSD X1,X2,X3", "VMINSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5eb5dd9", ""}, + {"VMAXSD X3,X4,X5", "VMAXSD", []Operand{vreg(t, "X3"), vreg(t, "X4"), vreg(t, "X5")}, "c5db5feb", ""}, + {"VADDSS X1,X2,X3", "VADDSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea58d9", ""}, + {"VSUBSS X1,X2,X3", "VSUBSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5cd9", ""}, + {"VMULSS X9,X10,X11", "VMULSS", []Operand{vreg(t, "X9"), vreg(t, "X10"), vreg(t, "X11")}, "c4412a59d9", ""}, + {"VDIVSS X1,X2,X3", "VDIVSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5ed9", ""}, + {"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""}, + {"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""}, + {"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""}, + // VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp). + {"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""}, + {"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""}, + {"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""}, + // Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3 + // prefix — see the table comment), DQ→PD. + {"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""}, + {"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""}, + {"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""}, + {"VCVTPS2PD X1,Y2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c5fc5ad1", ""}, + // PD→DQ conversions: the X/Y spellings fix the source length and the + // destination is always XMM; the decoder reports the base mnemonic. + {"VCVTPD2DQX X1,X2", "VCVTPD2DQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fbe6d1", "VCVTPD2DQ"}, + {"VCVTPD2DQY Y1,X2", "VCVTPD2DQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5ffe6d1", "VCVTPD2DQ"}, + {"VCVTTPD2DQX X3,X4", "VCVTTPD2DQX", []Operand{vreg(t, "X3"), vreg(t, "X4")}, "c5f9e6e3", "VCVTTPD2DQ"}, + {"VCVTTPD2DQY Y5,X6", "VCVTTPD2DQY", []Operand{vreg(t, "Y5"), vreg(t, "X6")}, "c5fde6f5", "VCVTTPD2DQ"}, + {"VCVTPD2DQY (AX),X1", "VCVTPD2DQY", []Operand{Ptr(AX, 0, 32), vreg(t, "X1")}, "c5ffe608", "VCVTPD2DQ"}, // No-operand. - {"VZEROUPPER", "VZEROUPPER", nil, "c5f877"}, + {"VZEROUPPER", "VZEROUPPER", nil, "c5f877", ""}, } for _, c := range cases { code, err := Encode(c.mnem, c.ops...) @@ -251,7 +290,11 @@ func TestVexGroundTruth(t *testing.T) { if inst.Len != len(code) { t.Errorf("%s: Decode consumed %d of %d bytes", c.name, inst.Len, len(code)) } - if inst.Op.String() != c.mnem { + wantOp := c.wantOp + if wantOp == "" { + wantOp = c.mnem + } + if inst.Op.String() != wantOp { t.Errorf("%s: decoded as %s", c.name, inst.Op.String()) } } diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index 86f24b5..b7f3afb 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -28,7 +28,7 @@ import ( // version is the release version, stamped at build time via // -ldflags "-X main.version=…" (defaulting to the current release). -var version = "0.9.0" +var version = "0.10.0" func main() { if len(os.Args) < 2 { diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 1eb0b05..98a889a 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -213,9 +213,13 @@ three-operand-plus-immediate form (`VSHUFPD`, `VPERM2I128`, `VINSERTI128`), the lane-extract form (`VEXTRACTI128`, `VEXTRACTF128`, where the YMM source occupies the reg field and the XMM or memory destination r/m), the direction-sensitive moves (`VMOVDQU`, `VMOVUPD`, -`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic (`VADDPD`, -`VMULPD`, `VXORPD`, `VUNPCKHPD`, the scalar `VADDSD`/`VMULSD`, `VCVTDQ2PD`, -`VFMADD231PD`) and the no-operand `VZEROUPPER` — together with `VPERMD` and +`VMOVD`, `VMOVQ`, `VMOVSD`), the floating-point and FMA arithmetic — the +packed double operations (`VADDPD`/`VSUBPD`/`VMULPD`/`VDIVPD`/`VMINPD`/ +`VMAXPD`), the unpacks (`VUNPCKHPD`/`VUNPCKLPD`), the scalar SD and SS +operations, `VMOVDDUP`, `VXORPD`, the width-changing conversions +(`VCVTDQ2PS`, `VCVTPS2PD`, `VCVTDQ2PD`, and the `VCVTPD2DQX`/`Y` and +`VCVTTPD2DQX`/`Y` spellings, whose length follows the wider source) and +`VFMADD231PD` — and the no-operand `VZEROUPPER`, together with `VPERMD` and the scalar families (`CMOVcc`, `SETcc`, `LZCNT`/`TZCNT`, the extending moves, `CVTSx2SD`, `IMUL3`) and the EVEX (AVX-512) prefix — the four-byte prefix with 5-bit register fields (Z0–Z31, X/Y 16–31, with the mod=11 quirk that carries @@ -224,7 +228,11 @@ explicit merging/zeroing masks — written the way Go writes them, as a K operand among the operands plus a `.Z` mnemonic suffix), and the compressed disp8×N displacement, whose multiplier follows the memory operand's size — covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use, -plus the common AVX-512 F/BW integer set. Every encoding is validated two ways: by +plus the common AVX-512 F/BW integer set and the floating-point and +conversion set (the packed double arithmetic, the scalar SD/SS forms — +whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, and the +width-changing conversions, including the `VCVTPD2DQ`/`VCVTTPD2DQ` family +whose length follows the wider source operand). Every encoding is validated two ways: by round-trip decoding through `golang.org/x/arch`, and byte-for-byte against the machine code the real Go assembler emits — a comparison that holds for whole functions: all 27 functions of both kernels assemble to exactly the Go @@ -237,7 +245,8 @@ behind the code and resolves references to them (`mask<>(SB)`) to RIP-relative loads whose displacements point inside the resulting image, so the bytes are self-consistent at any base address. External (non-file-local) symbols are rejected: they need object-file emission, which — together with -EVEX masking/zeroing and the other architectures — is the rest of Phase 2. +the remaining EVEX forms and the other architectures — is the rest of +Phase 2. ## Extension points diff --git a/justfile b/justfile index 718190c..eb6c862 100644 --- a/justfile +++ b/justfile @@ -3,7 +3,7 @@ # gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm). -version := "0.9.0" +version := "0.10.0" default: @just --list