From 1312122a9911bf7995c7351ab0479285f4fc4a0f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Mon, 20 Jul 2026 16:05:08 +0200 Subject: [PATCH] feat(asm): complete the EVEX conversions, narrowing and mask-vector moves Assisted-by: Qwen 3.8 Max Preview --- asm/evex.go | 111 +++++++++++++++++++++++++++++++++++++++++-- asm/evex_test.go | 90 +++++++++++++++++++++++++++++++++++ asm/vex.go | 15 +++++- cmd/gasm/main.go | 2 +- docs/ARCHITECTURE.md | 5 +- justfile | 2 +- 6 files changed, 216 insertions(+), 9 deletions(-) diff --git a/asm/evex.go b/asm/evex.go index 51372c4..e3e63b7 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -308,6 +308,75 @@ var evexTable = map[string]evexSpec{ // EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src, // rm=dst, imm8 — the extract layout). "VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}}, + + // EVEX — unsigned and truncating conversions. The PD sources are the + // wide operand (the bare names are 512-bit only, the X/Y spellings fix + // the length); the PS/UQQ destinations are wide and follow the + // destination. + "VCVTPD2PS": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}}, + "VCVTPD2PSX": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}}, + "VCVTPD2PSY": {1, 0x5A, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}}, + "VCVTPD2UDQ": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}}, + "VCVTPD2UDQX": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}}, + "VCVTPD2UDQY": {1, 0x79, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}}, + "VCVTTPD2UDQ": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}}, + "VCVTTPD2UDQX": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}}, + "VCVTTPD2UDQY": {1, 0x78, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}}, + "VCVTTPD2UQQ": {1, 0x78, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VCVTPS2UDQ": {1, 0x79, 0, 0, -1, vexRM, [3]int{16, 32, 64}}, + "VCVTTPS2UDQ": {1, 0x78, 0, 0, -1, vexRM, [3]int{16, 32, 64}}, + "VCVTPS2UQQ": {1, 0x79, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + "VCVTTPS2UQQ": {1, 0x78, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + "VCVTTPD2QQ": {1, 0x7A, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VCVTTPS2QQ": {1, 0x7A, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + "VCVTUQQ2PD": {1, 0x7A, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VCVTUQQ2PS": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}}, + "VCVTUQQ2PSX": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}}, + "VCVTUQQ2PSY": {1, 0x7A, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}}, + "VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}}, + "VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}}, + + // EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow + // source; disp8×N follows its size). + "VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}}, + "VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}}, + "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + "VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM, [3]int{4, 8, 16}}, + "VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM, [3]int{4, 8, 16}}, + "VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM, [3]int{2, 4, 8}}, + "VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + "VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}}, + "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + + // EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg, + // narrow destination in r/m): signed, unsigned and the D/Q truncations. + "VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, + "VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}}, + "VPMOVSDW": {2, 0x23, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, + "VPMOVSQW": {2, 0x24, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, + "VPMOVSQD": {2, 0x25, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, + "VPMOVSWB": {2, 0x20, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, + "VPMOVUSWB": {2, 0x10, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, + "VPMOVUSDB": {2, 0x11, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, + "VPMOVUSQB": {2, 0x12, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}}, + "VPMOVUSDW": {2, 0x13, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, + "VPMOVUSQW": {2, 0x14, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, + "VPMOVUSQD": {2, 0x15, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, + "VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, + "VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}}, + + // EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register + // into a vector (rm = K source, reg = vector destination), *2M does the + // reverse (reg = K destination, rm = vector source, the length follows + // the vector). + "VPMOVM2B": {2, 0x28, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VPMOVM2W": {2, 0x28, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VPMOVM2D": {2, 0x38, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VPMOVM2Q": {2, 0x38, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VPMOVB2M": {2, 0x29, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // operand is the narrow source, so disp8×N follows its size (8/16/32 for // the xmm/ymm/zmm destination lengths). @@ -543,6 +612,9 @@ var evexRound = map[string]bool{ "VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true, "VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true, "VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true, + "VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true, + "VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true, + "VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true, } // evexBcstN maps an instruction accepting .BCST to the broadcast element @@ -562,6 +634,9 @@ var evexBcstN = map[string]int{ "VRANGEPD": 8, "VRANGEPS": 4, "VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8, "VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4, + "VCVTPD2PS": 8, "VCVTPD2UDQ": 8, "VCVTTPD2UDQ": 8, "VCVTTPD2UQQ": 8, + "VCVTPS2UDQ": 4, "VCVTTPS2UDQ": 4, "VCVTPS2UQQ": 4, "VCVTTPS2UQQ": 4, + "VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8, } // splitMask extracts an explicit mask register (K1–K7) from the operand list, @@ -591,6 +666,18 @@ func splitMask(ops []Operand) ([]Operand, int, error) { // operands; the mnemonic suffix carries zeroing, rounding/SAE and // broadcast. func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error { + // The mask/vector conversions take the K register as a genuine operand + // (source or destination), not as a mask, and accept no suffixes. + if evexKOperand[mnemUpper] { + if sfx.any() { + return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper) + } + spec, ok := evexTable[mnemUpper] + if !ok { + return fmt.Errorf("unsupported instruction %q", mnemUpper) + } + return e.encodeEvexRM(spec, ops, 0, sfx) + } spec, inTable := evexTable[mnemUpper] if inTable { if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] { @@ -713,17 +800,25 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuf } // encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src, -// no vvvv), e.g. VCVTQQ2PD. +// no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the +// *2M mask conversions), in which case the vector length comes from the +// source. func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 2 { return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) - if !ok || !dstReg.isVec() { - return fmt.Errorf("EVEX destination must be a vector register") + if !ok || (!dstReg.isVec() && !dstReg.mask) { + return fmt.Errorf("EVEX destination must be a vector or mask register") } - return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx) + ll := dstReg.vecLenBit() + if dstReg.mask { + if r, ok := src.(Reg); ok && r.isVec() { + ll = r.vecLenBit() + } + } + return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx) } // encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst @@ -1260,6 +1355,14 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx) } +// evexKOperand lists the instructions whose K register is a genuine operand +// (the source or destination of a mask/vector conversion) rather than a +// mask modifier — the M2 and 2M conversions. They take no masking. +var evexKOperand = map[string]bool{ + "VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true, + "VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true, +} + // kmovSpec describes a KMOV width: the opcode depends on the operand // direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem), // gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory diff --git a/asm/evex_test.go b/asm/evex_test.go index dccc921..56c43e2 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -467,6 +467,96 @@ func TestEvexHelperGroundTruth(t *testing.T) { } } +// TestEvexConversionGroundTruth covers the unsigned and truncating VCVT* +// conversions, the remaining sign/zero-extending moves, the signed/unsigned +// narrowing stores and the mask/vector conversions, byte for byte against +// the Go assembler. +func TestEvexConversionGroundTruth(t *testing.T) { + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + // Unsigned and truncating conversions. + {"VCVTPD2PS", "VCVTPD2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fd485ad1"}, + {"VCVTPD2PSX", "VCVTPD2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f95ad1"}, + {"VCVTPD2PSY", "VCVTPD2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "c5fd5ad1"}, + {"VCVTPD2UDQ", "VCVTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4879d1"}, + {"VCVTPD2UDQX", "VCVTPD2UDQX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc0879d1"}, + {"VCVTTPD2UDQ", "VCVTTPD2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc4878d1"}, + {"VCVTTPD2UDQY", "VCVTTPD2UDQY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc2878d1"}, + {"VCVTTPD2UQQ", "VCVTTPD2UQQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd4878d1"}, + {"VCVTPS2UDQ", "VCVTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4879d1"}, + {"VCVTTPS2UDQ", "VCVTTPS2UDQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4878d1"}, + {"VCVTPS2UQQ", "VCVTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4879d1"}, + {"VCVTTPS2UQQ", "VCVTTPS2UQQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d4878d1"}, + {"VCVTTPD2QQ", "VCVTTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487ad1"}, + {"VCVTTPS2QQ", "VCVTTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487ad1"}, + {"VCVTUQQ2PD", "VCVTUQQ2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487ad1"}, + {"VCVTUQQ2PS", "VCVTUQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1ff487ad1"}, + {"VCVTUQQ2PSX", "VCVTUQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1ff087ad1"}, + {"VCVTQQ2PSX", "VCVTQQ2PSX", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "62f1fc085bd1"}, + {"VCVTQQ2PSY", "VCVTQQ2PSY", []Operand{vreg(t, "Y1"), vreg(t, "X2")}, "62f1fc285bd1"}, + // The remaining sign/zero-extending moves. + {"VPMOVSXBD", "VPMOVSXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d21d1"}, + {"VPMOVSXBQ evex", "VPMOVSXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4822d1"}, + {"VPMOVSXWQ", "VPMOVSXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d24d1"}, + {"VPMOVSXWD", "VPMOVSXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d23d1"}, + {"VPMOVZXBD", "VPMOVZXBD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d31d1"}, + {"VPMOVZXBQ evex", "VPMOVZXBQ", []Operand{vreg(t, "X1"), vreg(t, "Z2")}, "62f27d4832d1"}, + {"VPMOVZXWD", "VPMOVZXWD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d33d1"}, + {"VPMOVZXWQ", "VPMOVZXWQ", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d34d1"}, + // Signed narrowing stores. + {"VPMOVSDB", "VPMOVSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4821ca"}, + {"VPMOVSDW", "VPMOVSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4823ca"}, + {"VPMOVSQB", "VPMOVSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4822ca"}, + {"VPMOVSQD", "VPMOVSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4825ca"}, + {"VPMOVSQW", "VPMOVSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4824ca"}, + {"VPMOVSWB", "VPMOVSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4820ca"}, + // Unsigned narrowing stores. + {"VPMOVUSDB", "VPMOVUSDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4811ca"}, + {"VPMOVUSDW", "VPMOVUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4813ca"}, + {"VPMOVUSQB", "VPMOVUSQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4812ca"}, + {"VPMOVUSQD", "VPMOVUSQD", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4815ca"}, + {"VPMOVUSQW", "VPMOVUSQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4814ca"}, + {"VPMOVUSWB", "VPMOVUSWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4810ca"}, + {"VPMOVDB", "VPMOVDB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4831ca"}, + {"VPMOVQW", "VPMOVQW", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4834ca"}, + // Mask/vector conversions (the K register is an operand, not a + // mask). + {"VPMOVM2B", "VPMOVM2B", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0828d1"}, + {"VPMOVM2W", "VPMOVM2W", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f2fe0828d1"}, + {"VPMOVM2D", "VPMOVM2D", []Operand{vreg(t, "K1"), vreg(t, "X2")}, "62f27e0838d1"}, + {"VPMOVM2Q", "VPMOVM2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe4838d1"}, + {"VPMOVB2M", "VPMOVB2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f27e0829d1"}, + {"VPMOVW2M", "VPMOVW2M", []Operand{vreg(t, "X1"), vreg(t, "K2")}, "62f2fe0829d1"}, + {"VPMOVD2M", "VPMOVD2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f27e4839d1"}, + {"VPMOVQ2M", "VPMOVQ2M", []Operand{vreg(t, "Z1"), vreg(t, "K2")}, "62f2fe4839d1"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := hexCompact(code); got != c.want { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + continue + } + inst, err := x86asm.Decode(code, 64) + if err != nil { + t.Errorf("%s: Decode(%x): %v", c.name, code, err) + continue + } + want := c.mnem + got := inst.Op.String() + if got != want && !(len(want) > len(got) && want[:len(got)] == got) { + t.Errorf("%s: decoded as %s", c.name, got) + } + } +} + // TestEvexErrors checks the EVEX-specific error paths. func TestEvexErrors(t *testing.T) { cases := []struct { diff --git a/asm/vex.go b/asm/vex.go index 3916a40..d886e38 100644 --- a/asm/vex.go +++ b/asm/vex.go @@ -130,9 +130,15 @@ var vexTable = map[string]vexSpec{ // no vvvv). "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM}, - "VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM}, + "VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM}, + "VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM}, + "VPMOVSXWQ": {2, 0x24, 0, 1, -1, vexRM}, "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM}, "VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM}, + "VPMOVZXBD": {2, 0x31, 0, 1, -1, vexRM}, + "VPMOVZXBQ": {2, 0x32, 0, 1, -1, vexRM}, + "VPMOVZXWD": {2, 0x33, 0, 1, -1, vexRM}, + "VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, // VEX.128/256.F3.0F.WIG — signed dword to packed double conversion @@ -198,6 +204,11 @@ var vexTable = map[string]vexSpec{ // VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src). "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM}, "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM}, + // VEX.66.0F.WIG — packed double to packed single conversion, the X/Y + // spellings: the destination is always XMM and the spelling fixes the + // source length (X = 128, Y = 256). + "VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen}, + "VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen}, // VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift). "VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm}, @@ -222,6 +233,8 @@ var vexSrcLen = map[string]int{ "VCVTPD2DQY": 1, "VCVTTPD2DQX": 0, "VCVTTPD2DQY": 1, + "VCVTPD2PSX": 0, + "VCVTPD2PSY": 1, } // vexVarShift maps the shift mnemonics to their variable-count opcode — the diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index 4cdd39b..a209b47 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -28,7 +28,7 @@ import ( // version is the release version, stamped at build time via // -ldflags "-X main.version=…" (defaulting to the current release). -var version = "0.14.0" +var version = "0.15.0" func main() { if len(os.Args) < 2 { diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index a0598cc..268fc3f 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -240,8 +240,9 @@ expand/compress family, the broadcasts, the opmask-register instructions moves and the remaining extending/narrowing moves, the floating-point helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*, VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an -opmask destination, and the remaining VCVT* conversions including the -half-precision pair), and gather/scatter with VSIB addressing — both the +opmask destination, and the VCVT* conversions — signed, unsigned and +truncating, including the length-suffixed X/Y spellings and the +mask/vector conversions VPMOVM2*/VPMOV*2M), and gather/scatter with VSIB addressing — both the VEX spelling with a vector mask register and the EVEX spelling with an explicit K mask, where the EVEX length follows the VSIB index register, not the data register. The EVEX mnemonic diff --git a/justfile b/justfile index 62ef8f5..e96c825 100644 --- a/justfile +++ b/justfile @@ -3,7 +3,7 @@ # gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm). -version := "0.14.0" +version := "0.15.0" default: @just --list