From 5af12e15ac2782c48c8278f7e36bc42bbecd65c5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Tue, 21 Jul 2026 12:59:25 +0200 Subject: [PATCH] feat(asm): add the GPR-interchanging conversions, completing the amd64 EVEX set Assisted-by: Qwen 3.8 Max Preview --- asm/evex.go | 51 +++++++++++++++++++++++++++++---- asm/evex_test.go | 68 ++++++++++++++++++++++++++++++++++++++++++++ asm/vex.go | 18 ++++++++++++ asm/vex_test.go | 9 ++++-- cmd/gasm/main.go | 2 +- docs/ARCHITECTURE.md | 4 ++- justfile | 2 +- 7 files changed, 143 insertions(+), 11 deletions(-) diff --git a/asm/evex.go b/asm/evex.go index e3e63b7..9aed61b 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -377,6 +377,37 @@ var evexTable = map[string]evexSpec{ "VPMOVW2M": {2, 0x29, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, "VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, "VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, + + // EVEX — scalar conversions between vector and general-purpose + // registers. Vector to GPR (two operands: vec/mem source, GPR + // destination, vvvv unused): the signed and truncated pair, and the + // unsigned forms (EVEX only). + "VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM, [3]int{8, 8, 8}}, + "VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM, [3]int{8, 8, 8}}, + "VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM, [3]int{4, 4, 4}}, + "VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM, [3]int{4, 4, 4}}, + "VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM, [3]int{8, 8, 8}}, + "VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM, [3]int{8, 8, 8}}, + "VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM, [3]int{4, 4, 4}}, + "VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM, [3]int{4, 4, 4}}, + "VCVTSD2USIL": {1, 0x79, 0, 3, -1, vexRM, [3]int{8, 8, 8}}, + "VCVTSD2USIQ": {1, 0x79, 1, 3, -1, vexRM, [3]int{8, 8, 8}}, + "VCVTSS2USIL": {1, 0x79, 0, 2, -1, vexRM, [3]int{4, 4, 4}}, + "VCVTSS2USIQ": {1, 0x79, 1, 2, -1, vexRM, [3]int{4, 4, 4}}, + "VCVTTSD2USIL": {1, 0x78, 0, 3, -1, vexRM, [3]int{8, 8, 8}}, + "VCVTTSD2USIQ": {1, 0x78, 1, 3, -1, vexRM, [3]int{8, 8, 8}}, + "VCVTTSS2USIL": {1, 0x78, 0, 2, -1, vexRM, [3]int{4, 4, 4}}, + "VCVTTSS2USIQ": {1, 0x78, 1, 2, -1, vexRM, [3]int{4, 4, 4}}, + // GPR to vector (three operands: GPR/mem source in r/m, the preserved + // vector source in vvvv, vector destination in reg). + "VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}}, + "VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, + "VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, + "VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}}, + "VCVTUSI2SDL": {1, 0x7B, 0, 3, -1, vexNDS3, [3]int{4, 4, 4}}, + "VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, + "VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, + "VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}}, // EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // operand is the narrow source, so disp8×N follows its size (8/16/32 for // the xmm/ymm/zmm destination lengths). @@ -615,6 +646,12 @@ var evexRound = map[string]bool{ "VCVTPD2PS": true, "VCVTPD2UDQ": true, "VCVTTPD2UDQ": true, "VCVTTPD2UQQ": true, "VCVTPS2UDQ": true, "VCVTTPS2UDQ": true, "VCVTPS2UQQ": true, "VCVTTPS2UQQ": true, "VCVTTPD2QQ": true, "VCVTTPS2QQ": true, "VCVTUQQ2PD": true, "VCVTUQQ2PS": true, + "VCVTSD2SI": true, "VCVTSD2SIQ": true, "VCVTSS2SI": true, "VCVTSS2SIQ": true, + "VCVTSD2USIL": true, "VCVTSD2USIQ": true, "VCVTSS2USIL": true, "VCVTSS2USIQ": true, + "VCVTTSD2SI": true, "VCVTTSD2SIQ": true, "VCVTTSS2SI": true, "VCVTTSS2SIQ": true, + "VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true, + "VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true, + "VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true, } // evexBcstN maps an instruction accepting .BCST to the broadcast element @@ -801,19 +838,23 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuf // encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src, // no vvvv), e.g. VCVTQQ2PD. The destination may be an opmask register (the -// *2M mask conversions), in which case the vector length comes from the -// source. +// *2M mask conversions) or a general-purpose register (the scalar +// vector-to-GPR conversions); in both cases the vector length comes from +// the source. func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 2 { return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) - if !ok || (!dstReg.isVec() && !dstReg.mask) { - return fmt.Errorf("EVEX destination must be a vector or mask register") + if !ok { + return fmt.Errorf("EVEX destination must be a register") } ll := dstReg.vecLenBit() - if dstReg.mask { + if !dstReg.isVec() { + // Mask or GPR destination: the length follows the vector source + // (128 for a memory source). + ll = 0 if r, ok := src.(Reg); ok && r.isVec() { ll = r.vecLenBit() } diff --git a/asm/evex_test.go b/asm/evex_test.go index 56c43e2..cd2ca09 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -467,6 +467,74 @@ func TestEvexHelperGroundTruth(t *testing.T) { } } +// TestEvexGprGroundTruth covers the scalar conversions between vector and +// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI +// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector +// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte +// for byte against the Go assembler, including memory sources and extended +// GPRs. +func TestEvexGprGroundTruth(t *testing.T) { + mem := func(b Reg) Operand { return Ptr(b, 0, 8) } + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + {"VCVTSD2SI", "VCVTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2dc1"}, + {"VCVTSD2SIQ", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2dc1"}, + {"VCVTSS2SI", "VCVTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2dc1"}, + {"VCVTSS2SIQ", "VCVTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2dc1"}, + {"VCVTTSD2SI", "VCVTTSD2SI", []Operand{vreg(t, "X1"), AX}, "c5fb2cc1"}, + {"VCVTTSD2SIQ", "VCVTTSD2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fb2cc1"}, + {"VCVTTSS2SI", "VCVTTSS2SI", []Operand{vreg(t, "X1"), AX}, "c5fa2cc1"}, + {"VCVTTSS2SIQ", "VCVTTSS2SIQ", []Operand{vreg(t, "X1"), AX}, "c4e1fa2cc1"}, + {"VCVTSD2USIL", "VCVTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0879c1"}, + {"VCVTSD2USIQ", "VCVTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0879c1"}, + {"VCVTSS2USIL", "VCVTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0879c1"}, + {"VCVTSS2USIQ", "VCVTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0879c1"}, + {"VCVTTSD2USIL", "VCVTTSD2USIL", []Operand{vreg(t, "X1"), AX}, "62f17f0878c1"}, + {"VCVTTSD2USIQ", "VCVTTSD2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1ff0878c1"}, + {"VCVTTSS2USIL", "VCVTTSS2USIL", []Operand{vreg(t, "X1"), AX}, "62f17e0878c1"}, + {"VCVTTSS2USIQ", "VCVTTSS2USIQ", []Operand{vreg(t, "X1"), AX}, "62f1fe0878c1"}, + {"VCVTSI2SDL", "VCVTSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f32ad0"}, + {"VCVTSI2SDQ", "VCVTSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32ad0"}, + {"VCVTSI2SSL", "VCVTSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c5f22ad0"}, + {"VCVTSI2SSQ", "VCVTSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "c4e1f22ad0"}, + {"VCVTUSI2SDL", "VCVTUSI2SDL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f177087bd0"}, + {"VCVTUSI2SDQ", "VCVTUSI2SDQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f7087bd0"}, + {"VCVTUSI2SSL", "VCVTUSI2SSL", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f176087bd0"}, + {"VCVTUSI2SSQ", "VCVTUSI2SSQ", []Operand{AX, vreg(t, "X1"), vreg(t, "X2")}, "62f1f6087bd0"}, + {"VCVTSD2SI mem", "VCVTSD2SI", []Operand{mem(AX), BX}, "c5fb2d18"}, + {"VCVTSI2SDQ mem", "VCVTSI2SDQ", []Operand{mem(BX), vreg(t, "X1"), vreg(t, "X2")}, "c4e1f32a13"}, + {"VCVTSD2SIQ hi gpr", "VCVTSD2SIQ", []Operand{vreg(t, "X1"), vreg(t, "R9")}, "c461fb2dc9"}, + {"VCVTSI2SDQ hi gpr", "VCVTSI2SDQ", []Operand{vreg(t, "R10"), vreg(t, "X1"), vreg(t, "X2")}, "c4c1f32ad2"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := hexCompact(code); got != c.want { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + continue + } + inst, err := x86asm.Decode(code, 64) + if err != nil { + t.Errorf("%s: Decode(%x): %v", c.name, code, err) + continue + } + // The decoder does not distinguish the Plan 9 SIQ spelling (the + // 64-bit GPR destination) from the base name; the W bit carries it. + want := c.mnem + got := inst.Op.String() + if got != want && !(len(want) > len(got) && want[:len(got)] == got) { + t.Errorf("%s: decoded as %s", c.name, got) + } + } +} + // TestEvexConversionGroundTruth covers the unsigned and truncating VCVT* // conversions, the remaining sign/zero-extending moves, the signed/unsigned // narrowing stores and the mask/vector conversions, byte for byte against diff --git a/asm/vex.go b/asm/vex.go index d886e38..283d423 100644 --- a/asm/vex.go +++ b/asm/vex.go @@ -210,6 +210,24 @@ var vexTable = map[string]vexSpec{ "VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen}, "VCVTPD2PSY": {1, 0x5A, 0, 1, -1, vexRMSrcLen}, + // VEX scalar conversions between vector and general-purpose registers. + // Vector to GPR (two operands: vec/mem source, GPR destination, vvvv + // unused; the length follows the source). + "VCVTSD2SI": {1, 0x2D, 0, 3, -1, vexRM}, + "VCVTSD2SIQ": {1, 0x2D, 1, 3, -1, vexRM}, + "VCVTSS2SI": {1, 0x2D, 0, 2, -1, vexRM}, + "VCVTSS2SIQ": {1, 0x2D, 1, 2, -1, vexRM}, + "VCVTTSD2SI": {1, 0x2C, 0, 3, -1, vexRM}, + "VCVTTSD2SIQ": {1, 0x2C, 1, 3, -1, vexRM}, + "VCVTTSS2SI": {1, 0x2C, 0, 2, -1, vexRM}, + "VCVTTSS2SIQ": {1, 0x2C, 1, 2, -1, vexRM}, + // GPR to vector (three operands: GPR/mem source in r/m, the preserved + // vector source in vvvv, vector destination in reg). + "VCVTSI2SDL": {1, 0x2A, 0, 3, -1, vexNDS3}, + "VCVTSI2SDQ": {1, 0x2A, 1, 3, -1, vexNDS3}, + "VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3}, + "VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3}, + // VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift). "VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm}, "VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm}, diff --git a/asm/vex_test.go b/asm/vex_test.go index 4085b04..20a9ff7 100644 --- a/asm/vex_test.go +++ b/asm/vex_test.go @@ -39,11 +39,14 @@ func TestVexNDS3(t *testing.T) { } inst, err := x86asm.Decode(code, 64) if err != nil { - t.Errorf("%s: Decode(% x): %v", mnem, code, err) + t.Errorf("%s: Decode(% x): %v", mnem, err, code) continue } - if inst.Op.String() != mnem { - t.Errorf("%s: decoded as %s (% x)", mnem, inst.Op.String(), code) + // The decoder folds the Plan 9 L/Q GPR-width spellings (VCVTSI2SDL/ + // SDQ, SSL/SSQ) onto the base name; the W bit carries the width. + got := inst.Op.String() + if got != mnem && !(len(mnem) > len(got) && mnem[:len(got)] == got) { + t.Errorf("%s: decoded as %s (% x)", mnem, got, code) } } } diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index a209b47..9b9f876 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -28,7 +28,7 @@ import ( // version is the release version, stamped at build time via // -ldflags "-X main.version=…" (defaulting to the current release). -var version = "0.15.0" +var version = "0.16.0" func main() { if len(os.Args) < 2 { diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 268fc3f..107358e 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -242,7 +242,9 @@ helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*, VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an opmask destination, and the VCVT* conversions — signed, unsigned and truncating, including the length-suffixed X/Y spellings and the -mask/vector conversions VPMOVM2*/VPMOV*2M), and gather/scatter with VSIB addressing — both the +mask/vector conversions VPMOVM2*/VPMOV*2M, and the scalar conversions +between vector and general-purpose registers (VCVT{,T}S{D,S}2SI{,Q} and +the unsigned forms, VCVTSI2*/VCVTUSI2*), and gather/scatter with VSIB addressing — both the VEX spelling with a vector mask register and the EVEX spelling with an explicit K mask, where the EVEX length follows the VSIB index register, not the data register. The EVEX mnemonic diff --git a/justfile b/justfile index e96c825..6b969d8 100644 --- a/justfile +++ b/justfile @@ -3,7 +3,7 @@ # gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm). -version := "0.15.0" +version := "0.16.0" default: @just --list