diff --git a/asm/encode.go b/asm/encode.go index 8ae2553..3f88c48 100644 --- a/asm/encode.go +++ b/asm/encode.go @@ -57,7 +57,7 @@ func (e *enc) encode(mnem string, ops []Operand) error { if err != nil { return err } - if isVex(base) || isEvex(base) || isKOp(base) || base == "KMOVW" || base == "KMOVQ" { + if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" { return e.encodeVec(base, ops, sfx) } if sfx.any() { @@ -130,6 +130,12 @@ func splitSize(upper string) (base string, size int) { // takes EVEX when an operand demands it (a ZMM or K register, or an // EVEX-only mnemonic) and VEX otherwise. func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error { + if gs, ok := gatherTable[upper]; ok { + return e.encodeGather(upper, gs, ops, sfx) + } + if ss, ok := scatterTable[upper]; ok { + return e.encodeScatter(upper, ss, ops, sfx) + } if upper == "KMOVW" || upper == "KMOVQ" { if sfx.any() { return fmt.Errorf("%s takes no EVEX suffixes", upper) diff --git a/asm/evex.go b/asm/evex.go index c00dcb3..51372c4 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -234,6 +234,80 @@ var evexTable = map[string]evexSpec{ // EVEX W1 qword shifts. "VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}}, "VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, + + // EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src). + "VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + // EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is + // rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A + // forms, these take the 66 prefix; W selects double/single. + "VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, + "VRCP14SS": {2, 0x4D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, + "VRSQRT14SD": {2, 0x4F, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, + "VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, + "VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, + "VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, + // EVEX.66.0F38 — scale by a power of two (NDS form). + "VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}}, + "VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}}, + + // EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst, + // rm=src, imm8). + "VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + "VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + "VGETMANTPD": {3, 0x26, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + "VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + "VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + "VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, + // EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS + + // imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66 + // prefix; W selects double/single. + "VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, + "VRNDSCALESS": {3, 0x0A, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}}, + "VGETMANTSD": {3, 0x27, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, + "VGETMANTSS": {3, 0x27, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}}, + "VREDUCESD": {3, 0x57, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, + "VREDUCESS": {3, 0x57, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}}, + "VFIXUPIMMPD": {3, 0x54, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VFIXUPIMMPS": {3, 0x54, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VFIXUPIMMSD": {3, 0x55, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, + "VFIXUPIMMSS": {3, 0x55, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}}, + "VRANGEPD": {3, 0x50, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VRANGEPS": {3, 0x50, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}}, + "VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}}, + + // EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the + // reg field carries the opmask destination. The packed forms carry an + // explicit length in the mnemonic (X/Y/Z). + "VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}}, + "VFPCLASSPDY": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 32, 0}}, + "VFPCLASSPDZ": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{0, 0, 64}}, + "VFPCLASSPSX": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{16, 0, 0}}, + "VFPCLASSPSY": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 32, 0}}, + "VFPCLASSPSZ": {3, 0x66, 0, 1, -1, vexImmRM, [3]int{0, 0, 64}}, + "VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}}, + "VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}}, + + // EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit + // source sets the length); the rest follow the destination. + "VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}}, + "VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + "VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, + "VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, + // EVEX.66.0F38 — half-precision convert (half-width source). + "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + // EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src, + // rm=dst, imm8 — the extract layout). + "VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}}, // EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // operand is the narrow source, so disp8×N follows its size (8/16/32 for // the xmm/ymm/zmm destination lengths). @@ -466,6 +540,9 @@ var evexRound = map[string]bool{ "VMINSD": true, "VMAXSD": true, "VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true, "VMINSS": true, "VMAXSS": true, + "VSCALEFPD": true, "VSCALEFPS": true, "VSCALEFSD": true, "VSCALEFSS": true, + "VGETEXPPD": true, "VGETEXPPS": true, "VGETEXPSD": true, "VGETEXPSS": true, + "VCVTDQ2PS": true, "VCVTPS2QQ": true, "VCVTQQ2PS": true, "VCVTPD2UQQ": true, } // evexBcstN maps an instruction accepting .BCST to the broadcast element @@ -475,6 +552,16 @@ var evexBcstN = map[string]int{ "VMINPD": 8, "VMAXPD": 8, "VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4, "VMINPS": 4, "VMAXPS": 4, + "VRCP14PD": 8, "VRCP14PS": 4, "VRSQRT14PD": 8, "VRSQRT14PS": 4, + "VGETEXPPD": 8, "VGETEXPPS": 4, + "VSCALEFPD": 8, "VSCALEFPS": 4, + "VRNDSCALEPD": 8, "VRNDSCALEPS": 4, + "VGETMANTPD": 8, "VGETMANTPS": 4, + "VREDUCEPD": 8, "VREDUCEPS": 4, + "VFIXUPIMMPD": 8, "VFIXUPIMMPS": 4, + "VRANGEPD": 8, "VRANGEPS": 4, + "VCVTDQ2PS": 4, "VCVTPS2QQ": 4, "VCVTQQ2PS": 8, + "VCVTUDQ2PD": 4, "VCVTUDQ2PS": 4, } // splitMask extracts an explicit mask register (K1–K7) from the operand list, @@ -547,6 +634,10 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask { return kdst(e.encodeEvexNDS3Imm) } + case vexImmRM: + if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask { + return kdst(e.encodeEvexImmRM) + } } } @@ -636,7 +727,9 @@ func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffi } // encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst -// (reg = dst, rm = src, imm8), e.g. VPSHUFD. +// (reg = dst, rm = src, imm8), e.g. VPSHUFD. The destination may be an +// opmask register (VFPCLASS*), in which case the vector length comes from +// the source. func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 3 { return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops)) @@ -647,11 +740,15 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) - if !ok || !dstReg.isVec() { - return fmt.Errorf("shuffle destination must be a vector register") + if !ok || (!dstReg.isVec() && !dstReg.mask) { + return fmt.Errorf("shuffle destination must be a vector or mask register") } ll := dstReg.vecLenBit() - if r, ok := src.(Reg); ok && r.isVec() { + if dstReg.mask { + if r, ok := src.(Reg); ok && r.isVec() { + ll = r.vecLenBit() + } + } else if r, ok := src.(Reg); ok && r.isVec() { ll = r.vecLenBit() } immByte, err := imm8(int64(immVal)) @@ -1034,6 +1131,135 @@ func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil } +// gatherSpec describes a gather/scatter family member: all live in +// 66.0F38; the opcode and W select the index and data element widths, and n +// is the data element size (the EVEX disp8×N multiplier). +type gatherSpec struct { + opcode byte + w int + n int +} + +var gatherTable = map[string]gatherSpec{ + "VGATHERDPS": {0x92, 0, 4}, + "VGATHERDPD": {0x92, 1, 8}, + "VGATHERQPS": {0x93, 0, 4}, + "VGATHERQPD": {0x93, 1, 8}, + "VPGATHERDD": {0x90, 0, 4}, + "VPGATHERDQ": {0x90, 1, 8}, + "VPGATHERQD": {0x91, 0, 4}, + "VPGATHERQQ": {0x91, 1, 8}, +} + +var scatterTable = map[string]gatherSpec{ + "VSCATTERDPS": {0xA2, 0, 4}, + "VSCATTERDPD": {0xA2, 1, 8}, + "VSCATTERQPS": {0xA3, 0, 4}, + "VSCATTERQPD": {0xA3, 1, 8}, + "VPSCATTERDD": {0xA0, 0, 4}, + "VPSCATTERDQ": {0xA0, 1, 8}, + "VPSCATTERQD": {0xA1, 0, 4}, + "VPSCATTERQQ": {0xA1, 1, 8}, +} + +// isGather reports whether the mnemonic is a gather instruction. +func isGather(upper string) bool { + _, ok := gatherTable[upper] + return ok +} + +// isScatter reports whether the mnemonic is a scatter instruction. +func isScatter(upper string) bool { + _, ok := scatterTable[upper] + return ok +} + +// vsibLen validates a VSIB memory operand (the index must be a vector +// register) and returns it with the vector length the index selects — the +// EVEX L'L field follows the index register, not the data register. +func vsibLen(op Operand, what string) (Mem, int, error) { + m, ok := op.(Mem) + if !ok || !m.HasIndex || !m.Index.isVec() { + return Mem{}, 0, fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", what) + } + return m, m.Index.vecLenBit(), nil +} + +// encodeGather encodes a gather. The VEX spelling carries the mask in a +// vector register (OP mask, vsib, dst: vvvv = mask, rm = vsib, reg = dst, +// L follows the data register); the EVEX spelling carries it in aaa (OP +// vsib, K, dst: rm = vsib, reg = dst, L follows the VSIB index). +func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexSuffix) error { + rest, mask, err := splitMask(ops) + if err != nil { + return err + } + if mask != 0 || sfx.any() { + // EVEX form: OP vsib, K, dst. + if len(rest) != 2 { + return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops)) + } + vsib, ll, err := vsibLen(rest[0], upper) + if err != nil { + return err + } + dst, ok := rest[1].(Reg) + if !ok || !dst.isVec() { + return fmt.Errorf("%s: destination must be a vector register", upper) + } + evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}} + return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx) + } + // VEX form: OP mask, vsib, dst. + if len(rest) != 3 { + return fmt.Errorf("%s expects 3 operands (mask, vsib, dst), got %d", upper, len(rest)) + } + maskReg, ok := rest[0].(Reg) + if !ok || !maskReg.isVec() { + return fmt.Errorf("%s: mask must be a vector register", upper) + } + vsib, _, err := vsibLen(rest[1], upper) + if err != nil { + return err + } + dst, ok := rest[2].(Reg) + if !ok || !dst.isVec() { + return fmt.Errorf("%s: destination must be a vector register", upper) + } + spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1} + rBit := 0 + if dst.idx >= 8 { + rBit = 1 + } + return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib) +} + +// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src, +// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB +// index. +func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error { + rest, mask, err := splitMask(ops) + if err != nil { + return err + } + if mask == 0 { + return fmt.Errorf("%s requires a K mask register", upper) + } + if len(rest) != 2 { + return fmt.Errorf("%s expects 3 operands (src, K, vsib), got %d", upper, len(ops)) + } + src, ok := rest[0].(Reg) + if !ok || !src.isVec() { + return fmt.Errorf("%s: source must be a vector register", upper) + } + vsib, ll, err := vsibLen(rest[1], upper) + if err != nil { + return err + } + evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}} + return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx) +} + // kmovSpec describes a KMOV width: the opcode depends on the operand // direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem), // gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory diff --git a/asm/evex_test.go b/asm/evex_test.go index 74e0b98..dccc921 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -364,6 +364,109 @@ func TestEvexExtendedGroundTruth(t *testing.T) { } } +// TestEvexHelperGroundTruth covers the floating-point helper and conversion +// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef, +// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions — +// plus gather/scatter with VSIB addressing, byte for byte against the Go +// assembler. +func TestEvexHelperGroundTruth(t *testing.T) { + vsib := func(base, idx string, scale int) Operand { + return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0) + } + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + // Reciprocals and rsqrt (packed RM, scalar NDS). + {"VRCP14PD", "VRCP14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484cd1"}, + {"VRCP14PS", "VRCP14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484cd1"}, + {"VRCP14SD", "VRCP14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084dd9"}, + {"VRCP14SS", "VRCP14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084dd9"}, + {"VRSQRT14PD", "VRSQRT14PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd484ed1"}, + {"VRSQRT14PS", "VRSQRT14PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d484ed1"}, + {"VRSQRT14SD", "VRSQRT14SD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed084fd9"}, + {"VRSQRT14SS", "VRSQRT14SS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d084fd9"}, + // Getexp (packed RM, scalar NDS). + {"VGETEXPPD", "VGETEXPPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f2fd4842d1"}, + {"VGETEXPPS", "VGETEXPPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d4842d1"}, + {"VGETEXPSD", "VGETEXPSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed0843d9"}, + {"VGETEXPSS", "VGETEXPSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d0843d9"}, + // Scalef (NDS). + {"VSCALEFPD", "VSCALEFPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed482cd9"}, + {"VSCALEFPS", "VSCALEFPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482cd9"}, + {"VSCALEFSD", "VSCALEFSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f2ed082dd9"}, + {"VSCALEFSS", "VSCALEFSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f26d082dd9"}, + // Rndscale / getmant / reduce (packed $imm,src,dst; scalar NDS+imm). + {"VRNDSCALEPD", "VRNDSCALEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4809d104"}, + {"VRNDSCALEPS", "VRNDSCALEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4808d104"}, + {"VRNDSCALESD", "VRNDSCALESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed080bd904"}, + {"VRNDSCALESS", "VRNDSCALESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d080ad904"}, + {"VGETMANTPD", "VGETMANTPD", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4826d103"}, + {"VGETMANTPS", "VGETMANTPS", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4826d103"}, + {"VGETMANTSD", "VGETMANTSD", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0827d903"}, + {"VGETMANTSS", "VGETMANTSS", []Operand{Imm(3), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0827d903"}, + {"VREDUCEPD", "VREDUCEPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f3fd4856d104"}, + {"VREDUCEPS", "VREDUCEPS", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2")}, "62f37d4856d104"}, + {"VREDUCESD", "VREDUCESD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0857d904"}, + {"VREDUCESS", "VREDUCESS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0857d904"}, + // Fixupimm / range (NDS + imm8). + {"VFIXUPIMMPD", "VFIXUPIMMPD", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4854d902"}, + {"VFIXUPIMMPS", "VFIXUPIMMPS", []Operand{Imm(2), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4854d902"}, + {"VFIXUPIMMSD", "VFIXUPIMMSD", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0855d902"}, + {"VFIXUPIMMSS", "VFIXUPIMMSS", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0855d902"}, + {"VRANGEPD", "VRANGEPD", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4850d901"}, + {"VRANGEPS", "VRANGEPS", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4850d901"}, + {"VRANGESD", "VRANGESD", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f3ed0851d901"}, + {"VRANGESS", "VRANGESS", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "62f36d0851d901"}, + // FP class test ($imm, src, kdst; packed forms carry the length in + // the X/Y/Z mnemonic suffix the decoder drops). + {"VFPCLASSPDZ", "VFPCLASSPDZ", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "K2")}, "62f3fd4866d104"}, + {"VFPCLASSPSY", "VFPCLASSPSY", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "K2")}, "62f37d2866d104"}, + {"VFPCLASSSD", "VFPCLASSSD", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f3fd0867d104"}, + {"VFPCLASSSS", "VFPCLASSSS", []Operand{Imm(4), vreg(t, "X1"), vreg(t, "K2")}, "62f37d0867d104"}, + // Gather: VEX spelling (mask register, VSIB, destination) and EVEX + // spelling (VSIB, K mask, destination; L'L follows the VSIB index). + {"VGATHERDPS vex", "VGATHERDPS", []Operand{vreg(t, "X2"), vsib("SI", "X1", 4), vreg(t, "X3")}, "c4e269921c8e"}, + {"VPGATHERDD vex", "VPGATHERDD", []Operand{vreg(t, "Y2"), vsib("SI", "Y1", 4), vreg(t, "Y3")}, "c4e26d901c8e"}, + {"VGATHERDPS evex", "VGATHERDPS", []Operand{vsib("SI", "X1", 4), vreg(t, "K2"), vreg(t, "X3")}, "62f27d0a921c8e"}, + {"VPGATHERQD evex", "VPGATHERQD", []Operand{vsib("SI", "Z1", 8), vreg(t, "K2"), vreg(t, "Y3")}, "62f27d4a911cce"}, + // Scatter (EVEX only: source, K mask, VSIB). + {"VSCATTERDPS", "VSCATTERDPS", []Operand{vreg(t, "X3"), vreg(t, "K1"), vsib("SI", "X1", 4)}, "62f27d09a21c8e"}, + {"VSCATTERQPD", "VSCATTERQPD", []Operand{vreg(t, "Z3"), vreg(t, "K1"), vsib("SI", "Z1", 8)}, "62f2fd49a31cce"}, + // The remaining conversions. + {"VCVTDQ2PS", "VCVTDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c485bd1"}, + {"VCVTQQ2PS", "VCVTQQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f1fc485bd1"}, + {"VCVTPD2QQ", "VCVTPD2QQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fd487bd1"}, + {"VCVTPS2QQ", "VCVTPS2QQ", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f17d487bd1"}, + {"VCVTUDQ2PD", "VCVTUDQ2PD", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "62f17e287ad1"}, + {"VCVTPH2PS", "VCVTPH2PS", []Operand{vreg(t, "Y1"), vreg(t, "Z2")}, "62f27d4813d1"}, + {"VCVTPS2PH", "VCVTPS2PH", []Operand{Imm(4), vreg(t, "Y1"), vreg(t, "X2")}, "c4e37d1dca04"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := hexCompact(code); got != c.want { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + continue + } + inst, err := x86asm.Decode(code, 64) + if err != nil { + t.Errorf("%s: Decode(%x): %v", c.name, code, err) + continue + } + want := c.mnem + got := inst.Op.String() + if got != want && !(len(want) > len(got) && want[:len(got)] == got) { + t.Errorf("%s: decoded as %s", c.name, got) + } + } +} + // TestEvexErrors checks the EVEX-specific error paths. func TestEvexErrors(t *testing.T) { cases := []struct { diff --git a/asm/vex.go b/asm/vex.go index b78ae25..3916a40 100644 --- a/asm/vex.go +++ b/asm/vex.go @@ -178,6 +178,9 @@ var vexTable = map[string]vexSpec{ // VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8). "VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract}, "VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract}, + // VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst: + // reg=src, rm=XMM/memory dst, imm8 — the extract layout). + "VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract}, // VEX.128.0F.W0 — no operands. "VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero}, @@ -189,6 +192,9 @@ var vexTable = map[string]vexSpec{ // rm=scalar memory; SD is 256-bit only). "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM}, "VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM}, + // VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width + // source). + "VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM}, // VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src). "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM}, "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM}, diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index 9d38ba2..4cdd39b 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -28,7 +28,7 @@ import ( // version is the release version, stamped at build time via // -ldflags "-X main.version=…" (defaulting to the current release). -var version = "0.13.0" +var version = "0.14.0" func main() { if len(os.Args) < 2 { diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 0467410..a0598cc 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -237,7 +237,14 @@ operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts and extracts, compares with an opmask destination, the permutes, the expand/compress family, the broadcasts, the opmask-register instructions (KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned -moves and the remaining extending/narrowing moves. The EVEX mnemonic +moves and the remaining extending/narrowing moves, the floating-point +helper and conversion tail (VRCP14*, VRSQRT14*, VGETEXP*, VGETMANT*, +VSCALEF*, VRNDSCALE*, VREDUCE*, VFIXUPIMM*, VRANGE*, VFPCLASS* with an +opmask destination, and the remaining VCVT* conversions including the +half-precision pair), and gather/scatter with VSIB addressing — both the +VEX spelling with a vector mask register and the EVEX spelling with an +explicit K mask, where the EVEX length follows the VSIB index register, +not the data register. The EVEX mnemonic suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE), suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX b bit and the L'L rounding-control field (broadcast keeps the vector length diff --git a/justfile b/justfile index 9c4a723..62ef8f5 100644 --- a/justfile +++ b/justfile @@ -3,7 +3,7 @@ # gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm). -version := "0.13.0" +version := "0.14.0" default: @just --list