feat(arch): add the write mask to the packed amd64 destinations
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
631fb8a8d7
commit
c2adde948f
3 files changed
+225
-42
No files matched your search
+149
-42
@@ -23,8 +23,12 @@
|
||||
// packed ones with the {1toN} broadcast, base-relative operands with the
|
||||
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the
|
||||
// scaled index and the broadcast laying the SIB byte and EVEX.b over the
|
||||
// same displacement semantics. Write masking ({k1}{z}) and embedded
|
||||
// rounding still arrive with a later slice.
|
||||
// same displacement semantics. The packed destinations take the write mask
|
||||
// beside all of it, the {k1}{z} decorations the SDM spells: EVEX.aaa carries
|
||||
// the masking register and EVEX.z the zeroing bit, K0 masks nothing, and the
|
||||
// scalar forms, the compares, the opmask destinations and the load and store
|
||||
// shapes take no mask at all. Embedded rounding still arrives with a later
|
||||
// slice.
|
||||
|
||||
package arch
|
||||
|
||||
@@ -49,6 +53,17 @@ func ExtZmm(reg int) ExtOperand { return ExtOperand{Kind: ExtZMM, Reg: reg} }
|
||||
// register number runs 0..7.
|
||||
func ExtMask(reg int) ExtOperand { return ExtOperand{Kind: ExtKReg, Reg: reg} }
|
||||
|
||||
// ExtWriteMasked builds the write-masked spelling of a packed destination,
|
||||
// ZMM1{k7}{z} style: only the lanes mask selects take the result, and the
|
||||
// zeroing flag turns the inactive lanes into zeros instead of keeping the
|
||||
// destination's. The register runs 1..7, K0 never masks.
|
||||
func ExtWriteMasked(dest ExtOperand, mask int, zeroing bool) ExtOperand {
|
||||
dest.Mask = mask
|
||||
dest.HasMask = true
|
||||
dest.Zeroing = zeroing
|
||||
return dest
|
||||
}
|
||||
|
||||
// ExtGpr32 and ExtGpr64 build general-register operands, VCVTSI2SH XMM1,
|
||||
// XMM2, EAX style. The register number runs 0..15.
|
||||
func ExtGpr32(reg int) ExtOperand { return ExtOperand{Kind: ExtR32, Reg: reg} }
|
||||
@@ -135,6 +150,12 @@ func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, er
|
||||
if op.Qual != ExtQualNone {
|
||||
return 0, 0, fmt.Errorf("%s: operand %d carries a predicate qualifier, the amd64 layer takes none", in.Name, pos)
|
||||
}
|
||||
if op.HasMask {
|
||||
return 0, 0, fmt.Errorf("%s: operand %d carries a write mask, the memory operand takes none", in.Name, pos)
|
||||
}
|
||||
if op.Zeroing {
|
||||
return 0, 0, fmt.Errorf("%s: operand %d carries zeroing, the memory operand takes none", in.Name, pos)
|
||||
}
|
||||
if op.HasShift {
|
||||
return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos)
|
||||
}
|
||||
@@ -286,14 +307,60 @@ func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos in
|
||||
return amd64EncodeMemory(b, dest, vvvv, base, disp), nil
|
||||
}
|
||||
|
||||
// amd64WriteMask lifts the write mask spelling out of a destination operand:
|
||||
// the returned copy carries the register bits alone, the mask register and
|
||||
// the zeroing flag come back beside it. K0 never masks, zeroing is valid
|
||||
// only beside a mask, and an entry whose destination takes none refuses the
|
||||
// spellings outright, so nothing arrives at the encoding half-claimed.
|
||||
func (in ExtInstr) amd64WriteMask(op ExtOperand, pos int) (ExtOperand, int, bool, error) {
|
||||
stripped := op
|
||||
stripped.Mask, stripped.HasMask, stripped.Zeroing = 0, false, false
|
||||
if !op.HasMask && !op.Zeroing {
|
||||
return stripped, 0, false, nil
|
||||
}
|
||||
if !in.Mask {
|
||||
return stripped, 0, false, fmt.Errorf("%s: operand %d carries a write mask, the entry's destination takes none", in.Name, pos)
|
||||
}
|
||||
if !op.HasMask {
|
||||
return stripped, 0, false, fmt.Errorf("%s: operand %d carries zeroing without a write mask", in.Name, pos)
|
||||
}
|
||||
if op.Mask < 1 || op.Mask > 7 {
|
||||
return stripped, 0, false, fmt.Errorf("%s: operand %d names write mask k%d, outside the masking registers k1-k7", in.Name, pos, op.Mask)
|
||||
}
|
||||
return stripped, op.Mask, op.Zeroing, nil
|
||||
}
|
||||
|
||||
// amd64ApplyMask lays the validated write mask into the encoded word:
|
||||
// EVEX.aaa, bits two through nought of byte three, carries the masking
|
||||
// register and EVEX.z, bit seven, the zeroing flag. An unmasked destination
|
||||
// leaves the bits the template carries, which the template integrity keeps
|
||||
// zero.
|
||||
func amd64ApplyMask(b []byte, mask int, zeroing bool) {
|
||||
if mask > 0 {
|
||||
b[3] |= byte(mask)
|
||||
}
|
||||
if zeroing {
|
||||
b[3] |= 0x80
|
||||
}
|
||||
}
|
||||
|
||||
// amd64PlainReg checks the invariants every amd64 register operand carries:
|
||||
// no arm64 arrangement, no predicate qualifier, and a register number inside
|
||||
// the class the instruction encodes. A broadcast spelling names a memory
|
||||
// location, so a register position refuses it outright.
|
||||
// location, so a register position refuses it outright, and a write mask or
|
||||
// zeroing decoration belongs to a destination alone, which its encoder lifts
|
||||
// before the position checks see the operand: any spelling that survives to
|
||||
// here sits on a position that takes none.
|
||||
func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
|
||||
if op.Broadcast {
|
||||
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
|
||||
}
|
||||
if op.HasMask {
|
||||
return fmt.Errorf("%s: operand %d carries a write mask, the position takes none", in.Name, pos)
|
||||
}
|
||||
if op.Zeroing {
|
||||
return fmt.Errorf("%s: operand %d carries zeroing, the position takes none", in.Name, pos)
|
||||
}
|
||||
if op.Arr != ExtArrNone {
|
||||
return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
|
||||
}
|
||||
@@ -382,17 +449,35 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
dest, mask, zeroing, err := in.amd64WriteMask(ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return in.amd64MemBytes(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1], 2)
|
||||
if err := in.amd64Vector(dest, class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, ops[0].Reg, ops[1], 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return out, nil
|
||||
}
|
||||
for i, op := range ops[1:] {
|
||||
for i, op := range ops[1:2] {
|
||||
if err := in.amd64Vector(op, class, i+2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
|
||||
dest, mask, zeroing, err := in.amd64WriteMask(ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(dest, class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := amd64Encode(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg)
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// encodeAmdMemVec fills the memory-load form: mem, dest. VMOVSH X30,
|
||||
@@ -432,6 +517,10 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
if in.Form == ExtFormAmdVec2Half {
|
||||
destClass = amd64HalfClass(class)
|
||||
}
|
||||
dest, mask, zeroing, err := in.amd64WriteMask(ops[1], 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 1 && ops[0].Kind == ExtMem {
|
||||
// The memory spelling is validated before the destination: on the
|
||||
// half form the destination's narrowed class is the likelier
|
||||
@@ -439,24 +528,36 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
if _, _, err := in.amd64Memory(ops[0], 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
|
||||
if err := in.amd64Vector(dest, destClass, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1)
|
||||
out, err := in.amd64MemBytes(in.Bytes, dest.Reg, -1, ops[0], 1)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return out, nil
|
||||
}
|
||||
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, ops[1], 2)
|
||||
out, err := in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, dest, 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return out, nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
|
||||
if err := in.amd64Vector(dest, destClass, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil
|
||||
out := amd64Encode(in.Bytes, dest.Reg, -1, ops[0].Reg)
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// encodeAmdMask2 fills the mask-destination form: src1, src2, dest, where the
|
||||
@@ -494,6 +595,12 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
|
||||
if op.HasShift {
|
||||
return 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 imm8 forms take none", in.Name, pos)
|
||||
}
|
||||
if op.HasMask {
|
||||
return 0, fmt.Errorf("%s: operand %d carries a write mask, the immediate takes none", in.Name, pos)
|
||||
}
|
||||
if op.Zeroing {
|
||||
return 0, fmt.Errorf("%s: operand %d carries zeroing, the immediate takes none", in.Name, pos)
|
||||
}
|
||||
if op.Imm < 0 || op.Imm > 255 {
|
||||
return 0, fmt.Errorf("%s: operand %d is immediate %d, outside the unsigned byte range 0-255", in.Name, pos, op.Imm)
|
||||
}
|
||||
@@ -660,31 +767,31 @@ var amd64Extensions = []ExtInstr{
|
||||
// three-register convert carries F2 while the narrow convert and the dot
|
||||
// product carry F3, which the golden vectors pin byte for byte.
|
||||
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.512.F2.0F38.W0 72 /r)"},
|
||||
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.256.F2.0F38.W0 72 /r)"},
|
||||
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.128.F2.0F38.W0 72 /r)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.512.F3.0F38.W0 72 /r, YMM destination)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.256.F3.0F38.W0 72 /r, XMM destination)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"},
|
||||
|
||||
// AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask
|
||||
@@ -813,67 +920,67 @@ var amd64Extensions = []ExtInstr{
|
||||
// quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from
|
||||
// avx512_fp16_vl.d, on the same low registers the suite uses.
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"},
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"},
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"},
|
||||
|
||||
// AVX512-FP16 scalar, the imm8-control group: mantissa extraction,
|
||||
|
||||
@@ -523,6 +523,34 @@ var amd64GoldenRows = []amd64GoldenRow{
|
||||
{"vcvtne2ps2bf16 high registers", "VCVTNE2PS2BF16",
|
||||
[]ExtOperand{ExtZmm(21), ExtZmm(20), ExtZmm(23)},
|
||||
"62a2574072fc", ""},
|
||||
|
||||
// The write mask, the SDM's {k1}{z} decorations on the packed
|
||||
// destinations: EVEX.aaa carries the masking register, EVEX.z the
|
||||
// zeroing bit, laid over the words the unmasked rows above prove byte
|
||||
// for byte (the x86-64-avx512_fp16.d and avx512_bf16.d listings carry
|
||||
// the same {k7}{z} rows). Zeroing keeps the destination's inactive
|
||||
// lanes no longer: they become zeros.
|
||||
{"vaddph k7 zeroing", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)},
|
||||
"620514c758f4", ""},
|
||||
{"vaddph k1 merging", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 1, false)},
|
||||
"6205144158f4", ""},
|
||||
{"vaddph k7 zeroing, memory source", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(28), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)},
|
||||
"62451cc75831", ""},
|
||||
{"vcvtne2ps2bf16 k7 zeroing", "VCVTNE2PS2BF16",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtWriteMasked(ExtZmm(6), 7, true)},
|
||||
"62f257cf72f4", ""},
|
||||
{"vdpbf16ps k5 merging", "VDPBF16PS",
|
||||
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtWriteMasked(ExtZmm(6), 5, false)},
|
||||
"62f2564d52f4", ""},
|
||||
{"vcvtneps2bf16 k6 merging", "VCVTNEPS2BF16",
|
||||
[]ExtOperand{ExtYmm(5), ExtWriteMasked(ExtXmm(6), 6, false)},
|
||||
"62f27e2e72f5", ""},
|
||||
{"vsqrtph k3 zeroing", "VSQRTPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtWriteMasked(ExtZmm(30), 3, true)},
|
||||
"62057ccb51f5", ""},
|
||||
}
|
||||
|
||||
// amd64ResolveEntry finds the table entry a golden row exercises: the entry
|
||||
@@ -778,6 +806,30 @@ func TestAmd64ExtRejects(t *testing.T) {
|
||||
{"a scale the multipliers do not carry", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtScaledMemory(1, 14, 3, 0), ExtZmm(30)},
|
||||
"outside the byte multipliers"},
|
||||
{"K0 as the write mask", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(28), ExtZmm(30), ExtWriteMasked(ExtZmm(29), 0, true)},
|
||||
"outside the masking registers k1-k7"},
|
||||
{"zeroing without a write mask", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(28), ExtZmm(30), ExtOperand{Kind: ExtZMM, Reg: 29, Zeroing: true}},
|
||||
"carries zeroing without a write mask"},
|
||||
{"write mask on the scalar arithmetic", "VADDSH",
|
||||
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtWriteMasked(ExtXmm(30), 3, true)},
|
||||
"the entry's destination takes none"},
|
||||
{"write mask on the compare without a vector destination", "VCOMISH",
|
||||
[]ExtOperand{ExtXmm(29), ExtWriteMasked(ExtXmm(30), 2, false)},
|
||||
"the entry's destination takes none"},
|
||||
{"write mask on the intersection's vector source", "VP2INTERSECTD",
|
||||
[]ExtOperand{ExtWriteMasked(ExtZmm(2), 3, false), ExtZmm(1), ExtMask(0)},
|
||||
"the position takes none"},
|
||||
{"write mask on a source position", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtWriteMasked(ExtZmm(28), 3, false), ExtZmm(30)},
|
||||
"the position takes none"},
|
||||
{"write mask on the control form's destination", "VGETMANTSH",
|
||||
[]ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtXmm(28), ExtWriteMasked(ExtXmm(30), 7, true)},
|
||||
"the position takes none"},
|
||||
{"write mask where the destination is memory", "VCOMISH",
|
||||
[]ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, HasMask: true, Mask: 2}},
|
||||
"the entry's destination takes none"},
|
||||
} {
|
||||
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
|
||||
_, err := in.Encode(tt.ops)
|
||||
@@ -835,6 +887,12 @@ func TestAmd64ExtMemoryFormRejects(t *testing.T) {
|
||||
{"broadcast destination on the store", "VMOVSH", store,
|
||||
[]ExtOperand{ExtXmm(30), ExtBroadcast(9, 0)},
|
||||
"carries a broadcast, the entry's memory operand takes none"},
|
||||
{"write mask on the store's memory destination", "VMOVSH", store,
|
||||
[]ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, HasMask: true, Mask: 2}},
|
||||
"the memory operand takes none"},
|
||||
{"zeroing on the store's memory destination", "VMOVSH", store,
|
||||
[]ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, Zeroing: true}},
|
||||
"the memory operand takes none"},
|
||||
} {
|
||||
in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick)
|
||||
_, err := in.Encode(tt.ops)
|
||||
|
||||
@@ -183,6 +183,17 @@ type ExtOperand struct {
|
||||
Index int
|
||||
Scale int
|
||||
HasIndex bool
|
||||
// Mask spells the write mask of an amd64 EVEX destination, the {k1}
|
||||
// through {k7} decorations: only the masked lanes take the result.
|
||||
// HasMask separates a spelled mask from the unmasked destination, and
|
||||
// K0 never masks, so the register runs 1..7. Only the packed
|
||||
// destinations of the entries that carry Mask accept it.
|
||||
Mask int
|
||||
HasMask bool
|
||||
// Zeroing spells the {z} decoration beside a write mask: the inactive
|
||||
// lanes become zero instead of keeping the destination. It is valid
|
||||
// only together with a spelled mask.
|
||||
Zeroing bool
|
||||
}
|
||||
|
||||
// ExtVector builds a scalable vector operand, ADD Z1.S style.
|
||||
@@ -503,6 +514,13 @@ type ExtInstr struct {
|
||||
// forms and the full-width sources do not. The arm64 entries all
|
||||
// carry the zero value.
|
||||
Bcast bool
|
||||
// Mask records that the entry's packed destination takes the write
|
||||
// mask, the SDM's {k1}{z} decorations: EVEX.aaa carries the masking
|
||||
// register and EVEX.z the zeroing bit. The packed forms carry it; the
|
||||
// scalar forms, the compares without a vector destination, the
|
||||
// opmask-destination forms and the load and store shapes do not. The
|
||||
// arm64 entries all carry the zero value.
|
||||
Mask bool
|
||||
}
|
||||
|
||||
// Encode assembles the operands into the 4 little-endian bytes of the
|
||||
|
||||
Reference in new issue
Block a user