From c2adde948f81a2cd5795744a138610f8c9adf13a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 02:20:51 +0200 Subject: [PATCH] feat(arch): add the write mask to the packed amd64 destinations Assisted-by: GLM 5.3 Flash --- arch/amd64_ext.go | 191 ++++++++++++++++++++++++++++++++--------- arch/amd64_ext_test.go | 58 +++++++++++++ arch/arm64_ext.go | 18 ++++ 3 files changed, 225 insertions(+), 42 deletions(-) diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index f65fe25..e85ca81 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -23,8 +23,12 @@ // packed ones with the {1toN} broadcast, base-relative operands with the // ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the // scaled index and the broadcast laying the SIB byte and EVEX.b over the -// same displacement semantics. Write masking ({k1}{z}) and embedded -// rounding still arrive with a later slice. +// same displacement semantics. The packed destinations take the write mask +// beside all of it, the {k1}{z} decorations the SDM spells: EVEX.aaa carries +// the masking register and EVEX.z the zeroing bit, K0 masks nothing, and the +// scalar forms, the compares, the opmask destinations and the load and store +// shapes take no mask at all. Embedded rounding still arrives with a later +// slice. package arch @@ -49,6 +53,17 @@ func ExtZmm(reg int) ExtOperand { return ExtOperand{Kind: ExtZMM, Reg: reg} } // register number runs 0..7. func ExtMask(reg int) ExtOperand { return ExtOperand{Kind: ExtKReg, Reg: reg} } +// ExtWriteMasked builds the write-masked spelling of a packed destination, +// ZMM1{k7}{z} style: only the lanes mask selects take the result, and the +// zeroing flag turns the inactive lanes into zeros instead of keeping the +// destination's. The register runs 1..7, K0 never masks. +func ExtWriteMasked(dest ExtOperand, mask int, zeroing bool) ExtOperand { + dest.Mask = mask + dest.HasMask = true + dest.Zeroing = zeroing + return dest +} + // ExtGpr32 and ExtGpr64 build general-register operands, VCVTSI2SH XMM1, // XMM2, EAX style. The register number runs 0..15. func ExtGpr32(reg int) ExtOperand { return ExtOperand{Kind: ExtR32, Reg: reg} } @@ -135,6 +150,12 @@ func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, er if op.Qual != ExtQualNone { return 0, 0, fmt.Errorf("%s: operand %d carries a predicate qualifier, the amd64 layer takes none", in.Name, pos) } + if op.HasMask { + return 0, 0, fmt.Errorf("%s: operand %d carries a write mask, the memory operand takes none", in.Name, pos) + } + if op.Zeroing { + return 0, 0, fmt.Errorf("%s: operand %d carries zeroing, the memory operand takes none", in.Name, pos) + } if op.HasShift { return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos) } @@ -286,14 +307,60 @@ func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos in return amd64EncodeMemory(b, dest, vvvv, base, disp), nil } +// amd64WriteMask lifts the write mask spelling out of a destination operand: +// the returned copy carries the register bits alone, the mask register and +// the zeroing flag come back beside it. K0 never masks, zeroing is valid +// only beside a mask, and an entry whose destination takes none refuses the +// spellings outright, so nothing arrives at the encoding half-claimed. +func (in ExtInstr) amd64WriteMask(op ExtOperand, pos int) (ExtOperand, int, bool, error) { + stripped := op + stripped.Mask, stripped.HasMask, stripped.Zeroing = 0, false, false + if !op.HasMask && !op.Zeroing { + return stripped, 0, false, nil + } + if !in.Mask { + return stripped, 0, false, fmt.Errorf("%s: operand %d carries a write mask, the entry's destination takes none", in.Name, pos) + } + if !op.HasMask { + return stripped, 0, false, fmt.Errorf("%s: operand %d carries zeroing without a write mask", in.Name, pos) + } + if op.Mask < 1 || op.Mask > 7 { + return stripped, 0, false, fmt.Errorf("%s: operand %d names write mask k%d, outside the masking registers k1-k7", in.Name, pos, op.Mask) + } + return stripped, op.Mask, op.Zeroing, nil +} + +// amd64ApplyMask lays the validated write mask into the encoded word: +// EVEX.aaa, bits two through nought of byte three, carries the masking +// register and EVEX.z, bit seven, the zeroing flag. An unmasked destination +// leaves the bits the template carries, which the template integrity keeps +// zero. +func amd64ApplyMask(b []byte, mask int, zeroing bool) { + if mask > 0 { + b[3] |= byte(mask) + } + if zeroing { + b[3] |= 0x80 + } +} + // amd64PlainReg checks the invariants every amd64 register operand carries: // no arm64 arrangement, no predicate qualifier, and a register number inside // the class the instruction encodes. A broadcast spelling names a memory -// location, so a register position refuses it outright. +// location, so a register position refuses it outright, and a write mask or +// zeroing decoration belongs to a destination alone, which its encoder lifts +// before the position checks see the operand: any spelling that survives to +// here sits on a position that takes none. func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error { if op.Broadcast { return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos) } + if op.HasMask { + return fmt.Errorf("%s: operand %d carries a write mask, the position takes none", in.Name, pos) + } + if op.Zeroing { + return fmt.Errorf("%s: operand %d carries zeroing, the position takes none", in.Name, pos) + } if op.Arr != ExtArrNone { return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos) } @@ -382,17 +449,35 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { return nil, err } if in.Mem == 2 && ops[1].Kind == ExtMem { - if err := in.amd64Vector(ops[2], class, 3); err != nil { + dest, mask, zeroing, err := in.amd64WriteMask(ops[2], 3) + if err != nil { return nil, err } - return in.amd64MemBytes(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1], 2) + if err := in.amd64Vector(dest, class, 3); err != nil { + return nil, err + } + out, err := in.amd64MemBytes(in.Bytes, dest.Reg, ops[0].Reg, ops[1], 2) + if err != nil { + return nil, err + } + amd64ApplyMask(out, mask, zeroing) + return out, nil } - for i, op := range ops[1:] { + for i, op := range ops[1:2] { if err := in.amd64Vector(op, class, i+2); err != nil { return nil, err } } - return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil + dest, mask, zeroing, err := in.amd64WriteMask(ops[2], 3) + if err != nil { + return nil, err + } + if err := in.amd64Vector(dest, class, 3); err != nil { + return nil, err + } + out := amd64Encode(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg) + amd64ApplyMask(out, mask, zeroing) + return out, nil } // encodeAmdMemVec fills the memory-load form: mem, dest. VMOVSH X30, @@ -432,6 +517,10 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { if in.Form == ExtFormAmdVec2Half { destClass = amd64HalfClass(class) } + dest, mask, zeroing, err := in.amd64WriteMask(ops[1], 2) + if err != nil { + return nil, err + } if in.Mem == 1 && ops[0].Kind == ExtMem { // The memory spelling is validated before the destination: on the // half form the destination's narrowed class is the likelier @@ -439,24 +528,36 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { if _, _, err := in.amd64Memory(ops[0], 1); err != nil { return nil, err } - if err := in.amd64Vector(ops[1], destClass, 2); err != nil { + if err := in.amd64Vector(dest, destClass, 2); err != nil { return nil, err } - return in.amd64MemBytes(in.Bytes, ops[1].Reg, -1, ops[0], 1) + out, err := in.amd64MemBytes(in.Bytes, dest.Reg, -1, ops[0], 1) + if err != nil { + return nil, err + } + amd64ApplyMask(out, mask, zeroing) + return out, nil } if in.Mem == 2 && ops[1].Kind == ExtMem { if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err } - return in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, ops[1], 2) + out, err := in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, dest, 2) + if err != nil { + return nil, err + } + amd64ApplyMask(out, mask, zeroing) + return out, nil } if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err } - if err := in.amd64Vector(ops[1], destClass, 2); err != nil { + if err := in.amd64Vector(dest, destClass, 2); err != nil { return nil, err } - return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil + out := amd64Encode(in.Bytes, dest.Reg, -1, ops[0].Reg) + amd64ApplyMask(out, mask, zeroing) + return out, nil } // encodeAmdMask2 fills the mask-destination form: src1, src2, dest, where the @@ -494,6 +595,12 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) { if op.HasShift { return 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 imm8 forms take none", in.Name, pos) } + if op.HasMask { + return 0, fmt.Errorf("%s: operand %d carries a write mask, the immediate takes none", in.Name, pos) + } + if op.Zeroing { + return 0, fmt.Errorf("%s: operand %d carries zeroing, the immediate takes none", in.Name, pos) + } if op.Imm < 0 || op.Imm > 255 { return 0, fmt.Errorf("%s: operand %d is immediate %d, outside the unsigned byte range 0-255", in.Name, pos, op.Imm) } @@ -660,31 +767,31 @@ var amd64Extensions = []ExtInstr{ // three-register convert carries F2 while the narrow convert and the dot // product carry F3, which the golden vectors pin byte for byte. {Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating", - Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.512.F2.0F38.W0 72 /r)"}, {Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating", - Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.256.F2.0F38.W0 72 /r)"}, {Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating", - Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Mask: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.128.F2.0F38.W0 72 /r)"}, {Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination", - Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.512.F3.0F38.W0 72 /r, YMM destination)"}, {Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination", - Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.256.F3.0F38.W0 72 /r, XMM destination)"}, {Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination", - Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Mask: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"}, {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", - Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"}, {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", - Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"}, {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", - Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"}, // AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask @@ -813,67 +920,67 @@ var amd64Extensions = []ExtInstr{ // quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from // avx512_fp16_vl.d, on the same low registers the suite uses. {Name: "VADDPH", Summary: "Add packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"}, {Name: "VADDPH", Summary: "Add packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"}, {Name: "VADDPH", Summary: "Add packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"}, {Name: "VSUBPH", Summary: "Subtract packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"}, {Name: "VSUBPH", Summary: "Subtract packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"}, {Name: "VSUBPH", Summary: "Subtract packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"}, {Name: "VMULPH", Summary: "Multiply packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"}, {Name: "VMULPH", Summary: "Multiply packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"}, {Name: "VMULPH", Summary: "Multiply packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"}, {Name: "VDIVPH", Summary: "Divide packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"}, {Name: "VDIVPH", Summary: "Divide packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"}, {Name: "VDIVPH", Summary: "Divide packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"}, {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"}, {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"}, {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"}, {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"}, {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"}, {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"}, {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"}, {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"}, {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"}, // AVX512-FP16 scalar, the imm8-control group: mantissa extraction, diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 1f3379b..703c7bc 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -523,6 +523,34 @@ var amd64GoldenRows = []amd64GoldenRow{ {"vcvtne2ps2bf16 high registers", "VCVTNE2PS2BF16", []ExtOperand{ExtZmm(21), ExtZmm(20), ExtZmm(23)}, "62a2574072fc", ""}, + + // The write mask, the SDM's {k1}{z} decorations on the packed + // destinations: EVEX.aaa carries the masking register, EVEX.z the + // zeroing bit, laid over the words the unmasked rows above prove byte + // for byte (the x86-64-avx512_fp16.d and avx512_bf16.d listings carry + // the same {k7}{z} rows). Zeroing keeps the destination's inactive + // lanes no longer: they become zeros. + {"vaddph k7 zeroing", "VADDPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 7, true)}, + "620514c758f4", ""}, + {"vaddph k1 merging", "VADDPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtWriteMasked(ExtZmm(30), 1, false)}, + "6205144158f4", ""}, + {"vaddph k7 zeroing, memory source", "VADDPH", + []ExtOperand{ExtZmm(28), ExtMemory(9, 0), ExtWriteMasked(ExtZmm(30), 7, true)}, + "62451cc75831", ""}, + {"vcvtne2ps2bf16 k7 zeroing", "VCVTNE2PS2BF16", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtWriteMasked(ExtZmm(6), 7, true)}, + "62f257cf72f4", ""}, + {"vdpbf16ps k5 merging", "VDPBF16PS", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtWriteMasked(ExtZmm(6), 5, false)}, + "62f2564d52f4", ""}, + {"vcvtneps2bf16 k6 merging", "VCVTNEPS2BF16", + []ExtOperand{ExtYmm(5), ExtWriteMasked(ExtXmm(6), 6, false)}, + "62f27e2e72f5", ""}, + {"vsqrtph k3 zeroing", "VSQRTPH", + []ExtOperand{ExtZmm(29), ExtWriteMasked(ExtZmm(30), 3, true)}, + "62057ccb51f5", ""}, } // amd64ResolveEntry finds the table entry a golden row exercises: the entry @@ -778,6 +806,30 @@ func TestAmd64ExtRejects(t *testing.T) { {"a scale the multipliers do not carry", "VADDPH", []ExtOperand{ExtZmm(29), ExtScaledMemory(1, 14, 3, 0), ExtZmm(30)}, "outside the byte multipliers"}, + {"K0 as the write mask", "VADDPH", + []ExtOperand{ExtZmm(28), ExtZmm(30), ExtWriteMasked(ExtZmm(29), 0, true)}, + "outside the masking registers k1-k7"}, + {"zeroing without a write mask", "VADDPH", + []ExtOperand{ExtZmm(28), ExtZmm(30), ExtOperand{Kind: ExtZMM, Reg: 29, Zeroing: true}}, + "carries zeroing without a write mask"}, + {"write mask on the scalar arithmetic", "VADDSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtWriteMasked(ExtXmm(30), 3, true)}, + "the entry's destination takes none"}, + {"write mask on the compare without a vector destination", "VCOMISH", + []ExtOperand{ExtXmm(29), ExtWriteMasked(ExtXmm(30), 2, false)}, + "the entry's destination takes none"}, + {"write mask on the intersection's vector source", "VP2INTERSECTD", + []ExtOperand{ExtWriteMasked(ExtZmm(2), 3, false), ExtZmm(1), ExtMask(0)}, + "the position takes none"}, + {"write mask on a source position", "VADDPH", + []ExtOperand{ExtZmm(29), ExtWriteMasked(ExtZmm(28), 3, false), ExtZmm(30)}, + "the position takes none"}, + {"write mask on the control form's destination", "VGETMANTSH", + []ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtXmm(28), ExtWriteMasked(ExtXmm(30), 7, true)}, + "the position takes none"}, + {"write mask where the destination is memory", "VCOMISH", + []ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, HasMask: true, Mask: 2}}, + "the entry's destination takes none"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) @@ -835,6 +887,12 @@ func TestAmd64ExtMemoryFormRejects(t *testing.T) { {"broadcast destination on the store", "VMOVSH", store, []ExtOperand{ExtXmm(30), ExtBroadcast(9, 0)}, "carries a broadcast, the entry's memory operand takes none"}, + {"write mask on the store's memory destination", "VMOVSH", store, + []ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, HasMask: true, Mask: 2}}, + "the memory operand takes none"}, + {"zeroing on the store's memory destination", "VMOVSH", store, + []ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, Zeroing: true}}, + "the memory operand takes none"}, } { in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick) _, err := in.Encode(tt.ops) diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index 800c689..f01a812 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -183,6 +183,17 @@ type ExtOperand struct { Index int Scale int HasIndex bool + // Mask spells the write mask of an amd64 EVEX destination, the {k1} + // through {k7} decorations: only the masked lanes take the result. + // HasMask separates a spelled mask from the unmasked destination, and + // K0 never masks, so the register runs 1..7. Only the packed + // destinations of the entries that carry Mask accept it. + Mask int + HasMask bool + // Zeroing spells the {z} decoration beside a write mask: the inactive + // lanes become zero instead of keeping the destination. It is valid + // only together with a spelled mask. + Zeroing bool } // ExtVector builds a scalable vector operand, ADD Z1.S style. @@ -503,6 +514,13 @@ type ExtInstr struct { // forms and the full-width sources do not. The arm64 entries all // carry the zero value. Bcast bool + // Mask records that the entry's packed destination takes the write + // mask, the SDM's {k1}{z} decorations: EVEX.aaa carries the masking + // register and EVEX.z the zeroing bit. The packed forms carry it; the + // scalar forms, the compares without a vector destination, the + // opmask-destination forms and the load and store shapes do not. The + // arm64 entries all carry the zero value. + Mask bool } // Encode assembles the operands into the 4 little-endian bytes of the