feat(arch): add the {1toN} broadcast to the packed amd64 memory sources

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:06:11 +02:00
1 parent 9c1392dfc3
commit ccb155437e
4 files changed
+218 -32

No files matched your search

+68 -32
View File
@@ -19,10 +19,12 @@
// toolchain has is not an extension.
//
// The forms encode the unmasked shapes: register forms throughout, and the
// scalar FP16 memory forms beside them, base-relative operands with the
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand. A
// scaled index, write masking ({k1}{z}) and embedded rounding still arrive
// with a later slice.
// memory forms beside them, the scalar ones the manual spells m16 and the
// packed ones with the {1toN} broadcast, base-relative operands with the
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the
// broadcast laying EVEX.b over the same displacement semantics. A scaled
// index, write masking ({k1}{z}) and embedded rounding still arrive with a
// later slice.
package arch
@@ -120,7 +122,9 @@ func amd64Encode(b []byte, dest, vvvv, rm int) []byte {
// amd64Memory validates a memory operand of an amd64 entry: no arrangement
// and no qualifier, a base general register inside 0-15, a signed 32-bit
// displacement and no shift. The base number rides the operand's Reg and
// the displacement its Imm.
// the displacement its Imm. A broadcast spelling is refused unless the
// entry carries Bcast: the scalar forms read a plain m16 and the full-width
// sources a plain vector, and neither splats.
func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, err error) {
if op.Kind != ExtMem {
return 0, 0, fmt.Errorf("%s: operand %d wants a memory operand, got %s", in.Name, pos, op.Kind)
@@ -134,6 +138,9 @@ func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, er
if op.HasShift {
return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos)
}
if op.Broadcast && !in.Bcast {
return 0, 0, fmt.Errorf("%s: operand %d carries a broadcast, the entry's memory operand takes none", in.Name, pos)
}
if op.Reg < 0 || op.Reg > 15 {
return 0, 0, fmt.Errorf("%s: operand %d names base register %d, outside 0-15", in.Name, pos, op.Reg)
}
@@ -179,10 +186,26 @@ func amd64EncodeMemory(b []byte, dest, vvvv, base int, disp int64) []byte {
return append(out, tail...)
}
// amd64EncodeBroadcast returns the memory encoding with EVEX.b set: the
// {1toN} broadcast, whose single element the hardware splats across every
// lane of the destination. EVEX.b is bit 4 of byte three, and the ModR/M,
// SIB and displacement bytes keep the plain semantics amd64EncodeMemory
// chooses; only the prefix bit changes. The operand must have passed
// amd64Memory on an entry that carries Bcast.
func amd64EncodeBroadcast(b []byte, dest, vvvv, base int, disp int64) []byte {
out := amd64EncodeMemory(b, dest, vvvv, base, disp)
out[3] |= 0x10
return out
}
// amd64PlainReg checks the invariants every amd64 register operand carries:
// no arm64 arrangement, no predicate qualifier, and a register number inside
// the class the instruction encodes.
// the class the instruction encodes. A broadcast spelling names a memory
// location, so a register position refuses it outright.
func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
if op.Broadcast {
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
}
if op.Arr != ExtArrNone {
return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
}
@@ -196,7 +219,12 @@ func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
}
// amd64Vector checks one vector operand against the class the entry encodes.
// The broadcast spelling is named before the kind, so the diagnostic says
// what the operand carries rather than what the position wanted.
func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error {
if op.Broadcast {
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
}
if op.Kind != class {
article := "a"
if class == ExtXMM {
@@ -258,7 +286,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
// memory shape of that position too: the second source of the scalar
// arithmetic, spelled xmm3/m16 in the manual, may be a base-relative
// operand, which rides the r/m field with its displacement bytes after the
// opcode.
// opcode, and the packed entries lay the source's {1toN} broadcast over the
// same encoding as EVEX.b.
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
@@ -272,6 +301,9 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
if ops[1].Broadcast {
return amd64EncodeBroadcast(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
}
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
}
for i, op := range ops[1:] {
@@ -318,7 +350,8 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
// of destination, and at 128 bits the companion stays the class itself. An
// entry with Mem set takes the memory shape of that position too: the
// compares and the packed square root read their source from memory, and
// compares and the packed square root read their source from memory, the
// packed square root's source carrying the {1toN} broadcast as EVEX.b, and
// the narrow BF16 convert reads its full-width source there.
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
@@ -334,6 +367,9 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
return nil, err
}
if ops[0].Broadcast {
return amd64EncodeBroadcast(in.Bytes, ops[1].Reg, -1, base, disp), nil
}
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
@@ -580,13 +616,13 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"},
// AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask
@@ -715,67 +751,67 @@ var amd64Extensions = []ExtInstr{
// quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from
// avx512_fp16_vl.d, on the same low registers the suite uses.
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"},
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"},
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"},
// AVX512-FP16 scalar, the imm8-control group: mantissa extraction,