feat(arch): add the {1toN} broadcast to the packed amd64 memory sources
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
9c1392dfc3
commit
ccb155437e
4 files changed
+218
-32
No files matched your search
+68
-32
@@ -19,10 +19,12 @@
|
||||
// toolchain has is not an extension.
|
||||
//
|
||||
// The forms encode the unmasked shapes: register forms throughout, and the
|
||||
// scalar FP16 memory forms beside them, base-relative operands with the
|
||||
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand. A
|
||||
// scaled index, write masking ({k1}{z}) and embedded rounding still arrive
|
||||
// with a later slice.
|
||||
// memory forms beside them, the scalar ones the manual spells m16 and the
|
||||
// packed ones with the {1toN} broadcast, base-relative operands with the
|
||||
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the
|
||||
// broadcast laying EVEX.b over the same displacement semantics. A scaled
|
||||
// index, write masking ({k1}{z}) and embedded rounding still arrive with a
|
||||
// later slice.
|
||||
|
||||
package arch
|
||||
|
||||
@@ -120,7 +122,9 @@ func amd64Encode(b []byte, dest, vvvv, rm int) []byte {
|
||||
// amd64Memory validates a memory operand of an amd64 entry: no arrangement
|
||||
// and no qualifier, a base general register inside 0-15, a signed 32-bit
|
||||
// displacement and no shift. The base number rides the operand's Reg and
|
||||
// the displacement its Imm.
|
||||
// the displacement its Imm. A broadcast spelling is refused unless the
|
||||
// entry carries Bcast: the scalar forms read a plain m16 and the full-width
|
||||
// sources a plain vector, and neither splats.
|
||||
func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, err error) {
|
||||
if op.Kind != ExtMem {
|
||||
return 0, 0, fmt.Errorf("%s: operand %d wants a memory operand, got %s", in.Name, pos, op.Kind)
|
||||
@@ -134,6 +138,9 @@ func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, er
|
||||
if op.HasShift {
|
||||
return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos)
|
||||
}
|
||||
if op.Broadcast && !in.Bcast {
|
||||
return 0, 0, fmt.Errorf("%s: operand %d carries a broadcast, the entry's memory operand takes none", in.Name, pos)
|
||||
}
|
||||
if op.Reg < 0 || op.Reg > 15 {
|
||||
return 0, 0, fmt.Errorf("%s: operand %d names base register %d, outside 0-15", in.Name, pos, op.Reg)
|
||||
}
|
||||
@@ -179,10 +186,26 @@ func amd64EncodeMemory(b []byte, dest, vvvv, base int, disp int64) []byte {
|
||||
return append(out, tail...)
|
||||
}
|
||||
|
||||
// amd64EncodeBroadcast returns the memory encoding with EVEX.b set: the
|
||||
// {1toN} broadcast, whose single element the hardware splats across every
|
||||
// lane of the destination. EVEX.b is bit 4 of byte three, and the ModR/M,
|
||||
// SIB and displacement bytes keep the plain semantics amd64EncodeMemory
|
||||
// chooses; only the prefix bit changes. The operand must have passed
|
||||
// amd64Memory on an entry that carries Bcast.
|
||||
func amd64EncodeBroadcast(b []byte, dest, vvvv, base int, disp int64) []byte {
|
||||
out := amd64EncodeMemory(b, dest, vvvv, base, disp)
|
||||
out[3] |= 0x10
|
||||
return out
|
||||
}
|
||||
|
||||
// amd64PlainReg checks the invariants every amd64 register operand carries:
|
||||
// no arm64 arrangement, no predicate qualifier, and a register number inside
|
||||
// the class the instruction encodes.
|
||||
// the class the instruction encodes. A broadcast spelling names a memory
|
||||
// location, so a register position refuses it outright.
|
||||
func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
|
||||
if op.Broadcast {
|
||||
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
|
||||
}
|
||||
if op.Arr != ExtArrNone {
|
||||
return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
|
||||
}
|
||||
@@ -196,7 +219,12 @@ func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
|
||||
}
|
||||
|
||||
// amd64Vector checks one vector operand against the class the entry encodes.
|
||||
// The broadcast spelling is named before the kind, so the diagnostic says
|
||||
// what the operand carries rather than what the position wanted.
|
||||
func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error {
|
||||
if op.Broadcast {
|
||||
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
|
||||
}
|
||||
if op.Kind != class {
|
||||
article := "a"
|
||||
if class == ExtXMM {
|
||||
@@ -258,7 +286,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
||||
// memory shape of that position too: the second source of the scalar
|
||||
// arithmetic, spelled xmm3/m16 in the manual, may be a base-relative
|
||||
// operand, which rides the r/m field with its displacement bytes after the
|
||||
// opcode.
|
||||
// opcode, and the packed entries lay the source's {1toN} broadcast over the
|
||||
// same encoding as EVEX.b.
|
||||
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
@@ -272,6 +301,9 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if ops[1].Broadcast {
|
||||
return amd64EncodeBroadcast(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
|
||||
}
|
||||
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
|
||||
}
|
||||
for i, op := range ops[1:] {
|
||||
@@ -318,7 +350,8 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
|
||||
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
|
||||
// of destination, and at 128 bits the companion stays the class itself. An
|
||||
// entry with Mem set takes the memory shape of that position too: the
|
||||
// compares and the packed square root read their source from memory, and
|
||||
// compares and the packed square root read their source from memory, the
|
||||
// packed square root's source carrying the {1toN} broadcast as EVEX.b, and
|
||||
// the narrow BF16 convert reads its full-width source there.
|
||||
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
@@ -334,6 +367,9 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if ops[0].Broadcast {
|
||||
return amd64EncodeBroadcast(in.Bytes, ops[1].Reg, -1, base, disp), nil
|
||||
}
|
||||
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
|
||||
}
|
||||
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
||||
@@ -580,13 +616,13 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"},
|
||||
|
||||
// AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask
|
||||
@@ -715,67 +751,67 @@ var amd64Extensions = []ExtInstr{
|
||||
// quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from
|
||||
// avx512_fp16_vl.d, on the same low registers the suite uses.
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"},
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"},
|
||||
{Name: "VADDPH", Summary: "Add packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"},
|
||||
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"},
|
||||
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"},
|
||||
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"},
|
||||
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"},
|
||||
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"},
|
||||
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"},
|
||||
|
||||
// AVX512-FP16 scalar, the imm8-control group: mantissa extraction,
|
||||
|
||||
@@ -76,6 +76,8 @@ func TestAmd64ExtMemoryRejects(t *testing.T) {
|
||||
"arrangement"},
|
||||
{"predicate qualifier on the memory operand", ExtOperand{Kind: ExtMem, Reg: 8, Qual: ExtQualZeroing},
|
||||
"predicate qualifier"},
|
||||
{"broadcast on an entry that takes none", ExtBroadcast(8, 0),
|
||||
"takes none"},
|
||||
} {
|
||||
_, _, err := in.amd64Memory(tt.op, 1)
|
||||
if err == nil {
|
||||
@@ -86,6 +88,57 @@ func TestAmd64ExtMemoryRejects(t *testing.T) {
|
||||
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
|
||||
}
|
||||
}
|
||||
// The broadcast spelling passes the flag gate on an entry that carries
|
||||
// Bcast and then meets the same base and displacement checks.
|
||||
bcast := ExtInstr{Name: "TEST", Bcast: true}
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
op ExtOperand
|
||||
quote string
|
||||
}{
|
||||
{"broadcast base beyond r15", ExtOperand{Kind: ExtMem, Reg: 16, Imm: 0, Broadcast: true},
|
||||
"outside 0-15"},
|
||||
{"broadcast displacement past the signed 32-bit ceiling", ExtBroadcast(8, 1<<31),
|
||||
"outside the signed 32-bit range"},
|
||||
{"broadcast base under r0", ExtBroadcast(-1, 0),
|
||||
"outside 0-15"},
|
||||
} {
|
||||
if _, _, err := bcast.amd64Memory(tt.op, 1); err == nil {
|
||||
t.Errorf("%s: validation succeeded, want an error", tt.name)
|
||||
} else if !strings.Contains(err.Error(), tt.quote) {
|
||||
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestAmd64ExtBroadcastEncoding pins the broadcast layer over the memory
|
||||
// encoding: EVEX.b, bit 4 of byte three, set on the VADDPH 512-bit template
|
||||
// while the ModR/M mod bits, the SIB byte and the displacement choices keep
|
||||
// the plain semantics amd64EncodeMemory chooses.
|
||||
func TestAmd64ExtBroadcastEncoding(t *testing.T) {
|
||||
add := []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}
|
||||
for _, tt := range []struct {
|
||||
name string
|
||||
base int
|
||||
disp int64
|
||||
want string
|
||||
}{
|
||||
{"zero displacement keeps the mod-00 form under the broadcast bit", 9, 0,
|
||||
"624514505831"},
|
||||
{"disp8 semantics unchanged", 1, 127,
|
||||
"6265145058717f"},
|
||||
{"disp32 semantics unchanged", 2, 8128,
|
||||
"6265145058b2c01f0000"},
|
||||
{"RBP keeps the forced displacement", 5, 0,
|
||||
"62651450587500"},
|
||||
{"RSP keeps the SIB byte", 12, 0,
|
||||
"62451450583424"},
|
||||
} {
|
||||
got := amd64EncodeBroadcast(add, 30, 29, tt.base, tt.disp)
|
||||
if hex.EncodeToString(got) != tt.want {
|
||||
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestAmd64ExtMemoryVocabulary pins the names the shared layer gives the
|
||||
|
||||
@@ -387,6 +387,48 @@ var amd64GoldenRows = []amd64GoldenRow{
|
||||
[]ExtOperand{ExtMemory(2, -128), ExtXmm(30)},
|
||||
"62657c082e7280", "62 65 7c 08 2e 72 80 vucomish -0x100(%rdx),%xmm30 (Disp8(80))"},
|
||||
|
||||
// The {1toN} broadcast forms of the packed arithmetic: the same memory
|
||||
// encoding with EVEX.b set, one element the hardware splats across the
|
||||
// lanes. The zero-displacement rows quote the listings' broadcast rows
|
||||
// outright, the {1to32} spellings from x86-64-avx512_fp16.d and the
|
||||
// {1to16}/{1to8} ones from avx512_fp16_vl.d. The negative disp8 row
|
||||
// pins the bytes of a masked GNU row: its {k7}{z} rides the z and aaa
|
||||
// bits the layer leaves clear, and the spelling is N times the plain
|
||||
// displacement under the disp8*N scaling.
|
||||
{"vaddph broadcast source", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"624514505831", "62 45 14 50 58 31 vaddph (%r9){1to32},%zmm29,%zmm30"},
|
||||
{"vsubph broadcast source", "VSUBPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"624514505c31", "62 45 14 50 5c 31 vsubph (%r9){1to32},%zmm29,%zmm30"},
|
||||
{"vmulph broadcast source", "VMULPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"624514505931", "62 45 14 50 59 31 vmulph (%r9){1to32},%zmm29,%zmm30"},
|
||||
{"vdivph broadcast source", "VDIVPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"624514505e31", "62 45 14 50 5e 31 vdivph (%r9){1to32},%zmm29,%zmm30"},
|
||||
{"vminph broadcast source", "VMINPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"624514505d31", "62 45 14 50 5d 31 vminph (%r9){1to32},%zmm29,%zmm30"},
|
||||
{"vmaxph broadcast source", "VMAXPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"624514505f31", "62 45 14 50 5f 31 vmaxph (%r9){1to32},%zmm29,%zmm30"},
|
||||
{"vsqrtph broadcast source", "VSQRTPH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
|
||||
"62457c585131", "62 45 7c 58 51 31 vsqrtph (%r9){1to32},%zmm30"},
|
||||
{"vaddph broadcast source negative disp8", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtBroadcast(2, -128), ExtZmm(30)},
|
||||
"62651450587280", "62 65 14 d7 58 72 80 vaddph -0x100(%rdx){1to32},%zmm29,%zmm30{%k7}{z} (Disp8(80); the GNU row adds {k7}{z})"},
|
||||
{"vaddph ymm broadcast source", "VADDPH",
|
||||
[]ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)},
|
||||
"62f554385831", "62 f5 54 38 58 31 vaddph (%ecx){1to16},%ymm5,%ymm6"},
|
||||
{"vaddph xmm broadcast source", "VADDPH",
|
||||
[]ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)},
|
||||
"62f554185831", "62 f5 54 18 58 31 vaddph (%ecx){1to8},%xmm5,%xmm6"},
|
||||
{"vsqrtph ymm broadcast source", "VSQRTPH",
|
||||
[]ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)},
|
||||
"62f57c385131", "62 f5 7c 38 51 31 vsqrtph (%ecx){1to16},%ymm6"},
|
||||
|
||||
// The BF16 memory forms: the dot product reads its second source and
|
||||
// the narrow convert its full-width source from memory.
|
||||
{"vdpbf16ps memory source", "VDPBF16PS",
|
||||
@@ -408,6 +450,20 @@ var amd64GoldenRows = []amd64GoldenRow{
|
||||
// list, both destinations being XMM, so the resolver cannot tell them
|
||||
// apart and the register row above pins the 128-bit template alone.
|
||||
|
||||
// The BF16 dot product's broadcast forms, the m16bcst spelling the
|
||||
// manual gives beside the plain vector source: {1to16} on the 512-bit
|
||||
// row of avx512_bf16.d, the VL rows from avx512_bf16_vl.d. The narrow
|
||||
// convert takes no broadcast: its source is a full-width vector.
|
||||
{"vdpbf16ps broadcast source", "VDPBF16PS",
|
||||
[]ExtOperand{ExtZmm(5), ExtBroadcast(1, 0), ExtZmm(6)},
|
||||
"62f256585231", "62 f2 56 58 52 31 vdpbf16ps (%ecx){1to16},%zmm5,%zmm6"},
|
||||
{"vdpbf16ps ymm broadcast source", "VDPBF16PS",
|
||||
[]ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)},
|
||||
"62f256385231", "62 f2 56 38 52 31 vdpbf16ps (%ecx){1to8},%ymm5,%ymm6"},
|
||||
{"vdpbf16ps xmm broadcast source", "VDPBF16PS",
|
||||
[]ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)},
|
||||
"62f256185231", "62 f2 56 18 52 31 vdpbf16ps (%ecx){1to4},%xmm5,%xmm6"},
|
||||
|
||||
// The remaining scalar memory forms: the scale and exponent extracts,
|
||||
// the imm8-control group, and the integer converts, whose second
|
||||
// source the manual spells r/m32. The W1 integer converts take the
|
||||
@@ -683,6 +739,21 @@ func TestAmd64ExtRejects(t *testing.T) {
|
||||
{"memory as the narrow convert's destination", "VCVTNEPS2BF16",
|
||||
[]ExtOperand{ExtZmm(5), ExtMemory(9, 0)},
|
||||
"wants a YMM register"},
|
||||
{"broadcast in a register position", "VCVTNE2PS2BF16",
|
||||
[]ExtOperand{ExtZmm(5), ExtBroadcast(1, 0), ExtZmm(6)},
|
||||
"carries a broadcast, the position takes a register"},
|
||||
{"broadcast as the packed arithmetic's destination", "VADDPH",
|
||||
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtBroadcast(9, 0)},
|
||||
"carries a broadcast, the position takes a register"},
|
||||
{"broadcast on the scalar arithmetic's memory source", "VADDSH",
|
||||
[]ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)},
|
||||
"carries a broadcast, the entry's memory operand takes none"},
|
||||
{"broadcast on the scalar compare's memory operand", "VCOMISH",
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
|
||||
"carries a broadcast, the entry's memory operand takes none"},
|
||||
{"broadcast on the narrow convert's full-width source", "VCVTNEPS2BF16",
|
||||
[]ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)},
|
||||
"carries a broadcast, the entry's memory operand takes none"},
|
||||
} {
|
||||
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
|
||||
_, err := in.Encode(tt.ops)
|
||||
@@ -734,6 +805,12 @@ func TestAmd64ExtMemoryFormRejects(t *testing.T) {
|
||||
{"displacement past the signed 32-bit range on the store", "VMOVW", store,
|
||||
[]ExtOperand{ExtXmm(30), ExtMemory(8, 1<<32)},
|
||||
"outside the signed 32-bit range"},
|
||||
{"broadcast on the load's memory position", "VMOVSH", load,
|
||||
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
|
||||
"carries a broadcast, the entry's memory operand takes none"},
|
||||
{"broadcast destination on the store", "VMOVSH", store,
|
||||
[]ExtOperand{ExtXmm(30), ExtBroadcast(9, 0)},
|
||||
"carries a broadcast, the entry's memory operand takes none"},
|
||||
} {
|
||||
in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick)
|
||||
_, err := in.Encode(tt.ops)
|
||||
|
||||
@@ -170,6 +170,11 @@ type ExtOperand struct {
|
||||
// operand (the encoder may derive the sh bit from the value).
|
||||
Shift int
|
||||
HasShift bool
|
||||
// Broadcast spells the {1toN} broadcast on an amd64 memory operand: the
|
||||
// base-relative location holds one element the hardware splats across
|
||||
// every lane of the destination, which the encoder lays down as EVEX.b.
|
||||
// Only the memory positions of the entries that carry Bcast accept it.
|
||||
Broadcast bool
|
||||
}
|
||||
|
||||
// ExtVector builds a scalable vector operand, ADD Z1.S style.
|
||||
@@ -200,6 +205,14 @@ func ExtMemory(base int, disp int64) ExtOperand {
|
||||
return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp}
|
||||
}
|
||||
|
||||
// ExtBroadcast builds the {1toN} broadcast spelling of a base-relative memory
|
||||
// operand, the amd64 packed forms' m16bcst shape: the base is a 64-bit general
|
||||
// register number, 0..15, the displacement keeps the plain ModR/M semantics,
|
||||
// and the encoder sets EVEX.b so the single element splats across the lanes.
|
||||
func ExtBroadcast(base int, disp int64) ExtOperand {
|
||||
return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp, Broadcast: true}
|
||||
}
|
||||
|
||||
// ExtField is one named field of the 32-bit encoding word: a bit offset from
|
||||
// the least significant end and the field's width.
|
||||
type ExtField struct {
|
||||
@@ -467,6 +480,13 @@ type ExtInstr struct {
|
||||
// of their own, ExtFormAmdMemVec and ExtFormAmdVecMem, and need no
|
||||
// flag. The arm64 entries all carry the zero value.
|
||||
Mem int
|
||||
// Bcast records that the entry's memory position takes the {1toN}
|
||||
// broadcast beside the plain vector source, the SDM's m16bcst and
|
||||
// m32bcst spellings: one element the hardware splats across the lanes,
|
||||
// encoded as EVEX.b. The packed arithmetic carries it; the scalar
|
||||
// forms and the full-width sources do not. The arm64 entries all
|
||||
// carry the zero value.
|
||||
Bcast bool
|
||||
}
|
||||
|
||||
// Encode assembles the operands into the 4 little-endian bytes of the
|
||||
|
||||
Reference in new issue
Block a user