feat(arch): add the {1toN} broadcast to the packed amd64 memory sources

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:06:11 +02:00
1 parent 9c1392dfc3
commit ccb155437e
4 files changed
+218 -32

No files matched your search

+68 -32
View File
@@ -19,10 +19,12 @@
// toolchain has is not an extension.
//
// The forms encode the unmasked shapes: register forms throughout, and the
// scalar FP16 memory forms beside them, base-relative operands with the
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand. A
// scaled index, write masking ({k1}{z}) and embedded rounding still arrive
// with a later slice.
// memory forms beside them, the scalar ones the manual spells m16 and the
// packed ones with the {1toN} broadcast, base-relative operands with the
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the
// broadcast laying EVEX.b over the same displacement semantics. A scaled
// index, write masking ({k1}{z}) and embedded rounding still arrive with a
// later slice.
package arch
@@ -120,7 +122,9 @@ func amd64Encode(b []byte, dest, vvvv, rm int) []byte {
// amd64Memory validates a memory operand of an amd64 entry: no arrangement
// and no qualifier, a base general register inside 0-15, a signed 32-bit
// displacement and no shift. The base number rides the operand's Reg and
// the displacement its Imm.
// the displacement its Imm. A broadcast spelling is refused unless the
// entry carries Bcast: the scalar forms read a plain m16 and the full-width
// sources a plain vector, and neither splats.
func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, err error) {
if op.Kind != ExtMem {
return 0, 0, fmt.Errorf("%s: operand %d wants a memory operand, got %s", in.Name, pos, op.Kind)
@@ -134,6 +138,9 @@ func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, er
if op.HasShift {
return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos)
}
if op.Broadcast && !in.Bcast {
return 0, 0, fmt.Errorf("%s: operand %d carries a broadcast, the entry's memory operand takes none", in.Name, pos)
}
if op.Reg < 0 || op.Reg > 15 {
return 0, 0, fmt.Errorf("%s: operand %d names base register %d, outside 0-15", in.Name, pos, op.Reg)
}
@@ -179,10 +186,26 @@ func amd64EncodeMemory(b []byte, dest, vvvv, base int, disp int64) []byte {
return append(out, tail...)
}
// amd64EncodeBroadcast returns the memory encoding with EVEX.b set: the
// {1toN} broadcast, whose single element the hardware splats across every
// lane of the destination. EVEX.b is bit 4 of byte three, and the ModR/M,
// SIB and displacement bytes keep the plain semantics amd64EncodeMemory
// chooses; only the prefix bit changes. The operand must have passed
// amd64Memory on an entry that carries Bcast.
func amd64EncodeBroadcast(b []byte, dest, vvvv, base int, disp int64) []byte {
out := amd64EncodeMemory(b, dest, vvvv, base, disp)
out[3] |= 0x10
return out
}
// amd64PlainReg checks the invariants every amd64 register operand carries:
// no arm64 arrangement, no predicate qualifier, and a register number inside
// the class the instruction encodes.
// the class the instruction encodes. A broadcast spelling names a memory
// location, so a register position refuses it outright.
func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
if op.Broadcast {
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
}
if op.Arr != ExtArrNone {
return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
}
@@ -196,7 +219,12 @@ func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
}
// amd64Vector checks one vector operand against the class the entry encodes.
// The broadcast spelling is named before the kind, so the diagnostic says
// what the operand carries rather than what the position wanted.
func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error {
if op.Broadcast {
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
}
if op.Kind != class {
article := "a"
if class == ExtXMM {
@@ -258,7 +286,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
// memory shape of that position too: the second source of the scalar
// arithmetic, spelled xmm3/m16 in the manual, may be a base-relative
// operand, which rides the r/m field with its displacement bytes after the
// opcode.
// opcode, and the packed entries lay the source's {1toN} broadcast over the
// same encoding as EVEX.b.
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
@@ -272,6 +301,9 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
if ops[1].Broadcast {
return amd64EncodeBroadcast(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
}
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
}
for i, op := range ops[1:] {
@@ -318,7 +350,8 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
// of destination, and at 128 bits the companion stays the class itself. An
// entry with Mem set takes the memory shape of that position too: the
// compares and the packed square root read their source from memory, and
// compares and the packed square root read their source from memory, the
// packed square root's source carrying the {1toN} broadcast as EVEX.b, and
// the narrow BF16 convert reads its full-width source there.
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
@@ -334,6 +367,9 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
return nil, err
}
if ops[0].Broadcast {
return amd64EncodeBroadcast(in.Bytes, ops[1].Reg, -1, base, disp), nil
}
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
@@ -580,13 +616,13 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"},
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16,
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16,
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"},
// AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask
@@ -715,67 +751,67 @@ var amd64Extensions = []ExtInstr{
// quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from
// avx512_fp16_vl.d, on the same low registers the suite uses.
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"},
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"},
{Name: "VADDPH", Summary: "Add packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"},
{Name: "VSUBPH", Summary: "Subtract packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"},
{Name: "VMULPH", Summary: "Multiply packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"},
{Name: "VDIVPH", Summary: "Divide packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"},
{Name: "VMINPH", Summary: "Return the minimum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"},
{Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"},
{Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"},
// AVX512-FP16 scalar, the imm8-control group: mantissa extraction,
+53
View File
@@ -76,6 +76,8 @@ func TestAmd64ExtMemoryRejects(t *testing.T) {
"arrangement"},
{"predicate qualifier on the memory operand", ExtOperand{Kind: ExtMem, Reg: 8, Qual: ExtQualZeroing},
"predicate qualifier"},
{"broadcast on an entry that takes none", ExtBroadcast(8, 0),
"takes none"},
} {
_, _, err := in.amd64Memory(tt.op, 1)
if err == nil {
@@ -86,6 +88,57 @@ func TestAmd64ExtMemoryRejects(t *testing.T) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
// The broadcast spelling passes the flag gate on an entry that carries
// Bcast and then meets the same base and displacement checks.
bcast := ExtInstr{Name: "TEST", Bcast: true}
for _, tt := range []struct {
name string
op ExtOperand
quote string
}{
{"broadcast base beyond r15", ExtOperand{Kind: ExtMem, Reg: 16, Imm: 0, Broadcast: true},
"outside 0-15"},
{"broadcast displacement past the signed 32-bit ceiling", ExtBroadcast(8, 1<<31),
"outside the signed 32-bit range"},
{"broadcast base under r0", ExtBroadcast(-1, 0),
"outside 0-15"},
} {
if _, _, err := bcast.amd64Memory(tt.op, 1); err == nil {
t.Errorf("%s: validation succeeded, want an error", tt.name)
} else if !strings.Contains(err.Error(), tt.quote) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
}
// TestAmd64ExtBroadcastEncoding pins the broadcast layer over the memory
// encoding: EVEX.b, bit 4 of byte three, set on the VADDPH 512-bit template
// while the ModR/M mod bits, the SIB byte and the displacement choices keep
// the plain semantics amd64EncodeMemory chooses.
func TestAmd64ExtBroadcastEncoding(t *testing.T) {
add := []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}
for _, tt := range []struct {
name string
base int
disp int64
want string
}{
{"zero displacement keeps the mod-00 form under the broadcast bit", 9, 0,
"624514505831"},
{"disp8 semantics unchanged", 1, 127,
"6265145058717f"},
{"disp32 semantics unchanged", 2, 8128,
"6265145058b2c01f0000"},
{"RBP keeps the forced displacement", 5, 0,
"62651450587500"},
{"RSP keeps the SIB byte", 12, 0,
"62451450583424"},
} {
got := amd64EncodeBroadcast(add, 30, 29, tt.base, tt.disp)
if hex.EncodeToString(got) != tt.want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
}
}
}
// TestAmd64ExtMemoryVocabulary pins the names the shared layer gives the
+77
View File
@@ -387,6 +387,48 @@ var amd64GoldenRows = []amd64GoldenRow{
[]ExtOperand{ExtMemory(2, -128), ExtXmm(30)},
"62657c082e7280", "62 65 7c 08 2e 72 80 vucomish -0x100(%rdx),%xmm30 (Disp8(80))"},
// The {1toN} broadcast forms of the packed arithmetic: the same memory
// encoding with EVEX.b set, one element the hardware splats across the
// lanes. The zero-displacement rows quote the listings' broadcast rows
// outright, the {1to32} spellings from x86-64-avx512_fp16.d and the
// {1to16}/{1to8} ones from avx512_fp16_vl.d. The negative disp8 row
// pins the bytes of a masked GNU row: its {k7}{z} rides the z and aaa
// bits the layer leaves clear, and the spelling is N times the plain
// displacement under the disp8*N scaling.
{"vaddph broadcast source", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505831", "62 45 14 50 58 31 vaddph (%r9){1to32},%zmm29,%zmm30"},
{"vsubph broadcast source", "VSUBPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505c31", "62 45 14 50 5c 31 vsubph (%r9){1to32},%zmm29,%zmm30"},
{"vmulph broadcast source", "VMULPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505931", "62 45 14 50 59 31 vmulph (%r9){1to32},%zmm29,%zmm30"},
{"vdivph broadcast source", "VDIVPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505e31", "62 45 14 50 5e 31 vdivph (%r9){1to32},%zmm29,%zmm30"},
{"vminph broadcast source", "VMINPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505d31", "62 45 14 50 5d 31 vminph (%r9){1to32},%zmm29,%zmm30"},
{"vmaxph broadcast source", "VMAXPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505f31", "62 45 14 50 5f 31 vmaxph (%r9){1to32},%zmm29,%zmm30"},
{"vsqrtph broadcast source", "VSQRTPH",
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
"62457c585131", "62 45 7c 58 51 31 vsqrtph (%r9){1to32},%zmm30"},
{"vaddph broadcast source negative disp8", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(2, -128), ExtZmm(30)},
"62651450587280", "62 65 14 d7 58 72 80 vaddph -0x100(%rdx){1to32},%zmm29,%zmm30{%k7}{z} (Disp8(80); the GNU row adds {k7}{z})"},
{"vaddph ymm broadcast source", "VADDPH",
[]ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)},
"62f554385831", "62 f5 54 38 58 31 vaddph (%ecx){1to16},%ymm5,%ymm6"},
{"vaddph xmm broadcast source", "VADDPH",
[]ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)},
"62f554185831", "62 f5 54 18 58 31 vaddph (%ecx){1to8},%xmm5,%xmm6"},
{"vsqrtph ymm broadcast source", "VSQRTPH",
[]ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)},
"62f57c385131", "62 f5 7c 38 51 31 vsqrtph (%ecx){1to16},%ymm6"},
// The BF16 memory forms: the dot product reads its second source and
// the narrow convert its full-width source from memory.
{"vdpbf16ps memory source", "VDPBF16PS",
@@ -408,6 +450,20 @@ var amd64GoldenRows = []amd64GoldenRow{
// list, both destinations being XMM, so the resolver cannot tell them
// apart and the register row above pins the 128-bit template alone.
// The BF16 dot product's broadcast forms, the m16bcst spelling the
// manual gives beside the plain vector source: {1to16} on the 512-bit
// row of avx512_bf16.d, the VL rows from avx512_bf16_vl.d. The narrow
// convert takes no broadcast: its source is a full-width vector.
{"vdpbf16ps broadcast source", "VDPBF16PS",
[]ExtOperand{ExtZmm(5), ExtBroadcast(1, 0), ExtZmm(6)},
"62f256585231", "62 f2 56 58 52 31 vdpbf16ps (%ecx){1to16},%zmm5,%zmm6"},
{"vdpbf16ps ymm broadcast source", "VDPBF16PS",
[]ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)},
"62f256385231", "62 f2 56 38 52 31 vdpbf16ps (%ecx){1to8},%ymm5,%ymm6"},
{"vdpbf16ps xmm broadcast source", "VDPBF16PS",
[]ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)},
"62f256185231", "62 f2 56 18 52 31 vdpbf16ps (%ecx){1to4},%xmm5,%xmm6"},
// The remaining scalar memory forms: the scale and exponent extracts,
// the imm8-control group, and the integer converts, whose second
// source the manual spells r/m32. The W1 integer converts take the
@@ -683,6 +739,21 @@ func TestAmd64ExtRejects(t *testing.T) {
{"memory as the narrow convert's destination", "VCVTNEPS2BF16",
[]ExtOperand{ExtZmm(5), ExtMemory(9, 0)},
"wants a YMM register"},
{"broadcast in a register position", "VCVTNE2PS2BF16",
[]ExtOperand{ExtZmm(5), ExtBroadcast(1, 0), ExtZmm(6)},
"carries a broadcast, the position takes a register"},
{"broadcast as the packed arithmetic's destination", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtBroadcast(9, 0)},
"carries a broadcast, the position takes a register"},
{"broadcast on the scalar arithmetic's memory source", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)},
"carries a broadcast, the entry's memory operand takes none"},
{"broadcast on the scalar compare's memory operand", "VCOMISH",
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
"carries a broadcast, the entry's memory operand takes none"},
{"broadcast on the narrow convert's full-width source", "VCVTNEPS2BF16",
[]ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)},
"carries a broadcast, the entry's memory operand takes none"},
} {
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
_, err := in.Encode(tt.ops)
@@ -734,6 +805,12 @@ func TestAmd64ExtMemoryFormRejects(t *testing.T) {
{"displacement past the signed 32-bit range on the store", "VMOVW", store,
[]ExtOperand{ExtXmm(30), ExtMemory(8, 1<<32)},
"outside the signed 32-bit range"},
{"broadcast on the load's memory position", "VMOVSH", load,
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
"carries a broadcast, the entry's memory operand takes none"},
{"broadcast destination on the store", "VMOVSH", store,
[]ExtOperand{ExtXmm(30), ExtBroadcast(9, 0)},
"carries a broadcast, the entry's memory operand takes none"},
} {
in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick)
_, err := in.Encode(tt.ops)
+20
View File
@@ -170,6 +170,11 @@ type ExtOperand struct {
// operand (the encoder may derive the sh bit from the value).
Shift int
HasShift bool
// Broadcast spells the {1toN} broadcast on an amd64 memory operand: the
// base-relative location holds one element the hardware splats across
// every lane of the destination, which the encoder lays down as EVEX.b.
// Only the memory positions of the entries that carry Bcast accept it.
Broadcast bool
}
// ExtVector builds a scalable vector operand, ADD Z1.S style.
@@ -200,6 +205,14 @@ func ExtMemory(base int, disp int64) ExtOperand {
return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp}
}
// ExtBroadcast builds the {1toN} broadcast spelling of a base-relative memory
// operand, the amd64 packed forms' m16bcst shape: the base is a 64-bit general
// register number, 0..15, the displacement keeps the plain ModR/M semantics,
// and the encoder sets EVEX.b so the single element splats across the lanes.
func ExtBroadcast(base int, disp int64) ExtOperand {
return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp, Broadcast: true}
}
// ExtField is one named field of the 32-bit encoding word: a bit offset from
// the least significant end and the field's width.
type ExtField struct {
@@ -467,6 +480,13 @@ type ExtInstr struct {
// of their own, ExtFormAmdMemVec and ExtFormAmdVecMem, and need no
// flag. The arm64 entries all carry the zero value.
Mem int
// Bcast records that the entry's memory position takes the {1toN}
// broadcast beside the plain vector source, the SDM's m16bcst and
// m32bcst spellings: one element the hardware splats across the lanes,
// encoded as EVEX.b. The packed arithmetic carries it; the scalar
// forms and the full-width sources do not. The arm64 entries all
// carry the zero value.
Bcast bool
}
// Encode assembles the operands into the 4 little-endian bytes of the