diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index a1004a5..8c30562 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -19,10 +19,12 @@ // toolchain has is not an extension. // // The forms encode the unmasked shapes: register forms throughout, and the -// scalar FP16 memory forms beside them, base-relative operands with the -// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand. A -// scaled index, write masking ({k1}{z}) and embedded rounding still arrive -// with a later slice. +// memory forms beside them, the scalar ones the manual spells m16 and the +// packed ones with the {1toN} broadcast, base-relative operands with the +// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the +// broadcast laying EVEX.b over the same displacement semantics. A scaled +// index, write masking ({k1}{z}) and embedded rounding still arrive with a +// later slice. package arch @@ -120,7 +122,9 @@ func amd64Encode(b []byte, dest, vvvv, rm int) []byte { // amd64Memory validates a memory operand of an amd64 entry: no arrangement // and no qualifier, a base general register inside 0-15, a signed 32-bit // displacement and no shift. The base number rides the operand's Reg and -// the displacement its Imm. +// the displacement its Imm. A broadcast spelling is refused unless the +// entry carries Bcast: the scalar forms read a plain m16 and the full-width +// sources a plain vector, and neither splats. func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, err error) { if op.Kind != ExtMem { return 0, 0, fmt.Errorf("%s: operand %d wants a memory operand, got %s", in.Name, pos, op.Kind) @@ -134,6 +138,9 @@ func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, er if op.HasShift { return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos) } + if op.Broadcast && !in.Bcast { + return 0, 0, fmt.Errorf("%s: operand %d carries a broadcast, the entry's memory operand takes none", in.Name, pos) + } if op.Reg < 0 || op.Reg > 15 { return 0, 0, fmt.Errorf("%s: operand %d names base register %d, outside 0-15", in.Name, pos, op.Reg) } @@ -179,10 +186,26 @@ func amd64EncodeMemory(b []byte, dest, vvvv, base int, disp int64) []byte { return append(out, tail...) } +// amd64EncodeBroadcast returns the memory encoding with EVEX.b set: the +// {1toN} broadcast, whose single element the hardware splats across every +// lane of the destination. EVEX.b is bit 4 of byte three, and the ModR/M, +// SIB and displacement bytes keep the plain semantics amd64EncodeMemory +// chooses; only the prefix bit changes. The operand must have passed +// amd64Memory on an entry that carries Bcast. +func amd64EncodeBroadcast(b []byte, dest, vvvv, base int, disp int64) []byte { + out := amd64EncodeMemory(b, dest, vvvv, base, disp) + out[3] |= 0x10 + return out +} + // amd64PlainReg checks the invariants every amd64 register operand carries: // no arm64 arrangement, no predicate qualifier, and a register number inside -// the class the instruction encodes. +// the class the instruction encodes. A broadcast spelling names a memory +// location, so a register position refuses it outright. func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error { + if op.Broadcast { + return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos) + } if op.Arr != ExtArrNone { return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos) } @@ -196,7 +219,12 @@ func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error { } // amd64Vector checks one vector operand against the class the entry encodes. +// The broadcast spelling is named before the kind, so the diagnostic says +// what the operand carries rather than what the position wanted. func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error { + if op.Broadcast { + return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos) + } if op.Kind != class { article := "a" if class == ExtXMM { @@ -258,7 +286,8 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) { // memory shape of that position too: the second source of the scalar // arithmetic, spelled xmm3/m16 in the manual, may be a base-relative // operand, which rides the r/m field with its displacement bytes after the -// opcode. +// opcode, and the packed entries lay the source's {1toN} broadcast over the +// same encoding as EVEX.b. func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) if err := in.amd64Vector(ops[0], class, 1); err != nil { @@ -272,6 +301,9 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { if err := in.amd64Vector(ops[2], class, 3); err != nil { return nil, err } + if ops[1].Broadcast { + return amd64EncodeBroadcast(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil + } return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil } for i, op := range ops[1:] { @@ -318,7 +350,8 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) { // the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits // of destination, and at 128 bits the companion stays the class itself. An // entry with Mem set takes the memory shape of that position too: the -// compares and the packed square root read their source from memory, and +// compares and the packed square root read their source from memory, the +// packed square root's source carrying the {1toN} broadcast as EVEX.b, and // the narrow BF16 convert reads its full-width source there. func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) @@ -334,6 +367,9 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { if err := in.amd64Vector(ops[1], destClass, 2); err != nil { return nil, err } + if ops[0].Broadcast { + return amd64EncodeBroadcast(in.Bytes, ops[1].Reg, -1, base, disp), nil + } return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil } if in.Mem == 2 && ops[1].Kind == ExtMem { @@ -580,13 +616,13 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Mem: 1, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"}, {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", - Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"}, {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", - Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"}, {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", - Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureBF16, + Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureBF16, Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"}, // AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask @@ -715,67 +751,67 @@ var amd64Extensions = []ExtInstr{ // quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from // avx512_fp16_vl.d, on the same low registers the suite uses. {Name: "VADDPH", Summary: "Add packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"}, {Name: "VADDPH", Summary: "Add packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.256.MAP5.W0 58 /r)"}, {Name: "VADDPH", Summary: "Add packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"}, {Name: "VSUBPH", Summary: "Subtract packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"}, {Name: "VSUBPH", Summary: "Subtract packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.256.MAP5.W0 5C /r)"}, {Name: "VSUBPH", Summary: "Subtract packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"}, {Name: "VMULPH", Summary: "Multiply packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"}, {Name: "VMULPH", Summary: "Multiply packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.256.MAP5.W0 59 /r)"}, {Name: "VMULPH", Summary: "Multiply packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"}, {Name: "VDIVPH", Summary: "Divide packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"}, {Name: "VDIVPH", Summary: "Divide packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.256.MAP5.W0 5E /r)"}, {Name: "VDIVPH", Summary: "Divide packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"}, {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"}, {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.256.MAP5.W0 5D /r)"}, {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"}, {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"}, {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.256.MAP5.W0 5F /r)"}, {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"}, {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"}, {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.256.MAP5.W0 51 /r)"}, {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.128.MAP5.W0 51 /r)"}, // AVX512-FP16 scalar, the imm8-control group: mantissa extraction, diff --git a/arch/amd64_ext_mem_test.go b/arch/amd64_ext_mem_test.go index 731d3bf..11d982c 100644 --- a/arch/amd64_ext_mem_test.go +++ b/arch/amd64_ext_mem_test.go @@ -76,6 +76,8 @@ func TestAmd64ExtMemoryRejects(t *testing.T) { "arrangement"}, {"predicate qualifier on the memory operand", ExtOperand{Kind: ExtMem, Reg: 8, Qual: ExtQualZeroing}, "predicate qualifier"}, + {"broadcast on an entry that takes none", ExtBroadcast(8, 0), + "takes none"}, } { _, _, err := in.amd64Memory(tt.op, 1) if err == nil { @@ -86,6 +88,57 @@ func TestAmd64ExtMemoryRejects(t *testing.T) { t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote) } } + // The broadcast spelling passes the flag gate on an entry that carries + // Bcast and then meets the same base and displacement checks. + bcast := ExtInstr{Name: "TEST", Bcast: true} + for _, tt := range []struct { + name string + op ExtOperand + quote string + }{ + {"broadcast base beyond r15", ExtOperand{Kind: ExtMem, Reg: 16, Imm: 0, Broadcast: true}, + "outside 0-15"}, + {"broadcast displacement past the signed 32-bit ceiling", ExtBroadcast(8, 1<<31), + "outside the signed 32-bit range"}, + {"broadcast base under r0", ExtBroadcast(-1, 0), + "outside 0-15"}, + } { + if _, _, err := bcast.amd64Memory(tt.op, 1); err == nil { + t.Errorf("%s: validation succeeded, want an error", tt.name) + } else if !strings.Contains(err.Error(), tt.quote) { + t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote) + } + } +} + +// TestAmd64ExtBroadcastEncoding pins the broadcast layer over the memory +// encoding: EVEX.b, bit 4 of byte three, set on the VADDPH 512-bit template +// while the ModR/M mod bits, the SIB byte and the displacement choices keep +// the plain semantics amd64EncodeMemory chooses. +func TestAmd64ExtBroadcastEncoding(t *testing.T) { + add := []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0} + for _, tt := range []struct { + name string + base int + disp int64 + want string + }{ + {"zero displacement keeps the mod-00 form under the broadcast bit", 9, 0, + "624514505831"}, + {"disp8 semantics unchanged", 1, 127, + "6265145058717f"}, + {"disp32 semantics unchanged", 2, 8128, + "6265145058b2c01f0000"}, + {"RBP keeps the forced displacement", 5, 0, + "62651450587500"}, + {"RSP keeps the SIB byte", 12, 0, + "62451450583424"}, + } { + got := amd64EncodeBroadcast(add, 30, 29, tt.base, tt.disp) + if hex.EncodeToString(got) != tt.want { + t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want) + } + } } // TestAmd64ExtMemoryVocabulary pins the names the shared layer gives the diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 49b5a54..27cad27 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -387,6 +387,48 @@ var amd64GoldenRows = []amd64GoldenRow{ []ExtOperand{ExtMemory(2, -128), ExtXmm(30)}, "62657c082e7280", "62 65 7c 08 2e 72 80 vucomish -0x100(%rdx),%xmm30 (Disp8(80))"}, + // The {1toN} broadcast forms of the packed arithmetic: the same memory + // encoding with EVEX.b set, one element the hardware splats across the + // lanes. The zero-displacement rows quote the listings' broadcast rows + // outright, the {1to32} spellings from x86-64-avx512_fp16.d and the + // {1to16}/{1to8} ones from avx512_fp16_vl.d. The negative disp8 row + // pins the bytes of a masked GNU row: its {k7}{z} rides the z and aaa + // bits the layer leaves clear, and the spelling is N times the plain + // displacement under the disp8*N scaling. + {"vaddph broadcast source", "VADDPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "624514505831", "62 45 14 50 58 31 vaddph (%r9){1to32},%zmm29,%zmm30"}, + {"vsubph broadcast source", "VSUBPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "624514505c31", "62 45 14 50 5c 31 vsubph (%r9){1to32},%zmm29,%zmm30"}, + {"vmulph broadcast source", "VMULPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "624514505931", "62 45 14 50 59 31 vmulph (%r9){1to32},%zmm29,%zmm30"}, + {"vdivph broadcast source", "VDIVPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "624514505e31", "62 45 14 50 5e 31 vdivph (%r9){1to32},%zmm29,%zmm30"}, + {"vminph broadcast source", "VMINPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "624514505d31", "62 45 14 50 5d 31 vminph (%r9){1to32},%zmm29,%zmm30"}, + {"vmaxph broadcast source", "VMAXPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)}, + "624514505f31", "62 45 14 50 5f 31 vmaxph (%r9){1to32},%zmm29,%zmm30"}, + {"vsqrtph broadcast source", "VSQRTPH", + []ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)}, + "62457c585131", "62 45 7c 58 51 31 vsqrtph (%r9){1to32},%zmm30"}, + {"vaddph broadcast source negative disp8", "VADDPH", + []ExtOperand{ExtZmm(29), ExtBroadcast(2, -128), ExtZmm(30)}, + "62651450587280", "62 65 14 d7 58 72 80 vaddph -0x100(%rdx){1to32},%zmm29,%zmm30{%k7}{z} (Disp8(80); the GNU row adds {k7}{z})"}, + {"vaddph ymm broadcast source", "VADDPH", + []ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)}, + "62f554385831", "62 f5 54 38 58 31 vaddph (%ecx){1to16},%ymm5,%ymm6"}, + {"vaddph xmm broadcast source", "VADDPH", + []ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)}, + "62f554185831", "62 f5 54 18 58 31 vaddph (%ecx){1to8},%xmm5,%xmm6"}, + {"vsqrtph ymm broadcast source", "VSQRTPH", + []ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)}, + "62f57c385131", "62 f5 7c 38 51 31 vsqrtph (%ecx){1to16},%ymm6"}, + // The BF16 memory forms: the dot product reads its second source and // the narrow convert its full-width source from memory. {"vdpbf16ps memory source", "VDPBF16PS", @@ -408,6 +450,20 @@ var amd64GoldenRows = []amd64GoldenRow{ // list, both destinations being XMM, so the resolver cannot tell them // apart and the register row above pins the 128-bit template alone. + // The BF16 dot product's broadcast forms, the m16bcst spelling the + // manual gives beside the plain vector source: {1to16} on the 512-bit + // row of avx512_bf16.d, the VL rows from avx512_bf16_vl.d. The narrow + // convert takes no broadcast: its source is a full-width vector. + {"vdpbf16ps broadcast source", "VDPBF16PS", + []ExtOperand{ExtZmm(5), ExtBroadcast(1, 0), ExtZmm(6)}, + "62f256585231", "62 f2 56 58 52 31 vdpbf16ps (%ecx){1to16},%zmm5,%zmm6"}, + {"vdpbf16ps ymm broadcast source", "VDPBF16PS", + []ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)}, + "62f256385231", "62 f2 56 38 52 31 vdpbf16ps (%ecx){1to8},%ymm5,%ymm6"}, + {"vdpbf16ps xmm broadcast source", "VDPBF16PS", + []ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)}, + "62f256185231", "62 f2 56 18 52 31 vdpbf16ps (%ecx){1to4},%xmm5,%xmm6"}, + // The remaining scalar memory forms: the scale and exponent extracts, // the imm8-control group, and the integer converts, whose second // source the manual spells r/m32. The W1 integer converts take the @@ -683,6 +739,21 @@ func TestAmd64ExtRejects(t *testing.T) { {"memory as the narrow convert's destination", "VCVTNEPS2BF16", []ExtOperand{ExtZmm(5), ExtMemory(9, 0)}, "wants a YMM register"}, + {"broadcast in a register position", "VCVTNE2PS2BF16", + []ExtOperand{ExtZmm(5), ExtBroadcast(1, 0), ExtZmm(6)}, + "carries a broadcast, the position takes a register"}, + {"broadcast as the packed arithmetic's destination", "VADDPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtBroadcast(9, 0)}, + "carries a broadcast, the position takes a register"}, + {"broadcast on the scalar arithmetic's memory source", "VADDSH", + []ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)}, + "carries a broadcast, the entry's memory operand takes none"}, + {"broadcast on the scalar compare's memory operand", "VCOMISH", + []ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)}, + "carries a broadcast, the entry's memory operand takes none"}, + {"broadcast on the narrow convert's full-width source", "VCVTNEPS2BF16", + []ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)}, + "carries a broadcast, the entry's memory operand takes none"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) @@ -734,6 +805,12 @@ func TestAmd64ExtMemoryFormRejects(t *testing.T) { {"displacement past the signed 32-bit range on the store", "VMOVW", store, []ExtOperand{ExtXmm(30), ExtMemory(8, 1<<32)}, "outside the signed 32-bit range"}, + {"broadcast on the load's memory position", "VMOVSH", load, + []ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)}, + "carries a broadcast, the entry's memory operand takes none"}, + {"broadcast destination on the store", "VMOVSH", store, + []ExtOperand{ExtXmm(30), ExtBroadcast(9, 0)}, + "carries a broadcast, the entry's memory operand takes none"}, } { in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick) _, err := in.Encode(tt.ops) diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index d533d68..df82f3c 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -170,6 +170,11 @@ type ExtOperand struct { // operand (the encoder may derive the sh bit from the value). Shift int HasShift bool + // Broadcast spells the {1toN} broadcast on an amd64 memory operand: the + // base-relative location holds one element the hardware splats across + // every lane of the destination, which the encoder lays down as EVEX.b. + // Only the memory positions of the entries that carry Bcast accept it. + Broadcast bool } // ExtVector builds a scalable vector operand, ADD Z1.S style. @@ -200,6 +205,14 @@ func ExtMemory(base int, disp int64) ExtOperand { return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp} } +// ExtBroadcast builds the {1toN} broadcast spelling of a base-relative memory +// operand, the amd64 packed forms' m16bcst shape: the base is a 64-bit general +// register number, 0..15, the displacement keeps the plain ModR/M semantics, +// and the encoder sets EVEX.b so the single element splats across the lanes. +func ExtBroadcast(base int, disp int64) ExtOperand { + return ExtOperand{Kind: ExtMem, Reg: base, Imm: disp, Broadcast: true} +} + // ExtField is one named field of the 32-bit encoding word: a bit offset from // the least significant end and the field's width. type ExtField struct { @@ -467,6 +480,13 @@ type ExtInstr struct { // of their own, ExtFormAmdMemVec and ExtFormAmdVecMem, and need no // flag. The arm64 entries all carry the zero value. Mem int + // Bcast records that the entry's memory position takes the {1toN} + // broadcast beside the plain vector source, the SDM's m16bcst and + // m32bcst spellings: one element the hardware splats across the lanes, + // encoded as EVEX.b. The packed arithmetic carries it; the scalar + // forms and the full-width sources do not. The arm64 entries all + // carry the zero value. + Bcast bool } // Encode assembles the operands into the 4 little-endian bytes of the