feat(arch): add the {1toN} broadcast to the packed amd64 memory sources

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:06:11 +02:00
1 parent 9c1392dfc3
commit ccb155437e
4 files changed
+218 -32

No files matched your search

+77
View File
@@ -387,6 +387,48 @@ var amd64GoldenRows = []amd64GoldenRow{
[]ExtOperand{ExtMemory(2, -128), ExtXmm(30)},
"62657c082e7280", "62 65 7c 08 2e 72 80 vucomish -0x100(%rdx),%xmm30 (Disp8(80))"},
// The {1toN} broadcast forms of the packed arithmetic: the same memory
// encoding with EVEX.b set, one element the hardware splats across the
// lanes. The zero-displacement rows quote the listings' broadcast rows
// outright, the {1to32} spellings from x86-64-avx512_fp16.d and the
// {1to16}/{1to8} ones from avx512_fp16_vl.d. The negative disp8 row
// pins the bytes of a masked GNU row: its {k7}{z} rides the z and aaa
// bits the layer leaves clear, and the spelling is N times the plain
// displacement under the disp8*N scaling.
{"vaddph broadcast source", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505831", "62 45 14 50 58 31 vaddph (%r9){1to32},%zmm29,%zmm30"},
{"vsubph broadcast source", "VSUBPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505c31", "62 45 14 50 5c 31 vsubph (%r9){1to32},%zmm29,%zmm30"},
{"vmulph broadcast source", "VMULPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505931", "62 45 14 50 59 31 vmulph (%r9){1to32},%zmm29,%zmm30"},
{"vdivph broadcast source", "VDIVPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505e31", "62 45 14 50 5e 31 vdivph (%r9){1to32},%zmm29,%zmm30"},
{"vminph broadcast source", "VMINPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505d31", "62 45 14 50 5d 31 vminph (%r9){1to32},%zmm29,%zmm30"},
{"vmaxph broadcast source", "VMAXPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(9, 0), ExtZmm(30)},
"624514505f31", "62 45 14 50 5f 31 vmaxph (%r9){1to32},%zmm29,%zmm30"},
{"vsqrtph broadcast source", "VSQRTPH",
[]ExtOperand{ExtBroadcast(9, 0), ExtZmm(30)},
"62457c585131", "62 45 7c 58 51 31 vsqrtph (%r9){1to32},%zmm30"},
{"vaddph broadcast source negative disp8", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtBroadcast(2, -128), ExtZmm(30)},
"62651450587280", "62 65 14 d7 58 72 80 vaddph -0x100(%rdx){1to32},%zmm29,%zmm30{%k7}{z} (Disp8(80); the GNU row adds {k7}{z})"},
{"vaddph ymm broadcast source", "VADDPH",
[]ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)},
"62f554385831", "62 f5 54 38 58 31 vaddph (%ecx){1to16},%ymm5,%ymm6"},
{"vaddph xmm broadcast source", "VADDPH",
[]ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)},
"62f554185831", "62 f5 54 18 58 31 vaddph (%ecx){1to8},%xmm5,%xmm6"},
{"vsqrtph ymm broadcast source", "VSQRTPH",
[]ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)},
"62f57c385131", "62 f5 7c 38 51 31 vsqrtph (%ecx){1to16},%ymm6"},
// The BF16 memory forms: the dot product reads its second source and
// the narrow convert its full-width source from memory.
{"vdpbf16ps memory source", "VDPBF16PS",
@@ -408,6 +450,20 @@ var amd64GoldenRows = []amd64GoldenRow{
// list, both destinations being XMM, so the resolver cannot tell them
// apart and the register row above pins the 128-bit template alone.
// The BF16 dot product's broadcast forms, the m16bcst spelling the
// manual gives beside the plain vector source: {1to16} on the 512-bit
// row of avx512_bf16.d, the VL rows from avx512_bf16_vl.d. The narrow
// convert takes no broadcast: its source is a full-width vector.
{"vdpbf16ps broadcast source", "VDPBF16PS",
[]ExtOperand{ExtZmm(5), ExtBroadcast(1, 0), ExtZmm(6)},
"62f256585231", "62 f2 56 58 52 31 vdpbf16ps (%ecx){1to16},%zmm5,%zmm6"},
{"vdpbf16ps ymm broadcast source", "VDPBF16PS",
[]ExtOperand{ExtYmm(5), ExtBroadcast(1, 0), ExtYmm(6)},
"62f256385231", "62 f2 56 38 52 31 vdpbf16ps (%ecx){1to8},%ymm5,%ymm6"},
{"vdpbf16ps xmm broadcast source", "VDPBF16PS",
[]ExtOperand{ExtXmm(5), ExtBroadcast(1, 0), ExtXmm(6)},
"62f256185231", "62 f2 56 18 52 31 vdpbf16ps (%ecx){1to4},%xmm5,%xmm6"},
// The remaining scalar memory forms: the scale and exponent extracts,
// the imm8-control group, and the integer converts, whose second
// source the manual spells r/m32. The W1 integer converts take the
@@ -683,6 +739,21 @@ func TestAmd64ExtRejects(t *testing.T) {
{"memory as the narrow convert's destination", "VCVTNEPS2BF16",
[]ExtOperand{ExtZmm(5), ExtMemory(9, 0)},
"wants a YMM register"},
{"broadcast in a register position", "VCVTNE2PS2BF16",
[]ExtOperand{ExtZmm(5), ExtBroadcast(1, 0), ExtZmm(6)},
"carries a broadcast, the position takes a register"},
{"broadcast as the packed arithmetic's destination", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtBroadcast(9, 0)},
"carries a broadcast, the position takes a register"},
{"broadcast on the scalar arithmetic's memory source", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtBroadcast(9, 0), ExtXmm(30)},
"carries a broadcast, the entry's memory operand takes none"},
{"broadcast on the scalar compare's memory operand", "VCOMISH",
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
"carries a broadcast, the entry's memory operand takes none"},
{"broadcast on the narrow convert's full-width source", "VCVTNEPS2BF16",
[]ExtOperand{ExtBroadcast(1, 0), ExtYmm(6)},
"carries a broadcast, the entry's memory operand takes none"},
} {
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
_, err := in.Encode(tt.ops)
@@ -734,6 +805,12 @@ func TestAmd64ExtMemoryFormRejects(t *testing.T) {
{"displacement past the signed 32-bit range on the store", "VMOVW", store,
[]ExtOperand{ExtXmm(30), ExtMemory(8, 1<<32)},
"outside the signed 32-bit range"},
{"broadcast on the load's memory position", "VMOVSH", load,
[]ExtOperand{ExtBroadcast(9, 0), ExtXmm(30)},
"carries a broadcast, the entry's memory operand takes none"},
{"broadcast destination on the store", "VMOVSH", store,
[]ExtOperand{ExtXmm(30), ExtBroadcast(9, 0)},
"carries a broadcast, the entry's memory operand takes none"},
} {
in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick)
_, err := in.Encode(tt.ops)