feat(arch): add the scalar FP16 memory forms to the amd64 extension layer

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 01:35:48 +02:00
1 parent 7d69dda874
commit d275dee3ae
3 files changed
+228 -20

No files matched your search

+85 -14
View File
@@ -18,9 +18,11 @@
// live in the generated table and the EVEX encoder, and a mnemonic the
// toolchain has is not an extension.
//
// Memory operands, write masking ({k1}{z}) and embedded rounding arrive with
// a later slice; every form here encodes the unmasked register forms, which
// is what the golden-vector path exercises.
// The forms encode the unmasked shapes: register forms throughout, and the
// scalar FP16 memory forms beside them, base-relative operands with the
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand. A
// scaled index, write masking ({k1}{z}) and embedded rounding still arrive
// with a later slice.
package arch
@@ -196,7 +198,11 @@ func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
// amd64Vector checks one vector operand against the class the entry encodes.
func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error {
if op.Kind != class {
return fmt.Errorf("%s: operand %d wants a %s, got %s", in.Name, pos, class, op.Kind)
article := "a"
if class == ExtXMM {
article = "an"
}
return fmt.Errorf("%s: operand %d wants %s %s, got %s", in.Name, pos, article, class, op.Kind)
}
return in.amd64PlainReg(op, 31, pos)
}
@@ -234,6 +240,10 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
return in.encodeAmdVecGprVec(ops)
case ExtFormAmdGprVec, ExtFormAmdVecGpr:
return in.encodeAmdGprPair(ops)
case ExtFormAmdMemVec:
return in.encodeAmdMemVec(ops)
case ExtFormAmdVecMem:
return in.encodeAmdVecMem(ops)
case ExtFormAmdVec3Imm:
return in.encodeAmdVec3Imm(ops)
case ExtFormAmdMask2Imm:
@@ -244,17 +254,66 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
}
// encodeAmdVec3 fills the non-destructive three-vector form: src1, src2,
// dest, all under one register class.
// dest, all under one register class. An entry with Mem set takes the
// memory shape of that position too: the second source of the scalar
// arithmetic, spelled xmm3/m16 in the manual, may be a base-relative
// operand, which rides the r/m field with its displacement bytes after the
// opcode.
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
for i, op := range ops {
if err := in.amd64Vector(op, class, i+1); err != nil {
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[1], 2)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
}
for i, op := range ops[1:] {
if err := in.amd64Vector(op, class, i+2); err != nil {
return nil, err
}
}
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
}
// encodeAmdMemVec fills the memory-load form: mem, dest. VMOVSH X30,
// 4660(R8) shape, the manual's xmm1, m16 lines beside the register form.
// The form reads one value from memory, so the third register slot stays
// unused, which the encoding spells as vvvv 1111.
func (in ExtInstr) encodeAmdMemVec(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
base, disp, err := in.amd64Memory(ops[0], 1)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
}
// encodeAmdVecMem fills the memory-store form: src, mem. VMOVSH 4660(R9),
// X29 shape, the manual's m16, xmm1 lines. The register source sits in the
// ModR/M reg field and the memory destination in r/m, and vvvv stays
// unused.
func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
base, disp, err := in.amd64Memory(ops[1], 2)
if err != nil {
return nil, err
}
return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil
}
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
// of destination, and at 128 bits the companion stays the class itself.
@@ -500,32 +559,44 @@ var amd64Extensions = []ExtInstr{
{Name: "VMOVSH", Summary: "Move a scalar FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.NDS.LIG.F3.MAP5.W0 10 /r)"},
{Name: "VMOVSH", Summary: "Move a scalar FP16 value from memory into an XMM register",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdMemVec, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.LIG.F3.MAP5.W0 10 /r, m16 source)"},
{Name: "VMOVSH", Summary: "Move a scalar FP16 value from an XMM register to memory",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x11, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.LIG.F3.MAP5.W0 11 /r, m16 destination)"},
{Name: "VMOVW", Summary: "Move a word between a general register and an XMM register",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdGprVec, Feature: ExtFeatureFP16, Wig: true,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r)"},
{Name: "VMOVW", Summary: "Move a word between an XMM register and a general register",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, Wig: true,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r)"},
{Name: "VMOVW", Summary: "Move a word from memory into an XMM register",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdMemVec, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r, m16 source)"},
{Name: "VMOVW", Summary: "Move a word from an XMM register to memory",
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r, m16 destination)"},
{Name: "VADDSH", Summary: "Add scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VADDSH (EVEX.NDS.LIG.F3.MAP5.W0 58 /r)"},
{Name: "VSUBSH", Summary: "Subtract scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSUBSH (EVEX.NDS.LIG.F3.MAP5.W0 5C /r)"},
{Name: "VMULSH", Summary: "Multiply scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMULSH (EVEX.NDS.LIG.F3.MAP5.W0 59 /r)"},
{Name: "VDIVSH", Summary: "Divide scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VDIVSH (EVEX.NDS.LIG.F3.MAP5.W0 5E /r)"},
{Name: "VMINSH", Summary: "Return the minimum of scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMINSH (EVEX.NDS.LIG.F3.MAP5.W0 5D /r)"},
{Name: "VMAXSH", Summary: "Return the maximum of scalar FP16 values",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VMAXSH (EVEX.NDS.LIG.F3.MAP5.W0 5F /r)"},
{Name: "VSQRTSH", Summary: "Compute the square root of a scalar FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"},
{Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
+139 -2
View File
@@ -264,6 +264,75 @@ var amd64GoldenRows = []amd64GoldenRow{
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620314000af47b", "62 03 14 00 0a f4 7b vrndscalesh $0x7b,%xmm28,%xmm29,%xmm30"},
// The memory forms of the scalar moves and arithmetic, against the same
// listings' memory rows. The zero-displacement rows match the GNU
// source spellings outright and every base R8+ row exercises the EVEX.B
// high-base bit. The disp8 rows pin the bytes the listing lays down;
// binutils mainline encodes EVEX displacements with the APX disp8*N
// scaling, so its source spellings (0xfe for the m16 rows, 0x1fc0 for
// the m512 ones) are N times the plain SDM displacement those bytes
// carry, and the operand lists here hold the plain displacement. The
// rows with no GNU line are derived: the disp32 form the SDM ModR/M
// table defines and the source listings never emit plain, and the SIB
// byte the R12 base demands.
{"vmovsh load from r9", "VMOVSH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
"62457e081031", "62 45 7e 08 10 31 vmovsh (%r9),%xmm30"},
{"vmovsh load disp8", "VMOVSH",
[]ExtOperand{ExtMemory(1, 127), ExtXmm(30)},
"62657e0810717f", "62 65 7e 08 10 71 7f vmovsh 0xfe(%rcx),%xmm30 (Disp8(7f))"},
{"vmovsh store to r9", "VMOVSH",
[]ExtOperand{ExtXmm(30), ExtMemory(9, 0)},
"62457e081131", "62 45 7e 08 11 31 vmovsh %xmm30,(%r9)"},
{"vmovsh store disp8", "VMOVSH",
[]ExtOperand{ExtXmm(30), ExtMemory(1, 127)},
"62657e0811717f", "62 65 7e 08 11 71 7f vmovsh %xmm30,0xfe(%rcx) (Disp8(7f))"},
{"vmovsh store negative disp32", "VMOVSH",
[]ExtOperand{ExtXmm(30), ExtMemory(13, -200)},
"62457e0811b538ffffff", ""},
{"vmovw load from r9", "VMOVW",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
"62457d086e31", "62 45 7d 08 6e 31 vmovw (%r9),%xmm30"},
{"vmovw load disp8", "VMOVW",
[]ExtOperand{ExtMemory(1, 127), ExtXmm(30)},
"62657d086e717f", "62 65 7d 08 6e 71 7f vmovw 0xfe(%rcx),%xmm30 (Disp8(7f))"},
{"vmovw store to r9", "VMOVW",
[]ExtOperand{ExtXmm(30), ExtMemory(9, 0)},
"62457d087e31", "62 45 7d 08 7e 31 vmovw %xmm30,(%r9)"},
{"vmovw store disp8", "VMOVW",
[]ExtOperand{ExtXmm(30), ExtMemory(1, 127)},
"62657d087e717f", "62 65 7d 08 7e 71 7f vmovw %xmm30,0xfe(%rcx) (Disp8(7f))"},
{"vaddsh memory source", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005831", "62 45 16 00 58 31 vaddsh (%r9),%xmm29,%xmm30"},
{"vaddsh memory source disp8", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
"6265160058717f", "62 65 16 00 58 71 7f vaddsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
{"vaddsh memory source disp32", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtMemory(2, 8128), ExtXmm(30)},
"6265160058b2c01f0000", ""},
{"vsubsh memory source", "VSUBSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005c31", "62 45 16 00 5c 31 vsubsh (%r9),%xmm29,%xmm30"},
{"vmulsh memory source disp8", "VMULSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
"6265160059717f", "62 65 16 00 59 71 7f vmulsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
{"vdivsh memory source", "VDIVSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005e31", "62 45 16 00 5e 31 vdivsh (%r9),%xmm29,%xmm30"},
{"vminsh memory source disp8", "VMINSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
"626516005d717f", "62 65 16 00 5d 71 7f vminsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
{"vmaxsh memory source", "VMAXSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005f31", "62 45 16 00 5f 31 vmaxsh (%r9),%xmm29,%xmm30"},
{"vsqrtsh memory source", "VSQRTSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005131", "62 45 16 00 51 31 vsqrtsh (%r9),%xmm29,%xmm30"},
{"vsqrtsh memory source negative disp8", "VSQRTSH",
[]ExtOperand{ExtXmm(29), ExtMemory(2, -128), ExtXmm(30)},
"62651600517280", "62 65 16 87 51 72 80 vsqrtsh -0x100(%rdx),%xmm29,%xmm30 (Disp8(80); the GNU row adds {k7}{z})"},
// High registers in a 512-bit form exercise the EVEX extension bits:
// with both sources above 15 the B bar and X bar bits clear, while the
// destination zmm23 keeps R bar set in byte one (derived from the
@@ -386,6 +455,9 @@ func TestAmd64ExtTemplateIntegrity(t *testing.T) {
if in.Form.Arity() < 2 || in.Form.Arity() > 4 {
t.Errorf("%s: form %s carries an unusable arity %d", in.Name, in.Form, in.Form.Arity())
}
if in.Mem > in.Form.Arity() {
t.Errorf("%s: Mem names operand %d, outside the form's %d positions", in.Name, in.Mem, in.Form.Arity())
}
}
}
@@ -466,6 +538,27 @@ func TestAmd64ExtRejects(t *testing.T) {
{"mask beyond k7 on the compare", "VCMPSH",
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtMask(8)},
"outside 0-7"},
{"memory in the arithmetic's first source", "VADDSH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(30)},
"wants an XMM register"},
{"memory in the arithmetic's destination", "VADDSH",
[]ExtOperand{ExtXmm(28), ExtXmm(29), ExtMemory(9, 0)},
"wants an XMM register"},
{"memory where the general register belongs", "VCVTSI2SH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"wants a 32-bit general register"},
{"memory as the compare's second source", "VCMPSH",
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtMemory(9, 0), ExtMask(5)},
"wants an XMM register"},
{"memory as the intersect source", "VP2INTERSECTD",
[]ExtOperand{ExtZmm(2), ExtMemory(9, 0), ExtMask(0)},
"wants a ZMM register"},
{"memory as the convert's source", "VCVTSS2SH",
[]ExtOperand{ExtXmm(28), ExtMemory(9, 0), ExtXmm(30)},
"wants an XMM register"},
{"memory as the control byte", "VGETMANTSH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
"wants an immediate control byte"},
} {
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
_, err := in.Encode(tt.ops)
@@ -486,6 +579,50 @@ func TestAmd64ExtRejects(t *testing.T) {
}
}
// TestAmd64ExtMemoryFormRejects covers the shapes the memory forms refuse:
// a register in the load's memory position, a memory operand in the store's
// register position, and the out-of-range bases and displacements. The
// rows resolve against the load and store entries themselves, which the
// name-and-class lookup cannot pick alone: the register forms of the same
// mnemonics share the class.
func TestAmd64ExtMemoryFormRejects(t *testing.T) {
load := func(in ExtInstr) bool { return in.Form == ExtFormAmdMemVec }
store := func(in ExtInstr) bool { return in.Form == ExtFormAmdVecMem }
for _, tt := range []struct {
name string
mnem string
pick func(ExtInstr) bool
ops []ExtOperand
quote string
}{
{"vector in the load's memory position", "VMOVSH", load,
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
"wants a memory operand"},
{"memory in the store's register position", "VMOVSH", store,
[]ExtOperand{ExtMemory(9, 0), ExtMemory(1, 0)},
"wants an XMM register"},
{"vector in the store's memory position", "VMOVSH", store,
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
"wants a memory operand"},
{"base beyond r15 on the load", "VMOVW", load,
[]ExtOperand{ExtMemory(16, 0), ExtXmm(30)},
"outside 0-15"},
{"displacement past the signed 32-bit range on the store", "VMOVW", store,
[]ExtOperand{ExtXmm(30), ExtMemory(8, 1<<32)},
"outside the signed 32-bit range"},
} {
in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick)
_, err := in.Encode(tt.ops)
if err == nil {
t.Errorf("%s: encode succeeded, want an error", tt.name)
continue
}
if !strings.Contains(err.Error(), tt.quote) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
}
// operandClass names the vector class a row exercises, the key the entry
// lookup resolves with.
func operandClass(t *testing.T, ops []ExtOperand) ExtOperandKind {
@@ -554,7 +691,7 @@ func TestAmd64ExtArchBinding(t *testing.T) {
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
}
}
if got := Extensions(AMD64); len(got) != 66 {
t.Errorf("the amd64 layer registers %d instructions, want 66", len(got))
if got := Extensions(AMD64); len(got) != 70 {
t.Errorf("the amd64 layer registers %d instructions, want 70", len(got))
}
}
+4 -4
View File
@@ -25,8 +25,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
{"VDPBF16PS", 3},
{"VP2INTERSECTD", 3},
{"VP2INTERSECTQ", 3},
{"VMOVSH", 1},
{"VMOVW", 2},
{"VMOVSH", 3},
{"VMOVW", 4},
{"VADDSH", 1},
{"VSQRTSH", 1},
{"VCOMISH", 1},
@@ -54,8 +54,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) {
t.Errorf("the %s lookup is not case-insensitive", tt.mnem)
}
}
if got := arch.Extensions(arch.AMD64); len(got) != 66 {
t.Errorf("the amd64 layer registers %d instructions, want 66", len(got))
if got := arch.Extensions(arch.AMD64); len(got) != 70 {
t.Errorf("the amd64 layer registers %d instructions, want 70", len(got))
}
if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok {
t.Error("a non-extended mnemonic resolved")