feat(arch): add the scalar FP16 memory forms to the amd64 extension layer
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
7d69dda874
commit
d275dee3ae
3 files changed
+228
-20
No files matched your search
+85
-14
@@ -18,9 +18,11 @@
|
||||
// live in the generated table and the EVEX encoder, and a mnemonic the
|
||||
// toolchain has is not an extension.
|
||||
//
|
||||
// Memory operands, write masking ({k1}{z}) and embedded rounding arrive with
|
||||
// a later slice; every form here encodes the unmasked register forms, which
|
||||
// is what the golden-vector path exercises.
|
||||
// The forms encode the unmasked shapes: register forms throughout, and the
|
||||
// scalar FP16 memory forms beside them, base-relative operands with the
|
||||
// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand. A
|
||||
// scaled index, write masking ({k1}{z}) and embedded rounding still arrive
|
||||
// with a later slice.
|
||||
|
||||
package arch
|
||||
|
||||
@@ -196,7 +198,11 @@ func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
|
||||
// amd64Vector checks one vector operand against the class the entry encodes.
|
||||
func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error {
|
||||
if op.Kind != class {
|
||||
return fmt.Errorf("%s: operand %d wants a %s, got %s", in.Name, pos, class, op.Kind)
|
||||
article := "a"
|
||||
if class == ExtXMM {
|
||||
article = "an"
|
||||
}
|
||||
return fmt.Errorf("%s: operand %d wants %s %s, got %s", in.Name, pos, article, class, op.Kind)
|
||||
}
|
||||
return in.amd64PlainReg(op, 31, pos)
|
||||
}
|
||||
@@ -234,6 +240,10 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
||||
return in.encodeAmdVecGprVec(ops)
|
||||
case ExtFormAmdGprVec, ExtFormAmdVecGpr:
|
||||
return in.encodeAmdGprPair(ops)
|
||||
case ExtFormAmdMemVec:
|
||||
return in.encodeAmdMemVec(ops)
|
||||
case ExtFormAmdVecMem:
|
||||
return in.encodeAmdVecMem(ops)
|
||||
case ExtFormAmdVec3Imm:
|
||||
return in.encodeAmdVec3Imm(ops)
|
||||
case ExtFormAmdMask2Imm:
|
||||
@@ -244,17 +254,66 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
||||
}
|
||||
|
||||
// encodeAmdVec3 fills the non-destructive three-vector form: src1, src2,
|
||||
// dest, all under one register class.
|
||||
// dest, all under one register class. An entry with Mem set takes the
|
||||
// memory shape of that position too: the second source of the scalar
|
||||
// arithmetic, spelled xmm3/m16 in the manual, may be a base-relative
|
||||
// operand, which rides the r/m field with its displacement bytes after the
|
||||
// opcode.
|
||||
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
for i, op := range ops {
|
||||
if err := in.amd64Vector(op, class, i+1); err != nil {
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
||||
base, disp, err := in.amd64Memory(ops[1], 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
|
||||
}
|
||||
for i, op := range ops[1:] {
|
||||
if err := in.amd64Vector(op, class, i+2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
|
||||
}
|
||||
|
||||
// encodeAmdMemVec fills the memory-load form: mem, dest. VMOVSH X30,
|
||||
// 4660(R8) shape, the manual's xmm1, m16 lines beside the register form.
|
||||
// The form reads one value from memory, so the third register slot stays
|
||||
// unused, which the encoding spells as vvvv 1111.
|
||||
func (in ExtInstr) encodeAmdMemVec(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
base, disp, err := in.amd64Memory(ops[0], 1)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64EncodeMemory(in.Bytes, ops[1].Reg, -1, base, disp), nil
|
||||
}
|
||||
|
||||
// encodeAmdVecMem fills the memory-store form: src, mem. VMOVSH 4660(R9),
|
||||
// X29 shape, the manual's m16, xmm1 lines. The register source sits in the
|
||||
// ModR/M reg field and the memory destination in r/m, and vvvv stays
|
||||
// unused.
|
||||
func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
base, disp, err := in.amd64Memory(ops[1], 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64EncodeMemory(in.Bytes, ops[0].Reg, -1, base, disp), nil
|
||||
}
|
||||
|
||||
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
|
||||
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
|
||||
// of destination, and at 128 bits the companion stays the class itself.
|
||||
@@ -500,32 +559,44 @@ var amd64Extensions = []ExtInstr{
|
||||
{Name: "VMOVSH", Summary: "Move a scalar FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.NDS.LIG.F3.MAP5.W0 10 /r)"},
|
||||
{Name: "VMOVSH", Summary: "Move a scalar FP16 value from memory into an XMM register",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdMemVec, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.LIG.F3.MAP5.W0 10 /r, m16 source)"},
|
||||
{Name: "VMOVSH", Summary: "Move a scalar FP16 value from an XMM register to memory",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x11, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.LIG.F3.MAP5.W0 11 /r, m16 destination)"},
|
||||
{Name: "VMOVW", Summary: "Move a word between a general register and an XMM register",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdGprVec, Feature: ExtFeatureFP16, Wig: true,
|
||||
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r)"},
|
||||
{Name: "VMOVW", Summary: "Move a word between an XMM register and a general register",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, Wig: true,
|
||||
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r)"},
|
||||
{Name: "VMOVW", Summary: "Move a word from memory into an XMM register",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdMemVec, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r, m16 source)"},
|
||||
{Name: "VMOVW", Summary: "Move a word from an XMM register to memory",
|
||||
Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r, m16 destination)"},
|
||||
{Name: "VADDSH", Summary: "Add scalar FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VADDSH (EVEX.NDS.LIG.F3.MAP5.W0 58 /r)"},
|
||||
{Name: "VSUBSH", Summary: "Subtract scalar FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSUBSH (EVEX.NDS.LIG.F3.MAP5.W0 5C /r)"},
|
||||
{Name: "VMULSH", Summary: "Multiply scalar FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMULSH (EVEX.NDS.LIG.F3.MAP5.W0 59 /r)"},
|
||||
{Name: "VDIVSH", Summary: "Divide scalar FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VDIVSH (EVEX.NDS.LIG.F3.MAP5.W0 5E /r)"},
|
||||
{Name: "VMINSH", Summary: "Return the minimum of scalar FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMINSH (EVEX.NDS.LIG.F3.MAP5.W0 5D /r)"},
|
||||
{Name: "VMAXSH", Summary: "Return the maximum of scalar FP16 values",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VMAXSH (EVEX.NDS.LIG.F3.MAP5.W0 5F /r)"},
|
||||
{Name: "VSQRTSH", Summary: "Compute the square root of a scalar FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"},
|
||||
{Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
|
||||
Reference in new issue
Block a user