feat(arch): add the remaining scalar FP16 memory forms to the extension layer
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
454a21f5b7
commit
8d611bfdaf
2 files changed
+103
-24
No files matched your search
+58
-19
@@ -400,24 +400,42 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
|
||||
}
|
||||
|
||||
// encodeAmdVec3Imm fills the three-vector form with a control immediate:
|
||||
// imm, src1, src2, dest, the order the reference listings write it in.
|
||||
// imm, src1, src2, dest, the order the reference listings write it in. An
|
||||
// entry with Mem set takes the memory shape of the second source, xmm3/m16
|
||||
// in the manual.
|
||||
func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
imm, err := in.amd64Imm8(ops[0], 1)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for i, op := range ops[1:] {
|
||||
if err := in.amd64Vector(op, class, i+2); err != nil {
|
||||
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 3 && ops[2].Kind == ExtMem {
|
||||
base, disp, err := in.amd64Memory(ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[3], class, 4); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[3], class, 4); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
|
||||
// encodeAmdMask2Imm fills the opmask-destination form with a control
|
||||
// immediate: imm, src1, src2, dest.
|
||||
// immediate: imm, src1, src2, dest. An entry with Mem set takes the memory
|
||||
// shape of the second source.
|
||||
func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
imm, err := in.amd64Imm8(ops[0], 1)
|
||||
@@ -427,16 +445,25 @@ func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
|
||||
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if ops[3].Kind != ExtKReg {
|
||||
return nil, fmt.Errorf("%s: operand 4 wants an opmask register, got %s", in.Name, ops[3].Kind)
|
||||
}
|
||||
if err := in.amd64PlainReg(ops[3], 7, 4); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
|
||||
var out []byte
|
||||
if in.Mem == 3 && ops[2].Kind == ExtMem {
|
||||
base, disp, err := in.amd64Memory(ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
|
||||
return append(out, imm), nil
|
||||
}
|
||||
|
||||
@@ -471,12 +498,24 @@ var ExtFP16CmpPredicates = [32]string{
|
||||
}
|
||||
|
||||
// encodeAmdVecGprVec fills the conversion form with a general-register
|
||||
// source: src1, gpr, dest. VCVTSI2SH XMM1, XMM2, EAX style.
|
||||
// source: src1, gpr, dest. VCVTSI2SH X1, X2, EAX style. An entry with Mem
|
||||
// set takes the memory shape of the integer source, which the manual spells
|
||||
// r/m32: the value converts straight out of memory.
|
||||
func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
||||
base, disp, err := in.amd64Memory(ops[1], 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
|
||||
}
|
||||
if err := in.amd64Gpr(ops[1], 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -622,10 +661,10 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"},
|
||||
{Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VSCALEFSH (EVEX.NDS.LIG.66.MAP6.W0 2D /r)"},
|
||||
{Name: "VGETEXPSH", Summary: "Convert the exponent of a scalar FP16 value to an FP16 value",
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VGETEXPSH (EVEX.NDS.LIG.66.MAP6.W0 43 /r)"},
|
||||
{Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS",
|
||||
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
|
||||
@@ -646,16 +685,16 @@ var amd64Extensions = []ExtInstr{
|
||||
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"},
|
||||
{Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 2A /r)"},
|
||||
{Name: "VCVTSI2SH", Summary: "Convert one signed 64-bit integer to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 2A /r)"},
|
||||
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 32-bit integer to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 7B /r)"},
|
||||
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 64-bit integer to one FP16 value",
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"},
|
||||
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer",
|
||||
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
|
||||
@@ -749,15 +788,15 @@ var amd64Extensions = []ExtInstr{
|
||||
// ExtFP16CmpPredicates above; the reserved upper nibble of the mantissa
|
||||
// control is refused rather than encoded.
|
||||
{Name: "VCMPSH", Summary: "Compare scalar FP16 values into an opmask under an imm8 predicate",
|
||||
Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Mem: 3, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VCMPSH (EVEX.LLIG.F3.0F3A.W0 C2 /r /ib)"},
|
||||
{Name: "VGETMANTSH", Summary: "Extract the normalised mantissa of a scalar FP16 value under an imm8 control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VGETMANTSH (EVEX.LLIG.NP.0F3A.W0 27 /r /ib)"},
|
||||
{Name: "VREDUCESH", Summary: "Reduce a scalar FP16 value by imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VREDUCESH (EVEX.LLIG.NP.0F3A.W0 57 /r /ib)"},
|
||||
{Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control",
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
|
||||
Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"},
|
||||
}
|
||||
Reference in new issue
Block a user