feat(arch): add the remaining scalar FP16 memory forms to the extension layer

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 01:35:48 +02:00
1 parent 454a21f5b7
commit 8d611bfdaf
2 files changed
+103 -24

No files matched your search

+58 -19
View File
@@ -400,24 +400,42 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) {
}
// encodeAmdVec3Imm fills the three-vector form with a control immediate:
// imm, src1, src2, dest, the order the reference listings write it in.
// imm, src1, src2, dest, the order the reference listings write it in. An
// entry with Mem set takes the memory shape of the second source, xmm3/m16
// in the manual.
func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
imm, err := in.amd64Imm8(ops[0], 1)
if err != nil {
return nil, err
}
for i, op := range ops[1:] {
if err := in.amd64Vector(op, class, i+2); err != nil {
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
if in.Mem == 3 && ops[2].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[2], 3)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[3], class, 4); err != nil {
return nil, err
}
out := amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp)
return append(out, imm), nil
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
if err := in.amd64Vector(ops[3], class, 4); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
return append(out, imm), nil
}
// encodeAmdMask2Imm fills the opmask-destination form with a control
// immediate: imm, src1, src2, dest.
// immediate: imm, src1, src2, dest. An entry with Mem set takes the memory
// shape of the second source.
func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
imm, err := in.amd64Imm8(ops[0], 1)
@@ -427,16 +445,25 @@ func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) {
if err := in.amd64Vector(ops[1], class, 2); err != nil {
return nil, err
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
if ops[3].Kind != ExtKReg {
return nil, fmt.Errorf("%s: operand 4 wants an opmask register, got %s", in.Name, ops[3].Kind)
}
if err := in.amd64PlainReg(ops[3], 7, 4); err != nil {
return nil, err
}
out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
var out []byte
if in.Mem == 3 && ops[2].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[2], 3)
if err != nil {
return nil, err
}
out = amd64EncodeMemory(in.Bytes, ops[3].Reg, ops[1].Reg, base, disp)
return append(out, imm), nil
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
out = amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg)
return append(out, imm), nil
}
@@ -471,12 +498,24 @@ var ExtFP16CmpPredicates = [32]string{
}
// encodeAmdVecGprVec fills the conversion form with a general-register
// source: src1, gpr, dest. VCVTSI2SH XMM1, XMM2, EAX style.
// source: src1, gpr, dest. VCVTSI2SH X1, X2, EAX style. An entry with Mem
// set takes the memory shape of the integer source, which the manual spells
// r/m32: the value converts straight out of memory.
func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
class := amd64LengthClass(in.Bytes)
if err := in.amd64Vector(ops[0], class, 1); err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
base, disp, err := in.amd64Memory(ops[1], 2)
if err != nil {
return nil, err
}
if err := in.amd64Vector(ops[2], class, 3); err != nil {
return nil, err
}
return amd64EncodeMemory(in.Bytes, ops[2].Reg, ops[0].Reg, base, disp), nil
}
if err := in.amd64Gpr(ops[1], 2); err != nil {
return nil, err
}
@@ -622,10 +661,10 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"},
{Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VSCALEFSH (EVEX.NDS.LIG.66.MAP6.W0 2D /r)"},
{Name: "VGETEXPSH", Summary: "Convert the exponent of a scalar FP16 value to an FP16 value",
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETEXPSH (EVEX.NDS.LIG.66.MAP6.W0 43 /r)"},
{Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS",
Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16,
@@ -646,16 +685,16 @@ var amd64Extensions = []ExtInstr{
Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"},
{Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 2A /r)"},
{Name: "VCVTSI2SH", Summary: "Convert one signed 64-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 2A /r)"},
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 32-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 7B /r)"},
{Name: "VCVTUSI2SH", Summary: "Convert one unsigned 64-bit integer to one FP16 value",
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"},
{Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer",
Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16,
@@ -749,15 +788,15 @@ var amd64Extensions = []ExtInstr{
// ExtFP16CmpPredicates above; the reserved upper nibble of the mantissa
// control is refused rather than encoded.
{Name: "VCMPSH", Summary: "Compare scalar FP16 values into an opmask under an imm8 predicate",
Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Mem: 3, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VCMPSH (EVEX.LLIG.F3.0F3A.W0 C2 /r /ib)"},
{Name: "VGETMANTSH", Summary: "Extract the normalised mantissa of a scalar FP16 value under an imm8 control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VGETMANTSH (EVEX.LLIG.NP.0F3A.W0 27 /r /ib)"},
{Name: "VREDUCESH", Summary: "Reduce a scalar FP16 value by imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VREDUCESH (EVEX.LLIG.NP.0F3A.W0 57 /r /ib)"},
{Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control",
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Mem: 3, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16,
Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"},
}
+45 -5
View File
@@ -408,6 +408,43 @@ var amd64GoldenRows = []amd64GoldenRow{
// list, both destinations being XMM, so the resolver cannot tell them
// apart and the register row above pins the 128-bit template alone.
// The remaining scalar memory forms: the scale and exponent extracts,
// the imm8-control group, and the integer converts, whose second
// source the manual spells r/m32. The W1 integer converts take the
// same operand list as the W0 ones, memory carrying no width to pick
// between them, so the memory rows pin the W0 templates and the W1
// entries rest on their register rows.
{"vscalefsh memory source", "VSCALEFSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624615002d31", "62 46 15 00 2d 31 vscalefsh (%r9),%xmm29,%xmm30"},
{"vgetexpsh memory source", "VGETEXPSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624615004331", "62 46 15 00 43 31 vgetexpsh (%r9),%xmm29,%xmm30"},
{"vgetexpsh memory source disp8", "VGETEXPSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
"6266150043717f", "62 66 15 00 43 71 7f vgetexpsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
{"vcmpsh memory source", "VCMPSH",
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtMemory(9, 0), ExtMask(5)},
"62d31600c2297b", "62 d3 16 00 c2 29 7b vcmpsh $0x7b,(%r9),%xmm29,%k5"},
{"vcmpsh memory source disp8", "VCMPSH",
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtMemory(1, 127), ExtMask(5)},
"62f31600c2697f7b", "62 f3 16 00 c2 69 7f 7b vcmpsh $0x7b,0xfe(%rcx),%xmm29,%k5 (Disp8(7f))"},
{"vgetmantsh memory source", "VGETMANTSH",
[]ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"6243140027310b", "62 43 14 00 27 31 7b vgetmantsh $0x7b,(%r9),%xmm29,%xmm30 (opcode row only)"},
{"vreducesh memory source", "VREDUCESH",
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"6243140057317b", "62 43 14 00 57 31 7b vreducesh $0x7b,(%r9),%xmm29,%xmm30"},
{"vrndscalesh memory source", "VRNDSCALESH",
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624314000a317b", "62 43 14 00 0a 31 7b vrndscalesh $0x7b,(%r9),%xmm29,%xmm30"},
{"vcvtsi2sh memory source", "VCVTSI2SH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516002a31", "62 45 16 00 2a 31 vcvtsi2shl (%r9),%xmm29,%xmm30"},
{"vcvtusi2sh memory source", "VCVTUSI2SH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516007b31", "62 45 16 00 7b 31 vcvtusi2shl (%r9),%xmm29,%xmm30"},
// High registers in a 512-bit form exercise the EVEX extension bits:
// with both sources above 15 the B bar and X bar bits clear, while the
// destination zmm23 keeps R bar set in byte one (derived from the
@@ -619,11 +656,14 @@ func TestAmd64ExtRejects(t *testing.T) {
{"memory in the arithmetic's destination", "VADDSH",
[]ExtOperand{ExtXmm(28), ExtXmm(29), ExtMemory(9, 0)},
"wants an XMM register"},
{"memory where the general register belongs", "VCVTSI2SH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"wants a 32-bit general register"},
{"memory as the compare's second source", "VCMPSH",
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtMemory(9, 0), ExtMask(5)},
{"memory as the compare's mask destination", "VCMPSH",
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtMemory(9, 0)},
"wants an opmask register"},
{"memory as the convert's first source", "VCVTSI2SH",
[]ExtOperand{ExtMemory(9, 0), ExtGpr32(2), ExtXmm(30)},
"wants an XMM register"},
{"memory as the mantissa control's first source", "VGETMANTSH",
[]ExtOperand{ExtImmediate(0x0b), ExtMemory(9, 0), ExtXmm(29), ExtXmm(30)},
"wants an XMM register"},
{"memory as the intersect source", "VP2INTERSECTD",
[]ExtOperand{ExtZmm(2), ExtMemory(9, 0), ExtMask(0)},