diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index b0b3bb2..55409b4 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -8,8 +8,9 @@ // arch/amd64_gen.go stays untouched, and asm.Encodable keeps answering false // for every mnemonic here, so the layer stays out of the main encoders. // -// The families are AVX512-BF16, AVX512-VP2INTERSECT and the scalar core of -// AVX512-FP16, in their EVEX register forms. The encodings are transcribed +// The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the +// latter's scalar core with its imm8-control group and its packed 512-bit +// arithmetic, in their EVEX register forms. The encodings are transcribed // from the SDM instruction entries and cross-checked against binutils-gdb's // assembler testsuite; the golden vectors in amd64_ext_test.go pin the // bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, @@ -171,6 +172,10 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) { return in.encodeAmdVecGprVec(ops) case ExtFormAmdGprVec, ExtFormAmdVecGpr: return in.encodeAmdGprPair(ops) + case ExtFormAmdVec3Imm: + return in.encodeAmdVec3Imm(ops) + case ExtFormAmdMask2Imm: + return in.encodeAmdMask2Imm(ops) default: return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form) } @@ -225,6 +230,102 @@ func (in ExtInstr) encodeAmdMask2(ops []ExtOperand) ([]byte, error) { return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil } +// amd64Imm8 validates the leading immediate operand of an imm8-control form: +// an ExtImm with no shift, inside the unsigned byte range, and free of the +// bits the entry's control layout reserves. The reserved upper nibble of +// the VGETMANTSH control must encode as zero; the SDM marks every other +// layout here fully defined, and the VCMPSH hardware masks its predicate to +// five bits. +func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) { + if op.Kind != ExtImm { + return 0, fmt.Errorf("%s: operand %d wants an immediate control byte, got %s", in.Name, pos, op.Kind) + } + if op.Arr != ExtArrNone { + return 0, fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos) + } + if op.HasShift { + return 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 imm8 forms take none", in.Name, pos) + } + if op.Imm < 0 || op.Imm > 255 { + return 0, fmt.Errorf("%s: operand %d is immediate %d, outside the unsigned byte range 0-255", in.Name, pos, op.Imm) + } + if in.Imm8 == ExtImm8GetMant && op.Imm > 15 { + return 0, fmt.Errorf("%s: operand %d is immediate %d, the upper nibble of the mantissa control is reserved and must be zero", in.Name, pos, op.Imm) + } + return byte(op.Imm), nil +} + +// encodeAmdVec3Imm fills the three-vector form with a control immediate: +// imm, src1, src2, dest, the order the reference listings write it in. +func (in ExtInstr) encodeAmdVec3Imm(ops []ExtOperand) ([]byte, error) { + class := amd64LengthClass(in.Bytes) + imm, err := in.amd64Imm8(ops[0], 1) + if err != nil { + return nil, err + } + for i, op := range ops[1:] { + if err := in.amd64Vector(op, class, i+2); err != nil { + return nil, err + } + } + out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg) + return append(out, imm), nil +} + +// encodeAmdMask2Imm fills the opmask-destination form with a control +// immediate: imm, src1, src2, dest. +func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) { + class := amd64LengthClass(in.Bytes) + imm, err := in.amd64Imm8(ops[0], 1) + if err != nil { + return nil, err + } + if err := in.amd64Vector(ops[1], class, 2); err != nil { + return nil, err + } + if err := in.amd64Vector(ops[2], class, 3); err != nil { + return nil, err + } + if ops[3].Kind != ExtKReg { + return nil, fmt.Errorf("%s: operand 4 wants an opmask register, got %s", in.Name, ops[3].Kind) + } + if err := in.amd64PlainReg(ops[3], 7, 4); err != nil { + return nil, err + } + out := amd64Encode(in.Bytes, ops[3].Reg, ops[1].Reg, ops[2].Reg) + return append(out, imm), nil +} + +// ExtFP16RoundingModes names the two-bit rounding mode the round control of +// VRNDSCALESH and VREDUCESH carries, indexed by imm8[1:0], the SDM's RC +// field encoding. +var ExtFP16RoundingModes = [4]string{ + "round to nearest (even)", + "round down (toward -infinity)", + "round up (toward +infinity)", + "round toward zero (truncate)", +} + +// ExtFP16GetMantSigns names the sign control imm8[3:2] of the VGETMANTSH +// immediate, indexed by the field: the source's own sign, a forced positive, +// and the two encodings that yield the indefinite NaN on a negative source. +var ExtFP16GetMantSigns = [4]string{ + "the sign of the source", + "positive", + "the indefinite NaN when the source is negative", + "the indefinite NaN when the source is negative", +} + +// ExtFP16CmpPredicates names the 32 comparison predicates the VCMPSH +// immediate carries in imm8[4:0], in encoding order. The SDM's own +// spellings are the fixed vocabulary of the predicate suffixes. +var ExtFP16CmpPredicates = [32]string{ + "EQ_OQ", "LT_OS", "LE_OS", "UNORD_Q", "NEQ_UQ", "NLT_US", "NLE_US", "ORD_Q", + "EQ_UQ", "NGE_US", "NGT_US", "FALSE_OQ", "NEQ_OQ", "GE_OS", "GT_OS", "TRUE_UQ", + "EQ_OS", "LT_OQ", "LE_OQ", "UNORD_S", "NEQ_US", "NLT_UQ", "NLE_UQ", "ORD_S", + "EQ_US", "NGE_UQ", "NGT_UQ", "FALSE_OS", "NEQ_OS", "GE_OQ", "GT_OQ", "TRUE_US", +} + // encodeAmdVecGprVec fills the conversion form with a general-register // source: src1, gpr, dest. VCVTSI2SH XMM1, XMM2, EAX style. func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) { @@ -437,4 +538,26 @@ var amd64Extensions = []ExtInstr{ {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"}, + + // AVX512-FP16 scalar, the imm8-control group: mantissa extraction, + // reduction, rounding to fraction bits and the compare into an opmask. + // Each carries its control byte as the leading immediate operand, the + // order the reference listings write it in. The controls live in map + // 0F3A: the compare with the F3 prefix the manual gives the compare + // family, the other three unprefixed. The immediate layouts and their + // tables are ExtFP16RoundingModes, ExtFP16GetMantSigns and + // ExtFP16CmpPredicates above; the reserved upper nibble of the mantissa + // control is refused rather than encoded. + {Name: "VCMPSH", Summary: "Compare scalar FP16 values into an opmask under an imm8 predicate", + Bytes: []byte{0x62, 0x03, 0x06, 0x00, 0xC2, 0xC0}, Form: ExtFormAmdMask2Imm, Imm8: ExtImm8CmpPredicate, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCMPSH (EVEX.LLIG.F3.0F3A.W0 C2 /r /ib)"}, + {Name: "VGETMANTSH", Summary: "Extract the normalised mantissa of a scalar FP16 value under an imm8 control", + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x27, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8GetMant, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VGETMANTSH (EVEX.LLIG.NP.0F3A.W0 27 /r /ib)"}, + {Name: "VREDUCESH", Summary: "Reduce a scalar FP16 value by imm8 fraction bits under an imm8 round control", + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x57, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VREDUCESH (EVEX.LLIG.NP.0F3A.W0 57 /r /ib)"}, + {Name: "VRNDSCALESH", Summary: "Round a scalar FP16 value to imm8 fraction bits under an imm8 round control", + Bytes: []byte{0x62, 0x03, 0x04, 0x00, 0x0A, 0xC0}, Form: ExtFormAmdVec3Imm, Imm8: ExtImm8ScaleRound, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VRNDSCALESH (EVEX.LLIG.NP.0F3A.W0 0A /r /ib)"}, } diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 3994915..dea67c7 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -200,6 +200,27 @@ var amd64GoldenRows = []amd64GoldenRow{ []ExtOperand{ExtZmm(29), ExtZmm(30)}, "62057c4851f5", "62 05 7c 48 51 f5 vsqrtph %zmm29,%zmm30"}, + // The imm8-control group of the scalar core. The rows take the + // immediate first and the sources after it as src1, src2, the reverse + // of the listing's AT&T register order; every control byte is the $0x7b + // the suite drives through each imm8 form, save VGETMANTSH: the upper + // nibble of its control is reserved, so the layer enforces the SDM and + // encodes $0x0b where the suite's $0x7b would fault. The GNU line + // still proves the six opcode bytes, the immediate rides last as the + // operand it is. + {"vcmpsh", "VCMPSH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtMask(5)}, + "62931600c2ec7b", "62 93 16 00 c2 ec 7b vcmpsh $0x7b,%xmm28,%xmm29,%k5"}, + {"vgetmantsh", "VGETMANTSH", + []ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6203140027f40b", "62 03 14 00 27 f4 7b vgetmantsh $0x7b,%xmm28,%xmm29,%xmm30 (opcode row only)"}, + {"vreducesh", "VREDUCESH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6203140057f47b", "62 03 14 00 57 f4 7b vreducesh $0x7b,%xmm28,%xmm29,%xmm30"}, + {"vrndscalesh", "VRNDSCALESH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620314000af47b", "62 03 14 00 0a f4 7b vrndscalesh $0x7b,%xmm28,%xmm29,%xmm30"}, + // High registers in a 512-bit form exercise the EVEX extension bits: // with both sources above 15 the B bar and X bar bits clear, while the // destination zmm23 keeps R bar set in byte one (derived from the @@ -319,7 +340,7 @@ func TestAmd64ExtTemplateIntegrity(t *testing.T) { if in.Bytes[5]&0x3f != 0 || in.Bytes[5]&0xc0 != 0xc0 { t.Errorf("%s: byte five is %08b, want mod 11 with the reg and rm fields zero", in.Name, in.Bytes[5]) } - if in.Form.Arity() < 2 || in.Form.Arity() > 3 { + if in.Form.Arity() < 2 || in.Form.Arity() > 4 { t.Errorf("%s: form %s carries an unusable arity %d", in.Name, in.Form, in.Form.Arity()) } } @@ -381,6 +402,27 @@ func TestAmd64ExtRejects(t *testing.T) { {"predicate qualifier", "VCVTNEPS2BF16", []ExtOperand{{Kind: ExtZMM, Reg: 1, Qual: ExtQualZeroing}, ExtZmm(2)}, "predicate qualifier"}, + {"vector where the control byte belongs", "VGETMANTSH", + []ExtOperand{ExtXmm(28), ExtXmm(29), ExtXmm(30), ExtXmm(31)}, + "wants an immediate control byte"}, + {"reserved upper nibble on the mantissa control", "VGETMANTSH", + []ExtOperand{ExtImmediate(0x7b), ExtXmm(28), ExtXmm(29), ExtXmm(30)}, + "reserved and must be zero"}, + {"control byte under the floor", "VREDUCESH", + []ExtOperand{ExtImmediate(-1), ExtXmm(28), ExtXmm(29), ExtXmm(30)}, + "outside the unsigned byte range"}, + {"control byte over the top", "VRNDSCALESH", + []ExtOperand{ExtImmediate(256), ExtXmm(28), ExtXmm(29), ExtXmm(30)}, + "outside the unsigned byte range"}, + {"shift on the control byte", "VRNDSCALESH", + []ExtOperand{ExtShiftedImmediate(0x0b, 8), ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "take none"}, + {"vector in the mask position of the compare", "VCMPSH", + []ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtXmm(30)}, + "wants an opmask register"}, + {"mask beyond k7 on the compare", "VCMPSH", + []ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtMask(8)}, + "outside 0-7"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) @@ -415,6 +457,51 @@ func operandClass(t *testing.T, ops []ExtOperand) ExtOperandKind { return ExtXMM } +// TestAmd64ExtImm8Tables pins the imm8 semantics the layer carries as data +// against the SDM tables they are transcribed from: the rounding modes of +// the round control, the sign control of the mantissa extraction and the 32 +// comparison predicates, in encoding order. +func TestAmd64ExtImm8Tables(t *testing.T) { + roundModes := [4]string{ + "round to nearest (even)", + "round down (toward -infinity)", + "round up (toward +infinity)", + "round toward zero (truncate)", + } + if ExtFP16RoundingModes != roundModes { + t.Errorf("rounding modes %q, want the SDM RC field order", ExtFP16RoundingModes) + } + for i, sign := range ExtFP16GetMantSigns { + switch i { + case 0: + if sign != "the sign of the source" { + t.Errorf("sign control 0b00 = %q, want the source's own sign", sign) + } + case 1: + if sign != "positive" { + t.Errorf("sign control 0b01 = %q, want a forced positive", sign) + } + default: + if sign != "the indefinite NaN when the source is negative" { + t.Errorf("sign control 0b1x = %q, want the indefinite NaN branch", sign) + } + } + } + predicates := map[int]string{ + 0: "EQ_OQ", 1: "LT_OS", 2: "LE_OS", 3: "UNORD_Q", 4: "NEQ_UQ", + 5: "NLT_US", 6: "NLE_US", 7: "ORD_Q", 8: "EQ_UQ", 15: "TRUE_UQ", + 16: "EQ_OS", 23: "ORD_S", 24: "EQ_US", 27: "FALSE_OS", 31: "TRUE_US", + } + for i, want := range predicates { + if got := ExtFP16CmpPredicates[i]; got != want { + t.Errorf("predicate 0x%02x = %q, want %q", i, got, want) + } + } + if ExtFP16CmpPredicates[31] != "TRUE_US" { + t.Errorf("the predicate table ends at %q, want TRUE_US", ExtFP16CmpPredicates[31]) + } +} + // TestAmd64ExtArchBinding pins the layer's architecture binding: only riscv // and loong64 have no extended layer, arm64's lives in arm64_ext.go and the // amd64 one here. @@ -424,7 +511,7 @@ func TestAmd64ExtArchBinding(t *testing.T) { t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) } } - if got := Extensions(AMD64); len(got) != 48 { - t.Errorf("the amd64 layer registers %d instructions, want 48", len(got)) + if got := Extensions(AMD64); len(got) != 52 { + t.Errorf("the amd64 layer registers %d instructions, want 52", len(got)) } } diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index b64a5c2..4ba9dde 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -272,6 +272,16 @@ const ( // general-register destination: VMOVW EAX, X1 and VCVTSH2SI EAX, X1. // Operands: src, dest. ExtFormAmdVecGpr + // ExtFormAmdVec3Imm is the three-vector form with a control immediate, + // VGETMANTSH $11, X28, X29, X30 style: the immediate leads, the order + // the reference listings write it in. Operands: imm, src1, src2, dest. + // The immediate's layout is named by the entry's Imm8 kind. + ExtFormAmdVec3Imm + // ExtFormAmdMask2Imm is the opmask-destination form with a control + // immediate: VCMPSH $7, X28, X29, K5. Operands: imm, src1, src2, dest, + // where dest is an opmask register and the immediate's layout is named + // by the entry's Imm8 kind. + ExtFormAmdMask2Imm ) // Arity returns the operand count the form takes. @@ -285,6 +295,8 @@ func (f ExtForm) Arity() int { return 3 case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdGprVec, ExtFormAmdVecGpr: return 2 + case ExtFormAmdVec3Imm, ExtFormAmdMask2Imm: + return 4 default: return 0 } @@ -334,6 +346,10 @@ func (f ExtForm) String() string { return "general register, vector" case ExtFormAmdVecGpr: return "vector, general register" + case ExtFormAmdVec3Imm: + return "immediate, three vectors" + case ExtFormAmdMask2Imm: + return "immediate, two vectors into an opmask" default: return "unknown form" } @@ -350,6 +366,45 @@ const ( ExtFeatureSVE2 ExtFeature = "sve2" ) +// ExtImm8Kind names the imm8-control layout an amd64 extended entry carries +// in its immediate operand. The kind drives the validation at encode time: +// a value the manual reserves is an error, never a silent mis-encoding. +type ExtImm8Kind uint8 + +// The imm8-control layouts, ExtImm8None first as the zero value an entry +// without an immediate carries. +const ( + // ExtImm8None marks a form that takes no immediate operand. + ExtImm8None ExtImm8Kind = iota + // ExtImm8ScaleRound is the fraction-bits-plus-round-control layout of + // VRNDSCALESH and VREDUCESH: imm8[7:4] carries the number of fraction + // bits M, imm8[3] the precision-exception control, imm8[2] the rounding + // mode source and imm8[1:0] the rounding mode. Every byte encodes. + ExtImm8ScaleRound + // ExtImm8GetMant is the mantissa-extraction control of VGETMANTSH: + // imm8[3:2] the sign control, imm8[1:0] the normalisation interval, + // and imm8[7:4] reserved, which must encode as zero. + ExtImm8GetMant + // ExtImm8CmpPredicate is the comparison-predicate control of VCMPSH: + // imm8[4:0] names one of the 32 predicates and the bits above are + // masked away by the hardware, so every byte encodes. + ExtImm8CmpPredicate +) + +// String returns a short label for the imm8-control layout, for diagnostics. +func (k ExtImm8Kind) String() string { + switch k { + case ExtImm8ScaleRound: + return "fraction bits and round control" + case ExtImm8GetMant: + return "mantissa extraction control" + case ExtImm8CmpPredicate: + return "comparison predicate" + default: + return "no immediate" + } +} + // ExtInstr is one extended instruction: the metadata a lookup needs and the // encoding as data. Word holds the fixed bits of the 32-bit encoding with // every operand field and the size field zero; the form says which fields the @@ -372,6 +427,10 @@ type ExtInstr struct { // Wig records that the entry ignores the W bit in its general-register // position, so both 32-bit and 64-bit registers encode. Wig bool + // Imm8 names the imm8-control layout the amd64 entry's immediate + // operand carries, ExtImm8None when the form takes none. The arm64 + // entries all carry the zero value. + Imm8 ExtImm8Kind } // Encode assembles the operands into the 4 little-endian bytes of the @@ -393,7 +452,8 @@ func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) { case ExtFormSignedImmediate: return in.encodeSignedImmediate(ops) case ExtFormAmdVec3, ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdMask2, - ExtFormAmdVecGprVec, ExtFormAmdGprVec, ExtFormAmdVecGpr: + ExtFormAmdVecGprVec, ExtFormAmdGprVec, ExtFormAmdVecGpr, + ExtFormAmdVec3Imm, ExtFormAmdMask2Imm: return in.encodeAmd64(ops) default: return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form) diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index 92c579f..54b87cc 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -37,6 +37,10 @@ func TestAmd64ExtensionRegistry(t *testing.T) { {"VSQRTPH", 1}, {"VSCALEFSH", 1}, {"VGETEXPSH", 1}, + {"VCMPSH", 1}, + {"VGETMANTSH", 1}, + {"VREDUCESH", 1}, + {"VRNDSCALESH", 1}, } { cands, ok := LookupExtension(arch.AMD64, tt.mnem) if !ok { @@ -50,8 +54,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { t.Errorf("the %s lookup is not case-insensitive", tt.mnem) } } - if got := arch.Extensions(arch.AMD64); len(got) != 48 { - t.Errorf("the amd64 layer registers %d instructions, want 48", len(got)) + if got := arch.Extensions(arch.AMD64); len(got) != 52 { + t.Errorf("the amd64 layer registers %d instructions, want 52", len(got)) } if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { t.Error("a non-extended mnemonic resolved") @@ -115,6 +119,12 @@ func TestEncodeExtensionAmd64(t *testing.T) { {"word move into an xmm", "VMOVW", []arch.ExtOperand{arch.ExtGpr64(12), arch.ExtXmm(30)}, "62457d086ef4"}, + {"scalar compare into a mask", "VCMPSH", + []arch.ExtOperand{arch.ExtImmediate(0x7b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtMask(5)}, + "62931600c2ec7b"}, + {"mantissa extract with a control byte", "VGETMANTSH", + []arch.ExtOperand{arch.ExtImmediate(0x0b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}, + "6203140027f40b"}, } { got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...) if err != nil {