From fb6d01a7d05a5a5d14db32bfec6195066c5a41fe Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 13:08:37 +0200 Subject: [PATCH] feat(arch): encode the amd64 embedded rounding and SAE decorations Assisted-by: GLM 5.3 --- arch/amd64_ext.go | 302 ++++++++++++++++++++++++++---------- arch/amd64_ext_test.go | 137 +++++++++++++++- arch/arm64_ext.go | 17 ++ asm/extension_amd64_test.go | 6 + 4 files changed, 381 insertions(+), 81 deletions(-) diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index e85ca81..7336e69 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -9,26 +9,31 @@ // for every mnemonic here, so the layer stays out of the main encoders. // // The families are AVX512-BF16, AVX512-VP2INTERSECT and AVX512-FP16, the -// latter's scalar core with its imm8-control group and its packed 512-bit -// arithmetic, in their EVEX register forms. The encodings are transcribed -// from the SDM instruction entries and cross-checked against binutils-gdb's -// assembler testsuite; the golden vectors in amd64_ext_test.go pin the -// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, -// no longer belong here: the Go toolchain's assembler knows them today, they +// latter's scalar core with its imm8-control group, its packed 512-bit and +// VL arithmetic and the embedded rounding of its FP operations, in their +// EVEX register forms. The encodings are transcribed from the SDM +// instruction entries and cross-checked against binutils-gdb's assembler +// testsuite; the golden vectors in amd64_ext_test.go pin the bytes. +// VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, no +// longer belong here: the Go toolchain's assembler knows them today, they // live in the generated table and the EVEX encoder, and a mnemonic the // toolchain has is not an extension. // // The forms encode the unmasked shapes: register forms throughout, and the -// memory forms beside them, the scalar ones the manual spells m16 and the -// packed ones with the {1toN} broadcast, base-relative operands with the -// ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 demand, the -// scaled index and the broadcast laying the SIB byte and EVEX.b over the -// same displacement semantics. The packed destinations take the write mask -// beside all of it, the {k1}{z} decorations the SDM spells: EVEX.aaa carries -// the masking register and EVEX.z the zeroing bit, K0 masks nothing, and the -// scalar forms, the compares, the opmask destinations and the load and store -// shapes take no mask at all. Embedded rounding still arrives with a later -// slice. +// memory forms beside them, the scalar ones the manual spells m16, m32 and +// m64 and the packed ones with the {1toN} broadcast, base-relative operands +// with the ModR/M disp8 and disp32 choices and the SIB byte RSP and R12 +// demand, the scaled index and the broadcast laying the SIB byte and EVEX.b +// over the same displacement semantics. The packed destinations take the +// write mask beside all of it, the {k1}{z} decorations the SDM spells: +// EVEX.aaa carries the masking register and EVEX.z the zeroing bit, K0 masks +// nothing, and the scalar forms, the compares, the opmask destinations and +// the load and store shapes take no mask at all. The FP destinations take +// the embedded rounding beside that, the {sae} and {rn-sae} through +// {rz-sae} decorations: EVEX.b selects the rounding context and EVEX.RC, +// the L'L bits it replaces, the mode, on the 512-bit and scalar register +// forms alone, exactly the two lengths whose word has no vector length left +// to lose. package arch @@ -159,6 +164,9 @@ func (in ExtInstr) amd64Memory(op ExtOperand, pos int) (base int, disp int64, er if op.HasShift { return 0, 0, fmt.Errorf("%s: operand %d carries a shift, the amd64 memory forms take none", in.Name, pos) } + if op.Round != ExtRoundNone { + return 0, 0, fmt.Errorf("%s: operand %d carries a rounding control, the memory operand takes none", in.Name, pos) + } if op.Broadcast && !in.Bcast { return 0, 0, fmt.Errorf("%s: operand %d carries a broadcast, the entry's memory operand takes none", in.Name, pos) } @@ -307,27 +315,63 @@ func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos in return amd64EncodeMemory(b, dest, vvvv, base, disp), nil } -// amd64WriteMask lifts the write mask spelling out of a destination operand: -// the returned copy carries the register bits alone, the mask register and -// the zeroing flag come back beside it. K0 never masks, zeroing is valid -// only beside a mask, and an entry whose destination takes none refuses the -// spellings outright, so nothing arrives at the encoding half-claimed. -func (in ExtInstr) amd64WriteMask(op ExtOperand, pos int) (ExtOperand, int, bool, error) { +// amd64WriteMask lifts the decorations off a destination operand: the +// returned copy carries the register bits alone, while the mask register, +// the zeroing flag and the rounding control come back beside it. K0 never +// masks, zeroing is valid only beside a mask, and an entry whose destination +// takes neither decoration refuses the spellings outright, so nothing +// arrives at the encoding half-claimed. The rounding decoration is +// validated against the entry's capability: an entry that carries neither +// Er nor Sae takes none, an Er entry demands one of the four modes, and an +// entry that suppresses exceptions alone takes {sae} and refuses every +// mode, the split the manual's two decoration classes draw. +func (in ExtInstr) amd64WriteMask(op ExtOperand, pos int) (ExtOperand, int, bool, ExtRounding, error) { stripped := op stripped.Mask, stripped.HasMask, stripped.Zeroing = 0, false, false - if !op.HasMask && !op.Zeroing { - return stripped, 0, false, nil + round := op.Round + stripped.Round = ExtRoundNone + if !op.HasMask && !op.Zeroing && round == ExtRoundNone { + return stripped, 0, false, ExtRoundNone, nil } - if !in.Mask { - return stripped, 0, false, fmt.Errorf("%s: operand %d carries a write mask, the entry's destination takes none", in.Name, pos) + if op.HasMask || op.Zeroing { + if !in.Mask { + return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries a write mask, the entry's destination takes none", in.Name, pos) + } + if !op.HasMask { + return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries zeroing without a write mask", in.Name, pos) + } + if op.Mask < 1 || op.Mask > 7 { + return stripped, 0, false, round, fmt.Errorf("%s: operand %d names write mask k%d, outside the masking registers k1-k7", in.Name, pos, op.Mask) + } } - if !op.HasMask { - return stripped, 0, false, fmt.Errorf("%s: operand %d carries zeroing without a write mask", in.Name, pos) + if round != ExtRoundNone { + if !in.Er && !in.Sae { + return stripped, 0, false, round, fmt.Errorf("%s: operand %d carries a rounding control, the entry's destination takes none", in.Name, pos) + } + if in.Sae && round != ExtRoundSAE { + return stripped, 0, false, round, fmt.Errorf("%s: operand %d names the rounding mode %s, the entry suppresses exceptions alone and takes {sae}", in.Name, pos, round) + } + if in.Er && round == ExtRoundSAE { + return stripped, 0, false, round, fmt.Errorf("%s: operand %d spells {sae} without a mode, the embedded rounding wants {rn-sae} through {rz-sae}", in.Name, pos) + } } - if op.Mask < 1 || op.Mask > 7 { - return stripped, 0, false, fmt.Errorf("%s: operand %d names write mask k%d, outside the masking registers k1-k7", in.Name, pos, op.Mask) + return stripped, op.Mask, op.Zeroing, round, nil +} + +// amd64ApplyRounding lays the validated rounding decoration into the encoded +// word: EVEX.b, bit four of byte three, selects the rounding context, and +// EVEX.RC, the L'L bits, names the mode. The L'L bits the template carries +// are cleared, because a word with an embedded rounding mode has no vector +// length of its own to spell; the write mask keeps its bits under the +// decoration, EVEX.aaa and EVEX.z untouched. An undecorated destination +// leaves the word alone: its L'L carries the vector length. +func amd64ApplyRounding(b []byte, mode ExtRounding) { + if mode == ExtRoundNone { + return } - return stripped, op.Mask, op.Zeroing, nil + b[3] &^= 0x60 + b[3] |= 0x10 + b[3] |= amd64RoundingBits(mode) << 5 } // amd64ApplyMask lays the validated write mask into the encoded word: @@ -361,6 +405,9 @@ func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error { if op.Zeroing { return fmt.Errorf("%s: operand %d carries zeroing, the position takes none", in.Name, pos) } + if op.Round != ExtRoundNone { + return fmt.Errorf("%s: operand %d carries a rounding control, the position takes none", in.Name, pos) + } if op.Arr != ExtArrNone { return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos) } @@ -442,17 +489,23 @@ func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) { // arithmetic, spelled xmm3/m16 in the manual, may be a base-relative // operand, which rides the r/m field with its displacement bytes after the // opcode, and the packed entries lay the source's {1toN} broadcast over the -// same encoding as EVEX.b. +// same encoding as EVEX.b. An entry with Er or Sae set takes the rounding +// decoration on its register form alone: the memory shape refuses one, the +// embedded rounding and the broadcast share EVEX.b and cannot ride the same +// word. func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err } if in.Mem == 2 && ops[1].Kind == ExtMem { - dest, mask, zeroing, err := in.amd64WriteMask(ops[2], 3) + dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3) if err != nil { return nil, err } + if round != ExtRoundNone { + return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name) + } if err := in.amd64Vector(dest, class, 3); err != nil { return nil, err } @@ -468,7 +521,7 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { return nil, err } } - dest, mask, zeroing, err := in.amd64WriteMask(ops[2], 3) + dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3) if err != nil { return nil, err } @@ -477,6 +530,7 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { } out := amd64Encode(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg) amd64ApplyMask(out, mask, zeroing) + amd64ApplyRounding(out, round) return out, nil } @@ -507,27 +561,32 @@ func (in ExtInstr) encodeAmdVecMem(ops []ExtOperand) ([]byte, error) { // encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows // the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits // of destination, and at 128 bits the companion stays the class itself. An -// entry with Mem set takes the memory shape of that position too: the +// entry with Mem set takes the memory shape of the source position too: the // compares and the packed square root read their source from memory, the // packed square root's source carrying the {1toN} broadcast as EVEX.b, and -// the narrow BF16 convert reads its full-width source there. +// the narrow BF16 convert reads its full-width source there. An entry with +// Er or Sae set takes the rounding decoration on its register form alone; +// the memory shape refuses one. func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) destClass := class if in.Form == ExtFormAmdVec2Half { destClass = amd64HalfClass(class) } - dest, mask, zeroing, err := in.amd64WriteMask(ops[1], 2) + dest, mask, zeroing, round, err := in.amd64WriteMask(ops[1], 2) if err != nil { return nil, err } if in.Mem == 1 && ops[0].Kind == ExtMem { // The memory spelling is validated before the destination: on the - // half form the destination's narrowed class is the likelier + // narrow form the destination's narrowed class is the likelier // rejection, but a miswritten source spelling names itself first. if _, _, err := in.amd64Memory(ops[0], 1); err != nil { return nil, err } + if round != ExtRoundNone { + return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name) + } if err := in.amd64Vector(dest, destClass, 2); err != nil { return nil, err } @@ -538,17 +597,6 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { amd64ApplyMask(out, mask, zeroing) return out, nil } - if in.Mem == 2 && ops[1].Kind == ExtMem { - if err := in.amd64Vector(ops[0], class, 1); err != nil { - return nil, err - } - out, err := in.amd64MemBytes(in.Bytes, ops[0].Reg, -1, dest, 2) - if err != nil { - return nil, err - } - amd64ApplyMask(out, mask, zeroing) - return out, nil - } if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err } @@ -557,6 +605,7 @@ func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { } out := amd64Encode(in.Bytes, dest.Reg, -1, ops[0].Reg) amd64ApplyMask(out, mask, zeroing) + amd64ApplyRounding(out, round) return out, nil } @@ -601,6 +650,9 @@ func (in ExtInstr) amd64Imm8(op ExtOperand, pos int) (byte, error) { if op.Zeroing { return 0, fmt.Errorf("%s: operand %d carries zeroing, the immediate takes none", in.Name, pos) } + if op.Round != ExtRoundNone { + return 0, fmt.Errorf("%s: operand %d carries a rounding control, the immediate takes none", in.Name, pos) + } if op.Imm < 0 || op.Imm > 255 { return 0, fmt.Errorf("%s: operand %d is immediate %d, outside the unsigned byte range 0-255", in.Name, pos, op.Imm) } @@ -676,7 +728,7 @@ func (in ExtInstr) encodeAmdMask2Imm(ops []ExtOperand) ([]byte, error) { return append(out, imm), nil } -// ExtFP16RoundingModes names the two-bit rounding mode the round control of +// ExtRoundingMode names the two-bit rounding mode the round control of // VRNDSCALESH and VREDUCESH carries, indexed by imm8[1:0], the SDM's RC // field encoding. var ExtFP16RoundingModes = [4]string{ @@ -686,6 +738,87 @@ var ExtFP16RoundingModes = [4]string{ "round toward zero (truncate)", } +// ExtRounding names the rounding decoration an FP destination of the amd64 +// layer carries, the SDM's {sae} and {rn-sae} through {rz-sae} spellings. +// The decoration is the EVEX embedded rounding: EVEX.b, bit four of byte +// three, selects the rounding context, and EVEX.RC, the two bits above it +// that carry L'L in every other word, names the mode, in the very encoding +// the imm8 round control uses: 00 round to nearest even, 01 down toward +// negative infinity, 10 up toward positive infinity, 11 toward zero. The +// L'L bits the template carries are cleared when the decoration applies, +// which is why only the 512-bit and the scalar forms take one: a word with +// an embedded mode has no vector length of its own to spell. ExtRoundSAE +// suppresses the floating-point exceptions alone and leaves EVEX.RC zero; +// the entries that take it (Er) demand a mode, the entries that take SAE +// alone refuse every mode, exactly the split the manual's two decoration +// classes draw. +type ExtRounding uint8 + +// The rounding decorations, ExtRoundNone first as the zero value every +// undecorated destination carries. +const ( + // ExtRoundNone marks a destination without a rounding decoration: the + // MXCSR rules govern the operation. + ExtRoundNone ExtRounding = iota + // ExtRoundSAE is {sae}: suppress all floating-point exceptions, no mode + // named. + ExtRoundSAE + // ExtRoundNearest is {rn-sae}: round to nearest, ties to even, EVEX.RC 00. + ExtRoundNearest + // ExtRoundDown is {rd-sae}: round down toward negative infinity, EVEX.RC 01. + ExtRoundDown + // ExtRoundUp is {ru-sae}: round up toward positive infinity, EVEX.RC 10. + ExtRoundUp + // ExtRoundTruncate is {rz-sae}: round toward zero, EVEX.RC 11. + ExtRoundTruncate +) + +// String returns the spelling the decoration carries in the reference +// listings, for diagnostics. +func (r ExtRounding) String() string { + switch r { + case ExtRoundSAE: + return "{sae}" + case ExtRoundNearest: + return "{rn-sae}" + case ExtRoundDown: + return "{rd-sae}" + case ExtRoundUp: + return "{ru-sae}" + case ExtRoundTruncate: + return "{rz-sae}" + default: + return "no rounding control" + } +} + +// amd64RoundingBits lays the decoration's EVEX.RC encoding out: the four +// named modes in the SDM's RC field order, and {sae}, which names no mode, +// as the nearest-even bits the exception suppression shares. +func amd64RoundingBits(r ExtRounding) byte { + switch r { + case ExtRoundDown: + return 1 + case ExtRoundUp: + return 2 + case ExtRoundTruncate: + return 3 + default: + return 0 + } +} + +// ExtRounded builds the rounded spelling of an FP destination, ZMM1{k7}{z}, +// {rz-sae} style: the operation rounds by the named mode instead of the +// MXCSR, or suppresses its exceptions under {sae}. The mode rides the +// destination operand beside the write mask, the two decorations the +// reference listings spell on the one line; they compose, EVEX.aaa and +// EVEX.z keeping their bits under EVEX.RC. +func ExtRounded(dest ExtOperand, mode ExtRounding) ExtOperand { + dest.Round = mode + return dest +} + // ExtFP16GetMantSigns names the sign control imm8[3:2] of the VGETMANTSH // immediate, indexed by the field: the source's own sign, a forced positive, // and the two encodings that yield the indefinite NaN on a negative source. @@ -709,25 +842,36 @@ var ExtFP16CmpPredicates = [32]string{ // encodeAmdVecGprVec fills the conversion form with a general-register // source: src1, gpr, dest. VCVTSI2SH X1, X2, EAX style. An entry with Mem // set takes the memory shape of the integer source, which the manual spells -// r/m32: the value converts straight out of memory. +// r/m32: the value converts straight out of memory. The register form takes +// the rounding decoration the integer converts carry; the memory shape +// refuses one. func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) { class := amd64LengthClass(in.Bytes) if err := in.amd64Vector(ops[0], class, 1); err != nil { return nil, err } + dest, _, _, round, err := in.amd64WriteMask(ops[2], 3) + if err != nil { + return nil, err + } if in.Mem == 2 && ops[1].Kind == ExtMem { - if err := in.amd64Vector(ops[2], class, 3); err != nil { + if round != ExtRoundNone { + return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name) + } + if err := in.amd64Vector(dest, class, 3); err != nil { return nil, err } - return in.amd64MemBytes(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1], 2) + return in.amd64MemBytes(in.Bytes, dest.Reg, ops[0].Reg, ops[1], 2) } if err := in.amd64Gpr(ops[1], 2); err != nil { return nil, err } - if err := in.amd64Vector(ops[2], class, 3); err != nil { + if err := in.amd64Vector(dest, class, 3); err != nil { return nil, err } - return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil + out := amd64Encode(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg) + amd64ApplyRounding(out, round) + return out, nil } // encodeAmdGprPair fills the two-operand general-register forms: gpr, vec @@ -845,40 +989,40 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecMem, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r, m16 destination)"}, {Name: "VADDSH", Summary: "Add scalar FP16 values", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDSH (EVEX.NDS.LIG.F3.MAP5.W0 58 /r)"}, {Name: "VSUBSH", Summary: "Subtract scalar FP16 values", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBSH (EVEX.NDS.LIG.F3.MAP5.W0 5C /r)"}, {Name: "VMULSH", Summary: "Multiply scalar FP16 values", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULSH (EVEX.NDS.LIG.F3.MAP5.W0 59 /r)"}, {Name: "VDIVSH", Summary: "Divide scalar FP16 values", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVSH (EVEX.NDS.LIG.F3.MAP5.W0 5E /r)"}, {Name: "VMINSH", Summary: "Return the minimum of scalar FP16 values", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINSH (EVEX.NDS.LIG.F3.MAP5.W0 5D /r)"}, {Name: "VMAXSH", Summary: "Return the maximum of scalar FP16 values", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXSH (EVEX.NDS.LIG.F3.MAP5.W0 5F /r)"}, {Name: "VSQRTSH", Summary: "Compute the square root of a scalar FP16 value", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"}, {Name: "VSCALEFSH", Summary: "Scale a scalar FP16 value by the ratio of two others", - Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSCALEFSH (EVEX.NDS.LIG.66.MAP6.W0 2D /r)"}, {Name: "VGETEXPSH", Summary: "Convert the exponent of a scalar FP16 value to an FP16 value", - Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x06, 0x05, 0x00, 0x43, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Sae: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VGETEXPSH (EVEX.NDS.LIG.66.MAP6.W0 43 /r)"}, {Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCOMISH (EVEX.LIG.MAP5.W0 2F /r)"}, {Name: "VUCOMISH", Summary: "Unordered-compare a scalar FP16 value and set EFLAGS", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Sae: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"}, {Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"}, {Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value", Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, @@ -887,19 +1031,19 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"}, {Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"}, {Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 2A /r)"}, {Name: "VCVTSI2SH", Summary: "Convert one signed 64-bit integer to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 2A /r)"}, {Name: "VCVTUSI2SH", Summary: "Convert one unsigned 32-bit integer to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 7B /r)"}, {Name: "VCVTUSI2SH", Summary: "Convert one unsigned 64-bit integer to one FP16 value", - Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Mem: 2, Er: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"}, {Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer", Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, @@ -920,7 +1064,7 @@ var amd64Extensions = []ExtInstr{ // quoted from x86-64-avx512_fp16.d, the 256- and 128-bit ones from // avx512_fp16_vl.d, on the same low registers the suite uses. {Name: "VADDPH", Summary: "Add packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.512.MAP5.W0 58 /r)"}, {Name: "VADDPH", Summary: "Add packed FP16 values", Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, @@ -929,7 +1073,7 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VADDPH (EVEX.NDS.128.MAP5.W0 58 /r)"}, {Name: "VSUBPH", Summary: "Subtract packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.512.MAP5.W0 5C /r)"}, {Name: "VSUBPH", Summary: "Subtract packed FP16 values", Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, @@ -938,7 +1082,7 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSUBPH (EVEX.NDS.128.MAP5.W0 5C /r)"}, {Name: "VMULPH", Summary: "Multiply packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.512.MAP5.W0 59 /r)"}, {Name: "VMULPH", Summary: "Multiply packed FP16 values", Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, @@ -947,7 +1091,7 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMULPH (EVEX.NDS.128.MAP5.W0 59 /r)"}, {Name: "VDIVPH", Summary: "Divide packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.512.MAP5.W0 5E /r)"}, {Name: "VDIVPH", Summary: "Divide packed FP16 values", Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, @@ -956,7 +1100,7 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VDIVPH (EVEX.NDS.128.MAP5.W0 5E /r)"}, {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Sae: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.512.MAP5.W0 5D /r)"}, {Name: "VMINPH", Summary: "Return the minimum of packed FP16 values", Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, @@ -965,7 +1109,7 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMINPH (EVEX.NDS.128.MAP5.W0 5D /r)"}, {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Sae: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.512.MAP5.W0 5F /r)"}, {Name: "VMAXPH", Summary: "Return the maximum of packed FP16 values", Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, @@ -974,7 +1118,7 @@ var amd64Extensions = []ExtInstr{ Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Mem: 2, Mask: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VMAXPH (EVEX.NDS.128.MAP5.W0 5F /r)"}, {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", - Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, + Bytes: []byte{0x62, 0x05, 0x04, 0x40, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Er: true, Bcast: true, Feature: ExtFeatureFP16, Ref: "Intel SDM Vol. 2C, VSQRTPH (EVEX.512.MAP5.W0 51 /r)"}, {Name: "VSQRTPH", Summary: "Compute the square root of packed FP16 values", Bytes: []byte{0x62, 0x05, 0x04, 0x20, 0x51, 0xC0}, Form: ExtFormAmdVec2, Mem: 1, Mask: true, Bcast: true, Feature: ExtFeatureFP16, diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 703c7bc..a6122dc 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -551,6 +551,115 @@ var amd64GoldenRows = []amd64GoldenRow{ {"vsqrtph k3 zeroing", "VSQRTPH", []ExtOperand{ExtZmm(29), ExtWriteMasked(ExtZmm(30), 3, true)}, "62057ccb51f5", ""}, + + // The embedded rounding and the exception suppression, the EVEX.RC + // decorations the FP destinations of the 512-bit and the scalar register + // forms carry: EVEX.b selects the rounding context and EVEX.RC replaces + // L'L, naming the mode 00 nearest even through 11 toward zero in the very + // encoding the imm8 round control shares. Every row is quoted from the + // local GNU assembler's own output, whose FP16 table matches the SDM + // entry for entry; the AT&T listings spell the decoration ahead of the + // operands. The write mask composes under the decoration, its EVEX.aaa + // and EVEX.z bits untouched by EVEX.RC. + {"vaddph rn-sae", "VADDPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f5541858f4", "62 f5 54 18 58 f4 vaddph {rn-sae},%zmm4,%zmm5,%zmm6"}, + {"vaddph rd-sae", "VADDPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundDown)}, + "62f5543858f4", "62 f5 54 38 58 f4 vaddph {rd-sae},%zmm4,%zmm5,%zmm6"}, + {"vaddph ru-sae", "VADDPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundUp)}, + "62f5545858f4", "62 f5 54 58 58 f4 vaddph {ru-sae},%zmm4,%zmm5,%zmm6"}, + {"vaddph rz-sae", "VADDPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundTruncate)}, + "62f5547858f4", "62 f5 54 78 58 f4 vaddph {rz-sae},%zmm4,%zmm5,%zmm6"}, + {"vaddph rz-sae under k7", "VADDPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtWriteMasked(ExtZmm(6), 7, false), ExtRoundTruncate)}, + "62f5547f58f4", "62 f5 54 7f 58 f4 vaddph {rz-sae},%zmm4,%zmm5,%zmm6{%k7}"}, + {"vaddph rz-sae under k7, zeroing", "VADDPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtWriteMasked(ExtZmm(6), 7, true), ExtRoundTruncate)}, + "62f554ff58f4", "62 f5 54 ff 58 f4 vaddph {rz-sae},%zmm4,%zmm5,%zmm6{%k7}{z}"}, + {"vaddph rz-sae, high registers", "VADDPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtRounded(ExtZmm(30), ExtRoundTruncate)}, + "6205147058f4", "62 05 14 70 58 f4 vaddph {rz-sae},%zmm28,%zmm29,%zmm30"}, + {"vsubph rz-sae", "VSUBPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundTruncate)}, + "62f554785cf4", "62 f5 54 78 5c f4 vsubph {rz-sae},%zmm4,%zmm5,%zmm6"}, + {"vmulph rn-sae", "VMULPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f5541859f4", "62 f5 54 18 59 f4 vmulph {rn-sae},%zmm4,%zmm5,%zmm6"}, + {"vdivph rn-sae", "VDIVPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundNearest)}, + "62f554185ef4", "62 f5 54 18 5e f4 vdivph {rn-sae},%zmm4,%zmm5,%zmm6"}, + {"vminph sae", "VMINPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundSAE)}, + "62f554185df4", "62 f5 54 18 5d f4 vminph {sae},%zmm4,%zmm5,%zmm6"}, + {"vminph sae, high registers", "VMINPH", + []ExtOperand{ExtZmm(29), ExtZmm(28), ExtRounded(ExtZmm(30), ExtRoundSAE)}, + "620514105df4", "62 05 14 10 5d f4 vminph {sae},%zmm28,%zmm29,%zmm30"}, + {"vmaxph sae", "VMAXPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundSAE)}, + "62f554185ff4", "62 f5 54 18 5f f4 vmaxph {sae},%zmm4,%zmm5,%zmm6"}, + {"vsqrtph rz-sae", "VSQRTPH", + []ExtOperand{ExtZmm(4), ExtRounded(ExtZmm(5), ExtRoundTruncate)}, + "62f57c7851ec", "62 f5 7c 78 51 ec vsqrtph {rz-sae},%zmm4,%zmm5"}, + {"vaddsh rn-sae", "VADDSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundNearest)}, + "62f5561858f4", "62 f5 56 18 58 f4 vaddsh {rn-sae},%xmm4,%xmm5,%xmm6"}, + {"vaddsh rz-sae, high registers", "VADDSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundTruncate)}, + "6205167058f4", "62 05 16 70 58 f4 vaddsh {rz-sae},%xmm28,%xmm29,%xmm30"}, + {"vsubsh rd-sae", "VSUBSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundDown)}, + "62f556385cf4", "62 f5 56 38 5c f4 vsubsh {rd-sae},%xmm4,%xmm5,%xmm6"}, + {"vmulsh ru-sae", "VMULSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundUp)}, + "62f5565859f4", "62 f5 56 58 59 f4 vmulsh {ru-sae},%xmm4,%xmm5,%xmm6"}, + {"vdivsh rz-sae", "VDIVSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundTruncate)}, + "62f556785ef4", "62 f5 56 78 5e f4 vdivsh {rz-sae},%xmm4,%xmm5,%xmm6"}, + {"vminsh sae", "VMINSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundSAE)}, + "62f556185df4", "62 f5 56 18 5d f4 vminsh {sae},%xmm4,%xmm5,%xmm6"}, + {"vmaxsh sae", "VMAXSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundSAE)}, + "62f556185ff4", "62 f5 56 18 5f f4 vmaxsh {sae},%xmm4,%xmm5,%xmm6"}, + {"vsqrtsh rz-sae", "VSQRTSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundTruncate)}, + "62f5567851f4", "62 f5 56 78 51 f4 vsqrtsh {rz-sae},%xmm4,%xmm5,%xmm6"}, + {"vsqrtsh ru-sae, high registers", "VSQRTSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundUp)}, + "6205165051f4", "62 05 16 50 51 f4 vsqrtsh {ru-sae},%xmm28,%xmm29,%xmm30"}, + {"vscalefsh rn-sae", "VSCALEFSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundNearest)}, + "62f655182df4", "62 f6 55 18 2d f4 vscalefsh {rn-sae},%xmm4,%xmm5,%xmm6"}, + {"vgetexpsh sae", "VGETEXPSH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundSAE)}, + "62f6551843f4", "62 f6 55 18 43 f4 vgetexpsh {sae},%xmm4,%xmm5,%xmm6"}, + {"vcomish sae", "VCOMISH", + []ExtOperand{ExtXmm(29), ExtRounded(ExtXmm(30), ExtRoundSAE)}, + "62057c182ff5", "62 05 7c 18 2f f5 vcomish {sae},%xmm29,%xmm30"}, + {"vucomish sae", "VUCOMISH", + []ExtOperand{ExtXmm(29), ExtRounded(ExtXmm(30), ExtRoundSAE)}, + "62057c182ef5", "62 05 7c 18 2e f5 vucomish {sae},%xmm29,%xmm30"}, + {"vcvtss2sh rn-sae", "VCVTSS2SH", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtRounded(ExtXmm(6), ExtRoundNearest)}, + "62f554181df4", "62 f5 54 18 1d f4 vcvtss2sh {rn-sae},%xmm4,%xmm5,%xmm6"}, + {"vcvtsd2sh rd-sae, high registers", "VCVTSD2SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundDown)}, + "620597305af4", "62 05 97 30 5a f4 vcvtsd2sh {rd-sae},%xmm28,%xmm29,%xmm30"}, + {"vcvtsi2sh rn-sae, 32-bit", "VCVTSI2SH", + []ExtOperand{ExtXmm(5), ExtGpr32(0), ExtRounded(ExtXmm(6), ExtRoundNearest)}, + "62f556182af0", "62 f5 56 18 2a f0 vcvtsi2sh %eax,{rn-sae},%xmm5,%xmm6"}, + {"vcvtsi2sh rz-sae, 64-bit, high registers", "VCVTSI2SH", + []ExtOperand{ExtXmm(29), ExtGpr64(12), ExtRounded(ExtXmm(30), ExtRoundTruncate)}, + "624596702af4", "62 45 96 70 2a f4 vcvtsi2sh %r12,{rz-sae},%xmm29,%xmm30"}, + {"vcvtusi2sh ru-sae, 32-bit", "VCVTUSI2SH", + []ExtOperand{ExtXmm(5), ExtGpr32(2), ExtRounded(ExtXmm(6), ExtRoundUp)}, + "62f556587bf2", "62 f5 56 58 7b f2 vcvtusi2sh %edx,{ru-sae},%xmm5,%xmm6"}, + {"vcvtusi2sh rd-sae, 64-bit, high registers", "VCVTUSI2SH", + []ExtOperand{ExtXmm(29), ExtGpr64(12), ExtRounded(ExtXmm(30), ExtRoundDown)}, + "624596307bf4", "62 45 96 30 7b f4 vcvtusi2sh %r12,{rd-sae},%xmm29,%xmm30"}, } // amd64ResolveEntry finds the table entry a golden row exercises: the entry @@ -767,8 +876,8 @@ func TestAmd64ExtRejects(t *testing.T) { {"memory as the intersect source", "VP2INTERSECTD", []ExtOperand{ExtZmm(2), ExtMemory(9, 0), ExtMask(0)}, "wants a ZMM register"}, - {"memory as the convert's source", "VCVTSS2SH", - []ExtOperand{ExtXmm(28), ExtMemory(9, 0), ExtXmm(30)}, + {"memory as the convert's first source", "VCVTSS2SH", + []ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(30)}, "wants an XMM register"}, {"memory as the control byte", "VGETMANTSH", []ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(29), ExtXmm(30)}, @@ -830,6 +939,30 @@ func TestAmd64ExtRejects(t *testing.T) { {"write mask where the destination is memory", "VCOMISH", []ExtOperand{ExtXmm(30), ExtOperand{Kind: ExtMem, Reg: 9, HasMask: true, Mask: 2}}, "the entry's destination takes none"}, + {"rounding on the 256-bit arithmetic", "VADDPH", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtRounded(ExtYmm(6), ExtRoundNearest)}, + "the entry's destination takes none"}, + {"bare {sae} on the embedded-rounding arithmetic", "VADDPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundSAE)}, + "spells {sae} without a mode"}, + {"a named mode on the exception-suppressing minimum", "VMINPH", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtRounded(ExtZmm(6), ExtRoundDown)}, + "suppresses exceptions alone"}, + {"rounding over a memory source", "VADDSH", + []ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtRounded(ExtXmm(30), ExtRoundNearest)}, + "the memory form takes no rounding control"}, + {"rounding on a source position", "VADDPH", + []ExtOperand{ExtZmm(5), ExtRounded(ExtZmm(4), ExtRoundUp), ExtZmm(6)}, + "the position takes none"}, + {"rounding on the control byte", "VRNDSCALESH", + []ExtOperand{{Kind: ExtImm, Imm: 0x0b, Round: ExtRoundSAE}, ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "the immediate takes none"}, + {"rounding on the compare's memory operand", "VCOMISH", + []ExtOperand{ExtOperand{Kind: ExtMem, Reg: 9, Round: ExtRoundSAE}, ExtXmm(30)}, + "the memory operand takes none"}, + {"rounding on an entry without the capability", "VMOVSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtRounded(ExtXmm(30), ExtRoundNearest)}, + "the entry's destination takes none"}, } { in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) _, err := in.Encode(tt.ops) diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index 91d1b33..1ae6e8d 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -229,6 +229,12 @@ type ExtOperand struct { BaseVec bool // the first parenthesis spells a Z register Extend uint8 // 0 = none, 1 = UXTW, 2 = SXTW Off int // the second parenthesis's register, -1 when absent + // Round spells the rounding decoration of an amd64 EVEX destination, the + // {sae} and {rn-sae} through {rz-sae} spellings: EVEX.b selects the + // rounding context and EVEX.RC, the L'L bits it replaces, the mode. Only + // the destinations of the entries that carry Er or Sae accept it. The + // arm64 entries all carry the zero value. + Round ExtRounding } // ExtVector builds a scalable vector operand, ADD Z1.S style. @@ -967,6 +973,17 @@ type ExtInstr struct { // opmask-destination forms and the load and store shapes do not. The // arm64 entries all carry the zero value. Mask bool + // Er records that the entry's destination takes the embedded rounding + // decoration, the {rn-sae} through {rz-sae} spellings the FP arithmetic + // and the rounding conversions carry in their register forms: EVEX.b + // selects the rounding context, EVEX.RC names the mode, and the memory + // forms take none. The arm64 entries all carry the zero value. + Er bool + // Sae records that the entry's destination takes the exception-suppression + // decoration alone, the {sae} spelling the minima, maxima, compares and + // exponent extracts carry: no mode is named and EVEX.RC stays zero. The + // arm64 entries all carry the zero value. + Sae bool // PgQual names the qualifier the row's governing predicate requires, // where the form takes one: merging, zeroing or either (the MOVPRFX // classes spell both with one encoding). Forms without such a diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index 94abd98..57340be 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -125,6 +125,12 @@ func TestEncodeExtensionAmd64(t *testing.T) { {"mantissa extract with a control byte", "VGETMANTSH", []arch.ExtOperand{arch.ExtImmediate(0x0b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}, "6203140027f40b"}, + {"packed fp16 add under embedded rounding", "VADDPH", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundTruncate)}, + "62f5547858f4"}, + {"scalar fp16 minimum with exceptions suppressed", "VMINSH", + []arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundSAE)}, + "62f556185df4"}, } { got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...) if err != nil {