diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go index 002d589..456c778 100644 --- a/arch/amd64_ext.go +++ b/arch/amd64_ext.go @@ -8,13 +8,14 @@ // arch/amd64_gen.go stays untouched, and asm.Encodable keeps answering false // for every mnemonic here, so the layer stays out of the main encoders. // -// The first families are AVX512-BF16 and AVX512-VP2INTERSECT, in their EVEX -// register forms. The encodings are transcribed from the SDM instruction -// entries and cross-checked against binutils-gdb's assembler testsuite; the -// golden vectors in amd64_ext_test.go pin the bytes. VPOPCNTD and VPOPCNTQ, -// the third family of the 2026-09-19 survey, no longer belong here: the Go -// toolchain's assembler knows them today, they live in the generated table -// and the EVEX encoder, and a mnemonic the toolchain has is not an extension. +// The families are AVX512-BF16, AVX512-VP2INTERSECT and the scalar core of +// AVX512-FP16, in their EVEX register forms. The encodings are transcribed +// from the SDM instruction entries and cross-checked against binutils-gdb's +// assembler testsuite; the golden vectors in amd64_ext_test.go pin the +// bytes. VPOPCNTD and VPOPCNTQ, the third family of the 2026-09-19 survey, +// no longer belong here: the Go toolchain's assembler knows them today, they +// live in the generated table and the EVEX encoder, and a mnemonic the +// toolchain has is not an extension. // // Memory operands, write masking ({k1}{z}) and embedded rounding arrive with // a later slice; every form here encodes the unmasked register forms, which @@ -325,4 +326,84 @@ var amd64Extensions = []ExtInstr{ {Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors", Bytes: []byte{0x62, 0x02, 0x87, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT, Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W1 68 /r)"}, + + // AVX512-FP16, the scalar core: move, arithmetic, compare and + // conversion on one half-precision value in the low XMM lane, the + // register forms of the manual's scalar entries. The family lives in + // the EVEX maps five and six the toolchain has never emitted, with the + // mandatory prefixes the manual gives each entry; the golden vectors + // pin every prefix byte for byte. LIG encodes as L'L = 00, the XMM + // class alone. + {Name: "VMOVSH", Summary: "Move a scalar FP16 value", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMOVSH (EVEX.NDS.LIG.F3.MAP5.W0 10 /r)"}, + {Name: "VMOVW", Summary: "Move a word between a general register and an XMM register", + Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x6E, 0xC0}, Form: ExtFormAmdGprVec, Feature: ExtFeatureFP16, Wig: true, + Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 6E /r)"}, + {Name: "VMOVW", Summary: "Move a word between an XMM register and a general register", + Bytes: []byte{0x62, 0x05, 0x05, 0x00, 0x7E, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, Wig: true, + Ref: "Intel SDM Vol. 2C, VMOVW (EVEX.128.66.MAP5.WIG 7E /r)"}, + {Name: "VADDSH", Summary: "Add scalar FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x58, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VADDSH (EVEX.NDS.LIG.F3.MAP5.W0 58 /r)"}, + {Name: "VSUBSH", Summary: "Subtract scalar FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5C, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VSUBSH (EVEX.NDS.LIG.F3.MAP5.W0 5C /r)"}, + {Name: "VMULSH", Summary: "Multiply scalar FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x59, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMULSH (EVEX.NDS.LIG.F3.MAP5.W0 59 /r)"}, + {Name: "VDIVSH", Summary: "Divide scalar FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5E, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VDIVSH (EVEX.NDS.LIG.F3.MAP5.W0 5E /r)"}, + {Name: "VMINSH", Summary: "Return the minimum of scalar FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMINSH (EVEX.NDS.LIG.F3.MAP5.W0 5D /r)"}, + {Name: "VMAXSH", Summary: "Return the maximum of scalar FP16 values", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5F, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VMAXSH (EVEX.NDS.LIG.F3.MAP5.W0 5F /r)"}, + {Name: "VSQRTSH", Summary: "Compute the square root of a scalar FP16 value", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x51, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VSQRTSH (EVEX.NDS.LIG.F3.MAP5.W0 51 /r)"}, + {Name: "VCOMISH", Summary: "Compare a scalar FP16 value and set EFLAGS", + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2F, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCOMISH (EVEX.LIG.MAP5.W0 2F /r)"}, + {Name: "VUCOMISH", Summary: "Unordered-compare a scalar FP16 value and set EFLAGS", + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x2E, 0xC0}, Form: ExtFormAmdVec2, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VUCOMISH (EVEX.LIG.MAP5.W0 2E /r)"}, + {Name: "VCVTSS2SH", Summary: "Convert one FP32 value to one FP16 value", + Bytes: []byte{0x62, 0x05, 0x04, 0x00, 0x1D, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSS2SH (EVEX.NDS.LIG.MAP5.W0 1D /r)"}, + {Name: "VCVTSH2SS", Summary: "Convert a low FP16 value to an FP32 value", + Bytes: []byte{0x62, 0x06, 0x04, 0x00, 0x13, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSH2SS (EVEX.NDS.LIG.MAP6.W0 13 /r)"}, + {Name: "VCVTSH2SD", Summary: "Convert a low FP16 value to an FP64 value", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSH2SD (EVEX.NDS.LIG.F3.MAP5.W0 5A /r)"}, + {Name: "VCVTSD2SH", Summary: "Convert one FP64 value to one FP16 value", + Bytes: []byte{0x62, 0x05, 0x87, 0x00, 0x5A, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSD2SH (EVEX.NDS.LIG.F2.MAP5.W1 5A /r)"}, + {Name: "VCVTSI2SH", Summary: "Convert one signed 32-bit integer to one FP16 value", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 2A /r)"}, + {Name: "VCVTSI2SH", Summary: "Convert one signed 64-bit integer to one FP16 value", + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2A, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 2A /r)"}, + {Name: "VCVTUSI2SH", Summary: "Convert one unsigned 32-bit integer to one FP16 value", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W0 7B /r)"}, + {Name: "VCVTUSI2SH", Summary: "Convert one unsigned 64-bit integer to one FP16 value", + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x7B, 0xC0}, Form: ExtFormAmdVecGprVec, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTUSI2SH (EVEX.NDS.LIG.F3.MAP5.W1 7B /r)"}, + {Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 32-bit integer", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W0 2D /r)"}, + {Name: "VCVTSH2SI", Summary: "Convert a low FP16 value to a signed 64-bit integer", + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x2D, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSH2SI (EVEX.LIG.F3.MAP5.W1 2D /r)"}, + {Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 32-bit integer", + Bytes: []byte{0x62, 0x05, 0x06, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W0 79 /r)"}, + {Name: "VCVTSH2USI", Summary: "Convert a low FP16 value to an unsigned 64-bit integer", + Bytes: []byte{0x62, 0x05, 0x86, 0x00, 0x79, 0xC0}, Form: ExtFormAmdVecGpr, Feature: ExtFeatureFP16, + Ref: "Intel SDM Vol. 2C, VCVTSH2USI (EVEX.LIG.F3.MAP5.W1 79 /r)"}, } diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go index 22cca42..5ba4578 100644 --- a/arch/amd64_ext_test.go +++ b/arch/amd64_ext_test.go @@ -9,18 +9,209 @@ import ( "testing" ) -// The BF16 and VP2INTERSECT encodings have no toolchain oracle: go tool asm -// knows neither family. The golden words below are transcribed from the -// Intel SDM instruction entries and cross-checked against binutils-gdb's own -// assembler testsuite: every row marked "GNU" matches a vector in -// gas/testsuite/gas/i386/avx512_bf16.d, avx512_bf16_vl.d or -// x86-64-vp2intersect.d byte for byte, so no entry rests on transcription -// alone. The GNU dumps print AT&T order (sources first, destination last), -// which is the order the operands are built in here too. -func amd64ExtInstr(t *testing.T, mnem string, class ExtOperandKind) ExtInstr { +// The BF16, VP2INTERSECT and FP16 encodings have no toolchain oracle: go +// tool asm knows none of these families. The golden words below are +// transcribed from the Intel SDM instruction entries and cross-checked +// against binutils-gdb's own assembler testsuite: every row marked "GNU" +// matches a vector in gas/testsuite/gas/i386/avx512_bf16.d, +// avx512_bf16_vl.d, x86-64-vp2intersect.d or x86-64-avx512_fp16.d byte for +// byte, so no entry rests on transcription alone. The two VMOVW rows are +// class vectors: the GNU file proves the 66.MAP5 opcode row on the m16 +// memory forms, and the register form follows the manual's ModR/M reg row. +// The GNU dumps print AT&T order (sources first, destination last), which is +// the order the operands are built in here too. +type amd64GoldenRow struct { + name string + mnem string + ops []ExtOperand + want string // hex, little-endian bytes in memory order + GNU string // the matching binutils-gdb line, empty for a derived register form +} + +var amd64GoldenRows = []amd64GoldenRow{ + // AVX512-BF16, EVEX.NDS.F2.0F38.W0 for the three-register convert, + // EVEX.F3.0F38.W0 for the narrow convert and the dot product. + {"vcvtne2ps2bf16 zmm", "VCVTNE2PS2BF16", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)}, + "62f2574872f4", "62 f2 57 48 72 f4 vcvtne2ps2bf16 %zmm4,%zmm5,%zmm6"}, + {"vcvtne2ps2bf16 ymm", "VCVTNE2PS2BF16", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f2572872f4", "62 f2 57 28 72 f4 vcvtne2ps2bf16 %ymm4,%ymm5,%ymm6"}, + {"vcvtne2ps2bf16 xmm", "VCVTNE2PS2BF16", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f2570872f4", "62 f2 57 08 72 f4 vcvtne2ps2bf16 %xmm4,%xmm5,%xmm6"}, + {"vcvtneps2bf16 zmm to ymm", "VCVTNEPS2BF16", + []ExtOperand{ExtZmm(5), ExtYmm(6)}, + "62f27e4872f5", "62 f2 7e 48 72 f5 vcvtneps2bf16 %zmm5,%ymm6"}, + {"vcvtneps2bf16 ymm to xmm", "VCVTNEPS2BF16", + []ExtOperand{ExtYmm(5), ExtXmm(6)}, + "62f27e2872f5", "62 f2 7e 28 72 f5 vcvtneps2bf16 %ymm5,%xmm6"}, + {"vcvtneps2bf16 xmm to xmm", "VCVTNEPS2BF16", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f27e0872f5", "62 f2 7e 08 72 f5 vcvtneps2bf16 %xmm5,%xmm6"}, + {"vdpbf16ps zmm", "VDPBF16PS", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)}, + "62f2564852f4", "62 f2 56 48 52 f4 vdpbf16ps %zmm4,%zmm5,%zmm6"}, + {"vdpbf16ps ymm", "VDPBF16PS", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f2562852f4", "62 f2 56 28 52 f4 vdpbf16ps %ymm4,%ymm5,%ymm6"}, + {"vdpbf16ps xmm", "VDPBF16PS", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f2560852f4", "62 f2 56 08 52 f4 vdpbf16ps %xmm4,%xmm5,%xmm6"}, + + // AVX512-VP2INTERSECT, EVEX.NDS.F2.0F38. The mask destination is the + // ModR/M reg field, so a k register above k7 must refuse. + {"vp2intersectd zmm k0", "VP2INTERSECTD", + []ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)}, + "62f26f4868c1", "62 f2 6f 48 68 c1 vp2intersectd %zmm1,%zmm2,%k0"}, + {"vp2intersectd ymm k2", "VP2INTERSECTD", + []ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)}, + "62f26f2868d1", "62 f2 6f 28 68 d1 vp2intersectd %ymm1,%ymm2,%k2"}, + {"vp2intersectd xmm k4", "VP2INTERSECTD", + []ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)}, + "62f26f0868e1", "62 f2 6f 08 68 e1 vp2intersectd %xmm1,%xmm2,%k4"}, + {"vp2intersectq zmm k0", "VP2INTERSECTQ", + []ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)}, + "62f2ef4868c1", "62 f2 ef 48 68 c1 vp2intersectq %zmm1,%zmm2,%k0"}, + {"vp2intersectq ymm k2", "VP2INTERSECTQ", + []ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)}, + "62f2ef2868d1", "62 f2 ef 28 68 d1 vp2intersectq %ymm1,%ymm2,%k2"}, + {"vp2intersectq xmm k4", "VP2INTERSECTQ", + []ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)}, + "62f2ef0868e1", "62 f2 ef 08 68 e1 vp2intersectq %xmm1,%xmm2,%k4"}, + + // AVX512-FP16 scalar arithmetic, EVEX.NDS.LIG.F3.MAP5.W0. Every + // register in the GNU vector sits above 15, so the row exercises all + // four EVEX extension bits at once. + {"vmovsh", "VMOVSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6205160010f4", "62 05 16 00 10 f4 vmovsh %xmm28,%xmm29,%xmm30"}, + {"vaddsh", "VADDSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6205160058f4", "62 05 16 00 58 f4 vaddsh %xmm28,%xmm29,%xmm30"}, + {"vsubsh", "VSUBSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620516005cf4", "62 05 16 00 5c f4 vsubsh %xmm28,%xmm29,%xmm30"}, + {"vmulsh", "VMULSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6205160059f4", "62 05 16 00 59 f4 vmulsh %xmm28,%xmm29,%xmm30"}, + {"vdivsh", "VDIVSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620516005ef4", "62 05 16 00 5e f4 vdivsh %xmm28,%xmm29,%xmm30"}, + {"vminsh", "VMINSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620516005df4", "62 05 16 00 5d f4 vminsh %xmm28,%xmm29,%xmm30"}, + {"vmaxsh", "VMAXSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620516005ff4", "62 05 16 00 5f f4 vmaxsh %xmm28,%xmm29,%xmm30"}, + {"vsqrtsh", "VSQRTSH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6205160051f4", "62 05 16 00 51 f4 vsqrtsh %xmm28,%xmm29,%xmm30"}, + + // The scalar compares take two operands, EVEX.LIG.MAP5.W0. + {"vcomish", "VCOMISH", + []ExtOperand{ExtXmm(29), ExtXmm(30)}, + "62057c082ff5", "62 05 7c 08 2f f5 vcomish %xmm29,%xmm30"}, + {"vucomish", "VUCOMISH", + []ExtOperand{ExtXmm(29), ExtXmm(30)}, + "62057c082ef5", "62 05 7c 08 2e f5 vucomish %xmm29,%xmm30"}, + + // The floating-point conversions between the three scalar widths. The + // single-precision convert carries no prefix, the half-to-double convert + // carries F3, the double-to-half convert F2 and W1. + {"vcvtss2sh", "VCVTSS2SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620514001df4", "62 05 14 00 1d f4 vcvtss2sh %xmm28,%xmm29,%xmm30"}, + {"vcvtsh2ss", "VCVTSH2SS", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "6206140013f4", "62 06 14 00 13 f4 vcvtsh2ss %xmm28,%xmm29,%xmm30"}, + {"vcvtsh2sd", "VCVTSH2SD", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620516005af4", "62 05 16 00 5a f4 vcvtsh2sd %xmm28,%xmm29,%xmm30"}, + {"vcvtsd2sh", "VCVTSD2SH", + []ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)}, + "620597005af4", "62 05 97 00 5a f4 vcvtsd2sh %xmm28,%xmm29,%xmm30"}, + + // The integer conversions, one entry per W bit: the W bit picks the + // 32-bit or the 64-bit general register. + {"vcvtsi2sh edx", "VCVTSI2SH", + []ExtOperand{ExtXmm(29), ExtGpr32(2), ExtXmm(30)}, + "626516002af2", "62 65 16 00 2a f2 vcvtsi2sh %edx,%xmm29,%xmm30"}, + {"vcvtsi2sh r12", "VCVTSI2SH", + []ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)}, + "624596002af4", "62 45 96 00 2a f4 vcvtsi2sh %r12,%xmm29,%xmm30"}, + {"vcvtusi2sh edx", "VCVTUSI2SH", + []ExtOperand{ExtXmm(29), ExtGpr32(2), ExtXmm(30)}, + "626516007bf2", "62 65 16 00 7b f2 vcvtusi2sh %edx,%xmm29,%xmm30"}, + {"vcvtusi2sh r12", "VCVTUSI2SH", + []ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)}, + "624596007bf4", "62 45 96 00 7b f4 vcvtusi2sh %r12,%xmm29,%xmm30"}, + {"vcvtsh2si edx", "VCVTSH2SI", + []ExtOperand{ExtXmm(30), ExtGpr32(2)}, + "62957e082dd6", "62 95 7e 08 2d d6 vcvtsh2si %xmm30,%edx"}, + {"vcvtsh2si r12", "VCVTSH2SI", + []ExtOperand{ExtXmm(30), ExtGpr64(12)}, + "6215fe082de6", "62 15 fe 08 2d e6 vcvtsh2si %xmm30,%r12"}, + {"vcvtsh2usi edx", "VCVTSH2USI", + []ExtOperand{ExtXmm(30), ExtGpr32(2)}, + "62957e0879d6", "62 95 7e 08 79 d6 vcvtsh2usi %xmm30,%edx"}, + {"vcvtsh2usi r12", "VCVTSH2USI", + []ExtOperand{ExtXmm(30), ExtGpr64(12)}, + "6215fe0879e6", "62 15 fe 08 79 e6 vcvtsh2usi %xmm30,%r12"}, + + // VMOVW in both directions: the register forms are class vectors, the + // GNU file proves the opcode rows on the m16 memory forms. + {"vmovw into xmm", "VMOVW", + []ExtOperand{ExtGpr32(12), ExtXmm(30)}, + "62457d086ef4", ""}, + {"vmovw out of xmm", "VMOVW", + []ExtOperand{ExtXmm(30), ExtGpr32(12)}, + "62157d087ee6", ""}, + + // High registers in a 512-bit form exercise the EVEX extension bits: + // with both sources above 15 the B bar and X bar bits clear, while the + // destination zmm23 keeps R bar set in byte one (derived from the + // proven class above). + {"vcvtne2ps2bf16 high registers", "VCVTNE2PS2BF16", + []ExtOperand{ExtZmm(21), ExtZmm(20), ExtZmm(23)}, + "62a2574072fc", ""}, +} + +// amd64ResolveEntry finds the table entry a golden row exercises: the entry +// is the one that accepts the row's operands, which is what pins the bytes to +// a single template when a mnemonic registers one entry per W bit. +func amd64ResolveEntry(mnem string, ops []ExtOperand) (ExtInstr, bool) { + var first ExtInstr + for _, in := range Extensions(AMD64) { + if in.Name != mnem { + continue + } + if _, err := in.Encode(ops); err == nil { + return in, true + } + if first.Name == "" { + first = in + } + } + if first.Name != "" { + return first, true + } + return ExtInstr{}, false +} + +func amd64ExtInstr(t *testing.T, mnem string, class ExtOperandKind, preds ...func(ExtInstr) bool) ExtInstr { t.Helper() for _, in := range Extensions(AMD64) { - if in.Name == mnem && amd64LengthClass(in.Bytes) == class { + if in.Name != mnem || amd64LengthClass(in.Bytes) != class { + continue + } + match := true + for _, p := range preds { + if !p(in) { + match = false + } + } + if match { return in } } @@ -28,73 +219,24 @@ func amd64ExtInstr(t *testing.T, mnem string, class ExtOperandKind) ExtInstr { return ExtInstr{} } +// amd64W1 names the W1 encoding of a mnemonic registered once per W bit. +func amd64W1(in ExtInstr) bool { return in.Bytes[2]&0x80 != 0 } + +// withPred adapts an optional row predicate for the variadic lookup. +func withPred(p func(ExtInstr) bool) []func(ExtInstr) bool { + if p == nil { + return nil + } + return []func(ExtInstr) bool{p} +} + func TestAmd64ExtGoldenBytes(t *testing.T) { - for _, tt := range []struct { - name string - mnem string - ops []ExtOperand - want string // hex, little-endian bytes in memory order - GNU string // the matching binutils-gdb line, empty for a derived register form - }{ - // AVX512-BF16, EVEX.NDS.F2.0F38.W0. - {"vcvtne2ps2bf16 zmm", "VCVTNE2PS2BF16", - []ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)}, - "62f2574872f4", "62 f2 57 48 72 f4 vcvtne2ps2bf16 %zmm4,%zmm5,%zmm6"}, - {"vcvtne2ps2bf16 ymm", "VCVTNE2PS2BF16", - []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, - "62f2572872f4", "62 f2 57 28 72 f4 vcvtne2ps2bf16 %ymm4,%ymm5,%ymm6"}, - {"vcvtne2ps2bf16 xmm", "VCVTNE2PS2BF16", - []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, - "62f2570872f4", "62 f2 57 08 72 f4 vcvtne2ps2bf16 %xmm4,%xmm5,%xmm6"}, - {"vcvtneps2bf16 zmm to ymm", "VCVTNEPS2BF16", - []ExtOperand{ExtZmm(5), ExtYmm(6)}, - "62f27e4872f5", "62 f2 7e 48 72 f5 vcvtneps2bf16 %zmm5,%ymm6"}, - {"vcvtneps2bf16 ymm to xmm", "VCVTNEPS2BF16", - []ExtOperand{ExtYmm(5), ExtXmm(6)}, - "62f27e2872f5", "62 f2 7e 28 72 f5 vcvtneps2bf16 %ymm5,%xmm6"}, - {"vcvtneps2bf16 xmm to xmm", "VCVTNEPS2BF16", - []ExtOperand{ExtXmm(5), ExtXmm(6)}, - "62f27e0872f5", "62 f2 7e 08 72 f5 vcvtneps2bf16 %xmm5,%xmm6"}, - {"vdpbf16ps zmm", "VDPBF16PS", - []ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)}, - "62f2564852f4", "62 f2 56 48 52 f4 vdpbf16ps %zmm4,%zmm5,%zmm6"}, - {"vdpbf16ps ymm", "VDPBF16PS", - []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, - "62f2562852f4", "62 f2 56 28 52 f4 vdpbf16ps %ymm4,%ymm5,%ymm6"}, - {"vdpbf16ps xmm", "VDPBF16PS", - []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, - "62f2560852f4", "62 f2 56 08 52 f4 vdpbf16ps %xmm4,%xmm5,%xmm6"}, - - // AVX512-VP2INTERSECT, EVEX.NDS.F2.0F38. The mask destination is - // the ModR/M reg field, so a k register above k7 must refuse. - {"vp2intersectd zmm k0", "VP2INTERSECTD", - []ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)}, - "62f26f4868c1", "62 f2 6f 48 68 c1 vp2intersectd %zmm1,%zmm2,%k0"}, - {"vp2intersectd ymm k2", "VP2INTERSECTD", - []ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)}, - "62f26f2868d1", "62 f2 6f 28 68 d1 vp2intersectd %ymm1,%ymm2,%k2"}, - {"vp2intersectd xmm k4", "VP2INTERSECTD", - []ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)}, - "62f26f0868e1", "62 f2 6f 08 68 e1 vp2intersectd %xmm1,%xmm2,%k4"}, - {"vp2intersectq zmm k0", "VP2INTERSECTQ", - []ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)}, - "62f2ef4868c1", "62 f2 ef 48 68 c1 vp2intersectq %zmm1,%zmm2,%k0"}, - {"vp2intersectq ymm k2", "VP2INTERSECTQ", - []ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)}, - "62f2ef2868d1", "62 f2 ef 28 68 d1 vp2intersectq %ymm1,%ymm2,%k2"}, - {"vp2intersectq xmm k4", "VP2INTERSECTQ", - []ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)}, - "62f2ef0868e1", "62 f2 ef 08 68 e1 vp2intersectq %xmm1,%xmm2,%k4"}, - - // High registers exercise the EVEX extension bits: with both source - // registers above 15 the B bar and X bar bits clear, while the - // destination zmm23 keeps R bar set in byte one (derived from the - // proven class above). - {"vcvtne2ps2bf16 high registers", "VCVTNE2PS2BF16", - []ExtOperand{ExtZmm(21), ExtZmm(20), ExtZmm(23)}, - "62a2574072fc", ""}, - } { - in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) + for _, tt := range amd64GoldenRows { + in, ok := amd64ResolveEntry(tt.mnem, tt.ops) + if !ok { + t.Errorf("%s: no table entry for %s at the row's vector length", tt.name, tt.mnem) + continue + } got, err := in.Encode(tt.ops) if err != nil { t.Errorf("%s: encode: %v", tt.name, err) @@ -106,21 +248,6 @@ func TestAmd64ExtGoldenBytes(t *testing.T) { } } -// operandClass names the vector class a golden row exercises, the key the -// helper resolves the table entry with. A row without a vector operand -// (none today) would have nowhere to go. -func operandClass(t *testing.T, ops []ExtOperand) ExtOperandKind { - t.Helper() - for _, op := range ops { - switch op.Kind { - case ExtXMM, ExtYMM, ExtZMM: - return op.Kind - } - } - t.Fatal("the golden row carries no vector operand to pick the entry with") - return ExtXMM -} - // TestAmd64ExtTemplateIntegrity checks the metadata contract: every entry // names its manual reference, summary and feature, and every template carries // the fixed shape of an EVEX register form with the register-derived bits @@ -129,6 +256,7 @@ func TestAmd64ExtTemplateIntegrity(t *testing.T) { features := map[ExtFeature]bool{ ExtFeatureBF16: true, ExtFeatureVP2INTERSECT: true, + ExtFeatureFP16: true, } for _, in := range Extensions(AMD64) { if in.Name == "" || in.Summary == "" || in.Ref == "" { @@ -167,32 +295,19 @@ func TestAmd64ExtTemplateIntegrity(t *testing.T) { // TestAmd64ExtEveryEntryCarriesGoldenVector pins the measure the layer is // judged by: every registered entry is covered by at least one golden vector -// in the byte test above, so an entry without provenance cannot hide. +// whose resolved entry has the very template, so an entry without provenance +// cannot hide. func TestAmd64ExtEveryEntryCarriesGoldenVector(t *testing.T) { - covered := map[string]bool{} for _, in := range Extensions(AMD64) { - covered[in.Name+"|"+string(amd64LengthClass(in.Bytes))] = false - } - for _, tt := range []struct { - mnem string - class ExtOperandKind - }{ - {"VCVTNE2PS2BF16", ExtZMM}, {"VCVTNE2PS2BF16", ExtYMM}, {"VCVTNE2PS2BF16", ExtXMM}, - {"VCVTNEPS2BF16", ExtZMM}, {"VCVTNEPS2BF16", ExtYMM}, {"VCVTNEPS2BF16", ExtXMM}, - {"VDPBF16PS", ExtZMM}, {"VDPBF16PS", ExtYMM}, {"VDPBF16PS", ExtXMM}, - {"VP2INTERSECTD", ExtZMM}, {"VP2INTERSECTD", ExtYMM}, {"VP2INTERSECTD", ExtXMM}, - {"VP2INTERSECTQ", ExtZMM}, {"VP2INTERSECTQ", ExtYMM}, {"VP2INTERSECTQ", ExtXMM}, - } { - key := tt.mnem + "|" + string(tt.class) - if _, ok := covered[key]; !ok { - t.Errorf("the golden list covers %s, but the table registers no such entry", key) - continue + found := false + for _, tt := range amd64GoldenRows { + cand, ok := amd64ResolveEntry(tt.mnem, tt.ops) + if ok && cand.Name == in.Name && string(cand.Bytes) == string(in.Bytes) { + found = true + } } - covered[key] = true - } - for key, ok := range covered { - if !ok { - t.Errorf("%s has no golden vector", key) + if !found { + t.Errorf("%s (% x) has no golden vector", in.Name, in.Bytes) } } } @@ -214,13 +329,19 @@ func TestAmd64ExtRejects(t *testing.T) { []ExtOperand{ExtZmm(1), ExtZmm(2), ExtZmm(3)}, "wants an opmask register"}, {"mask register beyond k7", "VP2INTERSECTD", - []ExtOperand{ExtZmm(1), ExtZmm(2), ExtMask(8)}, + []ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(8)}, "outside 0-7"}, - {"vector where the general register belongs", "VCVTNE2PS2BF16", - []ExtOperand{ExtGpr32(0), ExtZmm(2), ExtZmm(3)}, - "wants a ZMM register"}, + {"vector where the general register belongs", "VCVTSH2SI", + []ExtOperand{ExtXmm(1), ExtXmm(2)}, + "wants a 32-bit general register"}, + {"64-bit register on the W0 convert", "VCVTSI2SH", + []ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)}, + "wants a 32-bit general register"}, + {"general register beyond r15", "VMOVW", + []ExtOperand{ExtGpr32(16), ExtXmm(30)}, + "outside 0-15"}, {"wrong arity", "VP2INTERSECTD", - []ExtOperand{ExtZmm(1), ExtZmm(2)}, + []ExtOperand{ExtZmm(2), ExtZmm(1)}, "takes 3 operands"}, {"arm64 arrangement suffix", "VCVTNE2PS2BF16", []ExtOperand{{Kind: ExtZMM, Reg: 1, Arr: ExtArrS}, ExtZmm(2), ExtZmm(3)}, @@ -239,4 +360,39 @@ func TestAmd64ExtRejects(t *testing.T) { t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote) } } + // The W1 convert refuses the 32-bit register the W0 entry takes. + in := amd64ExtInstr(t, "VCVTSH2SI", ExtXMM, amd64W1) + if _, err := in.Encode([]ExtOperand{ExtXmm(30), ExtGpr32(2)}); err == nil { + t.Error("a 32-bit register encoded on the W1 convert, want an error") + } else if !strings.Contains(err.Error(), "wants a 64-bit general register") { + t.Errorf("the W1 error %q does not name the 64-bit class", err) + } +} + +// operandClass names the vector class a row exercises, the key the entry +// lookup resolves with. +func operandClass(t *testing.T, ops []ExtOperand) ExtOperandKind { + t.Helper() + for _, op := range ops { + switch op.Kind { + case ExtXMM, ExtYMM, ExtZMM: + return op.Kind + } + } + t.Fatal("the row carries no vector operand to pick the entry with") + return ExtXMM +} + +// TestAmd64ExtArchBinding pins the layer's architecture binding: only riscv +// and loong64 have no extended layer, arm64's lives in arm64_ext.go and the +// amd64 one here. +func TestAmd64ExtArchBinding(t *testing.T) { + for _, a := range []Arch{RISCV, LOONG64, Unknown} { + if got := Extensions(a); len(got) != 0 { + t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) + } + } + if got := Extensions(AMD64); len(got) != 39 { + t.Errorf("the amd64 layer registers %d instructions, want 39", len(got)) + } } diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index 262ea3d..b64a5c2 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -266,11 +266,11 @@ const ( // dest. ExtFormAmdVecGprVec // ExtFormAmdGprVec is the two-operand form with a general-register - // source and a vector destination: VMOVW X1, EAX and VCVTSH2SI EAX, X1. - // Operands: gpr, dest. + // source and a vector destination: VMOVW X1, EAX. Operands: gpr, dest. ExtFormAmdGprVec // ExtFormAmdVecGpr is the two-operand form with a vector source and a - // general-register destination: VMOVW EAX, X1. Operands: src, dest. + // general-register destination: VMOVW EAX, X1 and VCVTSH2SI EAX, X1. + // Operands: src, dest. ExtFormAmdVecGpr ) diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go index e2ba5df..33515fd 100644 --- a/asm/extension_amd64_test.go +++ b/asm/extension_amd64_test.go @@ -12,8 +12,9 @@ import ( ) // TestAmd64ExtensionRegistry checks the mnemonic lookup for the amd64 layer: -// one mnemonic across several vector lengths resolves to every entry, the -// lookup is case-insensitive, and the counts match the registered families. +// one mnemonic across several vector lengths or W bits resolves to every +// entry, the lookup is case-insensitive, and the counts match the registered +// families. func TestAmd64ExtensionRegistry(t *testing.T) { for _, tt := range []struct { mnem string @@ -24,6 +25,14 @@ func TestAmd64ExtensionRegistry(t *testing.T) { {"VDPBF16PS", 3}, {"VP2INTERSECTD", 3}, {"VP2INTERSECTQ", 3}, + {"VMOVSH", 1}, + {"VMOVW", 2}, + {"VADDSH", 1}, + {"VSQRTSH", 1}, + {"VCOMISH", 1}, + {"VCVTSH2SS", 1}, + {"VCVTSI2SH", 2}, + {"VCVTSH2SI", 2}, } { cands, ok := LookupExtension(arch.AMD64, tt.mnem) if !ok { @@ -37,8 +46,8 @@ func TestAmd64ExtensionRegistry(t *testing.T) { t.Errorf("the %s lookup is not case-insensitive", tt.mnem) } } - if got := arch.Extensions(arch.AMD64); len(got) != 15 { - t.Errorf("the amd64 layer registers %d instructions, want 15", len(got)) + if got := arch.Extensions(arch.AMD64); len(got) != 39 { + t.Errorf("the amd64 layer registers %d instructions, want 39", len(got)) } if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { t.Error("a non-extended mnemonic resolved") @@ -87,6 +96,21 @@ func TestEncodeExtensionAmd64(t *testing.T) { {"intersect into a mask", "VP2INTERSECTD", []arch.ExtOperand{arch.ExtYmm(2), arch.ExtYmm(1), arch.ExtMask(2)}, "62f26f2868d1"}, + {"scalar fp16 add, high registers", "VADDSH", + []arch.ExtOperand{arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}, + "6205160058f4"}, + {"scalar compare", "VCOMISH", + []arch.ExtOperand{arch.ExtXmm(29), arch.ExtXmm(30)}, + "62057c082ff5"}, + {"integer into a scalar fp16", "VCVTSI2SH", + []arch.ExtOperand{arch.ExtXmm(29), arch.ExtGpr64(12), arch.ExtXmm(30)}, + "624596002af4"}, + {"scalar fp16 into an integer", "VCVTSH2SI", + []arch.ExtOperand{arch.ExtXmm(30), arch.ExtGpr32(2)}, + "62957e082dd6"}, + {"word move into an xmm", "VMOVW", + []arch.ExtOperand{arch.ExtGpr64(12), arch.ExtXmm(30)}, + "62457d086ef4"}, } { got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...) if err != nil {