698 lines
29 KiB
Go
698 lines
29 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package arch
|
|
|
|
import (
|
|
"encoding/hex"
|
|
"strings"
|
|
"testing"
|
|
)
|
|
|
|
// The BF16, VP2INTERSECT and FP16 encodings have no toolchain oracle: go
|
|
// tool asm knows none of these families. The golden words below are
|
|
// transcribed from the Intel SDM instruction entries and cross-checked
|
|
// against binutils-gdb's own assembler testsuite: every row marked "GNU"
|
|
// matches a vector in gas/testsuite/gas/i386/avx512_bf16.d,
|
|
// avx512_bf16_vl.d, x86-64-vp2intersect.d, x86-64-avx512_fp16.d or
|
|
// avx512_fp16_vl.d byte for byte, so no entry rests on transcription alone. The two VMOVW rows are
|
|
// class vectors: the GNU file proves the 66.MAP5 opcode row on the m16
|
|
// memory forms, and the register form follows the manual's ModR/M reg row.
|
|
// The GNU dumps print AT&T order (sources first, destination last), which is
|
|
// the order the operands are built in here too.
|
|
type amd64GoldenRow struct {
|
|
name string
|
|
mnem string
|
|
ops []ExtOperand
|
|
want string // hex, little-endian bytes in memory order
|
|
GNU string // the matching binutils-gdb line, empty for a derived register form
|
|
}
|
|
|
|
var amd64GoldenRows = []amd64GoldenRow{
|
|
// AVX512-BF16, EVEX.NDS.F2.0F38.W0 for the three-register convert,
|
|
// EVEX.F3.0F38.W0 for the narrow convert and the dot product.
|
|
{"vcvtne2ps2bf16 zmm", "VCVTNE2PS2BF16",
|
|
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)},
|
|
"62f2574872f4", "62 f2 57 48 72 f4 vcvtne2ps2bf16 %zmm4,%zmm5,%zmm6"},
|
|
{"vcvtne2ps2bf16 ymm", "VCVTNE2PS2BF16",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
|
"62f2572872f4", "62 f2 57 28 72 f4 vcvtne2ps2bf16 %ymm4,%ymm5,%ymm6"},
|
|
{"vcvtne2ps2bf16 xmm", "VCVTNE2PS2BF16",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
|
"62f2570872f4", "62 f2 57 08 72 f4 vcvtne2ps2bf16 %xmm4,%xmm5,%xmm6"},
|
|
{"vcvtneps2bf16 zmm to ymm", "VCVTNEPS2BF16",
|
|
[]ExtOperand{ExtZmm(5), ExtYmm(6)},
|
|
"62f27e4872f5", "62 f2 7e 48 72 f5 vcvtneps2bf16 %zmm5,%ymm6"},
|
|
{"vcvtneps2bf16 ymm to xmm", "VCVTNEPS2BF16",
|
|
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
|
|
"62f27e2872f5", "62 f2 7e 28 72 f5 vcvtneps2bf16 %ymm5,%xmm6"},
|
|
{"vcvtneps2bf16 xmm to xmm", "VCVTNEPS2BF16",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
|
"62f27e0872f5", "62 f2 7e 08 72 f5 vcvtneps2bf16 %xmm5,%xmm6"},
|
|
{"vdpbf16ps zmm", "VDPBF16PS",
|
|
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)},
|
|
"62f2564852f4", "62 f2 56 48 52 f4 vdpbf16ps %zmm4,%zmm5,%zmm6"},
|
|
{"vdpbf16ps ymm", "VDPBF16PS",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
|
"62f2562852f4", "62 f2 56 28 52 f4 vdpbf16ps %ymm4,%ymm5,%ymm6"},
|
|
{"vdpbf16ps xmm", "VDPBF16PS",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
|
"62f2560852f4", "62 f2 56 08 52 f4 vdpbf16ps %xmm4,%xmm5,%xmm6"},
|
|
|
|
// AVX512-VP2INTERSECT, EVEX.NDS.F2.0F38. The mask destination is the
|
|
// ModR/M reg field, so a k register above k7 must refuse.
|
|
{"vp2intersectd zmm k0", "VP2INTERSECTD",
|
|
[]ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)},
|
|
"62f26f4868c1", "62 f2 6f 48 68 c1 vp2intersectd %zmm1,%zmm2,%k0"},
|
|
{"vp2intersectd ymm k2", "VP2INTERSECTD",
|
|
[]ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)},
|
|
"62f26f2868d1", "62 f2 6f 28 68 d1 vp2intersectd %ymm1,%ymm2,%k2"},
|
|
{"vp2intersectd xmm k4", "VP2INTERSECTD",
|
|
[]ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)},
|
|
"62f26f0868e1", "62 f2 6f 08 68 e1 vp2intersectd %xmm1,%xmm2,%k4"},
|
|
{"vp2intersectq zmm k0", "VP2INTERSECTQ",
|
|
[]ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)},
|
|
"62f2ef4868c1", "62 f2 ef 48 68 c1 vp2intersectq %zmm1,%zmm2,%k0"},
|
|
{"vp2intersectq ymm k2", "VP2INTERSECTQ",
|
|
[]ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)},
|
|
"62f2ef2868d1", "62 f2 ef 28 68 d1 vp2intersectq %ymm1,%ymm2,%k2"},
|
|
{"vp2intersectq xmm k4", "VP2INTERSECTQ",
|
|
[]ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)},
|
|
"62f2ef0868e1", "62 f2 ef 08 68 e1 vp2intersectq %xmm1,%xmm2,%k4"},
|
|
|
|
// AVX512-FP16 scalar arithmetic, EVEX.NDS.LIG.F3.MAP5.W0. Every
|
|
// register in the GNU vector sits above 15, so the row exercises all
|
|
// four EVEX extension bits at once.
|
|
{"vmovsh", "VMOVSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"6205160010f4", "62 05 16 00 10 f4 vmovsh %xmm28,%xmm29,%xmm30"},
|
|
{"vaddsh", "VADDSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"6205160058f4", "62 05 16 00 58 f4 vaddsh %xmm28,%xmm29,%xmm30"},
|
|
{"vsubsh", "VSUBSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620516005cf4", "62 05 16 00 5c f4 vsubsh %xmm28,%xmm29,%xmm30"},
|
|
{"vmulsh", "VMULSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"6205160059f4", "62 05 16 00 59 f4 vmulsh %xmm28,%xmm29,%xmm30"},
|
|
{"vdivsh", "VDIVSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620516005ef4", "62 05 16 00 5e f4 vdivsh %xmm28,%xmm29,%xmm30"},
|
|
{"vminsh", "VMINSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620516005df4", "62 05 16 00 5d f4 vminsh %xmm28,%xmm29,%xmm30"},
|
|
{"vmaxsh", "VMAXSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620516005ff4", "62 05 16 00 5f f4 vmaxsh %xmm28,%xmm29,%xmm30"},
|
|
{"vsqrtsh", "VSQRTSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"6205160051f4", "62 05 16 00 51 f4 vsqrtsh %xmm28,%xmm29,%xmm30"},
|
|
|
|
// The scalar scale and exponent extracts, EVEX.NDS.LIG.66.MAP6.W0.
|
|
{"vscalefsh", "VSCALEFSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620615002df4", "62 06 15 00 2d f4 vscalefsh %xmm28,%xmm29,%xmm30"},
|
|
{"vgetexpsh", "VGETEXPSH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"6206150043f4", "62 06 15 00 43 f4 vgetexpsh %xmm28,%xmm29,%xmm30"},
|
|
|
|
// The scalar compares take two operands, EVEX.LIG.MAP5.W0.
|
|
{"vcomish", "VCOMISH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
|
|
"62057c082ff5", "62 05 7c 08 2f f5 vcomish %xmm29,%xmm30"},
|
|
{"vucomish", "VUCOMISH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
|
|
"62057c082ef5", "62 05 7c 08 2e f5 vucomish %xmm29,%xmm30"},
|
|
|
|
// The floating-point conversions between the three scalar widths. The
|
|
// single-precision convert carries no prefix, the half-to-double convert
|
|
// carries F3, the double-to-half convert F2 and W1.
|
|
{"vcvtss2sh", "VCVTSS2SH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620514001df4", "62 05 14 00 1d f4 vcvtss2sh %xmm28,%xmm29,%xmm30"},
|
|
{"vcvtsh2ss", "VCVTSH2SS",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"6206140013f4", "62 06 14 00 13 f4 vcvtsh2ss %xmm28,%xmm29,%xmm30"},
|
|
{"vcvtsh2sd", "VCVTSH2SD",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620516005af4", "62 05 16 00 5a f4 vcvtsh2sd %xmm28,%xmm29,%xmm30"},
|
|
{"vcvtsd2sh", "VCVTSD2SH",
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620597005af4", "62 05 97 00 5a f4 vcvtsd2sh %xmm28,%xmm29,%xmm30"},
|
|
|
|
// The integer conversions, one entry per W bit: the W bit picks the
|
|
// 32-bit or the 64-bit general register.
|
|
{"vcvtsi2sh edx", "VCVTSI2SH",
|
|
[]ExtOperand{ExtXmm(29), ExtGpr32(2), ExtXmm(30)},
|
|
"626516002af2", "62 65 16 00 2a f2 vcvtsi2sh %edx,%xmm29,%xmm30"},
|
|
{"vcvtsi2sh r12", "VCVTSI2SH",
|
|
[]ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)},
|
|
"624596002af4", "62 45 96 00 2a f4 vcvtsi2sh %r12,%xmm29,%xmm30"},
|
|
{"vcvtusi2sh edx", "VCVTUSI2SH",
|
|
[]ExtOperand{ExtXmm(29), ExtGpr32(2), ExtXmm(30)},
|
|
"626516007bf2", "62 65 16 00 7b f2 vcvtusi2sh %edx,%xmm29,%xmm30"},
|
|
{"vcvtusi2sh r12", "VCVTUSI2SH",
|
|
[]ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)},
|
|
"624596007bf4", "62 45 96 00 7b f4 vcvtusi2sh %r12,%xmm29,%xmm30"},
|
|
{"vcvtsh2si edx", "VCVTSH2SI",
|
|
[]ExtOperand{ExtXmm(30), ExtGpr32(2)},
|
|
"62957e082dd6", "62 95 7e 08 2d d6 vcvtsh2si %xmm30,%edx"},
|
|
{"vcvtsh2si r12", "VCVTSH2SI",
|
|
[]ExtOperand{ExtXmm(30), ExtGpr64(12)},
|
|
"6215fe082de6", "62 15 fe 08 2d e6 vcvtsh2si %xmm30,%r12"},
|
|
{"vcvtsh2usi edx", "VCVTSH2USI",
|
|
[]ExtOperand{ExtXmm(30), ExtGpr32(2)},
|
|
"62957e0879d6", "62 95 7e 08 79 d6 vcvtsh2usi %xmm30,%edx"},
|
|
{"vcvtsh2usi r12", "VCVTSH2USI",
|
|
[]ExtOperand{ExtXmm(30), ExtGpr64(12)},
|
|
"6215fe0879e6", "62 15 fe 08 79 e6 vcvtsh2usi %xmm30,%r12"},
|
|
|
|
// VMOVW in both directions: the register forms are class vectors, the
|
|
// GNU file proves the opcode rows on the m16 memory forms.
|
|
{"vmovw into xmm", "VMOVW",
|
|
[]ExtOperand{ExtGpr32(12), ExtXmm(30)},
|
|
"62457d086ef4", ""},
|
|
{"vmovw out of xmm", "VMOVW",
|
|
[]ExtOperand{ExtXmm(30), ExtGpr32(12)},
|
|
"62157d087ee6", ""},
|
|
|
|
// AVX512-FP16 packed arithmetic, EVEX.NDS.MAP5.W0 with no mandatory
|
|
// prefix. The 512-bit GNU rows sit on high registers, the VL rows
|
|
// quote avx512_fp16_vl.d on the suite's low registers.
|
|
{"vaddph", "VADDPH",
|
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
|
"6205144058f4", "62 05 14 40 58 f4 vaddph %zmm28,%zmm29,%zmm30"},
|
|
{"vaddph ymm", "VADDPH",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
|
"62f5542858f4", "62 f5 54 28 58 f4 vaddph %ymm4,%ymm5,%ymm6"},
|
|
{"vaddph xmm", "VADDPH",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
|
"62f5540858f4", "62 f5 54 08 58 f4 vaddph %xmm4,%xmm5,%xmm6"},
|
|
{"vsubph", "VSUBPH",
|
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
|
"620514405cf4", "62 05 14 40 5c f4 vsubph %zmm28,%zmm29,%zmm30"},
|
|
{"vsubph ymm", "VSUBPH",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
|
"62f554285cf4", "62 f5 54 28 5c f4 vsubph %ymm4,%ymm5,%ymm6"},
|
|
{"vsubph xmm", "VSUBPH",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
|
"62f554085cf4", "62 f5 54 08 5c f4 vsubph %xmm4,%xmm5,%xmm6"},
|
|
{"vmulph", "VMULPH",
|
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
|
"6205144059f4", "62 05 14 40 59 f4 vmulph %zmm28,%zmm29,%zmm30"},
|
|
{"vmulph ymm", "VMULPH",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
|
"62f5542859f4", "62 f5 54 28 59 f4 vmulph %ymm4,%ymm5,%ymm6"},
|
|
{"vmulph xmm", "VMULPH",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
|
"62f5540859f4", "62 f5 54 08 59 f4 vmulph %xmm4,%xmm5,%xmm6"},
|
|
{"vdivph", "VDIVPH",
|
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
|
"620514405ef4", "62 05 14 40 5e f4 vdivph %zmm28,%zmm29,%zmm30"},
|
|
{"vdivph ymm", "VDIVPH",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
|
"62f554285ef4", "62 f5 54 28 5e f4 vdivph %ymm4,%ymm5,%ymm6"},
|
|
{"vdivph xmm", "VDIVPH",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
|
"62f554085ef4", "62 f5 54 08 5e f4 vdivph %xmm4,%xmm5,%xmm6"},
|
|
{"vminph", "VMINPH",
|
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
|
"620514405df4", "62 05 14 40 5d f4 vminph %zmm28,%zmm29,%zmm30"},
|
|
{"vminph ymm", "VMINPH",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
|
"62f554285df4", "62 f5 54 28 5d f4 vminph %ymm4,%ymm5,%ymm6"},
|
|
{"vminph xmm", "VMINPH",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
|
"62f554085df4", "62 f5 54 08 5d f4 vminph %xmm4,%xmm5,%xmm6"},
|
|
{"vmaxph", "VMAXPH",
|
|
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
|
|
"620514405ff4", "62 05 14 40 5f f4 vmaxph %zmm28,%zmm29,%zmm30"},
|
|
{"vmaxph ymm", "VMAXPH",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
|
|
"62f554285ff4", "62 f5 54 28 5f f4 vmaxph %ymm4,%ymm5,%ymm6"},
|
|
{"vmaxph xmm", "VMAXPH",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
|
|
"62f554085ff4", "62 f5 54 08 5f f4 vmaxph %xmm4,%xmm5,%xmm6"},
|
|
{"vsqrtph", "VSQRTPH",
|
|
[]ExtOperand{ExtZmm(29), ExtZmm(30)},
|
|
"62057c4851f5", "62 05 7c 48 51 f5 vsqrtph %zmm29,%zmm30"},
|
|
{"vsqrtph ymm", "VSQRTPH",
|
|
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
|
|
"62f57c2851f5", "62 f5 7c 28 51 f5 vsqrtph %ymm5,%ymm6"},
|
|
{"vsqrtph xmm", "VSQRTPH",
|
|
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
|
|
"62f57c0851f5", "62 f5 7c 08 51 f5 vsqrtph %xmm5,%xmm6"},
|
|
|
|
// The imm8-control group of the scalar core. The rows take the
|
|
// immediate first and the sources after it as src1, src2, the reverse
|
|
// of the listing's AT&T register order; every control byte is the $0x7b
|
|
// the suite drives through each imm8 form, save VGETMANTSH: the upper
|
|
// nibble of its control is reserved, so the layer enforces the SDM and
|
|
// encodes $0x0b where the suite's $0x7b would fault. The GNU line
|
|
// still proves the six opcode bytes, the immediate rides last as the
|
|
// operand it is.
|
|
{"vcmpsh", "VCMPSH",
|
|
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtMask(5)},
|
|
"62931600c2ec7b", "62 93 16 00 c2 ec 7b vcmpsh $0x7b,%xmm28,%xmm29,%k5"},
|
|
{"vgetmantsh", "VGETMANTSH",
|
|
[]ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"6203140027f40b", "62 03 14 00 27 f4 7b vgetmantsh $0x7b,%xmm28,%xmm29,%xmm30 (opcode row only)"},
|
|
{"vreducesh", "VREDUCESH",
|
|
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"6203140057f47b", "62 03 14 00 57 f4 7b vreducesh $0x7b,%xmm28,%xmm29,%xmm30"},
|
|
{"vrndscalesh", "VRNDSCALESH",
|
|
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"620314000af47b", "62 03 14 00 0a f4 7b vrndscalesh $0x7b,%xmm28,%xmm29,%xmm30"},
|
|
|
|
// The memory forms of the scalar moves and arithmetic, against the same
|
|
// listings' memory rows. The zero-displacement rows match the GNU
|
|
// source spellings outright and every base R8+ row exercises the EVEX.B
|
|
// high-base bit. The disp8 rows pin the bytes the listing lays down;
|
|
// binutils mainline encodes EVEX displacements with the APX disp8*N
|
|
// scaling, so its source spellings (0xfe for the m16 rows, 0x1fc0 for
|
|
// the m512 ones) are N times the plain SDM displacement those bytes
|
|
// carry, and the operand lists here hold the plain displacement. The
|
|
// rows with no GNU line are derived: the disp32 form the SDM ModR/M
|
|
// table defines and the source listings never emit plain, and the SIB
|
|
// byte the R12 base demands.
|
|
{"vmovsh load from r9", "VMOVSH",
|
|
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
|
|
"62457e081031", "62 45 7e 08 10 31 vmovsh (%r9),%xmm30"},
|
|
{"vmovsh load disp8", "VMOVSH",
|
|
[]ExtOperand{ExtMemory(1, 127), ExtXmm(30)},
|
|
"62657e0810717f", "62 65 7e 08 10 71 7f vmovsh 0xfe(%rcx),%xmm30 (Disp8(7f))"},
|
|
{"vmovsh store to r9", "VMOVSH",
|
|
[]ExtOperand{ExtXmm(30), ExtMemory(9, 0)},
|
|
"62457e081131", "62 45 7e 08 11 31 vmovsh %xmm30,(%r9)"},
|
|
{"vmovsh store disp8", "VMOVSH",
|
|
[]ExtOperand{ExtXmm(30), ExtMemory(1, 127)},
|
|
"62657e0811717f", "62 65 7e 08 11 71 7f vmovsh %xmm30,0xfe(%rcx) (Disp8(7f))"},
|
|
{"vmovsh store negative disp32", "VMOVSH",
|
|
[]ExtOperand{ExtXmm(30), ExtMemory(13, -200)},
|
|
"62457e0811b538ffffff", ""},
|
|
{"vmovw load from r9", "VMOVW",
|
|
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
|
|
"62457d086e31", "62 45 7d 08 6e 31 vmovw (%r9),%xmm30"},
|
|
{"vmovw load disp8", "VMOVW",
|
|
[]ExtOperand{ExtMemory(1, 127), ExtXmm(30)},
|
|
"62657d086e717f", "62 65 7d 08 6e 71 7f vmovw 0xfe(%rcx),%xmm30 (Disp8(7f))"},
|
|
{"vmovw store to r9", "VMOVW",
|
|
[]ExtOperand{ExtXmm(30), ExtMemory(9, 0)},
|
|
"62457d087e31", "62 45 7d 08 7e 31 vmovw %xmm30,(%r9)"},
|
|
{"vmovw store disp8", "VMOVW",
|
|
[]ExtOperand{ExtXmm(30), ExtMemory(1, 127)},
|
|
"62657d087e717f", "62 65 7d 08 7e 71 7f vmovw %xmm30,0xfe(%rcx) (Disp8(7f))"},
|
|
{"vaddsh memory source", "VADDSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
|
"624516005831", "62 45 16 00 58 31 vaddsh (%r9),%xmm29,%xmm30"},
|
|
{"vaddsh memory source disp8", "VADDSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
|
|
"6265160058717f", "62 65 16 00 58 71 7f vaddsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
|
|
{"vaddsh memory source disp32", "VADDSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(2, 8128), ExtXmm(30)},
|
|
"6265160058b2c01f0000", ""},
|
|
{"vsubsh memory source", "VSUBSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
|
"624516005c31", "62 45 16 00 5c 31 vsubsh (%r9),%xmm29,%xmm30"},
|
|
{"vmulsh memory source disp8", "VMULSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
|
|
"6265160059717f", "62 65 16 00 59 71 7f vmulsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
|
|
{"vdivsh memory source", "VDIVSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
|
"624516005e31", "62 45 16 00 5e 31 vdivsh (%r9),%xmm29,%xmm30"},
|
|
{"vminsh memory source disp8", "VMINSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
|
|
"626516005d717f", "62 65 16 00 5d 71 7f vminsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
|
|
{"vmaxsh memory source", "VMAXSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
|
"624516005f31", "62 45 16 00 5f 31 vmaxsh (%r9),%xmm29,%xmm30"},
|
|
{"vsqrtsh memory source", "VSQRTSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
|
"624516005131", "62 45 16 00 51 31 vsqrtsh (%r9),%xmm29,%xmm30"},
|
|
{"vsqrtsh memory source negative disp8", "VSQRTSH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(2, -128), ExtXmm(30)},
|
|
"62651600517280", "62 65 16 87 51 72 80 vsqrtsh -0x100(%rdx),%xmm29,%xmm30 (Disp8(80); the GNU row adds {k7}{z})"},
|
|
|
|
// High registers in a 512-bit form exercise the EVEX extension bits:
|
|
// with both sources above 15 the B bar and X bar bits clear, while the
|
|
// destination zmm23 keeps R bar set in byte one (derived from the
|
|
// proven class above).
|
|
{"vcvtne2ps2bf16 high registers", "VCVTNE2PS2BF16",
|
|
[]ExtOperand{ExtZmm(21), ExtZmm(20), ExtZmm(23)},
|
|
"62a2574072fc", ""},
|
|
}
|
|
|
|
// amd64ResolveEntry finds the table entry a golden row exercises: the entry
|
|
// is the one that accepts the row's operands, which is what pins the bytes to
|
|
// a single template when a mnemonic registers one entry per W bit.
|
|
func amd64ResolveEntry(mnem string, ops []ExtOperand) (ExtInstr, bool) {
|
|
var first ExtInstr
|
|
for _, in := range Extensions(AMD64) {
|
|
if in.Name != mnem {
|
|
continue
|
|
}
|
|
if _, err := in.Encode(ops); err == nil {
|
|
return in, true
|
|
}
|
|
if first.Name == "" {
|
|
first = in
|
|
}
|
|
}
|
|
if first.Name != "" {
|
|
return first, true
|
|
}
|
|
return ExtInstr{}, false
|
|
}
|
|
|
|
func amd64ExtInstr(t *testing.T, mnem string, class ExtOperandKind, preds ...func(ExtInstr) bool) ExtInstr {
|
|
t.Helper()
|
|
for _, in := range Extensions(AMD64) {
|
|
if in.Name != mnem || amd64LengthClass(in.Bytes) != class {
|
|
continue
|
|
}
|
|
match := true
|
|
for _, p := range preds {
|
|
if !p(in) {
|
|
match = false
|
|
}
|
|
}
|
|
if match {
|
|
return in
|
|
}
|
|
}
|
|
t.Fatalf("no extended %s encoding at the %s vector length", mnem, class)
|
|
return ExtInstr{}
|
|
}
|
|
|
|
// amd64W1 names the W1 encoding of a mnemonic registered once per W bit.
|
|
func amd64W1(in ExtInstr) bool { return in.Bytes[2]&0x80 != 0 }
|
|
|
|
// withPred adapts an optional row predicate for the variadic lookup.
|
|
func withPred(p func(ExtInstr) bool) []func(ExtInstr) bool {
|
|
if p == nil {
|
|
return nil
|
|
}
|
|
return []func(ExtInstr) bool{p}
|
|
}
|
|
|
|
func TestAmd64ExtGoldenBytes(t *testing.T) {
|
|
for _, tt := range amd64GoldenRows {
|
|
in, ok := amd64ResolveEntry(tt.mnem, tt.ops)
|
|
if !ok {
|
|
t.Errorf("%s: no table entry for %s at the row's vector length", tt.name, tt.mnem)
|
|
continue
|
|
}
|
|
got, err := in.Encode(tt.ops)
|
|
if err != nil {
|
|
t.Errorf("%s: encode: %v", tt.name, err)
|
|
continue
|
|
}
|
|
if hex.EncodeToString(got) != tt.want {
|
|
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestAmd64ExtTemplateIntegrity checks the metadata contract: every entry
|
|
// names its manual reference, summary and feature, and every template carries
|
|
// the fixed shape of an EVEX register form with the register-derived bits
|
|
// zero, so a slip in the table is an error and not a stray byte.
|
|
func TestAmd64ExtTemplateIntegrity(t *testing.T) {
|
|
features := map[ExtFeature]bool{
|
|
ExtFeatureBF16: true,
|
|
ExtFeatureVP2INTERSECT: true,
|
|
ExtFeatureFP16: true,
|
|
}
|
|
for _, in := range Extensions(AMD64) {
|
|
if in.Name == "" || in.Summary == "" || in.Ref == "" {
|
|
t.Errorf("%+v: name, summary and reference are mandatory", in)
|
|
}
|
|
if !features[in.Feature] {
|
|
t.Errorf("%s: feature %q is not an amd64 extension feature", in.Name, in.Feature)
|
|
}
|
|
if len(in.Bytes) != 6 {
|
|
t.Errorf("%s: the template is %d bytes, want the 6-byte EVEX register form", in.Name, len(in.Bytes))
|
|
continue
|
|
}
|
|
if in.Bytes[0] != 0x62 {
|
|
t.Errorf("%s: the template opens with %02x, want the EVEX escape 62", in.Name, in.Bytes[0])
|
|
}
|
|
if in.Bytes[1]&0xf0 != 0 {
|
|
t.Errorf("%s: byte one carries register bits %04b, want them zero", in.Name, in.Bytes[1]>>4)
|
|
}
|
|
if in.Bytes[2]&0x78 != 0 {
|
|
t.Errorf("%s: byte two carries vvvv bits %04b, want them zero", in.Name, in.Bytes[2]>>3&0xf)
|
|
}
|
|
if in.Bytes[2]&0x04 == 0 {
|
|
t.Errorf("%s: byte two lacks the reserved one-bit", in.Name)
|
|
}
|
|
if in.Bytes[3]&0x9f != 0 {
|
|
t.Errorf("%s: byte three carries z, b, V prime or aaa bits, want them zero: %08b", in.Name, in.Bytes[3])
|
|
}
|
|
if in.Bytes[5]&0x3f != 0 || in.Bytes[5]&0xc0 != 0xc0 {
|
|
t.Errorf("%s: byte five is %08b, want mod 11 with the reg and rm fields zero", in.Name, in.Bytes[5])
|
|
}
|
|
if in.Form.Arity() < 2 || in.Form.Arity() > 4 {
|
|
t.Errorf("%s: form %s carries an unusable arity %d", in.Name, in.Form, in.Form.Arity())
|
|
}
|
|
if in.Mem > in.Form.Arity() {
|
|
t.Errorf("%s: Mem names operand %d, outside the form's %d positions", in.Name, in.Mem, in.Form.Arity())
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestAmd64ExtEveryEntryCarriesGoldenVector pins the measure the layer is
|
|
// judged by: every registered entry is covered by at least one golden vector
|
|
// whose resolved entry has the very template, so an entry without provenance
|
|
// cannot hide.
|
|
func TestAmd64ExtEveryEntryCarriesGoldenVector(t *testing.T) {
|
|
for _, in := range Extensions(AMD64) {
|
|
found := false
|
|
for _, tt := range amd64GoldenRows {
|
|
cand, ok := amd64ResolveEntry(tt.mnem, tt.ops)
|
|
if ok && cand.Name == in.Name && string(cand.Bytes) == string(in.Bytes) {
|
|
found = true
|
|
}
|
|
}
|
|
if !found {
|
|
t.Errorf("%s (% x) has no golden vector", in.Name, in.Bytes)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestAmd64ExtRejects(t *testing.T) {
|
|
for _, tt := range []struct {
|
|
name string
|
|
mnem string
|
|
ops []ExtOperand
|
|
quote string // a fragment the error carries
|
|
}{
|
|
{"wrong vector class", "VCVTNE2PS2BF16",
|
|
[]ExtOperand{ExtZmm(1), ExtZmm(2), ExtYmm(3)},
|
|
"wants a ZMM register"},
|
|
{"destination class is the source's on the narrow convert", "VCVTNEPS2BF16",
|
|
[]ExtOperand{ExtZmm(1), ExtZmm(2)},
|
|
"wants a YMM register"},
|
|
{"vector in the mask position", "VP2INTERSECTD",
|
|
[]ExtOperand{ExtZmm(1), ExtZmm(2), ExtZmm(3)},
|
|
"wants an opmask register"},
|
|
{"mask register beyond k7", "VP2INTERSECTD",
|
|
[]ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(8)},
|
|
"outside 0-7"},
|
|
{"vector where the general register belongs", "VCVTSH2SI",
|
|
[]ExtOperand{ExtXmm(1), ExtXmm(2)},
|
|
"wants a 32-bit general register"},
|
|
{"64-bit register on the W0 convert", "VCVTSI2SH",
|
|
[]ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)},
|
|
"wants a 32-bit general register"},
|
|
{"general register beyond r15", "VMOVW",
|
|
[]ExtOperand{ExtGpr32(16), ExtXmm(30)},
|
|
"outside 0-15"},
|
|
{"wrong arity", "VP2INTERSECTD",
|
|
[]ExtOperand{ExtZmm(2), ExtZmm(1)},
|
|
"takes 3 operands"},
|
|
{"arm64 arrangement suffix", "VCVTNE2PS2BF16",
|
|
[]ExtOperand{{Kind: ExtZMM, Reg: 1, Arr: ExtArrS}, ExtZmm(2), ExtZmm(3)},
|
|
"arrangement"},
|
|
{"predicate qualifier", "VCVTNEPS2BF16",
|
|
[]ExtOperand{{Kind: ExtZMM, Reg: 1, Qual: ExtQualZeroing}, ExtZmm(2)},
|
|
"predicate qualifier"},
|
|
{"vector where the control byte belongs", "VGETMANTSH",
|
|
[]ExtOperand{ExtXmm(28), ExtXmm(29), ExtXmm(30), ExtXmm(31)},
|
|
"wants an immediate control byte"},
|
|
{"reserved upper nibble on the mantissa control", "VGETMANTSH",
|
|
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
|
|
"reserved and must be zero"},
|
|
{"control byte under the floor", "VREDUCESH",
|
|
[]ExtOperand{ExtImmediate(-1), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
|
|
"outside the unsigned byte range"},
|
|
{"control byte over the top", "VRNDSCALESH",
|
|
[]ExtOperand{ExtImmediate(256), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
|
|
"outside the unsigned byte range"},
|
|
{"shift on the control byte", "VRNDSCALESH",
|
|
[]ExtOperand{ExtShiftedImmediate(0x0b, 8), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
|
|
"take none"},
|
|
{"vector in the mask position of the compare", "VCMPSH",
|
|
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
|
|
"wants an opmask register"},
|
|
{"mask beyond k7 on the compare", "VCMPSH",
|
|
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtMask(8)},
|
|
"outside 0-7"},
|
|
{"memory in the arithmetic's first source", "VADDSH",
|
|
[]ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(30)},
|
|
"wants an XMM register"},
|
|
{"memory in the arithmetic's destination", "VADDSH",
|
|
[]ExtOperand{ExtXmm(28), ExtXmm(29), ExtMemory(9, 0)},
|
|
"wants an XMM register"},
|
|
{"memory where the general register belongs", "VCVTSI2SH",
|
|
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
|
|
"wants a 32-bit general register"},
|
|
{"memory as the compare's second source", "VCMPSH",
|
|
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtMemory(9, 0), ExtMask(5)},
|
|
"wants an XMM register"},
|
|
{"memory as the intersect source", "VP2INTERSECTD",
|
|
[]ExtOperand{ExtZmm(2), ExtMemory(9, 0), ExtMask(0)},
|
|
"wants a ZMM register"},
|
|
{"memory as the convert's source", "VCVTSS2SH",
|
|
[]ExtOperand{ExtXmm(28), ExtMemory(9, 0), ExtXmm(30)},
|
|
"wants an XMM register"},
|
|
{"memory as the control byte", "VGETMANTSH",
|
|
[]ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
|
|
"wants an immediate control byte"},
|
|
} {
|
|
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
|
|
_, err := in.Encode(tt.ops)
|
|
if err == nil {
|
|
t.Errorf("%s: encode succeeded, want an error", tt.name)
|
|
continue
|
|
}
|
|
if !strings.Contains(err.Error(), tt.quote) {
|
|
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
|
|
}
|
|
}
|
|
// The W1 convert refuses the 32-bit register the W0 entry takes.
|
|
in := amd64ExtInstr(t, "VCVTSH2SI", ExtXMM, amd64W1)
|
|
if _, err := in.Encode([]ExtOperand{ExtXmm(30), ExtGpr32(2)}); err == nil {
|
|
t.Error("a 32-bit register encoded on the W1 convert, want an error")
|
|
} else if !strings.Contains(err.Error(), "wants a 64-bit general register") {
|
|
t.Errorf("the W1 error %q does not name the 64-bit class", err)
|
|
}
|
|
}
|
|
|
|
// TestAmd64ExtMemoryFormRejects covers the shapes the memory forms refuse:
|
|
// a register in the load's memory position, a memory operand in the store's
|
|
// register position, and the out-of-range bases and displacements. The
|
|
// rows resolve against the load and store entries themselves, which the
|
|
// name-and-class lookup cannot pick alone: the register forms of the same
|
|
// mnemonics share the class.
|
|
func TestAmd64ExtMemoryFormRejects(t *testing.T) {
|
|
load := func(in ExtInstr) bool { return in.Form == ExtFormAmdMemVec }
|
|
store := func(in ExtInstr) bool { return in.Form == ExtFormAmdVecMem }
|
|
for _, tt := range []struct {
|
|
name string
|
|
mnem string
|
|
pick func(ExtInstr) bool
|
|
ops []ExtOperand
|
|
quote string
|
|
}{
|
|
{"vector in the load's memory position", "VMOVSH", load,
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
|
|
"wants a memory operand"},
|
|
{"memory in the store's register position", "VMOVSH", store,
|
|
[]ExtOperand{ExtMemory(9, 0), ExtMemory(1, 0)},
|
|
"wants an XMM register"},
|
|
{"vector in the store's memory position", "VMOVSH", store,
|
|
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
|
|
"wants a memory operand"},
|
|
{"base beyond r15 on the load", "VMOVW", load,
|
|
[]ExtOperand{ExtMemory(16, 0), ExtXmm(30)},
|
|
"outside 0-15"},
|
|
{"displacement past the signed 32-bit range on the store", "VMOVW", store,
|
|
[]ExtOperand{ExtXmm(30), ExtMemory(8, 1<<32)},
|
|
"outside the signed 32-bit range"},
|
|
} {
|
|
in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick)
|
|
_, err := in.Encode(tt.ops)
|
|
if err == nil {
|
|
t.Errorf("%s: encode succeeded, want an error", tt.name)
|
|
continue
|
|
}
|
|
if !strings.Contains(err.Error(), tt.quote) {
|
|
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
|
|
}
|
|
}
|
|
}
|
|
|
|
// operandClass names the vector class a row exercises, the key the entry
|
|
// lookup resolves with.
|
|
func operandClass(t *testing.T, ops []ExtOperand) ExtOperandKind {
|
|
t.Helper()
|
|
for _, op := range ops {
|
|
switch op.Kind {
|
|
case ExtXMM, ExtYMM, ExtZMM:
|
|
return op.Kind
|
|
}
|
|
}
|
|
t.Fatal("the row carries no vector operand to pick the entry with")
|
|
return ExtXMM
|
|
}
|
|
|
|
// TestAmd64ExtImm8Tables pins the imm8 semantics the layer carries as data
|
|
// against the SDM tables they are transcribed from: the rounding modes of
|
|
// the round control, the sign control of the mantissa extraction and the 32
|
|
// comparison predicates, in encoding order.
|
|
func TestAmd64ExtImm8Tables(t *testing.T) {
|
|
roundModes := [4]string{
|
|
"round to nearest (even)",
|
|
"round down (toward -infinity)",
|
|
"round up (toward +infinity)",
|
|
"round toward zero (truncate)",
|
|
}
|
|
if ExtFP16RoundingModes != roundModes {
|
|
t.Errorf("rounding modes %q, want the SDM RC field order", ExtFP16RoundingModes)
|
|
}
|
|
for i, sign := range ExtFP16GetMantSigns {
|
|
switch i {
|
|
case 0:
|
|
if sign != "the sign of the source" {
|
|
t.Errorf("sign control 0b00 = %q, want the source's own sign", sign)
|
|
}
|
|
case 1:
|
|
if sign != "positive" {
|
|
t.Errorf("sign control 0b01 = %q, want a forced positive", sign)
|
|
}
|
|
default:
|
|
if sign != "the indefinite NaN when the source is negative" {
|
|
t.Errorf("sign control 0b1x = %q, want the indefinite NaN branch", sign)
|
|
}
|
|
}
|
|
}
|
|
predicates := map[int]string{
|
|
0: "EQ_OQ", 1: "LT_OS", 2: "LE_OS", 3: "UNORD_Q", 4: "NEQ_UQ",
|
|
5: "NLT_US", 6: "NLE_US", 7: "ORD_Q", 8: "EQ_UQ", 15: "TRUE_UQ",
|
|
16: "EQ_OS", 23: "ORD_S", 24: "EQ_US", 27: "FALSE_OS", 31: "TRUE_US",
|
|
}
|
|
for i, want := range predicates {
|
|
if got := ExtFP16CmpPredicates[i]; got != want {
|
|
t.Errorf("predicate 0x%02x = %q, want %q", i, got, want)
|
|
}
|
|
}
|
|
if ExtFP16CmpPredicates[31] != "TRUE_US" {
|
|
t.Errorf("the predicate table ends at %q, want TRUE_US", ExtFP16CmpPredicates[31])
|
|
}
|
|
}
|
|
|
|
// TestAmd64ExtArchBinding pins the layer's architecture binding: only riscv
|
|
// and loong64 have no extended layer, arm64's lives in arm64_ext.go and the
|
|
// amd64 one here.
|
|
func TestAmd64ExtArchBinding(t *testing.T) {
|
|
for _, a := range []Arch{RISCV, LOONG64, Unknown} {
|
|
if got := Extensions(a); len(got) != 0 {
|
|
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
|
|
}
|
|
}
|
|
if got := Extensions(AMD64); len(got) != 70 {
|
|
t.Errorf("the amd64 layer registers %d instructions, want 70", len(got))
|
|
}
|
|
}
|