Files
gasm-sdk/arch/amd64_ext_test.go
T

697 lines
29 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import (
"encoding/hex"
"strings"
"testing"
)
// The BF16, VP2INTERSECT and FP16 encodings have no toolchain oracle: go
// tool asm knows none of these families. The golden words below are
// transcribed from the Intel SDM instruction entries and cross-checked
// against binutils-gdb's own assembler testsuite: every row marked "GNU"
// matches a vector in gas/testsuite/gas/i386/avx512_bf16.d,
// avx512_bf16_vl.d, x86-64-vp2intersect.d, x86-64-avx512_fp16.d or
// avx512_fp16_vl.d byte for byte, so no entry rests on transcription alone. The two VMOVW rows are
// class vectors: the GNU file proves the 66.MAP5 opcode row on the m16
// memory forms, and the register form follows the manual's ModR/M reg row.
// The GNU dumps print AT&T order (sources first, destination last), which is
// the order the operands are built in here too.
type amd64GoldenRow struct {
name string
mnem string
ops []ExtOperand
want string // hex, little-endian bytes in memory order
GNU string // the matching binutils-gdb line, empty for a derived register form
}
var amd64GoldenRows = []amd64GoldenRow{
// AVX512-BF16, EVEX.NDS.F2.0F38.W0 for the three-register convert,
// EVEX.F3.0F38.W0 for the narrow convert and the dot product.
{"vcvtne2ps2bf16 zmm", "VCVTNE2PS2BF16",
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)},
"62f2574872f4", "62 f2 57 48 72 f4 vcvtne2ps2bf16 %zmm4,%zmm5,%zmm6"},
{"vcvtne2ps2bf16 ymm", "VCVTNE2PS2BF16",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f2572872f4", "62 f2 57 28 72 f4 vcvtne2ps2bf16 %ymm4,%ymm5,%ymm6"},
{"vcvtne2ps2bf16 xmm", "VCVTNE2PS2BF16",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f2570872f4", "62 f2 57 08 72 f4 vcvtne2ps2bf16 %xmm4,%xmm5,%xmm6"},
{"vcvtneps2bf16 zmm to ymm", "VCVTNEPS2BF16",
[]ExtOperand{ExtZmm(5), ExtYmm(6)},
"62f27e4872f5", "62 f2 7e 48 72 f5 vcvtneps2bf16 %zmm5,%ymm6"},
{"vcvtneps2bf16 ymm to xmm", "VCVTNEPS2BF16",
[]ExtOperand{ExtYmm(5), ExtXmm(6)},
"62f27e2872f5", "62 f2 7e 28 72 f5 vcvtneps2bf16 %ymm5,%xmm6"},
{"vcvtneps2bf16 xmm to xmm", "VCVTNEPS2BF16",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f27e0872f5", "62 f2 7e 08 72 f5 vcvtneps2bf16 %xmm5,%xmm6"},
{"vdpbf16ps zmm", "VDPBF16PS",
[]ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)},
"62f2564852f4", "62 f2 56 48 52 f4 vdpbf16ps %zmm4,%zmm5,%zmm6"},
{"vdpbf16ps ymm", "VDPBF16PS",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f2562852f4", "62 f2 56 28 52 f4 vdpbf16ps %ymm4,%ymm5,%ymm6"},
{"vdpbf16ps xmm", "VDPBF16PS",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f2560852f4", "62 f2 56 08 52 f4 vdpbf16ps %xmm4,%xmm5,%xmm6"},
// AVX512-VP2INTERSECT, EVEX.NDS.F2.0F38. The mask destination is the
// ModR/M reg field, so a k register above k7 must refuse.
{"vp2intersectd zmm k0", "VP2INTERSECTD",
[]ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)},
"62f26f4868c1", "62 f2 6f 48 68 c1 vp2intersectd %zmm1,%zmm2,%k0"},
{"vp2intersectd ymm k2", "VP2INTERSECTD",
[]ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)},
"62f26f2868d1", "62 f2 6f 28 68 d1 vp2intersectd %ymm1,%ymm2,%k2"},
{"vp2intersectd xmm k4", "VP2INTERSECTD",
[]ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)},
"62f26f0868e1", "62 f2 6f 08 68 e1 vp2intersectd %xmm1,%xmm2,%k4"},
{"vp2intersectq zmm k0", "VP2INTERSECTQ",
[]ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)},
"62f2ef4868c1", "62 f2 ef 48 68 c1 vp2intersectq %zmm1,%zmm2,%k0"},
{"vp2intersectq ymm k2", "VP2INTERSECTQ",
[]ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)},
"62f2ef2868d1", "62 f2 ef 28 68 d1 vp2intersectq %ymm1,%ymm2,%k2"},
{"vp2intersectq xmm k4", "VP2INTERSECTQ",
[]ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)},
"62f2ef0868e1", "62 f2 ef 08 68 e1 vp2intersectq %xmm1,%xmm2,%k4"},
// AVX512-FP16 scalar arithmetic, EVEX.NDS.LIG.F3.MAP5.W0. Every
// register in the GNU vector sits above 15, so the row exercises all
// four EVEX extension bits at once.
{"vmovsh", "VMOVSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6205160010f4", "62 05 16 00 10 f4 vmovsh %xmm28,%xmm29,%xmm30"},
{"vaddsh", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6205160058f4", "62 05 16 00 58 f4 vaddsh %xmm28,%xmm29,%xmm30"},
{"vsubsh", "VSUBSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620516005cf4", "62 05 16 00 5c f4 vsubsh %xmm28,%xmm29,%xmm30"},
{"vmulsh", "VMULSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6205160059f4", "62 05 16 00 59 f4 vmulsh %xmm28,%xmm29,%xmm30"},
{"vdivsh", "VDIVSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620516005ef4", "62 05 16 00 5e f4 vdivsh %xmm28,%xmm29,%xmm30"},
{"vminsh", "VMINSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620516005df4", "62 05 16 00 5d f4 vminsh %xmm28,%xmm29,%xmm30"},
{"vmaxsh", "VMAXSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620516005ff4", "62 05 16 00 5f f4 vmaxsh %xmm28,%xmm29,%xmm30"},
{"vsqrtsh", "VSQRTSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6205160051f4", "62 05 16 00 51 f4 vsqrtsh %xmm28,%xmm29,%xmm30"},
// The scalar scale and exponent extracts, EVEX.NDS.LIG.66.MAP6.W0.
{"vscalefsh", "VSCALEFSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620615002df4", "62 06 15 00 2d f4 vscalefsh %xmm28,%xmm29,%xmm30"},
{"vgetexpsh", "VGETEXPSH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6206150043f4", "62 06 15 00 43 f4 vgetexpsh %xmm28,%xmm29,%xmm30"},
// The scalar compares take two operands, EVEX.LIG.MAP5.W0.
{"vcomish", "VCOMISH",
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
"62057c082ff5", "62 05 7c 08 2f f5 vcomish %xmm29,%xmm30"},
{"vucomish", "VUCOMISH",
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
"62057c082ef5", "62 05 7c 08 2e f5 vucomish %xmm29,%xmm30"},
// The floating-point conversions between the three scalar widths. The
// single-precision convert carries no prefix, the half-to-double convert
// carries F3, the double-to-half convert F2 and W1.
{"vcvtss2sh", "VCVTSS2SH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620514001df4", "62 05 14 00 1d f4 vcvtss2sh %xmm28,%xmm29,%xmm30"},
{"vcvtsh2ss", "VCVTSH2SS",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6206140013f4", "62 06 14 00 13 f4 vcvtsh2ss %xmm28,%xmm29,%xmm30"},
{"vcvtsh2sd", "VCVTSH2SD",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620516005af4", "62 05 16 00 5a f4 vcvtsh2sd %xmm28,%xmm29,%xmm30"},
{"vcvtsd2sh", "VCVTSD2SH",
[]ExtOperand{ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620597005af4", "62 05 97 00 5a f4 vcvtsd2sh %xmm28,%xmm29,%xmm30"},
// The integer conversions, one entry per W bit: the W bit picks the
// 32-bit or the 64-bit general register.
{"vcvtsi2sh edx", "VCVTSI2SH",
[]ExtOperand{ExtXmm(29), ExtGpr32(2), ExtXmm(30)},
"626516002af2", "62 65 16 00 2a f2 vcvtsi2sh %edx,%xmm29,%xmm30"},
{"vcvtsi2sh r12", "VCVTSI2SH",
[]ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)},
"624596002af4", "62 45 96 00 2a f4 vcvtsi2sh %r12,%xmm29,%xmm30"},
{"vcvtusi2sh edx", "VCVTUSI2SH",
[]ExtOperand{ExtXmm(29), ExtGpr32(2), ExtXmm(30)},
"626516007bf2", "62 65 16 00 7b f2 vcvtusi2sh %edx,%xmm29,%xmm30"},
{"vcvtusi2sh r12", "VCVTUSI2SH",
[]ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)},
"624596007bf4", "62 45 96 00 7b f4 vcvtusi2sh %r12,%xmm29,%xmm30"},
{"vcvtsh2si edx", "VCVTSH2SI",
[]ExtOperand{ExtXmm(30), ExtGpr32(2)},
"62957e082dd6", "62 95 7e 08 2d d6 vcvtsh2si %xmm30,%edx"},
{"vcvtsh2si r12", "VCVTSH2SI",
[]ExtOperand{ExtXmm(30), ExtGpr64(12)},
"6215fe082de6", "62 15 fe 08 2d e6 vcvtsh2si %xmm30,%r12"},
{"vcvtsh2usi edx", "VCVTSH2USI",
[]ExtOperand{ExtXmm(30), ExtGpr32(2)},
"62957e0879d6", "62 95 7e 08 79 d6 vcvtsh2usi %xmm30,%edx"},
{"vcvtsh2usi r12", "VCVTSH2USI",
[]ExtOperand{ExtXmm(30), ExtGpr64(12)},
"6215fe0879e6", "62 15 fe 08 79 e6 vcvtsh2usi %xmm30,%r12"},
// VMOVW in both directions: the register forms are class vectors, the
// GNU file proves the opcode rows on the m16 memory forms.
{"vmovw into xmm", "VMOVW",
[]ExtOperand{ExtGpr32(12), ExtXmm(30)},
"62457d086ef4", ""},
{"vmovw out of xmm", "VMOVW",
[]ExtOperand{ExtXmm(30), ExtGpr32(12)},
"62157d087ee6", ""},
// AVX512-FP16 packed arithmetic, EVEX.NDS.MAP5.W0 with no mandatory
// prefix. The 512-bit GNU rows sit on high registers, the VL rows
// quote avx512_fp16_vl.d on the suite's low registers.
{"vaddph", "VADDPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
"6205144058f4", "62 05 14 40 58 f4 vaddph %zmm28,%zmm29,%zmm30"},
{"vaddph ymm", "VADDPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f5542858f4", "62 f5 54 28 58 f4 vaddph %ymm4,%ymm5,%ymm6"},
{"vaddph xmm", "VADDPH",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f5540858f4", "62 f5 54 08 58 f4 vaddph %xmm4,%xmm5,%xmm6"},
{"vsubph", "VSUBPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
"620514405cf4", "62 05 14 40 5c f4 vsubph %zmm28,%zmm29,%zmm30"},
{"vsubph ymm", "VSUBPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f554285cf4", "62 f5 54 28 5c f4 vsubph %ymm4,%ymm5,%ymm6"},
{"vsubph xmm", "VSUBPH",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f554085cf4", "62 f5 54 08 5c f4 vsubph %xmm4,%xmm5,%xmm6"},
{"vmulph", "VMULPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
"6205144059f4", "62 05 14 40 59 f4 vmulph %zmm28,%zmm29,%zmm30"},
{"vmulph ymm", "VMULPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f5542859f4", "62 f5 54 28 59 f4 vmulph %ymm4,%ymm5,%ymm6"},
{"vmulph xmm", "VMULPH",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f5540859f4", "62 f5 54 08 59 f4 vmulph %xmm4,%xmm5,%xmm6"},
{"vdivph", "VDIVPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
"620514405ef4", "62 05 14 40 5e f4 vdivph %zmm28,%zmm29,%zmm30"},
{"vdivph ymm", "VDIVPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f554285ef4", "62 f5 54 28 5e f4 vdivph %ymm4,%ymm5,%ymm6"},
{"vdivph xmm", "VDIVPH",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f554085ef4", "62 f5 54 08 5e f4 vdivph %xmm4,%xmm5,%xmm6"},
{"vminph", "VMINPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
"620514405df4", "62 05 14 40 5d f4 vminph %zmm28,%zmm29,%zmm30"},
{"vminph ymm", "VMINPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f554285df4", "62 f5 54 28 5d f4 vminph %ymm4,%ymm5,%ymm6"},
{"vminph xmm", "VMINPH",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f554085df4", "62 f5 54 08 5d f4 vminph %xmm4,%xmm5,%xmm6"},
{"vmaxph", "VMAXPH",
[]ExtOperand{ExtZmm(29), ExtZmm(28), ExtZmm(30)},
"620514405ff4", "62 05 14 40 5f f4 vmaxph %zmm28,%zmm29,%zmm30"},
{"vmaxph ymm", "VMAXPH",
[]ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)},
"62f554285ff4", "62 f5 54 28 5f f4 vmaxph %ymm4,%ymm5,%ymm6"},
{"vmaxph xmm", "VMAXPH",
[]ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)},
"62f554085ff4", "62 f5 54 08 5f f4 vmaxph %xmm4,%xmm5,%xmm6"},
{"vsqrtph", "VSQRTPH",
[]ExtOperand{ExtZmm(29), ExtZmm(30)},
"62057c4851f5", "62 05 7c 48 51 f5 vsqrtph %zmm29,%zmm30"},
{"vsqrtph ymm", "VSQRTPH",
[]ExtOperand{ExtYmm(5), ExtYmm(6)},
"62f57c2851f5", "62 f5 7c 28 51 f5 vsqrtph %ymm5,%ymm6"},
{"vsqrtph xmm", "VSQRTPH",
[]ExtOperand{ExtXmm(5), ExtXmm(6)},
"62f57c0851f5", "62 f5 7c 08 51 f5 vsqrtph %xmm5,%xmm6"},
// The imm8-control group of the scalar core. The rows take the
// immediate first and the sources after it as src1, src2, the reverse
// of the listing's AT&T register order; every control byte is the $0x7b
// the suite drives through each imm8 form, save VGETMANTSH: the upper
// nibble of its control is reserved, so the layer enforces the SDM and
// encodes $0x0b where the suite's $0x7b would fault. The GNU line
// still proves the six opcode bytes, the immediate rides last as the
// operand it is.
{"vcmpsh", "VCMPSH",
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtMask(5)},
"62931600c2ec7b", "62 93 16 00 c2 ec 7b vcmpsh $0x7b,%xmm28,%xmm29,%k5"},
{"vgetmantsh", "VGETMANTSH",
[]ExtOperand{ExtImmediate(0x0b), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6203140027f40b", "62 03 14 00 27 f4 7b vgetmantsh $0x7b,%xmm28,%xmm29,%xmm30 (opcode row only)"},
{"vreducesh", "VREDUCESH",
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"6203140057f47b", "62 03 14 00 57 f4 7b vreducesh $0x7b,%xmm28,%xmm29,%xmm30"},
{"vrndscalesh", "VRNDSCALESH",
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"620314000af47b", "62 03 14 00 0a f4 7b vrndscalesh $0x7b,%xmm28,%xmm29,%xmm30"},
// The memory forms of the scalar moves and arithmetic, against the same
// listings' memory rows. The zero-displacement rows match the GNU
// source spellings outright and every base R8+ row exercises the EVEX.B
// high-base bit. The disp8 rows pin the bytes the listing lays down;
// binutils mainline encodes EVEX displacements with the APX disp8*N
// scaling, so its source spellings (0xfe for the m16 rows, 0x1fc0 for
// the m512 ones) are N times the plain SDM displacement those bytes
// carry, and the operand lists here hold the plain displacement. The
// rows with no GNU line are derived: the disp32 form the SDM ModR/M
// table defines and the source listings never emit plain, and the SIB
// byte the R12 base demands.
{"vmovsh load from r9", "VMOVSH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
"62457e081031", "62 45 7e 08 10 31 vmovsh (%r9),%xmm30"},
{"vmovsh load disp8", "VMOVSH",
[]ExtOperand{ExtMemory(1, 127), ExtXmm(30)},
"62657e0810717f", "62 65 7e 08 10 71 7f vmovsh 0xfe(%rcx),%xmm30 (Disp8(7f))"},
{"vmovsh store to r9", "VMOVSH",
[]ExtOperand{ExtXmm(30), ExtMemory(9, 0)},
"62457e081131", "62 45 7e 08 11 31 vmovsh %xmm30,(%r9)"},
{"vmovsh store disp8", "VMOVSH",
[]ExtOperand{ExtXmm(30), ExtMemory(1, 127)},
"62657e0811717f", "62 65 7e 08 11 71 7f vmovsh %xmm30,0xfe(%rcx) (Disp8(7f))"},
{"vmovsh store negative disp32", "VMOVSH",
[]ExtOperand{ExtXmm(30), ExtMemory(13, -200)},
"62457e0811b538ffffff", ""},
{"vmovw load from r9", "VMOVW",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(30)},
"62457d086e31", "62 45 7d 08 6e 31 vmovw (%r9),%xmm30"},
{"vmovw load disp8", "VMOVW",
[]ExtOperand{ExtMemory(1, 127), ExtXmm(30)},
"62657d086e717f", "62 65 7d 08 6e 71 7f vmovw 0xfe(%rcx),%xmm30 (Disp8(7f))"},
{"vmovw store to r9", "VMOVW",
[]ExtOperand{ExtXmm(30), ExtMemory(9, 0)},
"62457d087e31", "62 45 7d 08 7e 31 vmovw %xmm30,(%r9)"},
{"vmovw store disp8", "VMOVW",
[]ExtOperand{ExtXmm(30), ExtMemory(1, 127)},
"62657d087e717f", "62 65 7d 08 7e 71 7f vmovw %xmm30,0xfe(%rcx) (Disp8(7f))"},
{"vaddsh memory source", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005831", "62 45 16 00 58 31 vaddsh (%r9),%xmm29,%xmm30"},
{"vaddsh memory source disp8", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
"6265160058717f", "62 65 16 00 58 71 7f vaddsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
{"vaddsh memory source disp32", "VADDSH",
[]ExtOperand{ExtXmm(29), ExtMemory(2, 8128), ExtXmm(30)},
"6265160058b2c01f0000", ""},
{"vsubsh memory source", "VSUBSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005c31", "62 45 16 00 5c 31 vsubsh (%r9),%xmm29,%xmm30"},
{"vmulsh memory source disp8", "VMULSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
"6265160059717f", "62 65 16 00 59 71 7f vmulsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
{"vdivsh memory source", "VDIVSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005e31", "62 45 16 00 5e 31 vdivsh (%r9),%xmm29,%xmm30"},
{"vminsh memory source disp8", "VMINSH",
[]ExtOperand{ExtXmm(29), ExtMemory(1, 127), ExtXmm(30)},
"626516005d717f", "62 65 16 00 5d 71 7f vminsh 0xfe(%rcx),%xmm29,%xmm30 (Disp8(7f))"},
{"vmaxsh memory source", "VMAXSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005f31", "62 45 16 00 5f 31 vmaxsh (%r9),%xmm29,%xmm30"},
{"vsqrtsh memory source", "VSQRTSH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"624516005131", "62 45 16 00 51 31 vsqrtsh (%r9),%xmm29,%xmm30"},
{"vsqrtsh memory source negative disp8", "VSQRTSH",
[]ExtOperand{ExtXmm(29), ExtMemory(2, -128), ExtXmm(30)},
"62651600517280", "62 65 16 87 51 72 80 vsqrtsh -0x100(%rdx),%xmm29,%xmm30 (Disp8(80); the GNU row adds {k7}{z})"},
// High registers in a 512-bit form exercise the EVEX extension bits:
// with both sources above 15 the B bar and X bar bits clear, while the
// destination zmm23 keeps R bar set in byte one (derived from the
// proven class above).
{"vcvtne2ps2bf16 high registers", "VCVTNE2PS2BF16",
[]ExtOperand{ExtZmm(21), ExtZmm(20), ExtZmm(23)},
"62a2574072fc", ""},
}
// amd64ResolveEntry finds the table entry a golden row exercises: the entry
// is the one that accepts the row's operands, which is what pins the bytes to
// a single template when a mnemonic registers one entry per W bit.
func amd64ResolveEntry(mnem string, ops []ExtOperand) (ExtInstr, bool) {
var first ExtInstr
for _, in := range Extensions(AMD64) {
if in.Name != mnem {
continue
}
if _, err := in.Encode(ops); err == nil {
return in, true
}
if first.Name == "" {
first = in
}
}
if first.Name != "" {
return first, true
}
return ExtInstr{}, false
}
func amd64ExtInstr(t *testing.T, mnem string, class ExtOperandKind, preds ...func(ExtInstr) bool) ExtInstr {
t.Helper()
for _, in := range Extensions(AMD64) {
if in.Name != mnem || amd64LengthClass(in.Bytes) != class {
continue
}
match := true
for _, p := range preds {
if !p(in) {
match = false
}
}
if match {
return in
}
}
t.Fatalf("no extended %s encoding at the %s vector length", mnem, class)
return ExtInstr{}
}
// amd64W1 names the W1 encoding of a mnemonic registered once per W bit.
func amd64W1(in ExtInstr) bool { return in.Bytes[2]&0x80 != 0 }
// withPred adapts an optional row predicate for the variadic lookup.
func withPred(p func(ExtInstr) bool) []func(ExtInstr) bool {
if p == nil {
return nil
}
return []func(ExtInstr) bool{p}
}
func TestAmd64ExtGoldenBytes(t *testing.T) {
for _, tt := range amd64GoldenRows {
in, ok := amd64ResolveEntry(tt.mnem, tt.ops)
if !ok {
t.Errorf("%s: no table entry for %s at the row's vector length", tt.name, tt.mnem)
continue
}
got, err := in.Encode(tt.ops)
if err != nil {
t.Errorf("%s: encode: %v", tt.name, err)
continue
}
if hex.EncodeToString(got) != tt.want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
}
}
}
// TestAmd64ExtTemplateIntegrity checks the metadata contract: every entry
// names its manual reference, summary and feature, and every template carries
// the fixed shape of an EVEX register form with the register-derived bits
// zero, so a slip in the table is an error and not a stray byte.
func TestAmd64ExtTemplateIntegrity(t *testing.T) {
features := map[ExtFeature]bool{
ExtFeatureBF16: true,
ExtFeatureVP2INTERSECT: true,
ExtFeatureFP16: true,
}
for _, in := range Extensions(AMD64) {
if in.Name == "" || in.Summary == "" || in.Ref == "" {
t.Errorf("%+v: name, summary and reference are mandatory", in)
}
if !features[in.Feature] {
t.Errorf("%s: feature %q is not an amd64 extension feature", in.Name, in.Feature)
}
if len(in.Bytes) != 6 {
t.Errorf("%s: the template is %d bytes, want the 6-byte EVEX register form", in.Name, len(in.Bytes))
continue
}
if in.Bytes[0] != 0x62 {
t.Errorf("%s: the template opens with %02x, want the EVEX escape 62", in.Name, in.Bytes[0])
}
if in.Bytes[1]&0xf0 != 0 {
t.Errorf("%s: byte one carries register bits %04b, want them zero", in.Name, in.Bytes[1]>>4)
}
if in.Bytes[2]&0x78 != 0 {
t.Errorf("%s: byte two carries vvvv bits %04b, want them zero", in.Name, in.Bytes[2]>>3&0xf)
}
if in.Bytes[2]&0x04 == 0 {
t.Errorf("%s: byte two lacks the reserved one-bit", in.Name)
}
if in.Bytes[3]&0x9f != 0 {
t.Errorf("%s: byte three carries z, b, V prime or aaa bits, want them zero: %08b", in.Name, in.Bytes[3])
}
if in.Bytes[5]&0x3f != 0 || in.Bytes[5]&0xc0 != 0xc0 {
t.Errorf("%s: byte five is %08b, want mod 11 with the reg and rm fields zero", in.Name, in.Bytes[5])
}
if in.Form.Arity() < 2 || in.Form.Arity() > 4 {
t.Errorf("%s: form %s carries an unusable arity %d", in.Name, in.Form, in.Form.Arity())
}
if in.Mem > in.Form.Arity() {
t.Errorf("%s: Mem names operand %d, outside the form's %d positions", in.Name, in.Mem, in.Form.Arity())
}
}
}
// TestAmd64ExtEveryEntryCarriesGoldenVector pins the measure the layer is
// judged by: every registered entry is covered by at least one golden vector
// whose resolved entry has the very template, so an entry without provenance
// cannot hide.
func TestAmd64ExtEveryEntryCarriesGoldenVector(t *testing.T) {
for _, in := range Extensions(AMD64) {
found := false
for _, tt := range amd64GoldenRows {
cand, ok := amd64ResolveEntry(tt.mnem, tt.ops)
if ok && cand.Name == in.Name && string(cand.Bytes) == string(in.Bytes) {
found = true
}
}
if !found {
t.Errorf("%s (% x) has no golden vector", in.Name, in.Bytes)
}
}
}
func TestAmd64ExtRejects(t *testing.T) {
for _, tt := range []struct {
name string
mnem string
ops []ExtOperand
quote string // a fragment the error carries
}{
{"wrong vector class", "VCVTNE2PS2BF16",
[]ExtOperand{ExtZmm(1), ExtZmm(2), ExtYmm(3)},
"wants a ZMM register"},
{"destination class is the source's on the narrow convert", "VCVTNEPS2BF16",
[]ExtOperand{ExtZmm(1), ExtZmm(2)},
"wants a YMM register"},
{"vector in the mask position", "VP2INTERSECTD",
[]ExtOperand{ExtZmm(1), ExtZmm(2), ExtZmm(3)},
"wants an opmask register"},
{"mask register beyond k7", "VP2INTERSECTD",
[]ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(8)},
"outside 0-7"},
{"vector where the general register belongs", "VCVTSH2SI",
[]ExtOperand{ExtXmm(1), ExtXmm(2)},
"wants a 32-bit general register"},
{"64-bit register on the W0 convert", "VCVTSI2SH",
[]ExtOperand{ExtXmm(29), ExtGpr64(12), ExtXmm(30)},
"wants a 32-bit general register"},
{"general register beyond r15", "VMOVW",
[]ExtOperand{ExtGpr32(16), ExtXmm(30)},
"outside 0-15"},
{"wrong arity", "VP2INTERSECTD",
[]ExtOperand{ExtZmm(2), ExtZmm(1)},
"takes 3 operands"},
{"arm64 arrangement suffix", "VCVTNE2PS2BF16",
[]ExtOperand{{Kind: ExtZMM, Reg: 1, Arr: ExtArrS}, ExtZmm(2), ExtZmm(3)},
"arrangement"},
{"predicate qualifier", "VCVTNEPS2BF16",
[]ExtOperand{{Kind: ExtZMM, Reg: 1, Qual: ExtQualZeroing}, ExtZmm(2)},
"predicate qualifier"},
{"vector where the control byte belongs", "VGETMANTSH",
[]ExtOperand{ExtXmm(28), ExtXmm(29), ExtXmm(30), ExtXmm(31)},
"wants an immediate control byte"},
{"reserved upper nibble on the mantissa control", "VGETMANTSH",
[]ExtOperand{ExtImmediate(0x7b), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
"reserved and must be zero"},
{"control byte under the floor", "VREDUCESH",
[]ExtOperand{ExtImmediate(-1), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
"outside the unsigned byte range"},
{"control byte over the top", "VRNDSCALESH",
[]ExtOperand{ExtImmediate(256), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
"outside the unsigned byte range"},
{"shift on the control byte", "VRNDSCALESH",
[]ExtOperand{ExtShiftedImmediate(0x0b, 8), ExtXmm(29), ExtXmm(28), ExtXmm(30)},
"take none"},
{"vector in the mask position of the compare", "VCMPSH",
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
"wants an opmask register"},
{"mask beyond k7 on the compare", "VCMPSH",
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtXmm(29), ExtMask(8)},
"outside 0-7"},
{"memory in the arithmetic's first source", "VADDSH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(30)},
"wants an XMM register"},
{"memory in the arithmetic's destination", "VADDSH",
[]ExtOperand{ExtXmm(28), ExtXmm(29), ExtMemory(9, 0)},
"wants an XMM register"},
{"memory where the general register belongs", "VCVTSI2SH",
[]ExtOperand{ExtXmm(29), ExtMemory(9, 0), ExtXmm(30)},
"wants a 32-bit general register"},
{"memory as the compare's second source", "VCMPSH",
[]ExtOperand{ExtImmediate(7), ExtXmm(28), ExtMemory(9, 0), ExtMask(5)},
"wants an XMM register"},
{"memory as the intersect source", "VP2INTERSECTD",
[]ExtOperand{ExtZmm(2), ExtMemory(9, 0), ExtMask(0)},
"wants a ZMM register"},
{"memory as the convert's source", "VCVTSS2SH",
[]ExtOperand{ExtXmm(28), ExtMemory(9, 0), ExtXmm(30)},
"wants an XMM register"},
{"memory as the control byte", "VGETMANTSH",
[]ExtOperand{ExtMemory(9, 0), ExtXmm(28), ExtXmm(29), ExtXmm(30)},
"wants an immediate control byte"},
} {
in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops))
_, err := in.Encode(tt.ops)
if err == nil {
t.Errorf("%s: encode succeeded, want an error", tt.name)
continue
}
if !strings.Contains(err.Error(), tt.quote) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
// The W1 convert refuses the 32-bit register the W0 entry takes.
in := amd64ExtInstr(t, "VCVTSH2SI", ExtXMM, amd64W1)
if _, err := in.Encode([]ExtOperand{ExtXmm(30), ExtGpr32(2)}); err == nil {
t.Error("a 32-bit register encoded on the W1 convert, want an error")
} else if !strings.Contains(err.Error(), "wants a 64-bit general register") {
t.Errorf("the W1 error %q does not name the 64-bit class", err)
}
}
// TestAmd64ExtMemoryFormRejects covers the shapes the memory forms refuse:
// a register in the load's memory position, a memory operand in the store's
// register position, and the out-of-range bases and displacements. The
// rows resolve against the load and store entries themselves, which the
// name-and-class lookup cannot pick alone: the register forms of the same
// mnemonics share the class.
func TestAmd64ExtMemoryFormRejects(t *testing.T) {
load := func(in ExtInstr) bool { return in.Form == ExtFormAmdMemVec }
store := func(in ExtInstr) bool { return in.Form == ExtFormAmdVecMem }
for _, tt := range []struct {
name string
mnem string
pick func(ExtInstr) bool
ops []ExtOperand
quote string
}{
{"vector in the load's memory position", "VMOVSH", load,
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
"wants a memory operand"},
{"memory in the store's register position", "VMOVSH", store,
[]ExtOperand{ExtMemory(9, 0), ExtMemory(1, 0)},
"wants an XMM register"},
{"vector in the store's memory position", "VMOVSH", store,
[]ExtOperand{ExtXmm(29), ExtXmm(30)},
"wants a memory operand"},
{"base beyond r15 on the load", "VMOVW", load,
[]ExtOperand{ExtMemory(16, 0), ExtXmm(30)},
"outside 0-15"},
{"displacement past the signed 32-bit range on the store", "VMOVW", store,
[]ExtOperand{ExtXmm(30), ExtMemory(8, 1<<32)},
"outside the signed 32-bit range"},
} {
in := amd64ExtInstr(t, tt.mnem, ExtXMM, tt.pick)
_, err := in.Encode(tt.ops)
if err == nil {
t.Errorf("%s: encode succeeded, want an error", tt.name)
continue
}
if !strings.Contains(err.Error(), tt.quote) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
}
// operandClass names the vector class a row exercises, the key the entry
// lookup resolves with.
func operandClass(t *testing.T, ops []ExtOperand) ExtOperandKind {
t.Helper()
for _, op := range ops {
switch op.Kind {
case ExtXMM, ExtYMM, ExtZMM:
return op.Kind
}
}
t.Fatal("the row carries no vector operand to pick the entry with")
return ExtXMM
}
// TestAmd64ExtImm8Tables pins the imm8 semantics the layer carries as data
// against the SDM tables they are transcribed from: the rounding modes of
// the round control, the sign control of the mantissa extraction and the 32
// comparison predicates, in encoding order.
func TestAmd64ExtImm8Tables(t *testing.T) {
roundModes := [4]string{
"round to nearest (even)",
"round down (toward -infinity)",
"round up (toward +infinity)",
"round toward zero (truncate)",
}
if ExtFP16RoundingModes != roundModes {
t.Errorf("rounding modes %q, want the SDM RC field order", ExtFP16RoundingModes)
}
for i, sign := range ExtFP16GetMantSigns {
switch i {
case 0:
if sign != "the sign of the source" {
t.Errorf("sign control 0b00 = %q, want the source's own sign", sign)
}
case 1:
if sign != "positive" {
t.Errorf("sign control 0b01 = %q, want a forced positive", sign)
}
default:
if sign != "the indefinite NaN when the source is negative" {
t.Errorf("sign control 0b1x = %q, want the indefinite NaN branch", sign)
}
}
}
predicates := map[int]string{
0: "EQ_OQ", 1: "LT_OS", 2: "LE_OS", 3: "UNORD_Q", 4: "NEQ_UQ",
5: "NLT_US", 6: "NLE_US", 7: "ORD_Q", 8: "EQ_UQ", 15: "TRUE_UQ",
16: "EQ_OS", 23: "ORD_S", 24: "EQ_US", 27: "FALSE_OS", 31: "TRUE_US",
}
for i, want := range predicates {
if got := ExtFP16CmpPredicates[i]; got != want {
t.Errorf("predicate 0x%02x = %q, want %q", i, got, want)
}
}
if ExtFP16CmpPredicates[31] != "TRUE_US" {
t.Errorf("the predicate table ends at %q, want TRUE_US", ExtFP16CmpPredicates[31])
}
}
// TestAmd64ExtArchBinding pins the layer's architecture binding: only riscv
// and loong64 have no extended layer, arm64's lives in arm64_ext.go and the
// amd64 one here.
func TestAmd64ExtArchBinding(t *testing.T) {
for _, a := range []Arch{RISCV, LOONG64, Unknown} {
if got := Extensions(a); len(got) != 0 {
t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got))
}
}
if got := Extensions(AMD64); len(got) != 70 {
t.Errorf("the amd64 layer registers %d instructions, want 70", len(got))
}
}