From 28bea9512892be2931b6944f7242f320c149e55e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Tue, 6 Oct 2026 23:43:23 +0200 Subject: [PATCH] feat(arch): add the amd64 extended-instruction layer with BF16 and VP2INTERSECT Assisted-by: GLM 5.3 Flash --- arch/amd64_ext.go | 328 ++++++++++++++++++++++++++++++++++++ arch/amd64_ext_test.go | 242 ++++++++++++++++++++++++++ arch/arm64_ext.go | 83 ++++++++- arch/arm64_ext_test.go | 9 +- asm/extension_amd64_test.go | 122 ++++++++++++++ asm/extension_test.go | 27 ++- 6 files changed, 803 insertions(+), 8 deletions(-) create mode 100644 arch/amd64_ext.go create mode 100644 arch/amd64_ext_test.go create mode 100644 asm/extension_amd64_test.go diff --git a/arch/amd64_ext.go b/arch/amd64_ext.go new file mode 100644 index 0000000..002d589 --- /dev/null +++ b/arch/amd64_ext.go @@ -0,0 +1,328 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// This file carries the amd64 side of the extended-instruction layer: +// instructions the Go toolchain does not know at all, described as data and +// validated against golden vectors from the Intel SDM rather than against the +// toolchain. It sits beside the generated table, never inside it: +// arch/amd64_gen.go stays untouched, and asm.Encodable keeps answering false +// for every mnemonic here, so the layer stays out of the main encoders. +// +// The first families are AVX512-BF16 and AVX512-VP2INTERSECT, in their EVEX +// register forms. The encodings are transcribed from the SDM instruction +// entries and cross-checked against binutils-gdb's assembler testsuite; the +// golden vectors in amd64_ext_test.go pin the bytes. VPOPCNTD and VPOPCNTQ, +// the third family of the 2026-09-19 survey, no longer belong here: the Go +// toolchain's assembler knows them today, they live in the generated table +// and the EVEX encoder, and a mnemonic the toolchain has is not an extension. +// +// Memory operands, write masking ({k1}{z}) and embedded rounding arrive with +// a later slice; every form here encodes the unmasked register forms, which +// is what the golden-vector path exercises. + +package arch + +import "fmt" + +// The features the amd64 layer covers. +const ( + ExtFeatureBF16 ExtFeature = "avx512bf16" + ExtFeatureVP2INTERSECT ExtFeature = "avx512vp2intersect" + ExtFeatureFP16 ExtFeature = "avx512fp16" +) + +// ExtXmm, ExtYmm and ExtZmm build vector operands of the three EVEX register +// widths, VADDPS ZMM1, ZMM2, ZMM3 style. The register number runs 0..31, +// XMM16 and above included: EVEX carries five register bits in every +// position, and the golden vectors exercise the high registers on purpose. +func ExtXmm(reg int) ExtOperand { return ExtOperand{Kind: ExtXMM, Reg: reg} } +func ExtYmm(reg int) ExtOperand { return ExtOperand{Kind: ExtYMM, Reg: reg} } +func ExtZmm(reg int) ExtOperand { return ExtOperand{Kind: ExtZMM, Reg: reg} } + +// ExtMask builds an opmask operand, VP2INTERSECTD K1, ZMM2, ZMM3 style. The +// register number runs 0..7. +func ExtMask(reg int) ExtOperand { return ExtOperand{Kind: ExtKReg, Reg: reg} } + +// ExtGpr32 and ExtGpr64 build general-register operands, VCVTSI2SH XMM1, +// XMM2, EAX style. The register number runs 0..15. +func ExtGpr32(reg int) ExtOperand { return ExtOperand{Kind: ExtR32, Reg: reg} } +func ExtGpr64(reg int) ExtOperand { return ExtOperand{Kind: ExtR64, Reg: reg} } + +// amd64LengthClass reads the vector length the template encodes out of the +// L'L field of the EVEX byte three and names the register class every vector +// operand of that entry must carry. +func amd64LengthClass(b []byte) ExtOperandKind { + switch (b[3] >> 5) & 3 { + case 0: + return ExtXMM + case 1: + return ExtYMM + default: + return ExtZMM + } +} + +// amd64HalfClass names the half-width companion of a vector class, the +// destination class of the narrow conversions. At 128 bits the companion is +// the class itself, which is what the manual gives for the narrowest form. +func amd64HalfClass(k ExtOperandKind) ExtOperandKind { + switch k { + case ExtZMM: + return ExtYMM + case ExtYMM: + return ExtXMM + default: + return ExtXMM + } +} + +// amd64Encode returns the template with the register-derived bits filled in: +// dest and rm are register numbers for the ModR/M reg and r/m fields, vvvv is +// the third-operand register or -1 when the form leaves it unused. The EVEX +// plumbing follows the encoder in asm: reg[3] rides R bar and reg[4] R prime +// bar, rm[3] rides B bar, and in a register form rm[4] rides X bar, while +// vvvv[4] rides V prime bar in byte three. +func amd64Encode(b []byte, dest, vvvv, rm int) []byte { + out := make([]byte, len(b)) + copy(out, b) + rBar, rPrimeBar := 1, 1 + if dest&8 != 0 { + rBar = 0 + } + if dest&16 != 0 { + rPrimeBar = 0 + } + xBar, bBar := 1, 1 + if rm&8 != 0 { + bBar = 0 + } + if rm&16 != 0 { + xBar = 0 + } + out[1] |= byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4) + vBar, vPrimeBar := 15, 1 + if vvvv >= 0 { + vBar = 15 - (vvvv & 15) + if vvvv&16 != 0 { + vPrimeBar = 0 + } + } + out[2] |= byte(vBar << 3) + out[3] |= byte(vPrimeBar << 3) + out[5] |= byte((dest&7)<<3 | rm&7) + return out +} + +// amd64PlainReg checks the invariants every amd64 register operand carries: +// no arm64 arrangement, no predicate qualifier, and a register number inside +// the class the instruction encodes. +func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error { + if op.Arr != ExtArrNone { + return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos) + } + if op.Qual != ExtQualNone { + return fmt.Errorf("%s: operand %d carries a predicate qualifier, the amd64 layer takes none", in.Name, pos) + } + if op.Reg < 0 || op.Reg > max { + return fmt.Errorf("%s: operand %d is register %d, outside 0-%d", in.Name, pos, op.Reg, max) + } + return nil +} + +// amd64Vector checks one vector operand against the class the entry encodes. +func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error { + if op.Kind != class { + return fmt.Errorf("%s: operand %d wants a %s, got %s", in.Name, pos, class, op.Kind) + } + return in.amd64PlainReg(op, 31, pos) +} + +// amd64Gpr checks the general-register operand against the width the entry +// encodes: the W bit picks 32-bit or 64-bit, unless the entry ignores W, and +// the general registers run 0..15. +func (in ExtInstr) amd64Gpr(op ExtOperand, pos int) error { + want := ExtR32 + if in.Bytes[2]&0x80 != 0 { + want = ExtR64 + } + if in.Wig { + if op.Kind != ExtR32 && op.Kind != ExtR64 { + return fmt.Errorf("%s: operand %d wants a 32-bit or 64-bit general register, got %s", in.Name, pos, op.Kind) + } + } else if op.Kind != want { + return fmt.Errorf("%s: operand %d wants a %s, got %s", in.Name, pos, want, op.Kind) + } + return in.amd64PlainReg(op, 15, pos) +} + +// encodeAmd64 encodes the amd64 forms: it validates the operand list against +// the class the template encodes and fills the register bits. An operand the +// form cannot carry is an error, never a silent mis-encoding. +func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) { + switch in.Form { + case ExtFormAmdVec3: + return in.encodeAmdVec3(ops) + case ExtFormAmdVec2, ExtFormAmdVec2Half: + return in.encodeAmdVec2(ops) + case ExtFormAmdMask2: + return in.encodeAmdMask2(ops) + case ExtFormAmdVecGprVec: + return in.encodeAmdVecGprVec(ops) + case ExtFormAmdGprVec, ExtFormAmdVecGpr: + return in.encodeAmdGprPair(ops) + default: + return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form) + } +} + +// encodeAmdVec3 fills the non-destructive three-vector form: src1, src2, +// dest, all under one register class. +func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) { + class := amd64LengthClass(in.Bytes) + for i, op := range ops { + if err := in.amd64Vector(op, class, i+1); err != nil { + return nil, err + } + } + return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil +} + +// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows +// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits +// of destination, and at 128 bits the companion stays the class itself. +func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) { + class := amd64LengthClass(in.Bytes) + destClass := class + if in.Form == ExtFormAmdVec2Half { + destClass = amd64HalfClass(class) + } + if err := in.amd64Vector(ops[0], class, 1); err != nil { + return nil, err + } + if err := in.amd64Vector(ops[1], destClass, 2); err != nil { + return nil, err + } + return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil +} + +// encodeAmdMask2 fills the mask-destination form: src1, src2, dest, where the +// destination is an opmask register and both sources share the class. +func (in ExtInstr) encodeAmdMask2(ops []ExtOperand) ([]byte, error) { + class := amd64LengthClass(in.Bytes) + if err := in.amd64Vector(ops[0], class, 1); err != nil { + return nil, err + } + if err := in.amd64Vector(ops[1], class, 2); err != nil { + return nil, err + } + if ops[2].Kind != ExtKReg { + return nil, fmt.Errorf("%s: operand 3 wants an opmask register, got %s", in.Name, ops[2].Kind) + } + if err := in.amd64PlainReg(ops[2], 7, 3); err != nil { + return nil, err + } + return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil +} + +// encodeAmdVecGprVec fills the conversion form with a general-register +// source: src1, gpr, dest. VCVTSI2SH XMM1, XMM2, EAX style. +func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) { + class := amd64LengthClass(in.Bytes) + if err := in.amd64Vector(ops[0], class, 1); err != nil { + return nil, err + } + if err := in.amd64Gpr(ops[1], 2); err != nil { + return nil, err + } + if err := in.amd64Vector(ops[2], class, 3); err != nil { + return nil, err + } + return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil +} + +// encodeAmdGprPair fills the two-operand general-register forms: gpr, vec +// (the move into a vector register and the integer conversions) and vec, gpr +// (the move out of one). In both orders the second operand is the +// destination in the reg field and the first the r/m source; the vector +// changes position with the form. +func (in ExtInstr) encodeAmdGprPair(ops []ExtOperand) ([]byte, error) { + class := amd64LengthClass(in.Bytes) + vecPos := 1 + if in.Form == ExtFormAmdVecGpr { + vecPos = 0 + } + if err := in.amd64Vector(ops[vecPos], class, vecPos+1); err != nil { + return nil, err + } + if err := in.amd64Gpr(ops[1-vecPos], 2-vecPos); err != nil { + return nil, err + } + return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil +} + +// --- the amd64 AVX512-BF16 and VP2INTERSECT table ----------------------------- + +// amd64Extensions is the extended-instruction layer of amd64. The encodings +// are transcribed from the Intel SDM instruction entries and cross-checked +// against binutils-gdb's assembler testsuite (gas/testsuite/gas/i386/ +// avx512_bf16.d, avx512_bf16_vl.d and x86-64-vp2intersect.d), whose register +// forms the golden vectors in amd64_ext_test.go quote byte for byte. Each +// template carries the fixed bits of one encoding with every register-derived +// bit zero: the map selection in byte one, the W bit, the mandatory prefix +// and the reserved one-bit in byte two, the vector length in byte three, and +// the ModR/M mod bits. +var amd64Extensions = []ExtInstr{ + // AVX512-BF16: the two-way packed single to BF16 conversion and the + // dot product accumulate. The prefixes differ inside the family, the + // three-register convert carries F2 while the narrow convert and the dot + // product carry F3, which the golden vectors pin byte for byte. + {Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating", + Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.512.F2.0F38.W0 72 /r)"}, + {Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating", + Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.256.F2.0F38.W0 72 /r)"}, + {Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating", + Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.128.F2.0F38.W0 72 /r)"}, + {Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination", + Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.512.F3.0F38.W0 72 /r, YMM destination)"}, + {Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination", + Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.256.F3.0F38.W0 72 /r, XMM destination)"}, + {Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination", + Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"}, + {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", + Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"}, + {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", + Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"}, + {Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision", + Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16, + Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"}, + + // AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask + // destination and two vector sources, EVEX.NDS.66.0F38. The instruction + // takes no write mask of its own. + {Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors", + Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT, + Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.512.F2.0F38.W0 68 /r)"}, + {Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors", + Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT, + Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.256.F2.0F38.W0 68 /r)"}, + {Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors", + Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT, + Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W0 68 /r)"}, + {Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors", + Bytes: []byte{0x62, 0x02, 0x87, 0x40, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT, + Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.512.F2.0F38.W1 68 /r)"}, + {Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors", + Bytes: []byte{0x62, 0x02, 0x87, 0x20, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT, + Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.256.F2.0F38.W1 68 /r)"}, + {Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors", + Bytes: []byte{0x62, 0x02, 0x87, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT, + Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W1 68 /r)"}, +} diff --git a/arch/amd64_ext_test.go b/arch/amd64_ext_test.go new file mode 100644 index 0000000..22cca42 --- /dev/null +++ b/arch/amd64_ext_test.go @@ -0,0 +1,242 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package arch + +import ( + "encoding/hex" + "strings" + "testing" +) + +// The BF16 and VP2INTERSECT encodings have no toolchain oracle: go tool asm +// knows neither family. The golden words below are transcribed from the +// Intel SDM instruction entries and cross-checked against binutils-gdb's own +// assembler testsuite: every row marked "GNU" matches a vector in +// gas/testsuite/gas/i386/avx512_bf16.d, avx512_bf16_vl.d or +// x86-64-vp2intersect.d byte for byte, so no entry rests on transcription +// alone. The GNU dumps print AT&T order (sources first, destination last), +// which is the order the operands are built in here too. +func amd64ExtInstr(t *testing.T, mnem string, class ExtOperandKind) ExtInstr { + t.Helper() + for _, in := range Extensions(AMD64) { + if in.Name == mnem && amd64LengthClass(in.Bytes) == class { + return in + } + } + t.Fatalf("no extended %s encoding at the %s vector length", mnem, class) + return ExtInstr{} +} + +func TestAmd64ExtGoldenBytes(t *testing.T) { + for _, tt := range []struct { + name string + mnem string + ops []ExtOperand + want string // hex, little-endian bytes in memory order + GNU string // the matching binutils-gdb line, empty for a derived register form + }{ + // AVX512-BF16, EVEX.NDS.F2.0F38.W0. + {"vcvtne2ps2bf16 zmm", "VCVTNE2PS2BF16", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)}, + "62f2574872f4", "62 f2 57 48 72 f4 vcvtne2ps2bf16 %zmm4,%zmm5,%zmm6"}, + {"vcvtne2ps2bf16 ymm", "VCVTNE2PS2BF16", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f2572872f4", "62 f2 57 28 72 f4 vcvtne2ps2bf16 %ymm4,%ymm5,%ymm6"}, + {"vcvtne2ps2bf16 xmm", "VCVTNE2PS2BF16", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f2570872f4", "62 f2 57 08 72 f4 vcvtne2ps2bf16 %xmm4,%xmm5,%xmm6"}, + {"vcvtneps2bf16 zmm to ymm", "VCVTNEPS2BF16", + []ExtOperand{ExtZmm(5), ExtYmm(6)}, + "62f27e4872f5", "62 f2 7e 48 72 f5 vcvtneps2bf16 %zmm5,%ymm6"}, + {"vcvtneps2bf16 ymm to xmm", "VCVTNEPS2BF16", + []ExtOperand{ExtYmm(5), ExtXmm(6)}, + "62f27e2872f5", "62 f2 7e 28 72 f5 vcvtneps2bf16 %ymm5,%xmm6"}, + {"vcvtneps2bf16 xmm to xmm", "VCVTNEPS2BF16", + []ExtOperand{ExtXmm(5), ExtXmm(6)}, + "62f27e0872f5", "62 f2 7e 08 72 f5 vcvtneps2bf16 %xmm5,%xmm6"}, + {"vdpbf16ps zmm", "VDPBF16PS", + []ExtOperand{ExtZmm(5), ExtZmm(4), ExtZmm(6)}, + "62f2564852f4", "62 f2 56 48 52 f4 vdpbf16ps %zmm4,%zmm5,%zmm6"}, + {"vdpbf16ps ymm", "VDPBF16PS", + []ExtOperand{ExtYmm(5), ExtYmm(4), ExtYmm(6)}, + "62f2562852f4", "62 f2 56 28 52 f4 vdpbf16ps %ymm4,%ymm5,%ymm6"}, + {"vdpbf16ps xmm", "VDPBF16PS", + []ExtOperand{ExtXmm(5), ExtXmm(4), ExtXmm(6)}, + "62f2560852f4", "62 f2 56 08 52 f4 vdpbf16ps %xmm4,%xmm5,%xmm6"}, + + // AVX512-VP2INTERSECT, EVEX.NDS.F2.0F38. The mask destination is + // the ModR/M reg field, so a k register above k7 must refuse. + {"vp2intersectd zmm k0", "VP2INTERSECTD", + []ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)}, + "62f26f4868c1", "62 f2 6f 48 68 c1 vp2intersectd %zmm1,%zmm2,%k0"}, + {"vp2intersectd ymm k2", "VP2INTERSECTD", + []ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)}, + "62f26f2868d1", "62 f2 6f 28 68 d1 vp2intersectd %ymm1,%ymm2,%k2"}, + {"vp2intersectd xmm k4", "VP2INTERSECTD", + []ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)}, + "62f26f0868e1", "62 f2 6f 08 68 e1 vp2intersectd %xmm1,%xmm2,%k4"}, + {"vp2intersectq zmm k0", "VP2INTERSECTQ", + []ExtOperand{ExtZmm(2), ExtZmm(1), ExtMask(0)}, + "62f2ef4868c1", "62 f2 ef 48 68 c1 vp2intersectq %zmm1,%zmm2,%k0"}, + {"vp2intersectq ymm k2", "VP2INTERSECTQ", + []ExtOperand{ExtYmm(2), ExtYmm(1), ExtMask(2)}, + "62f2ef2868d1", "62 f2 ef 28 68 d1 vp2intersectq %ymm1,%ymm2,%k2"}, + {"vp2intersectq xmm k4", "VP2INTERSECTQ", + []ExtOperand{ExtXmm(2), ExtXmm(1), ExtMask(4)}, + "62f2ef0868e1", "62 f2 ef 08 68 e1 vp2intersectq %xmm1,%xmm2,%k4"}, + + // High registers exercise the EVEX extension bits: with both source + // registers above 15 the B bar and X bar bits clear, while the + // destination zmm23 keeps R bar set in byte one (derived from the + // proven class above). + {"vcvtne2ps2bf16 high registers", "VCVTNE2PS2BF16", + []ExtOperand{ExtZmm(21), ExtZmm(20), ExtZmm(23)}, + "62a2574072fc", ""}, + } { + in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) + got, err := in.Encode(tt.ops) + if err != nil { + t.Errorf("%s: encode: %v", tt.name, err) + continue + } + if hex.EncodeToString(got) != tt.want { + t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want) + } + } +} + +// operandClass names the vector class a golden row exercises, the key the +// helper resolves the table entry with. A row without a vector operand +// (none today) would have nowhere to go. +func operandClass(t *testing.T, ops []ExtOperand) ExtOperandKind { + t.Helper() + for _, op := range ops { + switch op.Kind { + case ExtXMM, ExtYMM, ExtZMM: + return op.Kind + } + } + t.Fatal("the golden row carries no vector operand to pick the entry with") + return ExtXMM +} + +// TestAmd64ExtTemplateIntegrity checks the metadata contract: every entry +// names its manual reference, summary and feature, and every template carries +// the fixed shape of an EVEX register form with the register-derived bits +// zero, so a slip in the table is an error and not a stray byte. +func TestAmd64ExtTemplateIntegrity(t *testing.T) { + features := map[ExtFeature]bool{ + ExtFeatureBF16: true, + ExtFeatureVP2INTERSECT: true, + } + for _, in := range Extensions(AMD64) { + if in.Name == "" || in.Summary == "" || in.Ref == "" { + t.Errorf("%+v: name, summary and reference are mandatory", in) + } + if !features[in.Feature] { + t.Errorf("%s: feature %q is not an amd64 extension feature", in.Name, in.Feature) + } + if len(in.Bytes) != 6 { + t.Errorf("%s: the template is %d bytes, want the 6-byte EVEX register form", in.Name, len(in.Bytes)) + continue + } + if in.Bytes[0] != 0x62 { + t.Errorf("%s: the template opens with %02x, want the EVEX escape 62", in.Name, in.Bytes[0]) + } + if in.Bytes[1]&0xf0 != 0 { + t.Errorf("%s: byte one carries register bits %04b, want them zero", in.Name, in.Bytes[1]>>4) + } + if in.Bytes[2]&0x78 != 0 { + t.Errorf("%s: byte two carries vvvv bits %04b, want them zero", in.Name, in.Bytes[2]>>3&0xf) + } + if in.Bytes[2]&0x04 == 0 { + t.Errorf("%s: byte two lacks the reserved one-bit", in.Name) + } + if in.Bytes[3]&0x9f != 0 { + t.Errorf("%s: byte three carries z, b, V prime or aaa bits, want them zero: %08b", in.Name, in.Bytes[3]) + } + if in.Bytes[5]&0x3f != 0 || in.Bytes[5]&0xc0 != 0xc0 { + t.Errorf("%s: byte five is %08b, want mod 11 with the reg and rm fields zero", in.Name, in.Bytes[5]) + } + if in.Form.Arity() < 2 || in.Form.Arity() > 3 { + t.Errorf("%s: form %s carries an unusable arity %d", in.Name, in.Form, in.Form.Arity()) + } + } +} + +// TestAmd64ExtEveryEntryCarriesGoldenVector pins the measure the layer is +// judged by: every registered entry is covered by at least one golden vector +// in the byte test above, so an entry without provenance cannot hide. +func TestAmd64ExtEveryEntryCarriesGoldenVector(t *testing.T) { + covered := map[string]bool{} + for _, in := range Extensions(AMD64) { + covered[in.Name+"|"+string(amd64LengthClass(in.Bytes))] = false + } + for _, tt := range []struct { + mnem string + class ExtOperandKind + }{ + {"VCVTNE2PS2BF16", ExtZMM}, {"VCVTNE2PS2BF16", ExtYMM}, {"VCVTNE2PS2BF16", ExtXMM}, + {"VCVTNEPS2BF16", ExtZMM}, {"VCVTNEPS2BF16", ExtYMM}, {"VCVTNEPS2BF16", ExtXMM}, + {"VDPBF16PS", ExtZMM}, {"VDPBF16PS", ExtYMM}, {"VDPBF16PS", ExtXMM}, + {"VP2INTERSECTD", ExtZMM}, {"VP2INTERSECTD", ExtYMM}, {"VP2INTERSECTD", ExtXMM}, + {"VP2INTERSECTQ", ExtZMM}, {"VP2INTERSECTQ", ExtYMM}, {"VP2INTERSECTQ", ExtXMM}, + } { + key := tt.mnem + "|" + string(tt.class) + if _, ok := covered[key]; !ok { + t.Errorf("the golden list covers %s, but the table registers no such entry", key) + continue + } + covered[key] = true + } + for key, ok := range covered { + if !ok { + t.Errorf("%s has no golden vector", key) + } + } +} + +func TestAmd64ExtRejects(t *testing.T) { + for _, tt := range []struct { + name string + mnem string + ops []ExtOperand + quote string // a fragment the error carries + }{ + {"wrong vector class", "VCVTNE2PS2BF16", + []ExtOperand{ExtZmm(1), ExtZmm(2), ExtYmm(3)}, + "wants a ZMM register"}, + {"destination class is the source's on the narrow convert", "VCVTNEPS2BF16", + []ExtOperand{ExtZmm(1), ExtZmm(2)}, + "wants a YMM register"}, + {"vector in the mask position", "VP2INTERSECTD", + []ExtOperand{ExtZmm(1), ExtZmm(2), ExtZmm(3)}, + "wants an opmask register"}, + {"mask register beyond k7", "VP2INTERSECTD", + []ExtOperand{ExtZmm(1), ExtZmm(2), ExtMask(8)}, + "outside 0-7"}, + {"vector where the general register belongs", "VCVTNE2PS2BF16", + []ExtOperand{ExtGpr32(0), ExtZmm(2), ExtZmm(3)}, + "wants a ZMM register"}, + {"wrong arity", "VP2INTERSECTD", + []ExtOperand{ExtZmm(1), ExtZmm(2)}, + "takes 3 operands"}, + {"arm64 arrangement suffix", "VCVTNE2PS2BF16", + []ExtOperand{{Kind: ExtZMM, Reg: 1, Arr: ExtArrS}, ExtZmm(2), ExtZmm(3)}, + "arrangement"}, + {"predicate qualifier", "VCVTNEPS2BF16", + []ExtOperand{{Kind: ExtZMM, Reg: 1, Qual: ExtQualZeroing}, ExtZmm(2)}, + "predicate qualifier"}, + } { + in := amd64ExtInstr(t, tt.mnem, operandClass(t, tt.ops)) + _, err := in.Encode(tt.ops) + if err == nil { + t.Errorf("%s: encode succeeded, want an error", tt.name) + continue + } + if !strings.Contains(err.Error(), tt.quote) { + t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote) + } + } +} diff --git a/arch/arm64_ext.go b/arch/arm64_ext.go index fdeddc6..262ea3d 100644 --- a/arch/arm64_ext.go +++ b/arch/arm64_ext.go @@ -6,7 +6,8 @@ // golden vectors from the Arm Architecture Reference Manual rather than // against the toolchain. It sits beside the generated tables, never inside // them: arch/arm64_gen.go stays untouched, and Extensions returns the layer -// per architecture so a later amd64 table attaches through the same door. +// per architecture; the amd64 side of the layer lives in amd64_ext.go and +// attaches through the same door. // // The first entry is the arm64 SVE and SVE2 integer add/subtract/multiply // family (twenty-three forms over four word shapes). The encodings are @@ -26,6 +27,13 @@ const ( ExtZReg ExtOperandKind = iota // scalable vector register Z0-Z31 ExtPReg // predicate register P0-P15 ExtImm // immediate + // The amd64 layer's register kinds. + ExtXMM // 128-bit vector register XMM0-XMM31 + ExtYMM // 256-bit vector register YMM0-YMM31 + ExtZMM // 512-bit vector register ZMM0-ZMM31 + ExtKReg // opmask register K0-K7 + ExtR32 // 32-bit general register EAX-R15D + ExtR64 // 64-bit general register RAX-R15 ) // String returns a short label for the kind. @@ -37,6 +45,18 @@ func (k ExtOperandKind) String() string { return "predicate register" case ExtImm: return "immediate" + case ExtXMM: + return "XMM register" + case ExtYMM: + return "YMM register" + case ExtZMM: + return "ZMM register" + case ExtKReg: + return "opmask register" + case ExtR32: + return "32-bit general register" + case ExtR64: + return "64-bit general register" default: return "operand" } @@ -225,6 +245,33 @@ const ( // multiply (immediate) class: MUL $-128, Z0.B computes Z0 = Z0 * -128. // Operands: simm8, Zdn. No shift exists in this class. ExtFormSignedImmediate + // The amd64 layer's operand shapes, sources first, destination last. + // The vector register class an entry takes comes from the encoding + // template (the L'L field names it), not from the form. + // ExtFormAmdVec3 is the EVEX non-destructive three-vector form: + // VCVTNE2PS2BF16 Z6, Z5, Z4. Operands: src1, src2, dest. + ExtFormAmdVec3 + // ExtFormAmdVec2 is the two-vector form: VPOPCNTD Z1, Z2, VCOMISH X1, X2. + // Operands: src, dest. + ExtFormAmdVec2 + // ExtFormAmdVec2Half is the two-vector form whose destination is the + // half-width companion of the source, equal at 128 bits: VCVTNEPS2BF16 + // Y6, Z5. Operands: src, dest. + ExtFormAmdVec2Half + // ExtFormAmdMask2 is the form with an opmask destination and two vector + // sources: VP2INTERSECTD K0, Z2, Z1. Operands: src1, src2, dest. + ExtFormAmdMask2 + // ExtFormAmdVecGprVec is the three-operand conversion with a + // general-register source: VCVTSI2SH X1, X2, EAX. Operands: src1, gpr, + // dest. + ExtFormAmdVecGprVec + // ExtFormAmdGprVec is the two-operand form with a general-register + // source and a vector destination: VMOVW X1, EAX and VCVTSH2SI EAX, X1. + // Operands: gpr, dest. + ExtFormAmdGprVec + // ExtFormAmdVecGpr is the two-operand form with a vector source and a + // general-register destination: VMOVW EAX, X1. Operands: src, dest. + ExtFormAmdVecGpr ) // Arity returns the operand count the form takes. @@ -234,6 +281,10 @@ func (f ExtForm) Arity() int { return 3 case ExtFormImmediate, ExtFormSignedImmediate: return 2 + case ExtFormAmdVec3, ExtFormAmdMask2, ExtFormAmdVecGprVec: + return 3 + case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdGprVec, ExtFormAmdVecGpr: + return 2 default: return 0 } @@ -250,6 +301,9 @@ func (f ExtForm) Kinds() []ExtOperandKind { return []ExtOperandKind{ExtZReg, ExtPReg, ExtZReg} case ExtFormImmediate, ExtFormSignedImmediate: return []ExtOperandKind{ExtImm, ExtZReg} + // The amd64 forms return no kinds: the exact vector class depends on the + // entry's encoding template (its L'L field), which the form alone cannot + // name, and the encode paths diagnose the class themselves. default: return nil } @@ -266,6 +320,20 @@ func (f ExtForm) String() string { return "unsigned immediate" case ExtFormSignedImmediate: return "signed immediate" + case ExtFormAmdVec3: + return "three vectors" + case ExtFormAmdVec2: + return "two vectors" + case ExtFormAmdVec2Half: + return "two vectors, half-width destination" + case ExtFormAmdMask2: + return "two vectors into an opmask" + case ExtFormAmdVecGprVec: + return "vector, general register, vector" + case ExtFormAmdGprVec: + return "general register, vector" + case ExtFormAmdVecGpr: + return "vector, general register" default: return "unknown form" } @@ -296,6 +364,14 @@ type ExtInstr struct { Size ExtField // element-size field the arrangement fills Feature ExtFeature // sve or sve2 Ref string // the ARM ARM entry the encoding comes from + // Bytes is the amd64 encoding template: the EVEX prefix, opcode and + // ModR/M byte of one form, with every register-derived bit zero. The + // vector length and the general-register width an entry encodes are read + // back out of it at encode time. + Bytes []byte + // Wig records that the entry ignores the W bit in its general-register + // position, so both 32-bit and 64-bit registers encode. + Wig bool } // Encode assembles the operands into the 4 little-endian bytes of the @@ -316,6 +392,9 @@ func (in ExtInstr) Encode(ops []ExtOperand) ([]byte, error) { return in.encodeImmediate(ops) case ExtFormSignedImmediate: return in.encodeSignedImmediate(ops) + case ExtFormAmdVec3, ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdMask2, + ExtFormAmdVecGprVec, ExtFormAmdGprVec, ExtFormAmdVecGpr: + return in.encodeAmd64(ops) default: return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form) } @@ -616,6 +695,8 @@ var arm64Extensions = []ExtInstr{ // it simply offers no instruction where none is registered. func Extensions(a Arch) []ExtInstr { switch a { + case AMD64: + return amd64Extensions case ARM64: return arm64Extensions default: diff --git a/arch/arm64_ext_test.go b/arch/arm64_ext_test.go index 9cfc9d0..6ae1a73 100644 --- a/arch/arm64_ext_test.go +++ b/arch/arm64_ext_test.go @@ -386,10 +386,10 @@ func TestArm64ExtRejects(t *testing.T) { } // TestExtensionsArchBinding pins the registry's architecture binding: the -// extended layer exists for arm64 alone until an amd64 table attaches, and no -// other architecture sees a single SVE instruction. +// extended layer exists for arm64 and amd64 (the latter in amd64_ext.go), and +// no other architecture sees a single instruction of either. func TestExtensionsArchBinding(t *testing.T) { - for _, a := range []Arch{AMD64, RISCV, LOONG64, Unknown} { + for _, a := range []Arch{RISCV, LOONG64, Unknown} { if got := Extensions(a); len(got) != 0 { t.Errorf("Extensions(%s) carries %d instructions, want none", a, len(got)) } @@ -397,4 +397,7 @@ func TestExtensionsArchBinding(t *testing.T) { if got := Extensions(ARM64); len(got) == 0 { t.Error("Extensions(ARM64) is empty") } + if got := Extensions(AMD64); len(got) == 0 { + t.Error("Extensions(AMD64) is empty") + } } diff --git a/asm/extension_amd64_test.go b/asm/extension_amd64_test.go new file mode 100644 index 0000000..e2ba5df --- /dev/null +++ b/asm/extension_amd64_test.go @@ -0,0 +1,122 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "encoding/hex" + "strings" + "testing" + + "sourcedock.dev/petrbalvin/gasm-sdk/arch" +) + +// TestAmd64ExtensionRegistry checks the mnemonic lookup for the amd64 layer: +// one mnemonic across several vector lengths resolves to every entry, the +// lookup is case-insensitive, and the counts match the registered families. +func TestAmd64ExtensionRegistry(t *testing.T) { + for _, tt := range []struct { + mnem string + forms int + }{ + {"VCVTNE2PS2BF16", 3}, + {"VCVTNEPS2BF16", 3}, + {"VDPBF16PS", 3}, + {"VP2INTERSECTD", 3}, + {"VP2INTERSECTQ", 3}, + } { + cands, ok := LookupExtension(arch.AMD64, tt.mnem) + if !ok { + t.Fatalf("LookupExtension(AMD64, %s) found nothing", tt.mnem) + } + if len(cands) != tt.forms { + t.Errorf("%s registers %d forms, want %d", tt.mnem, len(cands), tt.forms) + } + lower, ok := LookupExtension(arch.AMD64, strings.ToLower(tt.mnem)) + if !ok || len(lower) != tt.forms { + t.Errorf("the %s lookup is not case-insensitive", tt.mnem) + } + } + if got := arch.Extensions(arch.AMD64); len(got) != 15 { + t.Errorf("the amd64 layer registers %d instructions, want 15", len(got)) + } + if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok { + t.Error("a non-extended mnemonic resolved") + } + // VPOPCNTD and VPOPCNTQ are toolchain instructions today: they stay in + // the generated table and out of the extension layer. + if _, ok := LookupExtension(arch.AMD64, "VPOPCNTD"); ok { + t.Error("VPOPCNTD is an extension, want it in the generated table alone") + } +} + +// TestAmd64ExtensionAboveGeneratedTable pins the layering twice over: no +// registered mnemonic sits in the generated amd64 table, and the encoder +// mirror asm.Encodable answers false for every one of them, so the layer +// stays out of the main encoders by test and not by promise. +func TestAmd64ExtensionAboveGeneratedTable(t *testing.T) { + for _, mnem := range ExtensionNames(arch.AMD64) { + if _, found := arch.ForArch(arch.AMD64).Lookup(mnem); found { + t.Errorf("%s leaked into the generated amd64 table", mnem) + } + if Encodable(mnem) { + t.Errorf("%s is encodable through the main encoder, the layer is not sealed", mnem) + } + } +} + +// TestEncodeExtensionAmd64 encodes through the registry and pins the same +// golden words the arch table tests pin, proving the registry resolves to the +// right encoding. +func TestEncodeExtensionAmd64(t *testing.T) { + for _, tt := range []struct { + name string + mnem string + ops []arch.ExtOperand + want string + }{ + {"bf16 convert", "VCVTNE2PS2BF16", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtZmm(6)}, + "62f2574872f4"}, + {"bf16 narrow convert", "VCVTNEPS2BF16", + []arch.ExtOperand{arch.ExtZmm(5), arch.ExtYmm(6)}, + "62f27e4872f5"}, + {"dot product", "VDPBF16PS", + []arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtXmm(6)}, + "62f2560852f4"}, + {"intersect into a mask", "VP2INTERSECTD", + []arch.ExtOperand{arch.ExtYmm(2), arch.ExtYmm(1), arch.ExtMask(2)}, + "62f26f2868d1"}, + } { + got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...) + if err != nil { + t.Errorf("%s: encode: %v", tt.name, err) + continue + } + if hex.EncodeToString(got) != tt.want { + t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want) + } + } +} + +// TestEncodeExtensionAmd64Errors checks the registry's diagnostics on the +// amd64 side: a wrong arity names the form's count and a mis-classed operand +// surfaces the entry's own message. +func TestEncodeExtensionAmd64Errors(t *testing.T) { + if _, err := EncodeExtension(arch.AMD64, "VP2INTERSECTD", arch.ExtZmm(1)); err == nil { + t.Error("one operand encoded, want an arity error") + } else if !strings.Contains(err.Error(), "3 operands") { + t.Errorf("arity error %q does not name the count", err) + } + _, err := EncodeExtension(arch.AMD64, "VCVTNEPS2BF16", arch.ExtZmm(1), arch.ExtZmm(2)) + if err == nil { + t.Fatal("a ZMM destination encoded on the narrow convert, want an error") + } + if !strings.Contains(err.Error(), "YMM register") { + t.Errorf("error %q does not name the YMM destination", err) + } + if _, err := EncodeExtension(arch.AMD64, "VCVTNE2PS2BF16"); err == nil || + !strings.Contains(err.Error(), "takes 3 operands, got 0") { + t.Errorf("zero-operand error = %v, want the operand-count diagnostic", err) + } +} diff --git a/asm/extension_test.go b/asm/extension_test.go index 15dc2b7..ee19de3 100644 --- a/asm/extension_test.go +++ b/asm/extension_test.go @@ -152,14 +152,15 @@ func TestExtensionEncodable(t *testing.T) { } } -// TestExtensionArchIsolation is the architecture-binding negative case: the -// extension layer is registered for arm64 alone, and no other architecture -// answers its queries, not even for a mnemonic the amd64 base table carries. +// TestExtensionArchIsolation is the architecture-binding negative case: no +// architecture answers the arm64 mnemonics but arm64, and the amd64 layer +// answers nothing of the arm64 family either (its own mnemonics live in +// extension_amd64_test.go). func TestExtensionArchIsolation(t *testing.T) { ops := []arch.ExtOperand{ arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), } - for _, a := range []arch.Arch{arch.AMD64, arch.RISCV, arch.LOONG64, arch.Unknown} { + for _, a := range []arch.Arch{arch.RISCV, arch.LOONG64, arch.Unknown} { if cands, ok := LookupExtension(a, "ADD"); ok || cands != nil { t.Errorf("LookupExtension(%s, ADD) offered %d candidates", a, len(cands)) } @@ -181,6 +182,24 @@ func TestExtensionArchIsolation(t *testing.T) { t.Errorf("arch.Extensions(%s) carries %d instructions", a, len(got)) } } + // The amd64 layer exists but stays silent about the arm64 family. + if cands, ok := LookupExtension(arch.AMD64, "ADD"); ok || cands != nil { + t.Errorf("LookupExtension(AMD64, ADD) offered %d candidates", len(cands)) + } + if got, err := EncodeExtension(arch.AMD64, "ADD", ops...); err == nil { + t.Errorf("EncodeExtension(AMD64, ADD) encoded %x, want a refusal", got) + } else if !strings.Contains(err.Error(), string(arch.AMD64)) { + t.Errorf("EncodeExtension(AMD64) error %q does not name the architecture", err) + } + if ExtensionEncodable(arch.AMD64, "ADD", ops...) { + t.Error("ExtensionEncodable(AMD64, ADD) reported true") + } + if names := ExtensionNames(arch.AMD64); len(names) == 0 { + t.Error("the amd64 layer registers no names") + } + if got := arch.Extensions(arch.AMD64); len(got) == 0 { + t.Error("arch.Extensions(AMD64) is empty") + } } // TestExtensionNamesARM64 checks the completion-facing name list: every