feat(arch): add the amd64 extended-instruction layer with BF16 and VP2INTERSECT
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
2d803e38d8
commit
28bea95128
6 files changed
+803
-8
No files matched your search
@@ -0,0 +1,328 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// This file carries the amd64 side of the extended-instruction layer:
|
||||
// instructions the Go toolchain does not know at all, described as data and
|
||||
// validated against golden vectors from the Intel SDM rather than against the
|
||||
// toolchain. It sits beside the generated table, never inside it:
|
||||
// arch/amd64_gen.go stays untouched, and asm.Encodable keeps answering false
|
||||
// for every mnemonic here, so the layer stays out of the main encoders.
|
||||
//
|
||||
// The first families are AVX512-BF16 and AVX512-VP2INTERSECT, in their EVEX
|
||||
// register forms. The encodings are transcribed from the SDM instruction
|
||||
// entries and cross-checked against binutils-gdb's assembler testsuite; the
|
||||
// golden vectors in amd64_ext_test.go pin the bytes. VPOPCNTD and VPOPCNTQ,
|
||||
// the third family of the 2026-09-19 survey, no longer belong here: the Go
|
||||
// toolchain's assembler knows them today, they live in the generated table
|
||||
// and the EVEX encoder, and a mnemonic the toolchain has is not an extension.
|
||||
//
|
||||
// Memory operands, write masking ({k1}{z}) and embedded rounding arrive with
|
||||
// a later slice; every form here encodes the unmasked register forms, which
|
||||
// is what the golden-vector path exercises.
|
||||
|
||||
package arch
|
||||
|
||||
import "fmt"
|
||||
|
||||
// The features the amd64 layer covers.
|
||||
const (
|
||||
ExtFeatureBF16 ExtFeature = "avx512bf16"
|
||||
ExtFeatureVP2INTERSECT ExtFeature = "avx512vp2intersect"
|
||||
ExtFeatureFP16 ExtFeature = "avx512fp16"
|
||||
)
|
||||
|
||||
// ExtXmm, ExtYmm and ExtZmm build vector operands of the three EVEX register
|
||||
// widths, VADDPS ZMM1, ZMM2, ZMM3 style. The register number runs 0..31,
|
||||
// XMM16 and above included: EVEX carries five register bits in every
|
||||
// position, and the golden vectors exercise the high registers on purpose.
|
||||
func ExtXmm(reg int) ExtOperand { return ExtOperand{Kind: ExtXMM, Reg: reg} }
|
||||
func ExtYmm(reg int) ExtOperand { return ExtOperand{Kind: ExtYMM, Reg: reg} }
|
||||
func ExtZmm(reg int) ExtOperand { return ExtOperand{Kind: ExtZMM, Reg: reg} }
|
||||
|
||||
// ExtMask builds an opmask operand, VP2INTERSECTD K1, ZMM2, ZMM3 style. The
|
||||
// register number runs 0..7.
|
||||
func ExtMask(reg int) ExtOperand { return ExtOperand{Kind: ExtKReg, Reg: reg} }
|
||||
|
||||
// ExtGpr32 and ExtGpr64 build general-register operands, VCVTSI2SH XMM1,
|
||||
// XMM2, EAX style. The register number runs 0..15.
|
||||
func ExtGpr32(reg int) ExtOperand { return ExtOperand{Kind: ExtR32, Reg: reg} }
|
||||
func ExtGpr64(reg int) ExtOperand { return ExtOperand{Kind: ExtR64, Reg: reg} }
|
||||
|
||||
// amd64LengthClass reads the vector length the template encodes out of the
|
||||
// L'L field of the EVEX byte three and names the register class every vector
|
||||
// operand of that entry must carry.
|
||||
func amd64LengthClass(b []byte) ExtOperandKind {
|
||||
switch (b[3] >> 5) & 3 {
|
||||
case 0:
|
||||
return ExtXMM
|
||||
case 1:
|
||||
return ExtYMM
|
||||
default:
|
||||
return ExtZMM
|
||||
}
|
||||
}
|
||||
|
||||
// amd64HalfClass names the half-width companion of a vector class, the
|
||||
// destination class of the narrow conversions. At 128 bits the companion is
|
||||
// the class itself, which is what the manual gives for the narrowest form.
|
||||
func amd64HalfClass(k ExtOperandKind) ExtOperandKind {
|
||||
switch k {
|
||||
case ExtZMM:
|
||||
return ExtYMM
|
||||
case ExtYMM:
|
||||
return ExtXMM
|
||||
default:
|
||||
return ExtXMM
|
||||
}
|
||||
}
|
||||
|
||||
// amd64Encode returns the template with the register-derived bits filled in:
|
||||
// dest and rm are register numbers for the ModR/M reg and r/m fields, vvvv is
|
||||
// the third-operand register or -1 when the form leaves it unused. The EVEX
|
||||
// plumbing follows the encoder in asm: reg[3] rides R bar and reg[4] R prime
|
||||
// bar, rm[3] rides B bar, and in a register form rm[4] rides X bar, while
|
||||
// vvvv[4] rides V prime bar in byte three.
|
||||
func amd64Encode(b []byte, dest, vvvv, rm int) []byte {
|
||||
out := make([]byte, len(b))
|
||||
copy(out, b)
|
||||
rBar, rPrimeBar := 1, 1
|
||||
if dest&8 != 0 {
|
||||
rBar = 0
|
||||
}
|
||||
if dest&16 != 0 {
|
||||
rPrimeBar = 0
|
||||
}
|
||||
xBar, bBar := 1, 1
|
||||
if rm&8 != 0 {
|
||||
bBar = 0
|
||||
}
|
||||
if rm&16 != 0 {
|
||||
xBar = 0
|
||||
}
|
||||
out[1] |= byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4)
|
||||
vBar, vPrimeBar := 15, 1
|
||||
if vvvv >= 0 {
|
||||
vBar = 15 - (vvvv & 15)
|
||||
if vvvv&16 != 0 {
|
||||
vPrimeBar = 0
|
||||
}
|
||||
}
|
||||
out[2] |= byte(vBar << 3)
|
||||
out[3] |= byte(vPrimeBar << 3)
|
||||
out[5] |= byte((dest&7)<<3 | rm&7)
|
||||
return out
|
||||
}
|
||||
|
||||
// amd64PlainReg checks the invariants every amd64 register operand carries:
|
||||
// no arm64 arrangement, no predicate qualifier, and a register number inside
|
||||
// the class the instruction encodes.
|
||||
func (in ExtInstr) amd64PlainReg(op ExtOperand, max, pos int) error {
|
||||
if op.Arr != ExtArrNone {
|
||||
return fmt.Errorf("%s: operand %d carries an arrangement suffix, the amd64 layer takes none", in.Name, pos)
|
||||
}
|
||||
if op.Qual != ExtQualNone {
|
||||
return fmt.Errorf("%s: operand %d carries a predicate qualifier, the amd64 layer takes none", in.Name, pos)
|
||||
}
|
||||
if op.Reg < 0 || op.Reg > max {
|
||||
return fmt.Errorf("%s: operand %d is register %d, outside 0-%d", in.Name, pos, op.Reg, max)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// amd64Vector checks one vector operand against the class the entry encodes.
|
||||
func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) error {
|
||||
if op.Kind != class {
|
||||
return fmt.Errorf("%s: operand %d wants a %s, got %s", in.Name, pos, class, op.Kind)
|
||||
}
|
||||
return in.amd64PlainReg(op, 31, pos)
|
||||
}
|
||||
|
||||
// amd64Gpr checks the general-register operand against the width the entry
|
||||
// encodes: the W bit picks 32-bit or 64-bit, unless the entry ignores W, and
|
||||
// the general registers run 0..15.
|
||||
func (in ExtInstr) amd64Gpr(op ExtOperand, pos int) error {
|
||||
want := ExtR32
|
||||
if in.Bytes[2]&0x80 != 0 {
|
||||
want = ExtR64
|
||||
}
|
||||
if in.Wig {
|
||||
if op.Kind != ExtR32 && op.Kind != ExtR64 {
|
||||
return fmt.Errorf("%s: operand %d wants a 32-bit or 64-bit general register, got %s", in.Name, pos, op.Kind)
|
||||
}
|
||||
} else if op.Kind != want {
|
||||
return fmt.Errorf("%s: operand %d wants a %s, got %s", in.Name, pos, want, op.Kind)
|
||||
}
|
||||
return in.amd64PlainReg(op, 15, pos)
|
||||
}
|
||||
|
||||
// encodeAmd64 encodes the amd64 forms: it validates the operand list against
|
||||
// the class the template encodes and fills the register bits. An operand the
|
||||
// form cannot carry is an error, never a silent mis-encoding.
|
||||
func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
||||
switch in.Form {
|
||||
case ExtFormAmdVec3:
|
||||
return in.encodeAmdVec3(ops)
|
||||
case ExtFormAmdVec2, ExtFormAmdVec2Half:
|
||||
return in.encodeAmdVec2(ops)
|
||||
case ExtFormAmdMask2:
|
||||
return in.encodeAmdMask2(ops)
|
||||
case ExtFormAmdVecGprVec:
|
||||
return in.encodeAmdVecGprVec(ops)
|
||||
case ExtFormAmdGprVec, ExtFormAmdVecGpr:
|
||||
return in.encodeAmdGprPair(ops)
|
||||
default:
|
||||
return nil, fmt.Errorf("%s: unknown form %d", in.Name, in.Form)
|
||||
}
|
||||
}
|
||||
|
||||
// encodeAmdVec3 fills the non-destructive three-vector form: src1, src2,
|
||||
// dest, all under one register class.
|
||||
func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
for i, op := range ops {
|
||||
if err := in.amd64Vector(op, class, i+1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
|
||||
}
|
||||
|
||||
// encodeAmdVec2 fills the two-vector form: src, dest. The half form narrows
|
||||
// the destination: VCVTNEPS2BF16 converts 512 bits of source into 256 bits
|
||||
// of destination, and at 128 bits the companion stays the class itself.
|
||||
func (in ExtInstr) encodeAmdVec2(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
destClass := class
|
||||
if in.Form == ExtFormAmdVec2Half {
|
||||
destClass = amd64HalfClass(class)
|
||||
}
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[1], destClass, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil
|
||||
}
|
||||
|
||||
// encodeAmdMask2 fills the mask-destination form: src1, src2, dest, where the
|
||||
// destination is an opmask register and both sources share the class.
|
||||
func (in ExtInstr) encodeAmdMask2(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[1], class, 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if ops[2].Kind != ExtKReg {
|
||||
return nil, fmt.Errorf("%s: operand 3 wants an opmask register, got %s", in.Name, ops[2].Kind)
|
||||
}
|
||||
if err := in.amd64PlainReg(ops[2], 7, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
|
||||
}
|
||||
|
||||
// encodeAmdVecGprVec fills the conversion form with a general-register
|
||||
// source: src1, gpr, dest. VCVTSI2SH XMM1, XMM2, EAX style.
|
||||
func (in ExtInstr) encodeAmdVecGprVec(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
if err := in.amd64Vector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Gpr(ops[1], 2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Vector(ops[2], class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64Encode(in.Bytes, ops[2].Reg, ops[0].Reg, ops[1].Reg), nil
|
||||
}
|
||||
|
||||
// encodeAmdGprPair fills the two-operand general-register forms: gpr, vec
|
||||
// (the move into a vector register and the integer conversions) and vec, gpr
|
||||
// (the move out of one). In both orders the second operand is the
|
||||
// destination in the reg field and the first the r/m source; the vector
|
||||
// changes position with the form.
|
||||
func (in ExtInstr) encodeAmdGprPair(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64LengthClass(in.Bytes)
|
||||
vecPos := 1
|
||||
if in.Form == ExtFormAmdVecGpr {
|
||||
vecPos = 0
|
||||
}
|
||||
if err := in.amd64Vector(ops[vecPos], class, vecPos+1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64Gpr(ops[1-vecPos], 2-vecPos); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64Encode(in.Bytes, ops[1].Reg, -1, ops[0].Reg), nil
|
||||
}
|
||||
|
||||
// --- the amd64 AVX512-BF16 and VP2INTERSECT table -----------------------------
|
||||
|
||||
// amd64Extensions is the extended-instruction layer of amd64. The encodings
|
||||
// are transcribed from the Intel SDM instruction entries and cross-checked
|
||||
// against binutils-gdb's assembler testsuite (gas/testsuite/gas/i386/
|
||||
// avx512_bf16.d, avx512_bf16_vl.d and x86-64-vp2intersect.d), whose register
|
||||
// forms the golden vectors in amd64_ext_test.go quote byte for byte. Each
|
||||
// template carries the fixed bits of one encoding with every register-derived
|
||||
// bit zero: the map selection in byte one, the W bit, the mandatory prefix
|
||||
// and the reserved one-bit in byte two, the vector length in byte three, and
|
||||
// the ModR/M mod bits.
|
||||
var amd64Extensions = []ExtInstr{
|
||||
// AVX512-BF16: the two-way packed single to BF16 conversion and the
|
||||
// dot product accumulate. The prefixes differ inside the family, the
|
||||
// three-register convert carries F2 while the narrow convert and the dot
|
||||
// product carry F3, which the golden vectors pin byte for byte.
|
||||
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.512.F2.0F38.W0 72 /r)"},
|
||||
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.256.F2.0F38.W0 72 /r)"},
|
||||
{Name: "VCVTNE2PS2BF16", Summary: "Convert two packed single-precision vectors to packed BF16, truncating",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNE2PS2BF16 (EVEX.NDS.128.F2.0F38.W0 72 /r)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.512.F3.0F38.W0 72 /r, YMM destination)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.256.F3.0F38.W0 72 /r, XMM destination)"},
|
||||
{Name: "VCVTNEPS2BF16", Summary: "Convert packed single precision to packed BF16, truncating, half-width destination",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x72, 0xC0}, Form: ExtFormAmdVec2Half, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VCVTNEPS2BF16 (EVEX.128.F3.0F38.W0 72 /r, XMM destination)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x40, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.512.F3.0F38.W0 52 /r)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x20, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.256.F3.0F38.W0 52 /r)"},
|
||||
{Name: "VDPBF16PS", Summary: "Multiply BF16 pairs and accumulate the dot product into single precision",
|
||||
Bytes: []byte{0x62, 0x02, 0x06, 0x00, 0x52, 0xC0}, Form: ExtFormAmdVec3, Feature: ExtFeatureBF16,
|
||||
Ref: "Intel SDM Vol. 2C, VDPBF16PS (EVEX.NDS.128.F3.0F38.W0 52 /r)"},
|
||||
|
||||
// AVX512-VP2INTERSECT: the pairwise intersection indices, one opmask
|
||||
// destination and two vector sources, EVEX.NDS.66.0F38. The instruction
|
||||
// takes no write mask of its own.
|
||||
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x40, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
||||
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.512.F2.0F38.W0 68 /r)"},
|
||||
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x20, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
||||
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.256.F2.0F38.W0 68 /r)"},
|
||||
{Name: "VP2INTERSECTD", Summary: "Store the indices of the first pairwise intersections of two dword vectors",
|
||||
Bytes: []byte{0x62, 0x02, 0x07, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
||||
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W0 68 /r)"},
|
||||
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
|
||||
Bytes: []byte{0x62, 0x02, 0x87, 0x40, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
||||
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.512.F2.0F38.W1 68 /r)"},
|
||||
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
|
||||
Bytes: []byte{0x62, 0x02, 0x87, 0x20, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
||||
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.256.F2.0F38.W1 68 /r)"},
|
||||
{Name: "VP2INTERSECTQ", Summary: "Store the indices of the first pairwise intersections of two qword vectors",
|
||||
Bytes: []byte{0x62, 0x02, 0x87, 0x00, 0x68, 0xC0}, Form: ExtFormAmdMask2, Feature: ExtFeatureVP2INTERSECT,
|
||||
Ref: "Intel SDM Vol. 2C, VP2INTERSECTD/VP2INTERSECTQ (EVEX.NDS.128.F2.0F38.W1 68 /r)"},
|
||||
}
|
||||
Reference in new issue
Block a user