feat(arch): add the VEX encoder to the extension layer

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 19:49:15 +02:00
1 parent 47d561b229
commit 96000dd64d
3 files changed
+249 -8

No files matched your search

+169
View File
@@ -92,6 +92,16 @@ func amd64LengthClass(b []byte) ExtOperandKind {
}
}
// amd64VexLengthClass reads the vector length a VEX template encodes out of
// VEX.L, bit two of byte two, and names the register class every vector
// operand of that entry must carry. VEX names no 512-bit class.
func amd64VexLengthClass(b []byte) ExtOperandKind {
if b[2]&0x04 != 0 {
return ExtYMM
}
return ExtXMM
}
// amd64HalfClass names the half-width companion of a vector class, the
// destination class of the narrow conversions. At 128 bits the companion is
// the class itself, which is what the manual gives for the narrowest form.
@@ -143,6 +153,76 @@ func amd64Encode(b []byte, dest, vvvv, rm int) []byte {
return out
}
// amd64EncodeVex returns a VEX template with the register-derived bits
// filled in, the mirror amd64Encode is over the EVEX layout: dest and rm
// are register numbers for the ModR/M reg and r/m fields and vvvv the third
// operand's register or -1 when the form leaves it unused. The three
// register bits VEX carries ride R bar, X bar and B bar in byte one beside
// the map, and vvvv keeps its complement in byte two, where the W, L and pp
// bits the template carries stay untouched. The registers run 0..15: a VEX
// word names no register above them, and the vector check upstream refuses
// one before the bits could wrap.
func amd64EncodeVex(b []byte, dest, vvvv, rm int) []byte {
out := make([]byte, len(b))
copy(out, b)
rBar, xBar, bBar := 1, 1, 1
if dest&8 != 0 {
rBar = 0
}
if rm&8 != 0 {
bBar = 0
}
if rm&16 != 0 {
xBar = 0
}
out[1] |= byte(rBar<<7 | xBar<<6 | bBar<<5)
vBar := 15
if vvvv >= 0 {
vBar = 15 - vvvv
}
out[2] |= byte(vBar << 3)
out[4] |= byte((dest&7)<<3 | rm&7)
return out
}
// amd64EncodeVexMemory returns a VEX template with a base-relative memory
// operand filled in: dest and vvvv keep their register meanings, the ModR/M
// r/m field carries the base, and the mod bits and displacement bytes follow
// the canonical choices amd64EncodeMemory makes over the EVEX words, the
// classic ones no VEX word scales. The operand must have passed
// amd64Memory first.
func amd64EncodeVexMemory(b []byte, dest, vvvv, base int, disp int64) []byte {
out := amd64EncodeVex(b, dest, vvvv, base)
rm := base & 7
mod, tail := amd64DispTail(disp, rm == 5)
if rm == 4 {
// RSP and R12 need the SIB byte: no index, base 100.
tail = append([]byte{0x24}, tail...)
}
out[4] = out[4]&0x3f | mod<<6
return append(out, tail...)
}
// amd64EncodeVexScaledMemory returns a VEX template with a
// base-plus-scaled-index memory operand filled in, the mirror
// amd64EncodeScaledMemory is: the SIB byte follows the ModR/M and carries
// the scale field, the index and the base, whose number rides the r/m field
// as 100, and VEX.X changes meaning from the register's bit four to the
// index's bit three, so it clears when the index sits above 7. The
// displacement choices stay the canonical ones, with the RBP and R13 bases
// keeping their forced displacement.
func amd64EncodeVexScaledMemory(b []byte, dest, vvvv, base, index, scale int, disp int64) []byte {
out := amd64EncodeVex(b, dest, vvvv, base)
if index&8 != 0 {
out[1] &^= 0x40
}
rm := base & 7
mod, tail := amd64DispTail(disp, rm == 5)
tail = append([]byte{amd64Sib(rm, index, scale)}, tail...)
out[4] = out[4]&0x38 | mod<<6 | 4
return append(out, tail...)
}
// amd64Memory validates a memory operand of an amd64 entry: no arrangement
// and no qualifier, a base general register inside 0-15, a signed 32-bit
// displacement and no shift. The base number rides the operand's Reg and
@@ -312,6 +392,25 @@ func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos in
return amd64EncodeMemory(b, dest, vvvv, base, disp), nil
}
// amd64VexMemBytes encodes one validated memory position of a VEX entry:
// the plain base-plus-displacement form and the scaled index over it, the
// layers amd64MemBytes lays over the EVEX words. The broadcast spelling is
// refused upstream: the entry carries no Bcast, which amd64Memory rejects.
func (in ExtInstr) amd64VexMemBytes(b []byte, dest, vvvv int, op ExtOperand, pos int) ([]byte, error) {
base, disp, err := in.amd64Memory(op, pos)
if err != nil {
return nil, err
}
if op.HasIndex {
index, scale, err := in.amd64Index(op, pos)
if err != nil {
return nil, err
}
return amd64EncodeVexScaledMemory(b, dest, vvvv, base, index, scale, disp), nil
}
return amd64EncodeVexMemory(b, dest, vvvv, base, disp), nil
}
// amd64WriteMask lifts the decorations off a destination operand: the
// returned copy carries the register bits alone, while the mask register,
// the zeroing flag and the rounding control come back beside it. K0 never
@@ -434,6 +533,23 @@ func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) err
return in.amd64PlainReg(op, 31, pos)
}
// amd64VexVector checks one vector operand against the class a VEX entry
// encodes: the register runs 0..15, VEX carrying four register bits where
// EVEX carries five.
func (in ExtInstr) amd64VexVector(op ExtOperand, class ExtOperandKind, pos int) error {
if op.Broadcast {
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
}
if op.Kind != class {
article := "a"
if class == ExtXMM {
article = "an"
}
return fmt.Errorf("%s: operand %d wants %s %s, got %s", in.Name, pos, article, class, op.Kind)
}
return in.amd64PlainReg(op, 15, pos)
}
// amd64Gpr checks the general-register operand against the width the entry
// encodes: the W bit picks 32-bit or 64-bit, unless the entry ignores W, and
// the general registers run 0..15.
@@ -458,6 +574,9 @@ func (in ExtInstr) amd64Gpr(op ExtOperand, pos int) error {
func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
switch in.Form {
case ExtFormAmdVec3:
if in.Vex {
return in.encodeAmdVexVec3(ops)
}
return in.encodeAmdVec3(ops)
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter,
ExtFormAmdVec2ToQuarter:
@@ -534,6 +653,56 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
return out, nil
}
// encodeAmdVexVec3 fills the VEX-encoded three-vector form: src1, src2,
// dest, the layout the AVX extensions carry over the two-byte VEX prefix.
// An entry with Mem set takes the memory shape of the second source, the
// base-relative and scaled-index spellings amd64MemBytes lays over the EVEX
// words and the classic displacement choices amd64EncodeVexMemory keeps.
// The registers run 0..15 and the decorations are refused: a VEX row
// carries no mask, broadcast or rounding capability, which the same
// validators that gate the EVEX entries enforce here, so a decorated
// operand is an error and never a silently dropped spelling.
func (in ExtInstr) encodeAmdVexVec3(ops []ExtOperand) ([]byte, error) {
class := amd64VexLengthClass(in.Bytes)
if err := in.amd64VexVector(ops[0], class, 1); err != nil {
return nil, err
}
if in.Mem == 2 && ops[1].Kind == ExtMem {
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if round != ExtRoundNone {
return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name)
}
if err := in.amd64VexVector(dest, class, 3); err != nil {
return nil, err
}
out, err := in.amd64VexMemBytes(in.Bytes, dest.Reg, ops[0].Reg, ops[1], 2)
if err != nil {
return nil, err
}
amd64ApplyMask(out, mask, zeroing)
return out, nil
}
for i, op := range ops[1:2] {
if err := in.amd64VexVector(op, class, i+2); err != nil {
return nil, err
}
}
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3)
if err != nil {
return nil, err
}
if err := in.amd64VexVector(dest, class, 3); err != nil {
return nil, err
}
out := amd64EncodeVex(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg)
amd64ApplyMask(out, mask, zeroing)
amd64ApplyRounding(out, round)
return out, nil
}
// encodeAmdMemVec fills the memory-load form: mem, dest. VMOVSH X30,
// 4660(R8) shape, the manual's xmm1, m16 lines beside the register form.
// The form reads one value from memory, so the third register slot stays