feat(arch): add the VEX encoder to the extension layer
Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
47d561b229
commit
96000dd64d
3 files changed
+249
-8
No files matched your search
@@ -92,6 +92,16 @@ func amd64LengthClass(b []byte) ExtOperandKind {
|
||||
}
|
||||
}
|
||||
|
||||
// amd64VexLengthClass reads the vector length a VEX template encodes out of
|
||||
// VEX.L, bit two of byte two, and names the register class every vector
|
||||
// operand of that entry must carry. VEX names no 512-bit class.
|
||||
func amd64VexLengthClass(b []byte) ExtOperandKind {
|
||||
if b[2]&0x04 != 0 {
|
||||
return ExtYMM
|
||||
}
|
||||
return ExtXMM
|
||||
}
|
||||
|
||||
// amd64HalfClass names the half-width companion of a vector class, the
|
||||
// destination class of the narrow conversions. At 128 bits the companion is
|
||||
// the class itself, which is what the manual gives for the narrowest form.
|
||||
@@ -143,6 +153,76 @@ func amd64Encode(b []byte, dest, vvvv, rm int) []byte {
|
||||
return out
|
||||
}
|
||||
|
||||
// amd64EncodeVex returns a VEX template with the register-derived bits
|
||||
// filled in, the mirror amd64Encode is over the EVEX layout: dest and rm
|
||||
// are register numbers for the ModR/M reg and r/m fields and vvvv the third
|
||||
// operand's register or -1 when the form leaves it unused. The three
|
||||
// register bits VEX carries ride R bar, X bar and B bar in byte one beside
|
||||
// the map, and vvvv keeps its complement in byte two, where the W, L and pp
|
||||
// bits the template carries stay untouched. The registers run 0..15: a VEX
|
||||
// word names no register above them, and the vector check upstream refuses
|
||||
// one before the bits could wrap.
|
||||
func amd64EncodeVex(b []byte, dest, vvvv, rm int) []byte {
|
||||
out := make([]byte, len(b))
|
||||
copy(out, b)
|
||||
rBar, xBar, bBar := 1, 1, 1
|
||||
if dest&8 != 0 {
|
||||
rBar = 0
|
||||
}
|
||||
if rm&8 != 0 {
|
||||
bBar = 0
|
||||
}
|
||||
if rm&16 != 0 {
|
||||
xBar = 0
|
||||
}
|
||||
out[1] |= byte(rBar<<7 | xBar<<6 | bBar<<5)
|
||||
vBar := 15
|
||||
if vvvv >= 0 {
|
||||
vBar = 15 - vvvv
|
||||
}
|
||||
out[2] |= byte(vBar << 3)
|
||||
out[4] |= byte((dest&7)<<3 | rm&7)
|
||||
return out
|
||||
}
|
||||
|
||||
// amd64EncodeVexMemory returns a VEX template with a base-relative memory
|
||||
// operand filled in: dest and vvvv keep their register meanings, the ModR/M
|
||||
// r/m field carries the base, and the mod bits and displacement bytes follow
|
||||
// the canonical choices amd64EncodeMemory makes over the EVEX words, the
|
||||
// classic ones no VEX word scales. The operand must have passed
|
||||
// amd64Memory first.
|
||||
func amd64EncodeVexMemory(b []byte, dest, vvvv, base int, disp int64) []byte {
|
||||
out := amd64EncodeVex(b, dest, vvvv, base)
|
||||
rm := base & 7
|
||||
mod, tail := amd64DispTail(disp, rm == 5)
|
||||
if rm == 4 {
|
||||
// RSP and R12 need the SIB byte: no index, base 100.
|
||||
tail = append([]byte{0x24}, tail...)
|
||||
}
|
||||
out[4] = out[4]&0x3f | mod<<6
|
||||
return append(out, tail...)
|
||||
}
|
||||
|
||||
// amd64EncodeVexScaledMemory returns a VEX template with a
|
||||
// base-plus-scaled-index memory operand filled in, the mirror
|
||||
// amd64EncodeScaledMemory is: the SIB byte follows the ModR/M and carries
|
||||
// the scale field, the index and the base, whose number rides the r/m field
|
||||
// as 100, and VEX.X changes meaning from the register's bit four to the
|
||||
// index's bit three, so it clears when the index sits above 7. The
|
||||
// displacement choices stay the canonical ones, with the RBP and R13 bases
|
||||
// keeping their forced displacement.
|
||||
func amd64EncodeVexScaledMemory(b []byte, dest, vvvv, base, index, scale int, disp int64) []byte {
|
||||
out := amd64EncodeVex(b, dest, vvvv, base)
|
||||
if index&8 != 0 {
|
||||
out[1] &^= 0x40
|
||||
}
|
||||
rm := base & 7
|
||||
mod, tail := amd64DispTail(disp, rm == 5)
|
||||
tail = append([]byte{amd64Sib(rm, index, scale)}, tail...)
|
||||
out[4] = out[4]&0x38 | mod<<6 | 4
|
||||
return append(out, tail...)
|
||||
}
|
||||
|
||||
// amd64Memory validates a memory operand of an amd64 entry: no arrangement
|
||||
// and no qualifier, a base general register inside 0-15, a signed 32-bit
|
||||
// displacement and no shift. The base number rides the operand's Reg and
|
||||
@@ -312,6 +392,25 @@ func (in ExtInstr) amd64MemBytes(b []byte, dest, vvvv int, op ExtOperand, pos in
|
||||
return amd64EncodeMemory(b, dest, vvvv, base, disp), nil
|
||||
}
|
||||
|
||||
// amd64VexMemBytes encodes one validated memory position of a VEX entry:
|
||||
// the plain base-plus-displacement form and the scaled index over it, the
|
||||
// layers amd64MemBytes lays over the EVEX words. The broadcast spelling is
|
||||
// refused upstream: the entry carries no Bcast, which amd64Memory rejects.
|
||||
func (in ExtInstr) amd64VexMemBytes(b []byte, dest, vvvv int, op ExtOperand, pos int) ([]byte, error) {
|
||||
base, disp, err := in.amd64Memory(op, pos)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if op.HasIndex {
|
||||
index, scale, err := in.amd64Index(op, pos)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return amd64EncodeVexScaledMemory(b, dest, vvvv, base, index, scale, disp), nil
|
||||
}
|
||||
return amd64EncodeVexMemory(b, dest, vvvv, base, disp), nil
|
||||
}
|
||||
|
||||
// amd64WriteMask lifts the decorations off a destination operand: the
|
||||
// returned copy carries the register bits alone, while the mask register,
|
||||
// the zeroing flag and the rounding control come back beside it. K0 never
|
||||
@@ -434,6 +533,23 @@ func (in ExtInstr) amd64Vector(op ExtOperand, class ExtOperandKind, pos int) err
|
||||
return in.amd64PlainReg(op, 31, pos)
|
||||
}
|
||||
|
||||
// amd64VexVector checks one vector operand against the class a VEX entry
|
||||
// encodes: the register runs 0..15, VEX carrying four register bits where
|
||||
// EVEX carries five.
|
||||
func (in ExtInstr) amd64VexVector(op ExtOperand, class ExtOperandKind, pos int) error {
|
||||
if op.Broadcast {
|
||||
return fmt.Errorf("%s: operand %d carries a broadcast, the position takes a register", in.Name, pos)
|
||||
}
|
||||
if op.Kind != class {
|
||||
article := "a"
|
||||
if class == ExtXMM {
|
||||
article = "an"
|
||||
}
|
||||
return fmt.Errorf("%s: operand %d wants %s %s, got %s", in.Name, pos, article, class, op.Kind)
|
||||
}
|
||||
return in.amd64PlainReg(op, 15, pos)
|
||||
}
|
||||
|
||||
// amd64Gpr checks the general-register operand against the width the entry
|
||||
// encodes: the W bit picks 32-bit or 64-bit, unless the entry ignores W, and
|
||||
// the general registers run 0..15.
|
||||
@@ -458,6 +574,9 @@ func (in ExtInstr) amd64Gpr(op ExtOperand, pos int) error {
|
||||
func (in ExtInstr) encodeAmd64(ops []ExtOperand) ([]byte, error) {
|
||||
switch in.Form {
|
||||
case ExtFormAmdVec3:
|
||||
if in.Vex {
|
||||
return in.encodeAmdVexVec3(ops)
|
||||
}
|
||||
return in.encodeAmdVec3(ops)
|
||||
case ExtFormAmdVec2, ExtFormAmdVec2Half, ExtFormAmdVec2Wide, ExtFormAmdVec2Quarter,
|
||||
ExtFormAmdVec2ToQuarter:
|
||||
@@ -534,6 +653,56 @@ func (in ExtInstr) encodeAmdVec3(ops []ExtOperand) ([]byte, error) {
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// encodeAmdVexVec3 fills the VEX-encoded three-vector form: src1, src2,
|
||||
// dest, the layout the AVX extensions carry over the two-byte VEX prefix.
|
||||
// An entry with Mem set takes the memory shape of the second source, the
|
||||
// base-relative and scaled-index spellings amd64MemBytes lays over the EVEX
|
||||
// words and the classic displacement choices amd64EncodeVexMemory keeps.
|
||||
// The registers run 0..15 and the decorations are refused: a VEX row
|
||||
// carries no mask, broadcast or rounding capability, which the same
|
||||
// validators that gate the EVEX entries enforce here, so a decorated
|
||||
// operand is an error and never a silently dropped spelling.
|
||||
func (in ExtInstr) encodeAmdVexVec3(ops []ExtOperand) ([]byte, error) {
|
||||
class := amd64VexLengthClass(in.Bytes)
|
||||
if err := in.amd64VexVector(ops[0], class, 1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if in.Mem == 2 && ops[1].Kind == ExtMem {
|
||||
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if round != ExtRoundNone {
|
||||
return nil, fmt.Errorf("%s: the memory form takes no rounding control, the decoration belongs to the register form", in.Name)
|
||||
}
|
||||
if err := in.amd64VexVector(dest, class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out, err := in.amd64VexMemBytes(in.Bytes, dest.Reg, ops[0].Reg, ops[1], 2)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
return out, nil
|
||||
}
|
||||
for i, op := range ops[1:2] {
|
||||
if err := in.amd64VexVector(op, class, i+2); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
dest, mask, zeroing, round, err := in.amd64WriteMask(ops[2], 3)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := in.amd64VexVector(dest, class, 3); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := amd64EncodeVex(in.Bytes, dest.Reg, ops[0].Reg, ops[1].Reg)
|
||||
amd64ApplyMask(out, mask, zeroing)
|
||||
amd64ApplyRounding(out, round)
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// encodeAmdMemVec fills the memory-load form: mem, dest. VMOVSH X30,
|
||||
// 4660(R8) shape, the manual's xmm1, m16 lines beside the register form.
|
||||
// The form reads one value from memory, so the third register slot stays
|
||||
|
||||
Reference in new issue
Block a user