feat(amd64): encode the GOROOT instruction families
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+144
-1
@@ -41,7 +41,7 @@ const (
|
||||
vexExtract
|
||||
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
||||
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
||||
// narrowing stores (VPMOVDW, VPMOVQD).
|
||||
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
|
||||
vexRMRev
|
||||
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
||||
// vector length follows the source: the packed-double → dword
|
||||
@@ -52,6 +52,15 @@ const (
|
||||
vexRMSrcLen
|
||||
// vexZero is the no-operand form (VZEROUPPER).
|
||||
vexZero
|
||||
// vexZeroAll is the no-operand form that zeroes the full upper state
|
||||
// (VZEROALL, the L = 1 twin of VZEROUPPER).
|
||||
vexZeroAll
|
||||
// vexNDS3GPR is the three-operand NDS form over general-purpose
|
||||
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
|
||||
vexNDS3GPR
|
||||
// vexImmRMGPR is the immediate form over general-purpose registers
|
||||
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
||||
vexImmRMGPR
|
||||
)
|
||||
|
||||
// vexSpec describes one VEX instruction's encoding parameters.
|
||||
@@ -125,6 +134,12 @@ var vexTable = map[string]vexSpec{
|
||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
||||
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
||||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
||||
// Scalar fused multiply-add (NDS form). The Go assembler carries the
|
||||
// same 66 prefix as the packed forms on every FMA row, and W1 on the
|
||||
// double-precision spellings, so SD shares PD's prefix/W pair and the
|
||||
// scalar width rides on the W bit.
|
||||
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
|
||||
|
||||
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
||||
// no vvvv).
|
||||
@@ -192,6 +207,31 @@ var vexTable = map[string]vexSpec{
|
||||
|
||||
// VEX.128.0F.W0, no operands.
|
||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
|
||||
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
|
||||
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
|
||||
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
|
||||
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
|
||||
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
|
||||
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
|
||||
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
|
||||
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
|
||||
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
|
||||
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
|
||||
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
|
||||
// falls through to the EVEX table.
|
||||
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
|
||||
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
|
||||
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
|
||||
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
|
||||
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
|
||||
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
|
||||
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
|
||||
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
||||
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
||||
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
||||
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||
|
||||
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||
@@ -200,6 +240,14 @@ var vexTable = map[string]vexSpec{
|
||||
// rm=scalar memory; SD is 256-bit only).
|
||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
|
||||
// YMM (the encoder rejects an XMM destination, as go tool asm does).
|
||||
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
|
||||
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
|
||||
// memory destination in rm).
|
||||
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
|
||||
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
|
||||
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
|
||||
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
||||
// source).
|
||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||
@@ -290,6 +338,8 @@ type vexMoveSpec struct {
|
||||
var vexMoveTable = map[string]vexMoveSpec{
|
||||
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
||||
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||
// VEX.128/256.66.0F.WIG, aligned integer move.
|
||||
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
||||
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
||||
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
||||
@@ -324,6 +374,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
||||
}
|
||||
}
|
||||
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
|
||||
// only; an XMM destination is rejected exactly as go tool asm does.
|
||||
if mnemUpper == "VBROADCASTI128" {
|
||||
dstReg, ok := ops[len(ops)-1].(Reg)
|
||||
if len(ops) != 2 || !ok || dstReg.size != 32 {
|
||||
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
|
||||
}
|
||||
}
|
||||
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
||||
return e.encodeVexMove(mnemUpper, ms, ops)
|
||||
}
|
||||
@@ -356,6 +414,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||
case vexZero:
|
||||
return e.encodeVexZero(mnemUpper, spec, ops)
|
||||
case vexZeroAll:
|
||||
return e.encodeVexZeroAll(mnemUpper, spec, ops)
|
||||
case vexNDS3GPR:
|
||||
return e.encodeVexNDS3GPR(spec, ops)
|
||||
case vexImmRMGPR:
|
||||
return e.encodeVexImmRMGPR(spec, ops)
|
||||
case vexRMRev:
|
||||
return e.encodeVexRMRev(spec, ops)
|
||||
}
|
||||
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
||||
}
|
||||
@@ -607,6 +673,83 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
|
||||
// twin of VZEROUPPER.
|
||||
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 0 {
|
||||
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
||||
}
|
||||
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
|
||||
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
|
||||
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
|
||||
// rm = src2 and L = 0.
|
||||
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
||||
}
|
||||
src2, src1, dst := ops[0], ops[1], ops[2]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
vvvvReg, ok := src1.(Reg)
|
||||
if !ok || vvvvReg.isVec() {
|
||||
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
|
||||
}
|
||||
|
||||
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
||||
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
|
||||
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
imm, src, dst := ops[0], ops[1], ops[2]
|
||||
immVal, ok := imm.(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("shift control must be an immediate")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
||||
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
||||
// a store with no register-destination form).
|
||||
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
srcReg, ok := ops[0].(Reg)
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("store source must be a vector register")
|
||||
}
|
||||
if !memOperand(ops[1]) {
|
||||
return fmt.Errorf("store destination must be memory")
|
||||
}
|
||||
rBit := 0
|
||||
if srcReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
|
||||
}
|
||||
|
||||
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
||||
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
||||
// move uses the store-form layout (reg = source, rm = destination), matching
|
||||
|
||||
Reference in New Issue
Block a user