feat(asm): encode the amd64 and loong64 tails of the corpus testdata
Assisted-by: GLM 5.3
This commit is contained in:
+123
-4
@@ -76,6 +76,10 @@ const (
|
||||
// carries a vector length, so the register the L'L field follows is the
|
||||
// XMM source.
|
||||
vexExtractGPR
|
||||
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
|
||||
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
|
||||
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
|
||||
vexBlend4
|
||||
)
|
||||
|
||||
// vexSpec describes one VEX instruction's encoding parameters.
|
||||
@@ -202,8 +206,28 @@ var vexTable = map[string]vexSpec{
|
||||
|
||||
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
|
||||
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
|
||||
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
|
||||
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
|
||||
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
|
||||
// double twin under op 01; the in-lane permutes under 04/05.
|
||||
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
|
||||
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM},
|
||||
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM},
|
||||
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM},
|
||||
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
|
||||
// pair, the AES key assistant and the string compares.
|
||||
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM},
|
||||
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM},
|
||||
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM},
|
||||
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM},
|
||||
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM},
|
||||
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM},
|
||||
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM},
|
||||
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
|
||||
// (reg = the XMM source, r/m = the destination).
|
||||
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR},
|
||||
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR},
|
||||
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
|
||||
// register in the /is4 byte.
|
||||
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4},
|
||||
|
||||
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
|
||||
// imm8).
|
||||
@@ -524,8 +548,16 @@ func isVex(mnemUpper string) bool {
|
||||
if _, ok := vexTable[mnemUpper]; ok {
|
||||
return true
|
||||
}
|
||||
_, ok := vexMoveTable[mnemUpper]
|
||||
return ok
|
||||
if _, ok := vexMoveTable[mnemUpper]; ok {
|
||||
return true
|
||||
}
|
||||
// The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand
|
||||
// count in encodeVex.
|
||||
switch mnemUpper {
|
||||
case "VMOVHPD", "VMOVLPD":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
|
||||
@@ -559,6 +591,22 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
|
||||
}
|
||||
}
|
||||
// The high/low double moves split by operand count: three operands
|
||||
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
|
||||
// the reversed store layout).
|
||||
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" {
|
||||
loadOp, storeOp := byte(0x16), byte(0x17)
|
||||
if mnemUpper == "VMOVLPD" {
|
||||
loadOp, storeOp = 0x12, 0x13
|
||||
}
|
||||
switch len(ops) {
|
||||
case 3:
|
||||
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops)
|
||||
case 2:
|
||||
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops)
|
||||
}
|
||||
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
|
||||
}
|
||||
spec := vexTable[mnemUpper]
|
||||
switch spec.form {
|
||||
case vexNDS3:
|
||||
@@ -573,6 +621,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return e.encodeVexNDS3Imm(spec, ops)
|
||||
case vexExtract:
|
||||
return e.encodeVexExtract(spec, ops)
|
||||
case vexExtractGPR:
|
||||
return e.encodeVexExtractGPR(spec, ops)
|
||||
case vexBlend4:
|
||||
return e.encodeVexBlend4(spec, ops)
|
||||
case vexRMSrcLen:
|
||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||
case vexZero:
|
||||
@@ -964,6 +1016,73 @@ func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
|
||||
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
|
||||
}
|
||||
|
||||
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
|
||||
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
|
||||
// ModRM.reg and the destination in r/m, L = 0.
|
||||
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
|
||||
}
|
||||
imm, src, dst := ops[0], ops[1], ops[2]
|
||||
immVal, ok := imm.(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("extract lane must be an immediate")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || !srcReg.isVec() || srcReg.size != 16 {
|
||||
return fmt.Errorf("extract source must be an XMM register")
|
||||
}
|
||||
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
|
||||
return fmt.Errorf("extract destination must be a register or memory")
|
||||
}
|
||||
rBit := 0
|
||||
if srcReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
|
||||
return err
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB):
|
||||
// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m =
|
||||
// src2 and the mask XMM register in the trailing /is4 byte.
|
||||
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 4 {
|
||||
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
|
||||
}
|
||||
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
|
||||
maskReg, ok := mask.(Reg)
|
||||
if !ok || !maskReg.isVec() || maskReg.size != 16 {
|
||||
return fmt.Errorf("blend mask must be an XMM register")
|
||||
}
|
||||
vvvvReg, ok := src1.(Reg)
|
||||
if !ok || !vvvvReg.isVec() {
|
||||
return fmt.Errorf("blend second source must be a vector register")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("blend destination must be a vector register")
|
||||
}
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
|
||||
return err
|
||||
}
|
||||
// The /is4 byte names the mask register: bits [3:0] its low nibble,
|
||||
// bit 7 the fourth register bit (X8-X15).
|
||||
e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4))
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
||||
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
||||
// move uses the store-form layout (reg = source, rm = destination), matching
|
||||
|
||||
Reference in New Issue
Block a user