feat(asm): encode the amd64 and loong64 tails of the corpus testdata

Assisted-by: GLM 5.3
This commit is contained in:
2026-10-02 00:40:43 +02:00
parent 2f679326c2
commit bafb2fd130
12 changed files with 1404 additions and 81 deletions
+123 -4
View File
@@ -76,6 +76,10 @@ const (
// carries a vector length, so the register the L'L field follows is the
// XMM source.
vexExtractGPR
// vexBlend4 is the four-operand variable blend `OP mask, src2, src1,
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
vexBlend4
)
// vexSpec describes one VEX instruction's encoding parameters.
@@ -202,8 +206,28 @@ var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8), and its
// double twin under op 01; the in-lane permutes under 04/05.
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM},
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM},
"VPERMILPD": {3, 0x05, 0, 1, -1, vexImmRM},
// VEX.66.0F3A.W0, the immediate-controlled AVX tail: the rounding
// pair, the AES key assistant and the string compares.
"VROUNDPD": {3, 0x09, 0, 1, -1, vexImmRM},
"VROUNDPS": {3, 0x08, 0, 1, -1, vexImmRM},
"VAESKEYGENASSIST": {3, 0xDF, 0, 1, -1, vexImmRM},
"VPCMPESTRI": {3, 0x61, 0, 1, -1, vexImmRM},
"VPCMPESTRM": {3, 0x60, 0, 1, -1, vexImmRM},
"VPCMPISTRI": {3, 0x63, 0, 1, -1, vexImmRM},
"VPCMPISTRM": {3, 0x62, 0, 1, -1, vexImmRM},
// VEX.128.66.0F3A.W0, the scalar lane extract to a GPR or memory
// (reg = the XMM source, r/m = the destination).
"VEXTRACTPS": {3, 0x17, 0, 1, -1, vexExtractGPR},
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR},
// VEX.128.66.0F3A.W0, the four-operand variable blend with its mask
// register in the /is4 byte.
"VPBLENDVB": {3, 0x4C, 0, 1, -1, vexBlend4},
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8).
@@ -524,8 +548,16 @@ func isVex(mnemUpper string) bool {
if _, ok := vexTable[mnemUpper]; ok {
return true
}
_, ok := vexMoveTable[mnemUpper]
return ok
if _, ok := vexMoveTable[mnemUpper]; ok {
return true
}
// The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand
// count in encodeVex.
switch mnemUpper {
case "VMOVHPD", "VMOVLPD":
return true
}
return false
}
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
@@ -559,6 +591,22 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
}
}
// The high/low double moves split by operand count: three operands
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
// the reversed store layout).
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" {
loadOp, storeOp := byte(0x16), byte(0x17)
if mnemUpper == "VMOVLPD" {
loadOp, storeOp = 0x12, 0x13
}
switch len(ops) {
case 3:
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops)
case 2:
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops)
}
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
}
spec := vexTable[mnemUpper]
switch spec.form {
case vexNDS3:
@@ -573,6 +621,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3Imm(spec, ops)
case vexExtract:
return e.encodeVexExtract(spec, ops)
case vexExtractGPR:
return e.encodeVexExtractGPR(spec, ops)
case vexBlend4:
return e.encodeVexBlend4(spec, ops)
case vexRMSrcLen:
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
@@ -964,6 +1016,73 @@ func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexExtractGPR encodes the lane extract to a general-purpose register
// or memory (VEXTRACTPS): OP $imm, xsrc, gpr/mem with the XMM source in
// ModRM.reg and the destination in r/m, L = 0.
func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("extract lane must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() || srcReg.size != 16 {
return fmt.Errorf("extract source must be an XMM register")
}
if _, isReg := dst.(Reg); !isReg && !memOperand(dst) {
return fmt.Errorf("extract destination must be a register or memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, 0, srcReg.idx&7, rBit, 15, dst); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB):
// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m =
// src2 and the mask XMM register in the trailing /is4 byte.
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
}
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
maskReg, ok := mask.(Reg)
if !ok || !maskReg.isVec() || maskReg.size != 16 {
return fmt.Errorf("blend mask must be an XMM register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("blend second source must be a vector register")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("blend destination must be a vector register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
return err
}
// The /is4 byte names the mask register: bits [3:0] its low nibble,
// bit 7 the fourth register bit (X8-X15).
e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4))
return nil
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching