Assisted-by: GLM 5.3 Flash
This commit is contained in:
+221
-7
@@ -61,6 +61,21 @@ const (
|
||||
// vexImmRMGPR is the immediate form over general-purpose registers
|
||||
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
||||
vexImmRMGPR
|
||||
// vexRMOpGPR is the two-operand /digit form over general-purpose
|
||||
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
|
||||
// (op0), VEX.vvvv = dst (op1), L = 0.
|
||||
vexRMOpGPR
|
||||
// vexCountGPR is the three-operand count form over general-purpose
|
||||
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
|
||||
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
|
||||
// family, with reg = dst (op2), L = 0.
|
||||
vexCountGPR
|
||||
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
|
||||
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
|
||||
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
|
||||
// carries a vector length, so the register the L'L field follows is the
|
||||
// XMM source.
|
||||
vexExtractGPR
|
||||
)
|
||||
|
||||
// vexSpec describes one VEX instruction's encoding parameters.
|
||||
@@ -230,8 +245,34 @@ var vexTable = map[string]vexSpec{
|
||||
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
||||
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
||||
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
||||
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
|
||||
// share the F7/F5 opcodes across W, the variable shifts carry their
|
||||
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
|
||||
// in F2/F3.
|
||||
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
|
||||
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
|
||||
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
|
||||
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
|
||||
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
|
||||
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
|
||||
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
|
||||
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
|
||||
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
|
||||
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
|
||||
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
|
||||
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
|
||||
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
|
||||
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
|
||||
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
|
||||
// rm = src, vvvv = dst).
|
||||
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
|
||||
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
|
||||
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
|
||||
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
|
||||
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
|
||||
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
|
||||
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||
|
||||
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||
@@ -290,6 +331,128 @@ var vexTable = map[string]vexSpec{
|
||||
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||
|
||||
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
|
||||
// spellings, read off the toolchain opcode tables ---
|
||||
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
|
||||
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
|
||||
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
|
||||
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
|
||||
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
|
||||
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
|
||||
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
|
||||
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
|
||||
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
|
||||
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
|
||||
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
|
||||
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
|
||||
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
|
||||
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
|
||||
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
|
||||
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
|
||||
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
|
||||
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
|
||||
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
|
||||
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
|
||||
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
|
||||
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
|
||||
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
|
||||
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
|
||||
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
|
||||
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
|
||||
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
|
||||
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
|
||||
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
|
||||
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
|
||||
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
|
||||
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
|
||||
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
|
||||
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
|
||||
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
|
||||
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
|
||||
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
|
||||
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
|
||||
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
|
||||
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
|
||||
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
|
||||
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
|
||||
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
|
||||
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
|
||||
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
|
||||
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
|
||||
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
|
||||
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
|
||||
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
|
||||
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
|
||||
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
|
||||
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
|
||||
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
|
||||
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
|
||||
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
|
||||
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
|
||||
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
|
||||
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
|
||||
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
|
||||
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
|
||||
|
||||
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
|
||||
// spellings (no 66 prefix; WIG, so W = 0).
|
||||
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
|
||||
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
|
||||
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
|
||||
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
|
||||
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
|
||||
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
|
||||
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
|
||||
// VEX.128.66.0F, the scalar and packed compare forms.
|
||||
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
|
||||
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
|
||||
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
|
||||
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
|
||||
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
|
||||
}
|
||||
|
||||
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
||||
@@ -420,6 +583,10 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
||||
return e.encodeVexNDS3GPR(spec, ops)
|
||||
case vexImmRMGPR:
|
||||
return e.encodeVexImmRMGPR(spec, ops)
|
||||
case vexRMOpGPR:
|
||||
return e.encodeVexRMOpGPR(spec, ops)
|
||||
case vexCountGPR:
|
||||
return e.encodeVexCountGPR(spec, ops)
|
||||
case vexRMRev:
|
||||
return e.encodeVexRMRev(spec, ops)
|
||||
}
|
||||
@@ -523,9 +690,10 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
||||
if !ok {
|
||||
return fmt.Errorf("shift count must be an immediate")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("shift source must be a vector register")
|
||||
// The count source is a vector register or memory; the VEX length
|
||||
// follows the destination register either way.
|
||||
if !vecOrMem(src) {
|
||||
return fmt.Errorf("shift source must be a vector register or memory")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
@@ -533,7 +701,7 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
||||
}
|
||||
|
||||
vvvvBar := 15 - (dstReg.idx & 15)
|
||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
|
||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
|
||||
return err
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
@@ -700,7 +868,11 @@ func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
||||
if !ok || vvvvReg.isVec() {
|
||||
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
|
||||
}
|
||||
|
||||
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
||||
@@ -729,6 +901,48 @@ func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
|
||||
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
|
||||
// ModRM.rm = src and VEX.vvvv = dst.
|
||||
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
|
||||
}
|
||||
|
||||
// encodeVexCountGPR encodes the three-operand count form over general-purpose
|
||||
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
|
||||
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
|
||||
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
|
||||
}
|
||||
src, count, dst := ops[0], ops[1], ops[2]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || dstReg.isVec() {
|
||||
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||
}
|
||||
countReg, ok := count.(Reg)
|
||||
if !ok || countReg.isVec() {
|
||||
return fmt.Errorf("VEX count operand must be a general-purpose register")
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
if !ok || srcReg.isVec() {
|
||||
return fmt.Errorf("VEX count source must be a general-purpose register")
|
||||
}
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
}
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
|
||||
}
|
||||
|
||||
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
||||
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
||||
// a store with no register-destination form).
|
||||
|
||||
Reference in New Issue
Block a user