feat(asm): add the wider EVEX set and the rounding, SAE and broadcast suffixes
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
+459
-71
@@ -55,6 +55,13 @@ var evexTable = map[string]evexSpec{
|
||||
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.0F.W0 — packed single arithmetic.
|
||||
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.66.0F.W1 — packed double unpack.
|
||||
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
@@ -119,6 +126,114 @@ var evexTable = map[string]evexSpec{
|
||||
"VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
|
||||
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
|
||||
|
||||
// EVEX.66.0F3A — ternary logic and lane shuffles (NDS + imm8).
|
||||
"VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VSHUFI64X2": {3, 0x43, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VSHUFF32X4": {3, 0x23, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.66.0F — the EVEX forms of the VEX two-source shuffle.
|
||||
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.66.0F3A — lane insert ($imm, xsrc, zsrc1, zdst).
|
||||
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||||
"VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
|
||||
"VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||||
"VINSERTF64X4": {3, 0x1A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
|
||||
"VINSERTI32X4": {3, 0x38, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||||
"VINSERTI32X8": {3, 0x3A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
|
||||
"VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||||
"VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
|
||||
|
||||
// EVEX.66.0F3A — lane extract (reg=source, rm=XMM/YMM destination,
|
||||
// imm8).
|
||||
"VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
|
||||
"VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
||||
"VEXTRACTF64X2": {3, 0x19, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
|
||||
"VEXTRACTI32X4": {3, 0x39, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
|
||||
"VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
|
||||
"VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
|
||||
|
||||
// EVEX.66.0F — compare with an opmask destination ($imm, src2, src1,
|
||||
// kdst): NDS3Imm with the K register in the reg field.
|
||||
"VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
|
||||
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
|
||||
|
||||
// EVEX.66.0F38 — permutes (NDS form).
|
||||
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.66.0F — the wider integer set (NDS form).
|
||||
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||
|
||||
// EVEX.66.0F38 — absolute values and replicating moves (reg=dst,
|
||||
// rm=src).
|
||||
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
// EVEX.F3.0F — replicate even/odd singles.
|
||||
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||
// EVEX.66.0F38 — sign/zero-extending moves; the memory source is the
|
||||
// narrow half (here byte to word).
|
||||
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||
// EVEX.66.0F — packed single conversions (reg=dst, rm=src).
|
||||
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
|
||||
// EVEX.66.0F38 — broadcast a single/double to all lanes (reg=dst,
|
||||
// rm=scalar memory; disp8×N is the element size).
|
||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}},
|
||||
|
||||
// EVEX.66.0F38 — expand loads (rm → vector register destination).
|
||||
"VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
|
||||
"VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
|
||||
"VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
|
||||
|
||||
// EVEX.66.0F38 — compress stores (vector register source → rm), and the
|
||||
// remaining narrowing stores.
|
||||
"VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
|
||||
"VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
|
||||
"VPCOMPRESSD": {2, 0x8B, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
|
||||
"VPCOMPRESSQ": {2, 0x8B, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
|
||||
"VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||
"VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
|
||||
|
||||
// EVEX.66.0F — rotates (immediate form: /0 right, /1 left).
|
||||
"VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
|
||||
"VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
|
||||
"VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
|
||||
"VPROLQ": {1, 0x72, 1, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
|
||||
// EVEX word shifts.
|
||||
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||||
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||||
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||||
// EVEX W1 qword shifts.
|
||||
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||||
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
|
||||
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
|
||||
// the xmm/ymm/zmm destination lengths).
|
||||
@@ -200,6 +315,10 @@ var evexBcastTable = map[string]evexBcastSpec{
|
||||
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes.
|
||||
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
|
||||
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
|
||||
// EVEX.128/256/512.66.0F38 — broadcast a byte/word (GPR or memory
|
||||
// source) to all lanes.
|
||||
"VPBROADCASTB": {2, 0x7A, 0x78, 0, 1},
|
||||
"VPBROADCASTW": {2, 0x7B, 0x79, 0, 2},
|
||||
}
|
||||
|
||||
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
||||
@@ -228,6 +347,15 @@ var evexMoveTable = map[string]evexMoveSpec{
|
||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
|
||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512 — aligned packed moves.
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.66.0F — aligned integer moves.
|
||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||||
// EVEX.128.F3.0F.W0 — scalar single move, memory operands (the
|
||||
// three-operand register form is not supported).
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
|
||||
}
|
||||
|
||||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||||
@@ -260,18 +388,93 @@ func evexRequired(upper string, ops []Operand) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// stripEvexSuffix splits a ".Z" zeroing suffix off the mnemonic. It is the
|
||||
// only EVEX suffix supported; Go writes masking as an explicit K operand, not
|
||||
// a suffix.
|
||||
func stripEvexSuffix(mnem string) (base string, zeroing bool, err error) {
|
||||
i := strings.LastIndexByte(mnem, '.')
|
||||
// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts:
|
||||
// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE),
|
||||
// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is
|
||||
// not a suffix — Go writes it as an explicit K operand.
|
||||
type evexSuffix struct {
|
||||
zeroing bool
|
||||
sae bool
|
||||
bcst bool
|
||||
rounding int // -1 = none; otherwise the EVEX rc value (0 RN, 1 RD, 2 RU, 3 RZ)
|
||||
}
|
||||
|
||||
// any reports whether any suffix is present.
|
||||
func (s evexSuffix) any() bool {
|
||||
return s.zeroing || s.sae || s.bcst || s.rounding >= 0
|
||||
}
|
||||
|
||||
// evexOnly reports whether the suffix forces the EVEX encoding (everything
|
||||
// but plain zeroing, which the dispatch checks separately).
|
||||
func (s evexSuffix) evexOnly() bool {
|
||||
return s.sae || s.bcst || s.rounding >= 0
|
||||
}
|
||||
|
||||
// parseEvexSuffix splits the EVEX suffix chain off the mnemonic
|
||||
// ("VADDPD.RN_SAE.Z" → base "VADDPD", rounding RN, zeroing), validating the
|
||||
// combinations the Go assembler allows: .Z last, no duplicates, no
|
||||
// broadcast together with rounding/SAE.
|
||||
func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
|
||||
sfx := evexSuffix{rounding: -1}
|
||||
i := strings.IndexByte(mnem, '.')
|
||||
if i < 0 {
|
||||
return mnem, false, nil
|
||||
return mnem, sfx, nil
|
||||
}
|
||||
if mnem[i+1:] == "Z" {
|
||||
return mnem[:i], true, nil
|
||||
base := mnem[:i]
|
||||
parts := strings.Split(mnem[i+1:], ".")
|
||||
seen := map[string]bool{}
|
||||
for j, p := range parts {
|
||||
if seen[p] {
|
||||
return "", sfx, fmt.Errorf("duplicate EVEX suffix %q", p)
|
||||
}
|
||||
seen[p] = true
|
||||
switch p {
|
||||
case "Z":
|
||||
if j != len(parts)-1 {
|
||||
return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", mnem[i+1:])
|
||||
}
|
||||
sfx.zeroing = true
|
||||
case "SAE":
|
||||
sfx.sae = true
|
||||
case "BCST":
|
||||
sfx.bcst = true
|
||||
case "RN_SAE":
|
||||
sfx.rounding = 0
|
||||
case "RD_SAE":
|
||||
sfx.rounding = 1
|
||||
case "RU_SAE":
|
||||
sfx.rounding = 2
|
||||
case "RZ_SAE":
|
||||
sfx.rounding = 3
|
||||
default:
|
||||
return "", sfx, fmt.Errorf("unsupported EVEX suffix %q", p)
|
||||
}
|
||||
}
|
||||
return "", false, fmt.Errorf("unsupported EVEX suffix %q", mnem[i+1:])
|
||||
if sfx.bcst && (sfx.sae || sfx.rounding >= 0) {
|
||||
return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", mnem[i+1:])
|
||||
}
|
||||
return base, sfx, nil
|
||||
}
|
||||
|
||||
// evexRound lists the instructions that accept a rounding mode or .SAE.
|
||||
var evexRound = map[string]bool{
|
||||
"VADDPD": true, "VSUBPD": true, "VMULPD": true, "VDIVPD": true,
|
||||
"VMINPD": true, "VMAXPD": true,
|
||||
"VADDPS": true, "VSUBPS": true, "VMULPS": true, "VDIVPS": true,
|
||||
"VMINPS": true, "VMAXPS": true,
|
||||
"VADDSD": true, "VSUBSD": true, "VMULSD": true, "VDIVSD": true,
|
||||
"VMINSD": true, "VMAXSD": true,
|
||||
"VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true,
|
||||
"VMINSS": true, "VMAXSS": true,
|
||||
}
|
||||
|
||||
// evexBcstN maps an instruction accepting .BCST to the broadcast element
|
||||
// size — the disp8×N multiplier for its memory operand.
|
||||
var evexBcstN = map[string]int{
|
||||
"VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8,
|
||||
"VMINPD": 8, "VMAXPD": 8,
|
||||
"VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4,
|
||||
"VMINPS": 4, "VMAXPS": 4,
|
||||
}
|
||||
|
||||
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
|
||||
@@ -298,20 +501,52 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
|
||||
|
||||
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
|
||||
// mask, when present, is an explicit K1–K7 operand anywhere among the
|
||||
// operands; zeroing comes from the .Z mnemonic suffix and requires a mask.
|
||||
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error {
|
||||
// Mask-destination comparisons (VPCMPEQD …, K1): the last operand is the
|
||||
// destination K register, and any mask sits among the preceding operands.
|
||||
if spec, ok := evexTable[mnemUpper]; ok && spec.form == vexNDS3 && len(ops) > 0 {
|
||||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||
rest, mask, err := splitMask(ops[:len(ops)-1])
|
||||
if err != nil {
|
||||
return err
|
||||
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
|
||||
// broadcast.
|
||||
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
|
||||
spec, inTable := evexTable[mnemUpper]
|
||||
if inTable {
|
||||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||||
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
|
||||
}
|
||||
if sfx.bcst {
|
||||
n, ok := evexBcstN[mnemUpper]
|
||||
if !ok {
|
||||
return fmt.Errorf("%s: broadcast is not supported for this instruction", mnemUpper)
|
||||
}
|
||||
if zeroing && mask == 0 {
|
||||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
|
||||
spec.n = [3]int{n, n, n}
|
||||
}
|
||||
} else if sfx.evexOnly() {
|
||||
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
|
||||
}
|
||||
|
||||
// Mask-destination comparisons (VPCMPEQD, VCMPPD $imm, …): the last
|
||||
// operand is the destination K register, and any mask sits among the
|
||||
// preceding operands.
|
||||
kdst := func(encode func(evexSpec, []Operand, int, evexSuffix) error) error {
|
||||
dst, ok := ops[len(ops)-1].(Reg)
|
||||
if !ok || !dst.mask {
|
||||
return nil // not a K-destination form; fall through
|
||||
}
|
||||
rest, mask, err := splitMask(ops[:len(ops)-1])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if sfx.zeroing && mask == 0 {
|
||||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
|
||||
}
|
||||
return encode(spec, append(rest, dst), mask, sfx)
|
||||
}
|
||||
if inTable && len(ops) > 0 {
|
||||
switch spec.form {
|
||||
case vexNDS3:
|
||||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||
return kdst(e.encodeEvexNDS3)
|
||||
}
|
||||
case vexNDS3Imm:
|
||||
if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask {
|
||||
return kdst(e.encodeEvexNDS3Imm)
|
||||
}
|
||||
return e.encodeEvexNDS3(spec, append(rest, dst), mask, zeroing)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -319,38 +554,43 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if zeroing && mask == 0 {
|
||||
if sfx.zeroing && mask == 0 {
|
||||
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper)
|
||||
}
|
||||
ops = rest
|
||||
|
||||
if bs, ok := evexBcastTable[mnemUpper]; ok {
|
||||
return e.encodeEvexBcast(bs, ops, mask, zeroing)
|
||||
if sfx.evexOnly() {
|
||||
return fmt.Errorf("%s: broadcast instructions take no rounding/SAE/broadcast suffix", mnemUpper)
|
||||
}
|
||||
return e.encodeEvexBcast(bs, ops, mask, sfx)
|
||||
}
|
||||
if ms, ok := evexMoveTable[mnemUpper]; ok {
|
||||
return e.encodeEvexMove(mnemUpper, ms, ops, mask, zeroing)
|
||||
if sfx.evexOnly() {
|
||||
return fmt.Errorf("%s: moves take no rounding/SAE/broadcast suffix", mnemUpper)
|
||||
}
|
||||
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
|
||||
}
|
||||
spec, ok := evexTable[mnemUpper]
|
||||
if !ok {
|
||||
if !inTable {
|
||||
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
||||
}
|
||||
switch spec.form {
|
||||
case vexNDS3:
|
||||
return e.encodeEvexNDS3(spec, ops, mask, zeroing)
|
||||
return e.encodeEvexNDS3(spec, ops, mask, sfx)
|
||||
case vexRM:
|
||||
return e.encodeEvexRM(spec, ops, mask, zeroing)
|
||||
return e.encodeEvexRM(spec, ops, mask, sfx)
|
||||
case vexRMRev:
|
||||
return e.encodeEvexRMRev(spec, ops, mask, zeroing)
|
||||
return e.encodeEvexRMRev(spec, ops, mask, sfx)
|
||||
case vexImmRM:
|
||||
return e.encodeEvexImmRM(spec, ops, mask, zeroing)
|
||||
return e.encodeEvexImmRM(spec, ops, mask, sfx)
|
||||
case vexShiftImm:
|
||||
return e.encodeEvexShiftImm(spec, ops, mask, zeroing)
|
||||
return e.encodeEvexShiftImm(spec, ops, mask, sfx)
|
||||
case vexNDS3Imm:
|
||||
return e.encodeEvexNDS3Imm(spec, ops, mask, zeroing)
|
||||
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
|
||||
case vexExtract:
|
||||
return e.encodeEvexExtract(spec, ops, mask, zeroing)
|
||||
return e.encodeEvexExtract(spec, ops, mask, sfx)
|
||||
case vexRMSrcLen:
|
||||
return e.encodeEvexRMSrcLen(spec, ops, mask, zeroing)
|
||||
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
|
||||
}
|
||||
return fmt.Errorf("unhandled EVEX form for %s", mnemUpper)
|
||||
}
|
||||
@@ -358,7 +598,7 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error {
|
||||
// encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The
|
||||
// destination may be an opmask register (VPCMPEQD), in which case the vector
|
||||
// length comes from the sources.
|
||||
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops))
|
||||
}
|
||||
@@ -378,12 +618,12 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, zeroing boo
|
||||
ll = r.vecLenBit()
|
||||
}
|
||||
}
|
||||
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, zeroing)
|
||||
return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx)
|
||||
}
|
||||
|
||||
// encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src,
|
||||
// no vvvv), e.g. VCVTQQ2PD.
|
||||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
@@ -392,12 +632,12 @@ func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, zeroing bool)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("EVEX destination must be a vector register")
|
||||
}
|
||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing)
|
||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
|
||||
}
|
||||
|
||||
// encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst
|
||||
// (reg = dst, rm = src, imm8), e.g. VPSHUFD.
|
||||
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
@@ -418,7 +658,7 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bo
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing); err != nil {
|
||||
if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
@@ -427,7 +667,7 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bo
|
||||
|
||||
// encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst
|
||||
// (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5.
|
||||
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||
}
|
||||
@@ -448,7 +688,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, zeroing); err != nil {
|
||||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
@@ -456,8 +696,10 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing
|
||||
}
|
||||
|
||||
// encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1,
|
||||
// rm=src2, imm8), e.g. VALIGND.
|
||||
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
// rm=src2, imm8), e.g. VALIGND. The destination may be an opmask register
|
||||
// (VCMPPD and friends), in which case the vector length comes from the
|
||||
// sources.
|
||||
func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 4 {
|
||||
return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops))
|
||||
}
|
||||
@@ -467,8 +709,8 @@ func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing
|
||||
return fmt.Errorf("shuffle control must be an immediate")
|
||||
}
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("destination must be a vector register")
|
||||
if !ok || (!dstReg.isVec() && !dstReg.mask) {
|
||||
return fmt.Errorf("destination must be a vector or mask register")
|
||||
}
|
||||
vvvvReg, ok := src1.(Reg)
|
||||
if !ok || !vvvvReg.isVec() {
|
||||
@@ -478,7 +720,14 @@ func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2, mask, zeroing); err != nil {
|
||||
ll := dstReg.vecLenBit()
|
||||
if dstReg.mask {
|
||||
ll = vvvvReg.vecLenBit()
|
||||
if r, ok := src2.(Reg); ok && r.isVec() {
|
||||
ll = r.vecLenBit()
|
||||
}
|
||||
}
|
||||
if err := e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
@@ -487,7 +736,7 @@ func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing
|
||||
|
||||
// encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory
|
||||
// destination, imm8), e.g. VEXTRACTI64X4.
|
||||
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops))
|
||||
}
|
||||
@@ -504,7 +753,7 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing); err != nil {
|
||||
if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
@@ -514,7 +763,7 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing
|
||||
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
||||
// the store-form opcode (reg = source, rm = destination), matching the Go
|
||||
// assembler.
|
||||
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
@@ -543,14 +792,14 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
|
||||
return fmt.Errorf("%s needs a vector register operand", mnem)
|
||||
}
|
||||
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||||
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, zeroing)
|
||||
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
|
||||
}
|
||||
|
||||
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
|
||||
// the destination always XMM and the length fixed by the mnemonic — the
|
||||
// single valid slot of spec.n names the vector length (and the disp8×N
|
||||
// multiplier) a register or memory source encodes.
|
||||
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("conversion expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
@@ -563,7 +812,7 @@ func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing)
|
||||
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
|
||||
}
|
||||
|
||||
// soleLen returns the vector-length index of the single valid slot of n —
|
||||
@@ -598,7 +847,7 @@ func memOperand(op Operand) bool {
|
||||
|
||||
// encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide
|
||||
// source in the reg field and the narrow destination in r/m (VPMOVDW/QD).
|
||||
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
@@ -607,12 +856,12 @@ func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, zeroing bo
|
||||
if !ok || !srcReg.isVec() {
|
||||
return fmt.Errorf("EVEX source must be a vector register")
|
||||
}
|
||||
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing)
|
||||
return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx)
|
||||
}
|
||||
|
||||
// encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory
|
||||
// source broadcast to every lane of the vector destination.
|
||||
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
@@ -631,7 +880,7 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing
|
||||
default:
|
||||
return fmt.Errorf("broadcast source must be a register or memory")
|
||||
}
|
||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing)
|
||||
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx)
|
||||
}
|
||||
|
||||
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
|
||||
@@ -639,7 +888,7 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing
|
||||
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the
|
||||
// vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and
|
||||
// zeroing fill the aaa and z bits of the P2 byte.
|
||||
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, zeroing bool) error {
|
||||
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error {
|
||||
if ll > 2 {
|
||||
return fmt.Errorf("invalid vector length")
|
||||
}
|
||||
@@ -702,12 +951,22 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
|
||||
}
|
||||
|
||||
z := 0
|
||||
if zeroing {
|
||||
if sfx.zeroing {
|
||||
z = 1
|
||||
}
|
||||
// The b bit and the L'L field carry the rounding/SAE/broadcast mode:
|
||||
// a rounding mode replaces L'L with the rc value, plain SAE and
|
||||
// broadcast keep the vector length.
|
||||
b, ll := 0, ll
|
||||
switch {
|
||||
case sfx.rounding >= 0:
|
||||
b, ll = 1, sfx.rounding
|
||||
case sfx.sae || sfx.bcst:
|
||||
b = 1
|
||||
}
|
||||
p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel)
|
||||
p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp)
|
||||
p2 := byte(z<<7 | ll<<5 | vPrimeBar<<3 | mask) // z, L'L, b=0, V', aaa
|
||||
p2 := byte(z<<7 | ll<<5 | b<<4 | vPrimeBar<<3 | mask) // z, L'L/rc, b, V', aaa
|
||||
e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm))
|
||||
if sib >= 0 {
|
||||
e.out = append(e.out, byte(sib))
|
||||
@@ -775,24 +1034,40 @@ func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte,
|
||||
return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil
|
||||
}
|
||||
|
||||
// encodeKmovw encodes KMOVW, whose opcode depends on the operand direction:
|
||||
// 90 (k/mem → K), 91 (K → mem), 92 (GPR → K), 93 (K → GPR); k → k uses 90.
|
||||
func (e *enc) encodeKmovw(ops []Operand) error {
|
||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
||||
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
|
||||
// prefix and W for the wider widths.
|
||||
type kmovSpec struct {
|
||||
kk, kmem, gprk, kgpr byte
|
||||
gprPP int
|
||||
w int
|
||||
}
|
||||
|
||||
var kmovTable = map[string]kmovSpec{
|
||||
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
|
||||
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
|
||||
}
|
||||
|
||||
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
|
||||
func (e *enc) encodeKmov(upper string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("KMOVW expects 2 operands, got %d", len(ops))
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
ks := kmovTable[upper]
|
||||
src, dst := ops[0], ops[1]
|
||||
srcReg, srcIsReg := src.(Reg)
|
||||
dstReg, dstIsReg := dst.(Reg)
|
||||
srcK := srcIsReg && srcReg.mask
|
||||
dstK := dstIsReg && dstReg.mask
|
||||
spec := vexSpec{mapSel: 1, w: 0, pp: 0, opdigit: -1}
|
||||
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
|
||||
switch {
|
||||
case srcK && dstK:
|
||||
spec.opcode = 0x90 // k ← k: reg = dst, rm = src
|
||||
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||||
case srcK && dstIsReg:
|
||||
spec.opcode = 0x93 // GPR ← k: reg = dst, rm = src
|
||||
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
|
||||
spec.pp = ks.gprPP
|
||||
rBit := 0
|
||||
if dstReg.idx >= 8 {
|
||||
rBit = 1
|
||||
@@ -800,13 +1075,126 @@ func (e *enc) encodeKmovw(ops []Operand) error {
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src)
|
||||
case srcK:
|
||||
if _, ok := dst.(Mem); !ok {
|
||||
return fmt.Errorf("KMOVW: invalid destination operand")
|
||||
return fmt.Errorf("%s: invalid destination operand", upper)
|
||||
}
|
||||
spec.opcode = 0x91 // mem ← k: reg = src, rm = dst
|
||||
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
|
||||
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
||||
case dstK:
|
||||
spec.opcode = 0x92 // k ← GPR/mem: reg = dst, rm = src
|
||||
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
|
||||
spec.pp = ks.gprPP
|
||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||||
}
|
||||
return fmt.Errorf("KMOVW requires a K register operand")
|
||||
return fmt.Errorf("%s requires a K register operand", upper)
|
||||
}
|
||||
|
||||
// kOpSpec describes the VEX encoding of an opmask-register instruction: the
|
||||
// L bit and the W/pp pair select the operand width, and the form the
|
||||
// operand layout.
|
||||
type kOpSpec struct {
|
||||
mapSel int
|
||||
opcode byte
|
||||
w int
|
||||
pp int
|
||||
ll int
|
||||
form vexForm
|
||||
}
|
||||
|
||||
var kOpsTable = map[string]kOpSpec{
|
||||
// k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask
|
||||
// registers).
|
||||
"KANDB": {1, 0x41, 0, 1, 1, vexNDS3},
|
||||
"KANDW": {1, 0x41, 0, 0, 1, vexNDS3},
|
||||
"KANDQ": {1, 0x41, 1, 0, 1, vexNDS3},
|
||||
"KORB": {1, 0x45, 0, 1, 1, vexNDS3},
|
||||
"KORD": {1, 0x45, 1, 1, 1, vexNDS3},
|
||||
"KXNORW": {1, 0x46, 0, 0, 1, vexNDS3},
|
||||
"KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3},
|
||||
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
|
||||
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
|
||||
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
|
||||
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
|
||||
"KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3},
|
||||
// k ← OP k (KNOT) and flags ← k OP k (KORTEST): reg = dst, rm = src.
|
||||
"KNOTB": {1, 0x44, 0, 1, 0, vexRM},
|
||||
"KORTESTD": {1, 0x98, 1, 1, 0, vexRM},
|
||||
// OP $imm, src, dst: reg = dst, rm = src, imm8.
|
||||
"KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM},
|
||||
}
|
||||
|
||||
// isKOp reports whether the mnemonic is an opmask-register instruction.
|
||||
func isKOp(upper string) bool {
|
||||
_, ok := kOpsTable[upper]
|
||||
return ok
|
||||
}
|
||||
|
||||
// encodeKOp encodes an opmask-register instruction; every operand is a K
|
||||
// register and the vector length is fixed by the instruction.
|
||||
func (e *enc) encodeKOp(upper string, ops []Operand) error {
|
||||
ks := kOpsTable[upper]
|
||||
spec := vexSpec{mapSel: ks.mapSel, opcode: ks.opcode, w: ks.w, pp: ks.pp, opdigit: -1}
|
||||
kreg := func(op Operand, what string) (Reg, error) {
|
||||
r, ok := op.(Reg)
|
||||
if !ok || !r.mask {
|
||||
return Reg{}, fmt.Errorf("%s: %s must be an opmask register", upper, what)
|
||||
}
|
||||
return r, nil
|
||||
}
|
||||
switch ks.form {
|
||||
case vexNDS3:
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("%s expects 3 operands, got %d", upper, len(ops))
|
||||
}
|
||||
src2, err := kreg(ops[0], "first source")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
src1, err := kreg(ops[1], "second source")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dst, err := kreg(ops[2], "destination")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15-src1.idx, src2)
|
||||
case vexRM:
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||
}
|
||||
src, err := kreg(ops[0], "source")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dst, err := kreg(ops[1], "destination")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src)
|
||||
case vexImmRM:
|
||||
if len(ops) != 3 {
|
||||
return fmt.Errorf("%s expects 3 operands ($imm, src, dst), got %d", upper, len(ops))
|
||||
}
|
||||
immVal, ok := ops[0].(Imm)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s: shift count must be an immediate", upper)
|
||||
}
|
||||
src, err := kreg(ops[1], "source")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dst, err := kreg(ops[2], "destination")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
immByte, err := imm8(int64(immVal))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src); err != nil {
|
||||
return err
|
||||
}
|
||||
e.out = append(e.out, immByte)
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("unhandled opmask form for %s", upper)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user