From ee68859beb5aa5d3f24f389be671bee274fa6a46 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sat, 18 Jul 2026 15:47:59 +0200 Subject: [PATCH] feat(asm): add the wider EVEX set and the rounding, SAE and broadcast suffixes Assisted-by: Qwen 3.8 Max Preview --- asm/encode.go | 35 +-- asm/evex.go | 530 +++++++++++++++++++++++++++++++++++++------ asm/evex_test.go | 126 +++++++++- asm/vex.go | 39 +++- cmd/gasm/main.go | 2 +- docs/ARCHITECTURE.md | 20 +- justfile | 2 +- 7 files changed, 655 insertions(+), 99 deletions(-) diff --git a/asm/encode.go b/asm/encode.go index 30a2f6d..8ae2553 100644 --- a/asm/encode.go +++ b/asm/encode.go @@ -51,16 +51,17 @@ func (e *enc) encode(mnem string, ops []Operand) error { // VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing // B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch - // before splitSize. A ".Z" suffix requests EVEX zeroing. - base, zeroing, err := stripEvexSuffix(upper) + // before splitSize. EVEX suffixes (.Z, .SAE, rounding, .BCST) split + // off the mnemonic too. + base, sfx, err := parseEvexSuffix(upper) if err != nil { return err } - if isVex(base) || isEvex(base) || base == "KMOVW" { - return e.encodeVec(base, ops, zeroing) + if isVex(base) || isEvex(base) || isKOp(base) || base == "KMOVW" || base == "KMOVQ" { + return e.encodeVec(base, ops, sfx) } - if zeroing { - return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", mnem) + if sfx.any() { + return fmt.Errorf("%s: the suffix requires an EVEX instruction", mnem) } // CMOVcc and SETcc carry the condition in the mnemonic (CMOVLGT, SETNE). @@ -128,20 +129,26 @@ func splitSize(upper string) (base string, size int) { // its own direction-dependent opcodes; KTESTW is always VEX; everything else // takes EVEX when an operand demands it (a ZMM or K register, or an // EVEX-only mnemonic) and VEX otherwise. -func (e *enc) encodeVec(upper string, ops []Operand, zeroing bool) error { - if upper == "KMOVW" { - if zeroing { - return fmt.Errorf("KMOVW takes no .Z suffix") +func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error { + if upper == "KMOVW" || upper == "KMOVQ" { + if sfx.any() { + return fmt.Errorf("%s takes no EVEX suffixes", upper) } - return e.encodeKmovw(ops) + return e.encodeKmov(upper, ops) } - if upper == "KTESTW" || !evexRequired(upper, ops) { - if zeroing { + if isKOp(upper) { + if sfx.any() { + return fmt.Errorf("%s takes no EVEX suffixes", upper) + } + return e.encodeKOp(upper, ops) + } + if upper == "KTESTW" || (!evexRequired(upper, ops) && !sfx.evexOnly()) { + if sfx.any() { return fmt.Errorf("%s: the .Z suffix requires an EVEX instruction", upper) } return e.encodeVex(upper, ops) } - return e.encodeEvex(upper, ops, zeroing) + return e.encodeEvex(upper, ops, sfx) } // --- instruction components ------------------------------------------------- diff --git a/asm/evex.go b/asm/evex.go index ae2ffb0..c00dcb3 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -55,6 +55,13 @@ var evexTable = map[string]evexSpec{ "VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + // EVEX.128/256/512.0F.W0 — packed single arithmetic. + "VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, + "VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — packed double unpack. "VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, @@ -119,6 +126,114 @@ var evexTable = map[string]evexSpec{ "VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}}, "VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}}, "VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}}, + + // EVEX.66.0F3A — ternary logic and lane shuffles (NDS + imm8). + "VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VSHUFI64X2": {3, 0x43, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VSHUFF32X4": {3, 0x23, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + + // EVEX.66.0F — the EVEX forms of the VEX two-source shuffle. + "VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + + // EVEX.66.0F3A — lane insert ($imm, xsrc, zsrc1, zdst). + "VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, + "VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}}, + "VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, + "VINSERTF64X4": {3, 0x1A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}}, + "VINSERTI32X4": {3, 0x38, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, + "VINSERTI32X8": {3, 0x3A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}}, + "VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}}, + "VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}}, + + // EVEX.66.0F3A — lane extract (reg=source, rm=XMM/YMM destination, + // imm8). + "VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}}, + "VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}}, + "VEXTRACTF64X2": {3, 0x19, 1, 1, -1, vexExtract, [3]int{0, 16, 16}}, + "VEXTRACTI32X4": {3, 0x39, 0, 1, -1, vexExtract, [3]int{0, 16, 16}}, + "VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}}, + "VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}}, + + // EVEX.66.0F — compare with an opmask destination ($imm, src2, src1, + // kdst): NDS3Imm with the K register in the reg field. + "VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}}, + "VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}}, + "VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}}, + + // EVEX.66.0F38 — permutes (NDS form). + "VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + + // EVEX.66.0F — the wider integer set (NDS form). + "VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + "VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, + + // EVEX.66.0F38 — absolute values and replicating moves (reg=dst, + // rm=src). + "VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}}, + // EVEX.F3.0F — replicate even/odd singles. + "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, + "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, + // EVEX.66.0F38 — sign/zero-extending moves; the memory source is the + // narrow half (here byte to word). + "VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + "VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, + // EVEX.66.0F — packed single conversions (reg=dst, rm=src). + "VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}}, + "VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}}, + // EVEX.66.0F38 — broadcast a single/double to all lanes (reg=dst, + // rm=scalar memory; disp8×N is the element size). + "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, + "VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}}, + + // EVEX.66.0F38 — expand loads (rm → vector register destination). + "VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}}, + "VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, + "VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}}, + "VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}}, + + // EVEX.66.0F38 — compress stores (vector register source → rm), and the + // remaining narrowing stores. + "VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}}, + "VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}}, + "VPCOMPRESSD": {2, 0x8B, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}}, + "VPCOMPRESSQ": {2, 0x8B, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}}, + "VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, + "VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}}, + + // EVEX.66.0F — rotates (immediate form: /0 right, /1 left). + "VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}}, + "VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}}, + "VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}}, + "VPROLQ": {1, 0x72, 1, 1, 1, vexShiftImm, [3]int{16, 32, 64}}, + // EVEX word shifts. + "VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}}, + "VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}}, + "VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, + // EVEX W1 qword shifts. + "VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}}, + "VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // operand is the narrow source, so disp8×N follows its size (8/16/32 for // the xmm/ymm/zmm destination lengths). @@ -200,6 +315,10 @@ var evexBcastTable = map[string]evexBcastSpec{ // EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes. "VPBROADCASTD": {2, 0x7C, 0x58, 0, 4}, "VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8}, + // EVEX.128/256/512.66.0F38 — broadcast a byte/word (GPR or memory + // source) to all lanes. + "VPBROADCASTB": {2, 0x7A, 0x78, 0, 1}, + "VPBROADCASTW": {2, 0x7B, 0x79, 0, 2}, } // evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX @@ -228,6 +347,15 @@ var evexMoveTable = map[string]evexMoveSpec{ "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — unaligned packed double move. "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}}, + // EVEX.128/256/512 — aligned packed moves. + "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}}, + "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}}, + // EVEX.128/256/512.66.0F — aligned integer moves. + "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, + "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, + // EVEX.128.F3.0F.W0 — scalar single move, memory operands (the + // three-operand register form is not supported). + "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}}, } // isEvex reports whether the mnemonic has an EVEX encoding we handle. @@ -260,18 +388,93 @@ func evexRequired(upper string, ops []Operand) bool { return false } -// stripEvexSuffix splits a ".Z" zeroing suffix off the mnemonic. It is the -// only EVEX suffix supported; Go writes masking as an explicit K operand, not -// a suffix. -func stripEvexSuffix(mnem string) (base string, zeroing bool, err error) { - i := strings.LastIndexByte(mnem, '.') +// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts: +// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE), +// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is +// not a suffix — Go writes it as an explicit K operand. +type evexSuffix struct { + zeroing bool + sae bool + bcst bool + rounding int // -1 = none; otherwise the EVEX rc value (0 RN, 1 RD, 2 RU, 3 RZ) +} + +// any reports whether any suffix is present. +func (s evexSuffix) any() bool { + return s.zeroing || s.sae || s.bcst || s.rounding >= 0 +} + +// evexOnly reports whether the suffix forces the EVEX encoding (everything +// but plain zeroing, which the dispatch checks separately). +func (s evexSuffix) evexOnly() bool { + return s.sae || s.bcst || s.rounding >= 0 +} + +// parseEvexSuffix splits the EVEX suffix chain off the mnemonic +// ("VADDPD.RN_SAE.Z" → base "VADDPD", rounding RN, zeroing), validating the +// combinations the Go assembler allows: .Z last, no duplicates, no +// broadcast together with rounding/SAE. +func parseEvexSuffix(mnem string) (string, evexSuffix, error) { + sfx := evexSuffix{rounding: -1} + i := strings.IndexByte(mnem, '.') if i < 0 { - return mnem, false, nil + return mnem, sfx, nil } - if mnem[i+1:] == "Z" { - return mnem[:i], true, nil + base := mnem[:i] + parts := strings.Split(mnem[i+1:], ".") + seen := map[string]bool{} + for j, p := range parts { + if seen[p] { + return "", sfx, fmt.Errorf("duplicate EVEX suffix %q", p) + } + seen[p] = true + switch p { + case "Z": + if j != len(parts)-1 { + return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", mnem[i+1:]) + } + sfx.zeroing = true + case "SAE": + sfx.sae = true + case "BCST": + sfx.bcst = true + case "RN_SAE": + sfx.rounding = 0 + case "RD_SAE": + sfx.rounding = 1 + case "RU_SAE": + sfx.rounding = 2 + case "RZ_SAE": + sfx.rounding = 3 + default: + return "", sfx, fmt.Errorf("unsupported EVEX suffix %q", p) + } } - return "", false, fmt.Errorf("unsupported EVEX suffix %q", mnem[i+1:]) + if sfx.bcst && (sfx.sae || sfx.rounding >= 0) { + return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", mnem[i+1:]) + } + return base, sfx, nil +} + +// evexRound lists the instructions that accept a rounding mode or .SAE. +var evexRound = map[string]bool{ + "VADDPD": true, "VSUBPD": true, "VMULPD": true, "VDIVPD": true, + "VMINPD": true, "VMAXPD": true, + "VADDPS": true, "VSUBPS": true, "VMULPS": true, "VDIVPS": true, + "VMINPS": true, "VMAXPS": true, + "VADDSD": true, "VSUBSD": true, "VMULSD": true, "VDIVSD": true, + "VMINSD": true, "VMAXSD": true, + "VADDSS": true, "VSUBSS": true, "VMULSS": true, "VDIVSS": true, + "VMINSS": true, "VMAXSS": true, +} + +// evexBcstN maps an instruction accepting .BCST to the broadcast element +// size — the disp8×N multiplier for its memory operand. +var evexBcstN = map[string]int{ + "VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8, + "VMINPD": 8, "VMAXPD": 8, + "VADDPS": 4, "VSUBPS": 4, "VMULPS": 4, "VDIVPS": 4, + "VMINPS": 4, "VMAXPS": 4, } // splitMask extracts an explicit mask register (K1–K7) from the operand list, @@ -298,20 +501,52 @@ func splitMask(ops []Operand) ([]Operand, int, error) { // encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The // mask, when present, is an explicit K1–K7 operand anywhere among the -// operands; zeroing comes from the .Z mnemonic suffix and requires a mask. -func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error { - // Mask-destination comparisons (VPCMPEQD …, K1): the last operand is the - // destination K register, and any mask sits among the preceding operands. - if spec, ok := evexTable[mnemUpper]; ok && spec.form == vexNDS3 && len(ops) > 0 { - if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask { - rest, mask, err := splitMask(ops[:len(ops)-1]) - if err != nil { - return err +// operands; the mnemonic suffix carries zeroing, rounding/SAE and +// broadcast. +func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error { + spec, inTable := evexTable[mnemUpper] + if inTable { + if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] { + return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper) + } + if sfx.bcst { + n, ok := evexBcstN[mnemUpper] + if !ok { + return fmt.Errorf("%s: broadcast is not supported for this instruction", mnemUpper) } - if zeroing && mask == 0 { - return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper) + spec.n = [3]int{n, n, n} + } + } else if sfx.evexOnly() { + return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper) + } + + // Mask-destination comparisons (VPCMPEQD, VCMPPD $imm, …): the last + // operand is the destination K register, and any mask sits among the + // preceding operands. + kdst := func(encode func(evexSpec, []Operand, int, evexSuffix) error) error { + dst, ok := ops[len(ops)-1].(Reg) + if !ok || !dst.mask { + return nil // not a K-destination form; fall through + } + rest, mask, err := splitMask(ops[:len(ops)-1]) + if err != nil { + return err + } + if sfx.zeroing && mask == 0 { + return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper) + } + return encode(spec, append(rest, dst), mask, sfx) + } + if inTable && len(ops) > 0 { + switch spec.form { + case vexNDS3: + if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask { + return kdst(e.encodeEvexNDS3) + } + case vexNDS3Imm: + if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask { + return kdst(e.encodeEvexNDS3Imm) } - return e.encodeEvexNDS3(spec, append(rest, dst), mask, zeroing) } } @@ -319,38 +554,43 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error { if err != nil { return err } - if zeroing && mask == 0 { + if sfx.zeroing && mask == 0 { return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper) } ops = rest if bs, ok := evexBcastTable[mnemUpper]; ok { - return e.encodeEvexBcast(bs, ops, mask, zeroing) + if sfx.evexOnly() { + return fmt.Errorf("%s: broadcast instructions take no rounding/SAE/broadcast suffix", mnemUpper) + } + return e.encodeEvexBcast(bs, ops, mask, sfx) } if ms, ok := evexMoveTable[mnemUpper]; ok { - return e.encodeEvexMove(mnemUpper, ms, ops, mask, zeroing) + if sfx.evexOnly() { + return fmt.Errorf("%s: moves take no rounding/SAE/broadcast suffix", mnemUpper) + } + return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx) } - spec, ok := evexTable[mnemUpper] - if !ok { + if !inTable { return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper) } switch spec.form { case vexNDS3: - return e.encodeEvexNDS3(spec, ops, mask, zeroing) + return e.encodeEvexNDS3(spec, ops, mask, sfx) case vexRM: - return e.encodeEvexRM(spec, ops, mask, zeroing) + return e.encodeEvexRM(spec, ops, mask, sfx) case vexRMRev: - return e.encodeEvexRMRev(spec, ops, mask, zeroing) + return e.encodeEvexRMRev(spec, ops, mask, sfx) case vexImmRM: - return e.encodeEvexImmRM(spec, ops, mask, zeroing) + return e.encodeEvexImmRM(spec, ops, mask, sfx) case vexShiftImm: - return e.encodeEvexShiftImm(spec, ops, mask, zeroing) + return e.encodeEvexShiftImm(spec, ops, mask, sfx) case vexNDS3Imm: - return e.encodeEvexNDS3Imm(spec, ops, mask, zeroing) + return e.encodeEvexNDS3Imm(spec, ops, mask, sfx) case vexExtract: - return e.encodeEvexExtract(spec, ops, mask, zeroing) + return e.encodeEvexExtract(spec, ops, mask, sfx) case vexRMSrcLen: - return e.encodeEvexRMSrcLen(spec, ops, mask, zeroing) + return e.encodeEvexRMSrcLen(spec, ops, mask, sfx) } return fmt.Errorf("unhandled EVEX form for %s", mnemUpper) } @@ -358,7 +598,7 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error { // encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The // destination may be an opmask register (VPCMPEQD), in which case the vector // length comes from the sources. -func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 3 { return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops)) } @@ -378,12 +618,12 @@ func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, zeroing boo ll = r.vecLenBit() } } - return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, zeroing) + return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx) } // encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src, // no vvvv), e.g. VCVTQQ2PD. -func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 2 { return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops)) } @@ -392,12 +632,12 @@ func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, zeroing bool) if !ok || !dstReg.isVec() { return fmt.Errorf("EVEX destination must be a vector register") } - return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing) + return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx) } // encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst // (reg = dst, rm = src, imm8), e.g. VPSHUFD. -func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 3 { return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops)) } @@ -418,7 +658,7 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bo if err != nil { return err } - if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing); err != nil { + if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx); err != nil { return err } e.out = append(e.out, immByte) @@ -427,7 +667,7 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bo // encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst // (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5. -func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 3 { return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops)) } @@ -448,7 +688,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing if err != nil { return err } - if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, zeroing); err != nil { + if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil { return err } e.out = append(e.out, immByte) @@ -456,8 +696,10 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing } // encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1, -// rm=src2, imm8), e.g. VALIGND. -func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing bool) error { +// rm=src2, imm8), e.g. VALIGND. The destination may be an opmask register +// (VCMPPD and friends), in which case the vector length comes from the +// sources. +func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 4 { return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops)) } @@ -467,8 +709,8 @@ func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) - if !ok || !dstReg.isVec() { - return fmt.Errorf("destination must be a vector register") + if !ok || (!dstReg.isVec() && !dstReg.mask) { + return fmt.Errorf("destination must be a vector or mask register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { @@ -478,7 +720,14 @@ func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing if err != nil { return err } - if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2, mask, zeroing); err != nil { + ll := dstReg.vecLenBit() + if dstReg.mask { + ll = vvvvReg.vecLenBit() + if r, ok := src2.(Reg); ok && r.isVec() { + ll = r.vecLenBit() + } + } + if err := e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, sfx); err != nil { return err } e.out = append(e.out, immByte) @@ -487,7 +736,7 @@ func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing // encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory // destination, imm8), e.g. VEXTRACTI64X4. -func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 3 { return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops)) } @@ -504,7 +753,7 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing if err != nil { return err } - if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing); err != nil { + if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx); err != nil { return err } e.out = append(e.out, immByte) @@ -514,7 +763,7 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing // encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses // the store-form opcode (reg = source, rm = destination), matching the Go // assembler. -func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 2 { return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops)) } @@ -543,14 +792,14 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i return fmt.Errorf("%s needs a vector register operand", mnem) } spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n} - return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, zeroing) + return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx) } // encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with // the destination always XMM and the length fixed by the mnemonic — the // single valid slot of spec.n names the vector length (and the disp8×N // multiplier) a register or memory source encodes. -func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 2 { return fmt.Errorf("conversion expects 2 operands, got %d", len(ops)) } @@ -563,7 +812,7 @@ func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing if err != nil { return err } - return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing) + return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx) } // soleLen returns the vector-length index of the single valid slot of n — @@ -598,7 +847,7 @@ func memOperand(op Operand) bool { // encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide // source in the reg field and the narrow destination in r/m (VPMOVDW/QD). -func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 2 { return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops)) } @@ -607,12 +856,12 @@ func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, zeroing bo if !ok || !srcReg.isVec() { return fmt.Errorf("EVEX source must be a vector register") } - return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing) + return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, sfx) } // encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory // source broadcast to every lane of the vector destination. -func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing bool) error { +func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx evexSuffix) error { if len(ops) != 2 { return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops)) } @@ -631,7 +880,7 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing default: return fmt.Errorf("broadcast source must be a register or memory") } - return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing) + return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, sfx) } // emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement @@ -639,7 +888,7 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing // unextended reg-field register index, or a /digit (0–7); vvvvIdx is the // vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and // zeroing fill the aaa and z bits of the P2 byte. -func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, zeroing bool) error { +func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error { if ll > 2 { return fmt.Errorf("invalid vector length") } @@ -702,12 +951,22 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, } z := 0 - if zeroing { + if sfx.zeroing { z = 1 } + // The b bit and the L'L field carry the rounding/SAE/broadcast mode: + // a rounding mode replaces L'L with the rc value, plain SAE and + // broadcast keep the vector length. + b, ll := 0, ll + switch { + case sfx.rounding >= 0: + b, ll = 1, sfx.rounding + case sfx.sae || sfx.bcst: + b = 1 + } p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel) p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp) - p2 := byte(z<<7 | ll<<5 | vPrimeBar<<3 | mask) // z, L'L, b=0, V', aaa + p2 := byte(z<<7 | ll<<5 | b<<4 | vPrimeBar<<3 | mask) // z, L'L/rc, b, V', aaa e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm)) if sib >= 0 { e.out = append(e.out, byte(sib)) @@ -775,24 +1034,40 @@ func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil } -// encodeKmovw encodes KMOVW, whose opcode depends on the operand direction: -// 90 (k/mem → K), 91 (K → mem), 92 (GPR → K), 93 (K → GPR); k → k uses 90. -func (e *enc) encodeKmovw(ops []Operand) error { +// kmovSpec describes a KMOV width: the opcode depends on the operand +// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem), +// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory +// prefix and W for the wider widths. +type kmovSpec struct { + kk, kmem, gprk, kgpr byte + gprPP int + w int +} + +var kmovTable = map[string]kmovSpec{ + "KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0}, + "KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1}, +} + +// encodeKmov encodes a KMOV width, selecting the opcode by direction. +func (e *enc) encodeKmov(upper string, ops []Operand) error { if len(ops) != 2 { - return fmt.Errorf("KMOVW expects 2 operands, got %d", len(ops)) + return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) } + ks := kmovTable[upper] src, dst := ops[0], ops[1] srcReg, srcIsReg := src.(Reg) dstReg, dstIsReg := dst.(Reg) srcK := srcIsReg && srcReg.mask dstK := dstIsReg && dstReg.mask - spec := vexSpec{mapSel: 1, w: 0, pp: 0, opdigit: -1} + spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1} switch { case srcK && dstK: - spec.opcode = 0x90 // k ← k: reg = dst, rm = src + spec.opcode = ks.kk // k ← k: reg = dst, rm = src return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src) case srcK && dstIsReg: - spec.opcode = 0x93 // GPR ← k: reg = dst, rm = src + spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src + spec.pp = ks.gprPP rBit := 0 if dstReg.idx >= 8 { rBit = 1 @@ -800,13 +1075,126 @@ func (e *enc) encodeKmovw(ops []Operand) error { return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src) case srcK: if _, ok := dst.(Mem); !ok { - return fmt.Errorf("KMOVW: invalid destination operand") + return fmt.Errorf("%s: invalid destination operand", upper) } - spec.opcode = 0x91 // mem ← k: reg = src, rm = dst + spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst) case dstK: - spec.opcode = 0x92 // k ← GPR/mem: reg = dst, rm = src + spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src + spec.pp = ks.gprPP return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src) } - return fmt.Errorf("KMOVW requires a K register operand") + return fmt.Errorf("%s requires a K register operand", upper) +} + +// kOpSpec describes the VEX encoding of an opmask-register instruction: the +// L bit and the W/pp pair select the operand width, and the form the +// operand layout. +type kOpSpec struct { + mapSel int + opcode byte + w int + pp int + ll int + form vexForm +} + +var kOpsTable = map[string]kOpSpec{ + // k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask + // registers). + "KANDB": {1, 0x41, 0, 1, 1, vexNDS3}, + "KANDW": {1, 0x41, 0, 0, 1, vexNDS3}, + "KANDQ": {1, 0x41, 1, 0, 1, vexNDS3}, + "KORB": {1, 0x45, 0, 1, 1, vexNDS3}, + "KORD": {1, 0x45, 1, 1, 1, vexNDS3}, + "KXNORW": {1, 0x46, 0, 0, 1, vexNDS3}, + "KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3}, + "KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3}, + "KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3}, + "KADDB": {1, 0x4A, 0, 1, 1, vexNDS3}, + "KADDW": {1, 0x4A, 0, 0, 1, vexNDS3}, + "KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3}, + // k ← OP k (KNOT) and flags ← k OP k (KORTEST): reg = dst, rm = src. + "KNOTB": {1, 0x44, 0, 1, 0, vexRM}, + "KORTESTD": {1, 0x98, 1, 1, 0, vexRM}, + // OP $imm, src, dst: reg = dst, rm = src, imm8. + "KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM}, +} + +// isKOp reports whether the mnemonic is an opmask-register instruction. +func isKOp(upper string) bool { + _, ok := kOpsTable[upper] + return ok +} + +// encodeKOp encodes an opmask-register instruction; every operand is a K +// register and the vector length is fixed by the instruction. +func (e *enc) encodeKOp(upper string, ops []Operand) error { + ks := kOpsTable[upper] + spec := vexSpec{mapSel: ks.mapSel, opcode: ks.opcode, w: ks.w, pp: ks.pp, opdigit: -1} + kreg := func(op Operand, what string) (Reg, error) { + r, ok := op.(Reg) + if !ok || !r.mask { + return Reg{}, fmt.Errorf("%s: %s must be an opmask register", upper, what) + } + return r, nil + } + switch ks.form { + case vexNDS3: + if len(ops) != 3 { + return fmt.Errorf("%s expects 3 operands, got %d", upper, len(ops)) + } + src2, err := kreg(ops[0], "first source") + if err != nil { + return err + } + src1, err := kreg(ops[1], "second source") + if err != nil { + return err + } + dst, err := kreg(ops[2], "destination") + if err != nil { + return err + } + return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15-src1.idx, src2) + case vexRM: + if len(ops) != 2 { + return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops)) + } + src, err := kreg(ops[0], "source") + if err != nil { + return err + } + dst, err := kreg(ops[1], "destination") + if err != nil { + return err + } + return e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src) + case vexImmRM: + if len(ops) != 3 { + return fmt.Errorf("%s expects 3 operands ($imm, src, dst), got %d", upper, len(ops)) + } + immVal, ok := ops[0].(Imm) + if !ok { + return fmt.Errorf("%s: shift count must be an immediate", upper) + } + src, err := kreg(ops[1], "source") + if err != nil { + return err + } + dst, err := kreg(ops[2], "destination") + if err != nil { + return err + } + immByte, err := imm8(int64(immVal)) + if err != nil { + return err + } + if err := e.emitVexFields(spec, ks.ll, dst.idx, 0, 15, src); err != nil { + return err + } + e.out = append(e.out, immByte) + return nil + } + return fmt.Errorf("unhandled opmask form for %s", upper) } diff --git a/asm/evex_test.go b/asm/evex_test.go index 39327bb..74e0b98 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -230,7 +230,11 @@ func TestEvexMasking(t *testing.T) { {"K0 mask", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K0"), vreg(t, "Z3")}}, {"two masks", "VPADDD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "K2"), vreg(t, "Z3")}}, {".Z on VEX-only", "VPSHUFD.Z", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}}, - {"unsupported suffix", "VPADDD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}}, + {"broadcast unsupported", "VPXORD.BCST", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}}, + {"rounding unsupported", "VPXORD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}}, + {"bcst with rounding", "VADDPD.BCST.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}}, + {"Z not last", "VADDPD.Z.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}}, + {"duplicate suffix", "VADDPD.Z.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}}, {"KMOVW.Z", "KMOVW.Z", []Operand{vreg(t, "K1"), vreg(t, "K2")}}, } for _, c := range bad { @@ -240,6 +244,126 @@ func TestEvexMasking(t *testing.T) { } } +// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary +// logic, lane shuffles/inserts/extracts, compares with a K destination, +// permutes, the wider integer families, expand/compress, broadcasts, +// rotates and word shifts, the opmask instructions, the EVEX suffixes +// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte +// against the Go assembler. +func TestEvexExtendedGroundTruth(t *testing.T) { + mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) } + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + // Ternary logic and lane shuffles (NDS + imm8). + {"VPTERNLOGD", "VPTERNLOGD", []Operand{Imm(0xE8), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d4825d9e8"}, + {"VPTERNLOGQ", "VPTERNLOGQ", []Operand{Imm(0x96), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4825d996"}, + {"VSHUFI32X4", "VSHUFI32X4", []Operand{Imm(0x4E), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "62f36d2843d94e"}, + {"VSHUFF64X2", "VSHUFF64X2", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed4823d901"}, + {"VPALIGNR", "VPALIGNR", []Operand{Imm(7), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d480fd907"}, + // Permutes. + {"VPERMB", "VPERMB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d488dd9"}, + {"VPERMW", "VPERMW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed488dd9"}, + {"VPERMI2D", "VPERMI2D", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4876d9"}, + {"VPERMT2PD", "VPERMT2PD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed487fd9"}, + // Compare with a K destination (and an immediate predicate). + {"VCMPPD", "VCMPPD", []Operand{Imm(4), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f1ed48c2d904"}, + {"VCMPPS", "VCMPPS", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K4")}, "62f16c28c2e100"}, + {"VCMPSD", "VCMPSD", []Operand{Imm(17), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K5")}, "62f1ef08c2e911"}, + // Rounding / SAE / broadcast suffixes. + {"VADDPD.RN_SAE", "VADDPD.RN_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed1858d9"}, + {"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"}, + {"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"}, + {"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"}, + // Packed single arithmetic (same opcodes, no mandatory prefix) — + // ZMM, YMM and XMM widths, rounding and broadcast. + {"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"}, + {"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"}, + {"VMAXPS", "VMAXPS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e85fd9"}, + {"VDIVPS.RD_SAE", "VDIVPS.RD_SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c385ed9"}, + {"VADDPS.BCST", "VADDPS.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f174585810"}, + // Compress / expand. + {"VCOMPRESSPD", "VCOMPRESSPD", []Operand{vreg(t, "Z1"), mem64(DI)}, "62f2fd488a0f"}, + {"VEXPANDPS", "VEXPANDPS", []Operand{mem64(SI), vreg(t, "Y2")}, "62f27d288816"}, + {"VPCOMPRESSD.Z", "VPCOMPRESSD.Z", []Operand{vreg(t, "Z1"), vreg(t, "K2"), mem64(DI)}, "62f27dca8b0f"}, + // Broadcasts. + {"VPBROADCASTB gpr", "VPBROADCASTB", []Operand{BX, vreg(t, "Z1")}, "62f27d487acb"}, + {"VPBROADCASTW mem", "VPBROADCASTW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d487910"}, + {"VBROADCASTSS", "VBROADCASTSS", []Operand{mem64(AX), vreg(t, "Y3")}, "c4e27d1818"}, + {"VBROADCASTSD", "VBROADCASTSD", []Operand{mem64(AX), vreg(t, "Z4")}, "62f2fd481920"}, + // Wider integer families. + {"VPMADDWD", "VPMADDWD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48f5d9"}, + {"VPMADDUBSW", "VPMADDUBSW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4804d9"}, + {"VPMULHUW", "VPMULHUW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d48e4d9"}, + {"VPSLLVW", "VPSLLVW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4812d9"}, + {"VPACKSSWB", "VPACKSSWB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16d4863d9"}, + {"VPACKUSDW", "VPACKUSDW", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d482bd9"}, + // Absolute values and replicating moves. + {"VPABSD", "VPABSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f27d481ed1"}, + {"VPABSQ mem", "VPABSQ", []Operand{mem64(AX), vreg(t, "Z2")}, "62f2fd481f10"}, + {"VMOVSLDUP", "VMOVSLDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa12d1"}, + {"VMOVSHDUP", "VMOVSHDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e4816d1"}, + // Rotates and word/qword shifts. + {"VPROLD", "VPROLD", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2")}, "62f16d4872c905"}, + {"VPRORQ", "VPRORQ", []Operand{Imm(63), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4872c13f"}, + {"VPSLLW", "VPSLLW", []Operand{Imm(9), vreg(t, "X1"), vreg(t, "X2")}, "c5e971f109"}, + {"VPSRLQ", "VPSRLQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ed4873d103"}, + // Opmask instructions (VEX-encoded, the width in the L/W/pp bits). + {"KANDW", "KANDW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec41d9"}, + {"KORD", "KORD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d545f4"}, + {"KXNORQ", "KXNORQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec46d9"}, + {"KNOTB", "KNOTB", []Operand{vreg(t, "K4"), vreg(t, "K5")}, "c5f944ec"}, + {"KUNPCKBW", "KUNPCKBW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed4bd9"}, + {"KSHIFTLW", "KSHIFTLW", []Operand{Imm(2), vreg(t, "K1"), vreg(t, "K2")}, "c4e3f932d102"}, + {"KADDQ", "KADDQ", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ec4ad9"}, + {"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"}, + {"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"}, + {"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"}, + // Lane extract / insert. + {"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"}, + {"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"}, + {"VINSERTF32X8", "VINSERTF32X8", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f36d481ad901"}, + {"VINSERTI64X4", "VINSERTI64X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f3ed483ad901"}, + // Aligned moves and the scalar single move. + {"VMOVAPS", "VMOVAPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17c4829ca"}, + {"VMOVDQA64 mem", "VMOVDQA64", []Operand{mem64(AX), vreg(t, "Z2")}, "62f1fd486f10"}, + {"VMOVSS mem", "VMOVSS", []Operand{mem64(AX), vreg(t, "X2")}, "c5fa1010"}, + // Conversions and extending/narrowing moves. + {"VCVTPS2DQ", "VCVTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17d485bd1"}, + {"VCVTTPS2DQ", "VCVTTPS2DQ", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17e485bd1"}, + {"VPMOVZXBW", "VPMOVZXBW", []Operand{vreg(t, "X1"), vreg(t, "Y2")}, "c4e27d30d1"}, + {"VPMOVSXBW mem", "VPMOVSXBW", []Operand{mem64(AX), vreg(t, "Z2")}, "62f27d482010"}, + {"VPMOVWB", "VPMOVWB", []Operand{vreg(t, "Z1"), vreg(t, "Y2")}, "62f27e4830ca"}, + {"VPMOVQB", "VPMOVQB", []Operand{vreg(t, "Z1"), vreg(t, "X2")}, "62f27e4832ca"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: Encode: %v", c.name, err) + continue + } + if got := hexCompact(code); got != c.want { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + continue + } + inst, err := x86asm.Decode(code, 64) + if err != nil { + t.Errorf("%s: Decode(%x): %v", c.name, code, err) + continue + } + want := c.mnem + if i := strings.IndexByte(want, '.'); i > 0 { + want = want[:i] + } + if inst.Op.String() != want { + t.Errorf("%s: decoded as %s", c.name, inst.Op.String()) + } + } +} + // TestEvexErrors checks the EVEX-specific error paths. func TestEvexErrors(t *testing.T) { cases := []struct { diff --git a/asm/vex.go b/asm/vex.go index 2719756..b78ae25 100644 --- a/asm/vex.go +++ b/asm/vex.go @@ -91,12 +91,19 @@ var vexTable = map[string]vexSpec{ "VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3}, // VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic. - "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, - "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, - "VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3}, - "VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3}, - "VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3}, - "VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3}, + "VADDPD": {1, 0x58, 0, 1, -1, vexNDS3}, + "VMULPD": {1, 0x59, 0, 1, -1, vexNDS3}, + "VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3}, + "VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3}, + "VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3}, + "VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3}, + // VEX.128/256.0F.WIG — packed single-precision arithmetic. + "VADDPS": {1, 0x58, 0, 0, -1, vexNDS3}, + "VMULPS": {1, 0x59, 0, 0, -1, vexNDS3}, + "VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3}, + "VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3}, + "VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3}, + "VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3}, "VXORPD": {1, 0x57, 0, 1, -1, vexNDS3}, "VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3}, "VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3}, @@ -123,7 +130,9 @@ var vexTable = map[string]vexSpec{ // no vvvv). "VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM}, "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM}, + "VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM}, "VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM}, + "VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM}, "VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM}, "VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM}, // VEX.128/256.F3.0F.WIG — signed dword to packed double conversion @@ -176,6 +185,19 @@ var vexTable = map[string]vexSpec{ // VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src). "KTESTW": {1, 0x99, 0, 0, -1, vexRM}, + // VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst, + // rm=scalar memory; SD is 256-bit only). + "VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM}, + "VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM}, + // VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src). + "VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM}, + "VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM}, + + // VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift). + "VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm}, + "VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm}, + "VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm}, + // VEX.F2.0F — packed double to packed dword conversions, truncating and // non-truncating. The destination is always XMM; the X/Y spellings fix // the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen. @@ -238,6 +260,11 @@ var vexMoveTable = map[string]vexMoveSpec{ // VEX.128.F2.0F.WIG — scalar double move, memory operands only (the // register form takes three operands and is not supported yet). "VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, + // VEX.128.F3.0F.WIG — scalar single move, memory operands only. + "VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true}, + // VEX.128/256 — aligned packed moves. + "VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, + "VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false}, } // isVex reports whether the mnemonic is a VEX-encoded instruction we handle. diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index c6b5904..9d38ba2 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -28,7 +28,7 @@ import ( // version is the release version, stamped at build time via // -ldflags "-X main.version=…" (defaulting to the current release). -var version = "0.12.0" +var version = "0.13.0" func main() { if len(os.Args) < 2 { diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index cb41b45..0467410 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -228,11 +228,21 @@ explicit merging/zeroing masks — written the way Go writes them, as a K operand among the operands plus a `.Z` mnemonic suffix), and the compressed disp8×N displacement, whose multiplier follows the memory operand's size — covering every instruction the go-flac and go-lz4 AVX2/AVX-512 kernels use, -plus the common AVX-512 F/BW integer set and the floating-point and -conversion set (the packed double arithmetic, the scalar SD/SS forms — -whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, and the -width-changing conversions, including the `VCVTPD2DQ`/`VCVTTPD2DQ` family -whose length follows the wider source operand). Every encoding is validated two ways: by +plus the common AVX-512 F/BW integer set, the floating-point and conversion +set (the packed double and single arithmetic, the scalar SD/SS forms — +whose EVEX encodings serve masked and zeroing use — `VMOVDDUP`, the +replicating moves, and the width-changing conversions, including the +`VCVTPD2DQ`/`VCVTTPD2DQ` family whose length follows the wider source +operand), and the wider AVX-512 set: ternary logic, lane shuffles, inserts +and extracts, compares with an opmask destination, the permutes, the +expand/compress family, the broadcasts, the opmask-register instructions +(KAND/KOR/KXNOR/KADD/KUNPCK/KNOT/KSHIFTL/KORTEST and KMOVQ), the aligned +moves and the remaining extending/narrowing moves. The EVEX mnemonic +suffixes — rounding modes (.RN_SAE/.RD_SAE/.RU_SAE/.RZ_SAE), +suppress-all-exceptions (.SAE) and memory broadcast (.BCST) — set the EVEX +b bit and the L'L rounding-control field (broadcast keeps the vector length +and scales disp8 by the element size), and combine with the .Z zeroing +suffix. Every encoding is validated two ways: by round-trip decoding through `golang.org/x/arch`, and byte-for-byte against the machine code the real Go assembler emits — a comparison that holds for whole functions: all 27 functions of both kernels assemble to exactly the Go diff --git a/justfile b/justfile index e778886..9c4a723 100644 --- a/justfile +++ b/justfile @@ -3,7 +3,7 @@ # gasm-devkit — developer tooling for Go's Plan 9 assembler (GAsm). -version := "0.12.0" +version := "0.13.0" default: @just --list