// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "fmt" "strings" ) // This file implements EVEX (AVX-512) instruction encoding: the four-byte // EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the // compressed disp8×N displacement, and the operand shapes the go-flac // AVX-512 kernels use plus the common floating-point and conversion set. // Masking follows the Go assembler's spelling: an explicit K1–K7 operand // anywhere among the operands (merging) plus a ".Z" mnemonic suffix for // zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are // supported too. // evexSpec describes one EVEX instruction's encoding parameters. The form // field reuses the vexForm shapes, which carry over unchanged. type evexSpec struct { mapSel int // 1 = 0F, 2 = 0F38, 3 = 0F3A opcode byte w int pp int // 0 = none, 1 = 66, 2 = F3, 3 = F2 opdigit int // ModRM.reg /digit, or -1 when reg is a register form vexForm // vexNDS3, vexRM, vexShiftImm, vexNDS3Imm, vexExtract n [3]int // disp8×N multiplier per vector length (128/256/512) } // evexTable maps an upper-case mnemonic to its EVEX encoding. Mnemonics // that also have a VEX form (VPADDD, VMOVUPD, …) are dispatched here only // when an operand demands EVEX (a ZMM or K register); EVEX-only mnemonics // (VPXORD, VALIGND, …) always encode through this table. The N multipliers // are taken from the Go assembler's opcode tables, which are authoritative // for byte-for-byte agreement. var evexTable = map[string]evexSpec{ // EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form. "VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSUBQ": {1, 0xFB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPUNPCKLDQ": {1, 0x62, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPXORD": {1, 0xEF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPXORQ": {1, 0xEF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — packed double arithmetic. "VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — packed double unpack. "VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with // an F2 pp; the EVEX forms exist for masked and zeroing use). The // memory operand is a single double, so disp8×N = 8. "VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VSUBSD": {1, 0x5C, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VMULSD": {1, 0x59, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VDIVSD": {1, 0x5E, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, "VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}}, // EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4). "VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, "VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}}, // EVEX.512.66.0F3A — align (NDS + imm8). "VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4). "VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ; // the W bit distinguishes it from VPSRAD's E2 form). "VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst, // rm=src, no vvvv). "VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}}, // EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst, // rm=src, no vvvv): a 128-bit destination reads a single double from // memory (disp8×8), the wider ones read the full operand. "VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}}, // EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst, // rm=src, no vvvv, no mandatory prefix — as in the VEX form). "VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}}, // EVEX.128/256/512.0F.W0 — packed single to packed double: the // destination is twice the source width and sets the length; disp8×N // follows the narrow memory source. No F3 prefix: the Go assembler // emits this instruction with pp = 00 (Intel's maps would call that // undefined) and gasm reproduces the Go assembler's bytes — its machine // code is the oracle, not the manual. "VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}}, // EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX // form of the VEX instruction; the destination sets the length, disp8×N // follows the narrow memory source). "VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}}, // EVEX packed double → dword conversions: the source is the wide // operand and the mnemonic fixes the length — the bare names are // 512-bit only (ZMM source, XMM destination), the X/Y spellings are // EVEX-128/256. Exactly one slot of n is valid; it names the vector // length (and the disp8×N multiplier) a register or memory source // encodes. "VCVTPD2DQ": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 0, 64}}, "VCVTTPD2DQ": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 0, 64}}, "VCVTPD2DQX": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{16, 0, 0}}, "VCVTPD2DQY": {1, 0xE6, 1, 3, -1, vexRMSrcLen, [3]int{0, 32, 0}}, "VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}}, "VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}}, // EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory // operand is the narrow source, so disp8×N follows its size (8/16/32 for // the xmm/ymm/zmm destination lengths). "VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}}, // EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory // destination, imm8). "VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}}, "VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}}, // EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q). "VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}}, // EVEX.128/256/512 — the wider integer set (AVX-512 F/BW): byte/word // arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and // variable shifts. All NDS form; W distinguishes element size. "VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPADDW": {1, 0xFD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSUBB": {1, 0xF8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSUBW": {1, 0xF9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMULLW": {1, 0xD5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPAVGB": {1, 0xE0, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPAVGW": {1, 0xE3, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMINUB": {1, 0xDA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMAXUB": {1, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMINSW": {1, 0xEA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMAXSW": {1, 0xEE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPANDD": {1, 0xDB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPANDQ": {1, 0xDB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPANDND": {1, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPANDNQ": {1, 0xDF, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMINSB": {2, 0x38, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMAXSB": {2, 0x3C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMINSQ": {2, 0x39, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMAXSQ": {2, 0x3D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMINUW": {2, 0x3A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMAXUW": {2, 0x3E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMINSD": {2, 0x39, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMAXSD": {2, 0x3D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMINUD": {2, 0x3B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMAXUD": {2, 0x3F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMINUQ": {2, 0x3B, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPMAXUQ": {2, 0x3F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSLLVD": {2, 0x47, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSLLVQ": {2, 0x47, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRLVD": {2, 0x45, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRLVQ": {2, 0x45, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRAVD": {2, 0x46, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, "VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX forms of instructions that also exist in VEX (selected when a ZMM // or K register, or indices 16–31, demand EVEX). "VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}}, "VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}}, // EVEX.66.0F — immediate shift (VPSLLD /6). "VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}}, // EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow // destination (VPMOVDW dword→word, VPMOVQD qword→dword). "VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, "VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}}, } // evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode // depends on the source kind — a GPR source uses opReg, a memory source uses // opMem with a disp8×N of n. type evexBcastSpec struct { mapSel int opReg byte opMem byte w int n int } var evexBcastTable = map[string]evexBcastSpec{ // EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes. "VPBROADCASTD": {2, 0x7C, 0x58, 0, 4}, "VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8}, } // evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX // move table). type evexMoveSpec struct { mapSel int pp int load byte // r/m → vector store byte // vector → r/m w int n [3]int } // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. var evexMoveTable = map[string]evexMoveSpec{ // EVEX.128/256/512.F3.0F.W0 — unaligned integer move. "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, // EVEX.128/256/512.F3.0F.W1 — unaligned qword move. "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, // EVEX.128/256/512.F2.0F.W0 — unaligned byte move (byte/word moves use the // F2 prefix, dword/qword moves F3; the element size only changes the tuple // semantics). "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, // EVEX.128/256/512.F2.0F.W1 — unaligned word move (shares the qword // encoding). "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, // EVEX.128/256/512.66.0F.W1 — unaligned packed double move. "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}}, } // isEvex reports whether the mnemonic has an EVEX encoding we handle. func isEvex(mnemUpper string) bool { if _, ok := evexTable[mnemUpper]; ok { return true } if _, ok := evexBcastTable[mnemUpper]; ok { return true } _, ok := evexMoveTable[mnemUpper] return ok } // evexRequired reports whether the operands force the EVEX encoding of a // mnemonic that also has a VEX form: ZMM and K registers do, and so do // register indices 16–31, which only EVEX can represent (X16–Y31 exist // solely under AVX-512). func evexRequired(upper string, ops []Operand) bool { _, inVex := vexTable[upper] _, inVexMove := vexMoveTable[upper] if !inVex && !inVexMove { return true // EVEX-only mnemonic } for _, op := range ops { if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) { return true } } return false } // stripEvexSuffix splits a ".Z" zeroing suffix off the mnemonic. It is the // only EVEX suffix supported; Go writes masking as an explicit K operand, not // a suffix. func stripEvexSuffix(mnem string) (base string, zeroing bool, err error) { i := strings.LastIndexByte(mnem, '.') if i < 0 { return mnem, false, nil } if mnem[i+1:] == "Z" { return mnem[:i], true, nil } return "", false, fmt.Errorf("unsupported EVEX suffix %q", mnem[i+1:]) } // splitMask extracts an explicit mask register (K1–K7) from the operand list, // returning the remaining operands and the mask index. K0 is not a usable // mask (aaa = 0 means "no mask"), matching the assembler. func splitMask(ops []Operand) ([]Operand, int, error) { var rest []Operand mask := 0 for _, op := range ops { if r, ok := op.(Reg); ok && r.mask { if mask != 0 { return nil, 0, fmt.Errorf("at most one mask register operand") } if r.idx == 0 { return nil, 0, fmt.Errorf("K0 is not a usable mask register") } mask = r.idx continue } rest = append(rest, op) } return rest, mask, nil } // encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The // mask, when present, is an explicit K1–K7 operand anywhere among the // operands; zeroing comes from the .Z mnemonic suffix and requires a mask. func (e *enc) encodeEvex(mnemUpper string, ops []Operand, zeroing bool) error { // Mask-destination comparisons (VPCMPEQD …, K1): the last operand is the // destination K register, and any mask sits among the preceding operands. if spec, ok := evexTable[mnemUpper]; ok && spec.form == vexNDS3 && len(ops) > 0 { if dst, ok := ops[len(ops)-1].(Reg); ok && dst.mask { rest, mask, err := splitMask(ops[:len(ops)-1]) if err != nil { return err } if zeroing && mask == 0 { return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper) } return e.encodeEvexNDS3(spec, append(rest, dst), mask, zeroing) } } rest, mask, err := splitMask(ops) if err != nil { return err } if zeroing && mask == 0 { return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnemUpper) } ops = rest if bs, ok := evexBcastTable[mnemUpper]; ok { return e.encodeEvexBcast(bs, ops, mask, zeroing) } if ms, ok := evexMoveTable[mnemUpper]; ok { return e.encodeEvexMove(mnemUpper, ms, ops, mask, zeroing) } spec, ok := evexTable[mnemUpper] if !ok { return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper) } switch spec.form { case vexNDS3: return e.encodeEvexNDS3(spec, ops, mask, zeroing) case vexRM: return e.encodeEvexRM(spec, ops, mask, zeroing) case vexRMRev: return e.encodeEvexRMRev(spec, ops, mask, zeroing) case vexImmRM: return e.encodeEvexImmRM(spec, ops, mask, zeroing) case vexShiftImm: return e.encodeEvexShiftImm(spec, ops, mask, zeroing) case vexNDS3Imm: return e.encodeEvexNDS3Imm(spec, ops, mask, zeroing) case vexExtract: return e.encodeEvexExtract(spec, ops, mask, zeroing) case vexRMSrcLen: return e.encodeEvexRMSrcLen(spec, ops, mask, zeroing) } return fmt.Errorf("unhandled EVEX form for %s", mnemUpper) } // encodeEvexNDS3 encodes the three-operand NDS form: OP src2, src1, dst. The // destination may be an opmask register (VPCMPEQD), in which case the vector // length comes from the sources. func (e *enc) encodeEvexNDS3(spec evexSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 3 { return fmt.Errorf("EVEX NDS instruction expects 3 operands, got %d", len(ops)) } src2, src1, dst := ops[0], ops[1], ops[2] dstReg, ok := dst.(Reg) if !ok || (!dstReg.isVec() && !dstReg.mask) { return fmt.Errorf("EVEX destination must be a vector or mask register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { return fmt.Errorf("EVEX vvvv operand must be a vector register") } ll := dstReg.vecLenBit() if dstReg.mask { ll = vvvvReg.vecLenBit() if r, ok := src2.(Reg); ok && r.isVec() { ll = r.vecLenBit() } } return e.emitEvexFields(spec, ll, dstReg.idx, vvvvReg.idx, src2, mask, zeroing) } // encodeEvexRM encodes the two-operand form: OP src, dst (reg=dst, rm=src, // no vvvv), e.g. VCVTQQ2PD. func (e *enc) encodeEvexRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 2 { return fmt.Errorf("EVEX two-operand instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("EVEX destination must be a vector register") } return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing) } // encodeEvexImmRM encodes the immediate shuffle form: OP $imm, src, dst // (reg = dst, rm = src, imm8), e.g. VPSHUFD. func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 3 { return fmt.Errorf("shuffle expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("shuffle destination must be a vector register") } ll := dstReg.vecLenBit() if r, ok := src.(Reg); ok && r.isVec() { ll = r.vecLenBit() } immByte, err := imm8(int64(immVal)) if err != nil { return err } if err := e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing); err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeEvexShiftImm encodes an immediate shift: OP $imm, src, dst // (ModRM.reg = /digit, vvvv = dst, rm = src, imm8), e.g. VPSRAD $31, Z3, Z5. func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 3 { return fmt.Errorf("EVEX shift expects 3 operands ($imm, src, dst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shift count must be an immediate") } srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("shift source must be a vector register") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("shift destination must be a vector register") } immByte, err := imm8(int64(immVal)) if err != nil { return err } if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, zeroing); err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeEvexNDS3Imm encodes OP $imm, src2, src1, dst (reg=dst, vvvv=src1, // rm=src2, imm8), e.g. VALIGND. func (e *enc) encodeEvexNDS3Imm(spec evexSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 4 { return fmt.Errorf("instruction expects 4 operands ($imm, src2, src1, dst), got %d", len(ops)) } imm, src2, src1, dst := ops[0], ops[1], ops[2], ops[3] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("shuffle control must be an immediate") } dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("destination must be a vector register") } vvvvReg, ok := src1.(Reg) if !ok || !vvvvReg.isVec() { return fmt.Errorf("second source must be a vector register") } immByte, err := imm8(int64(immVal)) if err != nil { return err } if err := e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, vvvvReg.idx, src2, mask, zeroing); err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeEvexExtract encodes OP $imm, zsrc, ydst (reg=ZMM source, rm=YMM/memory // destination, imm8), e.g. VEXTRACTI64X4. func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 3 { return fmt.Errorf("extract expects 3 operands ($imm, zsrc, ydst), got %d", len(ops)) } imm, src, dst := ops[0], ops[1], ops[2] immVal, ok := imm.(Imm) if !ok { return fmt.Errorf("extract lane must be an immediate") } srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("extract source must be a vector register") } immByte, err := imm8(int64(immVal)) if err != nil { return err } if err := e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing); err != nil { return err } e.out = append(e.out, immByte) return nil } // encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses // the store-form opcode (reg = source, rm = destination), matching the Go // assembler. func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 2 { return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcIsVec := vecReg(src) dstReg, dstIsVec := vecReg(dst) op := ms.store var reg Reg var rm Operand switch { case srcIsVec && dstIsVec: reg, rm = srcReg, dst case srcIsVec: if !memOperand(dst) { return fmt.Errorf("%s: invalid destination operand", mnem) } reg, rm = srcReg, dst case dstIsVec: if !memOperand(src) { return fmt.Errorf("%s: invalid source operand", mnem) } op = ms.load reg, rm = dstReg, src default: return fmt.Errorf("%s needs a vector register operand", mnem) } spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n} return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, zeroing) } // encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with // the destination always XMM and the length fixed by the mnemonic — the // single valid slot of spec.n names the vector length (and the disp8×N // multiplier) a register or memory source encodes. func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 2 { return fmt.Errorf("conversion expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("EVEX destination must be a vector register") } ll, err := soleLen(spec.n) if err != nil { return err } return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, zeroing) } // soleLen returns the vector-length index of the single valid slot of n — // the length a length-fixed mnemonic (the EVEX conversion spellings) encodes // regardless of its operands. func soleLen(n [3]int) (int, error) { ll := -1 for i, v := range n { if v == 0 { continue } if ll >= 0 { return 0, fmt.Errorf("ambiguous vector-length table %v", n) } ll = i } if ll < 0 { return 0, fmt.Errorf("empty vector-length table") } return ll, nil } // memOperand reports whether op is a memory reference (including a // static-symbol reference). func memOperand(op Operand) bool { switch op.(type) { case Mem, sbMem: return true } return false } // encodeEvexRMRev encodes the narrowing-store form: OP src, dst with the wide // source in the reg field and the narrow destination in r/m (VPMOVDW/QD). func (e *enc) encodeEvexRMRev(spec evexSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 2 { return fmt.Errorf("EVEX store instruction expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, ok := src.(Reg) if !ok || !srcReg.isVec() { return fmt.Errorf("EVEX source must be a vector register") } return e.emitEvexFields(spec, srcReg.vecLenBit(), srcReg.idx, -1, dst, mask, zeroing) } // encodeEvexBcast encodes VPBROADCASTD/Q: OP src, dst with the GPR or memory // source broadcast to every lane of the vector destination. func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, zeroing bool) error { if len(ops) != 2 { return fmt.Errorf("broadcast expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] dstReg, ok := dst.(Reg) if !ok || !dstReg.isVec() { return fmt.Errorf("broadcast destination must be a vector register") } spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1} switch src.(type) { case Mem, sbMem: spec.opcode = bs.opMem spec.n = [3]int{bs.n, bs.n, bs.n} case Reg: spec.opcode = bs.opReg default: return fmt.Errorf("broadcast source must be a register or memory") } return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, -1, src, mask, zeroing) } // emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement // (disp8×N compressed) for the given precomputed fields. regIdx is the // unextended reg-field register index, or a /digit (0–7); vvvvIdx is the // vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and // zeroing fill the aaa and z bits of the P2 byte. func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, zeroing bool) error { if ll > 2 { return fmt.Errorf("invalid vector length") } // reg-field extension bits (R̄, R'̄), inverted. rBar, rPrimeBar := 1, 1 if regIdx&8 != 0 { rBar = 0 } if regIdx&16 != 0 { rPrimeBar = 0 } // vvvv (inverted) and its extension bit V'̄. vBar, vPrimeBar := 15, 1 if vvvvIdx >= 0 { vBar = 15 - (vvvvIdx & 15) if vvvvIdx&16 != 0 { vPrimeBar = 0 } } var modrm, sib int var disp []byte xBar, bBar := 1, 1 var sb *sbRef switch r := rm.(type) { case Reg: // ModRM.mod = 11: rm[3] extends via B̄, and rm[4] via X̄ (the EVEX // register-register quirk). modrm = 0xC0 | (regIdx&7)<<3 | (r.idx & 7) sib = -1 if r.idx&8 != 0 { bBar = 0 } if r.idx&16 != 0 { xBar = 0 } if r.idx&16 != 0 { xBar = 0 } case Mem: var err error modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll]) if err != nil { return err } // An indexed memory operand carries index[4] in V'̄ (Go folds it // together with vvvv[4] into the same bit). if r.HasIndex && r.Index.idx&16 != 0 { vPrimeBar = 0 } case sbMem: // RIP-relative static-symbol reference; disp32 patched at link time // (no disp8 scaling for RIP-relative addressing). modrm = (regIdx&7)<<3 | 0x05 sib = -1 disp = le32(0) sb = &sbRef{name: r.name, addend: r.addend} default: return fmt.Errorf("invalid EVEX r/m operand") } z := 0 if zeroing { z = 1 } p0 := byte(rBar<<7 | xBar<<6 | bBar<<5 | rPrimeBar<<4 | spec.mapSel) p1 := byte(spec.w<<7 | vBar<<3 | 1<<2 | spec.pp) p2 := byte(z<<7 | ll<<5 | vPrimeBar<<3 | mask) // z, L'L, b=0, V', aaa e.out = append(e.out, 0x62, p0, p1, p2, spec.opcode, byte(modrm)) if sib >= 0 { e.out = append(e.out, byte(sib)) } if sb != nil { e.patches = append(e.patches, encPatch{off: len(e.out), name: sb.name, addend: sb.addend}) } e.out = append(e.out, disp...) return nil } // memComponentsEvex computes the ModR/M byte (with the given reg field), the // SIB byte (-1 if none), the displacement bytes and the (inverted sense) // index/base extension bits for an EVEX memory operand. The displacement is // compressed to disp8×N when it is a multiple of n and the quotient fits a // signed byte; otherwise a full disp32 is used. func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) { sib = -1 xBar, bBar = 1, 1 // inverted bits: 1 = no extension if !m.HasBase && !m.HasIndex { return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative } needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4) var mod int switch { case !m.HasBase: mod = 0 disp = le32(m.Disp) case m.Base.idx&7 == 5 && m.Disp == 0: mod = 1 disp = []byte{0} case m.Disp == 0: mod = 0 case n > 0 && m.Disp%int64(n) == 0 && m.Disp/int64(n) >= -128 && m.Disp/int64(n) <= 127: mod = 1 disp = []byte{byte(int8(m.Disp / int64(n)))} default: mod = 2 disp = le32(m.Disp) } if needSIB { idxField := 4 // 100 = no index if m.HasIndex { idxField = m.Index.idx & 7 if m.Index.idx&8 != 0 { xBar = 0 } } baseField := 5 // 101 = no base (with mod=00 → disp32) if m.HasBase { baseField = m.Base.idx & 7 if m.Base.idx&8 != 0 { bBar = 0 } } return mod<<6 | regField<<3 | 0x04, scaleBits(m.Scale)<<6 | idxField<<3 | baseField, disp, xBar, bBar, nil } if m.Base.idx&8 != 0 { bBar = 0 } return mod<<6 | regField<<3 | (m.Base.idx & 7), -1, disp, 1, bBar, nil } // encodeKmovw encodes KMOVW, whose opcode depends on the operand direction: // 90 (k/mem → K), 91 (K → mem), 92 (GPR → K), 93 (K → GPR); k → k uses 90. func (e *enc) encodeKmovw(ops []Operand) error { if len(ops) != 2 { return fmt.Errorf("KMOVW expects 2 operands, got %d", len(ops)) } src, dst := ops[0], ops[1] srcReg, srcIsReg := src.(Reg) dstReg, dstIsReg := dst.(Reg) srcK := srcIsReg && srcReg.mask dstK := dstIsReg && dstReg.mask spec := vexSpec{mapSel: 1, w: 0, pp: 0, opdigit: -1} switch { case srcK && dstK: spec.opcode = 0x90 // k ← k: reg = dst, rm = src return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src) case srcK && dstIsReg: spec.opcode = 0x93 // GPR ← k: reg = dst, rm = src rBit := 0 if dstReg.idx >= 8 { rBit = 1 } return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15, src) case srcK: if _, ok := dst.(Mem); !ok { return fmt.Errorf("KMOVW: invalid destination operand") } spec.opcode = 0x91 // mem ← k: reg = src, rm = dst return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst) case dstK: spec.opcode = 0x92 // k ← GPR/mem: reg = dst, rm = src return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src) } return fmt.Errorf("KMOVW requires a K register operand") }