feat(asm): encode the remaining amd64 VEX families

The SSE3 horizontal and add-subtract pairs, the SSSE3 sign and
horizontal integers, the masked moves in both directions, the
reciprocity and test pairs, the AVX imm8 tail (blends, dot products,
inserts, rounds, MPSADBW, the string compares), the four-operand
variable blends with their /is4 mask byte, the scalar three-operand
moves, the MXCSR accessors, the VPERMIL register controls and the
variable word shifts, plus the BMI2 count forms over memory.  Every
encoding is pinned byte for byte against go tool asm through every
corpus line the toolchain's own amd64enc.s carries for the families
(852 lines); the /is4 byte carries the mask register number in its high
nibble, the layout the toolchain emits.

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-06 23:59:47 +02:00
1 parent 6af3fd60d5
commit cfc3abb752
4 files changed
+1096 -25

No files matched your search

+208 -24
View File
@@ -80,6 +80,13 @@ const (
// dst` (VPBLENDVB): ModRM.reg = dst (op3), VEX.vvvv = src1 (op2),
// ModRM.rm = src2 (op1) and the mask register in the /is4 byte (op0).
vexBlend4
// vexNDS3Dst is the destination-first NDS form `OP dst, src1, src2`
// (VMASKMOVPS, VPMASKMOVD): ModRM.reg = dst (op0), VEX.vvvv = the mask
// source (op1), ModRM.rm = memory (op2).
vexNDS3Dst
// vexRMOpDigit is the two-operand /digit form over memory (VLDMXCSR,
// VSTMXCSR): ModRM.reg = /digit, ModRM.rm = the memory operand.
vexRMOpDigit
)
// vexSpec describes one VEX instruction's encoding parameters.
@@ -347,7 +354,6 @@ var vexTable = map[string]vexSpec{
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F, packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
@@ -477,6 +483,82 @@ var vexTable = map[string]vexSpec{
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
// --- the corpus families from amd64enc.s: the SSE3 horizontal and
// add-subtract pairs, the SSSE3 sign and horizontal integers, the AVX
// reciprocity and test pairs, the masked and non-temporal oddities ---
// VEX.128/256, the horizontal and add-subtract float pairs: PD carries
// 66, PS carries F2 (0F 7C/7D and 0F D0).
"VHADDPD": {1, 0x7C, 0, 1, -1, vexNDS3},
"VHADDPS": {1, 0x7C, 0, 3, -1, vexNDS3},
"VHSUBPD": {1, 0x7D, 0, 1, -1, vexNDS3},
"VHSUBPS": {1, 0x7D, 0, 3, -1, vexNDS3},
"VADDSUBPD": {1, 0xD0, 0, 1, -1, vexNDS3},
"VADDSUBPS": {1, 0xD0, 0, 3, -1, vexNDS3},
// VEX.128/256.0F38.W0, the SSSE3 horizontal integer family and the sign
// controls, all NDS over 66.
"VPHADDW": {2, 0x01, 0, 1, -1, vexNDS3},
"VPHADDD": {2, 0x02, 0, 1, -1, vexNDS3},
"VPHADDSW": {2, 0x03, 0, 1, -1, vexNDS3},
"VPHSUBW": {2, 0x05, 0, 1, -1, vexNDS3},
"VPHSUBD": {2, 0x06, 0, 1, -1, vexNDS3},
"VPHSUBSW": {2, 0x07, 0, 1, -1, vexNDS3},
"VPSIGNB": {2, 0x08, 0, 1, -1, vexNDS3},
"VPSIGNW": {2, 0x09, 0, 1, -1, vexNDS3},
"VPSIGND": {2, 0x0A, 0, 1, -1, vexNDS3},
// VEX.128/256.0F38.W0, the masked loads/stores whose mask source rides
// vvvv, memory in r/m: the Plan 9 order puts the destination first
// (dst, mask, src), its own form below. VMOVHLPS is the plain NDS
// register move under 0F 12, and the test pair and the AES inverse cube
// root are two-operand.
"VMOVHLPS": {1, 0x12, 0, 0, -1, vexNDS3},
"VMASKMOVPD": {2, 0x2F, 0, 1, -1, vexNDS3Dst},
"VMASKMOVPS": {2, 0x2E, 0, 1, -1, vexNDS3Dst},
"VPMASKMOVD": {2, 0x8E, 0, 1, -1, vexNDS3Dst},
"VPMASKMOVQ": {2, 0x8E, 1, 1, -1, vexNDS3Dst},
"VTESTPD": {2, 0x0F, 0, 1, -1, vexRM},
"VTESTPS": {2, 0x0E, 0, 1, -1, vexRM},
"VAESIMC": {2, 0xDB, 0, 1, -1, vexRM},
"VPHMINPOSUW": {2, 0x41, 0, 1, -1, vexRM},
// VEX.128/256.66.0F38.W0, broadcast a 128-bit lane into a YMM.
"VBROADCASTF128": {2, 0x1A, 0, 1, -1, vexRM},
// VEX.128/256, the reciprocal square root estimates: PS bare (two-op
// RM), SS F3-prefixed NDS (the scalar preserved source).
"VRCPPS": {1, 0x53, 0, 0, -1, vexRM},
"VRSQRTPS": {1, 0x52, 0, 0, -1, vexRM},
"VRCPSS": {1, 0x53, 0, 2, -1, vexNDS3},
"VRSQRTSS": {1, 0x52, 0, 2, -1, vexNDS3},
// VEX.128/256.F2.0F.WIG, the unaligned load with cache hints and the
// duplicated low double, F3 for the singles replica.
"VLDDQU": {1, 0xF0, 0, 3, -1, vexRM},
// VEX.128/256.66.0F, the move-mask twin of VMOVMSKPS and the masked
// store over the integer bank.
"VMOVMSKPD": {1, 0x50, 0, 1, -1, vexRM},
"VMASKMOVDQU": {1, 0xF7, 0, 1, -1, vexRM},
// VEX.128/256.0F3A.W0, the imm8 tail the legacy set carries and its
// variable blends over the is4 byte.
"VBLENDPD": {3, 0x0D, 0, 1, -1, vexNDS3Imm},
"VBLENDPS": {3, 0x0C, 0, 1, -1, vexNDS3Imm},
"VPBLENDW": {3, 0x0E, 0, 1, -1, vexNDS3Imm},
"VDPPD": {3, 0x41, 0, 1, -1, vexNDS3Imm},
"VDPPS": {3, 0x40, 0, 1, -1, vexNDS3Imm},
"VINSERTPS": {3, 0x21, 0, 1, -1, vexNDS3Imm},
"VMPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm},
"VROUNDSD": {3, 0x0B, 0, 1, -1, vexNDS3Imm},
"VROUNDSS": {3, 0x0A, 0, 1, -1, vexNDS3Imm},
"VPINSRB": {3, 0x20, 0, 1, -1, vexNDS3Imm},
// VEX.128.66.0F3A.W0, the four-operand variable blends over the /is4
// byte (the mask rides is4[7:4], the raw register number times 16).
"VBLENDVPS": {3, 0x4A, 0, 1, -1, vexBlend4},
"VBLENDVPD": {3, 0x4B, 0, 1, -1, vexBlend4},
// VEX.256.66.0F3A.W0, the lane insert, and the GPR insert the legacy
// set spells: VPINSRW rides plain 0F C4 with 66.
"VINSERTF128": {3, 0x18, 0, 1, -1, vexNDS3Imm},
"VPINSRW": {1, 0xC4, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.0F, the MXCSR accessors, memory alone, no prefix.
"VLDMXCSR": {1, 0xAE, 0, 0, 2, vexRMOpDigit},
"VSTMXCSR": {1, 0xAE, 0, 0, 3, vexRMOpDigit},
}
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
@@ -498,9 +580,20 @@ var vexSrcLen = map[string]int{
var vexVarShift = map[string]byte{
"VPSLLD": 0xF2,
"VPSLLQ": 0xF3,
"VPSLLW": 0xF1,
"VPSRAD": 0xE2,
"VPSRAW": 0xE1,
"VPSRLD": 0xD2,
"VPSRLQ": 0xD3,
"VPSRLW": 0xD1,
}
// vexPermilReg maps the VPERMIL register-control spellings to their 0F38
// NDS opcodes: the control rides vvvv, the immediate form the main table
// carries never enters this path.
var vexPermilReg = map[string]vexSpec{
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3},
"VPERMILPD": {2, 0x0D, 0, 1, -1, vexNDS3},
}
// vexMoveSpec describes a VEX move, which takes different opcodes (and
@@ -533,10 +626,10 @@ var vexMoveTable = map[string]vexMoveSpec{
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
// VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
// register form takes three operands and is not supported yet).
// VEX.128.F2.0F.WIG, scalar double move: two operands move against
// memory, three operands the NDS store-opcode form (see encodeVexMove).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG, scalar single move, memory operands only.
// VEX.128.F3.0F.WIG, scalar single move, the same two shapes.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
@@ -551,10 +644,10 @@ func isVex(mnemUpper string) bool {
if _, ok := vexMoveTable[mnemUpper]; ok {
return true
}
// The dual-shape moves (VMOVHPD/VMOVLPD) pick their VEX form by operand
// count in encodeVex.
// The dual-shape moves (VMOVHPD/VMOVLPD and the single-precision twins)
// pick their VEX form by operand count in encodeVex.
switch mnemUpper {
case "VMOVHPD", "VMOVLPD":
case "VMOVHPD", "VMOVLPD", "VMOVHPS", "VMOVLPS":
return true
}
return false
@@ -591,19 +684,30 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: op, pp: 1, opdigit: -1, form: vexNDS3}, ops)
}
}
// The VPERMIL register-control form: the control rides vvvv (an NDS
// encoding under 0F38), the immediate form the main table carries.
if spec, ok := vexPermilReg[mnemUpper]; ok && len(ops) == 3 {
if _, isImm := ops[0].(Imm); !isImm {
return e.encodeVexNDS3(spec, ops)
}
}
// The high/low double moves split by operand count: three operands
// load-and-insert (mem, src, dst, an NDS form), two store (xmm, m64,
// the reversed store layout).
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" {
if mnemUpper == "VMOVHPD" || mnemUpper == "VMOVLPD" || mnemUpper == "VMOVHPS" || mnemUpper == "VMOVLPS" {
loadOp, storeOp := byte(0x16), byte(0x17)
if mnemUpper == "VMOVLPD" {
if mnemUpper == "VMOVLPD" || mnemUpper == "VMOVLPS" {
loadOp, storeOp = 0x12, 0x13
}
pp := 1
if mnemUpper == "VMOVHPS" || mnemUpper == "VMOVLPS" {
pp = 0
}
switch len(ops) {
case 3:
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: 1, opdigit: -1, form: vexNDS3}, ops)
return e.encodeVexNDS3(vexSpec{mapSel: 1, opcode: loadOp, w: 0, pp: pp, opdigit: -1, form: vexNDS3}, ops)
case 2:
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: 1, opdigit: -1, form: vexRMRev}, ops)
return e.encodeVexRMRev(vexSpec{mapSel: 1, opcode: storeOp, w: 0, pp: pp, opdigit: -1, form: vexRMRev}, ops)
}
return fmt.Errorf("%s expects 2 or 3 operands, got %d", mnemUpper, len(ops))
}
@@ -641,10 +745,62 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexCountGPR(spec, ops)
case vexRMRev:
return e.encodeVexRMRev(spec, ops)
case vexNDS3Dst:
return e.encodeVexNDS3Dst(spec, ops)
case vexRMOpDigit:
return e.encodeVexRMOpDigit(mnemUpper, spec, ops)
}
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
}
// encodeVexNDS3Dst encodes the destination-first NDS forms (VMASKMOVPS,
// VPMASKMOVD): ModRM.reg = the vector register, VEX.vvvv = the mask source,
// ModRM.rm = memory. Both directions exist: (dst, mask, mem) stores under
// the table opcode, (mem, mask, dst) loads under its twin two lower (the
// opcode rows pair 2E/2C, 2F/2D and 8E/8C).
func (e *enc) encodeVexNDS3Dst(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
opcode := spec.opcode
first, second, third := ops[0], ops[1], ops[2]
if isX86Mem(first) {
first, third = third, first
opcode -= 2
}
dstReg, ok := first.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("VEX destination must be a vector register")
}
vvvvReg, ok := second.(Reg)
if !ok || !vvvvReg.isVec() {
return fmt.Errorf("VEX mask source must be a vector register")
}
if !vecOrMem(third) {
return fmt.Errorf("VEX memory source expected")
}
regField := dstReg.idx & 7
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
enc := spec
enc.opcode = opcode
return e.emitVexFields(enc, dstReg.vecLenBit(), regField, rBit, 15-(vvvvReg.idx&15), third)
}
// encodeVexRMOpDigit encodes the MXCSR accessors: the single memory operand
// rides r/m under the fixed /digit, the way the legacy 0F AE pair does.
func (e *enc) encodeVexRMOpDigit(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 memory operand, got %d", mnem, len(ops))
}
if !isX86Mem(ops[0]) {
return fmt.Errorf("%s requires a memory operand", mnem)
}
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15, ops[0])
}
// encodeVexNDS3 encodes the three-operand NDS form: OP src2, src1, dst.
func (e *enc) encodeVexNDS3(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
@@ -970,7 +1126,8 @@ func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
// encodeVexCountGPR encodes the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
// VEX.vvvv = src (op0), ModRM.rm = count (op1, register or memory),
// ModRM.reg = dst (op2).
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
@@ -980,9 +1137,13 @@ func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
countReg, ok := count.(Reg)
if !ok || countReg.isVec() {
return fmt.Errorf("VEX count operand must be a general-purpose register")
switch count.(type) {
case Reg, Mem, sbMem:
default:
return fmt.Errorf("VEX count operand must be a general-purpose register or memory")
}
if r, ok := count.(Reg); ok && r.isVec() {
return fmt.Errorf("VEX count operand must be a general-purpose register or memory")
}
srcReg, ok := src.(Reg)
if !ok || srcReg.isVec() {
@@ -1050,17 +1211,18 @@ func (e *enc) encodeVexExtractGPR(spec vexSpec, ops []Operand) error {
return nil
}
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB):
// OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv = src1, r/m =
// src2 and the mask XMM register in the trailing /is4 byte.
// encodeVexBlend4 encodes the four-operand variable blend (VPBLENDVB and the
// VBLENDV pair): OP mask, src2, src1, dst with ModRM.reg = dst, VEX.vvvv =
// src1, r/m = src2 and the mask register in the trailing /is4 byte, whose
// high nibble carries the mask's register number raw.
func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
if len(ops) != 4 {
return fmt.Errorf("blend expects 4 operands (mask, src2, src1, dst), got %d", len(ops))
}
mask, src2, src1, dst := ops[0], ops[1], ops[2], ops[3]
maskReg, ok := mask.(Reg)
if !ok || !maskReg.isVec() || maskReg.size != 16 {
return fmt.Errorf("blend mask must be an XMM register")
if !ok || !maskReg.isVec() {
return fmt.Errorf("blend mask must be a vector register")
}
vvvvReg, ok := src1.(Reg)
if !ok || !vvvvReg.isVec() {
@@ -1077,17 +1239,39 @@ func (e *enc) encodeVexBlend4(spec vexSpec, ops []Operand) error {
if err := e.emitVexFields(spec, dstReg.vecLenBit(), dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2); err != nil {
return err
}
// The /is4 byte names the mask register: bits [3:0] its low nibble,
// bit 7 the fourth register bit (X8-X15).
e.out = append(e.out, byte(maskReg.idx&7)|byte((maskReg.idx&8)<<4))
// The /is4 byte names the mask register: its number in the high nibble,
// the layout the Go assembler and the hardware agree on for X0-X15.
e.out = append(e.out, byte(maskReg.idx)<<4)
return nil
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
// the Go assembler.
// the Go assembler. The scalar moves (VMOVSD, VMOVSS) also carry a
// three-operand form, which the toolchain encodes with the store opcode:
// reg = the Plan 9 first operand, vvvv = the second, rm = the third.
func (e *enc) encodeVexMove(mnem string, ms vexMoveSpec, ops []Operand) error {
if len(ops) == 3 {
if !ms.xmmOnly {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
op0, ok0 := ops[0].(Reg)
op1, ok1 := ops[1].(Reg)
op2, ok2 := ops[2].(Reg)
if !ok0 || !ok1 || !ok2 || !op0.isVec() || !op1.isVec() || !op2.isVec() {
return fmt.Errorf("%s three-operand form takes three vector registers", mnem)
}
if ms.xmmOnly && (op0.size != 16 || op1.size != 16 || op2.size != 16) {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
rBit := 0
if op0.idx >= 8 {
rBit = 1
}
spec := vexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.storeW, pp: ms.pp, opdigit: -1}
return e.emitVexFields(spec, 0, op0.idx&7, rBit, 15-(op1.idx&15), op2)
}
if len(ops) != 2 {
return fmt.Errorf("VEX move expects 2 operands, got %d", len(ops))
}