feat(asm): encode the legacy amd64 SSE and MMX families

The packed integer and float binaries, the imm8-controlled SSE4.1 forms,
the variable blends with their X0 mask, the high/low half moves, the
sign-mask extractions, the non-temporal stores, the MOVQ bank crossings
and their odd spellings, the MMX shifts and shuffle and the cache-line
mask stores, plus the scalar leaves LEAVE, INVPCID and the RTM controls.
Every encoding is pinned byte for byte against go tool asm through every
corpus line the toolchain's own amd64enc.s carries for the families
(1465 lines).  Two corpus-wide gaps fell out of the comparison: the
64-bit MOV immediate uses the zero-extending form across the unsigned
32-bit span, and the MMX-to-GPR MOVQ puts the bank register in reg.

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-06 23:59:47 +02:00
1 parent 257feace6e
commit 6af3fd60d5
5 files changed
+2159 -13

No files matched your search

+39 -13
View File
@@ -182,13 +182,24 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// MMX register moves: MOVQ M0, mem and MOVQ mem, M0 are the MMX
// load/store pair 0F 6F/0F 7F (no prefix); a register pair takes the
// load opcode. The XMM MOVQ forms follow below.
// load opcode. The GPR crossings ride the MOVD opcodes with REX.W
// (0F 6E into the bank, 0F 7E out), and an XMM source crosses into the
// bank through the F2 0F D6 move. The XMM MOVQ forms follow below.
if m, ok := src.(Reg); ok && m.mmx {
switch d := dst.(type) {
case Reg:
if !d.mmx {
if !d.mmx && (d.isVec() || d.fp) {
return fmt.Errorf("MOV: MMX register moves stay inside the M bank")
}
if !d.mmx {
// M → GPR: 0F 7E with REX.W, the bank in reg and the GPR in
// r/m, the store layout the toolchain picks.
i := &instr{opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: true}
if err := setRM(i, m, d, 8); err != nil {
return err
}
return e.emit(i)
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, d, src, 8); err != nil {
return err
@@ -204,15 +215,30 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
return fmt.Errorf("MOV: invalid MMX destination")
}
if m, ok := dst.(Reg); ok && m.mmx {
srcM, ok := src.(Mem)
if !ok {
return fmt.Errorf("MOV: MMX load takes a memory source")
switch src.(type) {
case Mem, sbMem:
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, m, src, 8); err != nil {
return err
}
return e.emit(i)
case Reg:
if g := src.(Reg); g.isVec() {
// X → M: F2 0F D6, the bank in reg, the XMM source in r/m.
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, m, src, 8); err != nil {
return err
}
return e.emit(i)
}
// GPR → M: 0F 6E with REX.W, the bank in reg, the GPR in r/m.
i := &instr{opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: true}
if err := setRM(i, m, src, 8); err != nil {
return err
}
return e.emit(i)
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, m, srcM, 8); err != nil {
return err
}
return e.emit(i)
return fmt.Errorf("MOV: MMX load takes a register or memory source")
}
if srcVec || dstVec {
@@ -323,14 +349,14 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
case Imm:
if dstIsReg {
v := int64(src)
// The Go assembler compresses 64-bit moves whose immediate fits
// a signed int32, choosing per sign:
// The Go assembler compresses 64-bit moves whose immediate
// fits the zero-extending 32-bit span, choosing per sign:
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the
// hardware, REX.B still emitted for R8-R15);
// v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd
// form would zero-extend and corrupt the value).
// Out-of-range immediates keep the B8+rd imm64 form.
if size == 8 && v >= 0 && v <= (1<<31)-1 {
if size == 8 && v >= 0 && v <= (1<<32)-1 {
i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)})
i.rexB = dstReg.idx >= 8
i.imm = le32(v)