fix(lint): calibrate register-clobber to the Go ABI and add legacy SSE moves
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
@@ -93,6 +93,8 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
return e.encodeMovExtend(base, ops)
|
||||
case "CVTSL2SD", "CVTSQ2SD":
|
||||
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
||||
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||
}
|
||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||
}
|
||||
|
||||
@@ -127,6 +127,52 @@ func TestControl(t *testing.T) {
|
||||
checkOp(t, x86asm.JBE, "JLS", Imm(0))
|
||||
}
|
||||
|
||||
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
|
||||
// against the Go assembler. wantOp is the decoder's name, which differs from
|
||||
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
|
||||
func TestSSEMoveGroundTruth(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
wantOp string
|
||||
}{
|
||||
{"MOVOU (SI),X1", "MOVOU", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "f30f6f0e", "MOVDQU"},
|
||||
{"MOVOU X3,(DI)", "MOVOU", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "f30f7f1f", "MOVDQU"},
|
||||
{"MOVOU X1,X2", "MOVOU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f30f6fd1", "MOVDQU"},
|
||||
{"MOVOU (SI)(BX*4),X9", "MOVOU", []Operand{Idx(SI, BX, 4, 0, 16), vreg(t, "X9")}, "f3440f6f0c9e", "MOVDQU"},
|
||||
{"MOVO (SI),X1", "MOVO", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f6f0e", "MOVDQA"},
|
||||
{"MOVO X3,(DI)", "MOVO", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f7f1f", "MOVDQA"},
|
||||
{"MOVUPS (SI),X1", "MOVUPS", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "0f100e", "MOVUPS"},
|
||||
{"MOVAPS X3,(DI)", "MOVAPS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "0f291f", "MOVAPS"},
|
||||
{"MOVUPD (SI),X1", "MOVUPD", []Operand{Ptr(SI, 0, 16), vreg(t, "X1")}, "660f100e", "MOVUPD"},
|
||||
{"MOVAPD X3,(DI)", "MOVAPD", []Operand{vreg(t, "X3"), Ptr(DI, 0, 16)}, "660f291f", "MOVAPD"},
|
||||
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
|
||||
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
|
||||
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Encode: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := hexCompact(code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
continue
|
||||
}
|
||||
inst, err := x86asm.Decode(code, 64)
|
||||
if err != nil {
|
||||
t.Errorf("%s: Decode(%x): %v", c.name, code, err)
|
||||
continue
|
||||
}
|
||||
if inst.Op.String() != c.wantOp {
|
||||
t.Errorf("%s: decoded as %s", c.name, inst.Op.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
|
||||
// the encoder handles a realistic instruction sequence.
|
||||
func TestGoFlacScalarTail(t *testing.T) {
|
||||
|
||||
@@ -115,6 +115,8 @@ type evexMoveSpec struct {
|
||||
var evexMoveTable = map[string]evexMoveSpec{
|
||||
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move.
|
||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move.
|
||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||||
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
|
||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
|
||||
}
|
||||
|
||||
@@ -61,6 +61,10 @@ func TestEvexGroundTruth(t *testing.T) {
|
||||
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
|
||||
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
|
||||
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
|
||||
// VMOVDQU64 — the W1 qword variant.
|
||||
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
|
||||
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
|
||||
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
|
||||
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
|
||||
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
|
||||
// Conversions and narrowing stores (reg = wide source).
|
||||
|
||||
@@ -661,6 +661,66 @@ func (e *enc) encodeMovExtend(base string, ops []Operand) error {
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- legacy SSE moves --------------------------------------------------------
|
||||
|
||||
// sseMove describes a legacy (non-VEX) SSE move: a mandatory prefix plus a
|
||||
// load opcode (reg = destination, rm = source) and a store opcode (the
|
||||
// reverse). The Plan 9 names MOVOU/MOVO are the integer unaligned/aligned
|
||||
// octa moves (MOVDQU/MOVDQA), not the packed-single ones.
|
||||
type sseMove struct {
|
||||
prefix byte // 0, 0x66, 0xF2 or 0xF3
|
||||
load byte
|
||||
store byte
|
||||
}
|
||||
|
||||
var sseMoveTable = map[string]sseMove{
|
||||
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa
|
||||
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa
|
||||
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
||||
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
||||
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
||||
"MOVAPD": {0x66, 0x28, 0x29}, // aligned packed double
|
||||
"MOVSD": {0xF2, 0x10, 0x11}, // scalar double
|
||||
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
|
||||
}
|
||||
|
||||
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
|
||||
// load form (reg = destination), matching the Go assembler.
|
||||
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("SSE move expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
srcReg, srcVec := vecReg(src)
|
||||
dstReg, dstVec := vecReg(dst)
|
||||
op := m.store
|
||||
var reg Reg
|
||||
var rm Operand
|
||||
switch {
|
||||
case srcVec && dstVec:
|
||||
op = m.load
|
||||
reg, rm = dstReg, src
|
||||
case srcVec:
|
||||
if _, ok := dst.(Mem); !ok {
|
||||
return fmt.Errorf("SSE move: invalid destination operand")
|
||||
}
|
||||
reg, rm = srcReg, dst
|
||||
case dstVec:
|
||||
if _, ok := src.(Mem); !ok {
|
||||
return fmt.Errorf("SSE move: invalid source operand")
|
||||
}
|
||||
op = m.load
|
||||
reg, rm = dstReg, src
|
||||
default:
|
||||
return fmt.Errorf("SSE move needs a vector register operand")
|
||||
}
|
||||
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||
if err := setRM(i, reg, rm, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
|
||||
|
||||
// encodeCvtsi2sd encodes a signed integer to scalar double conversion
|
||||
|
||||
Reference in New Issue
Block a user