fix(asm): match the toolchain's bytes across the corpus sweep
A line-for-line byte comparison of the whole amd64enc.s corpus against go tool asm surfaced divergences the pass-only accounting never showed: PEXTRW's GPR form swapped its fields, PUSHW took an imm32 where the toolchain bounds the immediate to 16 bits, the double shift wrote the unmasked register number into the reg field, VCOMISS carried a 0x66 prefix, RORX dropped the destination's R bit, and the variable bit shifts used the manual's per-width opcodes where the toolchain consolidates each row on one opcode with the W bit. The VEX forms the toolchain prefers for plain vector registers (the SSE2/SSSE3/SSE4.1 AVX twins, the compare-with-predicate family, VMOVUPS, VSHUFPS, the variable shifts) now encode under VEX, with EVEX left to the ZMM, opmask and index-16+ spellings, and the mnemonics whose rows never offer the 2-byte prefix force it. Every line is pinned through the new corpus parity test (793 lines); the whole corpus file now assembles to the toolchain's bytes at every commented line (10022 of 10022). Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
cfc3abb752
commit
a69f8cf4a8
4 files changed
+1276
-337
No files matched your search
+23
-8
@@ -791,7 +791,7 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
|
||||
}
|
||||
i.imm = []byte{byte(imm)}
|
||||
}
|
||||
if err := setRMReg(i, srcReg.idx, srcReg.idx >= 8, false, dst, size); err != nil {
|
||||
if err := setRMReg(i, srcReg.idx&7, srcReg.idx >= 8, false, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
@@ -943,8 +943,12 @@ func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
|
||||
}
|
||||
// PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
|
||||
// immediate by the same signed/unsigned 32-bit span as every other
|
||||
// scalar immediate.
|
||||
immBytes, err := immediate(int64(op), 8, false)
|
||||
// scalar immediate. The W spelling takes imm16 alone.
|
||||
immWidth := 8
|
||||
if w16 {
|
||||
immWidth = 2
|
||||
}
|
||||
immBytes, err := immediate(int64(op), immWidth, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -1432,13 +1436,15 @@ type sseExtract struct {
|
||||
op []byte
|
||||
opMem []byte // used when the destination is memory; nil shares op
|
||||
rexW bool // PEXTRQ's REX.W
|
||||
rev bool // PEXTRW's GPR form swaps the fields: the register
|
||||
// destination rides reg and the XMM source r/m
|
||||
}
|
||||
|
||||
var sseExtractTable = map[string]sseExtract{
|
||||
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
|
||||
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
|
||||
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
|
||||
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
|
||||
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false, false},
|
||||
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false, false},
|
||||
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true, false},
|
||||
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false, true},
|
||||
}
|
||||
|
||||
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
|
||||
@@ -1925,8 +1931,17 @@ func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
|
||||
if m.opMem != nil && memOperand(ops[2]) {
|
||||
opcode = m.opMem
|
||||
}
|
||||
reg, rm := srcReg, ops[2]
|
||||
if m.rev && opcode[1] == 0xC5 {
|
||||
// the 0F C5 layout: the GPR destination rides reg, the XMM source r/m
|
||||
if d, ok := ops[2].(Reg); !ok {
|
||||
return fmt.Errorf("PEXTRW destination must be a register or memory")
|
||||
} else {
|
||||
reg, rm = d, ops[1]
|
||||
}
|
||||
}
|
||||
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
|
||||
if err := setRM(i, srcReg, ops[2], 8); err != nil {
|
||||
if err := setRM(i, reg, rm, 8); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{immByte}
|
||||
|
||||
Reference in new issue
Block a user