feat(asm): byte-identical go-flac AVX2 assembly with scalar families and jump relaxation
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
+181
-3
@@ -52,9 +52,10 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
switch src := src.(type) {
|
||||
case Reg:
|
||||
if dstIsReg {
|
||||
// MOV r, r/m: 0x8A/0x8B, reg=dst, rm=src.
|
||||
i := newInstr(size, []byte{movRR(size)})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst — the form the Go
|
||||
// assembler emits for register-to-register moves.
|
||||
i := newInstr(size, []byte{movRM(size)})
|
||||
if err := setRM(i, src, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
@@ -147,9 +148,37 @@ func (e *enc) encodeALU(op struct {
|
||||
return e.encodeALUImm(op.digit, src, int64(imm), size)
|
||||
}
|
||||
|
||||
// CMP records first − second without writing anywhere, so the first
|
||||
// operand must land as the minuend; every other ALU op writes its second
|
||||
// operand and follows the forms below.
|
||||
cmp := op.rr == 0x39
|
||||
dstReg, dstIsReg := dst.(Reg)
|
||||
srcReg, srcIsReg := src.(Reg)
|
||||
switch {
|
||||
case cmp && dstIsReg:
|
||||
// CMP x, reg: OP r/m, r (0x38/0x39) with rm = first operand, reg =
|
||||
// second, matching the Go assembler.
|
||||
opc := op.rr
|
||||
if size == 1 {
|
||||
opc = op.rr - 1
|
||||
}
|
||||
i := newInstr(size, []byte{opc})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case cmp && srcIsReg:
|
||||
// CMP reg, mem: OP r, r/m (0x3A/0x3B) with reg = first operand, rm =
|
||||
// second.
|
||||
opc := op.rr + 2
|
||||
if size == 1 {
|
||||
opc = op.rr + 1
|
||||
}
|
||||
i := newInstr(size, []byte{opc})
|
||||
if err := setRM(i, srcReg, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
case srcIsReg:
|
||||
// OP r/m, r: reg=src, rm=dst (dst is a register or memory). This is the
|
||||
// form the Go assembler prefers when the source is a register.
|
||||
@@ -497,3 +526,152 @@ func immediate(v int64, size int, full64 bool) []byte {
|
||||
return le32(v) // sign-extended imm32
|
||||
}
|
||||
}
|
||||
|
||||
// --- CMOVcc / SETcc ---------------------------------------------------------
|
||||
|
||||
// encodeCmov encodes a conditional move: CMOV + size (W/L/Q) + condition
|
||||
// (CMOVLGT, CMOVQEQ, …). The condition reads exactly like the Jcc spellings;
|
||||
// the instruction is 0F 40+cc with reg = dst, rm = src.
|
||||
func (e *enc) encodeCmov(upper string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("CMOVcc expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
rest := upper[len("CMOV"):]
|
||||
if len(rest) < 2 {
|
||||
return fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
var size int
|
||||
switch rest[0] {
|
||||
case 'W':
|
||||
size = 2
|
||||
case 'L':
|
||||
size = 4
|
||||
case 'Q':
|
||||
size = 8
|
||||
default:
|
||||
return fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
cc, ok := jccMap[rest[1:]]
|
||||
if !ok {
|
||||
return fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("CMOVcc destination must be a register")
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, byte(0x40 + cc)})
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …),
|
||||
// always a byte write — 0F 90+cc /0 into a register or memory operand.
|
||||
func (e *enc) encodeSet(upper string, ops []Operand) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
cond := upper[len("SET"):]
|
||||
cc, ok := jccMap[cond]
|
||||
if !ok || cond == "" {
|
||||
return fmt.Errorf("unsupported instruction %q", upper)
|
||||
}
|
||||
i := &instr{opcode: []byte{0x0F, byte(0x90 + cc)}, modrm: -1, sib: -1}
|
||||
if err := setRMDigit(i, 0, ops[0], 1); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- LZCNT / TZCNT ----------------------------------------------------------
|
||||
|
||||
// encodeCount encodes LZCNT/TZCNT (leading / trailing zero count): F3 0F BD
|
||||
// or F3 0F BC, with reg = dst and rm = src. The size suffix selects the
|
||||
// operand width (LZCNTW/LZCNTL/LZCNTQ).
|
||||
func (e *enc) encodeCount(base string, ops []Operand, size int) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||
}
|
||||
op := byte(0xBD)
|
||||
if base == "TZCNT" {
|
||||
op = 0xBC
|
||||
}
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s destination must be a register", base)
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, op})
|
||||
i.prefix = 0xF3
|
||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- mixed-width sign/zero-extending moves -----------------------------------
|
||||
|
||||
// movExtendOp maps Go's mixed-width move names to their opcode and destination
|
||||
// width. The source is narrower than the destination, so the plain size-suffix
|
||||
// convention does not apply to these names.
|
||||
var movExtendOp = map[string]struct {
|
||||
op []byte
|
||||
dst64 bool
|
||||
}{
|
||||
"MOVBLZX": {[]byte{0x0F, 0xB6}, false}, // byte → long, zero-extend
|
||||
"MOVBQZX": {[]byte{0x0F, 0xB6}, true}, // byte → quad, zero-extend
|
||||
"MOVWLZX": {[]byte{0x0F, 0xB7}, false}, // word → long, zero-extend
|
||||
"MOVWQZX": {[]byte{0x0F, 0xB7}, true}, // word → quad, zero-extend
|
||||
"MOVWLSX": {[]byte{0x0F, 0xBF}, false}, // word → long, sign-extend
|
||||
"MOVLQSX": {[]byte{0x63}, true}, // long → quad, sign-extend (MOVSXD)
|
||||
}
|
||||
|
||||
// encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider
|
||||
// operand), rm = src.
|
||||
func (e *enc) encodeMovExtend(base string, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||
}
|
||||
spec := movExtendOp[base]
|
||||
dstReg, ok := ops[1].(Reg)
|
||||
if !ok {
|
||||
return fmt.Errorf("%s destination must be a register", base)
|
||||
}
|
||||
size := 4
|
||||
if spec.dst64 {
|
||||
size = 8
|
||||
}
|
||||
i := newInstr(size, spec.op)
|
||||
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
// --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
|
||||
|
||||
// encodeCvtsi2sd encodes a signed integer to scalar double conversion
|
||||
// (CVTSL2SD from a 32-bit, CVTSQ2SD from a 64-bit source): F2 0F 2A with
|
||||
// reg = XMM dst, rm = GPR/memory src. The Go assembler emits the legacy SSE
|
||||
// encoding here, not the VEX form, so we match it byte for byte.
|
||||
func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
|
||||
if len(ops) != 2 {
|
||||
return fmt.Errorf("CVTSx2SD expects 2 operands, got %d", len(ops))
|
||||
}
|
||||
src, dst := ops[0], ops[1]
|
||||
dstReg, ok := dst.(Reg)
|
||||
if !ok || !dstReg.isVec() {
|
||||
return fmt.Errorf("CVTSx2SD destination must be a vector register")
|
||||
}
|
||||
size := 4
|
||||
if quad {
|
||||
size = 8
|
||||
}
|
||||
i := newInstr(size, []byte{0x0F, 0x2A})
|
||||
i.prefix = 0xF2
|
||||
if err := setRM(i, dstReg, src, size); err != nil {
|
||||
return err
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user