fix(amd64): correct guard displacements, frameless FP offsets and immediate ranges
Assisted-by: GLM 5.3
This commit is contained in:
+28
-28
@@ -114,6 +114,11 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
|
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
// A zero-operand jump parses; its arity is reported during
|
||||||
|
// emission (encodeJump), so the layout must not index Operands.
|
||||||
|
if len(s.Operands) != 1 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
name, ok := labelName(s.Operands[0])
|
name, ok := labelName(s.Operands[0])
|
||||||
if !ok {
|
if !ok {
|
||||||
continue // reported during emission
|
continue // reported during emission
|
||||||
@@ -137,28 +142,22 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
}
|
}
|
||||||
if fi.splitClass == 2 && !guardJBlong {
|
if fi.splitClass == 2 && !guardJBlong {
|
||||||
// The underflow JB sits before the CMPQ; its displacement spans
|
// The underflow JB sits before the CMPQ; its displacement spans
|
||||||
// the rest of the guard plus the prologue and the body.
|
// the rest of the guard plus the prologue and the body. The JB
|
||||||
jbLen := 2
|
// is still the short form this branch tests (relaxing it is this
|
||||||
if guardJBlong {
|
// branch's job), so guardLen is taken with a short JB and the
|
||||||
jbLen = 6
|
// subtraction drops the prefix and the JB's own 2 bytes.
|
||||||
}
|
rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2)
|
||||||
rest := fi.guardLen(guardJBlong, guardJBElong) - (9 + 3 + 7 + jbLen)
|
|
||||||
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
|
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
|
||||||
guardJBlong = true
|
guardJBlong = true
|
||||||
changed = true
|
changed = true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// The morestack JMP returns to the function start, so its
|
// The morestack JMP returns to the function start, so its
|
||||||
// displacement is the negated distance from its own end.
|
// displacement is the negated distance from its own end; while it is
|
||||||
if !moreJMPlong {
|
// still short, its own length is 2 bytes.
|
||||||
jmpLen := 2
|
if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) {
|
||||||
if moreJMPlong {
|
moreJMPlong = true
|
||||||
jmpLen = 5
|
changed = true
|
||||||
}
|
|
||||||
if !fits8(-int64(guard + len(fi.prologue) + bodyLen + 5 + jmpLen)) {
|
|
||||||
moreJMPlong = true
|
|
||||||
changed = true
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
if !changed {
|
if !changed {
|
||||||
break
|
break
|
||||||
@@ -181,7 +180,15 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
var out []byte
|
var out []byte
|
||||||
var patches []sbPatch
|
var patches []sbPatch
|
||||||
if fi.needSplit {
|
if fi.needSplit {
|
||||||
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+2)+len(fi.prologue)+bodyLen))
|
// The JBE ends the guard, so its displacement is the prologue plus
|
||||||
|
// the body; the underflow JB additionally spans the trailing CMPQ and
|
||||||
|
// JBE, whose combined length is guardLen minus the prefix and the
|
||||||
|
// JB's own length (2 short, 6 long).
|
||||||
|
jbLen := 2
|
||||||
|
if guardJBlong {
|
||||||
|
jbLen = 6
|
||||||
|
}
|
||||||
|
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen))
|
||||||
out = append(out, guard...)
|
out = append(out, guard...)
|
||||||
patches = append(patches, tlsPatch)
|
patches = append(patches, tlsPatch)
|
||||||
}
|
}
|
||||||
@@ -344,7 +351,10 @@ func computeFrame(t *ast.Text) frameInfo {
|
|||||||
// pass one extra slot, and the virtual SP is the hardware SP.
|
// pass one extra slot, and the virtual SP is the hardware SP.
|
||||||
fi.size = 8
|
fi.size = 8
|
||||||
fi.useFP = true
|
fi.useFP = true
|
||||||
fi.fpAdjust = int64(fi.size) + 16 // return address + saved BP + args base
|
// The push is the frame: the saved BP sits at SP+0 and the
|
||||||
|
// return address at SP+8, so arguments begin at SP+16. Unlike
|
||||||
|
// a SUBQ frame, the 8-byte size must not be added again.
|
||||||
|
fi.fpAdjust = 16
|
||||||
fi.spAdjust = 0
|
fi.spAdjust = 0
|
||||||
fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
|
fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
|
||||||
fi.epilogue = []byte{0x5D} // POPQ BP
|
fi.epilogue = []byte{0x5D} // POPQ BP
|
||||||
@@ -426,16 +436,6 @@ func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// moreLen returns the byte length of the trailing morestack block: the CALL
|
|
||||||
// (always rel32) plus the JMP back to the function start.
|
|
||||||
func moreLen(jmpLong bool) int {
|
|
||||||
jmp := 2
|
|
||||||
if jmpLong {
|
|
||||||
jmp = 5
|
|
||||||
}
|
|
||||||
return 5 + jmp
|
|
||||||
}
|
|
||||||
|
|
||||||
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
|
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
|
||||||
// already-computed displacements of the conditional branches that jump to the
|
// already-computed displacements of the conditional branches that jump to the
|
||||||
// morestack block (unused in classes without them). The TLS load carries a
|
// morestack block (unused in classes without them). The TLS load carries a
|
||||||
|
|||||||
@@ -160,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
|
||||||
|
// function containing a CALL receives: the PUSHQ BP prologue with no stack
|
||||||
|
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
|
||||||
|
// Go assembler produces. The push is the frame, so the offset must not count
|
||||||
|
// it twice.
|
||||||
|
func TestAssembleFramelessCall(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("frameless_call_amd64.s", `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·withcall(SB), NOSPLIT, $0-16
|
||||||
|
MOVQ x+0(FP), AX
|
||||||
|
CALL ·other(SB)
|
||||||
|
MOVQ AX, ret+8(FP)
|
||||||
|
RET
|
||||||
|
TEXT ·other(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
|
||||||
|
for _, r := range img.Funcs[0].Relocs {
|
||||||
|
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||||
|
code[j] = 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// From `go tool objdump` of the Go-assembled function:
|
||||||
|
// PUSHQ BP 55
|
||||||
|
// MOVQ SP, BP 4889e5
|
||||||
|
// MOVQ 0x10(SP), AX 488b442410
|
||||||
|
// CALL other e800000000
|
||||||
|
// MOVQ AX, 0x18(SP) 4889442418
|
||||||
|
// POPQ BP 5d
|
||||||
|
// RET c3
|
||||||
|
want := []byte{
|
||||||
|
0x55,
|
||||||
|
0x48, 0x89, 0xe5,
|
||||||
|
0x48, 0x8b, 0x44, 0x24, 0x10,
|
||||||
|
0xe8, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x48, 0x89, 0x44, 0x24, 0x18,
|
||||||
|
0x5d,
|
||||||
|
0xc3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
||||||
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
||||||
// the bytes the Go assembler produces.
|
// the bytes the Go assembler produces.
|
||||||
@@ -353,6 +404,19 @@ TEXT ·pf(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
|
||||||
|
// the parser does not arity-check mnemonics) is rejected with an error rather
|
||||||
|
// than panicking in the layout loop, which indexes Operands[0] before the
|
||||||
|
// emission pass gets a chance to diagnose the arity.
|
||||||
|
func TestAssembleBareJump(t *testing.T) {
|
||||||
|
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
|
||||||
|
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
|
||||||
|
if _, _, err := Assemble(fn); err == nil {
|
||||||
|
t.Errorf("%s with no operand: expected an error, got none", mnem)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
|
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
|
||||||
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
|
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
|
||||||
// larger. The intermediate 129..255 range used to encode an ADD with a
|
// larger. The intermediate 129..255 range used to encode an ADD with a
|
||||||
|
|||||||
+8
-3
@@ -36,10 +36,15 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
||||||
// alone (SETNE).
|
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
|
||||||
|
// so a spelling like CMOVBGT is not reported encodable when Encode
|
||||||
|
// would reject it.
|
||||||
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
||||||
if _, ok := jccMap[rest[1:]]; ok {
|
switch rest[0] {
|
||||||
return true
|
case 'W', 'L', 'Q':
|
||||||
|
if _, ok := jccMap[rest[1:]]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
||||||
|
|||||||
+8
-2
@@ -117,9 +117,9 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
case "IMUL", "IMUL3":
|
case "IMUL", "IMUL3":
|
||||||
return e.encodeImul(ops, size)
|
return e.encodeImul(ops, size)
|
||||||
case "PUSH":
|
case "PUSH":
|
||||||
return e.encodePushPop(ops, true)
|
return e.encodePushPop(ops, size, true)
|
||||||
case "POP":
|
case "POP":
|
||||||
return e.encodePushPop(ops, false)
|
return e.encodePushPop(ops, size, false)
|
||||||
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
||||||
return e.encodeCount(base, ops, size)
|
return e.encodeCount(base, ops, size)
|
||||||
case "BSWAP":
|
case "BSWAP":
|
||||||
@@ -345,6 +345,12 @@ func setMem(i *instr, regField int, m Mem) error {
|
|||||||
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
||||||
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
||||||
sib = -1
|
sib = -1
|
||||||
|
// A displacement wider than int32 fits no encoding form; truncating it
|
||||||
|
// would address a different location, and go tool asm reports "offset
|
||||||
|
// too large" for the same operand.
|
||||||
|
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
|
||||||
|
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
|
||||||
|
}
|
||||||
// RIP-relative: neither base nor index.
|
// RIP-relative: neither base nor index.
|
||||||
if !m.HasBase && !m.HasIndex {
|
if !m.HasBase && !m.HasIndex {
|
||||||
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
||||||
|
|||||||
@@ -149,6 +149,45 @@ func TestPushPop(t *testing.T) {
|
|||||||
checkSyntax(t, "push rbx", "PUSHQ", BX)
|
checkSyntax(t, "push rbx", "PUSHQ", BX)
|
||||||
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
|
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
|
||||||
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
|
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
|
||||||
|
// The W spelling carries the 0x66 operand-size prefix, byte for byte
|
||||||
|
// with go tool asm; the L and B spellings are illegal in 64-bit mode
|
||||||
|
// there and rejected here rather than silently widened.
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"PUSHW AX", "PUSHW", []Operand{AX}, "6650"},
|
||||||
|
{"POPW AX", "POPW", []Operand{AX}, "6658"},
|
||||||
|
{"PUSHW $5", "PUSHW", []Operand{Imm(5)}, "666a05"},
|
||||||
|
{"PUSHW (AX)", "PUSHW", []Operand{Ptr(AX, 0, 2)}, "66ff30"},
|
||||||
|
{"PUSHQ AX", "PUSHQ", []Operand{AX}, "50"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"PUSHL AX", "PUSHL", []Operand{AX}},
|
||||||
|
{"PUSHL R8", "PUSHL", []Operand{Reg{idx: 8, size: 8}}},
|
||||||
|
{"POPL BX", "POPL", []Operand{BX}},
|
||||||
|
{"PUSHB AX", "PUSHB", []Operand{AX}},
|
||||||
|
} {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestUnary(t *testing.T) {
|
func TestUnary(t *testing.T) {
|
||||||
@@ -397,6 +436,104 @@ func TestScalarErrors(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestImmediateOutOfRange pins the go-tool-asm parity of the immediate and
|
||||||
|
// displacement spans: a scalar immediate must fit a signed or unsigned 32-bit
|
||||||
|
// word (only MOVQ reg, $imm takes the full int64), a scalar shift count must
|
||||||
|
// be an unsigned byte, and a displacement must fit int32. Every rejected
|
||||||
|
// shape here is rejected by `go tool asm` too; every accepted one encodes the
|
||||||
|
// same bytes.
|
||||||
|
func TestImmediateOutOfRange(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"SHLQ count 300", "SHLQ", []Operand{Imm(300), AX}},
|
||||||
|
{"SHLQ count -1", "SHLQ", []Operand{Imm(-1), AX}},
|
||||||
|
{"SHLW count 256", "SHLW", []Operand{Imm(256), DX}},
|
||||||
|
{"SHLB count 300", "SHLB", []Operand{Imm(300), BL}},
|
||||||
|
{"MOVL imm32+", "MOVL", []Operand{Imm(4294967296), AX}},
|
||||||
|
{"MOVL imm32-", "MOVL", []Operand{Imm(-2147483649), AX}},
|
||||||
|
{"MOVW imm32+", "MOVW", []Operand{Imm(4294967296), AX}},
|
||||||
|
{"MOVB imm32+", "MOVB", []Operand{Imm(4294967296), AL}},
|
||||||
|
{"ADDB imm32+", "ADDB", []Operand{Imm(4294967296), AL}},
|
||||||
|
{"ADDL imm32+", "ADDL", []Operand{Imm(4294967296), AX}},
|
||||||
|
{"ADDQ imm32+", "ADDQ", []Operand{Imm(8589934592), AX}},
|
||||||
|
{"CMPQ imm32+", "CMPQ", []Operand{AX, Imm(4294967296)}},
|
||||||
|
{"CMPQ imm32-", "CMPQ", []Operand{AX, Imm(-2147483649)}},
|
||||||
|
{"TESTL imm32+", "TESTL", []Operand{Imm(4294967296), AX}},
|
||||||
|
{"IMUL3L imm32+", "IMUL3L", []Operand{Imm(4294967296), CX, DX}},
|
||||||
|
{"PUSHQ imm32+", "PUSHQ", []Operand{Imm(4294967296)}},
|
||||||
|
{"MOVQ mem imm32+", "MOVQ", []Operand{Imm(4294967296), Ptr(AX, 0, 8)}},
|
||||||
|
{"disp32+", "MOVQ", []Operand{Ptr(AX, 4294967296, 8), BX}},
|
||||||
|
{"disp32+ max", "MOVQ", []Operand{Ptr(AX, 2147483648, 8), BX}},
|
||||||
|
{"disp32-", "MOVQ", []Operand{Ptr(AX, -2147483649, 8), BX}},
|
||||||
|
{"VEX disp32+", "VMOVDQU", []Operand{Ptr(AX, 4294967296, 32), vreg(t, "Y1")}},
|
||||||
|
{"EVEX disp32+", "VMOVDQU32", []Operand{Ptr(AX, 4294967296, 64), vreg(t, "Z1")}},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestImmediateTruncation pins the toolchain-matching truncations inside the
|
||||||
|
// accepted 32-bit span: the narrower fields take the low bits silently, byte
|
||||||
|
// for byte with `go tool asm` (which rejects none of these).
|
||||||
|
func TestImmediateTruncation(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"ADDB $256,BL", "ADDB", []Operand{Imm(256), BL}, "80c300"},
|
||||||
|
{"ADDB $1000,BL", "ADDB", []Operand{Imm(1000), BL}, "80c3e8"},
|
||||||
|
{"MOVB $256,AL", "MOVB", []Operand{Imm(256), AL}, "b000"},
|
||||||
|
{"MOVB $-129,AL", "MOVB", []Operand{Imm(-129), AL}, "b07f"},
|
||||||
|
{"MOVW $65536,AX", "MOVW", []Operand{Imm(65536), AX}, "66b80000"},
|
||||||
|
{"MOVW $65535,AX", "MOVW", []Operand{Imm(65535), AX}, "66b8ffff"},
|
||||||
|
{"MOVW $-32769,AX", "MOVW", []Operand{Imm(-32769), AX}, "66b8ff7f"},
|
||||||
|
{"MOVL $4294967295,AX", "MOVL", []Operand{Imm(4294967295), AX}, "b8ffffffff"},
|
||||||
|
{"ADDQ $4294967295,AX", "ADDQ", []Operand{Imm(4294967295), AX}, "4805ffffffff"},
|
||||||
|
{"CMPB BL,$255", "CMPB", []Operand{BL, Imm(255)}, "80fbff"},
|
||||||
|
{"CMPQ AX,$4294967295", "CMPQ", []Operand{AX, Imm(4294967295)}, "483dffffffff"},
|
||||||
|
{"MOVQ $4294967295,0(AX)", "MOVQ", []Operand{Imm(4294967295), Ptr(AX, 0, 8)}, "48c700ffffffff"},
|
||||||
|
{"SHLQ $255,AX", "SHLQ", []Operand{Imm(255), AX}, "48c1e0ff"},
|
||||||
|
{"SHLQ $0,AX", "SHLQ", []Operand{Imm(0), AX}, "48c1e000"},
|
||||||
|
// The one form beyond the 32-bit span: the imm64 MOVQ register move.
|
||||||
|
{"MOVQ $4294967296,AX", "MOVQ", []Operand{Imm(4294967296), AX}, "48b80000000001000000"},
|
||||||
|
{"MOVQ disp32 max", "MOVQ", []Operand{Ptr(AX, 2147483647, 8), BX}, "488b98ffffff7f"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must
|
||||||
|
// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size
|
||||||
|
// letter) is not reported as encodable.
|
||||||
|
func TestEncodableCmovSize(t *testing.T) {
|
||||||
|
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} {
|
||||||
|
if Encodable(m) {
|
||||||
|
t.Errorf("Encodable(%q) = true, want false", m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} {
|
||||||
|
if !Encodable(m) {
|
||||||
|
t.Errorf("Encodable(%q) = false, want true", m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
||||||
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
||||||
func TestSSEBinGroundTruth(t *testing.T) {
|
func TestSSEBinGroundTruth(t *testing.T) {
|
||||||
|
|||||||
+37
-20
@@ -509,40 +509,43 @@ var evexBcastTable = map[string]evexBcastSpec{
|
|||||||
}
|
}
|
||||||
|
|
||||||
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
||||||
// move table).
|
// move table). vecOK and xmmOnly mirror the VEX twin's operand rules: a
|
||||||
|
// scalar move (vecOK false, xmmOnly true) takes XMM↔memory operands only.
|
||||||
type evexMoveSpec struct {
|
type evexMoveSpec struct {
|
||||||
mapSel int
|
mapSel int
|
||||||
pp int
|
pp int
|
||||||
load byte // r/m → vector
|
load byte // r/m → vector
|
||||||
store byte // vector → r/m
|
store byte // vector → r/m
|
||||||
w int
|
w int
|
||||||
n [3]int
|
n [3]int
|
||||||
|
vecOK bool // the non-memory operand may be a vector register
|
||||||
|
xmmOnly bool // wider than XMM registers are rejected
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
||||||
var evexMoveTable = map[string]evexMoveSpec{
|
var evexMoveTable = map[string]evexMoveSpec{
|
||||||
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
||||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
||||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
||||||
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
||||||
// semantics).
|
// semantics).
|
||||||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
||||||
// encoding).
|
// encoding).
|
||||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
||||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
|
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512, aligned packed moves.
|
// EVEX.128/256/512, aligned packed moves.
|
||||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
|
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
|
||||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
|
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128/256/512.66.0F, aligned integer moves.
|
// EVEX.128/256/512.66.0F, aligned integer moves.
|
||||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||||
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
||||||
// three-operand register form is not supported).
|
// three-operand register form is not supported).
|
||||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
|
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
|
||||||
}
|
}
|
||||||
|
|
||||||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||||||
@@ -1022,6 +1025,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
|
|||||||
var rm Operand
|
var rm Operand
|
||||||
switch {
|
switch {
|
||||||
case srcIsVec && dstIsVec:
|
case srcIsVec && dstIsVec:
|
||||||
|
// A store-form reg-reg move, the layout the Go assembler uses; a
|
||||||
|
// scalar move has no two-register form at all (the register form
|
||||||
|
// takes three operands), matching the VEX twin's vecOK rule.
|
||||||
|
if !ms.vecOK {
|
||||||
|
return fmt.Errorf("%s does not take two vector registers", mnem)
|
||||||
|
}
|
||||||
reg, rm = srcReg, dst
|
reg, rm = srcReg, dst
|
||||||
case srcIsVec:
|
case srcIsVec:
|
||||||
if !memOperand(dst) {
|
if !memOperand(dst) {
|
||||||
@@ -1037,6 +1046,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
|
|||||||
default:
|
default:
|
||||||
return fmt.Errorf("%s needs a vector register operand", mnem)
|
return fmt.Errorf("%s needs a vector register operand", mnem)
|
||||||
}
|
}
|
||||||
|
// The scalar move is 128-bit only, so the register the length follows
|
||||||
|
// must be an XMM (the VEX twin's xmmOnly rule; EVEX also reaches ZMM,
|
||||||
|
// hence the inequality rather than a YMM test).
|
||||||
|
if ms.xmmOnly && reg.size != 16 {
|
||||||
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
||||||
|
}
|
||||||
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||||||
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
|
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
|
||||||
}
|
}
|
||||||
@@ -1171,9 +1186,6 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
|
|||||||
if r.idx&16 != 0 {
|
if r.idx&16 != 0 {
|
||||||
xBar = 0
|
xBar = 0
|
||||||
}
|
}
|
||||||
if r.idx&16 != 0 {
|
|
||||||
xBar = 0
|
|
||||||
}
|
|
||||||
case Mem:
|
case Mem:
|
||||||
var err error
|
var err error
|
||||||
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
|
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
|
||||||
@@ -1232,6 +1244,11 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
|
|||||||
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
|
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
|
||||||
sib = -1
|
sib = -1
|
||||||
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
|
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
|
||||||
|
// The disp32 fallback bounds the displacement by int32, and the
|
||||||
|
// compressed disp8 form reaches at most ±127×64, well inside it.
|
||||||
|
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
|
||||||
|
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
|
||||||
|
}
|
||||||
if !m.HasBase && !m.HasIndex {
|
if !m.HasBase && !m.HasIndex {
|
||||||
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
|
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -675,6 +675,15 @@ func TestEvexErrors(t *testing.T) {
|
|||||||
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||||
// VEX-only mnemonics reject registers only EVEX can encode.
|
// VEX-only mnemonics reject registers only EVEX can encode.
|
||||||
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
||||||
|
// The scalar EVEX move matches its VEX twin and the Go assembler:
|
||||||
|
// XMM↔memory only, never reg-reg and never a wider register (the
|
||||||
|
// toolchain rejects every one of these shapes).
|
||||||
|
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
|
||||||
|
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
|
||||||
|
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
|
||||||
|
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
|
||||||
|
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
|
||||||
|
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
|
||||||
}
|
}
|
||||||
for _, c := range cases {
|
for _, c := range cases {
|
||||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
|||||||
+86
-21
@@ -173,7 +173,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
|||||||
if dstReg.needsREX(size) {
|
if dstReg.needsREX(size) {
|
||||||
i.rexForced = true
|
i.rexForced = true
|
||||||
}
|
}
|
||||||
i.imm = immediate(v, size, true)
|
imm, err := immediate(v, size, true)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = imm
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
|
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
|
||||||
@@ -185,7 +189,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
|||||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = immediate(int64(src), size, false)
|
imm, err := immediate(int64(src), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = imm
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("MOV: invalid operands")
|
return fmt.Errorf("MOV: invalid operands")
|
||||||
@@ -297,11 +305,15 @@ func (e *enc) encodeALU(op struct {
|
|||||||
|
|
||||||
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||||||
if size == 1 {
|
if size == 1 {
|
||||||
|
immBytes, err := immediate(imm, 1, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
i := newInstr(1, []byte{0x80})
|
i := newInstr(1, []byte{0x80})
|
||||||
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = []byte{byte(int8(imm))}
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
if fits8(imm) {
|
if fits8(imm) {
|
||||||
@@ -319,7 +331,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
|||||||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||||
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
|
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
|
||||||
i := newInstr(size, []byte{accOp})
|
i := newInstr(size, []byte{accOp})
|
||||||
i.imm = immediate(imm, size, false)
|
immBytes, err := immediate(imm, size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
// 0x81 /digit, imm16/imm32.
|
// 0x81 /digit, imm16/imm32.
|
||||||
@@ -327,7 +343,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
|||||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = immediate(imm, size, false)
|
immBytes, err := immediate(imm, size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -348,7 +368,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
|
|||||||
op = 0xA8
|
op = 0xA8
|
||||||
}
|
}
|
||||||
i := newInstr(size, []byte{op})
|
i := newInstr(size, []byte{op})
|
||||||
i.imm = immediate(int64(imm), size, false)
|
immBytes, err := immediate(int64(imm), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
op := byte(0xF7)
|
op := byte(0xF7)
|
||||||
@@ -359,7 +383,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
|
|||||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = immediate(int64(imm), size, false)
|
immBytes, err := immediate(int64(imm), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
srcReg, ok := src.(Reg)
|
srcReg, ok := src.(Reg)
|
||||||
@@ -457,7 +485,13 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
|||||||
}
|
}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
// 0xC0 (8-bit) / 0xC1, imm8.
|
// 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm
|
||||||
|
// rejects negative and ≥256 counts, and the hardware masks the count, so
|
||||||
|
// a silent truncation ($300 encoding 44) would shift by a different
|
||||||
|
// amount than the source states.
|
||||||
|
if imm < 0 || imm > 255 {
|
||||||
|
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
|
||||||
|
}
|
||||||
op := byte(0xC1)
|
op := byte(0xC1)
|
||||||
if size == 1 {
|
if size == 1 {
|
||||||
op = 0xC0
|
op = 0xC0
|
||||||
@@ -466,7 +500,7 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
|||||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = []byte{byte(int8(imm))}
|
i.imm = []byte{byte(imm)}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -508,7 +542,11 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
|
|||||||
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
i.imm = immediate(int64(imm), size, false)
|
immBytes, err := immediate(int64(imm), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
||||||
@@ -516,10 +554,21 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
|
|||||||
|
|
||||||
// --- PUSH / POP -------------------------------------------------------------
|
// --- PUSH / POP -------------------------------------------------------------
|
||||||
|
|
||||||
func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
|
||||||
if len(ops) != 1 {
|
if len(ops) != 1 {
|
||||||
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
|
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
|
||||||
}
|
}
|
||||||
|
// In 64-bit mode go tool asm knows the 64-bit push (the default, with or
|
||||||
|
// without the Q suffix) and the 16-bit W form with its 0x66 operand-size
|
||||||
|
// prefix, and rejects the B and L spellings outright ("illegal in 64-bit
|
||||||
|
// mode"); silently widening those would push a different width than the
|
||||||
|
// source states.
|
||||||
|
switch size {
|
||||||
|
case 0, 8, 2:
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode")
|
||||||
|
}
|
||||||
|
w16 := size == 2
|
||||||
switch op := ops[0].(type) {
|
switch op := ops[0].(type) {
|
||||||
case Reg:
|
case Reg:
|
||||||
base := byte(0x50) // PUSH r; POP is 0x58
|
base := byte(0x50) // PUSH r; POP is 0x58
|
||||||
@@ -527,7 +576,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
|||||||
base = 0x58
|
base = 0x58
|
||||||
}
|
}
|
||||||
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
|
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
|
||||||
i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
|
i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
|
||||||
i.rexB = op.idx >= 8
|
i.rexB = op.idx >= 8
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
case Mem:
|
case Mem:
|
||||||
@@ -537,7 +586,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
|||||||
opc = 0x8F // POP r/m: /0
|
opc = 0x8F // POP r/m: /0
|
||||||
digit = 0
|
digit = 0
|
||||||
}
|
}
|
||||||
i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1}
|
i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1}
|
||||||
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
@@ -547,10 +596,17 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
|||||||
return fmt.Errorf("POP does not take an immediate")
|
return fmt.Errorf("POP does not take an immediate")
|
||||||
}
|
}
|
||||||
if fits8(int64(op)) {
|
if fits8(int64(op)) {
|
||||||
i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
|
i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))}
|
// PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
|
||||||
|
// immediate by the same signed/unsigned 32-bit span as every other
|
||||||
|
// scalar immediate.
|
||||||
|
immBytes, err := immediate(int64(op), 8, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("PUSH/POP: invalid operand")
|
return fmt.Errorf("PUSH/POP: invalid operand")
|
||||||
@@ -639,19 +695,28 @@ func (e *enc) encodeJcc(cc int, ops []Operand) error {
|
|||||||
// immediate encodes an immediate of the given operand size. full64 selects the
|
// immediate encodes an immediate of the given operand size. full64 selects the
|
||||||
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
|
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
|
||||||
// sign-extended immediate is used for 64-bit operands.
|
// sign-extended immediate is used for 64-bit operands.
|
||||||
func immediate(v int64, size int, full64 bool) []byte {
|
//
|
||||||
|
// The span mirrors go tool asm: every scalar immediate must fit a signed or
|
||||||
|
// unsigned 32-bit word, and the narrower fields then take the low bits
|
||||||
|
// silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the
|
||||||
|
// imm64 form may exceed the span; anything wider elsewhere is an error rather
|
||||||
|
// than a truncation the source never asked for.
|
||||||
|
func immediate(v int64, size int, full64 bool) ([]byte, error) {
|
||||||
|
if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) {
|
||||||
|
return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v)
|
||||||
|
}
|
||||||
switch size {
|
switch size {
|
||||||
case 1:
|
case 1:
|
||||||
return []byte{byte(int8(v))}
|
return []byte{byte(int8(v))}, nil
|
||||||
case 2:
|
case 2:
|
||||||
return le16(v)
|
return le16(v), nil
|
||||||
case 4:
|
case 4:
|
||||||
return le32(v)
|
return le32(v), nil
|
||||||
default: // 8
|
default: // 8
|
||||||
if full64 {
|
if full64 {
|
||||||
return le64(v)
|
return le64(v), nil
|
||||||
}
|
}
|
||||||
return le32(v) // sign-extended imm32
|
return le32(v), nil // sign-extended imm32
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user