fix(amd64): correct guard displacements, frameless FP offsets and immediate ranges
Assisted-by: GLM 5.3
This commit is contained in:
+26
-26
@@ -114,6 +114,11 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
|
||||
continue
|
||||
}
|
||||
// A zero-operand jump parses; its arity is reported during
|
||||
// emission (encodeJump), so the layout must not index Operands.
|
||||
if len(s.Operands) != 1 {
|
||||
continue
|
||||
}
|
||||
name, ok := labelName(s.Operands[0])
|
||||
if !ok {
|
||||
continue // reported during emission
|
||||
@@ -137,29 +142,23 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
}
|
||||
if fi.splitClass == 2 && !guardJBlong {
|
||||
// The underflow JB sits before the CMPQ; its displacement spans
|
||||
// the rest of the guard plus the prologue and the body.
|
||||
jbLen := 2
|
||||
if guardJBlong {
|
||||
jbLen = 6
|
||||
}
|
||||
rest := fi.guardLen(guardJBlong, guardJBElong) - (9 + 3 + 7 + jbLen)
|
||||
// the rest of the guard plus the prologue and the body. The JB
|
||||
// is still the short form this branch tests (relaxing it is this
|
||||
// branch's job), so guardLen is taken with a short JB and the
|
||||
// subtraction drops the prefix and the JB's own 2 bytes.
|
||||
rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2)
|
||||
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
|
||||
guardJBlong = true
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
// The morestack JMP returns to the function start, so its
|
||||
// displacement is the negated distance from its own end.
|
||||
if !moreJMPlong {
|
||||
jmpLen := 2
|
||||
if moreJMPlong {
|
||||
jmpLen = 5
|
||||
}
|
||||
if !fits8(-int64(guard + len(fi.prologue) + bodyLen + 5 + jmpLen)) {
|
||||
// displacement is the negated distance from its own end; while it is
|
||||
// still short, its own length is 2 bytes.
|
||||
if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) {
|
||||
moreJMPlong = true
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
if !changed {
|
||||
break
|
||||
}
|
||||
@@ -181,7 +180,15 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
||||
var out []byte
|
||||
var patches []sbPatch
|
||||
if fi.needSplit {
|
||||
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+2)+len(fi.prologue)+bodyLen))
|
||||
// The JBE ends the guard, so its displacement is the prologue plus
|
||||
// the body; the underflow JB additionally spans the trailing CMPQ and
|
||||
// JBE, whose combined length is guardLen minus the prefix and the
|
||||
// JB's own length (2 short, 6 long).
|
||||
jbLen := 2
|
||||
if guardJBlong {
|
||||
jbLen = 6
|
||||
}
|
||||
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen))
|
||||
out = append(out, guard...)
|
||||
patches = append(patches, tlsPatch)
|
||||
}
|
||||
@@ -344,7 +351,10 @@ func computeFrame(t *ast.Text) frameInfo {
|
||||
// pass one extra slot, and the virtual SP is the hardware SP.
|
||||
fi.size = 8
|
||||
fi.useFP = true
|
||||
fi.fpAdjust = int64(fi.size) + 16 // return address + saved BP + args base
|
||||
// The push is the frame: the saved BP sits at SP+0 and the
|
||||
// return address at SP+8, so arguments begin at SP+16. Unlike
|
||||
// a SUBQ frame, the 8-byte size must not be added again.
|
||||
fi.fpAdjust = 16
|
||||
fi.spAdjust = 0
|
||||
fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
|
||||
fi.epilogue = []byte{0x5D} // POPQ BP
|
||||
@@ -426,16 +436,6 @@ func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
|
||||
}
|
||||
}
|
||||
|
||||
// moreLen returns the byte length of the trailing morestack block: the CALL
|
||||
// (always rel32) plus the JMP back to the function start.
|
||||
func moreLen(jmpLong bool) int {
|
||||
jmp := 2
|
||||
if jmpLong {
|
||||
jmp = 5
|
||||
}
|
||||
return 5 + jmp
|
||||
}
|
||||
|
||||
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
|
||||
// already-computed displacements of the conditional branches that jump to the
|
||||
// morestack block (unused in classes without them). The TLS load carries a
|
||||
|
||||
@@ -160,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
|
||||
// function containing a CALL receives: the PUSHQ BP prologue with no stack
|
||||
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
|
||||
// Go assembler produces. The push is the frame, so the offset must not count
|
||||
// it twice.
|
||||
func TestAssembleFramelessCall(t *testing.T) {
|
||||
f, errs := parser.Parse("frameless_call_amd64.s", `
|
||||
#include "textflag.h"
|
||||
TEXT ·withcall(SB), NOSPLIT, $0-16
|
||||
MOVQ x+0(FP), AX
|
||||
CALL ·other(SB)
|
||||
MOVQ AX, ret+8(FP)
|
||||
RET
|
||||
TEXT ·other(SB), NOSPLIT, $0-0
|
||||
RET
|
||||
`)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFile(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFile: %v", err)
|
||||
}
|
||||
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
|
||||
for _, r := range img.Funcs[0].Relocs {
|
||||
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||
code[j] = 0
|
||||
}
|
||||
}
|
||||
// From `go tool objdump` of the Go-assembled function:
|
||||
// PUSHQ BP 55
|
||||
// MOVQ SP, BP 4889e5
|
||||
// MOVQ 0x10(SP), AX 488b442410
|
||||
// CALL other e800000000
|
||||
// MOVQ AX, 0x18(SP) 4889442418
|
||||
// POPQ BP 5d
|
||||
// RET c3
|
||||
want := []byte{
|
||||
0x55,
|
||||
0x48, 0x89, 0xe5,
|
||||
0x48, 0x8b, 0x44, 0x24, 0x10,
|
||||
0xe8, 0x00, 0x00, 0x00, 0x00,
|
||||
0x48, 0x89, 0x44, 0x24, 0x18,
|
||||
0x5d,
|
||||
0xc3,
|
||||
}
|
||||
if hexBytes(code) != hexBytes(want) {
|
||||
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
|
||||
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
|
||||
// the bytes the Go assembler produces.
|
||||
@@ -353,6 +404,19 @@ TEXT ·pf(SB), NOSPLIT, $0
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
|
||||
// the parser does not arity-check mnemonics) is rejected with an error rather
|
||||
// than panicking in the layout loop, which indexes Operands[0] before the
|
||||
// emission pass gets a chance to diagnose the arity.
|
||||
func TestAssembleBareJump(t *testing.T) {
|
||||
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
|
||||
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
|
||||
if _, _, err := Assemble(fn); err == nil {
|
||||
t.Errorf("%s with no operand: expected an error, got none", mnem)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
|
||||
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
|
||||
// larger. The intermediate 129..255 range used to encode an ADD with a
|
||||
|
||||
+6
-1
@@ -36,12 +36,17 @@ func Encodable(mnemonic string) bool {
|
||||
}
|
||||
|
||||
// CMOV carries size then condition (CMOVLGT); SET carries the condition
|
||||
// alone (SETNE).
|
||||
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
|
||||
// so a spelling like CMOVBGT is not reported encodable when Encode
|
||||
// would reject it.
|
||||
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
|
||||
switch rest[0] {
|
||||
case 'W', 'L', 'Q':
|
||||
if _, ok := jccMap[rest[1:]]; ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
|
||||
if _, ok := jccMap[rest]; ok {
|
||||
return true
|
||||
|
||||
+8
-2
@@ -117,9 +117,9 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
||||
case "IMUL", "IMUL3":
|
||||
return e.encodeImul(ops, size)
|
||||
case "PUSH":
|
||||
return e.encodePushPop(ops, true)
|
||||
return e.encodePushPop(ops, size, true)
|
||||
case "POP":
|
||||
return e.encodePushPop(ops, false)
|
||||
return e.encodePushPop(ops, size, false)
|
||||
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
|
||||
return e.encodeCount(base, ops, size)
|
||||
case "BSWAP":
|
||||
@@ -345,6 +345,12 @@ func setMem(i *instr, regField int, m Mem) error {
|
||||
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
||||
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
|
||||
sib = -1
|
||||
// A displacement wider than int32 fits no encoding form; truncating it
|
||||
// would address a different location, and go tool asm reports "offset
|
||||
// too large" for the same operand.
|
||||
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
|
||||
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
|
||||
}
|
||||
// RIP-relative: neither base nor index.
|
||||
if !m.HasBase && !m.HasIndex {
|
||||
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
|
||||
|
||||
@@ -149,6 +149,45 @@ func TestPushPop(t *testing.T) {
|
||||
checkSyntax(t, "push rbx", "PUSHQ", BX)
|
||||
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
|
||||
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
|
||||
// The W spelling carries the 0x66 operand-size prefix, byte for byte
|
||||
// with go tool asm; the L and B spellings are illegal in 64-bit mode
|
||||
// there and rejected here rather than silently widened.
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"PUSHW AX", "PUSHW", []Operand{AX}, "6650"},
|
||||
{"POPW AX", "POPW", []Operand{AX}, "6658"},
|
||||
{"PUSHW $5", "PUSHW", []Operand{Imm(5)}, "666a05"},
|
||||
{"PUSHW (AX)", "PUSHW", []Operand{Ptr(AX, 0, 2)}, "66ff30"},
|
||||
{"PUSHQ AX", "PUSHQ", []Operand{AX}, "50"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"PUSHL AX", "PUSHL", []Operand{AX}},
|
||||
{"PUSHL R8", "PUSHL", []Operand{Reg{idx: 8, size: 8}}},
|
||||
{"POPL BX", "POPL", []Operand{BX}},
|
||||
{"PUSHB AX", "PUSHB", []Operand{AX}},
|
||||
} {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnary(t *testing.T) {
|
||||
@@ -397,6 +436,104 @@ func TestScalarErrors(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestImmediateOutOfRange pins the go-tool-asm parity of the immediate and
|
||||
// displacement spans: a scalar immediate must fit a signed or unsigned 32-bit
|
||||
// word (only MOVQ reg, $imm takes the full int64), a scalar shift count must
|
||||
// be an unsigned byte, and a displacement must fit int32. Every rejected
|
||||
// shape here is rejected by `go tool asm` too; every accepted one encodes the
|
||||
// same bytes.
|
||||
func TestImmediateOutOfRange(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
}{
|
||||
{"SHLQ count 300", "SHLQ", []Operand{Imm(300), AX}},
|
||||
{"SHLQ count -1", "SHLQ", []Operand{Imm(-1), AX}},
|
||||
{"SHLW count 256", "SHLW", []Operand{Imm(256), DX}},
|
||||
{"SHLB count 300", "SHLB", []Operand{Imm(300), BL}},
|
||||
{"MOVL imm32+", "MOVL", []Operand{Imm(4294967296), AX}},
|
||||
{"MOVL imm32-", "MOVL", []Operand{Imm(-2147483649), AX}},
|
||||
{"MOVW imm32+", "MOVW", []Operand{Imm(4294967296), AX}},
|
||||
{"MOVB imm32+", "MOVB", []Operand{Imm(4294967296), AL}},
|
||||
{"ADDB imm32+", "ADDB", []Operand{Imm(4294967296), AL}},
|
||||
{"ADDL imm32+", "ADDL", []Operand{Imm(4294967296), AX}},
|
||||
{"ADDQ imm32+", "ADDQ", []Operand{Imm(8589934592), AX}},
|
||||
{"CMPQ imm32+", "CMPQ", []Operand{AX, Imm(4294967296)}},
|
||||
{"CMPQ imm32-", "CMPQ", []Operand{AX, Imm(-2147483649)}},
|
||||
{"TESTL imm32+", "TESTL", []Operand{Imm(4294967296), AX}},
|
||||
{"IMUL3L imm32+", "IMUL3L", []Operand{Imm(4294967296), CX, DX}},
|
||||
{"PUSHQ imm32+", "PUSHQ", []Operand{Imm(4294967296)}},
|
||||
{"MOVQ mem imm32+", "MOVQ", []Operand{Imm(4294967296), Ptr(AX, 0, 8)}},
|
||||
{"disp32+", "MOVQ", []Operand{Ptr(AX, 4294967296, 8), BX}},
|
||||
{"disp32+ max", "MOVQ", []Operand{Ptr(AX, 2147483648, 8), BX}},
|
||||
{"disp32-", "MOVQ", []Operand{Ptr(AX, -2147483649, 8), BX}},
|
||||
{"VEX disp32+", "VMOVDQU", []Operand{Ptr(AX, 4294967296, 32), vreg(t, "Y1")}},
|
||||
{"EVEX disp32+", "VMOVDQU32", []Operand{Ptr(AX, 4294967296, 64), vreg(t, "Z1")}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", c.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestImmediateTruncation pins the toolchain-matching truncations inside the
|
||||
// accepted 32-bit span: the narrower fields take the low bits silently, byte
|
||||
// for byte with `go tool asm` (which rejects none of these).
|
||||
func TestImmediateTruncation(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
mnem string
|
||||
ops []Operand
|
||||
want string
|
||||
}{
|
||||
{"ADDB $256,BL", "ADDB", []Operand{Imm(256), BL}, "80c300"},
|
||||
{"ADDB $1000,BL", "ADDB", []Operand{Imm(1000), BL}, "80c3e8"},
|
||||
{"MOVB $256,AL", "MOVB", []Operand{Imm(256), AL}, "b000"},
|
||||
{"MOVB $-129,AL", "MOVB", []Operand{Imm(-129), AL}, "b07f"},
|
||||
{"MOVW $65536,AX", "MOVW", []Operand{Imm(65536), AX}, "66b80000"},
|
||||
{"MOVW $65535,AX", "MOVW", []Operand{Imm(65535), AX}, "66b8ffff"},
|
||||
{"MOVW $-32769,AX", "MOVW", []Operand{Imm(-32769), AX}, "66b8ff7f"},
|
||||
{"MOVL $4294967295,AX", "MOVL", []Operand{Imm(4294967295), AX}, "b8ffffffff"},
|
||||
{"ADDQ $4294967295,AX", "ADDQ", []Operand{Imm(4294967295), AX}, "4805ffffffff"},
|
||||
{"CMPB BL,$255", "CMPB", []Operand{BL, Imm(255)}, "80fbff"},
|
||||
{"CMPQ AX,$4294967295", "CMPQ", []Operand{AX, Imm(4294967295)}, "483dffffffff"},
|
||||
{"MOVQ $4294967295,0(AX)", "MOVQ", []Operand{Imm(4294967295), Ptr(AX, 0, 8)}, "48c700ffffffff"},
|
||||
{"SHLQ $255,AX", "SHLQ", []Operand{Imm(255), AX}, "48c1e0ff"},
|
||||
{"SHLQ $0,AX", "SHLQ", []Operand{Imm(0), AX}, "48c1e000"},
|
||||
// The one form beyond the 32-bit span: the imm64 MOVQ register move.
|
||||
{"MOVQ $4294967296,AX", "MOVQ", []Operand{Imm(4294967296), AX}, "48b80000000001000000"},
|
||||
{"MOVQ disp32 max", "MOVQ", []Operand{Ptr(AX, 2147483647, 8), BX}, "488b98ffffff7f"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
code, err := Encode(c.mnem, c.ops...)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must
|
||||
// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size
|
||||
// letter) is not reported as encodable.
|
||||
func TestEncodableCmovSize(t *testing.T) {
|
||||
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} {
|
||||
if Encodable(m) {
|
||||
t.Errorf("Encodable(%q) = true, want false", m)
|
||||
}
|
||||
}
|
||||
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} {
|
||||
if !Encodable(m) {
|
||||
t.Errorf("Encodable(%q) = false, want true", m)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
||||
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
||||
func TestSSEBinGroundTruth(t *testing.T) {
|
||||
|
||||
+31
-14
@@ -509,7 +509,8 @@ var evexBcastTable = map[string]evexBcastSpec{
|
||||
}
|
||||
|
||||
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
|
||||
// move table).
|
||||
// move table). vecOK and xmmOnly mirror the VEX twin's operand rules: a
|
||||
// scalar move (vecOK false, xmmOnly true) takes XMM↔memory operands only.
|
||||
type evexMoveSpec struct {
|
||||
mapSel int
|
||||
pp int
|
||||
@@ -517,32 +518,34 @@ type evexMoveSpec struct {
|
||||
store byte // vector → r/m
|
||||
w int
|
||||
n [3]int
|
||||
vecOK bool // the non-memory operand may be a vector register
|
||||
xmmOnly bool // wider than XMM registers are rejected
|
||||
}
|
||||
|
||||
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
||||
var evexMoveTable = map[string]evexMoveSpec{
|
||||
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
||||
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
||||
// semantics).
|
||||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
||||
// encoding).
|
||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
|
||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
|
||||
// EVEX.128/256/512, aligned packed moves.
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
|
||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
|
||||
// EVEX.128/256/512.66.0F, aligned integer moves.
|
||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
|
||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
|
||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
||||
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
||||
// three-operand register form is not supported).
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
|
||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
|
||||
}
|
||||
|
||||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||||
@@ -1022,6 +1025,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
|
||||
var rm Operand
|
||||
switch {
|
||||
case srcIsVec && dstIsVec:
|
||||
// A store-form reg-reg move, the layout the Go assembler uses; a
|
||||
// scalar move has no two-register form at all (the register form
|
||||
// takes three operands), matching the VEX twin's vecOK rule.
|
||||
if !ms.vecOK {
|
||||
return fmt.Errorf("%s does not take two vector registers", mnem)
|
||||
}
|
||||
reg, rm = srcReg, dst
|
||||
case srcIsVec:
|
||||
if !memOperand(dst) {
|
||||
@@ -1037,6 +1046,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
|
||||
default:
|
||||
return fmt.Errorf("%s needs a vector register operand", mnem)
|
||||
}
|
||||
// The scalar move is 128-bit only, so the register the length follows
|
||||
// must be an XMM (the VEX twin's xmmOnly rule; EVEX also reaches ZMM,
|
||||
// hence the inequality rather than a YMM test).
|
||||
if ms.xmmOnly && reg.size != 16 {
|
||||
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
||||
}
|
||||
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||||
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
|
||||
}
|
||||
@@ -1171,9 +1186,6 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
|
||||
if r.idx&16 != 0 {
|
||||
xBar = 0
|
||||
}
|
||||
if r.idx&16 != 0 {
|
||||
xBar = 0
|
||||
}
|
||||
case Mem:
|
||||
var err error
|
||||
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
|
||||
@@ -1232,6 +1244,11 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
|
||||
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
|
||||
sib = -1
|
||||
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
|
||||
// The disp32 fallback bounds the displacement by int32, and the
|
||||
// compressed disp8 form reaches at most ±127×64, well inside it.
|
||||
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
|
||||
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
|
||||
}
|
||||
if !m.HasBase && !m.HasIndex {
|
||||
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
|
||||
}
|
||||
|
||||
@@ -675,6 +675,15 @@ func TestEvexErrors(t *testing.T) {
|
||||
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
|
||||
// VEX-only mnemonics reject registers only EVEX can encode.
|
||||
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
|
||||
// The scalar EVEX move matches its VEX twin and the Go assembler:
|
||||
// XMM↔memory only, never reg-reg and never a wider register (the
|
||||
// toolchain rejects every one of these shapes).
|
||||
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
|
||||
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
|
||||
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
|
||||
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
|
||||
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
|
||||
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||
|
||||
+86
-21
@@ -173,7 +173,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
if dstReg.needsREX(size) {
|
||||
i.rexForced = true
|
||||
}
|
||||
i.imm = immediate(v, size, true)
|
||||
imm, err := immediate(v, size, true)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = imm
|
||||
return e.emit(i)
|
||||
}
|
||||
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
|
||||
@@ -185,7 +189,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immediate(int64(src), size, false)
|
||||
imm, err := immediate(int64(src), size, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = imm
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("MOV: invalid operands")
|
||||
@@ -297,11 +305,15 @@ func (e *enc) encodeALU(op struct {
|
||||
|
||||
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||||
if size == 1 {
|
||||
immBytes, err := immediate(imm, 1, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i := newInstr(1, []byte{0x80})
|
||||
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{byte(int8(imm))}
|
||||
i.imm = immBytes
|
||||
return e.emit(i)
|
||||
}
|
||||
if fits8(imm) {
|
||||
@@ -319,7 +331,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||||
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
|
||||
i := newInstr(size, []byte{accOp})
|
||||
i.imm = immediate(imm, size, false)
|
||||
immBytes, err := immediate(imm, size, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immBytes
|
||||
return e.emit(i)
|
||||
}
|
||||
// 0x81 /digit, imm16/imm32.
|
||||
@@ -327,7 +343,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immediate(imm, size, false)
|
||||
immBytes, err := immediate(imm, size, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immBytes
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
@@ -348,7 +368,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
|
||||
op = 0xA8
|
||||
}
|
||||
i := newInstr(size, []byte{op})
|
||||
i.imm = immediate(int64(imm), size, false)
|
||||
immBytes, err := immediate(int64(imm), size, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immBytes
|
||||
return e.emit(i)
|
||||
}
|
||||
op := byte(0xF7)
|
||||
@@ -359,7 +383,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
|
||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immediate(int64(imm), size, false)
|
||||
immBytes, err := immediate(int64(imm), size, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immBytes
|
||||
return e.emit(i)
|
||||
}
|
||||
srcReg, ok := src.(Reg)
|
||||
@@ -457,7 +485,13 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
||||
}
|
||||
return e.emit(i)
|
||||
}
|
||||
// 0xC0 (8-bit) / 0xC1, imm8.
|
||||
// 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm
|
||||
// rejects negative and ≥256 counts, and the hardware masks the count, so
|
||||
// a silent truncation ($300 encoding 44) would shift by a different
|
||||
// amount than the source states.
|
||||
if imm < 0 || imm > 255 {
|
||||
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
|
||||
}
|
||||
op := byte(0xC1)
|
||||
if size == 1 {
|
||||
op = 0xC0
|
||||
@@ -466,7 +500,7 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
||||
if err := setRMDigit(i, digit, dst, size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = []byte{byte(int8(imm))}
|
||||
i.imm = []byte{byte(imm)}
|
||||
return e.emit(i)
|
||||
}
|
||||
|
||||
@@ -508,7 +542,11 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
|
||||
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immediate(int64(imm), size, false)
|
||||
immBytes, err := immediate(int64(imm), size, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i.imm = immBytes
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
||||
@@ -516,10 +554,21 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
|
||||
|
||||
// --- PUSH / POP -------------------------------------------------------------
|
||||
|
||||
func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
||||
func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
|
||||
if len(ops) != 1 {
|
||||
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
// In 64-bit mode go tool asm knows the 64-bit push (the default, with or
|
||||
// without the Q suffix) and the 16-bit W form with its 0x66 operand-size
|
||||
// prefix, and rejects the B and L spellings outright ("illegal in 64-bit
|
||||
// mode"); silently widening those would push a different width than the
|
||||
// source states.
|
||||
switch size {
|
||||
case 0, 8, 2:
|
||||
default:
|
||||
return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode")
|
||||
}
|
||||
w16 := size == 2
|
||||
switch op := ops[0].(type) {
|
||||
case Reg:
|
||||
base := byte(0x50) // PUSH r; POP is 0x58
|
||||
@@ -527,7 +576,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
||||
base = 0x58
|
||||
}
|
||||
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
|
||||
i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
|
||||
i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
|
||||
i.rexB = op.idx >= 8
|
||||
return e.emit(i)
|
||||
case Mem:
|
||||
@@ -537,7 +586,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
||||
opc = 0x8F // POP r/m: /0
|
||||
digit = 0
|
||||
}
|
||||
i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1}
|
||||
i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1}
|
||||
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -547,10 +596,17 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
|
||||
return fmt.Errorf("POP does not take an immediate")
|
||||
}
|
||||
if fits8(int64(op)) {
|
||||
i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
|
||||
i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
|
||||
return e.emit(i)
|
||||
}
|
||||
i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))}
|
||||
// PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
|
||||
// immediate by the same signed/unsigned 32-bit span as every other
|
||||
// scalar immediate.
|
||||
immBytes, err := immediate(int64(op), 8, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes}
|
||||
return e.emit(i)
|
||||
}
|
||||
return fmt.Errorf("PUSH/POP: invalid operand")
|
||||
@@ -639,19 +695,28 @@ func (e *enc) encodeJcc(cc int, ops []Operand) error {
|
||||
// immediate encodes an immediate of the given operand size. full64 selects the
|
||||
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
|
||||
// sign-extended immediate is used for 64-bit operands.
|
||||
func immediate(v int64, size int, full64 bool) []byte {
|
||||
//
|
||||
// The span mirrors go tool asm: every scalar immediate must fit a signed or
|
||||
// unsigned 32-bit word, and the narrower fields then take the low bits
|
||||
// silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the
|
||||
// imm64 form may exceed the span; anything wider elsewhere is an error rather
|
||||
// than a truncation the source never asked for.
|
||||
func immediate(v int64, size int, full64 bool) ([]byte, error) {
|
||||
if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) {
|
||||
return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v)
|
||||
}
|
||||
switch size {
|
||||
case 1:
|
||||
return []byte{byte(int8(v))}
|
||||
return []byte{byte(int8(v))}, nil
|
||||
case 2:
|
||||
return le16(v)
|
||||
return le16(v), nil
|
||||
case 4:
|
||||
return le32(v)
|
||||
return le32(v), nil
|
||||
default: // 8
|
||||
if full64 {
|
||||
return le64(v)
|
||||
return le64(v), nil
|
||||
}
|
||||
return le32(v) // sign-extended imm32
|
||||
return le32(v), nil // sign-extended imm32
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user