fix(amd64): correct guard displacements, frameless FP offsets and immediate ranges

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-19 23:49:07 +02:00
parent 94e09e8070
commit 4258131a3a
8 changed files with 377 additions and 74 deletions
+28 -28
View File
@@ -114,6 +114,11 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] { if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
continue continue
} }
// A zero-operand jump parses; its arity is reported during
// emission (encodeJump), so the layout must not index Operands.
if len(s.Operands) != 1 {
continue
}
name, ok := labelName(s.Operands[0]) name, ok := labelName(s.Operands[0])
if !ok { if !ok {
continue // reported during emission continue // reported during emission
@@ -137,28 +142,22 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
} }
if fi.splitClass == 2 && !guardJBlong { if fi.splitClass == 2 && !guardJBlong {
// The underflow JB sits before the CMPQ; its displacement spans // The underflow JB sits before the CMPQ; its displacement spans
// the rest of the guard plus the prologue and the body. // the rest of the guard plus the prologue and the body. The JB
jbLen := 2 // is still the short form this branch tests (relaxing it is this
if guardJBlong { // branch's job), so guardLen is taken with a short JB and the
jbLen = 6 // subtraction drops the prefix and the JB's own 2 bytes.
} rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2)
rest := fi.guardLen(guardJBlong, guardJBElong) - (9 + 3 + 7 + jbLen)
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) { if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
guardJBlong = true guardJBlong = true
changed = true changed = true
} }
} }
// The morestack JMP returns to the function start, so its // The morestack JMP returns to the function start, so its
// displacement is the negated distance from its own end. // displacement is the negated distance from its own end; while it is
if !moreJMPlong { // still short, its own length is 2 bytes.
jmpLen := 2 if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) {
if moreJMPlong { moreJMPlong = true
jmpLen = 5 changed = true
}
if !fits8(-int64(guard + len(fi.prologue) + bodyLen + 5 + jmpLen)) {
moreJMPlong = true
changed = true
}
} }
if !changed { if !changed {
break break
@@ -181,7 +180,15 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
var out []byte var out []byte
var patches []sbPatch var patches []sbPatch
if fi.needSplit { if fi.needSplit {
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+2)+len(fi.prologue)+bodyLen)) // The JBE ends the guard, so its displacement is the prologue plus
// the body; the underflow JB additionally spans the trailing CMPQ and
// JBE, whose combined length is guardLen minus the prefix and the
// JB's own length (2 short, 6 long).
jbLen := 2
if guardJBlong {
jbLen = 6
}
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen))
out = append(out, guard...) out = append(out, guard...)
patches = append(patches, tlsPatch) patches = append(patches, tlsPatch)
} }
@@ -344,7 +351,10 @@ func computeFrame(t *ast.Text) frameInfo {
// pass one extra slot, and the virtual SP is the hardware SP. // pass one extra slot, and the virtual SP is the hardware SP.
fi.size = 8 fi.size = 8
fi.useFP = true fi.useFP = true
fi.fpAdjust = int64(fi.size) + 16 // return address + saved BP + args base // The push is the frame: the saved BP sits at SP+0 and the
// return address at SP+8, so arguments begin at SP+16. Unlike
// a SUBQ frame, the 8-byte size must not be added again.
fi.fpAdjust = 16
fi.spAdjust = 0 fi.spAdjust = 0
fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
fi.epilogue = []byte{0x5D} // POPQ BP fi.epilogue = []byte{0x5D} // POPQ BP
@@ -426,16 +436,6 @@ func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
} }
} }
// moreLen returns the byte length of the trailing morestack block: the CALL
// (always rel32) plus the JMP back to the function start.
func moreLen(jmpLong bool) int {
jmp := 2
if jmpLong {
jmp = 5
}
return 5 + jmp
}
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the // buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
// already-computed displacements of the conditional branches that jump to the // already-computed displacements of the conditional branches that jump to the
// morestack block (unused in classes without them). The TLS load carries a // morestack block (unused in classes without them). The TLS load carries a
+64
View File
@@ -160,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
} }
} }
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
// function containing a CALL receives: the PUSHQ BP prologue with no stack
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
// Go assembler produces. The push is the frame, so the offset must not count
// it twice.
func TestAssembleFramelessCall(t *testing.T) {
f, errs := parser.Parse("frameless_call_amd64.s", `
#include "textflag.h"
TEXT ·withcall(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
CALL ·other(SB)
MOVQ AX, ret+8(FP)
RET
TEXT ·other(SB), NOSPLIT, $0-0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
for _, r := range img.Funcs[0].Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
// From `go tool objdump` of the Go-assembled function:
// PUSHQ BP 55
// MOVQ SP, BP 4889e5
// MOVQ 0x10(SP), AX 488b442410
// CALL other e800000000
// MOVQ AX, 0x18(SP) 4889442418
// POPQ BP 5d
// RET c3
want := []byte{
0x55,
0x48, 0x89, 0xe5,
0x48, 0x8b, 0x44, 0x24, 0x10,
0xe8, 0x00, 0x00, 0x00, 0x00,
0x48, 0x89, 0x44, 0x24, 0x18,
0x5d,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style // TestAssembleFrame verifies a function with a non-zero frame: the Go-style
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against // prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
// the bytes the Go assembler produces. // the bytes the Go assembler produces.
@@ -353,6 +404,19 @@ TEXT ·pf(SB), NOSPLIT, $0
} }
} }
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
// the parser does not arity-check mnemonics) is rejected with an error rather
// than panicking in the layout loop, which indexes Operands[0] before the
// emission pass gets a chance to diagnose the arity.
func TestAssembleBareJump(t *testing.T) {
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
if _, _, err := Assemble(fn); err == nil {
t.Errorf("%s with no operand: expected an error, got none", mnem)
}
}
}
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm // TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything // emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
// larger. The intermediate 129..255 range used to encode an ADD with a // larger. The intermediate 129..255 range used to encode an ADD with a
+8 -3
View File
@@ -36,10 +36,15 @@ func Encodable(mnemonic string) bool {
} }
// CMOV carries size then condition (CMOVLGT); SET carries the condition // CMOV carries size then condition (CMOVLGT); SET carries the condition
// alone (SETNE). // alone (SETNE). The size letter is checked exactly as encodeCmov does,
// so a spelling like CMOVBGT is not reported encodable when Encode
// would reject it.
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 { if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
if _, ok := jccMap[rest[1:]]; ok { switch rest[0] {
return true case 'W', 'L', 'Q':
if _, ok := jccMap[rest[1:]]; ok {
return true
}
} }
} }
if rest, ok := strings.CutPrefix(upper, "SET"); ok { if rest, ok := strings.CutPrefix(upper, "SET"); ok {
+8 -2
View File
@@ -117,9 +117,9 @@ func (e *enc) encode(mnem string, ops []Operand) error {
case "IMUL", "IMUL3": case "IMUL", "IMUL3":
return e.encodeImul(ops, size) return e.encodeImul(ops, size)
case "PUSH": case "PUSH":
return e.encodePushPop(ops, true) return e.encodePushPop(ops, size, true)
case "POP": case "POP":
return e.encodePushPop(ops, false) return e.encodePushPop(ops, size, false)
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT": case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
return e.encodeCount(base, ops, size) return e.encodeCount(base, ops, size)
case "BSWAP": case "BSWAP":
@@ -345,6 +345,12 @@ func setMem(i *instr, regField int, m Mem) error {
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths. // a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) { func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
sib = -1 sib = -1
// A displacement wider than int32 fits no encoding form; truncating it
// would address a different location, and go tool asm reports "offset
// too large" for the same operand.
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
}
// RIP-relative: neither base nor index. // RIP-relative: neither base nor index.
if !m.HasBase && !m.HasIndex { if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101 return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
+137
View File
@@ -149,6 +149,45 @@ func TestPushPop(t *testing.T) {
checkSyntax(t, "push rbx", "PUSHQ", BX) checkSyntax(t, "push rbx", "PUSHQ", BX)
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8}) checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5)) checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
// The W spelling carries the 0x66 operand-size prefix, byte for byte
// with go tool asm; the L and B spellings are illegal in 64-bit mode
// there and rejected here rather than silently widened.
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"PUSHW AX", "PUSHW", []Operand{AX}, "6650"},
{"POPW AX", "POPW", []Operand{AX}, "6658"},
{"PUSHW $5", "PUSHW", []Operand{Imm(5)}, "666a05"},
{"PUSHW (AX)", "PUSHW", []Operand{Ptr(AX, 0, 2)}, "66ff30"},
{"PUSHQ AX", "PUSHQ", []Operand{AX}, "50"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"PUSHL AX", "PUSHL", []Operand{AX}},
{"PUSHL R8", "PUSHL", []Operand{Reg{idx: 8, size: 8}}},
{"POPL BX", "POPL", []Operand{BX}},
{"PUSHB AX", "PUSHB", []Operand{AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
} }
func TestUnary(t *testing.T) { func TestUnary(t *testing.T) {
@@ -397,6 +436,104 @@ func TestScalarErrors(t *testing.T) {
} }
} }
// TestImmediateOutOfRange pins the go-tool-asm parity of the immediate and
// displacement spans: a scalar immediate must fit a signed or unsigned 32-bit
// word (only MOVQ reg, $imm takes the full int64), a scalar shift count must
// be an unsigned byte, and a displacement must fit int32. Every rejected
// shape here is rejected by `go tool asm` too; every accepted one encodes the
// same bytes.
func TestImmediateOutOfRange(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
}{
{"SHLQ count 300", "SHLQ", []Operand{Imm(300), AX}},
{"SHLQ count -1", "SHLQ", []Operand{Imm(-1), AX}},
{"SHLW count 256", "SHLW", []Operand{Imm(256), DX}},
{"SHLB count 300", "SHLB", []Operand{Imm(300), BL}},
{"MOVL imm32+", "MOVL", []Operand{Imm(4294967296), AX}},
{"MOVL imm32-", "MOVL", []Operand{Imm(-2147483649), AX}},
{"MOVW imm32+", "MOVW", []Operand{Imm(4294967296), AX}},
{"MOVB imm32+", "MOVB", []Operand{Imm(4294967296), AL}},
{"ADDB imm32+", "ADDB", []Operand{Imm(4294967296), AL}},
{"ADDL imm32+", "ADDL", []Operand{Imm(4294967296), AX}},
{"ADDQ imm32+", "ADDQ", []Operand{Imm(8589934592), AX}},
{"CMPQ imm32+", "CMPQ", []Operand{AX, Imm(4294967296)}},
{"CMPQ imm32-", "CMPQ", []Operand{AX, Imm(-2147483649)}},
{"TESTL imm32+", "TESTL", []Operand{Imm(4294967296), AX}},
{"IMUL3L imm32+", "IMUL3L", []Operand{Imm(4294967296), CX, DX}},
{"PUSHQ imm32+", "PUSHQ", []Operand{Imm(4294967296)}},
{"MOVQ mem imm32+", "MOVQ", []Operand{Imm(4294967296), Ptr(AX, 0, 8)}},
{"disp32+", "MOVQ", []Operand{Ptr(AX, 4294967296, 8), BX}},
{"disp32+ max", "MOVQ", []Operand{Ptr(AX, 2147483648, 8), BX}},
{"disp32-", "MOVQ", []Operand{Ptr(AX, -2147483649, 8), BX}},
{"VEX disp32+", "VMOVDQU", []Operand{Ptr(AX, 4294967296, 32), vreg(t, "Y1")}},
{"EVEX disp32+", "VMOVDQU32", []Operand{Ptr(AX, 4294967296, 64), vreg(t, "Z1")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestImmediateTruncation pins the toolchain-matching truncations inside the
// accepted 32-bit span: the narrower fields take the low bits silently, byte
// for byte with `go tool asm` (which rejects none of these).
func TestImmediateTruncation(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ADDB $256,BL", "ADDB", []Operand{Imm(256), BL}, "80c300"},
{"ADDB $1000,BL", "ADDB", []Operand{Imm(1000), BL}, "80c3e8"},
{"MOVB $256,AL", "MOVB", []Operand{Imm(256), AL}, "b000"},
{"MOVB $-129,AL", "MOVB", []Operand{Imm(-129), AL}, "b07f"},
{"MOVW $65536,AX", "MOVW", []Operand{Imm(65536), AX}, "66b80000"},
{"MOVW $65535,AX", "MOVW", []Operand{Imm(65535), AX}, "66b8ffff"},
{"MOVW $-32769,AX", "MOVW", []Operand{Imm(-32769), AX}, "66b8ff7f"},
{"MOVL $4294967295,AX", "MOVL", []Operand{Imm(4294967295), AX}, "b8ffffffff"},
{"ADDQ $4294967295,AX", "ADDQ", []Operand{Imm(4294967295), AX}, "4805ffffffff"},
{"CMPB BL,$255", "CMPB", []Operand{BL, Imm(255)}, "80fbff"},
{"CMPQ AX,$4294967295", "CMPQ", []Operand{AX, Imm(4294967295)}, "483dffffffff"},
{"MOVQ $4294967295,0(AX)", "MOVQ", []Operand{Imm(4294967295), Ptr(AX, 0, 8)}, "48c700ffffffff"},
{"SHLQ $255,AX", "SHLQ", []Operand{Imm(255), AX}, "48c1e0ff"},
{"SHLQ $0,AX", "SHLQ", []Operand{Imm(0), AX}, "48c1e000"},
// The one form beyond the 32-bit span: the imm64 MOVQ register move.
{"MOVQ $4294967296,AX", "MOVQ", []Operand{Imm(4294967296), AX}, "48b80000000001000000"},
{"MOVQ disp32 max", "MOVQ", []Operand{Ptr(AX, 2147483647, 8), BX}, "488b98ffffff7f"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must
// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size
// letter) is not reported as encodable.
func TestEncodableCmovSize(t *testing.T) {
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} {
if Encodable(m) {
t.Errorf("Encodable(%q) = true, want false", m)
}
}
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} {
if !Encodable(m) {
t.Errorf("Encodable(%q) = false, want true", m)
}
}
}
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family // TestSSEBinGroundTruth checks the legacy packed/scalar binary family
// byte for byte (no prefix / 66 / F2 / F3 variants). // byte for byte (no prefix / 66 / F2 / F3 variants).
func TestSSEBinGroundTruth(t *testing.T) { func TestSSEBinGroundTruth(t *testing.T) {
+37 -20
View File
@@ -509,40 +509,43 @@ var evexBcastTable = map[string]evexBcastSpec{
} }
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX // evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
// move table). // move table). vecOK and xmmOnly mirror the VEX twin's operand rules: a
// scalar move (vecOK false, xmmOnly true) takes XMM↔memory operands only.
type evexMoveSpec struct { type evexMoveSpec struct {
mapSel int mapSel int
pp int pp int
load byte // r/m → vector load byte // r/m → vector
store byte // vector → r/m store byte // vector → r/m
w int w int
n [3]int n [3]int
vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected
} }
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{ var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move. // EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move. // EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple // F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics). // semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding). // encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move. // EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}}, "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512, aligned packed moves. // EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}}, "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}}, "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.66.0F, aligned integer moves. // EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the // EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported). // three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}}, "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
} }
// isEvex reports whether the mnemonic has an EVEX encoding we handle. // isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -1022,6 +1025,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
var rm Operand var rm Operand
switch { switch {
case srcIsVec && dstIsVec: case srcIsVec && dstIsVec:
// A store-form reg-reg move, the layout the Go assembler uses; a
// scalar move has no two-register form at all (the register form
// takes three operands), matching the VEX twin's vecOK rule.
if !ms.vecOK {
return fmt.Errorf("%s does not take two vector registers", mnem)
}
reg, rm = srcReg, dst reg, rm = srcReg, dst
case srcIsVec: case srcIsVec:
if !memOperand(dst) { if !memOperand(dst) {
@@ -1037,6 +1046,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
default: default:
return fmt.Errorf("%s needs a vector register operand", mnem) return fmt.Errorf("%s needs a vector register operand", mnem)
} }
// The scalar move is 128-bit only, so the register the length follows
// must be an XMM (the VEX twin's xmmOnly rule; EVEX also reaches ZMM,
// hence the inequality rather than a YMM test).
if ms.xmmOnly && reg.size != 16 {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n} spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx) return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
} }
@@ -1171,9 +1186,6 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
if r.idx&16 != 0 { if r.idx&16 != 0 {
xBar = 0 xBar = 0
} }
if r.idx&16 != 0 {
xBar = 0
}
case Mem: case Mem:
var err error var err error
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll]) modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
@@ -1232,6 +1244,11 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) { func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
sib = -1 sib = -1
xBar, bBar = 1, 1 // inverted bits: 1 = no extension xBar, bBar = 1, 1 // inverted bits: 1 = no extension
// The disp32 fallback bounds the displacement by int32, and the
// compressed disp8 form reaches at most ±127×64, well inside it.
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
}
if !m.HasBase && !m.HasIndex { if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
} }
+9
View File
@@ -675,6 +675,15 @@ func TestEvexErrors(t *testing.T) {
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}}, {"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
// VEX-only mnemonics reject registers only EVEX can encode. // VEX-only mnemonics reject registers only EVEX can encode.
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}}, {"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
// The scalar EVEX move matches its VEX twin and the Go assembler:
// XMM↔memory only, never reg-reg and never a wider register (the
// toolchain rejects every one of these shapes).
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
} }
for _, c := range cases { for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil { if _, err := Encode(c.mnem, c.ops...); err == nil {
+86 -21
View File
@@ -173,7 +173,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
if dstReg.needsREX(size) { if dstReg.needsREX(size) {
i.rexForced = true i.rexForced = true
} }
i.imm = immediate(v, size, true) imm, err := immediate(v, size, true)
if err != nil {
return err
}
i.imm = imm
return e.emit(i) return e.emit(i)
} }
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. // MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
@@ -185,7 +189,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
if err := setRMDigit(i, 0, dst, size); err != nil { if err := setRMDigit(i, 0, dst, size); err != nil {
return err return err
} }
i.imm = immediate(int64(src), size, false) imm, err := immediate(int64(src), size, false)
if err != nil {
return err
}
i.imm = imm
return e.emit(i) return e.emit(i)
} }
return fmt.Errorf("MOV: invalid operands") return fmt.Errorf("MOV: invalid operands")
@@ -297,11 +305,15 @@ func (e *enc) encodeALU(op struct {
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error { func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if size == 1 { if size == 1 {
immBytes, err := immediate(imm, 1, false)
if err != nil {
return err
}
i := newInstr(1, []byte{0x80}) i := newInstr(1, []byte{0x80})
if err := setRMDigit(i, digit, dst, 1); err != nil { if err := setRMDigit(i, digit, dst, 1); err != nil {
return err return err
} }
i.imm = []byte{byte(int8(imm))} i.imm = immBytes
return e.emit(i) return e.emit(i)
} }
if fits8(imm) { if fits8(imm) {
@@ -319,7 +331,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if r, ok := dst.(Reg); ok && r.idx == 0 { if r, ok := dst.(Reg); ok && r.idx == 0 {
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit] accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
i := newInstr(size, []byte{accOp}) i := newInstr(size, []byte{accOp})
i.imm = immediate(imm, size, false) immBytes, err := immediate(imm, size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i) return e.emit(i)
} }
// 0x81 /digit, imm16/imm32. // 0x81 /digit, imm16/imm32.
@@ -327,7 +343,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if err := setRMDigit(i, digit, dst, size); err != nil { if err := setRMDigit(i, digit, dst, size); err != nil {
return err return err
} }
i.imm = immediate(imm, size, false) immBytes, err := immediate(imm, size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i) return e.emit(i)
} }
@@ -348,7 +368,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
op = 0xA8 op = 0xA8
} }
i := newInstr(size, []byte{op}) i := newInstr(size, []byte{op})
i.imm = immediate(int64(imm), size, false) immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i) return e.emit(i)
} }
op := byte(0xF7) op := byte(0xF7)
@@ -359,7 +383,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
if err := setRMDigit(i, 0, dst, size); err != nil { if err := setRMDigit(i, 0, dst, size); err != nil {
return err return err
} }
i.imm = immediate(int64(imm), size, false) immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i) return e.emit(i)
} }
srcReg, ok := src.(Reg) srcReg, ok := src.(Reg)
@@ -457,7 +485,13 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
} }
return e.emit(i) return e.emit(i)
} }
// 0xC0 (8-bit) / 0xC1, imm8. // 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm
// rejects negative and ≥256 counts, and the hardware masks the count, so
// a silent truncation ($300 encoding 44) would shift by a different
// amount than the source states.
if imm < 0 || imm > 255 {
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
}
op := byte(0xC1) op := byte(0xC1)
if size == 1 { if size == 1 {
op = 0xC0 op = 0xC0
@@ -466,7 +500,7 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
if err := setRMDigit(i, digit, dst, size); err != nil { if err := setRMDigit(i, digit, dst, size); err != nil {
return err return err
} }
i.imm = []byte{byte(int8(imm))} i.imm = []byte{byte(imm)}
return e.emit(i) return e.emit(i)
} }
@@ -508,7 +542,11 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
if err := setRM(i, dstReg, ops[1], size); err != nil { if err := setRM(i, dstReg, ops[1], size); err != nil {
return err return err
} }
i.imm = immediate(int64(imm), size, false) immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i) return e.emit(i)
} }
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops)) return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
@@ -516,10 +554,21 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
// --- PUSH / POP ------------------------------------------------------------- // --- PUSH / POP -------------------------------------------------------------
func (e *enc) encodePushPop(ops []Operand, push bool) error { func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
if len(ops) != 1 { if len(ops) != 1 {
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops)) return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
} }
// In 64-bit mode go tool asm knows the 64-bit push (the default, with or
// without the Q suffix) and the 16-bit W form with its 0x66 operand-size
// prefix, and rejects the B and L spellings outright ("illegal in 64-bit
// mode"); silently widening those would push a different width than the
// source states.
switch size {
case 0, 8, 2:
default:
return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode")
}
w16 := size == 2
switch op := ops[0].(type) { switch op := ops[0].(type) {
case Reg: case Reg:
base := byte(0x50) // PUSH r; POP is 0x58 base := byte(0x50) // PUSH r; POP is 0x58
@@ -527,7 +576,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
base = 0x58 base = 0x58
} }
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed. // PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1} i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
i.rexB = op.idx >= 8 i.rexB = op.idx >= 8
return e.emit(i) return e.emit(i)
case Mem: case Mem:
@@ -537,7 +586,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
opc = 0x8F // POP r/m: /0 opc = 0x8F // POP r/m: /0
digit = 0 digit = 0
} }
i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1} i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1}
if err := setRMDigit(i, digit, ops[0], 8); err != nil { if err := setRMDigit(i, digit, ops[0], 8); err != nil {
return err return err
} }
@@ -547,10 +596,17 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
return fmt.Errorf("POP does not take an immediate") return fmt.Errorf("POP does not take an immediate")
} }
if fits8(int64(op)) { if fits8(int64(op)) {
i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}} i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
return e.emit(i) return e.emit(i)
} }
i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))} // PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
// immediate by the same signed/unsigned 32-bit span as every other
// scalar immediate.
immBytes, err := immediate(int64(op), 8, false)
if err != nil {
return err
}
i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes}
return e.emit(i) return e.emit(i)
} }
return fmt.Errorf("PUSH/POP: invalid operand") return fmt.Errorf("PUSH/POP: invalid operand")
@@ -639,19 +695,28 @@ func (e *enc) encodeJcc(cc int, ops []Operand) error {
// immediate encodes an immediate of the given operand size. full64 selects the // immediate encodes an immediate of the given operand size. full64 selects the
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit // 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
// sign-extended immediate is used for 64-bit operands. // sign-extended immediate is used for 64-bit operands.
func immediate(v int64, size int, full64 bool) []byte { //
// The span mirrors go tool asm: every scalar immediate must fit a signed or
// unsigned 32-bit word, and the narrower fields then take the low bits
// silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the
// imm64 form may exceed the span; anything wider elsewhere is an error rather
// than a truncation the source never asked for.
func immediate(v int64, size int, full64 bool) ([]byte, error) {
if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) {
return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v)
}
switch size { switch size {
case 1: case 1:
return []byte{byte(int8(v))} return []byte{byte(int8(v))}, nil
case 2: case 2:
return le16(v) return le16(v), nil
case 4: case 4:
return le32(v) return le32(v), nil
default: // 8 default: // 8
if full64 { if full64 {
return le64(v) return le64(v), nil
} }
return le32(v) // sign-extended imm32 return le32(v), nil // sign-extended imm32
} }
} }