From 4258131a3a49f0334d12964e40ed4c20b9fa263a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sat, 19 Sep 2026 23:49:07 +0200 Subject: [PATCH] fix(amd64): correct guard displacements, frameless FP offsets and immediate ranges Assisted-by: GLM 5.3 --- asm/assemble.go | 56 +++++++++--------- asm/assemble_test.go | 64 ++++++++++++++++++++ asm/encodable.go | 11 +++- asm/encode.go | 10 +++- asm/encode_test.go | 137 +++++++++++++++++++++++++++++++++++++++++++ asm/evex.go | 57 +++++++++++------- asm/evex_test.go | 9 +++ asm/instrs.go | 107 ++++++++++++++++++++++++++------- 8 files changed, 377 insertions(+), 74 deletions(-) diff --git a/asm/assemble.go b/asm/assemble.go index d8d31e0..bf7778b 100644 --- a/asm/assemble.go +++ b/asm/assemble.go @@ -114,6 +114,11 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [ if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] { continue } + // A zero-operand jump parses; its arity is reported during + // emission (encodeJump), so the layout must not index Operands. + if len(s.Operands) != 1 { + continue + } name, ok := labelName(s.Operands[0]) if !ok { continue // reported during emission @@ -137,28 +142,22 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [ } if fi.splitClass == 2 && !guardJBlong { // The underflow JB sits before the CMPQ; its displacement spans - // the rest of the guard plus the prologue and the body. - jbLen := 2 - if guardJBlong { - jbLen = 6 - } - rest := fi.guardLen(guardJBlong, guardJBElong) - (9 + 3 + 7 + jbLen) + // the rest of the guard plus the prologue and the body. The JB + // is still the short form this branch tests (relaxing it is this + // branch's job), so guardLen is taken with a short JB and the + // subtraction drops the prefix and the JB's own 2 bytes. + rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2) if !fits8(int64(rest + len(fi.prologue) + bodyLen)) { guardJBlong = true changed = true } } // The morestack JMP returns to the function start, so its - // displacement is the negated distance from its own end. - if !moreJMPlong { - jmpLen := 2 - if moreJMPlong { - jmpLen = 5 - } - if !fits8(-int64(guard + len(fi.prologue) + bodyLen + 5 + jmpLen)) { - moreJMPlong = true - changed = true - } + // displacement is the negated distance from its own end; while it is + // still short, its own length is 2 bytes. + if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) { + moreJMPlong = true + changed = true } if !changed { break @@ -181,7 +180,15 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [ var out []byte var patches []sbPatch if fi.needSplit { - guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+2)+len(fi.prologue)+bodyLen)) + // The JBE ends the guard, so its displacement is the prologue plus + // the body; the underflow JB additionally spans the trailing CMPQ and + // JBE, whose combined length is guardLen minus the prefix and the + // JB's own length (2 short, 6 long). + jbLen := 2 + if guardJBlong { + jbLen = 6 + } + guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen)) out = append(out, guard...) patches = append(patches, tlsPatch) } @@ -344,7 +351,10 @@ func computeFrame(t *ast.Text) frameInfo { // pass one extra slot, and the virtual SP is the hardware SP. fi.size = 8 fi.useFP = true - fi.fpAdjust = int64(fi.size) + 16 // return address + saved BP + args base + // The push is the frame: the saved BP sits at SP+0 and the + // return address at SP+8, so arguments begin at SP+16. Unlike + // a SUBQ frame, the 8-byte size must not be added again. + fi.fpAdjust = 16 fi.spAdjust = 0 fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP fi.epilogue = []byte{0x5D} // POPQ BP @@ -426,16 +436,6 @@ func (fi frameInfo) guardLen(jbLong, jbeLong bool) int { } } -// moreLen returns the byte length of the trailing morestack block: the CALL -// (always rel32) plus the JMP back to the function start. -func moreLen(jmpLong bool) int { - jmp := 2 - if jmpLong { - jmp = 5 - } - return 5 + jmp -} - // buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the // already-computed displacements of the conditional branches that jump to the // morestack block (unused in classes without them). The TLS load carries a diff --git a/asm/assemble_test.go b/asm/assemble_test.go index 18a44ae..acec9ba 100644 --- a/asm/assemble_test.go +++ b/asm/assemble_test.go @@ -160,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24 } } +// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame +// function containing a CALL receives: the PUSHQ BP prologue with no stack +// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the +// Go assembler produces. The push is the frame, so the offset must not count +// it twice. +func TestAssembleFramelessCall(t *testing.T) { + f, errs := parser.Parse("frameless_call_amd64.s", ` +#include "textflag.h" +TEXT ·withcall(SB), NOSPLIT, $0-16 + MOVQ x+0(FP), AX + CALL ·other(SB) + MOVQ AX, ret+8(FP) + RET +TEXT ·other(SB), NOSPLIT, $0-0 + RET +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFile(f) + if err != nil { + t.Fatalf("AssembleFile: %v", err) + } + code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...) + for _, r := range img.Funcs[0].Relocs { + for j := r.Off; j < r.Off+4 && j < len(code); j++ { + code[j] = 0 + } + } + // From `go tool objdump` of the Go-assembled function: + // PUSHQ BP 55 + // MOVQ SP, BP 4889e5 + // MOVQ 0x10(SP), AX 488b442410 + // CALL other e800000000 + // MOVQ AX, 0x18(SP) 4889442418 + // POPQ BP 5d + // RET c3 + want := []byte{ + 0x55, + 0x48, 0x89, 0xe5, + 0x48, 0x8b, 0x44, 0x24, 0x10, + 0xe8, 0x00, 0x00, 0x00, 0x00, + 0x48, 0x89, 0x44, 0x24, 0x18, + 0x5d, + 0xc3, + } + if hexBytes(code) != hexBytes(want) { + t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) + } +} + // TestAssembleFrame verifies a function with a non-zero frame: the Go-style // prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against // the bytes the Go assembler produces. @@ -353,6 +404,19 @@ TEXT ·pf(SB), NOSPLIT, $0 } } +// TestAssembleBareJump checks that a zero-operand jump (which parses, because +// the parser does not arity-check mnemonics) is rejected with an error rather +// than panicking in the layout loop, which indexes Operands[0] before the +// emission pass gets a chance to diagnose the arity. +func TestAssembleBareJump(t *testing.T) { + for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} { + fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n") + if _, _, err := Assemble(fn); err == nil { + t.Errorf("%s with no operand: expected an error, got none", mnem) + } + } +} + // TestSubSPEncodings pins the prologue SUB against the bytes go tool asm // emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything // larger. The intermediate 129..255 range used to encode an ADD with a diff --git a/asm/encodable.go b/asm/encodable.go index 0419b14..903f5df 100644 --- a/asm/encodable.go +++ b/asm/encodable.go @@ -36,10 +36,15 @@ func Encodable(mnemonic string) bool { } // CMOV carries size then condition (CMOVLGT); SET carries the condition - // alone (SETNE). + // alone (SETNE). The size letter is checked exactly as encodeCmov does, + // so a spelling like CMOVBGT is not reported encodable when Encode + // would reject it. if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 { - if _, ok := jccMap[rest[1:]]; ok { - return true + switch rest[0] { + case 'W', 'L', 'Q': + if _, ok := jccMap[rest[1:]]; ok { + return true + } } } if rest, ok := strings.CutPrefix(upper, "SET"); ok { diff --git a/asm/encode.go b/asm/encode.go index 0d85317..2a6e487 100644 --- a/asm/encode.go +++ b/asm/encode.go @@ -117,9 +117,9 @@ func (e *enc) encode(mnem string, ops []Operand) error { case "IMUL", "IMUL3": return e.encodeImul(ops, size) case "PUSH": - return e.encodePushPop(ops, true) + return e.encodePushPop(ops, size, true) case "POP": - return e.encodePushPop(ops, false) + return e.encodePushPop(ops, size, false) case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT": return e.encodeCount(base, ops, size) case "BSWAP": @@ -345,6 +345,12 @@ func setMem(i *instr, regField int, m Mem) error { // a memory operand. It is shared by the REX (scalar) and VEX (vector) paths. func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) { sib = -1 + // A displacement wider than int32 fits no encoding form; truncating it + // would address a different location, and go tool asm reports "offset + // too large" for the same operand. + if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 { + return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp) + } // RIP-relative: neither base nor index. if !m.HasBase && !m.HasIndex { return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101 diff --git a/asm/encode_test.go b/asm/encode_test.go index 09e7912..214c978 100644 --- a/asm/encode_test.go +++ b/asm/encode_test.go @@ -149,6 +149,45 @@ func TestPushPop(t *testing.T) { checkSyntax(t, "push rbx", "PUSHQ", BX) checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8}) checkSyntax(t, "push 0x5", "PUSHQ", Imm(5)) + // The W spelling carries the 0x66 operand-size prefix, byte for byte + // with go tool asm; the L and B spellings are illegal in 64-bit mode + // there and rejected here rather than silently widened. + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + {"PUSHW AX", "PUSHW", []Operand{AX}, "6650"}, + {"POPW AX", "POPW", []Operand{AX}, "6658"}, + {"PUSHW $5", "PUSHW", []Operand{Imm(5)}, "666a05"}, + {"PUSHW (AX)", "PUSHW", []Operand{Ptr(AX, 0, 2)}, "66ff30"}, + {"PUSHQ AX", "PUSHQ", []Operand{AX}, "50"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: %v", c.name, err) + continue + } + if got := fmt.Sprintf("%x", code); got != c.want { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + } + } + for _, c := range []struct { + name string + mnem string + ops []Operand + }{ + {"PUSHL AX", "PUSHL", []Operand{AX}}, + {"PUSHL R8", "PUSHL", []Operand{Reg{idx: 8, size: 8}}}, + {"POPL BX", "POPL", []Operand{BX}}, + {"PUSHB AX", "PUSHB", []Operand{AX}}, + } { + if _, err := Encode(c.mnem, c.ops...); err == nil { + t.Errorf("%s: expected an error, got none", c.name) + } + } } func TestUnary(t *testing.T) { @@ -397,6 +436,104 @@ func TestScalarErrors(t *testing.T) { } } +// TestImmediateOutOfRange pins the go-tool-asm parity of the immediate and +// displacement spans: a scalar immediate must fit a signed or unsigned 32-bit +// word (only MOVQ reg, $imm takes the full int64), a scalar shift count must +// be an unsigned byte, and a displacement must fit int32. Every rejected +// shape here is rejected by `go tool asm` too; every accepted one encodes the +// same bytes. +func TestImmediateOutOfRange(t *testing.T) { + cases := []struct { + name string + mnem string + ops []Operand + }{ + {"SHLQ count 300", "SHLQ", []Operand{Imm(300), AX}}, + {"SHLQ count -1", "SHLQ", []Operand{Imm(-1), AX}}, + {"SHLW count 256", "SHLW", []Operand{Imm(256), DX}}, + {"SHLB count 300", "SHLB", []Operand{Imm(300), BL}}, + {"MOVL imm32+", "MOVL", []Operand{Imm(4294967296), AX}}, + {"MOVL imm32-", "MOVL", []Operand{Imm(-2147483649), AX}}, + {"MOVW imm32+", "MOVW", []Operand{Imm(4294967296), AX}}, + {"MOVB imm32+", "MOVB", []Operand{Imm(4294967296), AL}}, + {"ADDB imm32+", "ADDB", []Operand{Imm(4294967296), AL}}, + {"ADDL imm32+", "ADDL", []Operand{Imm(4294967296), AX}}, + {"ADDQ imm32+", "ADDQ", []Operand{Imm(8589934592), AX}}, + {"CMPQ imm32+", "CMPQ", []Operand{AX, Imm(4294967296)}}, + {"CMPQ imm32-", "CMPQ", []Operand{AX, Imm(-2147483649)}}, + {"TESTL imm32+", "TESTL", []Operand{Imm(4294967296), AX}}, + {"IMUL3L imm32+", "IMUL3L", []Operand{Imm(4294967296), CX, DX}}, + {"PUSHQ imm32+", "PUSHQ", []Operand{Imm(4294967296)}}, + {"MOVQ mem imm32+", "MOVQ", []Operand{Imm(4294967296), Ptr(AX, 0, 8)}}, + {"disp32+", "MOVQ", []Operand{Ptr(AX, 4294967296, 8), BX}}, + {"disp32+ max", "MOVQ", []Operand{Ptr(AX, 2147483648, 8), BX}}, + {"disp32-", "MOVQ", []Operand{Ptr(AX, -2147483649, 8), BX}}, + {"VEX disp32+", "VMOVDQU", []Operand{Ptr(AX, 4294967296, 32), vreg(t, "Y1")}}, + {"EVEX disp32+", "VMOVDQU32", []Operand{Ptr(AX, 4294967296, 64), vreg(t, "Z1")}}, + } + for _, c := range cases { + if _, err := Encode(c.mnem, c.ops...); err == nil { + t.Errorf("%s: expected an error, got none", c.name) + } + } +} + +// TestImmediateTruncation pins the toolchain-matching truncations inside the +// accepted 32-bit span: the narrower fields take the low bits silently, byte +// for byte with `go tool asm` (which rejects none of these). +func TestImmediateTruncation(t *testing.T) { + cases := []struct { + name string + mnem string + ops []Operand + want string + }{ + {"ADDB $256,BL", "ADDB", []Operand{Imm(256), BL}, "80c300"}, + {"ADDB $1000,BL", "ADDB", []Operand{Imm(1000), BL}, "80c3e8"}, + {"MOVB $256,AL", "MOVB", []Operand{Imm(256), AL}, "b000"}, + {"MOVB $-129,AL", "MOVB", []Operand{Imm(-129), AL}, "b07f"}, + {"MOVW $65536,AX", "MOVW", []Operand{Imm(65536), AX}, "66b80000"}, + {"MOVW $65535,AX", "MOVW", []Operand{Imm(65535), AX}, "66b8ffff"}, + {"MOVW $-32769,AX", "MOVW", []Operand{Imm(-32769), AX}, "66b8ff7f"}, + {"MOVL $4294967295,AX", "MOVL", []Operand{Imm(4294967295), AX}, "b8ffffffff"}, + {"ADDQ $4294967295,AX", "ADDQ", []Operand{Imm(4294967295), AX}, "4805ffffffff"}, + {"CMPB BL,$255", "CMPB", []Operand{BL, Imm(255)}, "80fbff"}, + {"CMPQ AX,$4294967295", "CMPQ", []Operand{AX, Imm(4294967295)}, "483dffffffff"}, + {"MOVQ $4294967295,0(AX)", "MOVQ", []Operand{Imm(4294967295), Ptr(AX, 0, 8)}, "48c700ffffffff"}, + {"SHLQ $255,AX", "SHLQ", []Operand{Imm(255), AX}, "48c1e0ff"}, + {"SHLQ $0,AX", "SHLQ", []Operand{Imm(0), AX}, "48c1e000"}, + // The one form beyond the 32-bit span: the imm64 MOVQ register move. + {"MOVQ $4294967296,AX", "MOVQ", []Operand{Imm(4294967296), AX}, "48b80000000001000000"}, + {"MOVQ disp32 max", "MOVQ", []Operand{Ptr(AX, 2147483647, 8), BX}, "488b98ffffff7f"}, + } + for _, c := range cases { + code, err := Encode(c.mnem, c.ops...) + if err != nil { + t.Errorf("%s: %v", c.name, err) + continue + } + if got := fmt.Sprintf("%x", code); got != c.want { + t.Errorf("%s: bytes %s, want %s", c.name, got, c.want) + } + } +} + +// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must +// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size +// letter) is not reported as encodable. +func TestEncodableCmovSize(t *testing.T) { + for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} { + if Encodable(m) { + t.Errorf("Encodable(%q) = true, want false", m) + } + } + for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} { + if !Encodable(m) { + t.Errorf("Encodable(%q) = false, want true", m) + } + } +} + // TestSSEBinGroundTruth checks the legacy packed/scalar binary family // byte for byte (no prefix / 66 / F2 / F3 variants). func TestSSEBinGroundTruth(t *testing.T) { diff --git a/asm/evex.go b/asm/evex.go index 30cc2c3..27f7208 100644 --- a/asm/evex.go +++ b/asm/evex.go @@ -509,40 +509,43 @@ var evexBcastTable = map[string]evexBcastSpec{ } // evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX -// move table). +// move table). vecOK and xmmOnly mirror the VEX twin's operand rules: a +// scalar move (vecOK false, xmmOnly true) takes XMM↔memory operands only. type evexMoveSpec struct { - mapSel int - pp int - load byte // r/m → vector - store byte // vector → r/m - w int - n [3]int + mapSel int + pp int + load byte // r/m → vector + store byte // vector → r/m + w int + n [3]int + vecOK bool // the non-memory operand may be a vector register + xmmOnly bool // wider than XMM registers are rejected } // evexMoveTable maps an upper-case EVEX move mnemonic to its encoding. var evexMoveTable = map[string]evexMoveSpec{ // EVEX.128/256/512.F3.0F.W0, unaligned integer move. - "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, + "VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, // EVEX.128/256/512.F3.0F.W1, unaligned qword move. - "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, + "VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, // EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the // F2 prefix, dword/qword moves F3; the element size only changes the tuple // semantics). - "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, + "VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, // EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword // encoding). - "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, + "VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, // EVEX.128/256/512.66.0F.W1, unaligned packed double move. - "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}}, + "VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false}, // EVEX.128/256/512, aligned packed moves. - "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}}, - "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}}, + "VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false}, + "VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false}, // EVEX.128/256/512.66.0F, aligned integer moves. - "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}}, - "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}}, + "VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false}, + "VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false}, // EVEX.128.F3.0F.W0, scalar single move, memory operands (the // three-operand register form is not supported). - "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}}, + "VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true}, } // isEvex reports whether the mnemonic has an EVEX encoding we handle. @@ -1022,6 +1025,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i var rm Operand switch { case srcIsVec && dstIsVec: + // A store-form reg-reg move, the layout the Go assembler uses; a + // scalar move has no two-register form at all (the register form + // takes three operands), matching the VEX twin's vecOK rule. + if !ms.vecOK { + return fmt.Errorf("%s does not take two vector registers", mnem) + } reg, rm = srcReg, dst case srcIsVec: if !memOperand(dst) { @@ -1037,6 +1046,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i default: return fmt.Errorf("%s needs a vector register operand", mnem) } + // The scalar move is 128-bit only, so the register the length follows + // must be an XMM (the VEX twin's xmmOnly rule; EVEX also reaches ZMM, + // hence the inequality rather than a YMM test). + if ms.xmmOnly && reg.size != 16 { + return fmt.Errorf("%s operates on XMM registers only", mnem) + } spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n} return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx) } @@ -1171,9 +1186,6 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, if r.idx&16 != 0 { xBar = 0 } - if r.idx&16 != 0 { - xBar = 0 - } case Mem: var err error modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll]) @@ -1232,6 +1244,11 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) { sib = -1 xBar, bBar = 1, 1 // inverted bits: 1 = no extension + // The disp32 fallback bounds the displacement by int32, and the + // compressed disp8 form reaches at most ±127×64, well inside it. + if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 { + return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp) + } if !m.HasBase && !m.HasIndex { return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative } diff --git a/asm/evex_test.go b/asm/evex_test.go index ba8e4bb..605372a 100644 --- a/asm/evex_test.go +++ b/asm/evex_test.go @@ -675,6 +675,15 @@ func TestEvexErrors(t *testing.T) { {"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}}, // VEX-only mnemonics reject registers only EVEX can encode. {"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}}, + // The scalar EVEX move matches its VEX twin and the Go assembler: + // XMM↔memory only, never reg-reg and never a wider register (the + // toolchain rejects every one of these shapes). + {"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}}, + {"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}}, + {"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}}, + {"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}}, + {"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}}, + {"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}}, } for _, c := range cases { if _, err := Encode(c.mnem, c.ops...); err == nil { diff --git a/asm/instrs.go b/asm/instrs.go index a4878d0..5072724 100644 --- a/asm/instrs.go +++ b/asm/instrs.go @@ -173,7 +173,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error { if dstReg.needsREX(size) { i.rexForced = true } - i.imm = immediate(v, size, true) + imm, err := immediate(v, size, true) + if err != nil { + return err + } + i.imm = imm return e.emit(i) } // MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. @@ -185,7 +189,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error { if err := setRMDigit(i, 0, dst, size); err != nil { return err } - i.imm = immediate(int64(src), size, false) + imm, err := immediate(int64(src), size, false) + if err != nil { + return err + } + i.imm = imm return e.emit(i) } return fmt.Errorf("MOV: invalid operands") @@ -297,11 +305,15 @@ func (e *enc) encodeALU(op struct { func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error { if size == 1 { + immBytes, err := immediate(imm, 1, false) + if err != nil { + return err + } i := newInstr(1, []byte{0x80}) if err := setRMDigit(i, digit, dst, 1); err != nil { return err } - i.imm = []byte{byte(int8(imm))} + i.imm = immBytes return e.emit(i) } if fits8(imm) { @@ -319,7 +331,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error { if r, ok := dst.(Reg); ok && r.idx == 0 { accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit] i := newInstr(size, []byte{accOp}) - i.imm = immediate(imm, size, false) + immBytes, err := immediate(imm, size, false) + if err != nil { + return err + } + i.imm = immBytes return e.emit(i) } // 0x81 /digit, imm16/imm32. @@ -327,7 +343,11 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error { if err := setRMDigit(i, digit, dst, size); err != nil { return err } - i.imm = immediate(imm, size, false) + immBytes, err := immediate(imm, size, false) + if err != nil { + return err + } + i.imm = immBytes return e.emit(i) } @@ -348,7 +368,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error { op = 0xA8 } i := newInstr(size, []byte{op}) - i.imm = immediate(int64(imm), size, false) + immBytes, err := immediate(int64(imm), size, false) + if err != nil { + return err + } + i.imm = immBytes return e.emit(i) } op := byte(0xF7) @@ -359,7 +383,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error { if err := setRMDigit(i, 0, dst, size); err != nil { return err } - i.imm = immediate(int64(imm), size, false) + immBytes, err := immediate(int64(imm), size, false) + if err != nil { + return err + } + i.imm = immBytes return e.emit(i) } srcReg, ok := src.(Reg) @@ -457,7 +485,13 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error { } return e.emit(i) } - // 0xC0 (8-bit) / 0xC1, imm8. + // 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm + // rejects negative and ≥256 counts, and the hardware masks the count, so + // a silent truncation ($300 encoding 44) would shift by a different + // amount than the source states. + if imm < 0 || imm > 255 { + return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm)) + } op := byte(0xC1) if size == 1 { op = 0xC0 @@ -466,7 +500,7 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error { if err := setRMDigit(i, digit, dst, size); err != nil { return err } - i.imm = []byte{byte(int8(imm))} + i.imm = []byte{byte(imm)} return e.emit(i) } @@ -508,7 +542,11 @@ func (e *enc) encodeImul(ops []Operand, size int) error { if err := setRM(i, dstReg, ops[1], size); err != nil { return err } - i.imm = immediate(int64(imm), size, false) + immBytes, err := immediate(int64(imm), size, false) + if err != nil { + return err + } + i.imm = immBytes return e.emit(i) } return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops)) @@ -516,10 +554,21 @@ func (e *enc) encodeImul(ops []Operand, size int) error { // --- PUSH / POP ------------------------------------------------------------- -func (e *enc) encodePushPop(ops []Operand, push bool) error { +func (e *enc) encodePushPop(ops []Operand, size int, push bool) error { if len(ops) != 1 { return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops)) } + // In 64-bit mode go tool asm knows the 64-bit push (the default, with or + // without the Q suffix) and the 16-bit W form with its 0x66 operand-size + // prefix, and rejects the B and L spellings outright ("illegal in 64-bit + // mode"); silently widening those would push a different width than the + // source states. + switch size { + case 0, 8, 2: + default: + return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode") + } + w16 := size == 2 switch op := ops[0].(type) { case Reg: base := byte(0x50) // PUSH r; POP is 0x58 @@ -527,7 +576,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error { base = 0x58 } // PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed. - i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1} + i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1} i.rexB = op.idx >= 8 return e.emit(i) case Mem: @@ -537,7 +586,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error { opc = 0x8F // POP r/m: /0 digit = 0 } - i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1} + i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1} if err := setRMDigit(i, digit, ops[0], 8); err != nil { return err } @@ -547,10 +596,17 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error { return fmt.Errorf("POP does not take an immediate") } if fits8(int64(op)) { - i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}} + i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}} return e.emit(i) } - i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))} + // PUSH imm32, sign-extended to 64 bits; go tool asm bounds the + // immediate by the same signed/unsigned 32-bit span as every other + // scalar immediate. + immBytes, err := immediate(int64(op), 8, false) + if err != nil { + return err + } + i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes} return e.emit(i) } return fmt.Errorf("PUSH/POP: invalid operand") @@ -639,19 +695,28 @@ func (e *enc) encodeJcc(cc int, ops []Operand) error { // immediate encodes an immediate of the given operand size. full64 selects the // 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit // sign-extended immediate is used for 64-bit operands. -func immediate(v int64, size int, full64 bool) []byte { +// +// The span mirrors go tool asm: every scalar immediate must fit a signed or +// unsigned 32-bit word, and the narrower fields then take the low bits +// silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the +// imm64 form may exceed the span; anything wider elsewhere is an error rather +// than a truncation the source never asked for. +func immediate(v int64, size int, full64 bool) ([]byte, error) { + if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) { + return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v) + } switch size { case 1: - return []byte{byte(int8(v))} + return []byte{byte(int8(v))}, nil case 2: - return le16(v) + return le16(v), nil case 4: - return le32(v) + return le32(v), nil default: // 8 if full64 { - return le64(v) + return le64(v), nil } - return le32(v) // sign-extended imm32 + return le32(v), nil // sign-extended imm32 } }