851 lines
27 KiB
Go
851 lines
27 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package asm
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
|
)
|
|
|
|
// Assemble encodes the body of a TEXT function into x86-64 machine code,
|
|
// resolving local labels to relative jump offsets and translating the FP/SP
|
|
// pseudo-registers onto the hardware stack pointer (matching the Go
|
|
// assembler's default frame-pointer behaviour). Jumps start in the short
|
|
// (rel8) form and expand to rel32 when the settled displacement does not fit;
|
|
// sizes only grow, so the layout reaches a fixed point in a few passes. CALL
|
|
// has no short form and is always rel32.
|
|
//
|
|
// Supported operands: registers, memory (real base register), immediates,
|
|
// FP/SP frame-relative operands, and local-label jumps. SB (global symbol)
|
|
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
|
|
// integer and shuffle/extract/permute/move set is in.
|
|
//
|
|
// Like the other architectures, the stack-growth guard (the morestack check
|
|
// in the prologue and the call back into the runtime in the epilogue) is not
|
|
// emitted: the bytes match go tool asm only for NOSPLIT functions or
|
|
// zero-frame leaves, where the toolchain emits no guard either.
|
|
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
|
code, _, labels, _, _, err := assemble(t, nil)
|
|
return code, labels, err
|
|
}
|
|
|
|
// linkInfo carries file-level symbol context into a single-function assembly:
|
|
// the set of static symbols a GLOBL in the same file defines. A nil link
|
|
// rejects SB operands outright (single-function assembly cannot resolve
|
|
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
|
|
// file defines is recorded as an external relocation instead of failing
|
|
// the object-file emitters resolve it at link time.
|
|
type linkInfo struct {
|
|
symbols map[string]bool
|
|
allowExternal bool
|
|
}
|
|
|
|
// sbPatch is a function-relative static-symbol relocation: the disp32 field
|
|
// at off must become the symbol's address minus after, where after is the
|
|
// function-relative address just past the instruction.
|
|
type sbPatch struct {
|
|
off int
|
|
after int
|
|
name string
|
|
addend int64
|
|
kind RelocKind
|
|
}
|
|
|
|
// spadjStep is one stack-adjustment boundary within a function: Value is the
|
|
// SP delta from the entry state (just below the return address) in effect
|
|
// from PC (function-relative) until the next step. The steps feed the
|
|
// pcsp table of the object-file emitters.
|
|
type spadjStep struct {
|
|
pc int
|
|
value int
|
|
}
|
|
|
|
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
|
// patch sites (for the file-level layout to resolve), the label table and the
|
|
// stack-adjustment boundaries.
|
|
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
|
fi := computeFrame(t)
|
|
chain := jumpChain(t)
|
|
resolve := func(name string) string {
|
|
if r, ok := chain[name]; ok {
|
|
return r
|
|
}
|
|
return name
|
|
}
|
|
|
|
// Layout: iterate jump sizes to a fixed point. The stack-split guard
|
|
// prefix and the trailing morestack block participate in the iteration:
|
|
// their conditional branches relax from rel8 to rel32 when the body
|
|
// outgrows the short form.
|
|
long := make([]bool, len(t.Body))
|
|
sizes := make([]int, len(t.Body))
|
|
offsets := map[string]int{}
|
|
pcs := make([]int, len(t.Body))
|
|
var guardJBlong, guardJBElong, moreJMPlong bool
|
|
for {
|
|
guard := fi.guardLen(guardJBlong, guardJBElong)
|
|
pos := guard + len(fi.prologue)
|
|
for i, stmt := range t.Body {
|
|
switch s := stmt.(type) {
|
|
case *ast.Label:
|
|
offsets[s.Name.Text] = pos
|
|
case *ast.Instr:
|
|
sz, err := instrSize(s, fi, long[i], link)
|
|
if err != nil {
|
|
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
|
}
|
|
sizes[i] = sz
|
|
pcs[i] = pos
|
|
pos += sz
|
|
}
|
|
}
|
|
bodyLen := pos - (guard + len(fi.prologue))
|
|
// Expand any short jump whose displacement no longer fits rel8.
|
|
changed := false
|
|
for i, stmt := range t.Body {
|
|
s, ok := stmt.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
mnem := strings.ToUpper(s.Mnemonic.Text)
|
|
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
|
|
continue
|
|
}
|
|
// A zero-operand jump parses; its arity is reported during
|
|
// emission (encodeJump), so the layout must not index Operands.
|
|
if len(s.Operands) != 1 {
|
|
continue
|
|
}
|
|
name, ok := labelName(s.Operands[0])
|
|
if !ok {
|
|
continue // reported during emission
|
|
}
|
|
target, ok := offsets[resolve(name)]
|
|
if !ok {
|
|
continue // reported during emission
|
|
}
|
|
rel := int64(target - (pcs[i] + jumpSize(mnem, false)))
|
|
if !fits8(rel) {
|
|
long[i] = true
|
|
changed = true
|
|
}
|
|
}
|
|
// The guard's conditional branches target the morestack block, which
|
|
// starts right after the body: the JBE measures from the end of the
|
|
// guard, so its displacement is the prologue plus the body.
|
|
if !guardJBElong && !fits8(int64(len(fi.prologue)+bodyLen)) {
|
|
guardJBElong = true
|
|
changed = true
|
|
}
|
|
if fi.splitClass == 2 && !guardJBlong {
|
|
// The underflow JB sits before the CMPQ; its displacement spans
|
|
// the rest of the guard plus the prologue and the body. The JB
|
|
// is still the short form this branch tests (relaxing it is this
|
|
// branch's job), so guardLen is taken with a short JB and the
|
|
// subtraction drops the prefix and the JB's own 2 bytes.
|
|
rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2)
|
|
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
|
|
guardJBlong = true
|
|
changed = true
|
|
}
|
|
}
|
|
// The morestack JMP returns to the function start, so its
|
|
// displacement is the negated distance from its own end; while it is
|
|
// still short, its own length is 2 bytes.
|
|
if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) {
|
|
moreJMPlong = true
|
|
changed = true
|
|
}
|
|
if !changed {
|
|
break
|
|
}
|
|
}
|
|
|
|
// Pass 2: emit. The guard comes first, then the prologue, the body and
|
|
// the morestack block.
|
|
guardLen := fi.guardLen(guardJBlong, guardJBElong)
|
|
bodyLen := 0
|
|
{
|
|
pos := guardLen + len(fi.prologue)
|
|
for i, stmt := range t.Body {
|
|
if _, ok := stmt.(*ast.Instr); ok {
|
|
pos += sizes[i]
|
|
}
|
|
}
|
|
bodyLen = pos - (guardLen + len(fi.prologue))
|
|
}
|
|
var out []byte
|
|
var patches []sbPatch
|
|
if fi.needSplit {
|
|
// The JBE ends the guard, so its displacement is the prologue plus
|
|
// the body; the underflow JB additionally spans the trailing CMPQ and
|
|
// JBE, whose combined length is guardLen minus the prefix and the
|
|
// JB's own length (2 short, 6 long).
|
|
jbLen := 2
|
|
if guardJBlong {
|
|
jbLen = 6
|
|
}
|
|
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen))
|
|
out = append(out, guard...)
|
|
patches = append(patches, tlsPatch)
|
|
}
|
|
out = append(out, fi.prologue...)
|
|
var steps []spadjStep
|
|
var lines []LineEntry
|
|
if fi.useFP {
|
|
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
|
|
// changes nothing; SUBQ $size, SP completes the frame.
|
|
steps = append(steps,
|
|
spadjStep{guardLen + 1, 8},
|
|
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
|
|
)
|
|
}
|
|
pos := guardLen + len(fi.prologue)
|
|
for i, stmt := range t.Body {
|
|
s, ok := stmt.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
if strings.ToUpper(s.Mnemonic.Text) == "RET" && fi.useFP {
|
|
// The RET's epilogue prefix unwinds: ADDQ $size, SP restores
|
|
// the saved-BP-only stack, POPQ BP the entry state.
|
|
epi := len(fi.epilogue)
|
|
steps = append(steps,
|
|
spadjStep{pos + epi - 1, 8},
|
|
spadjStep{pos + epi, 0},
|
|
)
|
|
}
|
|
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
|
if err != nil {
|
|
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
|
}
|
|
if len(code) != sizes[i] {
|
|
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
|
}
|
|
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
|
|
for k := range ps {
|
|
ps[k].kind = RelCall
|
|
}
|
|
}
|
|
patches = append(patches, ps...)
|
|
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
|
out = append(out, code...)
|
|
pos += len(code)
|
|
}
|
|
if fi.needSplit {
|
|
// The morestack block: CALL runtime.morestack_noctxt, then a JMP
|
|
// back to the function entry.
|
|
jmpLen := 2
|
|
if moreJMPlong {
|
|
jmpLen = 5
|
|
}
|
|
jmpDisp := -int64(pos + 5 + jmpLen)
|
|
suffix, callPatch := buildMoreStack(int32(jmpDisp))
|
|
callPatch.off += pos
|
|
callPatch.after = pos + 5
|
|
patches = append(patches, callPatch)
|
|
out = append(out, suffix...)
|
|
pos += len(suffix)
|
|
}
|
|
_ = pos
|
|
return out, patches, offsets, steps, lines, nil
|
|
}
|
|
|
|
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
|
// is an unconditional local jump redirects its own jumpers to the ultimate
|
|
// target. The Go toolchain chases exactly these chains (the linker's xfol
|
|
// pass) before it encodes branches, so matching its bytes requires the same
|
|
// redirection.
|
|
func jumpChain(t *ast.Text) map[string]string {
|
|
// label → the target of its leading unconditional local JMP, if any.
|
|
leadsTo := map[string]string{}
|
|
for i, stmt := range t.Body {
|
|
l, ok := stmt.(*ast.Label)
|
|
if !ok {
|
|
continue
|
|
}
|
|
// Stacked labels share an address: skip to the first instruction.
|
|
j := i + 1
|
|
for j < len(t.Body) {
|
|
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
|
|
break
|
|
}
|
|
j++
|
|
}
|
|
if j >= len(t.Body) {
|
|
continue
|
|
}
|
|
in, ok := t.Body[j].(*ast.Instr)
|
|
if !ok || strings.ToUpper(in.Mnemonic.Text) != "JMP" || len(in.Operands) != 1 {
|
|
continue
|
|
}
|
|
if name, ok := labelName(in.Operands[0]); ok {
|
|
leadsTo[l.Name.Text] = name
|
|
}
|
|
}
|
|
// Chase each chain to its end, guarding against cycles.
|
|
chain := map[string]string{}
|
|
for name := range leadsTo {
|
|
visited := map[string]bool{name: true}
|
|
cur := name
|
|
for {
|
|
next, ok := leadsTo[cur]
|
|
if !ok || visited[next] {
|
|
break
|
|
}
|
|
visited[next] = true
|
|
cur = next
|
|
}
|
|
if cur != name {
|
|
chain[name] = cur
|
|
}
|
|
}
|
|
return chain
|
|
}
|
|
|
|
// frameInfo carries the frame layout derived from the TEXT directive.
|
|
type frameInfo struct {
|
|
size int // local frame size ($framesize)
|
|
useFP bool // a frame pointer (BP) is set up
|
|
fpAdjust int64 // added to x+N(FP) to reach the hardware SP-relative offset
|
|
spAdjust int64 // x-N(SP) becomes (spAdjust - N)(SP)
|
|
prologue []byte
|
|
epilogue []byte
|
|
|
|
// Stack-split guard state (matching the toolchain's stacksplit): needSplit
|
|
// is false for NOSPLIT functions and for leaf functions whose frame is
|
|
// below StackSmall, which the toolchain auto-marks NOSPLIT.
|
|
needSplit bool
|
|
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
|
|
framesize int // the size the guard checks: frame+8 for framed functions
|
|
}
|
|
|
|
// Stack-frame size classes from runtime/stack.go.
|
|
const (
|
|
stackSmall = 128
|
|
stackBig = 4096
|
|
)
|
|
|
|
// sbPatch gains a kind so the emitters can tell CALL and TLS patches from
|
|
// plain PC-relative displacements.
|
|
|
|
// computeFrame derives the frame layout, matching the Go assembler's default
|
|
// (a frame pointer is used whenever the function has a non-zero frame). It
|
|
// also decides whether the function needs the stack-split guard, mirroring
|
|
// obj6: a NOSPLIT function never splits, and a leaf function whose frame is
|
|
// below StackSmall is auto-marked NOSPLIT. One deliberate deviation: the
|
|
// toolchain treats zero-argument runtime calls (duffcopy and friends) as
|
|
// leaf-compatible; here any CALL makes the function a non-leaf.
|
|
func computeFrame(t *ast.Text) frameInfo {
|
|
fi := frameInfo{}
|
|
if t.Frame != nil && t.Frame.Imm.HasVal {
|
|
fi.size = int(t.Frame.Imm.Val)
|
|
}
|
|
if fi.size == 0 && hasCall(t) {
|
|
// The toolchain gives a frameless function containing a CALL an
|
|
// 8-byte frame for the pushed base pointer: the prologue saves BP
|
|
// with no stack adjustment, every RET pops it back, FP references
|
|
// pass one extra slot, and the virtual SP is the hardware SP.
|
|
fi.size = 8
|
|
fi.useFP = true
|
|
// The push is the frame: the saved BP sits at SP+0 and the
|
|
// return address at SP+8, so arguments begin at SP+16. Unlike
|
|
// a SUBQ frame, the 8-byte size must not be added again.
|
|
fi.fpAdjust = 16
|
|
fi.spAdjust = 0
|
|
fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
|
|
fi.epilogue = []byte{0x5D} // POPQ BP
|
|
} else if fi.size > 0 {
|
|
fi.useFP = true
|
|
fi.fpAdjust = int64(fi.size) + 16 // frame + saved BP + return address
|
|
fi.spAdjust = int64(fi.size)
|
|
fi.prologue = prologueBytes(fi.size)
|
|
fi.epilogue = epilogueBytes(fi.size)
|
|
} else {
|
|
fi.fpAdjust = 8 // return address only
|
|
}
|
|
|
|
noSplit := false
|
|
for _, f := range t.Flags {
|
|
if strings.EqualFold(f, "NOSPLIT") {
|
|
noSplit = true
|
|
}
|
|
}
|
|
// The toolchain's autoffset: the frame plus the saved base pointer.
|
|
framesize := fi.size
|
|
if framesize > 0 {
|
|
framesize += 8
|
|
}
|
|
switch {
|
|
case noSplit:
|
|
case framesize < stackSmall && !hasCall(t):
|
|
// Auto-NOSPLIT, as the toolchain's leaf search concludes.
|
|
default:
|
|
fi.needSplit = true
|
|
fi.framesize = framesize
|
|
switch {
|
|
case framesize <= stackSmall:
|
|
fi.splitClass = 0
|
|
case framesize <= stackBig:
|
|
fi.splitClass = 1
|
|
default:
|
|
fi.splitClass = 2
|
|
}
|
|
}
|
|
return fi
|
|
}
|
|
|
|
// hasCall reports whether the function body contains a CALL instruction.
|
|
func hasCall(t *ast.Text) bool {
|
|
for _, stmt := range t.Body {
|
|
in, ok := stmt.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
if strings.ToUpper(in.Mnemonic.Text) == "CALL" {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// guardLen returns the byte length of the stack-split guard prefix. The
|
|
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
|
|
// short form and 6 in the long form.
|
|
func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
|
|
if !fi.needSplit {
|
|
return 0
|
|
}
|
|
jb, jbe := 2, 2
|
|
if jbLong {
|
|
jb = 6
|
|
}
|
|
if jbeLong {
|
|
jbe = 6
|
|
}
|
|
switch fi.splitClass {
|
|
case 0:
|
|
return 9 + 4 + jbe
|
|
case 1:
|
|
return 9 + 8 + 4 + jbe
|
|
default:
|
|
return 9 + 3 + 7 + jb + 4 + jbe
|
|
}
|
|
}
|
|
|
|
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
|
|
// already-computed displacements of the conditional branches that jump to the
|
|
// morestack block (unused in classes without them). The TLS load carries a
|
|
// R_TLS_LE patch site at offset 5.
|
|
func buildGuard(fi frameInfo, jbeDisp, jbDisp int32) ([]byte, sbPatch) {
|
|
out := []byte{
|
|
0x64, 0x4c, 0x8b, 0x34, 0x25, // MOVQ FS:0, R14
|
|
0, 0, 0, 0, // TLS slot offset, filled by the linker
|
|
}
|
|
tls := sbPatch{off: 5, after: 9, kind: RelTLSLE}
|
|
jmp := func(op8, op32 byte, disp int32) []byte {
|
|
if disp >= -128 && disp <= 127 {
|
|
return []byte{op8, byte(disp)}
|
|
}
|
|
return append([]byte{0x0F, op32}, le32(int64(disp))...)
|
|
}
|
|
switch fi.splitClass {
|
|
case 0:
|
|
// CMPQ SP, 16(R14)
|
|
out = append(out, 0x49, 0x3b, 0x66, 0x10)
|
|
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
|
|
case 1:
|
|
// LEAQ -(framesize-StackSmall)(SP), R12; CMPQ R12, 16(R14)
|
|
out = append(out, 0x4c, 0x8d, 0xa4, 0x24)
|
|
out = append(out, le32(-int64(fi.framesize-stackSmall))...)
|
|
out = append(out, 0x4d, 0x3b, 0x66, 0x10)
|
|
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
|
|
default:
|
|
// MOVQ SP, R12; SUBQ $(framesize-StackSmall), R12; JB; CMPQ R12, 16(R14)
|
|
out = append(out, 0x49, 0x89, 0xe4)
|
|
out = append(out, 0x49, 0x81, 0xec)
|
|
out = append(out, le32(int64(fi.framesize-stackSmall))...)
|
|
out = append(out, jmp(0x72, 0x82, jbDisp)...)
|
|
out = append(out, 0x4d, 0x3b, 0x66, 0x10)
|
|
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
|
|
}
|
|
return out, tls
|
|
}
|
|
|
|
// buildMoreStack emits the trailing block: CALL runtime.morestack_noctxt
|
|
// (patched by the linker) and a JMP back to the function start.
|
|
func buildMoreStack(jmpDisp int32) ([]byte, sbPatch) {
|
|
out := []byte{0xE8, 0, 0, 0, 0}
|
|
call := sbPatch{off: 1, after: 5, name: "runtime\u00b7morestack_noctxt", kind: RelCall}
|
|
out = append(out, jmpBytes(jmpDisp)...)
|
|
return out, call
|
|
}
|
|
|
|
// jmpBytes encodes a near JMP in the short or long form.
|
|
func jmpBytes(disp int32) []byte {
|
|
if disp >= -128 && disp <= 127 {
|
|
return []byte{0xEB, byte(disp)}
|
|
}
|
|
return append([]byte{0xE9}, le32(int64(disp))...)
|
|
}
|
|
|
|
// prologueBytes emits: PUSHQ BP; MOVQ SP, BP; SUBQ $size, SP.
|
|
func prologueBytes(size int) []byte {
|
|
out := []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
|
|
return append(out, subSP(size)...)
|
|
}
|
|
|
|
// epilogueBytes emits: ADDQ $size, SP; POPQ BP.
|
|
func epilogueBytes(size int) []byte {
|
|
out := addSP(size)
|
|
return append(out, 0x5D) // POPQ BP
|
|
}
|
|
|
|
func subSP(size int) []byte { // SUBQ $size, SP
|
|
// imm8 holds -128..127; anything larger takes the imm32 form, exactly as
|
|
// the Go assembler encodes it (verified for 8, 128, 200 and 255).
|
|
if size >= -128 && size <= 127 {
|
|
return []byte{0x48, 0x83, 0xEC, byte(int8(size))}
|
|
}
|
|
return append([]byte{0x48, 0x81, 0xEC}, le32(int64(size))...)
|
|
}
|
|
|
|
func addSP(size int) []byte { // ADDQ $size, SP
|
|
if size >= -128 && size <= 127 {
|
|
return []byte{0x48, 0x83, 0xC4, byte(int8(size))}
|
|
}
|
|
return append([]byte{0x48, 0x81, 0xC4}, le32(int64(size))...)
|
|
}
|
|
|
|
// instrSize returns the encoded length of an instruction (layout pass).
|
|
// encodeInstr already includes the epilogue for a RET in a frame-pointer
|
|
// function; jumps use their short or long form (never an epilogue).
|
|
func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) {
|
|
mnem := strings.ToUpper(s.Mnemonic.Text)
|
|
if isJumpMnemonic(mnem) {
|
|
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
|
return 5, nil // opcode + rel32, always the long form
|
|
}
|
|
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
|
|
code, err := encodeIndirectJump(s, mnem)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
return len(code), nil
|
|
}
|
|
return jumpSize(mnem, long), nil
|
|
}
|
|
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
return len(code), nil
|
|
}
|
|
|
|
func isJumpMnemonic(mnem string) bool {
|
|
if mnem == "JMP" || mnem == "CALL" {
|
|
return true
|
|
}
|
|
_, ok := condCode(mnem)
|
|
return ok
|
|
}
|
|
|
|
// jumpSize returns the length of a jump instruction in the requested form:
|
|
// short (rel8) where available, otherwise the rel32 form. CALL is always
|
|
// rel32.
|
|
func jumpSize(mnem string, long bool) int {
|
|
if mnem == "CALL" {
|
|
return 5 // opcode + rel32
|
|
}
|
|
if !long {
|
|
return 2 // opcode + rel8
|
|
}
|
|
if mnem == "JMP" {
|
|
return 5 // E9 + rel32
|
|
}
|
|
return 6 // 0x0F 0x8x + rel32
|
|
}
|
|
|
|
// encodeInstr encodes one instruction, resolving jump targets against offsets
|
|
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
|
|
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
|
|
// jump label through the jump-to-jump chain before the offset lookup.
|
|
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) {
|
|
mnem := strings.ToUpper(s.Mnemonic.Text)
|
|
|
|
var prefix []byte
|
|
if mnem == "RET" && fi.useFP {
|
|
prefix = fi.epilogue
|
|
}
|
|
|
|
var code []byte
|
|
var ps []sbPatch
|
|
var err error
|
|
if isJumpMnemonic(mnem) {
|
|
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
|
// CALL/JMP sym(SB): a rel32 call (or tail call) against a
|
|
// static or external symbol, resolved by the file-level layout
|
|
// or the linker.
|
|
code, ps, err = encodeSBCall(s, link)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
for i := range ps {
|
|
ps[i].kind = RelCall
|
|
}
|
|
body := pc + len(prefix)
|
|
for i := range ps {
|
|
ps[i].off += body
|
|
ps[i].after = body + len(code)
|
|
}
|
|
return append(prefix, code...), ps, nil
|
|
}
|
|
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
|
|
// JMP/CALL through a register or memory: no relocation and no
|
|
// label to resolve, the operand fully determines the bytes.
|
|
code, err = encodeIndirectJump(s, mnem)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
return append(prefix, code...), nil, nil
|
|
}
|
|
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
|
|
} else {
|
|
code, ps, err = encodeNormal(s, fi, link)
|
|
}
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
// Anchor the patch fields at function-relative positions: off indexes the
|
|
// disp32 field, after is the address just past the instruction.
|
|
body := pc + len(prefix)
|
|
for i := range ps {
|
|
ps[i].off += body
|
|
ps[i].after = body + len(code)
|
|
}
|
|
return append(prefix, code...), ps, nil
|
|
}
|
|
|
|
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
|
|
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
|
|
if size == 0 {
|
|
size = 8
|
|
}
|
|
ops := make([]Operand, len(s.Operands))
|
|
for i, op := range s.Operands {
|
|
o, err := operandFromAST(op, size, fi, link)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
ops[i] = o
|
|
}
|
|
e := &enc{}
|
|
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
|
|
return nil, nil, err
|
|
}
|
|
ps := make([]sbPatch, len(e.patches))
|
|
for i, p := range e.patches {
|
|
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
|
|
}
|
|
return e.out, ps, nil
|
|
}
|
|
|
|
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
|
|
// target label, in the short (rel8) or long (rel32) form.
|
|
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string) ([]byte, error) {
|
|
if len(s.Operands) != 1 {
|
|
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
|
|
}
|
|
name, ok := labelName(s.Operands[0])
|
|
if !ok {
|
|
return nil, fmt.Errorf("jump target must be a local label")
|
|
}
|
|
if resolve != nil && mnem != "CALL" {
|
|
name = resolve(name)
|
|
}
|
|
target, ok := offsets[name]
|
|
if !ok {
|
|
return nil, fmt.Errorf("undefined label %q", name)
|
|
}
|
|
rel := int64(target - (pc + jumpSize(mnem, long)))
|
|
|
|
if !long {
|
|
if !fits8(rel) {
|
|
return nil, fmt.Errorf("jump to %q does not fit the short form", name)
|
|
}
|
|
if mnem == "JMP" {
|
|
return []byte{0xEB, byte(int8(rel))}, nil
|
|
}
|
|
cc, _ := condCode(mnem)
|
|
return []byte{0x70 + byte(cc), byte(int8(rel))}, nil
|
|
}
|
|
switch mnem {
|
|
case "JMP":
|
|
return append([]byte{0xE9}, le32(rel)...), nil
|
|
case "CALL":
|
|
return append([]byte{0xE8}, le32(rel)...), nil
|
|
default:
|
|
cc, _ := condCode(mnem)
|
|
return append([]byte{0x0F, 0x80 + byte(cc)}, le32(rel)...), nil
|
|
}
|
|
}
|
|
|
|
// isSBCall reports whether the CALL operand is a symbol reference.
|
|
func isSBCall(s *ast.Instr) bool {
|
|
return len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpAddr &&
|
|
s.Operands[0].Addr.Sym != nil && s.Operands[0].Addr.Sym.Pseudo == "SB"
|
|
}
|
|
|
|
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
|
|
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
|
|
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
m, ok := o.(sbMem)
|
|
if !ok {
|
|
return nil, nil, fmt.Errorf("CALL: unsupported operand")
|
|
}
|
|
opcode := []byte{0xE8}
|
|
if strings.ToUpper(s.Mnemonic.Text) == "JMP" {
|
|
opcode = []byte{0xE9} // a tail call, no return address pushed
|
|
}
|
|
e := &enc{}
|
|
if err := e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil {
|
|
return nil, nil, err
|
|
}
|
|
ps := make([]sbPatch, len(e.patches))
|
|
for i, p := range e.patches {
|
|
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelCall}
|
|
}
|
|
return e.out, ps, nil
|
|
}
|
|
|
|
// labelName extracts a local-label name from a jump operand.
|
|
func labelName(op *ast.Operand) (string, bool) {
|
|
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
|
|
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
|
|
return op.Addr.Sym.Name, true
|
|
}
|
|
return "", false
|
|
}
|
|
|
|
// indirectJumpTarget reports whether the JMP/CALL operand addresses a
|
|
// register or a memory location rather than a label or a static symbol.
|
|
// A bare identifier is a register when the register table knows the name and
|
|
// a label otherwise, which is exactly how the parser cannot distinguish them.
|
|
func indirectJumpTarget(s *ast.Instr) bool {
|
|
if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr {
|
|
return false
|
|
}
|
|
a := s.Operands[0].Addr
|
|
if a.Base != "" || a.Index != "" {
|
|
return true
|
|
}
|
|
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
|
|
if _, ok := ParseReg(a.Sym.Name); ok {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// encodeIndirectJump assembles a JMP/CALL through a register or memory
|
|
// operand, which carries no relocation and no label to resolve.
|
|
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
|
ops := make([]Operand, len(s.Operands))
|
|
for i, op := range s.Operands {
|
|
o, err := operandFromAST(op, 8, frameInfo{}, nil)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
ops[i] = o
|
|
}
|
|
e := &enc{}
|
|
if err := e.encodeIndirectBranch(mnem, ops); err != nil {
|
|
return nil, err
|
|
}
|
|
return e.out, nil
|
|
}
|
|
|
|
// spReg is the hardware stack pointer used to realise FP/SP pseudo-operands.
|
|
var spReg = Reg{idx: 4, size: 8}
|
|
|
|
// operandFromAST converts a parsed operand into an encoder Operand, applying
|
|
// the frame translation to FP/SP pseudo-register operands.
|
|
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
|
switch op.Kind {
|
|
case ast.OpImmediate:
|
|
if op.Imm.HasVal {
|
|
v := op.Imm.Val
|
|
if op.Imm.Neg {
|
|
v = -v
|
|
}
|
|
return Imm(v), nil
|
|
}
|
|
return nil, fmt.Errorf("non-integer immediate not supported")
|
|
|
|
case ast.OpAddr:
|
|
a := op.Addr
|
|
|
|
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
|
|
// symbol, not the address displacement.
|
|
if a.Sym != nil && a.Sym.Pseudo == "FP" {
|
|
off := a.Sym.Offset + fi.fpAdjust
|
|
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
|
|
}
|
|
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP).
|
|
if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" {
|
|
off := fi.spAdjust + a.Sym.Offset
|
|
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
|
|
}
|
|
// SB (global symbol): a symbol defined in the same file (GLOBL) is
|
|
// encoded RIP-relative and resolved by the file-level layout;
|
|
// anything not defined here needs object-file emission.
|
|
if a.Sym != nil && a.Sym.Pseudo == "SB" {
|
|
if link == nil || link.symbols == nil {
|
|
return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name)
|
|
}
|
|
if !link.symbols[a.Sym.Name] {
|
|
if a.Sym.Static {
|
|
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
|
|
}
|
|
if !link.allowExternal {
|
|
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
|
|
}
|
|
}
|
|
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
|
|
}
|
|
|
|
// Memory with a real base register: (base), off(base), (base)(index*scale).
|
|
if a.Base != "" {
|
|
base, ok := ParseReg(a.Base)
|
|
if !ok {
|
|
return nil, fmt.Errorf("unknown base register %q", a.Base)
|
|
}
|
|
m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size}
|
|
if a.Index != "" {
|
|
idx, ok := ParseReg(a.Index)
|
|
if !ok {
|
|
return nil, fmt.Errorf("unknown index register %q", a.Index)
|
|
}
|
|
m.Index = idx
|
|
m.Scale = a.Scale
|
|
m.HasIndex = true
|
|
}
|
|
return m, nil
|
|
}
|
|
// Bare register.
|
|
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
|
|
if r, ok := ParseReg(a.Sym.Name); ok {
|
|
return r, nil
|
|
}
|
|
}
|
|
return nil, fmt.Errorf("operand form not yet supported")
|
|
}
|
|
return nil, fmt.Errorf("unsupported operand")
|
|
}
|