// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "fmt" "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" ) // Assemble encodes the body of a TEXT function into x86-64 machine code, // resolving local labels to relative jump offsets and translating the FP/SP // pseudo-registers onto the hardware stack pointer (matching the Go // assembler's default frame-pointer behaviour). Jumps start in the short // (rel8) form and expand to rel32 when the settled displacement does not fit; // sizes only grow, so the layout reaches a fixed point in a few passes. CALL // has no short form and is always rel32. // // Supported operands: registers, memory (real base register), immediates, // FP/SP frame-relative operands, and local-label jumps. SB (global symbol) // operands require relocations and are not yet supported; the SIMD (VEX/AVX2) // integer and shuffle/extract/permute/move set is in. // // Like the other architectures, the stack-growth guard (the morestack check // in the prologue and the call back into the runtime in the epilogue) is not // emitted: the bytes match go tool asm only for NOSPLIT functions or // zero-frame leaves, where the toolchain emits no guard either. func Assemble(t *ast.Text) ([]byte, map[string]int, error) { code, _, labels, _, _, _, err := assemble(t, nil) return code, labels, err } // linkInfo carries file-level symbol context into a single-function assembly: // the set of static symbols a GLOBL in the same file defines. A nil link // rejects SB operands outright (single-function assembly cannot resolve // them). When allowExternal is set, a reference to a symbol no GLOBL in the // file defines is recorded as an external relocation instead of failing // the object-file emitters resolve it at link time. type linkInfo struct { symbols map[string]bool allowExternal bool } // sbPatch is a function-relative static-symbol relocation: the disp32 field // at off must become the symbol's address minus after, where after is the // function-relative address just past the instruction. type sbPatch struct { off int after int name string addend int64 kind RelocKind } // spadjStep is one stack-adjustment boundary within a function: Value is the // SP delta from the entry state (just below the return address) in effect // from PC (function-relative) until the next step. The steps feed the // pcsp table of the object-file emitters. type spadjStep struct { pc int value int } // assemble encodes a TEXT body, returning the machine code, the static-symbol // patch sites (for the file-level layout to resolve), the label table and the // stack-adjustment boundaries. func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, []floatPoolEntry, error) { if err := checkAdjspBalance(t); err != nil { return nil, nil, nil, nil, nil, nil, err } fi := computeFrame(t) chain := jumpChain(t) resolve := func(name string) string { if r, ok := chain[name]; ok { return r } return name } // Layout: iterate jump sizes to a fixed point. The stack-split guard // prefix and the trailing morestack block participate in the iteration: // their conditional branches relax from rel8 to rel32 when the body // outgrows the short form. long := make([]bool, len(t.Body)) sizes := make([]int, len(t.Body)) offsets := map[string]int{} pcs := make([]int, len(t.Body)) var guardJBlong, guardJBElong, moreJMPlong bool poolSeen := map[string]bool{} var poolList []floatPoolEntry for { guard := fi.guardLen(guardJBlong, guardJBElong) pos := guard + len(fi.prologue) for i, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: offsets[s.Name.Text] = pos case *ast.Instr: sz, err := instrSize(s, fi, long[i], link) if err != nil { return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) } sizes[i] = sz pcs[i] = pos pos += sz } } bodyLen := pos - (guard + len(fi.prologue)) // Expand any short jump whose displacement no longer fits rel8. changed := false for i, stmt := range t.Body { s, ok := stmt.(*ast.Instr) if !ok { continue } mnem := strings.ToUpper(s.Mnemonic.Text) if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] { continue } // A zero-operand jump parses; its arity is reported during // emission (encodeJump), so the layout must not index Operands. if len(s.Operands) != 1 { continue } name, ok := labelName(s.Operands[0]) if !ok { continue // reported during emission } target, ok := offsets[resolve(name)] if !ok { continue // reported during emission } rel := int64(target - (pcs[i] + jumpSize(mnem, false))) if !fits8(rel) { long[i] = true changed = true } } // The guard's conditional branches target the morestack block, which // starts right after the body: the JBE measures from the end of the // guard, so its displacement is the prologue plus the body. if !guardJBElong && !fits8(int64(len(fi.prologue)+bodyLen)) { guardJBElong = true changed = true } if fi.splitClass == 2 && !guardJBlong { // The underflow JB sits before the CMPQ; its displacement spans // the rest of the guard plus the prologue and the body. The JB // is still the short form this branch tests (relaxing it is this // branch's job), so guardLen is taken with a short JB and the // subtraction drops the prefix and the JB's own 2 bytes. rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2) if !fits8(int64(rest + len(fi.prologue) + bodyLen)) { guardJBlong = true changed = true } } // The morestack JMP returns to the function start, so its // displacement is the negated distance from its own end; while it is // still short, its own length is 2 bytes. if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) { moreJMPlong = true changed = true } if !changed { break } } // Pass 2: emit. The guard comes first, then the prologue, the body and // the morestack block. guardLen := fi.guardLen(guardJBlong, guardJBElong) bodyLen := 0 { pos := guardLen + len(fi.prologue) for i, stmt := range t.Body { if _, ok := stmt.(*ast.Instr); ok { pos += sizes[i] } } bodyLen = pos - (guardLen + len(fi.prologue)) } var out []byte var patches []sbPatch if fi.needSplit { // The JBE ends the guard, so its displacement is the prologue plus // the body; the underflow JB additionally spans the trailing CMPQ and // JBE, whose combined length is guardLen minus the prefix and the // JB's own length (2 short, 6 long). jbLen := 2 if guardJBlong { jbLen = 6 } guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen)) out = append(out, guard...) patches = append(patches, tlsPatch) } out = append(out, fi.prologue...) var steps []spadjStep var lines []LineEntry if fi.useFP { // PUSHQ BP saves the return-address-relative base (+8); the MOVQ // changes nothing; SUBQ $size, SP completes the frame. steps = append(steps, spadjStep{guardLen + 1, 8}, spadjStep{guardLen + len(fi.prologue), 8 + fi.size}, ) } // frameBase is the SP delta the prologue leaves: 8 for the saved base // pointer plus the frame, 0 frameless. bodyDelta tracks the ADJSP // statements' straight-line sum, so a mid-body step's value is the // frame base plus what the body has opened so far. frameBase, bodyDelta := 0, 0 if fi.useFP { frameBase = 8 + fi.size } pos := guardLen + len(fi.prologue) for i, stmt := range t.Body { s, ok := stmt.(*ast.Instr) if !ok { continue } if strings.ToUpper(s.Mnemonic.Text) == "RET" && fi.useFP { // The RET's epilogue prefix unwinds: ADDQ $size, SP restores // the saved-BP-only stack, POPQ BP the entry state. epi := len(fi.epilogue) steps = append(steps, spadjStep{pos + epi - 1, 8}, spadjStep{pos + epi, 0}, ) } code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link) if err != nil { return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err) } for _, entry := range pool { if !poolSeen[entry.name] { poolSeen[entry.name] = true poolList = append(poolList, entry) } } if len(code) != sizes[i] { return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i]) } if strings.ToUpper(s.Mnemonic.Text) == "CALL" { for k := range ps { ps[k].kind = RelCall } } if strings.ToUpper(s.Mnemonic.Text) == "ADJSP" && len(s.Operands) == 1 && s.Operands[0].Imm.HasVal { // The statement shifted SP mid-body: record the new running // delta as the value in effect from just past the instruction. v := s.Operands[0].Imm.Val if s.Operands[0].Imm.Neg { v = -v } bodyDelta += int(v) steps = append(steps, spadjStep{pos + len(code), frameBase + bodyDelta}) } patches = append(patches, ps...) lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line}) out = append(out, code...) pos += len(code) } if fi.needSplit { // The morestack block: CALL runtime.morestack_noctxt, then a JMP // back to the function entry. jmpLen := 2 if moreJMPlong { jmpLen = 5 } jmpDisp := -int64(pos + 5 + jmpLen) suffix, callPatch := buildMoreStack(int32(jmpDisp)) callPatch.off += pos callPatch.after = pos + 5 patches = append(patches, callPatch) out = append(out, suffix...) pos += len(suffix) } _ = pos return out, patches, offsets, steps, lines, poolList, nil } // jumpChain precomputes jump-to-jump folding: a label whose first instruction // is an unconditional local jump redirects its own jumpers to the ultimate // target. The Go toolchain chases exactly these chains (the linker's xfol // pass) before it encodes branches, so matching its bytes requires the same // redirection. func jumpChain(t *ast.Text) map[string]string { // label → the target of its leading unconditional local JMP, if any. leadsTo := map[string]string{} for i, stmt := range t.Body { l, ok := stmt.(*ast.Label) if !ok { continue } // Stacked labels share an address: skip to the first instruction. j := i + 1 for j < len(t.Body) { if _, isLabel := t.Body[j].(*ast.Label); !isLabel { break } j++ } if j >= len(t.Body) { continue } in, ok := t.Body[j].(*ast.Instr) if !ok || strings.ToUpper(in.Mnemonic.Text) != "JMP" || len(in.Operands) != 1 { continue } if name, ok := labelName(in.Operands[0]); ok { leadsTo[l.Name.Text] = name } } // Chase each chain to its end, guarding against cycles. chain := map[string]string{} for name := range leadsTo { visited := map[string]bool{name: true} cur := name for { next, ok := leadsTo[cur] if !ok || visited[next] { break } visited[next] = true cur = next } if cur != name { chain[name] = cur } } return chain } // frameInfo carries the frame layout derived from the TEXT directive. type frameInfo struct { size int // local frame size ($framesize) useFP bool // a frame pointer (BP) is set up fpAdjust int64 // added to x+N(FP) to reach the hardware SP-relative offset spAdjust int64 // x-N(SP) becomes (spAdjust - N)(SP) prologue []byte epilogue []byte // Stack-split guard state (matching the toolchain's stacksplit): needSplit // is false for NOSPLIT functions and for leaf functions whose frame is // below StackSmall, which the toolchain auto-marks NOSPLIT. needSplit bool splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig framesize int // the size the guard checks: frame+8 for framed functions } // Stack-frame size classes from runtime/stack.go. const ( stackSmall = 128 stackBig = 4096 ) // sbPatch gains a kind so the emitters can tell CALL and TLS patches from // plain PC-relative displacements. // computeFrame derives the frame layout, matching the Go assembler's default // (a frame pointer is used whenever the function has a non-zero frame). It // also decides whether the function needs the stack-split guard, mirroring // obj6: a NOSPLIT function never splits, and a leaf function whose frame is // below StackSmall is auto-marked NOSPLIT. One deliberate deviation: the // toolchain treats zero-argument runtime calls (duffcopy and friends) as // leaf-compatible; here any CALL makes the function a non-leaf. func computeFrame(t *ast.Text) frameInfo { fi := frameInfo{} if t.Frame != nil && t.Frame.Imm.HasVal { fi.size = int(t.Frame.Imm.Val) } if fi.size == 0 && hasCall(t) { // The toolchain gives a frameless function containing a CALL an // 8-byte frame for the pushed base pointer: the prologue saves BP // with no stack adjustment, every RET pops it back, FP references // pass one extra slot, and the virtual SP is the hardware SP. fi.size = 8 fi.useFP = true // The push is the frame: the saved BP sits at SP+0 and the // return address at SP+8, so arguments begin at SP+16. Unlike // a SUBQ frame, the 8-byte size must not be added again. fi.fpAdjust = 16 fi.spAdjust = 0 fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP fi.epilogue = []byte{0x5D} // POPQ BP } else if fi.size > 0 { fi.useFP = true fi.fpAdjust = int64(fi.size) + 16 // frame + saved BP + return address fi.spAdjust = int64(fi.size) fi.prologue = prologueBytes(fi.size) fi.epilogue = epilogueBytes(fi.size) } else { fi.fpAdjust = 8 // return address only } noSplit := false for _, f := range t.Flags { if strings.EqualFold(f, "NOSPLIT") { noSplit = true } } // The toolchain's autoffset: the frame plus the saved base pointer. framesize := fi.size if framesize > 0 { framesize += 8 } switch { case noSplit: case framesize < stackSmall && !hasCall(t): // Auto-NOSPLIT, as the toolchain's leaf search concludes. default: fi.needSplit = true fi.framesize = framesize switch { case framesize <= stackSmall: fi.splitClass = 0 case framesize <= stackBig: fi.splitClass = 1 default: fi.splitClass = 2 } } return fi } // hasCall reports whether the function body contains a CALL instruction. func hasCall(t *ast.Text) bool { for _, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } if strings.ToUpper(in.Mnemonic.Text) == "CALL" { return true } } return false } // checkAdjspBalance mirrors the toolchain's push/pop walk: every ADJSP // shifts SP away from the entry state and every RET must see the shifts // closed. The assembler's own prologue and epilogue contribute matching // deltas on both sides, so the statements' straight-line sum must be zero // at each RET; branches do not reset the walk, which runs over the program // list in source order. go tool asm reports an offender as "unbalanced // PUSH/POP" (verified against ADJSP $16 before a RET, accepted as a // $16/$-16 pair, per-RET rather than per-function). func checkAdjspBalance(t *ast.Text) error { delta := 0 for _, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } switch strings.ToUpper(in.Mnemonic.Text) { case "ADJSP": if len(in.Operands) != 1 || !in.Operands[0].Imm.HasVal { continue // reported during emission } v := in.Operands[0].Imm.Val if in.Operands[0].Imm.Neg { v = -v } delta += int(v) case "RET": if delta != 0 { return fmt.Errorf("unbalanced PUSH/POP") } } } return nil } // guardLen returns the byte length of the stack-split guard prefix. The // final conditional branch (JBE, and JB in the big class) is 2 bytes in the // short form and 6 in the long form. func (fi frameInfo) guardLen(jbLong, jbeLong bool) int { if !fi.needSplit { return 0 } jb, jbe := 2, 2 if jbLong { jb = 6 } if jbeLong { jbe = 6 } switch fi.splitClass { case 0: return 9 + 4 + jbe case 1: return 9 + 8 + 4 + jbe default: return 9 + 3 + 7 + jb + 4 + jbe } } // buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the // already-computed displacements of the conditional branches that jump to the // morestack block (unused in classes without them). The TLS load carries a // R_TLS_LE patch site at offset 5. func buildGuard(fi frameInfo, jbeDisp, jbDisp int32) ([]byte, sbPatch) { out := []byte{ 0x64, 0x4c, 0x8b, 0x34, 0x25, // MOVQ FS:0, R14 0, 0, 0, 0, // TLS slot offset, filled by the linker } tls := sbPatch{off: 5, after: 9, kind: RelTLSLE} jmp := func(op8, op32 byte, disp int32) []byte { if disp >= -128 && disp <= 127 { return []byte{op8, byte(disp)} } return append([]byte{0x0F, op32}, le32(int64(disp))...) } switch fi.splitClass { case 0: // CMPQ SP, 16(R14) out = append(out, 0x49, 0x3b, 0x66, 0x10) out = append(out, jmp(0x76, 0x86, jbeDisp)...) case 1: // LEAQ -(framesize-StackSmall)(SP), R12; CMPQ R12, 16(R14) out = append(out, 0x4c, 0x8d, 0xa4, 0x24) out = append(out, le32(-int64(fi.framesize-stackSmall))...) out = append(out, 0x4d, 0x3b, 0x66, 0x10) out = append(out, jmp(0x76, 0x86, jbeDisp)...) default: // MOVQ SP, R12; SUBQ $(framesize-StackSmall), R12; JB; CMPQ R12, 16(R14) out = append(out, 0x49, 0x89, 0xe4) out = append(out, 0x49, 0x81, 0xec) out = append(out, le32(int64(fi.framesize-stackSmall))...) out = append(out, jmp(0x72, 0x82, jbDisp)...) out = append(out, 0x4d, 0x3b, 0x66, 0x10) out = append(out, jmp(0x76, 0x86, jbeDisp)...) } return out, tls } // buildMoreStack emits the trailing block: CALL runtime.morestack_noctxt // (patched by the linker) and a JMP back to the function start. func buildMoreStack(jmpDisp int32) ([]byte, sbPatch) { out := []byte{0xE8, 0, 0, 0, 0} call := sbPatch{off: 1, after: 5, name: "runtime\u00b7morestack_noctxt", kind: RelCall} out = append(out, jmpBytes(jmpDisp)...) return out, call } // jmpBytes encodes a near JMP in the short or long form. func jmpBytes(disp int32) []byte { if disp >= -128 && disp <= 127 { return []byte{0xEB, byte(disp)} } return append([]byte{0xE9}, le32(int64(disp))...) } // prologueBytes emits: PUSHQ BP; MOVQ SP, BP; SUBQ $size, SP. func prologueBytes(size int) []byte { out := []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP return append(out, subSP(size)...) } // epilogueBytes emits: ADDQ $size, SP; POPQ BP. func epilogueBytes(size int) []byte { out := addSP(size) return append(out, 0x5D) // POPQ BP } func subSP(size int) []byte { // SUBQ $size, SP // imm8 holds -128..127; anything larger takes the imm32 form, exactly as // the Go assembler encodes it (verified for 8, 128, 200 and 255). if size >= -128 && size <= 127 { return []byte{0x48, 0x83, 0xEC, byte(int8(size))} } return append([]byte{0x48, 0x81, 0xEC}, le32(int64(size))...) } func addSP(size int) []byte { // ADDQ $size, SP if size >= -128 && size <= 127 { return []byte{0x48, 0x83, 0xC4, byte(int8(size))} } return append([]byte{0x48, 0x81, 0xC4}, le32(int64(size))...) } // instrSize returns the encoded length of an instruction (layout pass). // encodeInstr already includes the epilogue for a RET in a frame-pointer // function; jumps use their short or long form (never an epilogue). func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) { mnem := strings.ToUpper(s.Mnemonic.Text) if isJumpMnemonic(mnem) { if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) { return 5, nil // opcode + rel32, always the long form } if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) { code, err := encodeIndirectJump(s, mnem) if err != nil { return 0, err } return len(code), nil } return jumpSize(mnem, long), nil } code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link) if err != nil { return 0, err } return len(code), nil } func isJumpMnemonic(mnem string) bool { if mnem == "JMP" || mnem == "CALL" { return true } _, ok := condCode(mnem) return ok } // jumpSize returns the length of a jump instruction in the requested form: // short (rel8) where available, otherwise the rel32 form. CALL is always // rel32. func jumpSize(mnem string, long bool) int { if mnem == "CALL" { return 5 // opcode + rel32 } if !long { return 2 // opcode + rel8 } if mnem == "JMP" { return 5 // E9 + rel32 } return 6 // 0x0F 0x8x + rel32 } // encodeInstr encodes one instruction, resolving jump targets against offsets // (relative to pc, the instruction's own offset). A RET in a frame-pointer // function is prefixed with the epilogue. resolve, when non-nil, redirects a // jump label through the jump-to-jump chain before the offset lookup. func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) { mnem := strings.ToUpper(s.Mnemonic.Text) var prefix []byte if mnem == "RET" && fi.useFP { prefix = fi.epilogue } var code []byte var ps []sbPatch var pool []floatPoolEntry var err error if isJumpMnemonic(mnem) { if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) { // CALL/JMP sym(SB): a rel32 call (or tail call) against a // static or external symbol, resolved by the file-level layout // or the linker. code, ps, err = encodeSBCall(s, link) if err != nil { return nil, nil, nil, err } for i := range ps { ps[i].kind = RelCall } body := pc + len(prefix) for i := range ps { ps[i].off += body ps[i].after = body + len(code) } return append(prefix, code...), ps, nil, nil } if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) { // JMP/CALL through a register or memory: no relocation and no // label to resolve, the operand fully determines the bytes. code, err = encodeIndirectJump(s, mnem) if err != nil { return nil, nil, nil, err } return append(prefix, code...), nil, nil, nil } code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve) } else { code, ps, pool, err = encodeNormal(s, fi, link) } if err != nil { return nil, nil, nil, err } // Anchor the patch fields at function-relative positions: off indexes the // disp32 field, after is the address just past the instruction. body := pc + len(prefix) for i := range ps { ps[i].off += body ps[i].after = body + len(code) } return append(prefix, code...), ps, pool, nil } func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) { mnemUpper := strings.ToUpper(s.Mnemonic.Text) if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" { code, err := encodeBookkeeping(mnemUpper, s) if err != nil { return nil, nil, nil, err } return code, nil, nil, nil } _, size := splitSize(mnemUpper) if size == 0 { size = 8 } ops := make([]Operand, len(s.Operands)) for i, op := range s.Operands { o, err := operandFromAST(mnemUpper, op, size, fi, link) if err != nil { return nil, nil, nil, err } ops[i] = o } e := &enc{} if err := e.encode(s.Mnemonic.Text, ops); err != nil { return nil, nil, nil, err } ps := make([]sbPatch, len(e.patches)) for i, p := range e.patches { ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend} } return e.out, ps, e.floatPoolList(), nil } // encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement // level, before operand conversion: the toolchain's shapes are FUNCDATA // $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the // function body. The symbol reference must not run through the SB-operand // path, which demands file-level resolution the statement never needs. func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) { if len(s.Operands) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", upper, len(s.Operands)) } a, b := s.Operands[0], s.Operands[1] if a.Kind != ast.OpImmediate || !a.Imm.HasVal { return nil, fmt.Errorf("%s: first operand must be an integer immediate", upper) } switch upper { case "FUNCDATA": if b.Kind != ast.OpAddr || b.Addr.Sym == nil || b.Addr.Sym.Pseudo != "SB" { return nil, fmt.Errorf("FUNCDATA: second operand must be a symbol reference") } case "PCDATA": if b.Kind != ast.OpImmediate || !b.Imm.HasVal { return nil, fmt.Errorf("PCDATA: second operand must be an integer immediate") } } return nil, nil } // encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the // target label, in the short (rel8) or long (rel32) form. func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string) ([]byte, error) { if len(s.Operands) != 1 { return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands)) } name, ok := labelName(s.Operands[0]) if !ok { return nil, fmt.Errorf("jump target must be a local label") } if resolve != nil && mnem != "CALL" { name = resolve(name) } target, ok := offsets[name] if !ok { return nil, fmt.Errorf("undefined label %q", name) } rel := int64(target - (pc + jumpSize(mnem, long))) if !long { if !fits8(rel) { return nil, fmt.Errorf("jump to %q does not fit the short form", name) } if mnem == "JMP" { return []byte{0xEB, byte(int8(rel))}, nil } cc, _ := condCode(mnem) return []byte{0x70 + byte(cc), byte(int8(rel))}, nil } switch mnem { case "JMP": return append([]byte{0xE9}, le32(rel)...), nil case "CALL": return append([]byte{0xE8}, le32(rel)...), nil default: cc, _ := condCode(mnem) return append([]byte{0x0F, 0x80 + byte(cc)}, le32(rel)...), nil } } // isSBCall reports whether the CALL operand is a symbol reference. func isSBCall(s *ast.Instr) bool { return len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpAddr && s.Operands[0].Addr.Sym != nil && s.Operands[0].Addr.Sym.Pseudo == "SB" } // encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site. func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) { o, err := operandFromAST(strings.ToUpper(s.Mnemonic.Text), s.Operands[0], 8, frameInfo{}, link) if err != nil { return nil, nil, err } m, ok := o.(sbMem) if !ok { return nil, nil, fmt.Errorf("CALL: unsupported operand") } opcode := []byte{0xE8} if strings.ToUpper(s.Mnemonic.Text) == "JMP" { opcode = []byte{0xE9} // a tail call, no return address pushed } e := &enc{} if err := e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil { return nil, nil, err } ps := make([]sbPatch, len(e.patches)) for i, p := range e.patches { ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelCall} } return e.out, ps, nil } // labelName extracts a local-label name from a jump operand. func labelName(op *ast.Operand) (string, bool) { if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Sym.Name != "" { return op.Addr.Sym.Name, true } return "", false } // indirectJumpTarget reports whether the JMP/CALL operand addresses a // register or a memory location rather than a label or a static symbol. // A bare identifier is a register when the register table knows the name and // a label otherwise, which is exactly how the parser cannot distinguish them. func indirectJumpTarget(s *ast.Instr) bool { if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr { return false } a := s.Operands[0].Addr if a.Base != "" || a.Index != "" { return true } if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" { if _, ok := ParseReg(a.Sym.Name); ok { return true } } return false } // encodeIndirectJump assembles a JMP/CALL through a register or memory // operand, which carries no relocation and no label to resolve. func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) { ops := make([]Operand, len(s.Operands)) for i, op := range s.Operands { o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil) if err != nil { return nil, err } ops[i] = o } e := &enc{} if err := e.encodeIndirectBranch(mnem, ops); err != nil { return nil, err } return e.out, nil } // spReg is the hardware stack pointer used to realise FP/SP pseudo-operands. var spReg = Reg{idx: 4, size: 8} // operandFromAST converts a parsed operand into an encoder Operand, applying // the frame translation to FP/SP pseudo-register operands. mnemUpper is the // instruction's upper-case mnemonic, which the floating-point immediate gate // needs: only the SSE mnemonics whose encoding takes an XMM/memory source // accept one. func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) { switch op.Kind { case ast.OpImmediate: if op.Imm.HasVal { v := op.Imm.Val if op.Imm.Neg { v = -v } return Imm(v), nil } // A floating-point immediate: $1.5, $-1.0 or the parenthesised // $(-1.0) spelling (the constant-expression folder only folds // integers, so that shape arrives with an empty Immediate and only // the raw spelling carries the value). The toolchain rewrites it // into a pooled-constant read on the SSE scalar paths and rejects // it everywhere else. if text, neg, ok := floatImmText(op); ok { if !sseFloatImm[mnemUpper] { return nil, fmt.Errorf("%s does not take a floating-point immediate", mnemUpper) } return FloatImm{Text: text, Neg: neg}, nil } return nil, fmt.Errorf("non-integer immediate not supported") case ast.OpAddr: a := op.Addr // A bracketed register range, [Z0-Z3]: the four-register source of // the 4FMAPS/4VNNIW families. The range must span four consecutive // same-width vector registers, exactly what the toolchain's parser // takes; the EVEX quad-register emit path reads the low end. if a.Range != nil { lo, ok := ParseReg(a.Range.Lo) if !ok { return nil, fmt.Errorf("unknown register %q in range", a.Range.Lo) } hi, ok := ParseReg(a.Range.Hi) if !ok { return nil, fmt.Errorf("unknown register %q in range", a.Range.Hi) } if !lo.isVec() || lo.size != hi.size { return nil, fmt.Errorf("register range %q must span four same-width vector registers", op.Raw) } if hi.idx != lo.idx+3 { return nil, fmt.Errorf("register range %q must span four consecutive registers", op.Raw) } return RegList{Lo: lo, Hi: hi}, nil } // FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the // symbol, not the address displacement. if a.Sym != nil && a.Sym.Pseudo == "FP" { off := a.Sym.Offset + fi.fpAdjust return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil } // SP-relative local: x-N(SP) → (spAdjust + offset)(SP). if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" { off := fi.spAdjust + a.Sym.Offset return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil } // SB (global symbol): a symbol defined in the same file (GLOBL) is // encoded RIP-relative and resolved by the file-level layout; // anything not defined here needs object-file emission. if a.Sym != nil && a.Sym.Pseudo == "SB" { if link == nil || link.symbols == nil { return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name) } if !link.symbols[a.Sym.Name] { if a.Sym.Static { return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name) } if !link.allowExternal { return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name) } } return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil } // Memory with a real base register: (base), off(base), (base)(index*scale). if a.Base != "" { base, ok := ParseReg(a.Base) if !ok { return nil, fmt.Errorf("unknown base register %q", a.Base) } m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size} if a.Index != "" { idx, ok := ParseReg(a.Index) if !ok { return nil, fmt.Errorf("unknown index register %q", a.Index) } m.Index = idx m.Scale = a.Scale m.HasIndex = true } return m, nil } // Index-only memory: the VSIB form the gather/scatter families // read, 8(X4*1). A scaled vector index addresses memory with no // base register; the mod=00 SIB with base field 101 carries it. if a.Index != "" { idx, ok := ParseReg(a.Index) if !ok { return nil, fmt.Errorf("unknown index register %q", a.Index) } return Mem{Index: idx, Scale: a.Scale, Disp: a.Offset, HasIndex: true, Size: size}, nil } // Bare register. if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" { if r, ok := ParseReg(a.Sym.Name); ok { return r, nil } } return nil, fmt.Errorf("operand form not yet supported") } return nil, fmt.Errorf("unsupported operand") } // floatImmText recovers a floating-point immediate's magnitude and sign from // the parsed operand. The ordinary spellings arrive in Imm.Float; the // parenthesised $(-1.0) leaves the Immediate empty, because the integer // folder cannot read it, and only the verbatim operand text still carries // the value. Anything that is not a number a float parser accepts reports // not-ok, so every other shape keeps its existing diagnostic. func floatImmText(op *ast.Operand) (text string, neg bool, ok bool) { if op.Imm.Float != "" { return op.Imm.Float, op.Imm.Neg, true } if op.Imm.HasVal || op.Imm.Str != "" || op.Imm.Sym != nil { return "", false, false } // joinRaw spaced the token texts; the compact spelling is what matters. compact := strings.ReplaceAll(op.Raw, " ", "") inner, ok := strings.CutPrefix(compact, "$(") if !ok || !strings.HasSuffix(inner, ")") { return "", false, false } inner = strings.TrimSuffix(inner, ")") inner = strings.TrimPrefix(inner, "+") if s, ok := strings.CutPrefix(inner, "-"); ok { neg = true inner = s } if inner == "" || !strings.ContainsAny(inner, "0123456789") { return "", false, false } if _, err := strconv.ParseFloat(inner, 64); err != nil { return "", false, false } return inner, neg, true }