// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "bytes" "strings" "testing" "golang.org/x/arch/x86/x86asm" "sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/parser" ) // firstText parses src and returns its first TEXT function. func firstText(t *testing.T, src string) *ast.Text { t.Helper() f, errs := parser.Parse("f_amd64.s", src) if len(errs) > 0 { t.Fatalf("parse: %v", errs) } for _, d := range f.Decls { if txt, ok := d.(*ast.Text); ok { return txt } } t.Fatal("no TEXT function found") return nil } // disasm decodes a machine-code blob into Intel-syntax instruction strings. func disasm(t *testing.T, code []byte) []string { t.Helper() var out []string for len(code) > 0 { inst, err := x86asm.Decode(code, 64) if err != nil { t.Fatalf("decode %x: %v", code, err) } out = append(out, x86asm.IntelSyntax(inst, 0, nil)) code = code[inst.Len:] } return out } func hexBytes(b []byte) string { var sb strings.Builder for _, x := range b { sb.WriteString(" ") const hexdig = "0123456789abcdef" sb.WriteByte(hexdig[x>>4]) sb.WriteByte(hexdig[x&0xf]) } return strings.TrimSpace(sb.String()) } func TestAssembleLoop(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·f(SB), NOSPLIT, $0 XORQ AX, AX loop: ADDQ $1, AX CMPQ AX, $10 JLT loop RET `) code, labels, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } if _, ok := labels["loop"]; !ok { t.Fatalf("label 'loop' not recorded: %v", labels) } got := strings.Join(disasm(t, code), "\n") want := strings.Join([]string{ "xor rax, rax", "add rax, 0x1", "cmp rax, 0xa", "jl 0x0", "ret", }, "\n") gotLines := strings.Split(got, "\n") wantLines := strings.Split(want, "\n") if len(gotLines) != len(wantLines) { t.Fatalf("instruction count mismatch:\n got:\n%s\n want:\n%s", got, want) } for i := range wantLines { if strings.HasPrefix(wantLines[i], "jl") { if !strings.HasPrefix(gotLines[i], "jl") { t.Errorf("line %d: got %q, want a jl", i, gotLines[i]) } continue } if gotLines[i] != wantLines[i] { t.Errorf("line %d: got %q, want %q", i, gotLines[i], wantLines[i]) } } } func TestAssembleMemory(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·g(SB), NOSPLIT, $0 MOVQ (AX), BX MOVQ 8(AX), CX LEAQ (AX)(BX*4), DX RET `) code, _, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } got := strings.Join(disasm(t, code), "\n") want := strings.Join([]string{ "mov rbx, qword ptr [rax]", "mov rcx, qword ptr [rax+0x8]", "lea rdx, ptr [rax+4*rbx]", "ret", }, "\n") if got != want { t.Errorf("assemble memory:\n got:\n%s\n want:\n%s", got, want) } } // TestAssembleFP verifies the FP pseudo-register translation for a NOSPLIT $0 // function against the exact bytes the Go assembler produces (verified via // `go tool objdump`): x+N(FP) maps to (N+8)(SP). func TestAssembleFP(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·loadarg(SB), NOSPLIT, $0-24 MOVQ p+0(FP), AX MOVQ n+8(FP), CX ADDQ CX, AX MOVQ AX, ret+16(FP) RET `) code, _, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } // From `go tool objdump` of the Go-assembled function: // MOVQ 0x8(SP), AX 488b442408 // MOVQ 0x10(SP), CX 488b4c2410 // ADDQ CX, AX 4801c8 // MOVQ AX, 0x18(SP) 4889442418 // RET c3 want := []byte{ 0x48, 0x8b, 0x44, 0x24, 0x08, 0x48, 0x8b, 0x4c, 0x24, 0x10, 0x48, 0x01, 0xc8, 0x48, 0x89, 0x44, 0x24, 0x18, 0xc3, } if hexBytes(code) != hexBytes(want) { t.Errorf("FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } // TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame // function containing a CALL receives: the PUSHQ BP prologue with no stack // adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the // Go assembler produces. The push is the frame, so the offset must not count // it twice. func TestAssembleFramelessCall(t *testing.T) { f, errs := parser.Parse("frameless_call_amd64.s", ` #include "textflag.h" TEXT ·withcall(SB), NOSPLIT, $0-16 MOVQ x+0(FP), AX CALL ·other(SB) MOVQ AX, ret+8(FP) RET TEXT ·other(SB), NOSPLIT, $0-0 RET `) if len(errs) > 0 { t.Fatalf("parse: %v", errs) } img, err := AssembleFile(f) if err != nil { t.Fatalf("AssembleFile: %v", err) } code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...) for _, r := range img.Funcs[0].Relocs { for j := r.Off; j < r.Off+4 && j < len(code); j++ { code[j] = 0 } } // From `go tool objdump` of the Go-assembled function: // PUSHQ BP 55 // MOVQ SP, BP 4889e5 // MOVQ 0x10(SP), AX 488b442410 // CALL other e800000000 // MOVQ AX, 0x18(SP) 4889442418 // POPQ BP 5d // RET c3 want := []byte{ 0x55, 0x48, 0x89, 0xe5, 0x48, 0x8b, 0x44, 0x24, 0x10, 0xe8, 0x00, 0x00, 0x00, 0x00, 0x48, 0x89, 0x44, 0x24, 0x18, 0x5d, 0xc3, } if hexBytes(code) != hexBytes(want) { t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } // TestAssembleFrame verifies a function with a non-zero frame: the Go-style // prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against // the bytes the Go assembler produces. func TestAssembleFrame(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·withframe(SB), NOSPLIT, $16-16 MOVQ a+0(FP), AX MOVQ b+8(FP), CX ADDQ CX, AX MOVQ AX, ret+16(FP) RET `) code, _, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } // From `go tool objdump`: // PUSHQ BP 55 // MOVQ SP, BP 4889e5 // SUBQ $0x10, SP 4883ec10 // MOVQ 0x20(SP), AX 488b442420 (0 + 16 + 16) // MOVQ 0x28(SP), CX 488b4c2428 (8 + 16 + 16) // ADDQ CX, AX 4801c8 // MOVQ AX, 0x30(SP) 4889442430 (16 + 16 + 16) // ADDQ $0x10, SP 4883c410 // POPQ BP 5d // RET c3 want := []byte{ 0x55, 0x48, 0x89, 0xe5, 0x48, 0x83, 0xec, 0x10, 0x48, 0x8b, 0x44, 0x24, 0x20, 0x48, 0x8b, 0x4c, 0x24, 0x28, 0x48, 0x01, 0xc8, 0x48, 0x89, 0x44, 0x24, 0x30, 0x48, 0x83, 0xc4, 0x10, 0x5d, 0xc3, } if hexBytes(code) != hexBytes(want) { t.Errorf("frame translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } // TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac // kernels end with; exercising the VEX moves, shuffle and extract forms // through the full parser → encoder path; and checks the output is // byte-identical to the Go assembler's. func TestAssembleVexKernel(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·hsum(SB), NOSPLIT, $0 VPADDQ Y8, Y9, Y8 VEXTRACTI128 $1, Y8, X9 VPADDQ X9, X8, X8 VPSHUFD $0xEE, X8, X9 VPADDQ X9, X8, X8 VMOVQ X8, AX VZEROUPPER RET `) code, _, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } // From the Go-assembled function: // VPADDQ Y8, Y9, Y8 c44135d4c0 // VEXTRACTI128 $1, Y8, X9 c4437d39c101 // VPADDQ X9, X8, X8 c44139d4c1 // VPSHUFD $0xEE, X8, X9 c4417970c8ee // VPADDQ X9, X8, X8 c44139d4c1 // VMOVQ X8, AX c461f97ec0 // VZEROUPPER c5f877 // RET c3 want := []byte{ 0xc4, 0x41, 0x35, 0xd4, 0xc0, 0xc4, 0x43, 0x7d, 0x39, 0xc1, 0x01, 0xc4, 0x41, 0x39, 0xd4, 0xc1, 0xc4, 0x41, 0x79, 0x70, 0xc8, 0xee, 0xc4, 0x41, 0x39, 0xd4, 0xc1, 0xc4, 0x61, 0xf9, 0x7e, 0xc0, 0xc5, 0xf8, 0x77, 0xc3, } if hexBytes(code) != hexBytes(want) { t.Errorf("VEX kernel mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } // TestAssembleShortJumps checks that a tight loop settles on the short (rel8) // jump forms, byte for byte with the Go assembler. func TestAssembleShortJumps(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·loop(SB), NOSPLIT, $0 XORQ AX, AX l1: ADDQ $1, AX CMPQ AX, $10 JLT l1 RET `) code, _, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } // From the Go-assembled function: // XORQ AX, AX 4831c0 // ADDQ $1, AX 4883c001 // CMPQ AX, $10 4883f80a // JLT l1 7cf6 (short, rel8) // RET c3 want := []byte{ 0x48, 0x31, 0xc0, 0x48, 0x83, 0xc0, 0x01, 0x48, 0x83, 0xf8, 0x0a, 0x7c, 0xf6, 0xc3, } if hexBytes(code) != hexBytes(want) { t.Errorf("short-jump mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } // TestAssembleJumpFolding checks jump-to-jump folding: a conditional jump to a // label that only holds an unconditional jump is redirected to the ultimate // target, exactly as the Go toolchain does before it encodes branches. func TestAssembleJumpFolding(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·fold(SB), NOSPLIT, $0 XORQ AX, AX JGE done INCQ AX done: JMP end end: RET `) code, _, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } // From the Go-assembled function: the JGE skips past the done: trampoline // straight to end: // XORQ AX, AX 4831c0 // JGE end 7d05 (folded past done) // INCQ AX 48ffc0 // JMP end eb00 // RET c3 want := []byte{ 0x48, 0x31, 0xc0, 0x7d, 0x05, 0x48, 0xff, 0xc0, 0xeb, 0x00, 0xc3, } if hexBytes(code) != hexBytes(want) { t.Errorf("jump-folding mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } func TestAssemblePrefetch(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·pf(SB), NOSPLIT, $0 PREFETCHNTA (AX) PREFETCHT0 (BX) PREFETCHT1 8(CX) PREFETCHT2 -1(AX)(R12*1) RET `) code, _, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } got := strings.Join(disasm(t, code), "\n") want := strings.Join([]string{ "prefetchnta zmmword ptr [rax]", "prefetcht0 zmmword ptr [rbx]", "prefetcht1 zmmword ptr [rcx+0x8]", "prefetcht2 zmmword ptr [rax+r12-0x1]", "ret", }, "\n") if got != want { t.Errorf("prefetch disassembly mismatch:\n got:\n%s\n want:\n%s", got, want) } // Byte-level expectations: 0F 18 with the variant in the reg field. if hex := hexBytes(code[:3]); hex != "0f 18 00" { t.Errorf("PREFETCHNTA bytes: got %s, want 0f 18 00", hex) } if hex := hexBytes(code[3:6]); hex != "0f 18 0b" { t.Errorf("PREFETCHT0 bytes: got %s, want 0f 18 0b", hex) } } // TestAssembleBareJump checks that a zero-operand jump (which parses, because // the parser does not arity-check mnemonics) is rejected with an error rather // than panicking in the layout loop, which indexes Operands[0] before the // emission pass gets a chance to diagnose the arity. func TestAssembleBareJump(t *testing.T) { for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} { fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n") if _, _, err := Assemble(fn); err == nil { t.Errorf("%s with no operand: expected an error, got none", mnem) } } } // TestSubSPEncodings pins the prologue SUB against the bytes go tool asm // emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything // larger. The intermediate 129..255 range used to encode an ADD with a // truncated immediate, moving SP the wrong way. func TestSubSPEncodings(t *testing.T) { for _, tt := range []struct { size int want []byte }{ {8, []byte{0x48, 0x83, 0xEC, 0x08}}, {127, []byte{0x48, 0x83, 0xEC, 0x7F}}, {128, []byte{0x48, 0x81, 0xEC, 0x80, 0x00, 0x00, 0x00}}, {200, []byte{0x48, 0x81, 0xEC, 0xC8, 0x00, 0x00, 0x00}}, {255, []byte{0x48, 0x81, 0xEC, 0xFF, 0x00, 0x00, 0x00}}, {4096, []byte{0x48, 0x81, 0xEC, 0x00, 0x10, 0x00, 0x00}}, } { got := subSP(tt.size) if !bytes.Equal(got, tt.want) { t.Errorf("subSP(%d) = %x, want %x", tt.size, got, tt.want) } } } // TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the // full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It // asserts the three behaviours the toolchain shows: each prefix statement is // a standalone byte with a PC of its own (so a label placed on the LOCK // points at the F0), the data pseudo-ops write their literal bytes inline, // and END terminates nothing (the statements after it still belong to the // function and carry no trace of it). func TestAssemblePseudoStatements(t *testing.T) { fn := firstText(t, ` #include "textflag.h" TEXT ·pseudo(SB), NOSPLIT, $0-0 pfx: LOCK CMPXCHGQ AX, (BX) REP MOVSQ BYTE $0x0f BYTE $0x1f WORD $0x1234 END BYTE $0x02 RET `) code, labels, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } // go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3 want := []byte{ 0xf0, 0x48, 0x0f, 0xb1, 0x03, 0xf3, 0x48, 0xa5, 0x0f, 0x1f, 0x34, 0x12, 0x02, 0xc3, } if hexBytes(code) != hexBytes(want) { t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } // The label sits on the LOCK byte, exactly where the toolchain's PC // listing puts it. if off := labels["pfx"]; off != 0 { t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off) } // The trailing BYTE lands where the layout says: after the 8 bytes of // LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing // none. if code[12] != 0x02 { t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12]) } } // TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over // ADJSP: the straight-line sum of the adjustments must be zero at each // RET, branches in between counting for nothing (verified against go tool // asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a // $16/$-16 pair with a JMP in between assembles). func TestAssembleAdjspBalance(t *testing.T) { // Balanced pair with a branch in between, bytes pinned from go tool asm. fn := firstText(t, ` #include "textflag.h" TEXT ·adjsp(SB), NOSPLIT, $0-0 ADJSP $16 JMP body body: ADJSP $-16 RET `) code, _, err := Assemble(fn) if err != nil { t.Fatalf("Assemble: %v", err) } want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3} if hexBytes(code) != hexBytes(want) { t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } // Unbalanced at the RET: the toolchain diagnoses, so must we. _, _, err = Assemble(firstText(t, ` #include "textflag.h" TEXT ·unbalanced(SB), NOSPLIT, $0-0 ADJSP $16 RET `)) if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") { t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err) } // The check runs per RET: a closed pair before the first RET does not // excuse an open adjustment before the second. _, _, err = Assemble(firstText(t, ` #include "textflag.h" TEXT ·tworet(SB), NOSPLIT, $0-0 ADJSP $8 ADJSP $-8 RET mid: ADJSP $8 RET `)) if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") { t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err) } // A framed function: the assembler's own prologue and epilogue // contribute matching deltas, so the pair in the body still balances, // and the bytes match go tool asm end to end. fn = firstText(t, ` #include "textflag.h" TEXT ·framed(SB), $16-8 ADJSP $8 ADJSP $-8 RET `) code, _, err = Assemble(fn) if err != nil { t.Fatalf("Assemble framed: %v", err) } want = []byte{ 0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue 0x48, 0x83, 0xEC, 0x08, // ADJSP $8 0x48, 0x83, 0xC4, 0x08, // ADJSP $-8 0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue 0xC3, } if hexBytes(code) != hexBytes(want) { t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want)) } } // TestAssembleRegRange pins the bracketed register range at the statement // level: exactly four consecutive same-width vector registers assemble, the // toolchain's rejected shapes all report an error. func TestAssembleRegRange(t *testing.T) { asm := func(t *testing.T, op string) ([]byte, error) { t.Helper() f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n") if len(errs) > 0 { t.Fatalf("parse %s: %v", op, errs) } code, _, err := Assemble(f.Decls[0].(*ast.Text)) return code, err } for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} { if _, err := asm(t, op); err != nil { t.Errorf("%s: %v", op, err) } } for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} { if _, err := asm(t, op); err == nil { t.Errorf("%s: assembled, want an error", op) } } }