Compare commits

..
206 Commits
Author SHA1 Message Date
petrbalvin 9f4f949c1f chore: prepare release v0.34.0
Test / test (push) Successful in 2m11s
Release / gates (push) Successful in 2m11s
Release / build (amd64, linux) (push) Successful in 1m13s
Release / build (arm64, linux) (push) Successful in 1m10s
Release / build (loong64, linux) (push) Successful in 1m12s
Release / build (riscv64, linux) (push) Successful in 1m33s
Release / release (push) Successful in 58s
2026-09-20 01:44:23 +02:00
petrbalvin f0d5238c47 docs: state the validation status and correct claims the material contradicts
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin 2931bbd6b2 ci(release): refuse a tag the security policy does not name
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin 63562a503a test(justfile): run the CLI and debugger tests outside the coverage set
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin e836d6150d docs: changelog entry for the loong64 JIT enablement
Test / test (push) Successful in 2m7s
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin 8a51b060da feat(cmd): enable loong64 JIT execution, all trampolines qemu-validated
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin d3d47db727 test(verify): seed the arm64 ABI kernel arguments
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin 0758556b7d docs: changelog entries for the parity round and corpus number
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin ddb8440340 fix(cmd): padding-aware ground-truth comparison
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin f15ff66fb1 fix(riscv64): accept the g spelling of the goroutine register
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin 187e4856d3 feat(amd64): encode the mixed-width extend family and PMOVMSKB
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin d315a998ce fix(arm64): store-exclusive operand order and large-frame parity
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin a6f3828c02 docs: changelog entries for the review fixes
Test / test (push) Successful in 2m4s
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin e3b35bb817 style(testdata): canonical gasm formatting for the verify kernels
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin eb0a89e58d ci(release): state the version contract inline
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin dd32d9e66e chore(justfile): one-line install-man comment and long flag forms
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin 3a73acb20a docs: drop process labels and refresh the architecture and manual pages
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin 7604a9443f fix(cmd): usage exit codes, asm output file and cross-arch ground truth
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin b3908fc43d fix(lsp): parse-error survival, symbol ranges and UTF-16 positions
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin eb8b0cd316 fix(lint): trailing-label CFG guard and the goroutine alias
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin a8bfd54ed2 fix(debug): hardware watchpoints, signal stops and breakpoint restore
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin 375182ef1f fix(verify): arm64 stack save, adaptive canary and host gating
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin 87b1081c53 fix(goobj): external package and symbol indices and arm64 pair relocations
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin f3c8510a58 fix(elf): relocation records, DWARF tables and per-architecture frame data
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin ebdf14939f fix(loong64): FP immediates through R30 and unsigned branch forms
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin 79a2c16bac fix(riscv64): compressed store offsets, FENCE and branch range checks
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin 401386956c fix(arm64): encode shifts, divides and multiplies and align sizes with emission
Assisted-by: GLM 5.3
2026-09-19 23:49:07 +02:00
petrbalvin 4258131a3a fix(amd64): correct guard displacements, frameless FP offsets and immediate ranges
Assisted-by: GLM 5.3
2026-09-19 23:49:07 +02:00
petrbalvin 94e09e8070 fix(format): preserve flag separators and normalise CRLF input
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
petrbalvin 7aefe6a42d fix(parser): parse ABI markers and keep TEXT decls usable on errors
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
petrbalvin ac1c05c793 fix(lexer): tokenise the flag separator and handle NUL and invalid UTF-8
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
petrbalvin 93c47a312a feat(docs): man pages for gasm and every command, guarded against CLI drift
Test / test (push) Successful in 2m4s
Assisted-by: GLM 5.3 Flash
2026-09-19 21:18:43 +02:00
petrbalvin 708d0a0a5e docs: trim the changelog entries to user-visible deltas
Assisted-by: GLM 5.3 Flash
2026-09-19 20:54:56 +02:00
petrbalvin 3c8f7cb411 test(format): pin the fuzz-found crashers as regression seeds
Assisted-by: GLM 5.3 Flash
2026-09-19 20:48:51 +02:00
petrbalvin 7c5b7a1419 docs: add the changelog entries and the corpus number to the readme
Assisted-by: GLM 5.3 Flash
2026-09-19 20:48:51 +02:00
petrbalvin bc3f448738 feat(format): fuzz targets for the parser and formatter
Assisted-by: GLM 5.3 Flash
2026-09-19 20:41:43 +02:00
petrbalvin f37f183577 feat(riscv64): GOROOT instruction shapes, DATA order and offset expressions
Assisted-by: GLM 5.3 Flash
2026-09-19 19:58:43 +02:00
petrbalvin 1e77e58250 feat(gasm): audit a .s corpus with audit-instructions --corpus
Assisted-by: GLM 5.3 Flash
2026-09-19 19:27:30 +02:00
petrbalvin 1d0969ed64 feat(gasm): select the asm and diff architecture with -GOARCH
Assisted-by: GLM 5.3 Flash
2026-09-19 19:20:47 +02:00
petrbalvin 23c001be51 feat(asm): encode indirect JMP and CALL on all four architectures
Assisted-by: GLM 5.3 Flash
2026-09-19 19:17:07 +02:00
petrbalvin 96e81cc98d docs: add the Plan 9 assembly case and real-use note to the README
Test / test (push) Successful in 2m6s
2026-09-19 18:06:18 +02:00
petrbalvin c834d98210 docs: bring the document set into the standard shape
Test / test (push) Successful in 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 03d6d4da54 style: put the repository assembly in gasm fmt canonical form
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 0b42ce7952 style: use one spelling for colour across the CLI
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 288a64ccd2 ci: align the pipelines with the hand-written templates
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 5fddfa704b build: declare the exact toolchain and the canonical recipes
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin a2bb5eeb4e chore: drop the stale comment from the ignore list
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:14 +02:00
petrbalvin 48449b7a7f build: declare the go1.27.1 toolchain
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 3de043c494 docs: add SECURITY.md and record the round in the CHANGELOG
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 0078f7be5c style: purge em dashes from the produced text
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 6a7317d141 chore: trim the ignore list to the convention
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin d08523caa5 docs: move the recipe and version descriptions with the behaviour
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 20e4b8d9c4 ci: align the pipelines with the hand-written templates
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 61f4247cef refactor(gasm): report the toolchain-recorded version
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 049872ddff build: restore the canonical justfile recipe set
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 3669f64ff6 build: install the gasm binary into the user-local bin directory
Test / vet (push) Successful in 46s
Test / test (push) Successful in 2m44s
Test / build (push) Successful in 42s
2026-09-14 23:41:21 +02:00
petrbalvin fff9f75595 chore: prepare release v0.33.0
Release / build (amd64, linux) (push) Successful in 49s
Release / build (arm64, linux) (push) Successful in 43s
Release / build (loong64, linux) (push) Successful in 46s
Release / build (riscv64, linux) (push) Successful in 45s
Test / vet (push) Successful in 47s
Release / release (push) Successful in 18s
Test / test (push) Successful in 2m39s
Test / build (push) Successful in 43s
2026-09-14 23:36:19 +02:00
petrbalvin 40476546df fix(asm): close the oracle parity gaps in frame addressing and calls 2026-09-14 23:25:14 +02:00
petrbalvin 70218e84ba feat(asm): emit the loong64 stack-split guard for big frames 2026-09-14 22:38:58 +02:00
petrbalvin db50b98179 feat(asm): emit the loong64 stack-split guard for small and medium frames 2026-09-14 21:21:58 +02:00
petrbalvin 2e2c0b82a0 feat(asm): emit the riscv64 stack-split guard and fix large-frame addressing 2026-09-14 21:09:13 +02:00
petrbalvin 8dc1e98ca1 feat(asm): emit the arm64 stack-split guard and morestack block 2026-09-14 20:49:03 +02:00
petrbalvin 1d8e68c574 feat(asm): emit the amd64 stack-split guard and morestack block 2026-09-14 20:35:55 +02:00
petrbalvin 89fa6ea15e feat(lsp): resolve definition and references across open documents 2026-09-14 18:50:55 +02:00
petrbalvin edc20ffa97 feat(cmd): add gasm dis and share the decoder with the debugger 2026-09-14 18:47:08 +02:00
petrbalvin 50db6615b2 feat(cmd): add gofmt-style -l and -d modes to gasm fmt 2026-09-14 18:47:08 +02:00
petrbalvin 95f1d6f083 style: replace em dashes in the scaffold comments 2026-09-14 18:22:25 +02:00
petrbalvin f43e791e5a chore: add .qwen to the gitignore metadata block 2026-09-14 18:22:18 +02:00
petrbalvin 1691c81095 style: replace em and en dashes across sources 2026-09-14 18:22:18 +02:00
petrbalvin 2db563be07 refactor(cmd): consolidate cross-arch verify and drop dead code 2026-09-14 18:22:00 +02:00
petrbalvin 4f190ee1a2 refactor(debug): move watchpoint slot state into the session 2026-09-14 18:22:00 +02:00
petrbalvin 909f874797 fix(lsp): recover from handler panics and decode client uris 2026-09-14 18:22:00 +02:00
petrbalvin e307bf830f fix(lint): guard unnamed TEXT and refresh the textflag table 2026-09-14 18:22:00 +02:00
petrbalvin 953c258d6a fix(asm): make arm64 and loong64 relocations match the toolchain 2026-09-14 18:22:00 +02:00
petrbalvin c6f0286732 fix(asm): encode amd64 frame adjustments above 127 bytes with imm32 2026-09-14 18:22:00 +02:00
petrbalvin f5fc22d390 fix(parser): reject malformed TEXT frames and parse signed frame sizes 2026-09-14 18:22:00 +02:00
petrbalvin 22226d59a5 chore: prepare release v0.32.0
Release / build (amd64, linux) (push) Successful in 43s
Release / build (arm64, linux) (push) Successful in 42s
Release / build (loong64, linux) (push) Successful in 43s
Release / build (riscv64, linux) (push) Successful in 43s
Test / vet (push) Successful in 49s
Release / release (push) Successful in 18s
Test / test (push) Successful in 2m40s
Test / build (push) Successful in 42s
Assisted-by: GLM 5.3 Flash
2026-08-31 13:10:37 +02:00
petrbalvin a0ae0e4e37 docs: audit the development changelog against the release delta
Assisted-by: GLM 5.3 Flash
2026-08-31 12:53:36 +02:00
petrbalvin 94c4756d47 fix(verify): fix non-amd64 JIT trampolines and validate under qemu 2026-08-31 12:34:50 +02:00
petrbalvin a5a59d6503 fix(verify): gate JIT verification to amd64 until trampolines are hardened
Test / vet (push) Successful in 48s
Test / test (push) Successful in 2m34s
Test / build (push) Successful in 41s
2026-08-30 22:48:48 +02:00
petrbalvin 9cb1666b35 fix(debug): make ptrace sessions reliable on Go tracees 2026-08-30 22:35:22 +02:00
petrbalvin 96cc70731f docs: sync rule counts and feature lists with the new capabilities
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin b4da13d0f6 feat(lsp): pull diagnostics, include links and folding ranges
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin 41b387e54d feat(debug): instruction-level coverage and FP register display
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin 4171e412b5 feat(verify): save and replay fuzz corpora
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin 57c0ca8b09 feat(gasm): audit-instructions for arm64, riscv64 and loong64
Assisted-by: GLM 5.3 Flash
2026-08-30 21:42:12 +02:00
petrbalvin 75bd83fd52 feat(lint): flag writes to the platform-reserved register 2026-08-30 21:42:12 +02:00
petrbalvin 62f6fb4faf fix(debug): cross-compile for arm64, riscv64 and loong64
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m35s
Test / build (push) Successful in 40s
Assisted-by: GLM 5.3 Flash
2026-08-30 11:27:54 +02:00
petrbalvin 8f84dac10b feat(verify): ABI checks on arm64, riscv64 and loong64
Assisted-by: GLM 5.3 Flash
2026-08-30 11:27:54 +02:00
petrbalvin 6d7f10f13e refactor(cmd): re-enter child modes via environment instead of hidden flags
Assisted-by: GLM 5.3 Flash
2026-08-30 11:00:40 +02:00
petrbalvin 56f8babbce docs: sync README, CHANGELOG and docs with the current state 2026-08-30 10:44:18 +02:00
petrbalvin 6c1c8d9d96 fix: point the coverage gate at the format package 2026-08-30 10:43:55 +02:00
petrbalvin 93794c02f7 chore: untrack .idea files 2026-08-30 10:43:45 +02:00
petrbalvin 56ad158772 fix: restore iota blocks, asm --format flag and prose after the syntax pass
Test / vet (push) Successful in 48s
Test / test (push) Successful in 2m35s
Test / build (push) Successful in 40s
2026-08-29 17:12:53 +02:00
petrbalvin 15e8b88d32 style: modernize the new tooling code to match the repo conventions 2026-08-29 16:15:14 +02:00
petrbalvin 9beff4ae85 style: modernize to splitseq, cut, min, maps.copy and range-over-int 2026-08-29 16:04:32 +02:00
petrbalvin eacf33d0f7 fix: staticcheck and deadcode findings repo-wide, modernize counting loops 2026-08-29 15:25:15 +02:00
petrbalvin 9a34733615 style(parser): cutprefix, default case, comma and tokens rename 2026-08-29 15:14:49 +02:00
petrbalvin 970df7c32a style(lint): drop duplicate rule code declaration 2026-08-29 15:14:49 +02:00
petrbalvin 49572efe16 style: gofmt the audit command 2026-08-29 14:25:02 +02:00
petrbalvin 568c553986 docs: document the audit, scaffold, scalar-args and headless debug additions 2026-08-29 14:24:52 +02:00
petrbalvin 7a30e902fc feat(debug): label coverage report for headless runs 2026-08-29 14:17:26 +02:00
petrbalvin cb398d9498 feat(debug): headless script mode with timeout watchdog 2026-08-29 14:16:04 +02:00
petrbalvin e44162a749 feat(verify): scalar arguments for -call invocations 2026-08-29 14:03:45 +02:00
petrbalvin 6fb9629ab6 fix(gasm): scaffold shared param names and two-sided seed sets 2026-08-29 13:57:04 +02:00
petrbalvin 8bda4066e3 feat(gasm): audit-instructions command and scaffold generator 2026-08-29 13:52:32 +02:00
petrbalvin 685b150ecf feat(lint): flag table-known instructions the encoder cannot emit 2026-08-29 13:42:47 +02:00
petrbalvin c92e6bed3a feat(lint): nonportable amd64 register name rule 2026-08-29 13:39:22 +02:00
petrbalvin d75e6bcae6 feat(lint): abi0 register-args rule and go-asm width model 2026-08-29 13:36:28 +02:00
petrbalvin 6699ebd34f feat(asm): add prefetch hint encoding
Test / vet (push) Successful in 51s
Test / test (push) Successful in 2m37s
Test / build (push) Successful in 41s
2026-08-29 10:42:15 +02:00
petrbalvin ba4d961b20 fix(asm): ADDQ imm8 frame adjust for 128-255 byte frames 2026-08-28 22:11:24 +02:00
petrbalvin 78b12dd427 fix(asm): match go tool asm encodings and strictness 2026-08-28 19:55:25 +02:00
petrbalvin 19a26e049b feat(asm): add vpcmp compare, full opmask set, legacy sse integers and bswap
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m35s
Test / build (push) Successful in 41s
Assisted-by: GLM 5.3
2026-08-27 22:41:03 +02:00
petrbalvin 163480e283 feat(amd64): encode scalar/double conversion ops (CVTSS2SD/CVTSD2SS/CVTPS2PD/CVTPD2PS) 2026-08-27 22:02:11 +02:00
petrbalvin 0d62818db7 feat(amd64): encode legacy SSE binaries, imm8 shuffles and MOVQ xmm moves 2026-08-27 22:02:11 +02:00
petrbalvin be2ceaafb9 feat(amd64): encode legacy SSE packed binaries and imm8 shuffles
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m34s
Test / build (push) Successful in 41s
2026-08-27 17:15:10 +02:00
petrbalvin 941f7fa990 chore: set development version to 0.32.0-dev
Test / vet (push) Successful in 48s
Test / test (push) Successful in 3m2s
Test / build (push) Successful in 47s
Assisted-by: GLM 5.3
2026-08-24 21:03:30 +02:00
petrbalvin eb3c79c3c8 fix(verify): isolate smoke and abi sweeps in a child process
Assisted-by: GLM 5.3
2026-08-24 21:01:32 +02:00
petrbalvin eade875b53 fix(asm): compress movq immediates to the go-tool-asm imm32 forms
Assisted-by: GLM 5.3
2026-08-24 20:23:39 +02:00
petrbalvin b08005753e fix(asm): encode BSF, BSR and POPCNT
Assisted-by: GLM 5.3
2026-08-24 20:19:34 +02:00
petrbalvin 5e06d6a6aa feat(lint): add register-width-mismatch rule
Test / vet (push) Successful in 48s
Test / test (push) Successful in 2m31s
Test / build (push) Successful in 44s
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:38 +02:00
petrbalvin e4c9d78968 feat(lsp): add workspace symbol search
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:35 +02:00
petrbalvin f5fcaf9fa6 perf(verify): parallelize smoke and ABI checks
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:31 +02:00
petrbalvin 94b98468bb feat(debug): support register-register conditional breakpoints
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:25 +02:00
petrbalvin 746cef8100 feat(asm): add .debug_frame CFI section for stack unwinding
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:21 +02:00
petrbalvin f153be8158 feat(lsp): add code actions, signature help, and document highlights
Assisted-by: MiMo V2.5 Pro
2026-08-21 01:20:14 +02:00
petrbalvin 1bf95a169e fix: resolve audit findings — stale text, dead code, build tags, docs
Test / vet (push) Successful in 47s
Test / test (push) Successful in 2m39s
Test / build (push) Successful in 42s
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:50:31 +02:00
petrbalvin f9eb4021d6 docs: update CHANGELOG and README for development changes
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:39:38 +02:00
petrbalvin c4930438fd refactor(lsp): use format package for document formatting
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin f372e2db75 test(lsp): add tests for references, rename, formatting, inlay hints
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin ac02c83a86 feat(lint): add stack-imbalance rule
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin ce5ec24fa8 feat(asm): integrate DWARF5 sections into all ELF emitters
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 181d8e508c feat(verify): add Call trampolines for arm64, riscv64, loong64
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 1160c96427 chore: remove stale debug_linux_amd64.go
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin de9e211ff1 feat(debug): multi-architecture debugger support for arm64, riscv64, loong64
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 3acbdd6533 feat(asm): add DWARF5 debug info generation for ELF output
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin ba502c9b79 refactor(debug): make Regs and breakpoint arch-neutral for arm64
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 874e054ecb feat(lsp): add references, rename, formatting, and inlay hints
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin f1960febdc feat(lint): add unused-label and invalid-textflag rules
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin ae550cc05a fix(asm): add cross-package GOOBJ resolution for riscv64, loong64, arm64
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin cc50035375 fix(cmd): update asm help text to list arm64 as supported
Assisted-by: MiMo V2.5 Pro
2026-08-21 00:35:21 +02:00
petrbalvin 8054fff9ac chore: prepare release v0.31.1
Test / vet (push) Successful in 47s
Release / build (amd64, linux) (push) Successful in 55s
Release / build (arm64, linux) (push) Successful in 51s
Release / build (loong64, linux) (push) Successful in 46s
Release / build (riscv64, linux) (push) Successful in 42s
Test / test (push) Successful in 2m29s
Release / release (push) Successful in 19s
Test / build (push) Successful in 42s
Assisted-by: MiMo V2.5 Pro
2026-08-20 22:35:11 +02:00
petrbalvin 6a79c35bf7 fix(version): bump version to 0.31.0 in justfile and main.go
Assisted-by: MiMo V2.5 Pro
2026-08-20 22:35:11 +02:00
petrbalvin 6f4f2096e9 chore: prepare release v0.31.0
Release / build (amd64, linux) (push) Successful in 42s
Release / build (arm64, linux) (push) Successful in 40s
Release / build (loong64, linux) (push) Successful in 42s
Release / build (riscv64, linux) (push) Successful in 45s
Release / release (push) Successful in 18s
2026-08-20 16:24:33 +02:00
petrbalvin 56630f8624 chore(toolchain): upgrade to Go 1.27
Test / vet (push) Successful in 1m5s
Test / test (push) Successful in 2m33s
Test / build (push) Successful in 40s
2026-08-20 16:03:26 +02:00
petrbalvin 459f4a2b6e fix(test): add arm64 encoding tests for Go 1.26 coverage compatibility
Assisted-by: MiMo V2.5 Pro
2026-08-20 15:44:23 +02:00
petrbalvin 5d66343488 fix(goobj): make R_DWTXTADDR_U4 relocation type Go-version-aware
The relocation type number shifted between Go 1.26 (103) and Go 1.27
(106)
because new LoongArch relocations were inserted. Detect the Go version
at
runtime and use the correct value.
2026-08-20 15:35:39 +02:00
petrbalvin 48334c4d5a docs: remove completed roadmap phases, fix licence description 2026-08-20 15:01:57 +02:00
petrbalvin 7629963cab chore: fix project conventions — .gitignore, CHANGELOG categories, docs naming
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:47:39 +02:00
petrbalvin 97951cbeb6 feat(asm): extend arm64 encoder with atomics, bitfield, SIMD and more test kernels
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:31:15 +02:00
petrbalvin 6e73f59e78 feat(asm): extend arm64 encoder with FP, conditional select, CRC32 and tests
Assisted-by: MiMo V2.5 Pro
2026-08-20 14:07:12 +02:00
petrbalvin 4221ec5741 feat(asm): add AArch64 arm64 encoder with ground-truth verification
Assisted-by: MiMo V2.5 Pro
2026-08-20 13:33:39 +02:00
petrbalvin 01dcc3b86e revert(toolchain): restore go1.26 in CI
Test / vet (push) Successful in 43s
Test / test (push) Failing after 1m55s
Test / build (push) Skipped
Release / build (amd64, linux) (push) Successful in 40s
Release / build (arm64, linux) (push) Successful in 37s
Release / build (loong64, linux) (push) Successful in 38s
Release / build (riscv64, linux) (push) Successful in 58s
Release / release (push) Successful in 22s
2026-08-13 18:34:20 +02:00
petrbalvin b08885bd31 chore(release): prepare v0.30.0
Test / vet (push) Failing after 17s
Test / test (push) Skipped
Test / build (push) Skipped
Assisted-by: DeepSeek V4 Pro
2026-08-13 18:27:12 +02:00
petrbalvin c05c53452f fix(asm): encode RISC-V CALL sym(SB) as JAL
Test / vet (push) Successful in 47s
Test / test (push) Failing after 2m9s
Test / build (push) Skipped
Assisted-by: DeepSeek V4 Pro
2026-08-13 18:12:22 +02:00
petrbalvin 681a449c01 fix(asm): match RISC-V branch and jump encodings 2026-08-13 17:57:10 +02:00
petrbalvin 31a2cee382 fix(asm): materialise RISC-V MOV immediates
Test / vet (push) Successful in 44s
Test / test (push) Failing after 1m56s
Test / build (push) Skipped
2026-08-13 17:41:16 +02:00
petrbalvin 3bc7c18bc3 fix(asm): materialise large RISC-V immediates
Assisted-by: DeepSeek V4 Pro
2026-08-13 16:04:08 +02:00
petrbalvin f0512a4e1c fix(asm): complete RISC-V compressed loads/stores and word arithmetic
Assisted-by: DeepSeek V4 Pro
2026-08-13 15:42:38 +02:00
petrbalvin 373c09f725 fix(asm): correct RISC-V operand order and complete RVC compression
Assisted-by: DeepSeek V4 Pro
2026-08-13 15:13:31 +02:00
petrbalvin b78b6c5004 fix(asm): correct RISC-V frame layout and RVC encodings
Assisted-by: GLM 5.2
2026-08-13 14:41:57 +02:00
petrbalvin 9b5c878f9e feat(asm): emit RISC-V GOOBJ with the shared emitter
Assisted-by: DeepSeek V4 Pro
2026-08-13 12:07:29 +02:00
petrbalvin 31ee8e7941 feat(asm): add LoongArch encoder with ELF and GOOBJ emission
Assisted-by: DeepSeek V4 Pro
2026-08-13 11:24:44 +02:00
petrbalvin 2d1176e045 feat(asm): resolve external GOOBJ symbols from archive data
Test / vet (push) Successful in 1m2s
Test / test (push) Failing after 2m5s
Test / build (push) Skipped
2026-08-08 16:18:56 +02:00
petrbalvin 2a27a3a52b docs: add BSD-3-Clause headers to generated files and update CI docs 2026-08-07 22:43:40 +02:00
petrbalvin cde7d0f96a docs: document watchpoint slot tracking and update debugger commands 2026-08-07 22:27:49 +02:00
petrbalvin d7ee1b78d4 feat: drop Mach-O and macOS support, Linux-only 2026-08-07 22:20:26 +02:00
petrbalvin 30c53565a7 build: align just test coverage gate with CI package list
Test / vet (push) Successful in 43s
Release / build (amd64, linux) (push) Successful in 41s
Release / build (arm64, linux) (push) Successful in 38s
Release / build (loong64, linux) (push) Successful in 39s
Release / build (riscv64, linux) (push) Successful in 38s
Test / test (push) Successful in 1m58s
Release / release (push) Successful in 18s
Test / build (push) Successful in 37s
Assisted-by: GLM 5.2
2026-08-07 21:34:09 +02:00
petrbalvin ebd8ab8a3c test: gate go-libraries integration tests behind -tags=integration
Test / vet (push) Successful in 44s
Test / test (push) Successful in 2m8s
Test / build (push) Successful in 40s
Assisted-by: GLM 5.2
2026-08-07 21:28:24 +02:00
petrbalvin 176d856f67 docs: remove go-libraries kernel references from README and CHANGELOG
Test / vet (push) Successful in 45s
Test / test (push) Failing after 2m1s
Test / build (push) Skipped
2026-08-07 21:10:16 +02:00
petrbalvin eace06bbd6 fix(verify): remove all go-libraries kernel dependencies from tests 2026-08-07 21:06:14 +02:00
petrbalvin f97bea61c5 test(verify): add FuzzResult.String and arrayLen tests to lift coverage over 80%
Test / vet (push) Successful in 49s
Test / test (push) Failing after 2m14s
Test / build (push) Skipped
Assisted-by: GLM 5.2
2026-08-05 23:16:52 +02:00
petrbalvin 19b37569c0 fix(verify): fix flaky JIT tests with global buffers and KeepAlive
Test / vet (push) Successful in 49s
Test / test (push) Failing after 2m9s
Test / build (push) Skipped
2026-08-05 23:09:54 +02:00
petrbalvin 23b3d3e152 docs: fold development changes into v0.29.0, document --call/--buf/--map 2026-08-05 21:15:05 +02:00
petrbalvin b0c62be8ce feat(verify): add --call and --buf flags for single-function invocation
Assisted-by: GLM 5.2
2026-08-05 20:46:39 +02:00
petrbalvin ece0d3f127 feat(cli): add --map flag to diff for comparing differently-named
functions
2026-08-05 20:20:47 +02:00
petrbalvin ad6e3360df feat: release v0.29.0 with RISC-V GOOBJ, debugger enhancements, and new
Test / vet (push) Successful in 47s
Test / test (push) Failing after 2m4s
Test / build (push) Skipped
CLI commands
2026-08-05 20:08:56 +02:00
petrbalvin 49566de7fb feat(asm): add did-you-mean suggestions for undefined labels
Assisted-by: DeepSeek V4 Pro
2026-08-05 18:52:00 +02:00
petrbalvin 6228d77566 feat(cli): add gasm profile command for basic-block structure
Assisted-by: DeepSeek V4 Pro
2026-08-05 16:41:00 +02:00
petrbalvin c0e280ee3c feat(cli): add gasm diff command for comparing assembly encodings
Assisted-by: DeepSeek V4 Pro
2026-08-05 14:23:00 +02:00
petrbalvin 2d45dbf7ff feat(lsp): add go-to-definition for labels 2026-08-05 11:08:00 +02:00
petrbalvin 8ddac0135e feat(asm): add GOOBJ emission for RISC-V 2026-08-05 09:27:00 +02:00
petrbalvin f58a4fe51d feat(debug): add named buffer allocation with pattern filling
Test / vet (push) Successful in 45s
Test / test (push) Successful in 1m58s
Test / build (push) Successful in 42s
2026-08-04 23:55:25 +02:00
petrbalvin 5cb7e3e231 feat(verify): combine ABI checks with fuzzing for deep-path testing 2026-08-04 22:26:04 +02:00
petrbalvin 36bbc0c13b feat(verify): store crashing input in FuzzResult for reproducibility
Test / vet (push) Successful in 44s
Test / test (push) Successful in 1m59s
Test / build (push) Successful in 36s
Assisted-by: DeepSeek V4 Pro
2026-08-04 21:58:53 +02:00
petrbalvin 32afa3449f feat(debug): add YMM vector register display via PTRACE_GETFPREGS
Assisted-by: DeepSeek V4 Pro
2026-08-04 21:55:28 +02:00
petrbalvin 2ab6b9eb84 ci: add Gitea CI workflows and release pipeline
Test / vet (push) Successful in 44s
Release / build (amd64, linux) (push) Successful in 39s
Release / build (arm64, linux) (push) Successful in 38s
Release / build (loong64, linux) (push) Successful in 39s
Release / build (riscv64, linux) (push) Successful in 39s
Test / test (push) Successful in 1m58s
Release / release (push) Successful in 18s
Test / build (push) Successful in 37s
Assisted-by: DeepSeek V4 Pro
2026-08-03 19:42:05 +02:00
petrbalvin f41a86b660 feat(asm): add RISC-V ELF relocatable object emission and SB relocation
support
2026-08-03 08:51:00 +02:00
petrbalvin 7721353d44 feat(asm): add RVC compression for branches, arithmetic, and FP
Assisted-by: DeepSeek V4 Pro
2026-08-03 01:08:00 +02:00
petrbalvin 243b087116 feat(riscv): add MOV pseudo-instruction and RVC compressed encoding
Assisted-by: DeepSeek V4 Pro
2026-08-02 18:22:00 +02:00
petrbalvin eee7a6d4a4 feat(riscv): add RISC-V RV64A, FP and CSR instruction support
Assisted-by: Kimi K3
2026-08-02 11:45:00 +02:00
petrbalvin 801fb963c9 fix(verify): enlarge fuzz buffers so slice-based kernels can be fuzzed
directly
2026-08-02 06:12:00 +02:00
petrbalvin a2acc9b5a3 feat(riscv): add prologue, epilogue and frame pseudo-register support
Assisted-by: Kimi K3
2026-08-02 00:18:00 +02:00
petrbalvin f860bf8ce6 feat(asm): add RISC-V encoder with RV64I/RV64M instruction formats
Assisted-by: Kimi K3
2026-08-01 19:51:00 +02:00
petrbalvin e7df5e5225 test(debug): add unit tests for condition evaluation, line lookup and
RFLAGS decoding

Assisted-by: MiniMax M3
2026-08-01 14:33:00 +02:00
petrbalvin d114b3412c feat(debug): complete the interactive debugger with disassembly, breakpoints, watchpoints and execution control
Assisted-by: DeepSeek V4 Pro
2026-08-01 09:47:00 +02:00
petrbalvin c77d68018c docs: add the full documentation surface — AGENTS, CONTRIBUTING, cli and development references
Assisted-by: DeepSeek V4 Flash
2026-08-01 05:22:00 +02:00
petrbalvin 89d633f4bb feat(debug): add interactive ptrace debugger MVP — single-step, regs, breakpoints, labels
Assisted-by: Qwen 3.8 Max Preview
2026-08-01 02:34:00 +02:00
petrbalvin f20e0bf1e7 fix(verify): subprocess isolation for --fuzz, partial functions report CRASH gracefully
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin 1a45b66139 feat(verify): add universal --fuzz differential testing driven by // func signatures
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin 382efe538a feat(verify): add universal --ground-truth verification against go tool asm
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin f8d28a42ba feat(verify): complete the analyze family and add stereo16 differential tests
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin 51a2854d7f feat(verify): add analyzeO2/Res and decodeMono24 differential tests
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin c234c3dd5b feat(verify): add analyzeO1Range and fastStereoSums differential tests
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin 9262990ce5 feat(verify): extend differential tests to go-flac and AVX-512, add --abi/--profile CLI flags
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
petrbalvin d9f6167a4d feat(verify): add basic-block enumeration and path-diversity profiling
Assisted-by: Qwen 3.8 Max Preview
2026-08-02 23:11:30 +02:00
280 changed files with 41762 additions and 2452 deletions
+37
View File
@@ -0,0 +1,37 @@
# Race, Go. Dispatched by hand, and never a gate on a push or a tag: the release tag is
# cut only after `just gates` has already raced the tree, so this workflow is the
# explicit second opinion, not a step of the release.
#
# The race detector roughly doubles both time and memory, which the shared runner box
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
# task; here it is a decision rather than a routine.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Race
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
race:
runs-on: fedora
timeout-minutes: 20
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install gcc
# The race detector needs cgo and the runner image carries no C compiler.
run: dnf install -y gcc
- name: Race
run: go test -race -count=1 -timeout 10m ./...
+350
View File
@@ -0,0 +1,350 @@
# Release, Go binaries. Runs on version tags (v1.2.3) pushed to main.
#
# The module sits at the repository root: the toolchain records a version only for a root
# module, measured on go1.27.1, so a build of a module in a subdirectory reports (devel)
# even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never pass for
# it. A Go repository is one module at the root.
#
# The version contract these steps implement: nothing is injected. The toolchain records
# the tag into the binary's build information, so the build simply has to happen at the
# tag, which the trigger guarantees.
#
# The gates run in their own job, once, before the matrix, minus the race detector: race
# never runs on a push path or a tag, and the local gate raced this tree before the tag
# was cut. Putting the gates inside the matrix would run the whole suite once per target
# on the box that also hosts the forge. Each job validates the tag for itself rather than
# passing a value between jobs, so no workflow feature has to be trusted for the version
# to reach the file name.
name: Release
on:
push:
tags: ["v*"]
env:
# The box is shared with the forge, so parallelism is bounded on purpose. The gates job
# needs it most; the build jobs inherit it for their parallel compilation.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
gates:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install Perl
# Perl for the steps below. The install is a no-op where the package
# is already present.
run: dnf install -y perl
- name: Validate the tag
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
print qq{tag $v\n};
'
- name: Security policy names this release
# The supported-versions table is the one part of SECURITY.md that
# carries a version, so it goes stale the moment a tag is cut. Fail
# here rather than publish a policy naming the previous release.
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
(my $nv = $v) =~ s/^v//;
open(my $f, q{<}, q{SECURITY.md}) or die qq{SECURITY.md: $!\n};
local $/;
my $t = <$f>;
close $f;
$t =~ m{^\|\s*\Q$nv\E\s*\|\s*yes\s*\|}m
or die qq{ERROR: SECURITY.md does not name $nv as supported; update the table before releasing.\n};
print qq{SECURITY.md names $nv\n};
'
- name: Build
run: go build ./...
- name: Format
run: |
perl -e '
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
'
- name: Vet
run: go vet ./...
- name: Modernise
run: go fix -diff ./...
- name: Tests
# The same command as in test.yml, so the floor is the same number everywhere.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The same command as in test.yml: the CLI's exit codes and manual-page guard,
# and the debugger's architecture-neutral units, run outside the floor.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Coverage floor
run: |
perl -e '
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
'
build:
runs-on: fedora
timeout-minutes: 25
needs: gates
strategy:
fail-fast: false
matrix:
# Portable targets: amd64, arm64, loong64 and riscv64 on Linux, at the toolchain
# default level. No 32-bit, no wasm, no macOS, no Windows. FreeBSD stays out until
# verify/jit.go ports off syscall.Mprotect: the Go syscall package defines no
# Mprotect for freebsd, and verify/jit.go:50 calls it to drop the write bit from
# the JIT mapping, so every freebsd target fails to build with "undefined:
# syscall.Mprotect" (verified for amd64, arm64 and riscv64 on go1.27.1).
include:
- goos: linux
goarch: amd64
- goos: linux
goarch: arm64
- goos: linux
goarch: loong64
- goos: linux
goarch: riscv64
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install Perl
run: dnf install -y perl
- name: Validate the tag
id: version
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
(my $nv = $v) =~ s{^v}{};
open(my $o, q{>>}, $ENV{GITEA_OUTPUT}) or die qq{GITEA_OUTPUT: $!};
print $o qq{version_no_v=$nv\n};
close($o);
print qq{version $nv\n};
'
- name: Build
env:
VERSION_NO_V: ${{ steps.version.outputs.version_no_v }}
GOOS: ${{ matrix.goos }}
GOARCH: ${{ matrix.goarch }}
CGO_ENABLED: "0"
run: |
# Nothing is injected. The toolchain records the tag into the binary's build
# information, so the version is right because this build happens at the tag, and
# there is no path for anyone to get wrong. -s -w only strips symbols.
go build -ldflags "-s -w" -o "bin/gasm-${VERSION_NO_V}-${GOOS}-${GOARCH}" ./cmd/gasm
# Artifacts stay on v3: v4 and later detect Gitea as GHES and abort.
- name: Upload artifact
uses: actions/upload-artifact@v3
with:
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
path: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
if-no-files-found: error
- name: Smoke test
# Only a binary matching the runner can be run here. The check is not that --version
# exits cleanly but that it reports the tag and nothing more: a build outside version
# control reports (devel), and a build whose tree was dirty reports +dirty, and both
# would otherwise be published.
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
env:
TAG: ${{ gitea.ref_name }}
BIN: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
run: |
perl -e '
my $want = $ENV{TAG} // die qq{ERROR: no tag\n};
open(my $bin, q{-|}, $ENV{BIN}, q{--version}) or die qq{$ENV{BIN}: $!};
my $got = <$bin>;
close($bin);
$got = defined $got ? $got : q{};
chomp $got;
index($got, $want) >= 0
or die qq{ERROR: the binary printed "$got", which does not contain $want. Version control was disabled, so there is no recorded version.\n};
index($got, q{+dirty}) < 0
or die qq{ERROR: the binary printed "$got". The tree was dirty at build time, which means the checkout was not the tag, or the build artefacts are not ignored.\n};
print qq{$ENV{BIN} reports $got\n};
'
release:
runs-on: fedora
timeout-minutes: 15
needs: build
permissions:
# contents: read is required for the checkout: a job that declares any
# permissions gets a token scoped to exactly those, and releases: write
# alone leaves the fetch with no read access, which Gitea answers with
# a 404 "Repository not found". Verified on the instance 2026-09-16.
contents: read
releases: write
steps:
- uses: actions/checkout@v7
- name: Download all artifacts
uses: actions/download-artifact@v3
with:
path: dist
- name: Install Perl
run: dnf install -y perl
- name: Extract the CHANGELOG section
env:
VERSION: ${{ gitea.ref_name }}
run: |
# Each step derives what it needs from the tag, so no value has to travel between
# jobs.
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ s{^v}{};
open(my $vout, q{>}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
print $vout $v;
close($vout);
open(my $in, q{<}, q{CHANGELOG.md}) or die qq{CHANGELOG.md: $!};
my @lines = <$in>;
close($in);
my ($start, $end) = (-1, scalar @lines);
for my $i (0 .. $#lines) {
if ($start < 0) { $start = $i if $lines[$i] =~ m{^##\s+\[\Q$v\E\]} }
elsif ($lines[$i] =~ m{^##\s+\[}) { $end = $i; last }
}
$start >= 0 or die qq{ERROR: no CHANGELOG section for $v, expected a heading like: ## [$v] - YYYY-MM-DD\n};
my @body = grep { m{\S} } @lines[$start + 1 .. $end - 1];
@body or die qq{ERROR: the CHANGELOG section for $v is empty\n};
open(my $out, q{>}, q{release-body.md}) or die qq{release-body.md: $!};
print $out @body;
close($out);
printf qq{notes for %s: %d lines\n}, $v, scalar @body;
'
- name: Build the release request
run: |
perl -e '
open(my $vin, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
my $v = <$vin>;
close($vin);
chomp $v;
open(my $in, q{<:raw}, q{release-body.md}) or die qq{release-body.md: $!};
my $body = do { local $/; <$in> };
close($in);
# Byte-oriented escaping: JSON is UTF-8, so non-ASCII passes through and only the
# characters JSON forbids are rewritten.
$body =~ s/([\\"])/\\$1/g;
$body =~ s/\t/\\t/g;
$body =~ s/\r//g;
$body =~ s/\n/\\n/g;
$body =~ s/([\x00-\x08\x0b\x0c\x0e-\x1f])/sprintf(q{\u%04x}, ord($1))/ge;
my $json = sprintf(qq{{"tag_name":"v%s","name":"v%s","body":"%s","draft":false,"prerelease":false}}, $v, $v, $body);
open(my $out, q{>}, q{release.json}) or die qq{release.json: $!};
print $out $json;
close($out);
print qq{release.json written for v$v\n};
'
- name: Create the release
env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
GITEA_SERVER_URL: ${{ gitea.server_url }}
GITEA_REPOSITORY: ${{ gitea.repository }}
run: |
perl -e '
my @cmd = (q{curl}, q{-sS}, q{-o}, q{response.json}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/json},
q{-X}, q{POST},
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases},
q{--data-binary}, q{@release.json});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>;
my $ok = close($curl);
my $exit = $? >> 8;
$code = defined $code ? $code : q{};
$ok or die qq{ERROR: curl failed (exit $exit) calling $ENV{GITEA_SERVER_URL}\n};
open(my $r, q{<:raw}, q{response.json}) or die qq{response.json: $!};
my $body = do { local $/; <$r> };
close($r);
$code eq q{201} or die qq{ERROR: the release was not created, HTTP $code: $body\n};
$body =~ m{"id"\s*:\s*([0-9]+)} or die qq{ERROR: no release id in the response: $body\n};
open(my $o, q{>}, q{release-id.txt}) or die qq{release-id.txt: $!};
print $o $1;
close($o);
print qq{release id $1\n};
'
- name: Upload assets
env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
GITEA_SERVER_URL: ${{ gitea.server_url }}
GITEA_REPOSITORY: ${{ gitea.repository }}
run: |
perl -e '
open(my $f, q{<}, q{release-id.txt}) or die qq{release-id.txt: $!};
my $id = <$f>;
close($f);
chomp $id;
my @files = grep { -f $_ } glob(q{dist/*/*});
@files or die qq{ERROR: no assets under dist/\n};
my $bad = 0;
for my $path (@files) {
(my $name = $path) =~ s{.*/}{};
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/octet-stream},
q{-X}, q{POST}, q{--data-binary}, qq{@$path},
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>;
my $ok = close($curl);
my $exit = $? >> 8;
$code = defined $code ? $code : q{};
unless ($ok) {
printf qq{%s: curl failed (exit %d)\n}, $name, $exit;
$bad = 1;
next;
}
printf qq{%s: HTTP %s\n}, $name, $code;
$bad = 1 if $code ne q{201};
}
exit($bad ? 1 : 0);
'
+110
View File
@@ -0,0 +1,110 @@
# Test, Go. Push and pull request to development. Never on main.
#
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
# runner box cannot afford the race detector on every push, so it lives in race.yml.
# The box is one core and 2 GB beside Gitea, so parallelism is bounded on purpose and
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
# the dependency download three times without buying any parallelism.
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
# on ash, and it is one language instead of two. The Perl uses builtins only, because
# Fedora packages the Perl modules separately and nothing beyond `perl` itself may be
# assumed present.
name: Test
on:
push:
branches: [development]
pull_request:
branches: [development]
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
# A superseded run of the same ref is cancelled instead of queueing behind one that
# no longer matters. Verified on Gitea 1.27.1 on 2026-09-17: a queued run whose ref
# moved on is cancelled before it ever reaches the runner, while a run already
# dispatched there runs to completion.
concurrency:
group: ${{ gitea.workflow }}-${{ gitea.ref }}
cancel-in-progress: true
jobs:
test:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
cache: true
- name: Install Perl
# The runner images are minimal and Perl is not guaranteed. The install is a
# no-op where it is already present; drop this step once verified on the box.
run: dnf install -y perl
# The steps follow the `gates` order of the justfile contract: build, format,
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
- name: Build
run: go build ./...
- name: Format
run: |
perl -e '
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
'
- name: Vet
run: go vet ./...
- name: Modernise
# Exits non-zero when it has something to rewrite, so it needs no output capture.
run: go fix -diff ./...
- name: Tests
# The suite must be fast: a push pipeline that cannot finish in a few minutes moves
# its heavy part behind a dispatch. The inner timeout matches the job's, so a
# hanging test reports its own goroutine dump rather than a silent job kill.
# The pattern is `packages` in the project's justfile: the logic packages, since a
# thin cmd/ would drag the total under the floor. release.yml runs the same
# command, so the floor is the same number everywhere. ./verify/... carries the
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth
# suites); the runner's Go setup provides both the tool and GOROOT.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The CLI and the debugger sit outside `packages` because a thin main and a
# ptrace-bound package pull the total under the floor, but their tests guard
# shipped surfaces: the command exit codes, the manual pages against the
# binary's own help, and the debugger's architecture-neutral units. They run
# here so the floor stays a product measure and nothing is left untested.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Oracle parity
# Re-run the live go-tool-asm comparison as its own step so that a parity
# regression names the gate that failed instead of hiding inside the suite.
run: go test -count=1 -timeout 10m -run 'TestGroundTruth' ./verify/...
- name: Coverage floor
run: |
perl -e '
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
'
+9 -8
View File
@@ -1,12 +1,13 @@
# Binaries
/gasm
/bin/
*.exe
.idea/
.zcode/
# Test and coverage artefacts
# Build output
/bin/
/gasm
coverage.out
*.test
# Editor detritus
*.swp
.DS_Store
# Crash dumps from the emulator runs
core
core.*
*.core
+1429
View File
File diff suppressed because it is too large Load Diff
+134
View File
@@ -0,0 +1,134 @@
# Contributing
Contributions to **gasm-devkit** are governed by the Contributor terms
below; submitting one means you accept them.
## Contributor terms
1. This project belongs to its owner alone. The owner decides what is
accepted, in what form and when; the decision is final and needs no
justification.
2. By submitting a contribution you assign to Petr Balvín
<opensource@petrbalvin.org> all present and future copyright and
related rights in it, worldwide, for the full term of the rights,
with the right to relicense and sublicense without restriction,
including under proprietary terms.
3. Where that assignment is not effective, it counts as a perpetual,
irrevocable, royalty-free licence with the same scope.
4. To the fullest extent permitted by law, you waive any right of
attribution and integrity in the contribution. The project names no
contributors and keeps no credits list.
5. By submitting you represent that the work is yours and that you
hold the rights to assign it as above.
## Development setup
Requirements: Go 1.27.1, the exact version the `go` directive in `go.mod`
declares, [just](https://github.com/casey/just) for the recipes, and a C
compiler (gcc), because `just gates` includes `just race` and the race
detector needs cgo.
```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
cd gasm-devkit
just build
just gates
```
## Workflow
1. Branch from `development`. Never commit directly to `main`, which is release-only.
2. Commit in [Conventional Commits](https://www.conventionalcommits.org/) form:
`type(scope): description`, subject line only, imperative mood, lowercase after the
colon, no trailing full stop. Allowed types: `feat`, `fix`, `docs`, `style`,
`refactor`, `perf`, `test`, `chore`, `ci`, `build`, `revert`.
3. One logical change per commit. A refactor, a behaviour change and a formatting pass
are three commits, never one.
4. Record every user-visible change in `CHANGELOG.md` under `## [development]`.
5. Add or update tests. Coverage stays at 80 percent or more; it is a hard gate.
6. Update the documentation when the public API, the configuration or the behaviour
changes.
7. Open a pull request against `development`.
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The release
workflow builds the assets and publishes the release and its notes.
## Code style
`gofmt` and `go vet` run through `just fmt` and `just vet`, with zero diff and zero
warnings tolerated. `just vet` is two gates, `go vet ./...` and `go fix -diff ./...`,
so the modernisation rewrites are enforced too. `just gates` is the definition of done in
one command, and the recipe file names what it contains. Errors are checked explicitly,
wrapped as `fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. The
recipe file holds the commands, and the language and standard-library surface is the one
the `go` directive in `go.mod` pins.
- `golang.org/x/arch` is the one module dependency, and it is linked into the binary:
`gasm dis` and the debugger's listings decode through it. Everything else is the
standard library.
- No cgo and no C. The standalone encoder paths (`gasm asm --format raw` and `--format
elf`) need no Go installation; `gasm verify --ground-truth`, `gasm verify --fuzz`,
`gasm audit-instructions` and `gasm asm --format goobj` resolve through the installed
Go toolchain.
- The parser, lexer and formatter are hand-written; the `arch` instruction tables are
generated only by `_gen/gen.go` (`just gen`) and never edited by hand.
- Assembly committed to the repository goes through `gasm fmt` and `gasm lint`, so a
`.s` file that `gasm fmt -l .` lists is unfinished.
New source files open with the project's two-line licence header, whose SPDX
identifier matches `LICENSE`. Configuration files, workflows and dotfiles do not carry
it.
## AI contribution policy
AI tools are welcome as productivity aids and are a normal part of modern software
development. What matters is that the contribution stays understandable, reviewable and
genuinely useful.
- **Disclose the assistance.** If AI helped draft any part of a commit, issue, pull
request or review, say so.
- **Commit messages carry exactly one trailer**, as a git trailer on the line after a
blank line that closes the subject:
```
Assisted-by: MODEL
```
Name the model that did the work, spelled the way its maker spells it, for example
`GLM 5.3`, `DeepSeek V4.1 Flash` or `Qwen 3.8 Flash`. No `Co-Authored-By`, no `Signed-off-by`,
no other trailers, and no prose: the trailer is the disclosure.
- **Issues and pull requests** attribute the assistance in a comment, for example
`_Assisted-by: GLM 5.3_`. It does not belong in the pull request description.
- **Take responsibility.** You are accountable for the accuracy, completeness and
intent of everything you submit, whether or not AI produced it.
- **Review before marking ready.** Read the diff carefully, run it locally, and add the
tests it needs. Do not mark a pull request ready until you can defend every change in
it.
- **Quality over quantity.** Contributions that look like un-reviewed output, or whose
author cannot engage substantively during review, may be closed.
- **Preferred models.** Prefer open-weight models with transparent training data and
minimal output filtering.
AI assists. It does not replace judgement.
## Continuous integration
Workflows live in `.gitea/workflows/` and run on the project's own runners:
| Workflow | Trigger | What it does |
|---|---|---|
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor, the CLI and debugger tests outside the profile, then the oracle-parity rerun against `go tool asm` |
| Release | a `v*` tag | the same gates as Test minus the oracle-parity step, then the matrix build, the version smoke test and the release itself; the race detector runs locally in `just gates` before the tag is cut |
The local equivalent is `just gates`, which is the same set plus the race detector. The
race detector also has its own workflow, dispatched by hand; it never runs on a push or a
tag, where it would double the time and the memory a shared runner cannot spare.
## Reporting bugs
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the
version, the operating system and architecture, the exact command, the full output,
and the expected against the actual behaviour.
**Security issues do not go in the issue tracker.** Report them as
[SECURITY.md](SECURITY.md) describes, to **opensource@petrbalvin.org**.
+291
View File
@@ -0,0 +1,291 @@
# Plan 9 assembly tooling, inside and outside Go
> **Warning: this is an experiment.** gasm-devkit is under active
> development and is not stable. The version is 0.x.x: commands, flags,
> output formats and behaviour can change without warning at any time.
> A 1.0.0 release is light years away. Nothing in this document is a
> stability promise. For all of that, this is not a paper project: gasm
> is already in active use and is tested on real assembly work. Only
> amd64 is validated on real hardware; the other three architectures run
> under emulation ([Validation status](#validation-status)).
**GAsm** is Go's Plan 9 assembler, and Go ships it without tooling:
there is no formatter, no linter and no debugger for `.s` files, and no
assembler that works without a Go installation. Developers write
assembly blind, validate it by benchmark, and debug it by print
statement. gasm-devkit is the missing toolkit: a single, self-contained
binary, `gasm`, that serves both purposes.
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
dynamic verification, a source-level debugger and a language server,
for `.s` files in Go programs.
- **Use Plan 9 assembly outside the Go toolchain.** `gasm asm` encodes
on its own and writes raw images or linkable ELF objects with DWARF5
debug sections, with no Go installation in the loop; the Go
toolchain's own GOOBJ format, which `go build` consumes in place of
the toolchain's output, needs the installed toolchain.
## Why Plan 9 assembly
Plan 9 assembly is the quiet triumph of the field. One syntax across
every architecture Go builds for: the same source-first operand order,
the same four pseudo-registers, the same frame convention, whether the
target is x86, ARM, RISC-V or LoongArch. Learn it once and you can
read a kernel on any of them.
Compare the alternatives. Intel syntax and AT&T syntax disagree on the
one question every instruction answers, which operand is the source
and which is the destination, so half the world writes it one way,
half the other, and every assembly programmer carries both in their
head forever. GNU as settles the argument with directives that switch
dialects mid-file (`.intel_syntax noprefix`), a percent sign on every
register and a dollar on every immediate: punctuation that carries
nothing the operand order did not already say. And the x86 family
fragments again underneath: NASM is not MASM is not GAS, each with its
own directive zoo and macro language, so every project picks a dialect
and every reader learns a different one by accident.
Plan 9 assembly has none of it. Registers are bare names. Memory is
one notation, `offset(base)`, extended by an index and a scale when
the instruction needs it. Arguments arrive named and offset-checked:
`x+0(FP)` is the argument x, on every architecture, and `go vet`
polices the offsets against the Go prototype.
```text
AT&T (GNU as): movq %rax, -16(%rbp)
Plan 9 (Go): MOVQ AX, total-16(SP)
```
The same lines, but only one of them tells you what the number is for.
The syntax is uppercase, regular and boring, which is the highest
compliment a language for machine code can earn. gasm-devkit exists
to give that syntax the tooling it deserves.
## Features
- **Front end.** A hand-written lexer and an error-tolerant parser produce a
typed AST with source positions; `gasm tokens` and `gasm parse` expose them
directly.
- **Formatter.** `gasm fmt` canonicalises indentation, operand spacing,
per-function mnemonic alignment and blank-line layout: `gofmt` for assembly,
operating recursively on directories the way `go fmt` does. `-l` lists
files whose formatting differs and `-d` prints a unified diff.
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
`undefined-label`, `abi-argsize` (declared argument area vs the `// func`
signature), `register-clobber` (Go ABI register liveness over the
control-flow graph), `stack-imbalance`, `abi0-register-args` and
`unencodable-instruction`.
- **Standalone assembler.** `gasm asm` encodes all four architectures without
the Go toolchain and writes raw images or linkable ELF objects (with DWARF5
debug sections) with no Go installation needed, or the Go toolchain's own
GOOBJ format, which needs the installed toolchain and which `go build`
consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too.
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
offsets after assembling, or disassembles raw bytes from a file or stdin.
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
executable memory: smoke calls, ABI checks (sentinel registers, red-zone
canary), differential fuzzing against the `go tool asm` build, and
byte-for-byte ground-truth comparison of the machine code.
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
breakpoints (optionally conditional), hardware watchpoints, register and
memory inspection, and headless script runs that report instruction and
label coverage.
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature
help, document highlights, workspace symbol search, #include document
links and folding ranges over stdio; definition, references and rename
work across every open document.
- **Comparators and audits.** `gasm diff` compares the machine code of two
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
`gasm audit-instructions` diffs the encoder against the installed toolchain,
and `gasm scaffold` generates a differential test skeleton for a kernel.
### Architecture support
Four architectures, the four that matter in practice:
| Architecture | GOARCH | File suffix | Instructions recognised |
|--------------|-------------|--------------|---------------------------------------------|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
"Common opcodes" are the instructions shared by every architecture (`RET`,
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
...) that the assembler accepts as aliases. The tables are generated from
the Go toolchain's own assembler source (`just gen` refreshes them), so
every mnemonic the real assembler accepts is recognised; what the encoder
can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 127 of 627 files (20.3 %)
assembling for every target architecture today, with the top failure
reasons per architecture; the number moves with every release.
### Validation status
**Only amd64 is validated on real hardware.** The other three
architectures are validated under qemu-user emulation, because the
project owns no arm64, riscv64 or loong64 machine, and emulation is the
only substitute available for the hardware. The distinction matters and
is stated rather than implied: everything below is a claim about what has
actually been executed.
| Layer | amd64 | arm64, riscv64, loong64 |
|---|---|---|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more |
Consequences, stated plainly. An emulator is a model of a CPU, not the
CPU: instruction semantics are implemented in software and can differ
from silicon in ways a test suite does not reveal. A kernel that passes
under qemu-user is therefore not proven correct on real hardware, and a
discrepancy found on real hardware is a defect in gasm, reported like any
other. Encoding parity is the exception: the byte comparison against the
toolchain runs on the host for every architecture, so no emulator stands
between the claim and the evidence. The debugger is the weakest case: on
the three emulated architectures its per-architecture ptrace code has
been compiled and read, never executed. Its architecture-neutral units
run under `go test ./...`, which the race workflow and a manual run
perform; the default `just test` gate does not sweep `./debug/...`.
## Direction
The plan, in the order it is being worked:
- **Extended instruction support.** Two layers. First, encoding
coverage for every mnemonic the Go toolchain itself accepts, closed in
order of how often real code needs each instruction;
`gasm audit-instructions` measures the gap. Second, the larger work:
an extended instruction set the toolchain does not know at all. The
toolchain-derived tables stay generated and untouched; only the
extended instructions are hand-maintained, with their own spellings
and encoders, verified by execution (on real hardware for amd64, under
emulation for the rest, per the validation status above) because the
toolchain offers no ground truth to compare against. The gaps exist
on every architecture, amd64 included.
- **Full GOOBJ and ELF compilation.** The destination is a complete,
standalone compilation path: linkable ELF objects for consumers outside
Go, and GOOBJ objects that `go build` links directly. Through GOOBJ, a
Go program will be able to use machine instructions that the Go
toolchain itself does not support; through ELF, Plan 9 assembly becomes
usable outside Go entirely.
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
architectures and is where the binary builds. FreeBSD follows: the
JIT's executable-memory mapping and the ptrace debugger layer are the
two pieces of porting work. Other unix systems may follow those two.
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
No others are planned.
## Install
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
linux/loong64 are on the
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
From source (Go 1.27.1):
```sh
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
```
Or from a repository checkout:
```sh
just install
```
The installed binary reports the version the toolchain recorded: the tag
on a tagged checkout, a pseudo-version naming the commit below one.
## Quick start
```sh
cat > hello_amd64.s <<'EOF'
#include "textflag.h"
// func add(a, b int) int
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
ADDQ b+8(FP), AX
MOVQ AX, ret+16(FP)
RET
EOF
gasm lint hello_amd64.s # static checks
gasm asm -o hello.bin hello_amd64.s # assemble to a raw image
gasm verify --call add --args a=2,b=3 hello_amd64.s # JIT-call it with arguments
```
## Usage
```sh
gasm fmt # reformat every .s below here, like go fmt
gasm fmt -w kernel_amd64.s # canonicalise one file in place
gasm fmt -l *.s # list files whose formatting differs
gasm fmt -d kernel_amd64.s # print a unified diff instead
gasm lint *.s # static checks
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
gasm asm --format goobj -p pkg/path -o k.o k.s # Go object, consumed by go build
gasm dis k.s # assemble, then list each function
gasm dis -a amd64 - < dump.bin # disassemble raw bytes from stdin
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
gasm debug --func name k.s # interactive debugger
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
gasm debug --func name --cover k.s # instruction and label coverage
gasm diff a.s b.s # compare machine code byte-for-byte
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
gasm profile k.s # show basic-block structure
gasm audit-instructions # encoder vs go tool asm name diff
gasm scaffold differential k.s # generate a differential test skeleton
```
Run `gasm --help` for the command overview and `gasm <command> -h` for a
command's flags. [docs/CLI.md](docs/CLI.md) is the full reference.
### Editor integration
`gasm lsp` speaks the Language Server Protocol over standard input/output, so
any LSP-capable editor can use it: point your editor's LSP client at the
binary and associate it with `.s` files. Syntax highlighting is delivered as
LSP semantic tokens, so no editor-specific grammar is required. The server
infers the target architecture from the file-name suffix
(`_amd64.s` / `_arm64.s` / `_riscv64.s` / `_loong64.s`).
## Development
```sh
just build # compile, zero errors and zero warnings
just test # the suite, no cache, the 80 % coverage floor
just gates # build, fmt-check, vet, test, race: the definition of done
just fmt # gofmt the tree
just gen # regenerate the instruction tables from the Go toolchain
```
See [CONTRIBUTING.md](CONTRIBUTING.md) for the development workflow and
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for setup details and every
recipe.
## Documentation
- [docs/CLI.md](docs/CLI.md): full command reference
- man pages: `just install-man` installs gasm(1) and one page per command
except `version`, which is documented inside gasm(1) instead, into
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
them
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
- [CHANGELOG.md](CHANGELOG.md): release history
## Licence
BSD-3-Clause; see [LICENSE](LICENSE).
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
+41
View File
@@ -0,0 +1,41 @@
# Security policy
## Supported versions
Security fixes go to the newest release and to the `development` branch. Older
releases do not receive them.
| Version | Supported |
|---|---|
| 0.34.0 | yes |
| older releases | no |
## Reporting a vulnerability
**Do not open a public issue for a security problem.** A public report tells everyone
about the flaw before there is a fix. Report it privately to
**opensource@petrbalvin.org**.
Include:
- the version or commit you tested, and the platform
- what the problem is, and what an attacker gains from it
- the smallest reproducer you have, ideally a test or a single command
- a suggested fix, if you have one
## What to expect
- A human reads the report, and you get an acknowledgement.
- You are kept informed while the fix is being made, and told when it ships.
- The fix is released before the details are published, and the timing is agreed with
you.
- The fix ships without naming you: the project keeps no credits list, so the release
notes, the changelog and the commits name no reporter.
## Out of scope
- Findings that require the attacker to already run code as the user, or to have local
access.
- Missing hardening with no demonstrated impact.
- Flaws in a third-party dependency: report them to that project, and to this one only
when this project's use of it makes them reachable.
+8 -2
View File
@@ -86,7 +86,10 @@ func filterCommon(names []string) []string {
func writeCommon(names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
b.WriteString("package arch\n\n")
b.WriteString("// commonGeneratedInstrs is the set of opcodes shared by every architecture\n")
b.WriteString("// (RET, JMP, NOP, CALL, TEXT, FUNCDATA, PCDATA, …).\n")
@@ -154,7 +157,10 @@ func stringLit(elt ast.Expr) string {
func writeGen(arch, sub string, names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
b.WriteString("// SPDX-License-Identifier: BSD-3-Clause\n\n")
b.WriteString("package arch\n\n")
b.WriteString("// " + arch + "GeneratedInstrs is the complete set of " + arch +
" mnemonics accepted by\n// Go's Plan 9 assembler.\n")
+2
View File
@@ -270,6 +270,8 @@ func amd64Curated() []Instr {
"VMINPD", "VMINPS", "VMINSD", "VMINSS", "VMAXPD", "VMAXPS", "VMAXSD", "VMAXSS",
"VXORPD", "VXORPS", "VANDPD", "VANDPS", "VANDNPD", "VANDNPS", "VORPD", "VORPS",
"VUNPCKHPD", "VUNPCKLPD", "VUNPCKHPS", "VUNPCKLPS",
"PSHUFD", "PSHUFHW", "PSHUFLW", "SHUFPS", "SHUFPD",
"UNPCKLPS", "UNPCKHPS", "UNPCKLPD", "UNPCKHPD",
"VSQRTPD", "VSQRTPS", "VSQRTSD", "VSQRTSS", "VRSQRTPS", "VRCPPS",
"VCMPPD", "VCMPPS", "VCMPSD", "VCMPSS",
} {
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+3
View File
@@ -55,6 +55,7 @@ const (
Mask // AVX-512 mask register (K)
Float // arm64 floating-point register (F)
VecARM // arm64 SIMD/vector register (V)
VecSIMD // architecture-neutral SIMD/vector register (LoongArch LSX/LASX)
Special // architecture-special register
)
@@ -73,6 +74,8 @@ func (c RegClass) String() string {
return "float"
case VecARM:
return "vector (arm64)"
case VecSIMD:
return "vector"
case Special:
return "special"
default:
+2 -2
View File
@@ -29,7 +29,7 @@ func arm64Registers() []Register {
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
}
// General-purpose integer registers R0–R30.
// General-purpose integer registers R0-R30.
for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
}
@@ -145,7 +145,7 @@ func arm64Curated() []Instr {
for _, op := range []string{
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL",
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL",
} {
t = append(t, i(op, "Atomic memory operation"))
}
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/util.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+2 -2
View File
@@ -31,10 +31,10 @@ func loong64Registers() []Register {
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register")
add(fmt.Sprintf("V%d", i), VecSIMD, "LSX 128-bit vector register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register")
add(fmt.Sprintf("X%d", i), VecSIMD, "LASX 256-bit vector register")
}
return regs
}
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+3
View File
@@ -1,5 +1,8 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
+247
View File
@@ -0,0 +1,247 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
// AArch64 GOOBJ: the preamble, the magic, the block offsets and the
// non-package symbol definitions.
func TestGOObjectAARCH64Structure(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Check preamble.
idx := strings.Index(string(obj), "\n!\n")
if idx < 0 {
t.Fatal("missing preamble separator")
}
preamble := string(obj[:idx])
if !strings.HasPrefix(preamble, "go object") {
t.Errorf("preamble = %q, want 'go object ...'", preamble)
}
// Check GOOBJ magic.
magicIdx := idx + 3
if magicIdx+8 > len(obj) || string(obj[magicIdx:magicIdx+8]) != "\x00go120ld" {
t.Error("missing GOOBJ magic")
}
// The object should contain the function's code.
if len(img.Code) == 0 {
t.Error("no code generated")
}
}
// TestGOObjectAARCH64PairReloc pins the ADRP-pair relocation shape against
// the toolchain's own object for the same source: exactly one R_ADDRARM64
// of Siz 8 at the ADRP word (cmd/internal/obj/arm64/asm7.go adds a single
// Siz-8 relocation per pair and the linker patches both instructions from
// it). gasm's assembler records the ADRP+ADD form as two word relocs; the
// emitter must coalesce them, not emit two Siz-4 records.
func TestGOObjectAARCH64PairReloc(t *testing.T) {
f, errs := parser.Parse("gv_arm64.s", `
#include "textflag.h"
TEXT ·getv(SB), NOSPLIT, $0-8
MOVD $v<>(SB), R4
MOVD R4, ret+0(FP)
RET
GLOBL v<>(SB), RODATA, $8
DATA v<>+0(SB)/8, $7
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.GOObjectAARCH64("main", "gv_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
v := openGoobj(t, obj)
relocs := v.blk(blkReloc)
le := binary.LittleEndian
// Two DWARF relocs on the lines/DIE symbols, then the code's one pair
// relocation.
if len(relocs) != 3*23 {
t.Fatalf("relocs = %d bytes, want three entries", len(relocs))
}
cr := relocs[2*23:]
if off := int32(le.Uint32(cr[0:])); off != 0 {
t.Errorf("pair reloc off = %d, want 0 (the ADRP word)", off)
}
if siz := cr[4]; siz != 8 {
t.Errorf("pair reloc siz = %d, want 8", siz)
}
if typ := le.Uint16(cr[5:]); typ != relocArm64Addr {
t.Errorf("pair reloc type = %d, want %d (R_ADDRARM64)", typ, relocArm64Addr)
}
if pkg := le.Uint32(cr[15:]); pkg != pkgIdxSelf {
t.Errorf("pair reloc PkgIdx = %#x, want pkgIdxSelf", pkg)
}
// The GLOBL is the first package definition.
if sym := le.Uint32(cr[19:]); sym != 0 {
t.Errorf("pair reloc SymIdx = %d, want 0 (the GLOBL definition)", sym)
}
}
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
// Go program for arm64, substitutes the gasm-produced object into the package
// archive, re-links with cmd/link, and verifies the symbol appears in the
// resulting binary. The binary is not executed (no arm64 host or qemu).
// Skipped when no Go toolchain is available.
func TestGOObjectAARCH64Link(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
TEXT ·getv(SB), NOSPLIT, $0-8
MOVD $v<>(SB), R4
MOVD R4, ret+0(FP)
RET
GLOBL v<>(SB), RODATA, $8
DATA v<>+0(SB)/8, $7
`
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func getv() *int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
if getv() == nil {
panic("bad getv")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64link\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the cross build (GOARCH=arm64): the package archive and the
// link line.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=arm64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
// Expand $WORK in the object path.
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
// Read the toolchain-produced object and assemble the same source with gasm.
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse("main_arm64.s", string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Replace the toolchain-produced object with gasm's.
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
// Re-link.
if linkLine == "" {
t.Skip("could not find link command in build log")
}
// Expand $WORK in the link command.
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// Verify the binary exists and contains the symbol.
binPath := filepath.Join(dir, "prog")
if _, err := os.Stat(binPath); err != nil {
t.Fatalf("binary not found: %v", err)
}
binData, err := os.ReadFile(binPath)
if err != nil {
t.Fatalf("read binary: %v", err)
}
if !strings.Contains(string(binData), "add") && !strings.Contains(string(binData), "a64link") {
t.Error("binary does not contain expected symbol")
}
}
File diff suppressed because it is too large Load Diff
+693
View File
@@ -0,0 +1,693 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 (AArch64) instruction encoding.
//
// The encoder is data-driven: each mnemonic maps to an instruction format and
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own arm64
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
// exactly, the ground-truth oracle for the verify suite.
//
// All AArch64 instructions are 32 bits, little-endian. The formats used here
// (per the ARM Architecture Reference Manual):
//
// DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
// DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd
// Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd
// Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd
// Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt
// LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt)
// LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt
// Branch-imm 0<<31 | 0x5<<26 | imm26 (B)
// Branch-imm 1<<31 | 0x5<<26 | imm26 (BL)
// Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond)
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
import "maps"
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
// runtime's assembly uses. Returns -1 for an unrecognised name.
func arm64RegNum(name string) int {
switch name {
case "R0":
return 0
case "R1":
return 1
case "R2":
return 2
case "R3":
return 3
case "R4":
return 4
case "R5":
return 5
case "R6":
return 6
case "R7":
return 7
case "R8":
return 8
case "R9":
return 9
case "R10":
return 10
case "R11":
return 11
case "R12":
return 12
case "R13":
return 13
case "R14":
return 14
case "R15":
return 15
case "R16":
return 16
case "R17":
return 17
case "R18":
return 18
case "R19":
return 19
case "R20":
return 20
case "R21":
return 21
case "R22":
return 22
case "R23":
return 23
case "R24":
return 24
case "R25":
return 25
case "R26", "REGCTXT", "CTXT":
return 26
case "R27", "REGTMP", "TMP":
return 27
case "R28", "REGG", "g":
return 28
case "R29", "FP":
return 29
case "R30", "LR", "LINK":
return 30
case "R31", "ZR":
return 31
case "SP", "RSP":
// RSP is the toolchain's spelling for register 31 (it rejects
// R31 in an operand); SP stays for sources that spell it the
// amd64 way. SP and ZR share encoding 31; context determines
// the meaning.
return 31
}
// F0-F31.
if len(name) >= 1 && name[0] == 'F' {
n := 0
for i := 1; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return -1
}
n = n*10 + int(name[i]-'0')
}
if n <= 31 {
return n
}
}
return -1
}
// ---- format helpers ----
// a64wordLE encodes a uint32 as 4 little-endian bytes.
func a64wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// a64WordsLE concatenates one or more instruction words as little-endian bytes.
func a64WordsLE(ws ...uint32) []byte {
var out []byte
for _, w := range ws {
out = append(out, a64wordLE(w)...)
}
return out
}
// ---- data-processing (immediate) ----
// a64AddSub encodes an ADD/SUB (immediate) instruction:
// sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd.
func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 {
return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd
}
// ---- move wide ----
// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction:
// sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd.
func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
}
// ---- load/store (unsigned immediate, scaled) ----
// a64LSU encodes a load/store register (unsigned immediate, scaled):
// size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt.
// (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.)
func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 {
return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt
}
// ---- load/store (unscaled immediate) ----
// a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed):
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt.
// Note: the 0<<24 distinguishes unscaled from the pre/post-index forms.
func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// ---- load/store pair ----
// a64LSP encodes a load/store pair instruction (signed offset):
// opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
// opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD.
// L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs).
func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
}
// ---- branches ----
// a64Branch encodes an unconditional branch (B/BL):
// op<<31 | 0x5<<26 | imm26.
func a64Branch(op uint32, imm26 int32) uint32 {
return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF)
}
// a64BranchCond encodes a conditional branch (B.cond):
// 0x2A<<25 | imm19<<5 | cond.
func a64BranchCond(imm19 int32, cond uint32) uint32 {
return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF
}
// a64UncondBranch encodes an unconditional branch register (BR/BLR/RET):
// 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd.
// opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30).
func a64UncondBranch(opc, rn, rd uint32) uint32 {
return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd
}
// ---- ADR/ADRP ----
// a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1):
// p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd.
func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 {
return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd
}
// ---- system ----
// a64NOP encodes a NOP: 0xd503201f.
const a64NOP uint32 = 0xd503201f
// a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5.
func a64BRK(imm16 uint32) uint32 {
return 0xd4200000 | imm16<<5
}
// ---- condition codes ----
const (
a64CondEQ = 0x0
a64CondNE = 0x1
a64CondCS = 0x2
a64CondHS = 0x2
a64CondCC = 0x3
a64CondLO = 0x3
a64CondMI = 0x4
a64CondPL = 0x5
a64CondVS = 0x6
a64CondVC = 0x7
a64CondHI = 0x8
a64CondLS = 0x9
a64CondGE = 0xa
a64CondLT = 0xb
a64CondGT = 0xc
a64CondLE = 0xd
)
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
var arm64CondMap = map[string]uint32{
"EQ": a64CondEQ,
"NE": a64CondNE,
"CS": a64CondCS,
"HS": a64CondHS,
"CC": a64CondCC,
"LO": a64CondLO,
"MI": a64CondMI,
"PL": a64CondPL,
"VS": a64CondVS,
"VC": a64CondVC,
"HI": a64CondHI,
"LS": a64CondLS,
"GE": a64CondGE,
"LT": a64CondLT,
"GT": a64CondGT,
"LE": a64CondLE,
}
// ---- instruction format tags ----
type a64Format uint8
const (
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
// opcode constant, positioned at its exact bit range.
type a64Enc struct {
format a64Format
op uint32 // the pre-positioned opcode bits
}
// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to
// their encoding. The base integer, memory, floating-point and SIMD
// instruction sets are covered.
var a64InstrTable = map[string]a64Enc{}
func init() {
// ---- data-processing (shifted register) ----
// Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
dpsr := map[string]uint32{
// Add/Sub
"ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default)
"ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0
"ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24,
"ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24,
"SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24,
"SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24,
"SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24,
"SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24,
// Logical (shifted register)
"AND": 1<<31 | 0<<29 | 0x0a<<24,
"ANDW": 0<<31 | 0<<29 | 0x0a<<24,
"BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21,
"BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21,
"ORR": 1<<31 | 1<<29 | 0x0a<<24,
"ORRW": 0<<31 | 1<<29 | 0x0a<<24,
"ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21,
"ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21,
"EOR": 1<<31 | 2<<29 | 0x0a<<24,
"EORW": 0<<31 | 2<<29 | 0x0a<<24,
"EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21,
"EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21,
"ANDS": 1<<31 | 3<<29 | 0x0a<<24,
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
// Divide (data-processing 2 source): the opcode occupies bits 15:10
// of the 0xd6<<21 fixed field, UDIV=0b0010 and SDIV=0b0011 (ARM ARM
// "Data-processing (2 source)"; the toolchain spells them OPDP2(2)
// and OPDP2(3)). sf=1 selects the X forms.
"SDIV": 1<<31 | 0xd6<<21 | 3<<10,
"SDIVW": 0<<31 | 0xd6<<21 | 3<<10,
"UDIV": 1<<31 | 0xd6<<21 | 2<<10,
"UDIVW": 0<<31 | 0xd6<<21 | 2<<10,
// Conditional select
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
"CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10,
"CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10,
"CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10,
"CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10,
"CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10,
"CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10,
}
for m, op := range dpsr {
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
}
// Aliases that map to the same encoding as their target.
a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]}
a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]}
a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]}
a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]}
a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]}
a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]}
a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]}
a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]}
a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]}
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
// ---- shifts ----
// The mnemonic serves both forms: with an immediate the aliases of the
// data-processing (immediate) group apply (ARM ARM "Shifts"), with a
// register the data-processing (2 source) LSLV/LSRV/ASRV/RORV. The op
// field carries the immediate-alias base; encodeARM64Shift derives both
// it and the two-source opcode. Identities, W = 64 (X) or 32 (W):
//
// LSL $sh, Rn, Rd = UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh
// LSR $sh, Rn, Rd = UBFM Rd, Rn, #sh, #(W-1)
// ASR $sh, Rn, Rd = SBFM Rd, Rn, #sh, #(W-1)
// ROR $sh, Rn, Rd = EXTR Rd, Rn, Rn, #sh
shifts := map[string]a64Enc{
"LSL": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
"LSLW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
"LSR": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
"LSRW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
"ASR": {format: a64FShift, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}, // SBFM X
"ASRW": {format: a64FShift, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}, // SBFM W
"ROR": {format: a64FShift, op: 1<<31 | 0x27<<23 | 1<<22}, // EXTR X
"RORW": {format: a64FShift, op: 0<<31 | 0x27<<23 | 0<<22}, // EXTR W
}
maps.Copy(a64InstrTable, shifts)
// ---- multiply accumulate ----
// MADD/MSUB Rm, Ra, Rn, Rd: sf 00 11011 o0(15) Rm Ra Rn Rd. The
// toolchain's optab has no shorter row, so all four operands are
// mandatory, and Ra is the SECOND operand.
a64InstrTable["MADD"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24}
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
// ---- move wide ----
// MOVZ/MOVN/MOVK
a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23}
a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23}
a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23}
a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23}
a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23}
a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23}
// ---- ADR/ADRP ----
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
// Load/store mnemonics never enter this table: the MOV pseudo-instruction
// dispatch handles them through a64LoadTable, which also carries the store
// opcode (integer and FP stores both use opc=00, differing only in V).
// ---- branches ----
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26}
// Conditional branches.
condBranches := map[string]uint32{
"BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2,
"BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5,
"BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9,
"BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd,
}
for name, cond := range condBranches {
a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond}
}
// Unconditional branch register (BR/BLR/RET).
a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21}
a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21}
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
// ---- system ----
// NOP/NOOP/UNDEF are spelled out in encodeARM64Instr's pseudo switch,
// so they carry no table entry; a64NOP and a64BRK are the encoders.
// ---- EXTR ----
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22}
// ---- bitfield ----
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ----
fp3 := map[string]uint32{
"FADDS": 0x1e202800, "FADDD": 0x1e602800,
"FSUBS": 0x1e203800, "FSUBD": 0x1e603800,
"FMULS": 0x1e200800, "FMULD": 0x1e600800,
"FDIVS": 0x1e201800, "FDIVD": 0x1e601800,
"FMAXS": 0x1e204800, "FMAXD": 0x1e604800,
"FMINS": 0x1e205800, "FMIND": 0x1e605800,
"FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800,
"FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800,
"FNMULS": 0x1e208800, "FNMULD": 0x1e608800,
}
for m, op := range fp3 {
a64InstrTable[m] = a64Enc{format: a64FFP3, op: op}
}
// ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ----
fp1 := map[string]uint32{
"FMOVS": 0x1e204000, "FMOVD": 0x1e604000,
"FABSS": 0x1e20c000, "FABSD": 0x1e60c000,
"FNEGS": 0x1e214000, "FNEGD": 0x1e614000,
"FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000,
"FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000,
"FRINTNS": 0x1e244000, "FRINTND": 0x1e644000,
"FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000,
"FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000,
"FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000,
"FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000,
"FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000,
"FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000,
}
for m, op := range fp1 {
a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op}
}
// ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ----
fp4 := map[string]uint32{
"FMADDS": 0x1f000000, "FMADDD": 0x1f400000,
"FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000,
"FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000,
"FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000,
}
for m, op := range fp4 {
a64InstrTable[m] = a64Enc{format: a64FFP4, op: op}
}
// ---- FP compare (Rm, Rn or #0, Rn) ----
fpcmp := map[string]uint32{
"FCMPS": 0x1e202000, "FCMPD": 0x1e602000,
"FCMPES": 0x1e202010, "FCMPED": 0x1e602010,
}
for m, op := range fpcmp {
a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op}
}
// ---- FP conditional compare (Rm, Rn, #nzcv, cond) ----
fpccmp := map[string]uint32{
"FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400,
"FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410,
}
for m, op := range fpccmp {
a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op}
}
// ---- FP conditional select (Rm, Rn, Rd, cond) ----
a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00}
a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00}
// ---- FP ↔ integer conversion ----
fpcvt := map[string]uint32{
"FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000,
"FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000,
"FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000,
"FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000,
"SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000,
"SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000,
"UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000,
"UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000,
}
for m, op := range fpcvt {
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
}
// FMOV between GP and FP registers needs no table entry: the MOV
// pseudo-instruction dispatches it by operand class (encodeARM64RegMove).
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
csel := map[string]uint32{
"CSEL": 0x9a800000, "CSELW": 0x1a800000,
"CSINC": 0x9a800400, "CSINCW": 0x1a800400,
"CSINV": 0xda800000, "CSINVW": 0x5a800000,
"CSNEG": 0xda800400, "CSNEGW": 0x5a800400,
}
for m, op := range csel {
a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op}
}
// Aliases
a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000}
a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000}
a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400}
a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400}
// ---- CRC32 ----
crc32 := map[string]uint32{
"CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400,
"CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00,
"CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400,
"CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00,
}
for m, op := range crc32 {
a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op}
}
// ---- exclusive load/store ----
// Single-register forms pre-set the unused Rs and Rt2 fields to 31 (the
// 0x7c00/0x1f0000 halves of the constants below); the register-pair
// forms carry a real Rt2 in bits 14:10, so their opcodes pre-set
// neither field.
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00}
a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00}
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
// Pair loads, LDSTX(sz, 0, l=1, o1=1, o0) in asm7.go: LDXP/ LDXPW have
// o0=0, LDAXP/LDAXPW o0=1 (bit 15). Rs (bits 20:16) stays 31.
a64InstrTable["LDXP"] = a64Enc{format: a64FExcl, op: 0xc8600000}
a64InstrTable["LDXPW"] = a64Enc{format: a64FExcl, op: 0x88600000}
a64InstrTable["LDAXP"] = a64Enc{format: a64FExcl, op: 0xc8608000}
a64InstrTable["LDAXPW"] = a64Enc{format: a64FExcl, op: 0x88608000}
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00}
a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00}
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
// Pair stores, LDSTX(sz, 0, l=0, o1=1, o0): STXP/STXPW have o0=0,
// STLXP/STLXPW o0=1 (bit 15). Both Rs and Rt2 are real fields.
a64InstrTable["STXP"] = a64Enc{format: a64FExcl, op: 0xc8200000}
a64InstrTable["STXPW"] = a64Enc{format: a64FExcl, op: 0x88200000}
a64InstrTable["STLXP"] = a64Enc{format: a64FExcl, op: 0xc8208000}
a64InstrTable["STLXPW"] = a64Enc{format: a64FExcl, op: 0x88208000}
// ---- LSE atomics ----
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10}
a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10}
a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10}
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
// ---- SIMD basics ----
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
}
// ---- load/store helper tables ----
// a64LSType describes the load/store parameters for a MOV width mnemonic.
type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
}
// a64StoreOpc returns the store opc for a given load type: integer and FP
// stores both encode opc=00 (the load's signedness bit sits in opc[1], which
// the store form clears; FP registers are selected by V, not opc).
func a64StoreOpc(t a64LSType) int {
return 0
}
// arm64RegClass discriminates integer (R), floating-point (F) registers for
// the MOV pseudo-instruction.
type arm64RegClass int
const (
arm64ClsNone arm64RegClass = iota
arm64ClsGR
arm64ClsFP
)
// arm64RegClassOf reports the register class of a register operand name.
func arm64RegClassOf(name string) arm64RegClass {
switch {
case name == "":
return arm64ClsNone
case len(name) >= 1 && name[0] == 'F':
return arm64ClsFP
default:
return arm64ClsGR
}
}
// arm64Movcon returns the shift (in units of 16 bits) at which a non-zero
// 16-bit chunk of v sits, or -1 if v cannot be represented as a single
// MOVZ/MOVN immediate. This is the Go toolchain's movcon function.
func arm64Movcon(v int64) int {
for s := 0; s < 64; s += 16 {
if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 {
return s
}
}
return -1
}
+990
View File
@@ -0,0 +1,990 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
func TestArm64LDRSTREncoding(t *testing.T) {
tests := []struct {
name string
got uint32
want uint32
}{
{"LDR X4, [SP, #56]", a64LSU(3, 0, 1, 7, 31, 4), 0xf9401fe4},
{"STR X4, [SP, #64]", a64LSU(3, 0, 0, 8, 31, 4), 0xf90023e4},
{"STR X5, [SP, #32]", a64LSU(3, 0, 0, 4, 31, 5), 0xf90013e5},
{"LDR X6, [SP, #32]", a64LSU(3, 0, 1, 4, 31, 6), 0xf94013e6},
}
for _, tt := range tests {
if tt.got != tt.want {
t.Errorf("%s: got %08x, want %08x", tt.name, tt.got, tt.want)
}
}
}
func TestArm64PrologueEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
pro := arm64Prologue(fi)
if len(pro) != 12 {
t.Fatalf("prologue length: got %d, want 12", len(pro))
}
expected := []uint32{0xf81d0ffe, 0xf81f83fd, 0xd10023fd}
for i, w := range leWords(pro) {
if w != expected[i] {
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
}
}
}
func TestArm64EpilogueSmallEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32, leaf: false}
ret := arm64Return(fi)
if len(ret) != 12 {
t.Fatalf("epilogue length: got %d, want 12", len(ret))
}
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #48; RET
expected := []uint32{0xf85f83fd, 0xf84307fe, 0xd65f03c0}
for i, w := range leWords(ret) {
if w != expected[i] {
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, expected[i])
}
}
}
func TestArm64LargeFrameEncoding(t *testing.T) {
fi := arm64FrameInfo{autosize: 272, frame: 256, leaf: false}
pro := arm64Prologue(fi)
if len(pro) != 16 {
t.Fatalf("prologue length: got %d, want 16", len(pro))
}
expected := []uint32{0xd10443f4, 0xa93ffa9d, 0x9100029f, 0xd10023fd}
for i, w := range leWords(pro) {
if w != expected[i] {
t.Errorf("prologue word %d: got %08x, want %08x", i, w, expected[i])
}
}
epi := arm64Return(fi)
if len(epi) != 12 {
t.Fatalf("epilogue length: got %d, want 12", len(epi))
}
eexpected := []uint32{0xa97ffbfd, 0x910443ff, 0xd65f03c0}
for i, w := range leWords(epi) {
if w != eexpected[i] {
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, eexpected[i])
}
}
}
func TestArm64NoFrame(t *testing.T) {
fi := arm64FrameInfo{autosize: 0, frame: 0, leaf: true}
pro := arm64Prologue(fi)
if len(pro) != 0 {
t.Errorf("no-frame prologue: got %d bytes, want 0", len(pro))
}
ret := arm64Return(fi)
if len(ret) != 4 {
t.Fatalf("no-frame return: got %d bytes, want 4", len(ret))
}
if leWord(ret) != 0xd65f03c0 {
t.Errorf("no-frame RET: got %08x, want d65f03c0", leWord(ret))
}
}
func TestArm64RegNum(t *testing.T) {
tests := []struct {
name string
want int
}{
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
{"F0", 0}, {"F4", 4}, {"F31", 31},
{"INVALID", -1}, {"X0", -1}, {"", -1},
}
for _, tt := range tests {
got := arm64RegNum(tt.name)
if got != tt.want {
t.Errorf("arm64RegNum(%q) = %d, want %d", tt.name, got, tt.want)
}
}
}
func TestArm64ComputeFrame(t *testing.T) {
src := "TEXT ·f(SB), NOSPLIT, $32-0\n\tADD\tR4, R5\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fi := arm64ComputeFrame(f.Decls[0].(*ast.Text))
if fi.frame != 32 {
t.Errorf("frame: got %d, want 32", fi.frame)
}
if fi.autosize != 48 { // 32+8=40, aligned to48
t.Errorf("autosize: got %d, want 48", fi.autosize)
}
// ADD + RET with no CALL/BL → leaf
if !fi.leaf {
t.Error("expected leaf")
}
}
func TestArm64IsLeaf(t *testing.T) {
src := "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR4, R5\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if !arm64IsLeaf(f.Decls[0].(*ast.Text)) {
t.Error("expected leaf")
}
src2 := "TEXT ·f(SB), NOSPLIT, $0-0\n\tBL\tother(SB)\n\tRET\n"
f2, errs := parser.Parse("test_arm64.s", src2)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if arm64IsLeaf(f2.Decls[0].(*ast.Text)) {
t.Error("expected non-leaf")
}
}
func TestArm64Bitmask(t *testing.T) {
tests := []struct {
v uint64
sf int
N, immr, imms uint32
ok bool
}{
{1, 1, 1, 0, 0, true}, // single bit at pos 0
{2, 1, 1, 63, 0, true}, // single bit at pos 1 (immr = esize-1)
{0, 1, 0, 0, 0, false}, // zero is not a bitmask
{0xFFFFFFFFFFFFFFFF, 1, 0, 0, 0, false}, // all ones is not a bitmask
{0x5555555555555555, 1, 0, 0, 0x3E, true}, // alternating bits (esize=2, ones=1)
{0xFFFFFFFF00000000, 1, 1, 32, 31, true}, // upper 32 bits set (esize=64, ones=32)
}
for _, tt := range tests {
N, immr, imms, ok := arm64Bitmask(tt.v, tt.sf)
if ok != tt.ok {
t.Errorf("arm64Bitmask(%#x, %d): ok=%v, want %v", tt.v, tt.sf, ok, tt.ok)
continue
}
if ok && (N != tt.N || immr != tt.immr || imms != tt.imms) {
t.Errorf("arm64Bitmask(%#x, %d): N=%d immr=%d imms=%d, want N=%d immr=%d imms=%d",
tt.v, tt.sf, N, immr, imms, tt.N, tt.immr, tt.imms)
}
}
}
func TestArm64AssembleFile(t *testing.T) {
src := `#include "textflag.h"
TEXT ·simple(SB), NOSPLIT, $0-0
MOV R4, R5
ADD R4, R5, R6
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
}
fn := img.Funcs[0]
if fn.Name != "simple" {
t.Errorf("func name: got %q, want %q", fn.Name, "simple")
}
//3 instructions ×4 bytes =12
if fn.Size != 12 {
t.Errorf("func size: got %d, want 12", fn.Size)
}
}
func TestArm64AssembleFileWithFrame(t *testing.T) {
src := `#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $16-8
MOVD arg+0(FP), R4
ADD $1, R4, R4
MOVD R4, ret+0(FP)
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
}
fn := img.Funcs[0]
if fn.Frame != 16 {
t.Errorf("frame: got %d, want 16", fn.Frame)
}
// Prologue (3×4=12) + body (3×4=12) + RET epilogue (3×4=12) = 36
if fn.Size != 36 {
t.Errorf("func size: got %d, want 36", fn.Size)
}
}
func TestArm64AssembleFileWithBranches(t *testing.T) {
src := `#include "textflag.h"
TEXT ·branch(SB), NOSPLIT, $0-0
BEQ done
BNE skip
skip:
ADD R4, R5
done:
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 16 {
t.Errorf("func size: got %d, want 16", fn.Size)
}
}
func TestArm64AssembleFileWithJumpChain(t *testing.T) {
src := `#include "textflag.h"
TEXT ·chain(SB), NOSPLIT, $0-0
BNE skip
ADD R4, R5
RET
skip:
B target
target:
ADD R6, R7
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// BNE should be redirected past skip→target to target directly.
if img.Funcs[0].Size != 24 {
t.Errorf("func size: got %d, want 24", img.Funcs[0].Size)
}
}
func TestArm64AssembleErrors(t *testing.T) {
tests := []struct {
name string
src string
}{
{"unsupported", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4, R5\n\tRET\n"},
{"undefined label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
f, errs := parser.Parse("test_arm64.s", tt.src)
if len(errs) > 0 {
return // parse error, that's fine
}
_, err := AssembleFileARM64(f)
if err == nil {
t.Error("expected error, got nil")
}
})
}
}
func TestArm64Movcon(t *testing.T) {
tests := []struct {
v int64
want int
}{
{0, 0}, // 0 fits at shift 0
{1, 0}, // single bit at shift 0
{0x10000, 16}, // single bit at shift 16
{0x100000000, 32}, // single bit at shift 32
{0xFF, 0}, // 0xFF fits at shift 0
{0x12345, -1}, // multiple chunks, not movcon
}
for _, tt := range tests {
got := arm64Movcon(tt.v)
if got != tt.want {
t.Errorf("arm64Movcon(%#x) = %d, want %d", tt.v, got, tt.want)
}
}
}
func TestArm64RegClassOf(t *testing.T) {
if arm64RegClassOf("R4") != arm64ClsGR {
t.Error("R4 should be GR")
}
if arm64RegClassOf("F4") != arm64ClsFP {
t.Error("F4 should be FP")
}
if arm64RegClassOf("") != arm64ClsNone {
t.Error("empty should be None")
}
}
func TestArm64ResolvePseudo(t *testing.T) {
fi := arm64FrameInfo{autosize: 48, frame: 32}
// FP: offset = sym.Offset + autosize +8
base, off := arm64ResolvePseudo(&ast.Symbol{Pseudo: "FP", Offset: 0}, fi)
if base != 31 || off != 56 {
t.Errorf("FP: base=%d off=%d, want 31, 56", base, off)
}
// SP: offset = sym.Offset + frame +8
base, off = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SP", Offset: -8}, fi)
if base != 31 || off != 32 {
t.Errorf("SP: base=%d off=%d, want 31, 32", base, off)
}
// SB: unresolved
base, _ = arm64ResolvePseudo(&ast.Symbol{Pseudo: "SB"}, fi)
if base != -1 {
t.Errorf("SB: base=%d, want -1", base)
}
}
// TestArm64FPSel tests FP conditional select encoding.
func TestArm64FPSel(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
FCSELD GE, F10, F11, F12
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// FCSELD should be 4 bytes + RET 4 bytes = 8
if img.Funcs[0].Size != 8 {
t.Errorf("size: got %d, want 8", img.Funcs[0].Size)
}
}
// TestArm64FPCvt tests FP conversion encoding.
func TestArm64FPCvt(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
FCVTZSD F4, R0
SCVTFD R4, F8
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12 {
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
}
}
// TestArm64CSEL tests conditional select encoding.
func TestArm64CSEL(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
CSEL EQ, R0, R1, R2
CSET NE, R3
CINC GE, R4, R5
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 16 {
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
}
}
// TestArm64CRC32 tests CRC32 encoding.
func TestArm64CRC32(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
CRC32B R0, R2
CRC32W R6, R8
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12 {
t.Errorf("size: got %d, want 12", img.Funcs[0].Size)
}
}
// TestArm64Bitfield tests bitfield/shift encoding.
func TestArm64Bitfield(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
ASR $4, R0, R1
LSL $12, R4, R5
EXTR $8, R0, R1, R2
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 16 {
t.Errorf("size: got %d, want 16", img.Funcs[0].Size)
}
}
// TestArm64SIMD tests SIMD encoding (via the instruction table).
func TestArm64SIMD(t *testing.T) {
// Verify SIMD instructions are in the table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
if _, ok := a64InstrTable[mnem]; !ok {
t.Errorf("%s not in instruction table", mnem)
}
}
}
// TestArm64LoadImm64 tests 64-bit immediate loading.
func TestArm64LoadImm64(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
MOVD $0x123456789ABCDEF0, R0
MOVD $0, R1
MOVD $1, R2
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// $0x123456789ABCDEF0 needs 4 MOVZ/MOVK instructions (16 bytes)
// $0 is 1 instruction (4 bytes)
// $1 is 1 bitmask instruction (4 bytes)
// RET is 1 instruction (4 bytes)
if img.Funcs[0].Size != 28 {
t.Errorf("size: got %d, want 28", img.Funcs[0].Size)
}
}
// TestArm64BranchCond tests conditional branch encoding.
func TestArm64BranchCond(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
BEQ done
BNE done
BGE done
BLT done
ADD R4, R5
done:
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// 4 branches + 1 ADD + 1 RET = 24 bytes
if img.Funcs[0].Size != 24 {
t.Errorf("size: got %d, want 24", img.Funcs[0].Size)
}
}
// TestArm64Errors tests error paths.
func TestArm64Errors(t *testing.T) {
tests := []struct {
name string
src string
}{
{"bad mnemonic", "TEXT ·f(SB), NOSPLIT, $0-0\n\tINVALID\tR4\n\tRET\n"},
{"bad label", "TEXT ·f(SB), NOSPLIT, $0-0\n\tB\tnosuch\n\tRET\n"},
{"bad register", "TEXT ·f(SB), NOSPLIT, $0-0\n\tADD\tR99, R0\n\tRET\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
f, errs := parser.Parse("test_arm64.s", tt.src)
if len(errs) > 0 {
return
}
_, err := AssembleFileARM64(f)
if err == nil {
t.Error("expected error, got nil")
}
})
}
}
// leWord reads a little-endian uint32 from b.
func leWord(b []byte) uint32 {
return uint32(b[0]) | uint32(b[1])<<8 | uint32(b[2])<<16 | uint32(b[3])<<24
}
// leWords reads all little-endian uint32s from b.
func leWords(b []byte) []uint32 {
n := len(b) / 4
w := make([]uint32, n)
for i := range w {
w[i] = leWord(b[i*4:])
}
return w
}
// TestArm64IndirectBranch pins the indirect branch forms in a leaf function:
// JMP (Rn) lowers to BR Rn, matching the toolchain's spelling, and the raw
// BR/BLR mnemonics encode directly (a gasm superset the toolchain's front
// end does not accept). CALL (Rn) shares the BLR path and its non-leaf
// prologue parity is covered by the ground-truth kernel.
func TestArm64IndirectBranch(t *testing.T) {
src := `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
JMP (R0)
BR R5
BLR R6
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
want := []uint32{
0xd61f0000, // BR R0
0xd61f00a0, // BR R5
0xd63f00c0, // BLR R6
0xd65f03c0, // RET (BR LR)
}
got := leWords(img.Code)
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// arm64Words assembles a single NOSPLIT leaf body and returns its words.
func arm64Words(t *testing.T, body string) []uint32 {
t.Helper()
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
return leWords(img.Code)
}
// TestArm64ShiftEncodings pins the shift words against `go tool asm -S`
// output (Go 1.27, arm64): immediate forms alias SBFM/UBFM with ROR as EXTR,
// register forms are the two-source LSLV/LSRV/ASRV/RORV.
func TestArm64ShiftEncodings(t *testing.T) {
got := arm64Words(t, "\tLSL $4, R0, R1\n\tLSR $8, R0, R2\n\tASR $4, R0, R3\n\tROR $12, R0, R4\n"+
"\tLSLW $4, R0, R5\n\tLSRW $8, R0, R6\n\tASRW $4, R0, R7\n\tRORW $12, R0, R8\n")
want := []uint32{
0xd37cec01, // LSL $4 = UBFM X1, X0, #60, #59
0xd348fc02, // LSR $8 = UBFM X2, X0, #8, #63
0x9344fc03, // ASR $4 = SBFM X3, X0, #4, #63
0x93c03004, // ROR $12 = EXTR X4, X0, X0, #12
0x531c6c05, // LSLW $4 = UBFM W5, W0, #28, #27
0x53087c06, // LSRW $8 = UBFM W6, W0, #8, #31
0x13047c07, // ASRW $4 = SBFM W7, W0, #4, #31
0x13803008, // RORW $12 = EXTR W8, W0, W0, #12
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("imm shift word %d = %08x, want %08x", i, got[i], want[i])
}
}
got = arm64Words(t, "\tLSL R9, R0, R10\n\tLSR R9, R0, R11\n\tASR R9, R0, R12\n\tROR R9, R0, R13\n"+
"\tLSLW R9, R0, R14\n\tLSRW R9, R0, R15\n\tASRW R9, R0, R16\n\tRORW R9, R0, R17\n")
want = []uint32{
0x9ac9200a, // LSLV X10, X0, X9
0x9ac9240b, // LSRV X11, X0, X9
0x9ac9280c, // ASRV X12, X0, X9
0x9ac92c0d, // RORV X13, X0, X9
0x1ac9200e, // LSLV W14, W0, W9
0x1ac9240f, // LSRV W15, W0, W9
0x1ac92810, // ASRV W16, W0, W9
0x1ac92c11, // RORV W17, W0, W9
0xd65f03c0, // RET
}
for i := range want {
if got[i] != want[i] {
t.Errorf("reg shift word %d = %08x, want %08x", i, got[i], want[i])
}
}
// Two-operand spellings fold to Rn = Rd.
got = arm64Words(t, "\tLSL $4, R1\n\tLSR R9, R1\n\tASR $4, R1\n\tROR R9, R1\n\tLSLW $4, R1\n\tRORW R9, R1\n")
want = []uint32{
0xd37cec21, // LSL $4, R1 = UBFM X1, X1, #60, #59
0x9ac92421, // LSRV X1, X1, X9
0x9344fc21, // ASR $4, R1 = SBFM X1, X1, #4, #63
0x9ac92c21, // RORV X1, X1, X9
0x531c6c21, // LSLW $4, R1 = UBFM W1, W1, #28, #27
0x1ac92c21, // RORV W1, W1, W9
0xd65f03c0, // RET
}
for i := range want {
if got[i] != want[i] {
t.Errorf("2op shift word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64ShiftRangeErrors: the toolchain reports "illegal bit number" for
// shift amounts at or above the operand width.
func TestArm64ShiftRangeErrors(t *testing.T) {
for _, src := range []string{
"\tLSL $64, R0, R1\n",
"\tLSRW $32, R0, R1\n",
"\tRORW $32, R0, R1\n",
"\tASR $-1, R0, R1\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
// TestArm64DivEncodings pins SDIV/UDIV in both widths: the 2-source opcode
// field (bits 15:10 of the 0xd6<<21 fixed field) is UDIV=0b0010, SDIV=0b0011.
func TestArm64DivEncodings(t *testing.T) {
got := arm64Words(t, "\tSDIV R1, R2, R3\n\tUDIV R1, R2, R3\n\tSDIVW R1, R2, R3\n\tUDIVW R1, R2, R3\n")
want := []uint32{
0x9ac10c43, // SDIV X3, X2, X1
0x9ac10843, // UDIV X3, X2, X1
0x1ac10c43, // SDIV W3, W2, W1
0x1ac10843, // UDIV W3, W2, W1
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("div word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64MAddSub pins the four-operand MADD/MSUB words (Rm, Ra, Rn, Rd,
// with Ra in bits 14:10) and rejects the shorter spellings the toolchain
// also rejects.
func TestArm64MAddSub(t *testing.T) {
got := arm64Words(t, "\tMADD R1, R2, R3, R4\n\tMSUB R1, R2, R3, R4\n\tMADDW R1, R2, R3, R5\n\tMSUBW R1, R2, R3, R5\n")
want := []uint32{
0x9b010864, // MADD X4, X3, X1, X2 (Rm=1, Ra=2, Rn=3)
0x9b018864, // MSUB X4, X3, X1, X2
0x1b010865, // MADD W5, W3, W1, W2
0x1b018865, // MSUB W5, W3, W1, W2
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("madd word %d = %08x, want %08x", i, got[i], want[i])
}
}
// The accumulate operand is mandatory: 2- and 3-operand forms error
// rather than silently reading R0 or ZR as the accumulator.
for _, body := range []string{
"\tMADD R1, R2\n",
"\tMADD R1, R2, R3\n",
"\tMSUBW R1, R2, R3\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64MovImmWidth pins the immediate classifications whose size pass
// once disagreed with the encoder: negative and 0xFFFFFFFF W values go
// through MOVN after 32-bit truncation, and 3- to 4-chunk constants expand
// to one word per non-zero chunk.
func TestArm64MovImmWidth(t *testing.T) {
got := arm64Words(t, "\tMOVW $-1, R0\n\tMOVW $0xFFFFFFFF, R3\n")
want := []uint32{
0x12800000, // MOVN W0, #0
0x12800003, // MOVN W3, #0
0xd65f03c0, // RET
}
for i := range want {
if got[i] != want[i] {
t.Errorf("movw word %d = %08x, want %08x", i, got[i], want[i])
}
}
for _, tt := range []struct {
body string
words int
}{
{"\tMOVD $0x0001000200030000, R2\n", 3}, // three chunks
{"\tMOVD $0x0001000200030004, R1\n", 4}, // four chunks
{"\tMOVW $-1, R0\n", 1}, // MOVN after truncation
} {
if got := arm64Words(t, tt.body); len(got) != tt.words+1 {
t.Errorf("%s: %d words, want %d (including RET)", tt.body, len(got), tt.words+1)
}
}
}
// TestArm64ExclOffsetErrors: exclusive and atomic encodings carry no
// immediate field, so a non-zero offset is rejected the way the toolchain
// reports "illegal combination" for it, never silently dropped.
func TestArm64ExclOffsetErrors(t *testing.T) {
for _, body := range []string{
"\tLDXR 8(R1), R2\n",
"\tLDAXR 8(R1), R2\n",
"\tSTXR R3, 8(R1), R4\n",
"\tSTLXR R3, 8(R1), R4\n",
"\tCASD R3, 8(R1), R4\n",
"\tLDADDD R3, 8(R1), R4\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64ExclNoOffset pins the plain (Rn) forms, byte-for-byte against
// go tool asm. The toolchain parses the FIRST register of a store as the
// data register and the LAST as the status register (asm7.go case 59), and
// the pair forms as (Rt1, Rt2) (case 58/59):
//
// STXR R3, (R1), R4 → c8047c23 (Rt=3, Rn=1, Rs=4)
// STXP (R3, R4), (R1), R5 → c8251023 (Rt=3, Rt2=4, Rn=1, Rs=5)
// LDXP (R1), (R3, R4) → c87f1023 (Rn=1, Rt=3, Rt2=4)
func TestArm64ExclNoOffset(t *testing.T) {
got := arm64Words(t, "\tLDXR (R1), R2\n\tSTXR R3, (R1), R4\n"+
"\tSTXP (R3, R4), (R1), R5\n\tSTXPW (R3, R4), (R1), R5\n"+
"\tLDXP (R1), (R3, R4)\n\tLDXPW (R1), (R3, R4)\n"+
"\tSTXR R3, (RSP), R4\n\tLDXR (RSP), R2\n")
want := []uint32{
0xc85f7c22, // LDXR X2, [X1]
0xc8047c23, // STXR W3, [X1], W4 with Rt = R3, Rs = R4
0xc8251023, // STXP (R3, R4), [X1], R5
0x88251023, // STXPW (R3, R4), [X1], R5
0xc87f1023, // LDXP [X1], (R3, R4)
0x887f1023, // LDXPW [X1], (R3, R4)
0xc8047fe3, // STXR R3, [SP], R4
0xc85f7fe2, // LDXR [SP], R2
0xd65f03c0, // RET
}
for i := range want {
if got[i] != want[i] {
t.Errorf("excl word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
// rejected instead of wrapping through int32.
func TestArm64AddSubImmRange(t *testing.T) {
for _, body := range []string{
"\tADD $0x100000000, R0, R1\n",
"\tSUB $-0x100000000, R0, R1\n",
"\tCMP $0x100000000, R0\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64LargeRegisterOffset pins the large-offset path for a register
// base: the ADD offsets from the operand's own base, not from SP, matching
// the toolchain's `ADD $(256<<12), R2, R27; MOVD (R27), R3`.
func TestArm64LargeRegisterOffset(t *testing.T) {
got := arm64Words(t, "\tMOVD 0x100000(R2), R3\n\tMOVD R3, 0x100000(R2)\n")
want := []uint32{
0x9144005b, // ADD $(256<<12), R2, R27
0xf9400363, // MOVD (R27), R3
0x9144005b, // ADD $(256<<12), R2, R27
0xf9000363, // MOVD R3, (R27)
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("large offset word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64LargeFrameSpadj checks the stack-adjustment boundaries of a frame
// whose autosize must be materialised into REGTMP: $5000 rounds the autosize
// to 5024, so the prologue is [MOVD $5024, R27][SUB R27, RSP, R20][STP][ADD
// R20, SP][SUB $8] and SP moves only at its fourth word, while the RET's
// epilogue is [LDP][MOVD $5024, R27][ADD R27, RSP, RSP] before the final
// RET. These PCs feed the DWARF CFA rules and the goobj stack maps.
func TestArm64LargeFrameSpadj(t *testing.T) {
f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), $5000-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
fn := img.Funcs[0]
// autosize 5024: class-2 guard of 6 words (24 bytes), a 5-word prologue
// whose ADD R20, SP sits at byte 8 inside it, a one-instruction body,
// then a 3-word epilogue before the final RET.
wantSpadj := []SpadjStep{{PC: 24 + 12, Value: 5024}, {PC: 24 + 20 + 4 + 12, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
// The words those PCs point between: the prologue's ADD R20, SP at byte
// 36, and the epilogue's materialised ADD R27, RSP, RSP right before the
// final RET at byte 60.
words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
if got := words[(24+12)/4]; got != 0x9100029f {
t.Errorf("prologue word at byte 36 = %08x, want 9100029f (ADD R20, SP)", got)
}
if got := words[(24+20+4+8)/4]; got != 0x8b3b63ff {
t.Errorf("epilogue word at byte 56 = %08x, want 8b3b63ff (ADD R27, RSP, RSP)", got)
}
if got := words[(24+20+4+12)/4]; got != 0xd65f03c0 {
t.Errorf("final RET word at byte 60 = %08x, want d65f03c0", got)
}
}
// TestArm64SplitFrameSpadj pins the addcon2 band, where neither imm12 form
// nor a single MOVZ carries the autosize and the toolchain splits the
// prologue SUB into two imm12 instructions (asm7.go case 48) while the
// non-leaf RET still materialises the value into REGTMP (obj7.go ARET,
// issue 73259). $65664 rounds the autosize to 65680 = 144 + 16<<12:
//
// [SUB $144, RSP, R20][SUB $(16<<12), R20, R20][STP][MOVD R20, SP][SUB $8]
// [CALL]
// [LDP][MOVD $144, R27][MOVK $(1<<16), R27][ADD R27, RSP, RSP][RET]
//
// SP moves at the fourth word (byte 12) and returns to zero at the final
// RET (byte 40); the words are go tool asm's own for the same source.
func TestArm64SplitFrameSpadj(t *testing.T) {
f, errs := parser.Parse("frame_arm64.s", "#include \"textflag.h\"\n\nTEXT ·framed(SB), NOSPLIT, $65664-0\n\tCALL ·other(SB)\n\tRET\n\nTEXT ·other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
fn := img.Funcs[0]
wantSpadj := []SpadjStep{{PC: 12, Value: 65680}, {PC: 40, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
want := []uint32{
0xd10243f4, // SUB $144, RSP, R20
0xd1404294, // SUB $(16<<12), R20, R20
0xa93ffa9d, // STP (R29, R30), -8(R20)
0x9100029f, // MOVD R20, RSP
0xd10023fd, // SUB $8, RSP, R29
0x94000000, // CALL (relocation masked at link time)
0xa97ffbfd, // LDP -8(RSP), (R29, R30)
0xd280121b, // MOVD $144, R27
0xf2a0003b, // MOVK $(1<<16), R27
0x8b3b63ff, // ADD R27, RSP, RSP
0xd65f03c0, // RET
}
words := leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
if len(words) != len(want) {
t.Fatalf("framed = %d words, want %d", len(words), len(want))
}
for i, w := range want {
if words[i] != w {
t.Errorf("word %d = %08x, want %08x", i, words[i], w)
}
}
}
+474
View File
@@ -0,0 +1,474 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 frame mapping, matching the Go toolchain's arm64 backend.
//
// Go's arm64 functions use R29 as the frame pointer (FP) and R30 as the link
// register (LR). R31 is the stack pointer (SP). FP and SP in the source
// are synthetic pseudo-registers resolved against the hardware SP and the
// frame size.
//
// The autosize is the real stack adjustment: the declared local frame plus
// 8 bytes for the saved link register, rounded up to a 16-byte multiple.
// The toolchain adds an "extrasize" to align: if autosize%16 == 8, add 8;
// if autosize%16 == 0, add 16.
//
// Prologue (autosize > 0, small frame ≤ 0xf0):
//
// MOVD.W LR, -autosize(SP) // pre-index: SP -= autosize, store LR at SP
// MOVD FP, -8(SP) // store FP at SP-8
// SUB $8, SP, FP // FP = SP - 8
//
// Prologue (autosize > 0, large frame > 0xf0):
//
// SUB $autosize, SP, R20 // R20 = SP - autosize
// STP (FP, LR), -8(R20) // store FP,LR at R20-8
// MOVD R20, SP // SP = R20
// SUB $8, SP, FP // FP = SP - 8
//
// Epilogue (non-leaf, small frame):
//
// ADD $autosize-8, SP, FP // restore FP
// ADD $autosize, SP, SP // deallocate frame
// MOVD -8(SP), FP // (actually the reverse of prologue)
// Actually:
// MOVD -8(SP), FP // load FP from SP-8
// MOVD.P autosize(SP), LR // post-index: load LR, SP += autosize
//
// Epilogue (non-leaf, large frame):
// ADD $autosize-8, SP, FP
// ADD $autosize, SP, SP
// Actually:
// LDP -8(SP), (FP, LR) // load FP,LR
// ADD $autosize, SP, SP // deallocate frame
//
// Epilogue (leaf with frame):
// ADD $autosize-8, SP, FP
// ADD $autosize, SP, SP
//
// RET always emits as BR LR (0xd65f03c0).
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
type arm64FrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR + alignment)
frame int // the declared $framesize
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
// Stack-split guard state: needSplit mirrors the toolchain, which skips
// the check for NOSPLIT functions and auto-marks leaf functions with an
// autosize below StackSmall as NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// arm64ComputeFrame derives the frame layout for a TEXT function.
func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
fi := arm64FrameInfo{
frame: frameSize(t),
args: argsSize(t),
}
for _, f := range t.Flags {
if f == "NOSPLIT" {
fi.noSplit = true
}
}
fi.leaf = arm64IsLeaf(t)
if fi.frame != 0 || !fi.leaf {
fi.autosize = fi.frame + 8 // space for the saved LR
// The toolchain always adds an extrasize: 8 when the total leaves a
// 16-byte alignment gap, another 16 when already aligned.
switch fi.autosize % 16 {
case 8:
fi.autosize += 8
case 0:
fi.autosize += 16
default:
// The toolchain rejects unaligned frames; round up so such
// sources still assemble.
fi.autosize += 16 - (fi.autosize % 16)
}
}
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
default:
fi.needSplit = true
switch {
case fi.autosize <= stackSmall:
fi.splitClass = 0
case fi.autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
// arm64GuardLen returns the byte length of the stack-split guard prefix
// (zero when the function needs no guard). The big class materialises
// framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies.
func arm64GuardLen(fi arm64FrameInfo) int {
if !fi.needSplit {
return 0
}
switch fi.splitClass {
case 0:
return 12
case 1:
return 16
default:
n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall))
if err != nil {
return 0
}
return 4 + n + 4 + 4 + 4 + 4
}
}
// arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that
// loads v into a register.
func arm64LoadImmLen(v int64) (int, error) {
b, err := encodeARM64LoadImm(27, v, "MOVD")
if err != nil {
return 0, err
}
return len(b), nil
}
// arm64IsLeaf reports whether a function contains no call instructions
// (BL/CALL), matching the toolchain's LEAF mark.
func arm64IsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "BL", "CALL":
return false
}
}
return true
}
// arm64Prologue returns the prologue bytes for an arm64 function.
func arm64Prologue(fi arm64FrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
if fi.autosize <= 0xf0 {
// Small frame: MOVD.W LR, -autosize(SP); MOVD FP, -8(SP); SUB $8, SP, FP
return a64WordsLE(
arm64PreStoreImm(3, 0, int32(-fi.autosize), 31, 30), // STR.W LR, -autosize(SP) (pre-index store)
arm64UnscaledStore(3, 0, -8, 31, 29), // STUR FP, [SP, #-8]
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
)
}
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
ws := arm64SubImmWords(uint32(fi.autosize), 20)
ws = append(ws,
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
)
return a64WordsLE(ws...)
}
// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
// two imm12 instructions instead of materialising it into REGTMP
// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
// unsigned and be neither encodable as one imm12 (checked by the callers
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
// logical immediate, because conclass tests all three before C_ADDCON2.
func arm64SplitImm12(imm uint32) bool {
if imm > 0xFFFFFF {
return false
}
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
return false
}
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
}
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
// immediate form when the value fits imm12 (plain, or shifted left by 12
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
// logical immediate, or one wider than 24 bits is materialised into REGTMP
// (R27) and subtracted in the extended-register form; everything else up to
// 0xFFFFFF is split into two imm12 instructions:
//
// SUB $(imm&0xfff), SP, Rd
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
}
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
}
if !arm64SplitImm12(imm) {
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
}
return []uint32{
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
}
}
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
// split and REGTMP ladder as arm64SubImmWords.
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
}
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
}
if !arm64SplitImm12(imm) {
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
}
return []uint32{
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
}
}
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
// large frame. The toolchain adds the frame back with a single instruction:
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
// materialised into REGTMP and added as a register, so the epilogue never
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
// the leaf epilogue's plain ADD instructions.
func arm64RetAddWords(autosize uint32) []uint32 {
if autosize < 1<<12 {
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
}
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
}
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
// deallocate the frame when present) followed by RET (BR LR).
func arm64Return(fi arm64FrameInfo) []byte {
var ws []uint32
if fi.autosize != 0 {
if fi.leaf {
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...)
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
} else if fi.autosize <= 0xf0 {
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
ws = append(ws,
arm64UnscaledLoad(3, 0, -8, 31, 29), // LDR FP, [SP, #-8]
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
)
} else {
// Large frame: LDP -8(SP), (FP, LR), then deallocate.
ws = append(ws,
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
)
ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
}
}
// RET: BR LR (0xd65f03c0)
ws = append(ws, a64UncondBranch(2, 30, 0)) // opc=2(RET), Rn=LR(30), Rd=0
return a64WordsLE(ws...)
}
// arm64PrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
if fi.autosize == 0 {
return 0
}
if fi.autosize <= 0xf0 {
return 4 // MOVD.W instruction decrements SP
}
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
// position depends on how many words the SUB itself took (immediate,
// shifted immediate, the two-word imm12 split, or a materialised REGTMP
// sequence).
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
}
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final RET instruction. The lengths are read from
// the same word-emitting helpers the epilogue uses rather than assumed: the
// leaf path shares the prologue's immediate ladder, and a materialised
// autosize costs its MOV words plus the ADD itself.
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
if fi.autosize == 0 {
return 0
}
if fi.leaf {
return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) +
len(arm64AddImmWords(uint32(fi.autosize), 31)))
}
if fi.autosize <= 0xf0 {
return 8 // LDR + LDR.P
}
// LDP + the deallocation emitted by arm64RetAddWords, so the length
// tracks whatever the MOVD ladder needs.
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
}
// arm64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x+N(SP) → (N + frame + 8)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
//
// The Go toolchain resolves all pseudo-register references against the
// hardware stack pointer (R31/SP): FP references add autosize+8 (the
// distance from SP after the prologue to the caller's argument area),
// SP references add frame+8 (the distance to the local area).
func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
switch sym.Pseudo {
case "FP":
return 31, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
return 31, int32(sym.Offset) + int32(fi.frame) + 8
case "SB":
return -1, int32(sym.Offset)
}
return -1, 0
}
// arm64PreStoreImm encodes a pre-index store (STR with writeback):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
3<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64UnscaledStore encodes an unscaled store (STUR):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64UnscaledStore(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 0<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64UnscaledLoad encodes an unscaled load (LDUR):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 0<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64UnscaledLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// arm64PostLoad encodes a post-index load (LDR with post-increment):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 0<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// Data-processing (shifted register) base opcodes for the guard blocks.
const (
arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24
arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24
arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24
)
// arm64DPSRWords builds one data-processing (shifted register) word:
// OP Rm, Rn, Rd in the Go assembler's operand order.
func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 {
return base | rm<<16 | rn<<5 | rd
}
// arm64DPExtWords builds one data-processing (extended register) word, the
// form the toolchain picks when a large immediate was materialised into
// REGTMP before the operation: base | 1<<21 | Rm<<16 | UXTX<<13 | Rn<<5 | Rd.
func arm64DPExtWords(base, rm, rn, rd uint32) uint32 {
return base | 1<<21 | rm<<16 | 3<<13 | rn<<5 | rd
}
// wordsOf converts little-endian instruction bytes back to words.
func wordsOf(b []byte) []uint32 {
ws := make([]uint32, 0, len(b)/4)
for i := 0; i+4 <= len(b); i += 4 {
ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24)
}
return ws
}
// arm64GuardBytes emits the stack-split guard prefix; blockStart is the
// function-relative byte address of the morestack block the branches target.
func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
// MOVD 16(R28), R16 (g.stackguard0)
ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)}
br := func(from int, cond uint32) uint32 {
return a64BranchCond(int32((blockStart-from)>>2), cond)
}
switch fi.splitClass {
case 0:
// CMP R16, RSP in the exact encoding go tool asm emits for it.
ws = append(ws, 0xeb3063ff)
ws = append(ws, br(8, a64CondLS))
case 1:
ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(12, a64CondLS))
default:
mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD")
if err != nil {
mov = nil
}
ws = append(ws, wordsOf(mov)...)
ml := len(mov) / 4
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
// The branches sit at fixed byte offsets in the guard prefix: after
// the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO,
// then a further B.LO word and the CMP for B.LS.
ws = append(ws, br(8+4*ml, a64CondLO))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(16+4*ml, a64CondLS))
}
return a64WordsLE(ws...)
}
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
// the R_CALLARM64 relocation.
func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) {
ws := []uint32{
1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3
a64Branch(1, 0), // BL, patched by the linker
}
bPC := blockStart + 8
ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry
reloc := Reloc{
Off: blockStart + 4,
After: blockStart + 8,
Name: "runtime\u00b7morestack_noctxt",
Kind: RelArm64Branch,
}
return a64WordsLE(ws...), reloc
}
+126
View File
@@ -0,0 +1,126 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// parseArm64File is a helper assembling one arm64 source file.
func parseArm64File(t *testing.T, src string) *Image {
t.Helper()
f, errs := parser.Parse("k_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
return img
}
// TestArm64RelocOffsetsIncludePrologue pins the function-relative relocation
// offsets of a framed function: the offsets used to exclude the prologue, so
// every relocation landed on a prologue instruction in the GOOBJ/ELF output.
// The function calls an external, so it is a non-leaf and carries the
// stack-split guard (12 bytes, small class) before the prologue.
func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7f(SB), $16-0\n"+
"\tBL ext\u00b7foo(SB)\n"+
"\tMOVD $gdata(SB), R5\n"+
"\tMOVD $extsym(SB), R6\n"+
"\tRET\n"+
"GLOBL gdata(SB), $8\n")
fn := img.Funcs[0]
// Layout: 12-byte guard, 12-byte prologue, BL (24), ADRP+ADD (28, 32),
// ADRP+ADD (36, 40), 12-byte epilogue with RET, 12-byte morestack block.
want := []struct {
off int
after int
name string
kind RelocKind
external bool
}{
{24, 28, "foo", RelArm64Branch, true},
{28, 28, "gdata", RelArm64Addr, false},
{32, 32, "gdata", RelArm64Addr, false},
{36, 36, "extsym", RelArm64Addr, true},
{40, 40, "extsym", RelArm64Addr, true},
{60, 64, "runtime\u00b7morestack_noctxt", RelArm64Branch, true},
}
if len(fn.Relocs) != len(want) {
t.Fatalf("relocs = %d, want %d", len(fn.Relocs), len(want))
}
for i, w := range want {
r := fn.Relocs[i]
if r.Off != w.off || r.After != w.after || r.Name != w.name || r.Kind != w.kind || r.External != w.external {
t.Errorf("reloc %d = {off %d after %d name %q kind %d ext %v}, want {off %d after %d name %q kind %d ext %v}",
i, r.Off, r.After, r.Name, r.Kind, r.External, w.off, w.after, w.name, w.kind, w.external)
}
}
// The BL with a zero offset sits exactly at the first reloc site.
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if w := binary.LittleEndian.Uint32(code[24:28]); w != 0x94000000 {
t.Errorf("BL word = %08x, want 94000000", w)
}
}
// TestArm64SBLoadStoreMatchesToolchain pins the ADRP scratch register
// (REGTMP, R27) and the LDST64 relocation kind for sym loads and stores,
// against the bytes go tool asm emits for MOVD sym(SB), R5.
func TestArm64SBLoadStoreMatchesToolchain(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
"\tMOVD sym(SB), R5\n"+
"\tMOVD R5, sym(SB)\n"+
"\tRET\n"+
"GLOBL sym(SB), $8\n")
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
// go tool asm: ADRP 0(PC), R27 (9000001b); MOVD (R27), R5 (f9400365);
// ADRP 0(PC), R27; MOVD R5, (R27) (f9000365).
for off, want := range map[int]uint32{0: 0x9000001b, 4: 0xf9400365, 8: 0x9000001b, 12: 0xf9000365} {
if got := binary.LittleEndian.Uint32(code[off : off+4]); got != want {
t.Errorf("word at %d = %08x, want %08x", off, got, want)
}
}
if len(fn.Relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
}
for i, w := range []struct{ off, after int }{{0, 8}, {8, 16}} {
r := fn.Relocs[i]
if r.Kind != RelArm64LDST64 {
t.Errorf("reloc %d kind = %d, want RelArm64LDST64 (%d)", i, r.Kind, RelArm64LDST64)
}
if r.Off != w.off || r.After != w.after {
t.Errorf("reloc %d = {off %d after %d}, want {off %d after %d}", i, r.Off, r.After, w.off, w.after)
}
}
}
// TestArm64GOObjRelocTypes checks that GOOBJ emission succeeds with the new
// relocation kinds in play; the detailed layout is covered by the goobj tests.
func TestArm64GOObjRelocTypes(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
"\tMOVD sym(SB), R5\n"+
"\tMOVD R5, sym(SB)\n"+
"\tRET\n"+
"GLOBL sym(SB), $8\n")
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
if len(obj) == 0 {
t.Fatal("empty object")
}
// The detailed layout is covered by the goobj tests; here we only pin
// that emission succeeds with the new relocation kinds in play.
}
+373 -16
View File
@@ -22,8 +22,13 @@ import (
// FP/SP frame-relative operands, and local-label jumps. SB (global symbol)
// operands require relocations and are not yet supported; the SIMD (VEX/AVX2)
// integer and shuffle/extract/permute/move set is in.
//
// Like the other architectures, the stack-growth guard (the morestack check
// in the prologue and the call back into the runtime in the epilogue) is not
// emitted: the bytes match go tool asm only for NOSPLIT functions or
// zero-frame leaves, where the toolchain emits no guard either.
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
code, _, labels, _, err := assemble(t, nil)
code, _, labels, _, _, err := assemble(t, nil)
return code, labels, err
}
@@ -31,7 +36,7 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
// the set of static symbols a GLOBL in the same file defines. A nil link
// rejects SB operands outright (single-function assembly cannot resolve
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
// file defines is recorded as an external relocation instead of failing —
// file defines is recorded as an external relocation instead of failing
// the object-file emitters resolve it at link time.
type linkInfo struct {
symbols map[string]bool
@@ -46,6 +51,7 @@ type sbPatch struct {
after int
name string
addend int64
kind RelocKind
}
// spadjStep is one stack-adjustment boundary within a function: Value is the
@@ -60,7 +66,7 @@ type spadjStep struct {
// assemble encodes a TEXT body, returning the machine code, the static-symbol
// patch sites (for the file-level layout to resolve), the label table and the
// stack-adjustment boundaries.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, error) {
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
fi := computeFrame(t)
chain := jumpChain(t)
resolve := func(name string) string {
@@ -70,13 +76,18 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
return name
}
// Layout: iterate jump sizes to a fixed point.
// Layout: iterate jump sizes to a fixed point. The stack-split guard
// prefix and the trailing morestack block participate in the iteration:
// their conditional branches relax from rel8 to rel32 when the body
// outgrows the short form.
long := make([]bool, len(t.Body))
sizes := make([]int, len(t.Body))
offsets := map[string]int{}
pcs := make([]int, len(t.Body))
var guardJBlong, guardJBElong, moreJMPlong bool
for {
pos := len(fi.prologue)
guard := fi.guardLen(guardJBlong, guardJBElong)
pos := guard + len(fi.prologue)
for i, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
@@ -84,13 +95,14 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
case *ast.Instr:
sz, err := instrSize(s, fi, long[i], link)
if err != nil {
return nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
sizes[i] = sz
pcs[i] = pos
pos += sz
}
}
bodyLen := pos - (guard + len(fi.prologue))
// Expand any short jump whose displacement no longer fits rel8.
changed := false
for i, stmt := range t.Body {
@@ -102,6 +114,11 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
if !isJumpMnemonic(mnem) || mnem == "CALL" || long[i] {
continue
}
// A zero-operand jump parses; its arity is reported during
// emission (encodeJump), so the layout must not index Operands.
if len(s.Operands) != 1 {
continue
}
name, ok := labelName(s.Operands[0])
if !ok {
continue // reported during emission
@@ -116,24 +133,77 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
changed = true
}
}
// The guard's conditional branches target the morestack block, which
// starts right after the body: the JBE measures from the end of the
// guard, so its displacement is the prologue plus the body.
if !guardJBElong && !fits8(int64(len(fi.prologue)+bodyLen)) {
guardJBElong = true
changed = true
}
if fi.splitClass == 2 && !guardJBlong {
// The underflow JB sits before the CMPQ; its displacement spans
// the rest of the guard plus the prologue and the body. The JB
// is still the short form this branch tests (relaxing it is this
// branch's job), so guardLen is taken with a short JB and the
// subtraction drops the prefix and the JB's own 2 bytes.
rest := fi.guardLen(false, guardJBElong) - (9 + 3 + 7 + 2)
if !fits8(int64(rest + len(fi.prologue) + bodyLen)) {
guardJBlong = true
changed = true
}
}
// The morestack JMP returns to the function start, so its
// displacement is the negated distance from its own end; while it is
// still short, its own length is 2 bytes.
if !moreJMPlong && !fits8(-int64(guard+len(fi.prologue)+bodyLen+5+2)) {
moreJMPlong = true
changed = true
}
if !changed {
break
}
}
// Pass 2: emit.
out := append([]byte(nil), fi.prologue...)
// Pass 2: emit. The guard comes first, then the prologue, the body and
// the morestack block.
guardLen := fi.guardLen(guardJBlong, guardJBElong)
bodyLen := 0
{
pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body {
if _, ok := stmt.(*ast.Instr); ok {
pos += sizes[i]
}
}
bodyLen = pos - (guardLen + len(fi.prologue))
}
var out []byte
var patches []sbPatch
if fi.needSplit {
// The JBE ends the guard, so its displacement is the prologue plus
// the body; the underflow JB additionally spans the trailing CMPQ and
// JBE, whose combined length is guardLen minus the prefix and the
// JB's own length (2 short, 6 long).
jbLen := 2
if guardJBlong {
jbLen = 6
}
guard, tlsPatch := buildGuard(fi, int32(len(fi.prologue)+bodyLen), int32(fi.guardLen(guardJBlong, guardJBElong)-(9+3+7+jbLen)+len(fi.prologue)+bodyLen))
out = append(out, guard...)
patches = append(patches, tlsPatch)
}
out = append(out, fi.prologue...)
var steps []spadjStep
var lines []LineEntry
if fi.useFP {
// PUSHQ BP saves the return-address-relative base (+8); the MOVQ
// changes nothing; SUBQ $size, SP completes the frame.
steps = append(steps,
spadjStep{1, 8},
spadjStep{len(fi.prologue), 8 + fi.size},
spadjStep{guardLen + 1, 8},
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
)
}
pos := len(fi.prologue)
pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
if !ok {
@@ -150,16 +220,38 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
}
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
if err != nil {
return nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
}
if len(code) != sizes[i] {
return nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
}
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
for k := range ps {
ps[k].kind = RelCall
}
}
patches = append(patches, ps...)
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
out = append(out, code...)
pos += len(code)
}
return out, patches, offsets, steps, nil
if fi.needSplit {
// The morestack block: CALL runtime.morestack_noctxt, then a JMP
// back to the function entry.
jmpLen := 2
if moreJMPlong {
jmpLen = 5
}
jmpDisp := -int64(pos + 5 + jmpLen)
suffix, callPatch := buildMoreStack(int32(jmpDisp))
callPatch.off += pos
callPatch.after = pos + 5
patches = append(patches, callPatch)
out = append(out, suffix...)
pos += len(suffix)
}
_ = pos
return out, patches, offsets, steps, lines, nil
}
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
@@ -222,16 +314,51 @@ type frameInfo struct {
spAdjust int64 // x-N(SP) becomes (spAdjust - N)(SP)
prologue []byte
epilogue []byte
// Stack-split guard state (matching the toolchain's stacksplit): needSplit
// is false for NOSPLIT functions and for leaf functions whose frame is
// below StackSmall, which the toolchain auto-marks NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
framesize int // the size the guard checks: frame+8 for framed functions
}
// Stack-frame size classes from runtime/stack.go.
const (
stackSmall = 128
stackBig = 4096
)
// sbPatch gains a kind so the emitters can tell CALL and TLS patches from
// plain PC-relative displacements.
// computeFrame derives the frame layout, matching the Go assembler's default
// (a frame pointer is used whenever the function has a non-zero frame).
// (a frame pointer is used whenever the function has a non-zero frame). It
// also decides whether the function needs the stack-split guard, mirroring
// obj6: a NOSPLIT function never splits, and a leaf function whose frame is
// below StackSmall is auto-marked NOSPLIT. One deliberate deviation: the
// toolchain treats zero-argument runtime calls (duffcopy and friends) as
// leaf-compatible; here any CALL makes the function a non-leaf.
func computeFrame(t *ast.Text) frameInfo {
fi := frameInfo{}
if t.Frame != nil && t.Frame.Imm.HasVal {
fi.size = int(t.Frame.Imm.Val)
}
if fi.size > 0 {
if fi.size == 0 && hasCall(t) {
// The toolchain gives a frameless function containing a CALL an
// 8-byte frame for the pushed base pointer: the prologue saves BP
// with no stack adjustment, every RET pops it back, FP references
// pass one extra slot, and the virtual SP is the hardware SP.
fi.size = 8
fi.useFP = true
// The push is the frame: the saved BP sits at SP+0 and the
// return address at SP+8, so arguments begin at SP+16. Unlike
// a SUBQ frame, the 8-byte size must not be added again.
fi.fpAdjust = 16
fi.spAdjust = 0
fi.prologue = []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
fi.epilogue = []byte{0x5D} // POPQ BP
} else if fi.size > 0 {
fi.useFP = true
fi.fpAdjust = int64(fi.size) + 16 // frame + saved BP + return address
fi.spAdjust = int64(fi.size)
@@ -240,9 +367,131 @@ func computeFrame(t *ast.Text) frameInfo {
} else {
fi.fpAdjust = 8 // return address only
}
noSplit := false
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
noSplit = true
}
}
// The toolchain's autoffset: the frame plus the saved base pointer.
framesize := fi.size
if framesize > 0 {
framesize += 8
}
switch {
case noSplit:
case framesize < stackSmall && !hasCall(t):
// Auto-NOSPLIT, as the toolchain's leaf search concludes.
default:
fi.needSplit = true
fi.framesize = framesize
switch {
case framesize <= stackSmall:
fi.splitClass = 0
case framesize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
// hasCall reports whether the function body contains a CALL instruction.
func hasCall(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
if strings.ToUpper(in.Mnemonic.Text) == "CALL" {
return true
}
}
return false
}
// guardLen returns the byte length of the stack-split guard prefix. The
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
// short form and 6 in the long form.
func (fi frameInfo) guardLen(jbLong, jbeLong bool) int {
if !fi.needSplit {
return 0
}
jb, jbe := 2, 2
if jbLong {
jb = 6
}
if jbeLong {
jbe = 6
}
switch fi.splitClass {
case 0:
return 9 + 4 + jbe
case 1:
return 9 + 8 + 4 + jbe
default:
return 9 + 3 + 7 + jb + 4 + jbe
}
}
// buildGuard emits the stack-split guard prefix. jbeDisp and jbDisp are the
// already-computed displacements of the conditional branches that jump to the
// morestack block (unused in classes without them). The TLS load carries a
// R_TLS_LE patch site at offset 5.
func buildGuard(fi frameInfo, jbeDisp, jbDisp int32) ([]byte, sbPatch) {
out := []byte{
0x64, 0x4c, 0x8b, 0x34, 0x25, // MOVQ FS:0, R14
0, 0, 0, 0, // TLS slot offset, filled by the linker
}
tls := sbPatch{off: 5, after: 9, kind: RelTLSLE}
jmp := func(op8, op32 byte, disp int32) []byte {
if disp >= -128 && disp <= 127 {
return []byte{op8, byte(disp)}
}
return append([]byte{0x0F, op32}, le32(int64(disp))...)
}
switch fi.splitClass {
case 0:
// CMPQ SP, 16(R14)
out = append(out, 0x49, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
case 1:
// LEAQ -(framesize-StackSmall)(SP), R12; CMPQ R12, 16(R14)
out = append(out, 0x4c, 0x8d, 0xa4, 0x24)
out = append(out, le32(-int64(fi.framesize-stackSmall))...)
out = append(out, 0x4d, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
default:
// MOVQ SP, R12; SUBQ $(framesize-StackSmall), R12; JB; CMPQ R12, 16(R14)
out = append(out, 0x49, 0x89, 0xe4)
out = append(out, 0x49, 0x81, 0xec)
out = append(out, le32(int64(fi.framesize-stackSmall))...)
out = append(out, jmp(0x72, 0x82, jbDisp)...)
out = append(out, 0x4d, 0x3b, 0x66, 0x10)
out = append(out, jmp(0x76, 0x86, jbeDisp)...)
}
return out, tls
}
// buildMoreStack emits the trailing block: CALL runtime.morestack_noctxt
// (patched by the linker) and a JMP back to the function start.
func buildMoreStack(jmpDisp int32) ([]byte, sbPatch) {
out := []byte{0xE8, 0, 0, 0, 0}
call := sbPatch{off: 1, after: 5, name: "runtime\u00b7morestack_noctxt", kind: RelCall}
out = append(out, jmpBytes(jmpDisp)...)
return out, call
}
// jmpBytes encodes a near JMP in the short or long form.
func jmpBytes(disp int32) []byte {
if disp >= -128 && disp <= 127 {
return []byte{0xEB, byte(disp)}
}
return append([]byte{0xE9}, le32(int64(disp))...)
}
// prologueBytes emits: PUSHQ BP; MOVQ SP, BP; SUBQ $size, SP.
func prologueBytes(size int) []byte {
out := []byte{0x55, 0x48, 0x89, 0xE5} // PUSHQ BP; MOVQ SP, BP
@@ -256,6 +505,8 @@ func epilogueBytes(size int) []byte {
}
func subSP(size int) []byte { // SUBQ $size, SP
// imm8 holds -128..127; anything larger takes the imm32 form, exactly as
// the Go assembler encodes it (verified for 8, 128, 200 and 255).
if size >= -128 && size <= 127 {
return []byte{0x48, 0x83, 0xEC, byte(int8(size))}
}
@@ -275,6 +526,16 @@ func addSP(size int) []byte { // ADDQ $size, SP
func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, error) {
mnem := strings.ToUpper(s.Mnemonic.Text)
if isJumpMnemonic(mnem) {
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
return 5, nil // opcode + rel32, always the long form
}
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
code, err := encodeIndirectJump(s, mnem)
if err != nil {
return 0, err
}
return len(code), nil
}
return jumpSize(mnem, long), nil
}
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
@@ -324,6 +585,33 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
var ps []sbPatch
var err error
if isJumpMnemonic(mnem) {
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
// CALL/JMP sym(SB): a rel32 call (or tail call) against a
// static or external symbol, resolved by the file-level layout
// or the linker.
code, ps, err = encodeSBCall(s, link)
if err != nil {
return nil, nil, err
}
for i := range ps {
ps[i].kind = RelCall
}
body := pc + len(prefix)
for i := range ps {
ps[i].off += body
ps[i].after = body + len(code)
}
return append(prefix, code...), ps, nil
}
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
// JMP/CALL through a register or memory: no relocation and no
// label to resolve, the operand fully determines the bytes.
code, err = encodeIndirectJump(s, mnem)
if err != nil {
return nil, nil, err
}
return append(prefix, code...), nil, nil
}
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
} else {
code, ps, err = encodeNormal(s, fi, link)
@@ -405,6 +693,37 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long
}
}
// isSBCall reports whether the CALL operand is a symbol reference.
func isSBCall(s *ast.Instr) bool {
return len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpAddr &&
s.Operands[0].Addr.Sym != nil && s.Operands[0].Addr.Sym.Pseudo == "SB"
}
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link)
if err != nil {
return nil, nil, err
}
m, ok := o.(sbMem)
if !ok {
return nil, nil, fmt.Errorf("CALL: unsupported operand")
}
opcode := []byte{0xE8}
if strings.ToUpper(s.Mnemonic.Text) == "JMP" {
opcode = []byte{0xE9} // a tail call, no return address pushed
}
e := &enc{}
if err := e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}}); err != nil {
return nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for i, p := range e.patches {
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelCall}
}
return e.out, ps, nil
}
// labelName extracts a local-label name from a jump operand.
func labelName(op *ast.Operand) (string, bool) {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
@@ -414,6 +733,44 @@ func labelName(op *ast.Operand) (string, bool) {
return "", false
}
// indirectJumpTarget reports whether the JMP/CALL operand addresses a
// register or a memory location rather than a label or a static symbol.
// A bare identifier is a register when the register table knows the name and
// a label otherwise, which is exactly how the parser cannot distinguish them.
func indirectJumpTarget(s *ast.Instr) bool {
if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr {
return false
}
a := s.Operands[0].Addr
if a.Base != "" || a.Index != "" {
return true
}
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
if _, ok := ParseReg(a.Sym.Name); ok {
return true
}
}
return false
}
// encodeIndirectJump assembles a JMP/CALL through a register or memory
// operand, which carries no relocation and no label to resolve.
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands {
o, err := operandFromAST(op, 8, frameInfo{}, nil)
if err != nil {
return nil, err
}
ops[i] = o
}
e := &enc{}
if err := e.encodeIndirectBranch(mnem, ops); err != nil {
return nil, err
}
return e.out, nil
}
// spReg is the hardware stack pointer used to realise FP/SP pseudo-operands.
var spReg = Reg{idx: 4, size: 8}
+125 -3
View File
@@ -4,6 +4,7 @@
package asm
import (
"bytes"
"strings"
"testing"
@@ -62,7 +63,7 @@ TEXT ·f(SB), NOSPLIT, $0
XORQ AX, AX
loop:
ADDQ $1, AX
CMPQ $10, AX
CMPQ AX, $10
JLT loop
RET
`)
@@ -159,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
}
}
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
// function containing a CALL receives: the PUSHQ BP prologue with no stack
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
// Go assembler produces. The push is the frame, so the offset must not count
// it twice.
func TestAssembleFramelessCall(t *testing.T) {
f, errs := parser.Parse("frameless_call_amd64.s", `
#include "textflag.h"
TEXT ·withcall(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
CALL ·other(SB)
MOVQ AX, ret+8(FP)
RET
TEXT ·other(SB), NOSPLIT, $0-0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
for _, r := range img.Funcs[0].Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
// From `go tool objdump` of the Go-assembled function:
// PUSHQ BP 55
// MOVQ SP, BP 4889e5
// MOVQ 0x10(SP), AX 488b442410
// CALL other e800000000
// MOVQ AX, 0x18(SP) 4889442418
// POPQ BP 5d
// RET c3
want := []byte{
0x55,
0x48, 0x89, 0xe5,
0x48, 0x8b, 0x44, 0x24, 0x10,
0xe8, 0x00, 0x00, 0x00, 0x00,
0x48, 0x89, 0x44, 0x24, 0x18,
0x5d,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
// the bytes the Go assembler produces.
@@ -201,8 +253,8 @@ TEXT ·withframe(SB), NOSPLIT, $16-16
}
// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac
// kernels end with — exercising the VEX moves, shuffle and extract forms
// through the full parser → encoder path — and checks the output is
// kernels end with; exercising the VEX moves, shuffle and extract forms
// through the full parser → encoder path; and checks the output is
// byte-identical to the Go assembler's.
func TestAssembleVexKernel(t *testing.T) {
fn := firstText(t, `
@@ -317,3 +369,73 @@ end:
t.Errorf("jump-folding mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
func TestAssemblePrefetch(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·pf(SB), NOSPLIT, $0
PREFETCHNTA (AX)
PREFETCHT0 (BX)
PREFETCHT1 8(CX)
PREFETCHT2 -1(AX)(R12*1)
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
got := strings.Join(disasm(t, code), "\n")
want := strings.Join([]string{
"prefetchnta zmmword ptr [rax]",
"prefetcht0 zmmword ptr [rbx]",
"prefetcht1 zmmword ptr [rcx+0x8]",
"prefetcht2 zmmword ptr [rax+r12-0x1]",
"ret",
}, "\n")
if got != want {
t.Errorf("prefetch disassembly mismatch:\n got:\n%s\n want:\n%s", got, want)
}
// Byte-level expectations: 0F 18 with the variant in the reg field.
if hex := hexBytes(code[:3]); hex != "0f 18 00" {
t.Errorf("PREFETCHNTA bytes: got %s, want 0f 18 00", hex)
}
if hex := hexBytes(code[3:6]); hex != "0f 18 0b" {
t.Errorf("PREFETCHT0 bytes: got %s, want 0f 18 0b", hex)
}
}
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
// the parser does not arity-check mnemonics) is rejected with an error rather
// than panicking in the layout loop, which indexes Operands[0] before the
// emission pass gets a chance to diagnose the arity.
func TestAssembleBareJump(t *testing.T) {
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
if _, _, err := Assemble(fn); err == nil {
t.Errorf("%s with no operand: expected an error, got none", mnem)
}
}
}
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
// larger. The intermediate 129..255 range used to encode an ADD with a
// truncated immediate, moving SP the wrong way.
func TestSubSPEncodings(t *testing.T) {
for _, tt := range []struct {
size int
want []byte
}{
{8, []byte{0x48, 0x83, 0xEC, 0x08}},
{127, []byte{0x48, 0x83, 0xEC, 0x7F}},
{128, []byte{0x48, 0x81, 0xEC, 0x80, 0x00, 0x00, 0x00}},
{200, []byte{0x48, 0x81, 0xEC, 0xC8, 0x00, 0x00, 0x00}},
{255, []byte{0x48, 0x81, 0xEC, 0xFF, 0x00, 0x00, 0x00}},
{4096, []byte{0x48, 0x81, 0xEC, 0x00, 0x10, 0x00, 0x00}},
} {
got := subSP(tt.size)
if !bytes.Equal(got, tt.want) {
t.Errorf("subSP(%d) = %x, want %x", tt.size, got, tt.want)
}
}
}
+80 -8
View File
@@ -12,12 +12,11 @@ import (
// Image: a .text section holding the function bodies, a .data section
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol
// a .rela.text relocation table, one R_X86_64_PC32 entry per static-symbol
// reference, internal references resolving against the local data symbols
// and external ones against undefined globals. The output links with the
// system toolchain (cc/ld) the way a hand-assembled .o would.
// ELF constants (ELF64, little-endian, System V).
const (
elfClass64 = 2
elfDataLSB = 1
@@ -36,18 +35,17 @@ const (
shfAlloc = 2
shfExecInstr = 4
stbLocal = 0
stbGlobal = 1
sttNotype = 0
sttObject = 1
sttFunc = 2
sttSection = 3
stInfoShift = 4
shnUndef = 0
rX8664PC32 = 2
// R_X86_64_TPOFF32 (debug/elf): the local-exec TLS offset the stack
// guard loads from FS. 20 is R_X86_64_TLSLD, a different relocation.
rX8664TPOFF32 = 23
)
// elfSym is one symbol-table entry in construction.
@@ -76,7 +74,7 @@ func (img *Image) ELFObject() ([]byte, error) {
// Build the symbol table: the null entry and the two section symbols
// come first, then the local symbols (static TEXT and GLOBL), then the
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF
// globals (exported TEXT and GLOBL, and the undefined externals), ELF
// requires every local to precede every global, and sh_info records the
// boundary. symIdx maps a symbol name to its index for the relocations.
var locals, globals []elfSym
@@ -130,11 +128,19 @@ func (img *Image) ELFObject() ([]byte, error) {
type elfRela struct {
off uint64
sym int
typ uint32
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
var typ uint32 = rX8664PC32
if r.Kind == RelTLSLE {
// R_X86_64_TPOFF32 resolves to the local-exec TLS offset and
// carries no symbol.
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), sym: 0, typ: rX8664TPOFF32})
continue
}
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
@@ -142,6 +148,7 @@ func (img *Image) ELFObject() ([]byte, error) {
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
sym: idx,
typ: typ,
// R_X86_64_PC32 computes S + A − P with P the patch site; the
// assembler measures the symbol from the instruction end,
// After − Off bytes past the field, so the addend carries
@@ -160,6 +167,9 @@ func (img *Image) ELFObject() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
// Section presence: .rela.text only when there are relocations.
hasRela := len(relas) > 0
@@ -211,7 +221,7 @@ func (img *Image) ELFObject() ([]byte, error) {
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
@@ -220,6 +230,34 @@ func (img *Image) ELFObject() ([]byte, error) {
shstrOff := len(out)
out = append(out, stSections.bytes()...)
// DWARF debug sections; the address placeholders they leave are carried
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiAMD64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rX8664Abs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
shoff := len(out)
@@ -248,6 +286,40 @@ func (img *Image) ELFObject() ([]byte, error) {
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
// The ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
+406
View File
@@ -0,0 +1,406 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
)
// DWARF5 section generation for ELF output. Unlike the GOOBJ path (where
// the linker assembles the final DWARF), the ELF path must emit complete,
// self-contained sections because the system linker only performs fixup
// relocations, not assembly.
// DWARF5 attribute, form and line-table constants (the values the
// toolchain uses, cmd/internal/dwarf/dwarf_defs.go; the DIE streams below
// are written against these forms).
const (
dwAtName = 0x03 // DW_AT_name
dwAtStmtList = 0x10 // DW_AT_stmt_list
dwAtLowPC = 0x11 // DW_AT_low_pc
dwAtHighPC = 0x12 // DW_AT_high_pc
dwAtDeclFile = 0x3a // DW_AT_decl_file
dwAtDeclLine = 0x3b // DW_AT_decl_line
dwAtExternal = 0x3f // DW_AT_external
dwAtFrameBase = 0x40 // DW_AT_frame_base
dwTagSubprog = 0x2e // DW_TAG_subprogram
dwTagCompUnit = 0x11 // DW_TAG_compile_unit
dwFormAddr = 0x01 // DW_FORM_addr
dwFormData8 = 0x07 // DW_FORM_data8
dwFormString = 0x08 // DW_FORM_string
dwFormData1 = 0x0b // DW_FORM_data1
dwFormUdata = 0x0f // DW_FORM_udata
dwFormSecOff = 0x17 // DW_FORM_sec_offset
dwFormExprloc = 0x18 // DW_FORM_exprloc
dwFormLineStrp = 0x1f // DW_FORM_line_strp
dwLnctPath = 0x01 // DW_LNCT_path
dwLnctDirIndex = 0x02 // DW_LNCT_directory_index
)
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries. The
// attribute/form pairs must match the DIE streams dwarfBuildInfoSection
// writes byte for byte, in the same order, or every consumer's parse of
// .debug_info desynchronises.
func dwarfAbbrevTable() []byte {
var b []byte
// Abbrev 1: DW_TAG_compile_unit.
b = append(b, 1) // abbreviation code
b = appendUleb(b, dwTagCompUnit) // DW_TAG_compile_unit
b = append(b, 1) // DW_CHILDREN_yes
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
b = appendUleb(b, dwFormData8) // DW_FORM_data8
b = appendUleb(b, dwAtStmtList) // DW_AT_stmt_list
b = appendUleb(b, dwFormSecOff) // DW_FORM_sec_offset (4 bytes here)
b = appendUleb(b, dwAtName) // DW_AT_name
b = appendUleb(b, dwFormString) // DW_FORM_string
b = appendUleb(b, 0) // end of attributes: attr 0
b = appendUleb(b, 0) // ... paired with form 0
// Abbrev 2: DW_TAG_subprogram.
b = append(b, 2) // abbreviation code
b = appendUleb(b, dwTagSubprog) // DW_TAG_subprogram
b = append(b, 0) // DW_CHILDREN_no
b = appendUleb(b, dwAtName) // DW_AT_name
b = appendUleb(b, dwFormString) // DW_FORM_string
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
b = appendUleb(b, dwFormData8) // DW_FORM_data8
b = appendUleb(b, dwAtFrameBase) // DW_AT_frame_base
b = appendUleb(b, dwFormExprloc) // DW_FORM_exprloc
b = appendUleb(b, dwAtDeclFile) // DW_AT_decl_file
b = appendUleb(b, dwFormData1) // DW_FORM_data1
b = appendUleb(b, dwAtDeclLine) // DW_AT_decl_line
b = appendUleb(b, dwFormData1) // DW_FORM_data1
b = appendUleb(b, dwAtExternal) // DW_AT_external
b = appendUleb(b, 0x0c) // DW_FORM_flag (one byte, 0 or 1)
b = appendUleb(b, 0) // end of attributes: attr 0
b = appendUleb(b, 0) // ... paired with form 0
// End of table.
b = append(b, 0)
return b
}
// dwarfSections holds the generated DWARF section payloads and their
// relocations (byte offsets within .debug_info and .debug_line that need
// fixup against .text symbols).
type dwarfSections struct {
debugAbbrev []byte
debugInfo []byte
debugLine []byte
debugLineStr []byte
debugFrame []byte
// Relocations for .debug_info: (offset, symbol name, addend).
infoRelocs []dwarfReloc
// Relocations for .debug_line: (offset, symbol name, addend).
lineRelocs []dwarfReloc
// Relocations for .debug_frame: (offset, symbol name, addend), one per
// FDE initial_location.
frameRelocs []dwarfReloc
}
type dwarfReloc struct {
off uint64
name string
addend int64
}
// emitDWARF generates complete DWARF5 sections for the image. cfi carries
// the architecture's .debug_frame register conventions.
func emitDWARF(img *Image, srcFile string, cfi cfiArch) *dwarfSections {
ds := &dwarfSections{}
ds.debugAbbrev = dwarfAbbrevTable()
// Build the string table for .debug_line_str.
lineStr := newElfStrtab()
lineStr.add(srcFile)
ds.debugLineStr = lineStr.bytes()
// Build .debug_line; the file table references the source name through
// its offset in .debug_line_str.
ds.debugLine = dwarfBuildLineSection(img, uint32(lineStr.at(srcFile)), ds)
// Build .debug_info.
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
// Build .debug_frame.
ds.debugFrame = dwarfBuildFrameSection(img, cfi, ds)
return ds
}
// dwarfBuildLineSection builds a complete .debug_line section. srcStrOff is
// the source file name's offset in .debug_line_str.
func dwarfBuildLineSection(img *Image, srcStrOff uint32, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
// We'll build the header first, then the programs, then patch the length.
headerStart := len(b)
b = append(b, 0, 0, 0, 0) // unit_length (placeholder)
b = le.AppendUint16(b, 5) // version (DWARF5)
b = append(b, 8) // address_size
b = append(b, 0) // segment_selector_size
b = append(b, 0, 0, 0, 0) // header_length (placeholder)
// Line program parameters.
b = append(b, 1) // minimum_instruction_length
b = append(b, 1) // maximum_ops_per_instruction
b = append(b, 1) // default_is_stmt
b = append(b, byte(dwLineBase&0xFF)) // line_base (-4 as unsigned)
b = append(b, uint8(dwLineRange)) // line_range
b = append(b, uint8(dwOpcodeBase)) // opcode_base
// Standard opcode lengths (opcode 1..opcode_base-1).
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
// Directory table (DWARF5 §6.2.4): entry format descriptors followed by
// the entries. One directory, the compilation directory, whose path is
// the empty string at .debug_line_str offset 0.
b = append(b, 1) // directory_entry_format_count
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
b = appendUleb(b, 1) // directories_count
b = le.AppendUint32(b, 0) // .debug_line_str offset of ""
// File table (DWARF5 §6.2.5). v5 indexes files from 0, so the source
// file is entry 0, matching the DW_AT_decl_file value 0 the DIEs carry.
b = append(b, 2) // file_name_entry_format_count
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
b = appendUleb(b, dwLnctDirIndex) // DW_LNCT_directory_index
b = appendUleb(b, dwFormUdata) // DW_FORM_udata
b = appendUleb(b, 1) // file_names_count
b = le.AppendUint32(b, srcStrOff) // .debug_line_str offset of the source name
b = appendUleb(b, 0) // directory index 0 (the compilation directory)
headerEnd := len(b)
// Per-function line programs.
for _, fn := range img.Funcs {
// LNE_set_address with the function's offset in .text.
b = append(b, 0, 9, 2) // extended opcode, length 9, DW_LNE_set_address
addrOff := len(b)
b = le.AppendUint64(b, 0) // placeholder for address
ds.lineRelocs = append(ds.lineRelocs, dwarfReloc{
off: uint64(addrOff),
name: fn.Name,
addend: 0,
})
// Build the line entries.
pts := make([]LineEntry, 0, len(fn.Lines)+1)
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
}
pts = append(pts, fn.Lines...)
line := int64(1)
pc := uint64(0)
for _, p := range pts {
if p.Line == 0 || uint64(p.Offset) < pc {
continue
}
if int64(p.Line) == line {
continue
}
deltaPC := uint64(p.Offset) - pc
deltaLC := int64(p.Line) - line
b = dwPutPCLCDelta(b, deltaPC, deltaLC)
line, pc = int64(p.Line), uint64(p.Offset)
}
// Advance to end of function.
if end := uint64(fn.Size) - pc; end > 0 {
b = append(b, 2) // DW_LNS_advance_pc
b = appendUleb(b, end)
}
b = append(b, 0, 1, 1) // LNE_end_sequence
}
// Patch unit_length.
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
// Patch header_length. In the v5 header it follows the one-byte
// address_size and segment_selector_size (offset 8, not the DWARF2-4
// offset 6), and counts from just past itself to the first program
// byte.
le.PutUint32(b[headerStart+8:], uint32(headerEnd-headerStart-12))
return b
}
// dwarfBuildInfoSection builds a complete .debug_info section.
func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
cuStart := len(b)
b = append(b, 0, 0, 0, 0) // unit_length (placeholder)
b = le.AppendUint16(b, 5) // version (DWARF5)
b = append(b, 0x01) // unit_type (DW_UT_compile)
b = append(b, 8) // address_size
b = le.AppendUint32(b, 0) // debug_abbrev_offset (0 since single CU)
// DW_TAG_compile_unit (abbrev 1).
b = append(b, 1) // abbreviation code
// DW_AT_low_pc: address of .text start. A data-only image has no
// functions to relocate against; its CU covers no code, so the base
// stays zero (the DWARF "no base address" value) with no relocation.
b = le.AppendUint64(b, 0) // placeholder
if len(img.Funcs) > 0 {
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
off: uint64(len(b) - 8),
name: img.Funcs[0].Name,
})
}
// DW_AT_high_pc: size of .text.
b = le.AppendUint64(b, uint64(len(img.Code)))
// DW_AT_stmt_list: offset into .debug_line (0).
b = le.AppendUint32(b, 0)
// DW_AT_name: source file name.
b = append(b, srcFile...)
b = append(b, 0)
// DW_TAG_subprogram entries (abbrev 2).
for _, fn := range img.Funcs {
b = append(b, 2) // abbreviation code
// DW_AT_name.
b = append(b, fn.Name...)
b = append(b, 0)
// DW_AT_low_pc.
addrOff := len(b)
b = le.AppendUint64(b, 0) // placeholder
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
off: uint64(addrOff),
name: fn.Name,
addend: 0,
})
// DW_AT_high_pc: function size.
b = le.AppendUint64(b, uint64(fn.Size))
// DW_AT_frame_base: DW_OP_call_frame_cfa.
b = append(b, 1, 0x9c)
// DW_AT_decl_file: the single file-table entry, index 0 (v5 indexes
// files from 0).
b = append(b, 0)
// DW_AT_decl_line.
b = append(b, uint8(fn.Line))
// DW_AT_external.
if fn.Static {
b = append(b, 0)
} else {
b = append(b, 1)
}
}
// End of compile unit children.
b = append(b, 0)
// Patch unit_length.
le.PutUint32(b[cuStart:], uint32(len(b)-cuStart-4))
return b
}
func appendUleb(b []byte, v uint64) []byte {
return binary.AppendUvarint(b, v)
}
// appendSleb appends v in signed LEB128, the encoding DWARF specifies:
// two's-complement sign extension, which is NOT Go's zigzag varint
// (binary.AppendVarint(-8) encodes 15, where DWARF wants 0x78).
func appendSleb(b []byte, v int64) []byte {
for {
c := byte(v & 0x7f)
v >>= 7
if (v == 0 && c&0x40 == 0) || (v == -1 && c&0x40 != 0) {
return append(b, c)
}
b = append(b, c|0x80)
}
}
// cfiArch carries the .debug_frame CIE parameters that differ per
// architecture: the DWARF register numbers of the stack pointer the initial
// CFA rule names and of the return address. The values are the ones the Go
// linker writes into its own CIE (cmd/link/internal/ld/dwarf.go uses
// Dwarfregsp and Dwarfreglr; the per-architecture constants live in
// cmd/link/internal/<arch>/l.go).
type cfiArch struct {
name string
cfaReg byte // the stack-pointer register the initial CFA rule names
raReg byte // the return-address register
}
var (
cfiAMD64 = cfiArch{"amd64", 7, 16} // RSP, RIP
cfiARM64 = cfiArch{"arm64", 31, 30} // SP (X31), LR (X30)
cfiRISCV64 = cfiArch{"riscv64", 2, 1} // X2 (sp), X1 (ra)
cfiLOONG64 = cfiArch{"loong64", 3, 1} // $r3 (sp), $r1 (ra)
)
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
// unwinding. It emits one CIE and one FDE per function, encoding the
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
// boundary recorded in FuncLayout.Spadj.
func dwarfBuildFrameSection(img *Image, cfi cfiArch, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
// CIE (Common Information Entry).
cieStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
b = append(b, 3) // version (DWARF3, widely supported)
b = append(b, 0) // augmentation (empty)
b = appendUleb(b, 1) // code alignment
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
b = appendUleb(b, uint64(cfi.raReg)) // return address register
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
b = append(b, 0x0c) // DW_CFA_def_cfa
b = appendUleb(b, uint64(cfi.cfaReg)) // the architecture's stack pointer
b = appendUleb(b, 0) // offset: 0
b = append(b, 0) // DW_CFA_nop (padding)
// Patch CIE length.
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
// FDEs (Frame Description Entries), one per function.
for _, fn := range img.Funcs {
fdeStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
// Initial location: function offset in .text, referenced through
// the function's symbol so the linker relocates it.
ds.frameRelocs = append(ds.frameRelocs, dwarfReloc{
off: uint64(fdeStart + 8),
name: fn.Name,
})
b = le.AppendUint64(b, uint64(fn.Offset))
// Address range: function size.
b = le.AppendUint64(b, uint64(fn.Size))
// Emit CFA rule changes at each Spadj boundary.
for _, step := range fn.Spadj {
if step.Value == 0 {
continue
}
// DW_CFA_def_cfa_offset: set CFA = SP + |delta|.
// The delta is negative (stack grows down), so CFA offset = -delta.
offset := -step.Value
if offset > 0 {
b = append(b, 0x0e) // DW_CFA_def_cfa_offset
b = appendUleb(b, uint64(offset))
}
}
// Pad to alignment.
for len(b)%4 != 0 {
b = append(b, 0) // DW_CFA_nop
}
// Patch FDE length.
le.PutUint32(b[fdeStart:], uint32(len(b)-fdeStart-4))
}
return b
}
+179
View File
@@ -0,0 +1,179 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "encoding/binary"
// Absolute 64-bit relocation types for the DWARF address fixups, one per
// supported architecture (the numbers debug/elf carries).
const (
rX8664Abs64 = 1 // R_X86_64_64
rAARCH64Abs64 = 257 // R_AARCH64_ABS64
rRISCVAbs64 = 2 // R_RISCV_64
rLarchAbs64 = 2 // R_LARCH_64
)
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
// in an ELF file.
type dwarfELFSections struct {
abbrevOff, abbrevSize int
infoOff, infoSize int
lineOff, lineSize int
lineStrOff, lineStrSize int
frameOff, frameSize int
// .rela.debug_info and .rela.debug_line contents: file offsets and
// entry counts (zero count: the section is absent).
infoRelaOff, infoRelaCount int
lineRelaOff, lineRelaCount int
frameRelaOff, frameRelaCount int
// Relocations for .debug_info address references, offsets relative to
// the section start (what an r_offset in .rela.debug_info means).
infoRelocs []elfDwarfReloc
// Relocations for .debug_line address references, section-relative.
lineRelocs []elfDwarfReloc
// Relocations for .debug_frame FDE initial locations, section-relative.
frameRelocs []elfDwarfReloc
}
type elfDwarfReloc struct {
off uint64 // offset within the target section
sym int // symbol index in .symtab
addend int64
}
// appendDWARFSections generates and appends DWARF5 debug sections to the ELF
// output. It returns the section offsets/sizes and relocations for the caller
// to emit section headers and relocation records.
//
// symIdx maps function names to their .symtab indices (needed for relocations
// against .text symbols). The map uses objectName format (pkg.name); the
// DWARF code uses bare function names, so we build a reverse lookup. cfi
// carries the architecture's .debug_frame register conventions.
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int), cfi cfiArch) *dwarfELFSections {
// Build a lookup from bare function name to symbol index.
nameToIdx := make(map[string]int, len(symIdx))
for name, idx := range symIdx {
// Strip package prefix: "pkg.name" → "name".
if i := len(name) - 1; i >= 0 {
for j := len(name) - 1; j >= 0; j-- {
if name[j] == '.' {
nameToIdx[name[j+1:]] = idx
break
}
}
}
nameToIdx[name] = idx
}
ds := emitDWARF(img, srcFile, cfi)
if ds == nil || len(ds.debugAbbrev) == 0 {
return nil
}
result := &dwarfELFSections{}
// .debug_abbrev
align(1)
result.abbrevOff = len(*out)
result.abbrevSize = len(ds.debugAbbrev)
*out = append(*out, ds.debugAbbrev...)
// .debug_line_str
align(1)
result.lineStrOff = len(*out)
result.lineStrSize = len(ds.debugLineStr)
*out = append(*out, ds.debugLineStr...)
// .debug_line
align(1)
result.lineOff = len(*out)
result.lineSize = len(ds.debugLine)
*out = append(*out, ds.debugLine...)
for _, dr := range ds.lineRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
off: dr.off,
sym: idx,
addend: dr.addend,
})
}
}
// .debug_info
align(1)
result.infoOff = len(*out)
result.infoSize = len(ds.debugInfo)
*out = append(*out, ds.debugInfo...)
for _, dr := range ds.infoRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
off: dr.off,
sym: idx,
addend: dr.addend,
})
}
}
// .debug_frame: the section header declares alignment 8, so the data is
// padded to 8, matching it.
if len(ds.debugFrame) > 0 {
align(8)
result.frameOff = len(*out)
result.frameSize = len(ds.debugFrame)
*out = append(*out, ds.debugFrame...)
for _, dr := range ds.frameRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.frameRelocs = append(result.frameRelocs, elfDwarfReloc{
off: dr.off,
sym: idx,
addend: dr.addend,
})
}
}
}
return result
}
// appendDWARFRelas writes the .rela.debug_info and .rela.debug_line section
// bodies from the relocations appendDWARFSections recorded, with the
// architecture's absolute 64-bit relocation type, and records their file
// offsets and entry counts on dw. Called after the DWARF sections
// themselves so the r_offsets (section-relative) need no adjustment.
func appendDWARFRelas(out *[]byte, dw *dwarfELFSections, abs64 uint32, align func(int)) {
le := binary.LittleEndian
write := func(relas []elfDwarfReloc) (off, count int) {
if len(relas) == 0 {
return 0, 0
}
align(8)
off = len(*out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(abs64))
le.PutUint64(b[16:], uint64(r.addend))
*out = append(*out, b[:]...)
}
return off, len(relas)
}
dw.infoRelaOff, dw.infoRelaCount = write(dw.infoRelocs)
dw.lineRelaOff, dw.lineRelaCount = write(dw.lineRelocs)
dw.frameRelaOff, dw.frameRelaCount = write(dw.frameRelocs)
}
// dwarfSourceName returns the source name the DWARF sections record: the
// image's source path when the assembler captured one, "gasm.s" otherwise.
func dwarfSourceName(img *Image) string {
if img.SourcePath != "" {
return img.SourcePath
}
return "gasm.s"
}
// dwarfSectionNames returns the DWARF section names for the string table.
var dwarfSectionNames = []string{
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
".debug_frame", ".rela.debug_info", ".rela.debug_line",
".rela.debug_frame",
}
+372
View File
@@ -0,0 +1,372 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
// encoding.
type ulebIter struct {
b []byte
i int
}
func (r *ulebIter) uleb(t *testing.T) uint64 {
t.Helper()
v, n := binary.Uvarint(r.b[r.i:])
if n <= 0 {
t.Fatalf("bad ULEB at %d", r.i)
}
r.i += n
return v
}
func (r *ulebIter) byteAt(t *testing.T) byte {
t.Helper()
if r.i >= len(r.b) {
t.Fatalf("read past end at %d", r.i)
}
c := r.b[r.i]
r.i++
return c
}
func (r *ulebIter) uint32At(t *testing.T) uint32 {
t.Helper()
v := binary.LittleEndian.Uint32(r.b[r.i:])
r.i += 4
return v
}
// sleb reads a signed LEB128, the DWARF encoding (sign-extended two's
// complement, not Go's zigzag varint).
func (r *ulebIter) sleb(t *testing.T) int64 {
t.Helper()
var v int64
var shift uint
for {
c := r.byteAt(t)
v |= int64(c&0x7f) << shift
shift += 7
if c&0x80 == 0 {
if c&0x40 != 0 {
v |= -1 << shift
}
return v
}
}
}
// dwarfAttr is one attribute/form pair of an abbreviation.
type dwarfAttr struct{ attr, form uint64 }
// dwarfAbbrev is one parsed abbreviation declaration.
type dwarfAbbrev struct {
code uint64
tag uint64
children bool
attrs []dwarfAttr
}
// parseAbbrevs walks a .debug_abbrev table: abbreviation code, tag,
// children flag, then attr/form ULEB pairs terminated by a double zero.
func parseAbbrevs(t *testing.T, b []byte) map[uint64]dwarfAbbrev {
t.Helper()
out := map[uint64]dwarfAbbrev{}
r := &ulebIter{b: b}
for {
code := r.uleb(t)
if code == 0 {
return out
}
ab := dwarfAbbrev{code: code, tag: r.uleb(t)}
ab.children = r.byteAt(t) == 1
for {
attr := r.uleb(t)
form := r.uleb(t)
if attr == 0 && form == 0 {
break
}
if attr == 0 || form == 0 {
t.Fatalf("abbrev %d: half-terminated attr/form pair (%d, %d)", code, attr, form)
}
ab.attrs = append(ab.attrs, dwarfAttr{attr, form})
}
out[code] = ab
}
}
func eqAttrs(t *testing.T, ab dwarfAbbrev, want []dwarfAttr) {
t.Helper()
if len(ab.attrs) != len(want) {
t.Fatalf("abbrev %d attrs = %v, want %v", ab.code, ab.attrs, want)
}
for i, w := range want {
if ab.attrs[i] != w {
t.Fatalf("abbrev %d attr %d = (%#x, %#x), want (%#x, %#x)", ab.code, i, ab.attrs[i].attr, ab.attrs[i].form, w.attr, w.form)
}
}
}
// TestDwarfAbbrevTable walks the abbreviation table as a consumer does and
// checks the attribute/form sets against the constants the toolchain uses
// (cmd/internal/dwarf/dwarf_defs.go). A wrong constant here renames an
// attribute (0x1b is comp_dir, not low_pc; 0x29 and 0x37 are bounds and
// count) and a wrong form desynchronises the DIE parse: 0x25 is strx1, one
// byte, where the writer emits four for a section offset.
func TestDwarfAbbrevTable(t *testing.T) {
abbrev := dwarfAbbrevTable()
if len(abbrev) == 0 {
t.Fatal("empty abbrev table")
}
// Must end with a zero byte (end of table).
if abbrev[len(abbrev)-1] != 0 {
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
}
abs := parseAbbrevs(t, abbrev)
if len(abs) != 2 {
t.Fatalf("abbreviations = %d, want 2", len(abs))
}
cu, ok := abs[1]
if !ok {
t.Fatal("missing abbreviation 1 (compile unit)")
}
if cu.tag != dwTagCompUnit || !cu.children {
t.Errorf("abbrev 1: tag %#x children %v, want compile unit with children", cu.tag, cu.children)
}
eqAttrs(t, cu, []dwarfAttr{
{dwAtLowPC, dwFormAddr},
{dwAtHighPC, dwFormData8},
{dwAtStmtList, dwFormSecOff},
{dwAtName, dwFormString},
})
sp, ok := abs[2]
if !ok {
t.Fatal("missing abbreviation 2 (subprogram)")
}
if sp.tag != dwTagSubprog || sp.children {
t.Errorf("abbrev 2: tag %#x children %v, want subprogram without children", sp.tag, sp.children)
}
eqAttrs(t, sp, []dwarfAttr{
{dwAtName, dwFormString},
{dwAtLowPC, dwFormAddr},
{dwAtHighPC, dwFormData8},
{dwAtFrameBase, dwFormExprloc},
{dwAtDeclFile, dwFormData1},
{dwAtDeclLine, dwFormData1},
{dwAtExternal, 0x0c}, // DW_FORM_flag
})
}
// TestDwarfLineHeaderV5 parses the .debug_line header under DWARF5 rules:
// the directory and file tables are format-descriptor lists, not the
// DWARF2-4 shape of null-terminated strings, and the file entry references
// the source name through .debug_line_str.
func TestDwarfLineHeaderV5(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
MOVQ AX, ret+16(FP)
RET
`
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
r := &ulebIter{b: ds.debugLine}
r.uint32At(t) // unit_length
if v := binary.LittleEndian.Uint16(ds.debugLine[4:]); v != 5 {
t.Fatalf("version = %d, want 5", v)
}
r.i = 6
r.byteAt(t) // address_size
r.byteAt(t) // segment_selector_size
r.uint32At(t) // header_length
r.byteAt(t) // minimum_instruction_length
r.byteAt(t) // maximum_ops_per_instruction
r.byteAt(t) // default_is_stmt
r.byteAt(t) // line_base
r.byteAt(t) // line_range
opcodeBase := r.byteAt(t)
for range int(opcodeBase) - 1 {
r.byteAt(t) // standard opcode lengths
}
// Directory table (DWARF5 §6.2.4).
if n := r.byteAt(t); n != 1 {
t.Fatalf("directory_entry_format_count = %d, want 1", n)
}
if lnct := r.uleb(t); lnct != dwLnctPath {
t.Errorf("directory content type = %#x, want DW_LNCT_path", lnct)
}
if form := r.uleb(t); form != dwFormLineStrp {
t.Errorf("directory form = %#x, want DW_FORM_line_strp", form)
}
if n := r.uleb(t); n != 1 {
t.Fatalf("directories_count = %d, want 1", n)
}
if off := r.uint32At(t); off != 0 {
t.Errorf("compilation directory line_strp = %d, want 0 (the empty string)", off)
}
// File table (DWARF5 §6.2.5).
if n := r.byteAt(t); n != 2 {
t.Fatalf("file_name_entry_format_count = %d, want 2", n)
}
if lnct := r.uleb(t); lnct != dwLnctPath {
t.Errorf("file content type = %#x, want DW_LNCT_path", lnct)
}
if form := r.uleb(t); form != dwFormLineStrp {
t.Errorf("file path form = %#x, want DW_FORM_line_strp", form)
}
if lnct := r.uleb(t); lnct != dwLnctDirIndex {
t.Errorf("file content type = %#x, want DW_LNCT_directory_index", lnct)
}
if form := r.uleb(t); form != dwFormUdata {
t.Errorf("file dir-index form = %#x, want DW_FORM_udata", form)
}
if n := r.uleb(t); n != 1 {
t.Fatalf("file_names_count = %d, want 1", n)
}
strOff := r.uint32At(t)
if dirIdx := r.uleb(t); dirIdx != 0 {
t.Errorf("file directory index = %d, want 0", dirIdx)
}
// The file entry's line_strp must resolve to the source name.
end := int(strOff) + len("test_amd64.s")
if int(strOff) >= len(ds.debugLineStr) || !bytes.Equal(ds.debugLineStr[strOff:end], []byte("test_amd64.s")) {
t.Errorf("file entry line_strp %d does not name the source: %q", strOff, ds.debugLineStr)
}
// The fixed header fields: address_size 8 and a header_length that
// points just past the file table (the patch site is offset 8 in the
// v5 header, and the field counts from its own end).
if ds.debugLine[6] != 8 || ds.debugLine[7] != 0 {
t.Errorf("address_size/segment_selector = %d/%d, want 8/0", ds.debugLine[6], ds.debugLine[7])
}
if hl := binary.LittleEndian.Uint32(ds.debugLine[8:]); hl != uint32(r.i-12) {
t.Errorf("header_length = %d, want %d (the byte after the file table is %d)", hl, r.i-12, r.i)
}
}
// TestDwarfFrameCIEArch checks the shared CIE carries each architecture's
// stack-pointer and return-address registers: the values the Go linker
// writes (cmd/link/internal/<arch>/l.go dwarfRegSP/dwarfRegLR).
func TestDwarfFrameCIEArch(t *testing.T) {
for _, tc := range []struct {
name string
cfi cfiArch
}{
{"amd64", cfiAMD64},
{"arm64", cfiARM64},
{"riscv64", cfiRISCV64},
{"loong64", cfiLOONG64},
} {
frame := dwarfBuildFrameSection(&Image{}, tc.cfi, &dwarfSections{})
r := &ulebIter{b: frame}
r.uint32At(t) // length
if cid := r.uint32At(t); cid != 0xFFFFFFFF {
t.Errorf("%s: CIE id = %#x, want 0xffffffff", tc.name, cid)
}
if v := r.byteAt(t); v != 3 {
t.Errorf("%s: CIE version = %d, want 3", tc.name, v)
}
if aug := r.byteAt(t); aug != 0 {
t.Errorf("%s: CIE augmentation = %d, want 0", tc.name, aug)
}
if ca := r.uleb(t); ca != 1 {
t.Errorf("%s: code alignment = %d, want 1", tc.name, ca)
}
if da := r.sleb(t); da != -8 {
t.Errorf("%s: data alignment = %d, want -8 (signed LEB128, not zigzag)", tc.name, da)
}
if ra := r.uleb(t); ra != uint64(tc.cfi.raReg) {
t.Errorf("%s: return-address register = %d, want %d", tc.name, ra, tc.cfi.raReg)
}
if op := r.byteAt(t); op != 0x0c {
t.Errorf("%s: expected DW_CFA_def_cfa, got opcode %#x", tc.name, op)
}
if cfa := r.uleb(t); cfa != uint64(tc.cfi.cfaReg) {
t.Errorf("%s: CFA register = %d, want %d", tc.name, cfa, tc.cfi.cfaReg)
}
if off := r.uleb(t); off != 0 {
t.Errorf("%s: CFA offset = %d, want 0", tc.name, off)
}
}
}
func TestEmitDWARF(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
MOVQ AX, ret+16(FP)
RET
`
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
// .debug_abbrev must not be empty and must start with abbrev code 1.
if len(ds.debugAbbrev) == 0 {
t.Fatal("empty .debug_abbrev")
}
if ds.debugAbbrev[0] != 1 {
t.Fatalf(".debug_abbrev first byte = %d, want 1", ds.debugAbbrev[0])
}
// .debug_info must have a compile unit header (DWARF5 version 5).
if len(ds.debugInfo) < 12 {
t.Fatalf(".debug_info too short: %d bytes", len(ds.debugInfo))
}
// Version field at offset 4 (after unit_length).
if ds.debugInfo[4] != 5 || ds.debugInfo[5] != 0 {
t.Fatalf(".debug_info version = %d, want 5", uint16(ds.debugInfo[4])|uint16(ds.debugInfo[5])<<8)
}
// .debug_line must have a header.
if len(ds.debugLine) < 20 {
t.Fatalf(".debug_line too short: %d bytes", len(ds.debugLine))
}
// Version at offset 4.
if ds.debugLine[4] != 5 || ds.debugLine[5] != 0 {
t.Fatalf(".debug_line version = %d, want 5", uint16(ds.debugLine[4])|uint16(ds.debugLine[5])<<8)
}
// .debug_line_str must contain the source file name.
if len(ds.debugLineStr) == 0 {
t.Fatal("empty .debug_line_str")
}
// Relocations must reference the function.
if len(ds.lineRelocs) == 0 {
t.Fatal("no .debug_line relocations")
}
if len(ds.infoRelocs) == 0 {
t.Fatal("no .debug_info relocations")
}
}
+444 -2
View File
@@ -12,6 +12,7 @@ import (
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
@@ -52,7 +53,7 @@ func elfTestImage(t *testing.T) *Image {
}
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
// defines is recorded as an external relocation instead of failing — the
// defines is recorded as an external relocation instead of failing; the
// raw image leaves the displacement zero, the object emitters carry it.
func TestAssembleFileExternals(t *testing.T) {
img := elfTestImage(t)
@@ -187,7 +188,7 @@ func TestELFObject(t *testing.T) {
end := bytes.IndexByte(strtabRaw[stName:], 0)
return string(strtabRaw[stName : int(stName)+end])
}
for i := 0; i < 2; i++ {
for i := range 2 {
e := raw[i*24 : (i+1)*24]
off := binary.LittleEndian.Uint64(e[0:])
info := binary.LittleEndian.Uint64(e[8:])
@@ -211,6 +212,75 @@ func TestELFObject(t *testing.T) {
}
}
// TestELFObjectTLSGuardReloc checks that a non-NOSPLIT function's stack
// guard carries an R_X86_64_TPOFF32 relocation against the null symbol in
// .rela.text. The serialisation must honour the record's type field: a
// hardcoded R_X86_64_PC32 mislinks the TLS load as an ordinary
// PC-relative reference.
func TestELFObjectTLSGuardReloc(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", `
#include "textflag.h"
TEXT ·grow(SB), $0
CALL ·other(SB)
RET
TEXT ·other(SB), NOSPLIT, $0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
var haveTLS bool
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.Kind == RelTLSLE {
haveTLS = true
}
}
}
if !haveTLS {
t.Fatal("test source produced no RelTLSLE relocation")
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
found := false
for i := 0; i+24 <= len(raw); i += 24 {
e := raw[i:]
info := binary.LittleEndian.Uint64(e[8:])
typ := info & 0xffffffff
sym := int(info >> 32)
if typ == uint64(elf.R_X86_64_TPOFF32) {
found = true
if sym != 0 {
t.Errorf("TPOFF32 relocation against symbol %d, want 0 (the null symbol)", sym)
}
}
}
if !found {
t.Errorf("no R_X86_64_TPOFF32 relocation in .rela.text (%d bytes)", len(raw))
}
}
// TestELFObjectNoRelocations checks a file with no static-symbol references
// emits a valid object without a .rela.text section.
func TestELFObjectNoRelocations(t *testing.T) {
@@ -253,6 +323,238 @@ TEXT ·nop(SB), NOSPLIT, $0
}
}
// elfSectionHeaderCount returns the e_shnum the ELF header declares.
func elfSectionHeaderCount(t *testing.T, obj []byte) int {
t.Helper()
return int(binary.LittleEndian.Uint16(obj[60:]))
}
// checkELFSectionAccounting verifies the number of section headers the
// writer physically laid out equals e_shnum: every DWARF section written
// after .shstrtab must be counted, or the last ones (always .debug_frame)
// are invisible to every consumer, debug/elf included.
func checkELFSectionAccounting(t *testing.T, obj []byte) {
t.Helper()
shoff := int(binary.LittleEndian.Uint64(obj[40:]))
shentsize := int(binary.LittleEndian.Uint16(obj[58:]))
shnum := elfSectionHeaderCount(t, obj)
if shentsize != 64 {
t.Fatalf("e_shentsize = %d, want 64", shentsize)
}
if (len(obj)-shoff)%shentsize != 0 {
t.Fatalf("section header table is not a whole number of entries: shoff=%d len=%d", shoff, len(obj))
}
if present := (len(obj) - shoff) / shentsize; present != shnum {
t.Errorf("e_shnum = %d but %d section headers are laid out", shnum, present)
}
}
// TestELFDWARFSectionAccounting runs the header accounting check over all
// four architecture emitters, and additionally checks the .debug_frame
// section is visible (its data aligned as its header declares).
func TestELFDWARFSectionAccounting(t *testing.T) {
parse := func(name, src string) *ast.File {
f, errs := parser.Parse(name, src)
if len(errs) > 0 {
t.Fatalf("parse %s: %v", name, errs)
}
return f
}
cases := []struct {
name string
img *Image
emit func(*Image) ([]byte, error)
}{
{"amd64", elfTestImage(t), (*Image).ELFObject},
{"arm64", mustImage(t, func() (*Image, error) {
return AssembleFileARM64(parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`))
}), (*Image).ELFAARCH64Object},
{"riscv64", mustImage(t, func() (*Image, error) {
return AssembleFileRISCV(parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`))
}), (*Image).ELFRISCVObject},
{"loong64", mustImage(t, func() (*Image, error) {
return AssembleFileLOONG64(parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`))
}), (*Image).ELFLOONG64Object},
}
for _, tc := range cases {
obj, err := tc.emit(tc.img)
if err != nil {
t.Fatalf("%s: emit: %v", tc.name, err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("%s: parse emitted object: %v", tc.name, err)
}
frame := ef.Section(".debug_frame")
if frame == nil {
t.Errorf("%s: .debug_frame invisible to debug/elf (e_shnum too small?)", tc.name)
ef.Close()
continue
}
if frame.Offset%8 != 0 || frame.Addralign != 8 {
t.Errorf("%s: .debug_frame offset %d align %d, want offset%%8==0 align 8", tc.name, frame.Offset, frame.Addralign)
}
ef.Close()
}
}
func mustImage(t *testing.T, f func() (*Image, error)) *Image {
t.Helper()
img, err := f()
if err != nil {
t.Fatal(err)
}
return img
}
// TestELFDWARFRelocations checks the .rela.debug_info and .rela.debug_line
// sections exist and carry absolute 64-bit relocations against the
// function symbols, with r_offsets inside their target sections.
func TestELFDWARFRelocations(t *testing.T) {
img := elfTestImage(t)
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
// The DWARF must record the assembled file's path (threaded through
// Image.SourcePath), not a placeholder name.
info, err := ef.Section(".debug_info").Data()
if err != nil {
t.Fatal(err)
}
if img.SourcePath != "t_amd64.s" || !bytes.Contains(info, []byte(img.SourcePath)) {
t.Errorf("DWARF compilation unit does not name the source %q", img.SourcePath)
}
for _, tc := range []struct {
rela string
target string
want uint32
}{
{".rela.debug_info", ".debug_info", rX8664Abs64},
{".rela.debug_line", ".debug_line", rX8664Abs64},
{".rela.debug_frame", ".debug_frame", rX8664Abs64},
} {
rs := ef.Section(tc.rela)
if rs == nil {
t.Fatalf("missing %s", tc.rela)
}
if rs.Type != elf.SHT_RELA {
t.Errorf("%s: type %v, want SHT_RELA", tc.rela, rs.Type)
}
target := ef.Section(tc.target)
if target == nil {
t.Fatalf("missing %s", tc.target)
}
if rs.Link == 0 || ef.Sections[rs.Info] != target {
t.Errorf("%s: link %d info %d, want the symtab and %s", tc.rela, rs.Link, rs.Info, tc.target)
}
b, err := rs.Data()
if err != nil {
t.Fatal(err)
}
// .debug_line has one address per function; .debug_info adds the
// compile unit's own low_pc.
want := len(img.Funcs)
if tc.target == ".debug_info" {
want++
}
if len(b)/24 != want {
t.Errorf("%s: %d entries, want %d", tc.rela, len(b)/24, want)
}
for i := 0; i+24 <= len(b); i += 24 {
r_offset := binary.LittleEndian.Uint64(b[i:])
info := binary.LittleEndian.Uint64(b[i+8:])
typ := uint32(info)
sym := int(info >> 32)
if typ != tc.want {
t.Errorf("%s entry %d: type %d, want R_X86_64_64 (%d)", tc.rela, i/24, typ, tc.want)
}
if r_offset >= uint64(target.Size) {
t.Errorf("%s entry %d: r_offset %d outside %s (%d bytes)", tc.rela, i/24, r_offset, tc.target, target.Size)
}
if sym == 0 {
t.Errorf("%s entry %d: against the null symbol", tc.rela, i/24)
}
}
}
}
// TestELFDataOnly checks a source with GLOBL data and no TEXT emits a valid
// ELF object: the DWARF compilation unit of a code-less image has no
// function to relocate against and must not reach for one.
func TestELFDataOnly(t *testing.T) {
f, errs := parser.Parse("d0_amd64.s", `
GLOBL table<>(SB), RODATA, $8
DATA table<>+0(SB)/8, $12345
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
found := false
for _, s := range syms {
if s.Name == "table" && s.Size == 8 {
found = true
}
}
if !found {
t.Errorf("data symbol table missing: %v", syms)
}
if ef.Section(".rela.debug_info") != nil || ef.Section(".rela.debug_line") != nil {
t.Error("data-only image must not emit DWARF address relocations")
}
}
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
// link the emitted object with a C driver that defines the external symbol,
// and run the result. Skipped when no C compiler is available.
@@ -307,4 +609,144 @@ int main(void) {
if got := string(run); got != "42 42 7\n" {
t.Errorf("output %q, want \"42 42 7\\n\"", got)
}
// The DWARF addresses must have resolved at link time: the .debug_info
// placeholders were carried by .rela.debug_info, so every subprogram's
// low_pc must now equal its linked symbol address.
bin, err := os.ReadFile(appPath)
if err != nil {
t.Fatal(err)
}
lef, err := elf.NewFile(bytes.NewReader(bin))
if err != nil {
t.Fatalf("parse linked binary: %v", err)
}
defer lef.Close()
syms, err := lef.Symbols()
if err != nil {
t.Fatal(err)
}
addrByName := map[string]uint64{}
for _, s := range syms {
if elf.ST_TYPE(s.Info) == elf.STT_FUNC && s.Value != 0 {
addrByName[s.Name] = s.Value
}
}
lowPCs := dwarfSubprogramLowPCs(t, lef)
if len(lowPCs) == 0 {
t.Fatal("no subprogram DW_AT_low_pc parsed from the linked binary")
}
for name, pc := range lowPCs {
addr, ok := addrByName[name]
if !ok {
t.Errorf("subprogram %q not in the linked symbol table", name)
continue
}
if pc != addr {
t.Errorf("subprogram %q: DW_AT_low_pc = %#x, linked address %#x (DWARF relocation unresolved)", name, pc, addr)
}
}
}
// dwarfSubprogramLowPCs walks the linked binary's .debug_info with its own
// .debug_abbrev and returns each DW_TAG_subprogram's DW_AT_low_pc by name.
func dwarfSubprogramLowPCs(t *testing.T, ef *elf.File) map[string]uint64 {
t.Helper()
abbrevSec := ef.Section(".debug_abbrev")
infoSec := ef.Section(".debug_info")
if abbrevSec == nil || infoSec == nil {
t.Fatal("linked binary lacks .debug_abbrev or .debug_info")
}
abbrev, err := abbrevSec.Data()
if err != nil {
t.Fatal(err)
}
info, err := infoSec.Data()
if err != nil {
t.Fatal(err)
}
abs := parseAbbrevs(t, abbrev)
le := binary.LittleEndian
out := map[string]uint64{}
r := &ulebIter{b: info}
r.uint32At(t) // unit_length
if v := le.Uint16(info[4:]); v != 5 {
t.Fatalf(".debug_info version %d, want 5", v)
}
r.i = 6
r.byteAt(t) // unit_type
r.byteAt(t) // address_size
r.uint32At(t) // debug_abbrev_offset
var name string
var lowPC uint64
for r.i < len(r.b) {
code := r.uleb(t)
if code == 0 {
continue // end of the CU's children
}
ab, ok := abs[code]
if !ok {
t.Fatalf("unknown abbreviation code %d", code)
}
name, lowPC = "", 0
for _, a := range ab.attrs {
switch a.attr {
case dwAtName:
readFormKeep(t, r, a.form, &name, nil)
case dwAtLowPC:
readFormKeep(t, r, a.form, nil, &lowPC)
default:
readFormSkip(t, r, a.form)
}
}
if ab.tag == dwTagSubprog && name != "" {
out[name] = lowPC
}
}
return out
}
// readFormKeep reads one DIE attribute value, keeping a string or an
// address into the pointer it was given (nil keeps nothing).
func readFormKeep(t *testing.T, r *ulebIter, form uint64, name *string, addr *uint64) {
t.Helper()
switch form {
case dwFormString:
end := r.i
for end < len(r.b) && r.b[end] != 0 {
end++
}
if name != nil {
*name = string(r.b[r.i:end])
}
r.i = end + 1
case dwFormAddr:
if addr != nil {
*addr = binary.LittleEndian.Uint64(r.b[r.i:])
}
r.i += 8
default:
readFormSkip(t, r, form)
}
}
func readFormSkip(t *testing.T, r *ulebIter, form uint64) {
t.Helper()
switch form {
case dwFormString:
for r.i < len(r.b) && r.b[r.i] != 0 {
r.i++
}
r.i++
case dwFormAddr, dwFormData8:
r.i += 8
case dwFormSecOff:
r.i += 4
case dwFormExprloc:
r.i += int(r.uleb(t))
case dwFormData1, 0x0c:
r.i++
default:
t.Fatalf("unsupported form %#x", form)
}
}
+313
View File
@@ -0,0 +1,313 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// AArch64 ELF64 relocatable object emission.
const (
emAARCH64 = 183 // EM_AARCH64
// AArch64 relocation types (the ELF psABI).
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
)
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
func (img *Image) ELFAARCH64Object() ([]byte, error) {
le := binary.LittleEndian
const (
secText = 1
secData = 2
)
// Build symbol table.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{},
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms)
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build relocations. Each SB reference is an ADRP pair:
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21 at the ADRP
// ADD → R_AARCH64_ADD_ABS_LO12_NC at the ADD word
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC at the LDR/STR word
// BL → R_AARCH64_CALL26
// cmd/link's own conversion emits the HI21 at sectoff and the LO12 at
// sectoff+4 (cmd/link/internal/arm64/asm.go), so the ADD or load word
// carries the page-offset relocation, never a second HI21. The
// assembler records two RelArm64Addr relocs per ADRP+ADD pair (one per
// word), so the second of the pair is consumed here.
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
// against S+A, and CALL26 branches take the branch instruction's own
// place as the PC-relative base, so subtracting the field width (the
// amd64 R_PCREL convention) would misplace every branch by 4 bytes.
type elfRela struct {
off uint64
typ uint32
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for i := 0; i < len(fn.Relocs); i++ {
r := fn.Relocs[i]
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
switch r.Kind {
case RelArm64Branch:
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off), typ: rArm64Call26, sym: idx, addend: r.Addend,
})
case RelArm64Addr:
// ADRP+ADD: the pair's second reloc (at Off+4) is the
// assembler's twin of the same pair; skip it.
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64AddAbsLo12NC, sym: idx, addend: r.Addend},
)
i++
case RelArm64LDST64:
// ADRP+LDR/STR: one assembler reloc covers the pair.
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64Ldst64Lo12NC, sym: idx, addend: r.Addend},
)
default:
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
}
}
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
// DWARF debug sections; the address placeholders they leave are carried
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiARM64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rAARCH64Abs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
shoff := len(out)
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0)
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
// ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emAARCH64)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
le.PutUint16(hdr[58:], 64)
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
+199
View File
@@ -0,0 +1,199 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
// relocatable object: sections, the symbol table (bindings, types, values,
// sizes) and the .rela.text relocation pair for the static-symbol load,
// parsed back with debug/elf.
func TestELFAARCH64Object(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
TEXT ·getanswer(SB), NOSPLIT, $0-8
MOVD answer<>(SB), R4
MOVD $answer<>(SB), R5
MOVD R4, ret+0(FP)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_AARCH64 {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_AARCH64", ef.Type, ef.Machine)
}
text := ef.Section(".text")
data := ef.Section(".data")
if text == nil || data == nil {
t.Fatal("missing .text or .data section")
}
if text.Size == 0 {
t.Error(".text section is empty")
}
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
foundAdd, foundGetanswer, foundAnswer := false, false, false
for _, s := range syms {
switch s.Name {
case "add":
foundAdd = true
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
t.Errorf("add: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
}
case "getanswer":
foundGetanswer = true
if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL {
t.Errorf("getanswer: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info)
}
case "answer":
foundAnswer = true
if elf.SymType(s.Info&0xf) != elf.STT_OBJECT || elf.SymBind(s.Info>>4) != elf.STB_LOCAL {
t.Errorf("answer: info=0x%02x, want STT_OBJECT|STB_LOCAL", s.Info)
}
}
}
if !foundAdd {
t.Error("symbol 'add' not found")
}
if !foundGetanswer {
t.Error("symbol 'getanswer' not found")
}
if !foundAnswer {
t.Error("symbol 'answer' not found")
}
// Check that .rela.text exists (getanswer has SB reference).
relaText := ef.Section(".rela.text")
if relaText == nil {
t.Fatal("missing .rela.text section")
}
// The SB references of getanswer form two ADRP pairs: the load
// (MOVD answer<>(SB), R4) is ADRP+LDR carrying HI21 at the ADRP and
// LDST64_ABS_LO12_NC at the LDR word, and the address-of
// (MOVD $answer<>(SB), R5) is ADRP+ADD carrying HI21 and
// ADD_ABS_LO12_NC. cmd/link's own conversion emits exactly this
// sectoff / sectoff+4 pairing; a second HI21 at the ADD or LDR word
// corrupts the pair.
raw, err := relaText.Data()
if err != nil {
t.Fatal(err)
}
if len(raw)%24 != 0 || len(raw)/24 != 4 {
t.Fatalf(".rela.text has %d bytes, want four 24-byte entries", len(raw))
}
wantRela := []struct {
typ elf.R_AARCH64
off uint64 // relative to the getanswer function start
}{
{elf.R_AARCH64_ADR_PREL_PG_HI21, 0},
{elf.R_AARCH64_LDST64_ABS_LO12_NC, 4},
{elf.R_AARCH64_ADR_PREL_PG_HI21, 8},
{elf.R_AARCH64_ADD_ABS_LO12_NC, 12},
}
getanswer := byNameElf(t, ef, "getanswer")
for i, w := range wantRela {
e := raw[i*24 : (i+1)*24]
off := binary.LittleEndian.Uint64(e[0:])
info := binary.LittleEndian.Uint64(e[8:])
typ := elf.R_AARCH64(info & 0xffffffff)
sym := int(info >> 32)
if typ != w.typ || off != getanswer.Value+w.off {
t.Errorf("reloc %d: type %v off %d, want %v at %d", i, typ, off, w.typ, getanswer.Value+w.off)
}
if sym != 3 { // NULL, .text, .data, then the first local: answer
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
}
}
}
// byNameElf returns the symbol table entry for name from the raw .symtab,
// which carries every entry including the null and section symbols in order.
func byNameElf(t *testing.T, ef *elf.File, name string) elf.Symbol {
t.Helper()
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
for _, s := range syms {
if s.Name == name {
return s
}
}
t.Fatalf("symbol %q not found", name)
return elf.Symbol{}
}
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
// static-symbol references (no .rela.text section).
func TestELFAARCH64ObjectNoRelocations(t *testing.T) {
f, errs := parser.Parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Section(".rela.text") != nil {
t.Error("unexpected .rela.text section when there are no relocations")
}
}
+295
View File
@@ -0,0 +1,295 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// LoongArch ELF64 relocatable object emission.
const (
emLOONGARCH = 258 // EM_LOONGARCH
// EF_LOONGARCH_ABI_DOUBLE_FLOAT | EF_LOONGARCH_OBJABI_V1: the flags the
// Go toolchain writes (cmd/link/internal/ld/elf.go: Flags = 0x43 for
// Loong64). System linkers refuse to merge ET_REL objects whose float
// ABI differs, so 0 (soft-float) would make the object unlinkable.
efLarchAbiDoubleObjV1 = 0x43
// LoongArch relocation types (the ELF psABI).
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
)
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
func (img *Image) ELFLOONG64Object() ([]byte, error) {
le := binary.LittleEndian
const (
secText = 1
secData = 2
)
// Build symbol table.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{},
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms)
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build relocations. Each SB reference is a pcalau12i pair:
// pcalau12i rd, 0 → R_LARCH_PCALA_HI20
// addi.d/ld/st → R_LARCH_PCALA_LO12
type elfRela struct {
off uint64
typ uint32
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
typ := uint32(rLarchPCALAHI20)
switch r.Kind {
case RelLoong64AddrLo:
typ = rLarchPCALALO12
case RelLoong64Branch:
typ = rLarchB26
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend,
})
}
}
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiLOONG64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rLarchAbs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
shoff := len(out)
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0)
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
// ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emLOONGARCH)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], efLarchAbiDoubleObjV1)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
le.PutUint16(hdr[58:], 64)
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
+247
View File
@@ -0,0 +1,247 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
// relocatable object: sections, the symbol table (bindings, types, values,
// sizes) and the .rela.text relocation pair for the static-symbol load,
// parsed back with debug/elf.
func TestELFLOONG64Object(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
TEXT ·getanswer(SB), NOSPLIT, $0-8
MOVV answer<>(SB), R4
MOVV R4, ret+0(FP)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
}
// The double-float ABI plus OBJABI_V1 flags the Go toolchain writes;
// system linkers refuse ABI-mismatched merges.
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efLarchAbiDoubleObjV1 {
t.Errorf("e_flags = %#x, want %#x (double-float, OBJABI_V1)", flags, efLarchAbiDoubleObjV1)
}
text := ef.Section(".text")
data := ef.Section(".data")
if text == nil || data == nil {
t.Fatal("missing .text or .data section")
}
if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 {
t.Errorf(".text flags = %v", text.Flags)
}
if data.Flags&elf.SHF_WRITE == 0 {
t.Errorf(".data flags = %v", data.Flags)
}
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
if !bytes.Equal(textData, img.Code) {
t.Errorf(".text contents differ from the image code")
}
dataData, err := data.Data()
if err != nil {
t.Fatal(err)
}
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
byName := map[string]elf.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ {
t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ)
}
if s.Section != section {
t.Errorf("%s: section = %v, want %v", name, s.Section, section)
}
if s.Size != size {
t.Errorf("%s: size = %d, want %d", name, s.Size, size)
}
}
if ef.Sections[1].Name != ".text" || ef.Sections[2].Name != ".data" {
t.Fatalf("section layout = %s, %s; want .text, .data", ef.Sections[1].Name, ef.Sections[2].Name)
}
textIdx := elf.SectionIndex(1)
dataIdx := elf.SectionIndex(2)
wantSym("add", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 20)
wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 16)
wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8)
// The data section carries 16-byte alignment padding; the answer
// symbol sits at its padded offset.
ans := byName["answer"]
if ans.Value+8 > uint64(len(dataData)) {
t.Fatalf("answer value %d outside .data (%d bytes)", ans.Value, len(dataData))
}
if got := dataData[ans.Value : ans.Value+8]; !bytes.Equal(got, []byte{42, 0, 0, 0, 0, 0, 0, 0}) {
t.Errorf("answer data = % x, want $42", got)
}
// Relocations: the static-symbol load is a pcalau12i+ld.d pair, so one
// R_LARCH_PCALA_HI20 and one R_LARCH_PCALA_LO12, both against the local
// data symbol. debug/elf does not surface rela entries, so read the
// section directly.
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
le := binary.LittleEndian
for i := range 2 {
e := raw[i*24 : (i+1)*24]
off := le.Uint64(e[0:])
info := le.Uint64(e[8:])
typ := info & 0xffffffff
sym := int(info >> 32)
if i == 0 && (typ != uint64(elf.R_LARCH_PCALA_HI20) || off != 20) {
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_HI20 at 20", i, typ, off)
}
if i == 1 && (typ != uint64(elf.R_LARCH_PCALA_LO12) || off != 24) {
t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_LO12 at 24", i, typ, off)
}
if sym != 3 { // NULL, .text, .data, then the first local: answer
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
}
}
}
// TestELFLOONG64ObjectNoRelocations checks a file with no static-symbol
// references emits a valid object without a .rela.text section.
func TestELFLOONG64ObjectNoRelocations(t *testing.T) {
f, errs := parser.Parse("n_loong64.s", `
#include "textflag.h"
TEXT ·nop(SB), NOSPLIT, $0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
if ef.Section(".rela.text") != nil {
t.Error("unexpected .rela.text section")
}
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
found := false
for _, s := range syms {
if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC {
found = true
}
}
if !found {
t.Error("function symbol nop not found")
}
}
// TestELFLOONG64BranchRelocation checks that the morestack call and an
// internal CALL both carry R_LARCH_B26 in the emitted object, matching the
// Go linker's mapping of its call relocation.
func TestELFLOONG64BranchRelocation(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
// The guard's morestack call plus the body's CALL to other.
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
le := binary.LittleEndian
for i := range 2 {
info := le.Uint64(raw[i*24+8:])
if elf.R_LARCH(info&0xffffffff) != elf.R_LARCH_B26 {
t.Errorf("relocation %d type = %v, want R_LARCH_B26", i, elf.R_LARCH(info&0xffffffff))
}
}
}
+307
View File
@@ -0,0 +1,307 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// RISC-V ELF64 relocatable object emission.
const (
emRISCV = 243 // EM_RISCV
// EF_RISCV_FLOAT_ABI_DOUBLE: the double-precision float ABI the Go
// toolchain targets (cmd/link/internal/ld/elf.go writes Flags = 0x4 for
// RISCV64). System linkers refuse to merge ET_REL objects whose float
// ABI differs, so 0 (soft-float) would make the object unlinkable.
efRISCVFloatAbiDouble = 0x4
// RISC-V relocation types.
rRISCVJAL = 17 // R_RISCV_JAL
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
)
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
func (img *Image) ELFRISCVObject() ([]byte, error) {
le := binary.LittleEndian
const (
secText = 1
secData = 2
)
// Build symbol table.
var locals, globals []elfSym
for _, fn := range img.Funcs {
s := elfSym{
name: objectName(fn.Pkg, fn.Name),
info: sttFunc,
shndx: secText,
value: uint64(fn.Offset),
size: uint64(fn.Size),
}
if fn.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := elfSym{
name: objectName(d.Pkg, d.Name),
info: sttObject,
shndx: secData,
value: uint64(d.Offset),
size: uint64(d.Size),
}
if d.Static {
locals = append(locals, s)
} else {
s.info |= stbGlobal << stInfoShift
globals = append(globals, s)
}
}
for _, name := range img.Externals {
globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift})
}
syms := []elfSym{
{},
{name: ".text", info: sttSection, shndx: secText},
{name: ".data", info: sttSection, shndx: secData},
}
syms = append(syms, locals...)
shInfo := len(syms)
syms = append(syms, globals...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Build relocations. Each SB reference is an AUIPC + second-instruction
// pair carrying a single relocation kind; the ELF writer expands it into
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
// The HI20 carries the symbol and its addend. The LO12's symbol must
// denote the AUIPC site the HI20 relocates (psABI §8.4.9: the pair is
// resolved against the label of the AUIPC, not the target symbol;
// cmd/link generates one local text symbol per AUIPC for exactly this,
// cmd/link/internal/riscv64/asm.go). The .text section symbol with the
// AUIPC's section-relative offset as addend gives S + A = the AUIPC
// address, which is that label.
const secSymText = 1 // syms[1], the .text section symbol
type elfRela struct {
off uint64
typ uint32
sym int
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
auipc := int64(fn.Offset + r.Off)
switch r.Kind {
case RelRISCVPCRELIType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: secSymText, addend: auipc},
)
case RelRISCVPCRELSType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: secSymText, addend: auipc},
)
case RelRISCVJal:
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
default:
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
}
}
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
stNames.add(s.name)
}
stSections := newElfStrtab()
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
align := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
align(16)
textOff := len(out)
out = append(out, img.Code...)
align(16)
dataOff := len(out)
out = append(out, img.Data...)
align(8)
symtabOff := len(out)
for _, s := range syms {
var b [24]byte
le.PutUint32(b[0:], uint32(stNames.at(s.name)))
b[4] = s.info
b[5] = 0
le.PutUint16(b[6:], s.shndx)
le.PutUint64(b[8:], s.value)
le.PutUint64(b[16:], s.size)
out = append(out, b[:]...)
}
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiRISCV64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rRISCVAbs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
shoff := len(out)
putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) {
var b [64]byte
le.PutUint32(b[0:], uint32(stSections.at(name)))
le.PutUint32(b[4:], uint32(typ))
le.PutUint64(b[8:], flags)
le.PutUint64(b[16:], 0)
le.PutUint64(b[24:], uint64(off))
le.PutUint64(b[32:], uint64(size))
le.PutUint32(b[40:], uint32(link))
le.PutUint32(b[44:], uint32(info))
le.PutUint64(b[48:], alignV)
le.PutUint64(b[56:], entsize)
out = append(out, b[:]...)
}
putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0)
putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0)
putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0)
putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24)
putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0)
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
// ELF header.
hdr := out[:64]
copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0})
le.PutUint16(hdr[16:], etREL)
le.PutUint16(hdr[18:], emRISCV)
le.PutUint32(hdr[20:], elfVersion)
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], efRISCVFloatAbiDouble)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
le.PutUint16(hdr[58:], 64)
le.PutUint16(hdr[60:], uint16(nSections))
le.PutUint16(hdr[62:], uint16(secShstr))
return out, nil
}
+101
View File
@@ -0,0 +1,101 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "strings"
// Encodable reports whether the amd64 encoder knows how to encode the
// mnemonic. It mirrors the dispatch in (*enc).encode: the fixed-name
// instructions, conditional jumps, the CMOV/SET condition families, the
// VEX/EVEX/opmask/gather/scatter vector paths, the legacy SSE tables and the
// explicit scalar cases. A mnemonic that parses (is in the architecture
// table) but is not encodable would otherwise surface only at assembly time,
// deep inside a build; the linter uses this predicate to flag it at edit
// time.
func Encodable(mnemonic string) bool {
upper := strings.ToUpper(mnemonic)
// Fixed-name instructions (no size suffix).
switch upper {
case "RET", "NOP", "CALL", "JMP":
return true
}
if _, ok := condCode(upper); ok {
return true
}
// VEX/EVEX and friends: the trailing B/W/L/Q/D is part of the mnemonic.
base, _, err := parseEvexSuffix(upper)
if err != nil {
return false
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" {
return true
}
// CMOV carries size then condition (CMOVLGT); SET carries the condition
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
// so a spelling like CMOVBGT is not reported encodable when Encode
// would reject it.
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
switch rest[0] {
case 'W', 'L', 'Q':
if _, ok := jccMap[rest[1:]]; ok {
return true
}
}
}
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
if _, ok := jccMap[rest]; ok {
return true
}
}
// Legacy SSE shuffles and packed binaries dispatch on the full name.
if _, ok := sseShufTable[upper]; ok {
return true
}
if _, ok := sseBinTable[upper]; ok {
return true
}
// The size-suffix split: retry the tables and the scalar switch on the
// base.
base2, size := splitSize(upper)
if size == 0 {
size = 8
}
_ = size
if base2 != upper {
if _, ok := sseBinTable[base2]; ok {
return true
}
}
switch base2 {
case "MOV",
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
"TEST",
"LEA",
"INC", "DEC", "NEG", "NOT",
"SHL", "SHR", "SAR",
"IMUL", "IMUL3",
"PUSH", "POP",
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
"BSWAP",
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
"CVTSL2SD", "CVTSQ2SD",
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return true
}
// Full-name dispatches the size split would eat (a trailing width
// letter that is part of the mnemonic).
switch upper {
case "PMOVMSKB":
return true
}
return false
}
+82 -10
View File
@@ -40,10 +40,20 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeRet()
case upper == "NOP":
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
case upper == "CALL":
return e.encodeJmpRel(ops, []byte{0xE8})
case upper == "JMP":
return e.encodeJmpRel(ops, []byte{0xE9})
case upper == "CALL" || upper == "JMP":
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
// Anything else is a rel32 against a label resolved by the assembler.
if len(ops) == 1 {
switch ops[0].(type) {
case Reg, Mem:
return e.encodeIndirectBranch(upper, ops)
}
}
opcode := []byte{0xE8}
if upper == "JMP" {
opcode = []byte{0xE9}
}
return e.encodeJmpRel(ops, opcode)
}
if cc, ok := condCode(upper); ok {
return e.encodeJcc(cc, ops)
@@ -76,6 +86,26 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if size == 0 {
size = 8 // default operand size in 64-bit mode (e.g. PUSHQ)
}
// Legacy SSE imm8 shuffles whose names end in W/H (PSHUFLW,
// PSHUFHW) must dispatch BEFORE the size-suffix split, and the
// others ride along.
if m, ok := sseShufTable[upper]; ok {
return e.encodeSSEShuf(m, ops)
}
// Legacy SSE packed binaries dispatch on the full name: the packed
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
// which the size split must not eat.
if m, ok := sseBinTable[upper]; ok {
return e.encodeSSEBin(m, ops)
}
if m, ok := sseBinTable[base]; ok {
return e.encodeSSEBin(m, ops)
}
// PMOVMSKB ends in a width letter the size split would eat, so it
// dispatches on the full name like the packed binaries above.
if upper == "PMOVMSKB" {
return e.encodePmovmskb(upper, ops)
}
switch base {
case "MOV":
return e.encodeMov(ops, size)
@@ -92,12 +122,17 @@ func (e *enc) encode(mnem string, ops []Operand) error {
case "IMUL", "IMUL3":
return e.encodeImul(ops, size)
case "PUSH":
return e.encodePushPop(ops, true)
return e.encodePushPop(ops, size, true)
case "POP":
return e.encodePushPop(ops, false)
case "LZCNT", "TZCNT":
return e.encodePushPop(ops, size, false)
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
return e.encodeCount(base, ops, size)
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
case "BSWAP":
return e.encodeBswap(ops, size)
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
return e.encodePrefetch(base, ops)
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX":
return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD":
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
@@ -107,6 +142,30 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return fmt.Errorf("unsupported instruction %q", mnem)
}
// encodePrefetch emits the 0F 18 /r prefetch hints: the reg field selects
// the locality (NTA=0, T0=1, T1=2, T2=3) and the single operand is memory.
func (e *enc) encodePrefetch(base string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects one memory operand", base)
}
m, ok := ops[0].(Mem)
if !ok {
return fmt.Errorf("%s requires a memory operand", base)
}
i := newInstr(0, []byte{0x0F, 0x18})
if err := setMem(i, prefetchVariant[base], m); err != nil {
return err
}
return e.emit(i)
}
var prefetchVariant = map[string]int{
"PREFETCHNTA": 0,
"PREFETCHT0": 1,
"PREFETCHT1": 2,
"PREFETCHT2": 3,
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) {
if upper == "" {
@@ -241,7 +300,7 @@ func setRM(i *instr, reg Reg, rm Operand, opSize int) error {
}
// setRMDigit fills in the ModR/M for an instruction whose reg field is an
// opcode /digit extension (0–7), which carries none of the register REX rules.
// opcode /digit extension (0-7), which carries none of the register REX rules.
func setRMDigit(i *instr, digit int, rm Operand, opSize int) error {
return setRMReg(i, digit, false, false, rm, opSize)
}
@@ -292,11 +351,24 @@ func setMem(i *instr, regField int, m Mem) error {
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
sib = -1
// A displacement wider than int32 fits no encoding form; truncating it
// would address a different location, and go tool asm reports "offset
// too large" for the same operand.
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
}
// RIP-relative: neither base nor index.
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
}
// The SIB scale field only encodes 1/2/4/8; the Go assembler rejects
// anything else ("bad scale: 16"), so a silent fallback to scale 1 here
// would mis-assemble the operand instead of reporting it.
if m.HasIndex && m.Scale != 1 && m.Scale != 2 && m.Scale != 4 && m.Scale != 8 {
return 0, -1, nil, 0, 0, fmt.Errorf("bad scale: %d", m.Scale)
}
needSIB := m.HasIndex || (m.HasBase && m.Base.idx&7 == 4)
var mod int
@@ -369,7 +441,7 @@ func le16(v int64) []byte {
func le64(v int64) []byte {
u := uint64(v)
b := make([]byte, 8)
for i := 0; i < 8; i++ {
for i := range 8 {
b[i] = byte(u >> (8 * i))
}
return b
+337 -4
View File
@@ -4,6 +4,7 @@
package asm
import (
"fmt"
"strings"
"testing"
@@ -57,8 +58,11 @@ func TestMov(t *testing.T) {
checkSyntax(t, "mov qword ptr [rbx], rax", "MOVQ", AX, Ptr(BX, 0, 8))
checkSyntax(t, "mov rbx, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), BX)
checkSyntax(t, "mov rbx, qword ptr [rsi+4*rbx]", "MOVQ", Idx(SI, BX, 4, 0, 8), BX)
checkSyntax(t, "mov rax, 0x5", "MOVQ", Imm(5), AX)
checkSyntax(t, "mov r8, 0x5", "MOVQ", Imm(5), Reg{idx: 8, size: 8})
// A small positive immediate compresses to the 32-bit zero-extending
// form (matching go tool asm), so the disassembler renders the 32-bit
// register name even for MOVQ.
checkSyntax(t, "mov eax, 0x5", "MOVQ", Imm(5), AX)
checkSyntax(t, "mov r8d, 0x5", "MOVQ", Imm(5), Reg{idx: 8, size: 8})
checkSyntax(t, "mov qword ptr [rax], 0x5", "MOVQ", Imm(5), Ptr(AX, 0, 8))
checkSyntax(t, "mov r12, r13", "MOVQ", Reg{idx: 13, size: 8}, Reg{idx: 12, size: 8})
}
@@ -74,13 +78,63 @@ func TestALU(t *testing.T) {
checkSyntax(t, "cmp rsi, r10", "CMPQ", SI, Reg{idx: 10, size: 8})
checkSyntax(t, "add rbx, qword ptr [rax]", "ADDQ", Ptr(AX, 0, 8), BX)
checkSyntax(t, "add qword ptr [rax], rbx", "ADDQ", BX, Ptr(AX, 0, 8))
checkSyntax(t, "cmp rbx, -0x20", "CMPQ", Imm(-32), BX)
// The Go assembler rejects the immediate-first CMP spelling outright,
// so Encode errors instead of silently emitting the swapped form.
if _, err := Encode("CMPQ", Imm(-32), BX); err == nil {
t.Errorf("Encode(CMPQ imm-first) should error, got success")
}
// The Go assembler's own spelling: immediate second.
checkSyntax(t, "cmp ecx, 0x1f", "CMPL", CX, Imm(31))
checkSyntax(t, "cmp ecx, -0x80000000", "CMPL", CX, Imm(-2147483648))
checkSyntax(t, "cmp r9, -0x80000000", "CMPQ", Reg{idx: 9, size: 8}, Imm(-2147483648))
}
// TestScalarXmmRegMoves pins the Go-assembler byte forms of scalar
// MOVQ/MOVL between GPRs and XMM registers (66 REX.W 0F 6E/0F 7E) and the
// memory forms (F3 0F 7E load, 66 0F D6 store), all byte-for-byte.
func TestScalarXmmRegMoves(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MOVQ AX,X1", "MOVQ", []Operand{AX, vreg(t, "X1")}, "66480f6ec8"},
{"MOVQ DX,X2", "MOVQ", []Operand{DX, vreg(t, "X2")}, "66480f6ed2"},
{"MOVQ X1,AX", "MOVQ", []Operand{vreg(t, "X1"), AX}, "66480f7ec8"},
{"MOVQ X0,DX", "MOVQ", []Operand{vreg(t, "X0"), DX}, "66480f7ec2"},
{"MOVL AX,X1", "MOVL", []Operand{AX, vreg(t, "X1")}, "660f6ec8"},
{"MOVL X1,AX", "MOVL", []Operand{vreg(t, "X1"), AX}, "660f7ec8"},
{"MOVQ (SI),X1", "MOVQ", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f30f7e0e"},
{"MOVQ X3,(DI)", "MOVQ", []Operand{vreg(t, "X3"), Ptr(DI, 0, 8)}, "660fd61f"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestBadScale pins the go-tool-asm parity of rejecting SIB scales the
// hardware cannot encode.
func TestBadScale(t *testing.T) {
for _, sc := range []int{3, 5, 16, 32} {
if _, err := Encode("LEAQ", Idx(SI, BX, sc, 0, 8), AX); err == nil {
t.Errorf("LEAQ scale %d: expected error, got success", sc)
}
}
for _, sc := range []int{1, 2, 4, 8} {
if _, err := Encode("LEAQ", Idx(SI, BX, sc, 0, 8), AX); err != nil {
t.Errorf("LEAQ scale %d: %v", sc, err)
}
}
}
func TestLea(t *testing.T) {
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
checkSyntax(t, "lea rax, ptr [rbx+0x8]", "LEAQ", Ptr(BX, 0x8, 8), AX)
@@ -95,6 +149,45 @@ func TestPushPop(t *testing.T) {
checkSyntax(t, "push rbx", "PUSHQ", BX)
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
// The W spelling carries the 0x66 operand-size prefix, byte for byte
// with go tool asm; the L and B spellings are illegal in 64-bit mode
// there and rejected here rather than silently widened.
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"PUSHW AX", "PUSHW", []Operand{AX}, "6650"},
{"POPW AX", "POPW", []Operand{AX}, "6658"},
{"PUSHW $5", "PUSHW", []Operand{Imm(5)}, "666a05"},
{"PUSHW (AX)", "PUSHW", []Operand{Ptr(AX, 0, 2)}, "66ff30"},
{"PUSHQ AX", "PUSHQ", []Operand{AX}, "50"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"PUSHL AX", "PUSHL", []Operand{AX}},
{"PUSHL R8", "PUSHL", []Operand{Reg{idx: 8, size: 8}}},
{"POPL BX", "POPL", []Operand{BX}},
{"PUSHB AX", "PUSHB", []Operand{AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
func TestUnary(t *testing.T) {
@@ -127,6 +220,39 @@ func TestControl(t *testing.T) {
checkOp(t, x86asm.JBE, "JLS", Imm(0))
}
// TestIndirectControlFlow pins the indirect JMP/CALL forms: FF /4 for JMP and
// FF /2 for CALL through a register or memory. A REX appears only for the
// extended registers, never REX.W: the branch operand size is fixed at 64
// bits in long mode.
func TestIndirectControlFlow(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"JMP AX", "JMP", []Operand{AX}, "ffe0"},
{"CALL AX", "CALL", []Operand{AX}, "ffd0"},
{"JMP (BX)", "JMP", []Operand{Ptr(BX, 0, 8)}, "ff23"},
{"CALL (BX)", "CALL", []Operand{Ptr(BX, 0, 8)}, "ff13"},
{"JMP 8(BX)", "JMP", []Operand{Ptr(BX, 8, 8)}, "ff6308"},
{"CALL -16(BX)", "CALL", []Operand{Ptr(BX, -16, 8)}, "ff53f0"},
{"JMP R8", "JMP", []Operand{Reg{idx: 8, size: 2}}, "41ffe0"},
{"CALL R9", "CALL", []Operand{Reg{idx: 9, size: 2}}, "41ffd1"},
{"JMP R15", "JMP", []Operand{Reg{idx: 15, size: 2}}, "41ffe7"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
// against the Go assembler. wantOp is the decoder's name, which differs from
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
@@ -176,7 +302,7 @@ func TestSSEMoveGroundTruth(t *testing.T) {
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
// the encoder handles a realistic instruction sequence.
func TestGoFlacScalarTail(t *testing.T) {
// MOVQ swin_base+0(FP), SI — modelled as MOVQ disp(reg), reg.
// MOVQ swin_base+0(FP), SI; modelled as MOVQ disp(reg), reg.
checkSyntax(t, "mov rsi, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), SI)
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
@@ -203,6 +329,21 @@ func TestScalarGroundTruth(t *testing.T) {
{"LZCNTQ R8,R9", "LZCNTQ", []Operand{r8, r9}, "f34d0fbdc8", "LZCNT"},
{"LZCNTW AX,CX", "LZCNTW", []Operand{AX, CX}, "66f30fbdc8", "LZCNT"},
{"TZCNTL AX,CX", "TZCNTL", []Operand{AX, CX}, "f30fbcc8", "TZCNT"},
// Bit scan: BSF/BSR are the unprefixed forms of TZCNT/LZCNT's map.
{"BSFL AX,CX", "BSFL", []Operand{AX, CX}, "0fbcc8", "BSF"},
{"BSFQ R8,R9", "BSFQ", []Operand{r8, r9}, "4d0fbcc8", "BSF"},
{"BSFW AX,CX", "BSFW", []Operand{AX, CX}, "660fbcc8", "BSF"},
{"BSRL AX,CX", "BSRL", []Operand{AX, CX}, "0fbdc8", "BSR"},
{"BSRQ AX,CX", "BSRQ", []Operand{AX, CX}, "480fbdc8", "BSR"},
{"POPCNTL AX,CX", "POPCNTL", []Operand{AX, CX}, "f30fb8c8", "POPCNT"},
{"POPCNTQ R8,R9", "POPCNTQ", []Operand{r8, r9}, "f34d0fb8c8", "POPCNT"},
// A 64-bit immediate that fits a signed int32 is compressed exactly
// as the Go assembler does: positive via B8+rd without REX.W
// (zero-extended), negative via REX.W C7 /0 (sign-extended).
{"MOVQ $4,BX", "MOVQ", []Operand{Imm(4), BX}, "bb04000000", "MOV"},
{"MOVQ $4,R8", "MOVQ", []Operand{Imm(4), r8}, "41b804000000", "MOV"},
{"MOVQ $-1,BX", "MOVQ", []Operand{Imm(-1), BX}, "48c7c3ffffffff", "MOV"},
{"MOVQ big,BX", "MOVQ", []Operand{Imm(0x1122334455667788), BX}, "48bb8877665544332211", "MOV"},
{"CMOVLGT CX,AX", "CMOVLGT", []Operand{CX, AX}, "0f4fc1", "CMOVG"},
{"CMOVLEQ CX,AX", "CMOVLEQ", []Operand{CX, AX}, "0f44c1", "CMOVE"},
{"CMOVQGT R9,R8", "CMOVQGT", []Operand{r9, r8}, "4d0f4fc1", "CMOVG"},
@@ -216,6 +357,18 @@ func TestScalarGroundTruth(t *testing.T) {
{"MOVBQZX AL,R8", "MOVBQZX", []Operand{AL, r8}, "4c0fb6c0", "MOVZX"},
{"MOVWLZX AX,CX", "MOVWLZX", []Operand{AX, CX}, "0fb7c8", "MOVZX"},
{"MOVWQZX AX,R8", "MOVWQZX", []Operand{AX, r8}, "4c0fb7c0", "MOVZX"},
// The width pairs the toolchain accepts and GOROOT uses; bytes
// pinned from go tool asm (see testdata/verify/widen_amd64.s).
{"MOVBWZX (BX),R11W", "MOVBWZX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fb61b", "MOVZX"},
{"MOVBWSX (BX),R11W", "MOVBWSX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fbe1b", "MOVSX"},
{"MOVBLSX (BX),AX", "MOVBLSX", []Operand{Ptr(BX, 0, 1), AX}, "0fbe03", "MOVSX"},
{"MOVBQSX (BX),R8", "MOVBQSX", []Operand{Ptr(BX, 0, 1), r8}, "4c0fbe03", "MOVSX"},
{"MOVWQSX (BX),R9", "MOVWQSX", []Operand{Ptr(BX, 0, 2), r9}, "4c0fbf0b", "MOVSX"},
// A long to quad zero-extend is a plain 32-bit move.
{"MOVLQZX (BX),DX", "MOVLQZX", []Operand{Ptr(BX, 0, 4), DX}, "8b13", "MOV"},
{"MOVLQZX AX,DX", "MOVLQZX", []Operand{AX, DX}, "8bd0", "MOV"},
{"PMOVMSKB X1,AX", "PMOVMSKB", []Operand{vreg(t, "X1"), AX}, "660fd7c1", "PMOVMSKB"},
{"PMOVMSKB X11,CX", "PMOVMSKB", []Operand{vreg(t, "X11"), CX}, "66410fd7cb", "PMOVMSKB"},
{"CVTSL2SD R8,X13", "CVTSL2SD", []Operand{r8, vreg(t, "X13")}, "f2450f2ae8", "CVTSI2SD"},
{"CVTSL2SD AX,X0", "CVTSL2SD", []Operand{AX, vreg(t, "X0")}, "f20f2ac0", "CVTSI2SD"},
{"CVTSQ2SD R8,X13", "CVTSQ2SD", []Operand{r8, vreg(t, "X13")}, "f24d0f2ae8", "CVTSI2SD"},
@@ -294,3 +447,183 @@ func TestScalarErrors(t *testing.T) {
}
}
}
// TestImmediateOutOfRange pins the go-tool-asm parity of the immediate and
// displacement spans: a scalar immediate must fit a signed or unsigned 32-bit
// word (only MOVQ reg, $imm takes the full int64), a scalar shift count must
// be an unsigned byte, and a displacement must fit int32. Every rejected
// shape here is rejected by `go tool asm` too; every accepted one encodes the
// same bytes.
func TestImmediateOutOfRange(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
}{
{"SHLQ count 300", "SHLQ", []Operand{Imm(300), AX}},
{"SHLQ count -1", "SHLQ", []Operand{Imm(-1), AX}},
{"SHLW count 256", "SHLW", []Operand{Imm(256), DX}},
{"SHLB count 300", "SHLB", []Operand{Imm(300), BL}},
{"MOVL imm32+", "MOVL", []Operand{Imm(4294967296), AX}},
{"MOVL imm32-", "MOVL", []Operand{Imm(-2147483649), AX}},
{"MOVW imm32+", "MOVW", []Operand{Imm(4294967296), AX}},
{"MOVB imm32+", "MOVB", []Operand{Imm(4294967296), AL}},
{"ADDB imm32+", "ADDB", []Operand{Imm(4294967296), AL}},
{"ADDL imm32+", "ADDL", []Operand{Imm(4294967296), AX}},
{"ADDQ imm32+", "ADDQ", []Operand{Imm(8589934592), AX}},
{"CMPQ imm32+", "CMPQ", []Operand{AX, Imm(4294967296)}},
{"CMPQ imm32-", "CMPQ", []Operand{AX, Imm(-2147483649)}},
{"TESTL imm32+", "TESTL", []Operand{Imm(4294967296), AX}},
{"IMUL3L imm32+", "IMUL3L", []Operand{Imm(4294967296), CX, DX}},
{"PUSHQ imm32+", "PUSHQ", []Operand{Imm(4294967296)}},
{"MOVQ mem imm32+", "MOVQ", []Operand{Imm(4294967296), Ptr(AX, 0, 8)}},
{"disp32+", "MOVQ", []Operand{Ptr(AX, 4294967296, 8), BX}},
{"disp32+ max", "MOVQ", []Operand{Ptr(AX, 2147483648, 8), BX}},
{"disp32-", "MOVQ", []Operand{Ptr(AX, -2147483649, 8), BX}},
{"VEX disp32+", "VMOVDQU", []Operand{Ptr(AX, 4294967296, 32), vreg(t, "Y1")}},
{"EVEX disp32+", "VMOVDQU32", []Operand{Ptr(AX, 4294967296, 64), vreg(t, "Z1")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestImmediateTruncation pins the toolchain-matching truncations inside the
// accepted 32-bit span: the narrower fields take the low bits silently, byte
// for byte with `go tool asm` (which rejects none of these).
func TestImmediateTruncation(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ADDB $256,BL", "ADDB", []Operand{Imm(256), BL}, "80c300"},
{"ADDB $1000,BL", "ADDB", []Operand{Imm(1000), BL}, "80c3e8"},
{"MOVB $256,AL", "MOVB", []Operand{Imm(256), AL}, "b000"},
{"MOVB $-129,AL", "MOVB", []Operand{Imm(-129), AL}, "b07f"},
{"MOVW $65536,AX", "MOVW", []Operand{Imm(65536), AX}, "66b80000"},
{"MOVW $65535,AX", "MOVW", []Operand{Imm(65535), AX}, "66b8ffff"},
{"MOVW $-32769,AX", "MOVW", []Operand{Imm(-32769), AX}, "66b8ff7f"},
{"MOVL $4294967295,AX", "MOVL", []Operand{Imm(4294967295), AX}, "b8ffffffff"},
{"ADDQ $4294967295,AX", "ADDQ", []Operand{Imm(4294967295), AX}, "4805ffffffff"},
{"CMPB BL,$255", "CMPB", []Operand{BL, Imm(255)}, "80fbff"},
{"CMPQ AX,$4294967295", "CMPQ", []Operand{AX, Imm(4294967295)}, "483dffffffff"},
{"MOVQ $4294967295,0(AX)", "MOVQ", []Operand{Imm(4294967295), Ptr(AX, 0, 8)}, "48c700ffffffff"},
{"SHLQ $255,AX", "SHLQ", []Operand{Imm(255), AX}, "48c1e0ff"},
{"SHLQ $0,AX", "SHLQ", []Operand{Imm(0), AX}, "48c1e000"},
// The one form beyond the 32-bit span: the imm64 MOVQ register move.
{"MOVQ $4294967296,AX", "MOVQ", []Operand{Imm(4294967296), AX}, "48b80000000001000000"},
{"MOVQ disp32 max", "MOVQ", []Operand{Ptr(AX, 2147483647, 8), BX}, "488b98ffffff7f"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must
// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size
// letter) is not reported as encodable.
func TestEncodableCmovSize(t *testing.T) {
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} {
if Encodable(m) {
t.Errorf("Encodable(%q) = true, want false", m)
}
}
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} {
if !Encodable(m) {
t.Errorf("Encodable(%q) = false, want true", m)
}
}
}
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
// byte for byte (no prefix / 66 / F2 / F3 variants).
func TestSSEBinGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MULPS X0,X1", "MULPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f59c8"},
{"MULPS (DI),X1", "MULPS", []Operand{Ptr(DI, 0, 16), vreg(t, "X1")}, "0f590f"},
{"ADDPD X1,X2", "ADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "660f58d1"},
{"XORPS X0,X0", "XORPS", []Operand{vreg(t, "X0"), vreg(t, "X0")}, "0f57c0"},
{"UNPCKLPS X0,X0", "UNPCKLPS", []Operand{vreg(t, "X0"), vreg(t, "X0")}, "0f14c0"},
{"MULSD X1,X2", "MULSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f59d1"},
{"ADDSS (DI),X0", "ADDSS", []Operand{Ptr(DI, 0, 4), vreg(t, "X0")}, "f30f5807"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
}
// TestSSEShuffleGroundTruth checks the imm8 shuffle family: immediate
// first in Plan 9 order, encoded last on the wire.
func TestSSEShuffleGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"SHUFPS $0,X0,X0", "SHUFPS", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X0")}, "0fc6c000"},
{"SHUFPS $27,X1,X2", "SHUFPS", []Operand{Imm(27), vreg(t, "X1"), vreg(t, "X2")}, "0fc6d11b"},
{"PSHUFD $0,X0,X0", "PSHUFD", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X0")}, "660f70c000"},
{"PSHUFLW $3,(DI),X1", "PSHUFLW", []Operand{Imm(3), Ptr(DI, 0, 8), vreg(t, "X1")}, "f20f700f03"},
{"PSHUFHW $2,X1,X2", "PSHUFHW", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2")}, "f30f70d102"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
}
// TestMOVQXMMGroundTruth pins the SSE2 packed-quadword move encodings:
// loads and register moves on F3 0F 7E, stores on 66 0F D6; the forms
// the GPR-move fallback silently corrupted.
func TestMOVQXMMGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MOVQ (DI),X0", "MOVQ", []Operand{Ptr(DI, 0, 8), vreg(t, "X0")}, "f30f7e07"},
{"MOVQ X1,X2", "MOVQ", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f30f7ed1"},
{"MOVQ X0,(DI)", "MOVQ", []Operand{vreg(t, "X0"), Ptr(DI, 0, 8)}, "660fd607"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
}
+178 -112
View File
@@ -9,10 +9,10 @@ import (
)
// This file implements EVEX (AVX-512) instruction encoding: the four-byte
// EVEX prefix with 5-bit vector register fields (Z0–Z31, X/Y 16–31), the
// EVEX prefix with 5-bit vector register fields (Z0-Z31, X/Y 16-31), the
// compressed disp8×N displacement, and the operand shapes the go-flac
// AVX-512 kernels use plus the common floating-point and conversion set.
// Masking follows the Go assembler's spelling: an explicit K1–K7 operand
// Masking follows the Go assembler's spelling: an explicit K1-K7 operand
// anywhere among the operands (merging) plus a ".Z" mnemonic suffix for
// zeroing. K-register operands (mask destinations, KMOVW, KTESTW) are
// supported too.
@@ -36,7 +36,7 @@ type evexSpec struct {
// are taken from the Go assembler's opcode tables, which are authoritative
// for byte-for-byte agreement.
var evexTable = map[string]evexSpec{
// EVEX.128/256/512.66.0F — integer arithmetic / logic, NDS form.
// EVEX.128/256/512.66.0F, integer arithmetic / logic, NDS form.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPADDQ": {1, 0xD4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -48,25 +48,25 @@ var evexTable = map[string]evexSpec{
"VPCMPEQD": {1, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double arithmetic.
// EVEX.128/256/512.66.0F.W1, packed double arithmetic.
"VADDPD": {1, 0x58, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPD": {1, 0x59, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPD": {1, 0x5C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPD": {1, 0x5E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPD": {1, 0x5D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPD": {1, 0x5F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single arithmetic.
// EVEX.128/256/512.0F.W0, packed single arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VDIVPS": {1, 0x5E, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMINPS": {1, 0x5D, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
"VMAXPS": {1, 0x5F, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — packed double unpack.
// EVEX.128/256/512.66.0F.W1, packed double unpack.
"VUNPCKLPD": {1, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VUNPCKHPD": {1, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128.F2.0F.W1 — scalar double arithmetic (the packed opcodes with
// EVEX.128.F2.0F.W1, scalar double arithmetic (the packed opcodes with
// an F2 pp; the EVEX forms exist for masked and zeroing use). The
// memory operand is a single double, so disp8×N = 8.
"VADDSD": {1, 0x58, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
@@ -76,7 +76,7 @@ var evexTable = map[string]evexSpec{
"VMINSD": {1, 0x5D, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VMAXSD": {1, 0x5F, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128.F3.0F.W0 — scalar single arithmetic (disp8×N = 4).
// EVEX.128.F3.0F.W0, scalar single arithmetic (disp8×N = 4).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMULSS": {1, 0x59, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
@@ -84,38 +84,38 @@ var evexTable = map[string]evexSpec{
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.512.66.0F3A — align (NDS + imm8).
// EVEX.512.66.0F3A, align (NDS + imm8).
"VALIGND": {3, 0x03, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — immediate shift (VPSRAD /4).
// EVEX.128/256/512.66.0F, immediate shift (VPSRAD /4).
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — variable shift with an XMM count (VPSRAQ;
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
// the W bit distinguishes it from VPSRAD's E2 form).
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — signed qword to packed double (reg=dst,
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
// rm=src, no vvvv).
"VCVTQQ2PD": {1, 0xE6, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — duplicate the low double (reg=dst,
// EVEX.128/256/512.F2.0F.W1, duplicate the low double (reg=dst,
// rm=src, no vvvv): a 128-bit destination reads a single double from
// memory (disp8×8), the wider ones read the full operand.
"VMOVDDUP": {1, 0x12, 1, 3, -1, vexRM, [3]int{8, 32, 64}},
// EVEX.128/256/512.0F.W0 — signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix — as in the VEX form).
// EVEX.128/256/512.0F.W0, signed dword to packed single (reg=dst,
// rm=src, no vvvv, no mandatory prefix, as in the VEX form).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.128/256/512.0F.W0 — packed single to packed double: the
// EVEX.128/256/512.0F.W0, packed single to packed double: the
// destination is twice the source width and sets the length; disp8×N
// follows the narrow memory source. No F3 prefix: the Go assembler
// emits this instruction with pp = 00 (Intel's maps would call that
// undefined) and gasm reproduces the Go assembler's bytes — its machine
// undefined) and gasm reproduces the Go assembler's bytes, its machine
// code is the oracle, not the manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.128/256/512.F3.0F.W0 — signed dword to packed double (the EVEX
// EVEX.128/256/512.F3.0F.W0, signed dword to packed double (the EVEX
// form of the VEX instruction; the destination sets the length, disp8×N
// follows the narrow memory source).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
// EVEX packed double → dword conversions: the source is the wide
// operand and the mnemonic fixes the length — the bare names are
// operand and the mnemonic fixes the length, the bare names are
// 512-bit only (ZMM source, XMM destination), the X/Y spellings are
// EVEX-128/256. Exactly one slot of n is valid; it names the vector
// length (and the disp8×N multiplier) a register or memory source
@@ -127,7 +127,7 @@ var evexTable = map[string]evexSpec{
"VCVTTPD2DQX": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTTPD2DQY": {1, 0xE6, 1, 1, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F3A — ternary logic and lane shuffles (NDS + imm8).
// EVEX.66.0F3A, ternary logic and lane shuffles (NDS + imm8).
"VPTERNLOGD": {3, 0x25, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPTERNLOGQ": {3, 0x25, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFI32X4": {3, 0x43, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
@@ -136,11 +136,11 @@ var evexTable = map[string]evexSpec{
"VSHUFF64X2": {3, 0x23, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F — the EVEX forms of the VEX two-source shuffle.
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F3A — lane insert ($imm, xsrc, zsrc1, zdst).
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTF32X8": {3, 0x1A, 0, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
"VINSERTF64X2": {3, 0x18, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
@@ -150,7 +150,7 @@ var evexTable = map[string]evexSpec{
"VINSERTI64X2": {3, 0x38, 1, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
"VINSERTI64X4": {3, 0x3A, 1, 1, -1, vexNDS3Imm, [3]int{0, 0, 32}},
// EVEX.66.0F3A — lane extract (reg=source, rm=XMM/YMM destination,
// EVEX.66.0F3A, lane extract (reg=source, rm=XMM/YMM destination,
// imm8).
"VEXTRACTF32X4": {3, 0x19, 0, 1, -1, vexExtract, [3]int{0, 16, 16}},
"VEXTRACTF32X8": {3, 0x1B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
@@ -159,14 +159,27 @@ var evexTable = map[string]evexSpec{
"VEXTRACTI32X8": {3, 0x3B, 0, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTI64X2": {3, 0x39, 1, 1, -1, vexExtract, [3]int{0, 16, 16}},
// EVEX.66.0F — compare with an opmask destination ($imm, src2, src1,
// EVEX.66.0F, compare with an opmask destination ($imm, src2, src1,
// kdst): NDS3Imm with the K register in the reg field.
"VCMPPD": {1, 0xC2, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPPS": {1, 0xC2, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VCMPSD": {1, 0xC2, 1, 3, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VCMPSS": {1, 0xC2, 0, 2, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F38 — permutes (NDS form).
// EVEX.66.0F3A, integer compares with an opmask destination, the same
// NDS3Imm-with-k-reg shape as the floating-point compares; W selects the
// operand width (byte/word vs dword/qword), the opcode the signedness.
// The memory form takes a full vector, so disp8×N is 16/32/64.
"VPCMPB": {3, 0x3F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUB": {3, 0x3E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPW": {3, 0x3F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUW": {3, 0x3E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPD": {3, 0x1F, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUD": {3, 0x1E, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPQ": {3, 0x1F, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38, permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -175,7 +188,7 @@ var evexTable = map[string]evexSpec{
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — the wider integer set (NDS form).
// EVEX.66.0F, the wider integer set (NDS form).
"VPMADDWD": {1, 0xF5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -186,34 +199,34 @@ var evexTable = map[string]evexSpec{
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPACKUSDW": {2, 0x2B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38 — absolute values and replicating moves (reg=dst,
// EVEX.66.0F38, absolute values and replicating moves (reg=dst,
// rm=src).
"VPABSB": {2, 0x1C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSW": {2, 0x1D, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSD": {2, 0x1E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPABSQ": {2, 0x1F, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.F3.0F — replicate even/odd singles.
// EVEX.F3.0F, replicate even/odd singles.
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — sign/zero-extending moves; the memory source is the
// EVEX.66.0F38, sign/zero-extending moves; the memory source is the
// narrow half (here byte to word).
"VPMOVSXBW": {2, 0x20, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VPMOVZXBW": {2, 0x30, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F — packed single conversions (reg=dst, rm=src).
// EVEX.66.0F, packed single conversions (reg=dst, rm=src).
"VCVTPS2DQ": {1, 0x5B, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VCVTTPS2DQ": {1, 0x5B, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — broadcast a single/double to all lanes (reg=dst,
// EVEX.66.0F38, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; disp8×N is the element size).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VBROADCASTSD": {2, 0x19, 1, 1, -1, vexRM, [3]int{0, 8, 8}},
// EVEX.66.0F38 — expand loads (rm → vector register destination).
// EVEX.66.0F38, expand loads (rm → vector register destination).
"VEXPANDPD": {2, 0x88, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
"VEXPANDPS": {2, 0x88, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDD": {2, 0x89, 0, 1, -1, vexRM, [3]int{4, 4, 4}},
"VPEXPANDQ": {2, 0x89, 1, 1, -1, vexRM, [3]int{8, 8, 8}},
// EVEX.66.0F38 — compress stores (vector register source → rm), and the
// EVEX.66.0F38, compress stores (vector register source → rm), and the
// remaining narrowing stores.
"VCOMPRESSPD": {2, 0x8A, 1, 1, -1, vexRMRev, [3]int{8, 8, 8}},
"VCOMPRESSPS": {2, 0x8A, 0, 1, -1, vexRMRev, [3]int{4, 4, 4}},
@@ -222,7 +235,7 @@ var evexTable = map[string]evexSpec{
"VPMOVWB": {2, 0x30, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQB": {2, 0x32, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
// EVEX.66.0F — rotates (immediate form: /0 right, /1 left).
// EVEX.66.0F, rotates (immediate form: /0 right, /1 left).
"VPRORD": {1, 0x72, 0, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPRORQ": {1, 0x72, 1, 1, 0, vexShiftImm, [3]int{16, 32, 64}},
"VPROLD": {1, 0x72, 0, 1, 1, vexShiftImm, [3]int{16, 32, 64}},
@@ -235,14 +248,14 @@ var evexTable = map[string]evexSpec{
"VPSRLQ": {1, 0x73, 1, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
"VPSLLQ": {1, 0x73, 1, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, packed (reg=dst, rm=src).
// EVEX.66.0F38, floating-point helpers, packed (reg=dst, rm=src).
"VRCP14PD": {2, 0x4C, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRCP14PS": {2, 0x4C, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PD": {2, 0x4E, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VRSQRT14PS": {2, 0x4E, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPD": {2, 0x42, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
"VGETEXPPS": {2, 0x42, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F38 — floating-point helpers, scalar (NDS form: src2 is
// EVEX.66.0F38, floating-point helpers, scalar (NDS form: src2 is
// rm, src1 is vvvv, the XMM destination is reg). Like the scalar 0F3A
// forms, these take the 66 prefix; W selects double/single.
"VRCP14SD": {2, 0x4D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
@@ -251,13 +264,13 @@ var evexTable = map[string]evexSpec{
"VRSQRT14SS": {2, 0x4F, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
"VGETEXPSD": {2, 0x43, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VGETEXPSS": {2, 0x43, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F38 — scale by a power of two (NDS form).
// EVEX.66.0F38, scale by a power of two (NDS form).
"VSCALEFPD": {2, 0x2C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFPS": {2, 0x2C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VSCALEFSD": {2, 0x2D, 1, 1, -1, vexNDS3, [3]int{8, 8, 8}},
"VSCALEFSS": {2, 0x2D, 0, 1, -1, vexNDS3, [3]int{4, 4, 4}},
// EVEX.66.0F3A — packed round/getmant/reduce ($imm, src, dst: reg=dst,
// EVEX.66.0F3A, packed round/getmant/reduce ($imm, src, dst: reg=dst,
// rm=src, imm8).
"VRNDSCALEPD": {3, 0x09, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VRNDSCALEPS": {3, 0x08, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
@@ -265,7 +278,7 @@ var evexTable = map[string]evexSpec{
"VGETMANTPS": {3, 0x26, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPD": {3, 0x56, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VREDUCEPS": {3, 0x56, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
// EVEX.66.0F3A — scalar round/getmant/reduce and fixup/range (NDS +
// EVEX.66.0F3A, scalar round/getmant/reduce and fixup/range (NDS +
// imm8: $imm, src2, src1, dst). The scalar 0F3A forms all take the 66
// prefix; W selects double/single.
"VRNDSCALESD": {3, 0x0B, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
@@ -283,7 +296,7 @@ var evexTable = map[string]evexSpec{
"VRANGESD": {3, 0x51, 1, 1, -1, vexNDS3Imm, [3]int{8, 8, 8}},
"VRANGESS": {3, 0x51, 0, 1, -1, vexNDS3Imm, [3]int{4, 4, 4}},
// EVEX.66.0F3A — floating-point class test ($imm, src, kdst): the
// EVEX.66.0F3A, floating-point class test ($imm, src, kdst): the
// reg field carries the opmask destination. The packed forms carry an
// explicit length in the mnemonic (X/Y/Z).
"VFPCLASSPDX": {3, 0x66, 1, 1, -1, vexImmRM, [3]int{16, 0, 0}},
@@ -295,7 +308,7 @@ var evexTable = map[string]evexSpec{
"VFPCLASSSD": {3, 0x67, 1, 1, -1, vexImmRM, [3]int{8, 0, 0}},
"VFPCLASSSS": {3, 0x67, 0, 1, -1, vexImmRM, [3]int{4, 0, 0}},
// EVEX — the remaining conversions. VCVTQQ2PS narrows (the 512-bit
// EVEX, the remaining conversions. VCVTQQ2PS narrows (the 512-bit
// source sets the length); the rest follow the destination.
"VCVTQQ2PS": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 0, 64}},
"VCVTPD2QQ": {1, 0x7B, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
@@ -303,13 +316,13 @@ var evexTable = map[string]evexSpec{
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F38 — half-precision convert (half-width source).
// EVEX.66.0F38, half-precision convert (half-width source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.66.0F3A — half-precision convert back ($imm, src, dst: reg=src,
// rm=dst, imm8 — the extract layout).
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
// rm=dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract, [3]int{8, 16, 32}},
// EVEX — unsigned and truncating conversions. The PD sources are the
// EVEX, unsigned and truncating conversions. The PD sources are the
// wide operand (the bare names are 512-bit only, the X/Y spellings fix
// the length); the PS/UQQ destinations are wide and follow the
// destination.
@@ -336,7 +349,7 @@ var evexTable = map[string]evexSpec{
"VCVTQQ2PSX": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{16, 0, 0}},
"VCVTQQ2PSY": {1, 0x5B, 1, 0, -1, vexRMSrcLen, [3]int{0, 32, 0}},
// EVEX.66.0F38 — the remaining sign/zero-extending moves (narrow
// EVEX.66.0F38, the remaining sign/zero-extending moves (narrow
// source; disp8×N follows its size).
"VPMOVSXBD": {2, 0x21, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVSXBQ": {2, 0x22, 0, 1, -1, vexRM, [3]int{2, 4, 8}},
@@ -348,7 +361,7 @@ var evexTable = map[string]evexSpec{
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM, [3]int{4, 8, 16}},
"VPMOVZXDQ": {2, 0x35, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.F3.0F38 — the remaining narrowing stores (vector source in reg,
// EVEX.F3.0F38, the remaining narrowing stores (vector source in reg,
// narrow destination in r/m): signed, unsigned and the D/Q truncations.
"VPMOVSDB": {2, 0x21, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVSQB": {2, 0x22, 0, 2, -1, vexRMRev, [3]int{2, 4, 8}},
@@ -365,7 +378,7 @@ var evexTable = map[string]evexSpec{
"VPMOVDB": {2, 0x31, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
"VPMOVQW": {2, 0x34, 0, 2, -1, vexRMRev, [3]int{4, 8, 16}},
// EVEX.F3.0F38 — mask/vector conversions: M2* moves an opmask register
// EVEX.F3.0F38, mask/vector conversions: M2* moves an opmask register
// into a vector (rm = K source, reg = vector destination), *2M does the
// reverse (reg = K destination, rm = vector source, the length follows
// the vector).
@@ -378,7 +391,7 @@ var evexTable = map[string]evexSpec{
"VPMOVD2M": {2, 0x39, 0, 2, -1, vexRM, [3]int{16, 32, 64}},
"VPMOVQ2M": {2, 0x39, 1, 2, -1, vexRM, [3]int{16, 32, 64}},
// EVEX — scalar conversions between vector and general-purpose
// EVEX, scalar conversions between vector and general-purpose
// registers. Vector to GPR (two operands: vec/mem source, GPR
// destination, vvvv unused): the signed and truncated pair, and the
// unsigned forms (EVEX only).
@@ -408,22 +421,22 @@ var evexTable = map[string]evexSpec{
"VCVTUSI2SDQ": {1, 0x7B, 1, 3, -1, vexNDS3, [3]int{8, 8, 8}},
"VCVTUSI2SSL": {1, 0x7B, 0, 2, -1, vexNDS3, [3]int{4, 4, 4}},
"VCVTUSI2SSQ": {1, 0x7B, 1, 2, -1, vexNDS3, [3]int{8, 8, 8}},
// EVEX.128/256/512.66.0F38.W0 — sign-extend dwords to qwords; the memory
// EVEX.128/256/512.66.0F38.W0, sign-extend dwords to qwords; the memory
// operand is the narrow source, so disp8×N follows its size (8/16/32 for
// the xmm/ymm/zmm destination lengths).
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
// EVEX.512.66.0F3A.W1 — lane extract (reg=ZMM source, rm=YMM/memory
// EVEX.512.66.0F3A.W1, lane extract (reg=ZMM source, rm=YMM/memory
// destination, imm8).
"VEXTRACTI64X4": {3, 0x3B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
"VEXTRACTF64X4": {3, 0x1B, 1, 1, -1, vexExtract, [3]int{0, 0, 32}},
// EVEX.66.0F38 — more integer NDS forms (W distinguishes D/Q).
// EVEX.66.0F38, more integer NDS forms (W distinguishes D/Q).
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPMULLQ": {2, 0x40, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
// EVEX.128/256/512 — the wider integer set (AVX-512 F/BW): byte/word
// EVEX.128/256/512, the wider integer set (AVX-512 F/BW): byte/word
// arithmetic, the bitwise ops with D/Q suffixes, min/max, averages and
// variable shifts. All NDS form; W distinguishes element size.
"VPADDB": {1, 0xFC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -461,21 +474,21 @@ var evexTable = map[string]evexSpec{
"VPSRAVQ": {2, 0x46, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX forms of instructions that also exist in VEX (selected when a ZMM
// or K register, or indices 16–31, demand EVEX).
// or K register, or indices 16-31, demand EVEX).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F — immediate shift (VPSLLD /6).
// EVEX.66.0F, immediate shift (VPSLLD /6).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm, [3]int{16, 32, 64}},
// EVEX.F3.0F38.W0 — narrowing stores: reg = wide source, rm = narrow
// EVEX.F3.0F38.W0, narrowing stores: reg = wide source, rm = narrow
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
}
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
// depends on the source kind — a GPR source uses opReg, a memory source uses
// depends on the source kind, a GPR source uses opReg, a memory source uses
// opMem with a disp8×N of n.
type evexBcastSpec struct {
mapSel int
@@ -486,50 +499,53 @@ type evexBcastSpec struct {
}
var evexBcastTable = map[string]evexBcastSpec{
// EVEX.128/256/512.66.0F38 — broadcast a dword/qword to all lanes.
// EVEX.128/256/512.66.0F38, broadcast a dword/qword to all lanes.
"VPBROADCASTD": {2, 0x7C, 0x58, 0, 4},
"VPBROADCASTQ": {2, 0x7C, 0x59, 1, 8},
// EVEX.128/256/512.66.0F38 — broadcast a byte/word (GPR or memory
// EVEX.128/256/512.66.0F38, broadcast a byte/word (GPR or memory
// source) to all lanes.
"VPBROADCASTB": {2, 0x7A, 0x78, 0, 1},
"VPBROADCASTW": {2, 0x7B, 0x79, 0, 2},
}
// evexMoveSpec describes an EVEX move (load and store opcodes, like the VEX
// move table).
// move table). vecOK and xmmOnly mirror the VEX twin's operand rules: a
// scalar move (vecOK false, xmmOnly true) takes XMM↔memory operands only.
type evexMoveSpec struct {
mapSel int
pp int
load byte // r/m → vector
store byte // vector → r/m
w int
n [3]int
mapSel int
pp int
load byte // r/m → vector
store byte // vector → r/m
w int
n [3]int
vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0 — unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F3.0F.W1 — unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W0 — unaligned byte move (byte/word moves use the
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
// EVEX.128/256/512.F2.0F.W1 — unaligned word move (shares the qword
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F.W1 — unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512 — aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}},
// EVEX.128/256/512.66.0F — aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}},
// EVEX.128.F3.0F.W0 — scalar single move, memory operands (the
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}},
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -546,7 +562,7 @@ func isEvex(mnemUpper string) bool {
// evexRequired reports whether the operands force the EVEX encoding of a
// mnemonic that also has a VEX form: ZMM and K registers do, and so do
// register indices 16–31, which only EVEX can represent (X16–Y31 exist
// register indices 16-31, which only EVEX can represent (X16-Y31 exist
// solely under AVX-512).
func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper]
@@ -565,7 +581,7 @@ func evexRequired(upper string, ops []Operand) bool {
// evexSuffix carries the EVEX mnemonic suffixes the Go assembler accepts:
// zeroing (.Z), a rounding mode (.RN_SAE, .RD_SAE, .RU_SAE, .RZ_SAE),
// suppress-all-exceptions (.SAE) and memory broadcast (.BCST). Masking is
// not a suffix — Go writes it as an explicit K operand.
// not a suffix, Go writes it as an explicit K operand.
type evexSuffix struct {
zeroing bool
sae bool
@@ -590,12 +606,12 @@ func (s evexSuffix) evexOnly() bool {
// broadcast together with rounding/SAE.
func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
sfx := evexSuffix{rounding: -1}
i := strings.IndexByte(mnem, '.')
if i < 0 {
before, after, ok := strings.Cut(mnem, ".")
if !ok {
return mnem, sfx, nil
}
base := mnem[:i]
parts := strings.Split(mnem[i+1:], ".")
base := before
parts := strings.Split(after, ".")
seen := map[string]bool{}
for j, p := range parts {
if seen[p] {
@@ -605,7 +621,7 @@ func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
switch p {
case "Z":
if j != len(parts)-1 {
return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", mnem[i+1:])
return "", sfx, fmt.Errorf("the .Z suffix must come last in %q", after)
}
sfx.zeroing = true
case "SAE":
@@ -625,7 +641,7 @@ func parseEvexSuffix(mnem string) (string, evexSuffix, error) {
}
}
if sfx.bcst && (sfx.sae || sfx.rounding >= 0) {
return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", mnem[i+1:])
return "", sfx, fmt.Errorf("cannot combine .BCST with rounding or SAE in %q", after)
}
return base, sfx, nil
}
@@ -655,7 +671,7 @@ var evexRound = map[string]bool{
}
// evexBcstN maps an instruction accepting .BCST to the broadcast element
// size — the disp8×N multiplier for its memory operand.
// size, the disp8×N multiplier for its memory operand.
var evexBcstN = map[string]int{
"VADDPD": 8, "VSUBPD": 8, "VMULPD": 8, "VDIVPD": 8,
"VMINPD": 8, "VMAXPD": 8,
@@ -676,7 +692,7 @@ var evexBcstN = map[string]int{
"VCVTTPD2QQ": 8, "VCVTTPS2QQ": 4, "VCVTUQQ2PD": 8, "VCVTUQQ2PS": 8,
}
// splitMask extracts an explicit mask register (K1–K7) from the operand list,
// splitMask extracts an explicit mask register (K1-K7) from the operand list,
// returning the remaining operands and the mask index. K0 is not a usable
// mask (aaa = 0 means "no mask"), matching the assembler.
func splitMask(ops []Operand) ([]Operand, int, error) {
@@ -688,7 +704,7 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
return nil, 0, fmt.Errorf("at most one mask register operand")
}
if r.idx == 0 {
return nil, 0, fmt.Errorf("K0 is not a usable mask register")
return nil, 0, fmt.Errorf("k0 is not a usable mask register")
}
mask = r.idx
continue
@@ -699,7 +715,7 @@ func splitMask(ops []Operand) ([]Operand, int, error) {
}
// encodeEvex encodes an EVEX instruction with operands in Plan 9 order. The
// mask, when present, is an explicit K1–K7 operand anywhere among the
// mask, when present, is an explicit K1-K7 operand anywhere among the
// operands; the mnemonic suffix carries zeroing, rounding/SAE and
// broadcast.
func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error {
@@ -1009,6 +1025,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
var rm Operand
switch {
case srcIsVec && dstIsVec:
// A store-form reg-reg move, the layout the Go assembler uses; a
// scalar move has no two-register form at all (the register form
// takes three operands), matching the VEX twin's vecOK rule.
if !ms.vecOK {
return fmt.Errorf("%s does not take two vector registers", mnem)
}
reg, rm = srcReg, dst
case srcIsVec:
if !memOperand(dst) {
@@ -1024,12 +1046,18 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
default:
return fmt.Errorf("%s needs a vector register operand", mnem)
}
// The scalar move is 128-bit only, so the register the length follows
// must be an XMM (the VEX twin's xmmOnly rule; EVEX also reaches ZMM,
// hence the inequality rather than a YMM test).
if ms.xmmOnly && reg.size != 16 {
return fmt.Errorf("%s operates on XMM registers only", mnem)
}
spec := evexSpec{mapSel: ms.mapSel, opcode: op, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
return e.emitEvexFields(spec, reg.vecLenBit(), reg.idx, -1, rm, mask, sfx)
}
// encodeEvexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the length fixed by the mnemonic — the
// the destination always XMM and the length fixed by the mnemonic, the
// single valid slot of spec.n names the vector length (and the disp8×N
// multiplier) a register or memory source encodes.
func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
@@ -1048,7 +1076,7 @@ func (e *enc) encodeEvexRMSrcLen(spec evexSpec, ops []Operand, mask int, sfx eve
return e.emitEvexFields(spec, ll, dstReg.idx, -1, src, mask, sfx)
}
// soleLen returns the vector-length index of the single valid slot of n —
// soleLen returns the vector-length index of the single valid slot of n
// the length a length-fixed mnemonic (the EVEX conversion spellings) encodes
// regardless of its operands.
func soleLen(n [3]int) (int, error) {
@@ -1118,8 +1146,8 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
// emitEvexFields emits the EVEX prefix, opcode, ModR/M, SIB and displacement
// (disp8×N compressed) for the given precomputed fields. regIdx is the
// unextended reg-field register index, or a /digit (0–7); vvvvIdx is the
// vvvv register index, or -1 when unused. mask (K1–K7, 0 = unmasked) and
// unextended reg-field register index, or a /digit (0-7); vvvvIdx is the
// vvvv register index, or -1 when unused. mask (K1-K7, 0 = unmasked) and
// zeroing fill the aaa and z bits of the P2 byte.
func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand, mask int, sfx evexSuffix) error {
if ll > 2 {
@@ -1158,9 +1186,6 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
if r.idx&16 != 0 {
xBar = 0
}
if r.idx&16 != 0 {
xBar = 0
}
case Mem:
var err error
modrm, sib, disp, xBar, bBar, err = memComponentsEvex(regIdx&7, r, spec.n[ll])
@@ -1190,7 +1215,7 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
// The b bit and the L'L field carry the rounding/SAE/broadcast mode:
// a rounding mode replaces L'L with the rc value, plain SAE and
// broadcast keep the vector length.
b, ll := 0, ll
b := 0
switch {
case sfx.rounding >= 0:
b, ll = 1, sfx.rounding
@@ -1219,6 +1244,11 @@ func (e *enc) emitEvexFields(spec evexSpec, ll, regIdx, vvvvIdx int, rm Operand,
func memComponentsEvex(regField int, m Mem, n int) (modrm, sib int, disp []byte, xBar, bBar int, err error) {
sib = -1
xBar, bBar = 1, 1 // inverted bits: 1 = no extension
// The disp32 fallback bounds the displacement by int32, and the
// compressed disp8 form reaches at most ±127×64, well inside it.
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
}
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 1, 1, nil // RIP-relative
}
@@ -1311,7 +1341,7 @@ func isScatter(upper string) bool {
}
// vsibLen validates a VSIB memory operand (the index must be a vector
// register) and returns it with the vector length the index selects — the
// register) and returns it with the vector length the index selects, the
// EVEX L'L field follows the index register, not the data register.
func vsibLen(op Operand, what string) (Mem, int, error) {
m, ok := op.(Mem)
@@ -1370,7 +1400,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
}
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib — reg = src,
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
// rm = the VSIB memory operand, the K mask in aaa and L following the VSIB
// index.
func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evexSuffix) error {
@@ -1398,15 +1428,15 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
// evexKOperand lists the instructions whose K register is a genuine operand
// (the source or destination of a mask/vector conversion) rather than a
// mask modifier — the M2 and 2M conversions. They take no masking.
// mask modifier, the M2 and 2M conversions. They take no masking.
var evexKOperand = map[string]bool{
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
// direction — kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR) — and the GPR forms carry a mandatory
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
// prefix and W for the wider widths.
type kmovSpec struct {
kk, kmem, gprk, kgpr byte
@@ -1471,24 +1501,60 @@ type kOpSpec struct {
var kOpsTable = map[string]kOpSpec{
// k ← k OP k: reg = dst, vvvv = src1, rm = src2 (three opmask
// registers).
// registers). Byte/word widths share W0 and differ by the 66 prefix;
// dword/qword share the W bit selection the Go assembler emits.
"KANDB": {1, 0x41, 0, 1, 1, vexNDS3},
"KANDW": {1, 0x41, 0, 0, 1, vexNDS3},
"KANDD": {1, 0x41, 1, 1, 1, vexNDS3},
"KANDQ": {1, 0x41, 1, 0, 1, vexNDS3},
"KANDNB": {1, 0x42, 0, 1, 1, vexNDS3},
"KANDNW": {1, 0x42, 0, 0, 1, vexNDS3},
"KANDND": {1, 0x42, 1, 1, 1, vexNDS3},
"KANDNQ": {1, 0x42, 1, 0, 1, vexNDS3},
"KORB": {1, 0x45, 0, 1, 1, vexNDS3},
"KORW": {1, 0x45, 0, 0, 1, vexNDS3},
"KORD": {1, 0x45, 1, 1, 1, vexNDS3},
"KORQ": {1, 0x45, 1, 0, 1, vexNDS3},
"KXNORB": {1, 0x46, 0, 1, 1, vexNDS3},
"KXNORW": {1, 0x46, 0, 0, 1, vexNDS3},
"KXNORD": {1, 0x46, 1, 1, 1, vexNDS3},
"KXNORQ": {1, 0x46, 1, 0, 1, vexNDS3},
"KXORB": {1, 0x47, 0, 1, 1, vexNDS3},
"KXORW": {1, 0x47, 0, 0, 1, vexNDS3},
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
"KADDD": {1, 0x4A, 1, 1, 1, vexNDS3},
"KADDQ": {1, 0x4A, 1, 0, 1, vexNDS3},
// k ← OP k (KNOT) and flags ← k OP k (KORTEST): reg = dst, rm = src.
// k ← OP k (KNOT), k ← k AND~ k (KTEST-style RM) and flags ← k OP k
// (KORTEST): reg = dst, rm = src.
"KNOTB": {1, 0x44, 0, 1, 0, vexRM},
"KNOTW": {1, 0x44, 0, 0, 0, vexRM},
"KNOTD": {1, 0x44, 1, 1, 0, vexRM},
"KNOTQ": {1, 0x44, 1, 0, 0, vexRM},
"KORTESTB": {1, 0x98, 0, 1, 0, vexRM},
"KORTESTW": {1, 0x98, 0, 0, 0, vexRM},
"KORTESTD": {1, 0x98, 1, 1, 0, vexRM},
// OP $imm, src, dst: reg = dst, rm = src, imm8.
"KORTESTQ": {1, 0x98, 1, 0, 0, vexRM},
"KTESTB": {1, 0x99, 0, 1, 0, vexRM},
"KTESTW": {1, 0x99, 0, 0, 0, vexRM},
"KTESTD": {1, 0x99, 1, 1, 0, vexRM},
"KTESTQ": {1, 0x99, 1, 0, 0, vexRM},
// OP $imm, src, dst: reg = dst, rm = src, imm8. The opcodes split by
// direction (0x32/0x33 left, 0x30/0x31 right) and within each by
// element half (0x32 byte/word, 0x33 dword/qword); W picks byte/dword
// (W0) against word/qword (W1).
"KSHIFTLB": {3, 0x32, 0, 1, 0, vexImmRM},
"KSHIFTLW": {3, 0x32, 1, 1, 0, vexImmRM},
"KSHIFTLD": {3, 0x33, 0, 1, 0, vexImmRM},
"KSHIFTLQ": {3, 0x33, 1, 1, 0, vexImmRM},
"KSHIFTRB": {3, 0x30, 0, 1, 0, vexImmRM},
"KSHIFTRW": {3, 0x30, 1, 1, 0, vexImmRM},
"KSHIFTRD": {3, 0x31, 0, 1, 0, vexImmRM},
"KSHIFTRQ": {3, 0x31, 1, 1, 0, vexImmRM},
}
// isKOp reports whether the mnemonic is an opmask-register instruction.
+59 -95
View File
@@ -4,13 +4,10 @@
package asm
import (
"os"
"strings"
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestEvexGroundTruth checks the EVEX (AVX-512) encodings byte for byte
@@ -19,7 +16,7 @@ import (
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
// register fields (X/Y 16–31, Z 0–31).
// register fields (X/Y 16-31, Z 0-31).
func TestEvexGroundTruth(t *testing.T) {
cases := []struct {
name string
@@ -61,7 +58,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
// VMOVDQU64 — the W1 qword variant.
// VMOVDQU64; the W1 qword variant.
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
@@ -80,7 +77,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
// Indices 16–31: rm[4] rides in X̄ for register operands.
// Indices 16-31: rm[4] rides in X̄ for register operands.
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
@@ -99,7 +96,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
// Register indices 16–31 exist only in EVEX encodings.
// Register indices 16-31 exist only in EVEX encodings.
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
// Packed double arithmetic / unpack (EVEX forms carry W=1).
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
@@ -110,7 +107,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
// VMOVDDUP; duplicate the low double; disp8×N = 64 at 512 bits, and
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
@@ -153,7 +150,7 @@ func TestEvexGroundTruth(t *testing.T) {
}
}
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
// TestEvexMasking checks the AVX-512 mask operand (K1-K7, placed freely among
// the operands) and the .Z zeroing suffix, byte for byte against the Go
// assembler.
func TestEvexMasking(t *testing.T) {
@@ -244,11 +241,11 @@ func TestEvexMasking(t *testing.T) {
}
}
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set; ternary
// logic, lane shuffles/inserts/extracts, compares with a K destination,
// permutes, the wider integer families, expand/compress, broadcasts,
// rotates and word shifts, the opmask instructions, the EVEX suffixes
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
// (rounding/SAE/broadcast) and the aligned/scalar moves; byte for byte
// against the Go assembler.
func TestEvexExtendedGroundTruth(t *testing.T) {
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
@@ -278,7 +275,7 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
// Packed single arithmetic (same opcodes, no mandatory prefix) —
// Packed single arithmetic (same opcodes, no mandatory prefix);
// ZMM, YMM and XMM widths, rounding and broadcast.
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
@@ -322,6 +319,43 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
{"KORTESTD", "KORTESTD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f998d1"},
{"KMOVQ k,k", "KMOVQ", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f890d1"},
{"KMOVQ gpr,k", "KMOVQ", []Operand{BX, vreg(t, "K1")}, "c4e1fb92cb"},
// Completed opmask families (ANDN, NOT, OR/XOR word+qword, TEST,
// word-width shifts; byte-exact against go tool asm).
{"KANDNW", "KANDNW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec42d9"},
{"KANDNB", "KANDNB", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c5d542f4"},
{"KANDND", "KANDND", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed42d9"},
{"KANDNQ", "KANDNQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d442f4"},
{"KANDD", "KANDD", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c4e1ed41d9"},
{"KADDD", "KADDD", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d54af4"},
{"KNOTW", "KNOTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f844d1"},
{"KNOTD", "KNOTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f944e3"},
{"KNOTQ", "KNOTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f844f5"},
{"KORW", "KORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec45d9"},
{"KORQ", "KORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d445f4"},
{"KXNORB", "KXNORB", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ed46d9"},
{"KXORW", "KXORW", []Operand{vreg(t, "K1"), vreg(t, "K2"), vreg(t, "K3")}, "c5ec47d9"},
{"KXORQ", "KXORQ", []Operand{vreg(t, "K4"), vreg(t, "K5"), vreg(t, "K6")}, "c4e1d447f4"},
{"KORTESTW", "KORTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f898d1"},
{"KORTESTB", "KORTESTB", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c5f998e3"},
{"KORTESTQ", "KORTESTQ", []Operand{vreg(t, "K5"), vreg(t, "K6")}, "c4e1f898f5"},
{"KTESTW", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f899d1"},
{"KTESTD", "KTESTD", []Operand{vreg(t, "K3"), vreg(t, "K4")}, "c4e1f999e3"},
{"KSHIFTLB", "KSHIFTLB", []Operand{Imm(1), vreg(t, "K1"), vreg(t, "K2")}, "c4e37932d101"},
{"KSHIFTLD", "KSHIFTLD", []Operand{Imm(2), vreg(t, "K3"), vreg(t, "K4")}, "c4e37933e302"},
{"KSHIFTLQ", "KSHIFTLQ", []Operand{Imm(3), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f933f503"},
{"KSHIFTRB", "KSHIFTRB", []Operand{Imm(4), vreg(t, "K1"), vreg(t, "K2")}, "c4e37930d104"},
{"KSHIFTRW", "KSHIFTRW", []Operand{Imm(5), vreg(t, "K3"), vreg(t, "K4")}, "c4e3f930e305"},
{"KSHIFTRQ", "KSHIFTRQ", []Operand{Imm(6), vreg(t, "K5"), vreg(t, "K6")}, "c4e3f931f506"},
// Integer compares with an opmask destination (0F3A map, the
// go-bzip2 partition kernel's classify instructions).
{"VPCMPUB", "VPCMPUB", []Operand{Imm(1), vreg(t, "X1"), vreg(t, "X0"), vreg(t, "K1")}, "62f37d083ec901"},
{"VPCMPB", "VPCMPB", []Operand{Imm(2), vreg(t, "Y2"), vreg(t, "Y3"), vreg(t, "K2")}, "62f365283fd202"},
{"VPCMPUW", "VPCMPUW", []Operand{Imm(5), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3")}, "62f3ed483ed905"},
{"VPCMPW", "VPCMPW", []Operand{Imm(6), vreg(t, "X3"), vreg(t, "X4"), vreg(t, "K4")}, "62f3dd083fe306"},
{"VPCMPD", "VPCMPD", []Operand{Imm(0), vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "K1")}, "62f36d281fc900"},
{"VPCMPUD", "VPCMPUD", []Operand{Imm(1), vreg(t, "Z2"), vreg(t, "Z3"), vreg(t, "K2")}, "62f365481ed201"},
{"VPCMPQ", "VPCMPQ", []Operand{Imm(2), vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K3")}, "62f3ed081fd902"},
{"VPCMPUQ", "VPCMPUQ", []Operand{Imm(3), vreg(t, "Y3"), vreg(t, "Y4"), vreg(t, "K4")}, "62f3dd281ee303"},
// Lane extract / insert.
{"VEXTRACTF32X4", "VEXTRACTF32X4", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f37d2819ca01"},
{"VEXTRACTI64X2", "VEXTRACTI64X2", []Operand{Imm(1), vreg(t, "Y1"), vreg(t, "X2")}, "62f3fd2839ca01"},
@@ -365,8 +399,8 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
}
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
// tail of the EVEX set; reciprocals, rsqrt, getexp/getmant, scalef,
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions;
// plus gather/scatter with VSIB addressing, byte for byte against the Go
// assembler.
func TestEvexHelperGroundTruth(t *testing.T) {
@@ -468,9 +502,9 @@ func TestEvexHelperGroundTruth(t *testing.T) {
}
// TestEvexGprGroundTruth covers the scalar conversions between vector and
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
// general-purpose registers; the signed and truncated VCVT{,T}S{D,S}2SI
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv; byte
// for byte against the Go assembler, including memory sources and extended
// GPRs.
func TestEvexGprGroundTruth(t *testing.T) {
@@ -641,6 +675,15 @@ func TestEvexErrors(t *testing.T) {
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
// VEX-only mnemonics reject registers only EVEX can encode.
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
// The scalar EVEX move matches its VEX twin and the Go assembler:
// XMM↔memory only, never reg-reg and never a wider register (the
// toolchain rejects every one of these shapes).
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
@@ -649,71 +692,6 @@ func TestEvexErrors(t *testing.T) {
}
}
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
// kernel — all functions plus the file-global idx16 constant — and checks
// that the static-symbol load resolves to the right bytes in the image.
// Skipped when the sibling repository is not checked out.
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
}
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
if len(img.Funcs) != 10 {
t.Errorf("functions = %d, want 10", len(img.Funcs))
}
// idx16 as the DATA directives define it: dwords 1..16.
idx := make([]byte, 0, 64)
for i := 1; i <= 16; i++ {
idx = append(idx, byte(i), 0, 0, 0)
}
image := img.Bytes()
base := img.Symbols["idx16"]
if base == 0 {
t.Fatal("idx16 not laid out")
}
if got := image[base : base+64]; hexCompact(got) != hexCompact(idx) {
t.Errorf("idx16 contents %x, want %x", got, idx)
}
// The VMOVDQU32 idx16(SB), Z13 load (62 71 7e 48 6f 2d + rel32) must
// resolve to idx16 within the image.
loads := 0
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
pat := []byte{0x62, 0x71, 0x7e, 0x48, 0x6f, 0x2d}
for pos := 0; ; {
i := indexOf(code[pos:], pat)
if i < 0 {
break
}
i += pos
rel := int32(uint32(code[i+6]) | uint32(code[i+7])<<8 | uint32(code[i+8])<<16 | uint32(code[i+9])<<24)
target := fn.Offset + i + 10 + int(rel)
if target != base {
t.Errorf("%s: idx16 load at +%d targets 0x%x, want 0x%x", fn.Name, i, target, base)
}
loads++
pos = i + 10
}
}
if loads != 1 {
t.Errorf("idx16 loads found = %d, want 1", loads)
}
}
// hexCompact renders bytes as a lowercase hex string without separators.
func hexCompact(b []byte) string {
const hexdig = "0123456789abcdef"
@@ -724,17 +702,3 @@ func hexCompact(b []byte) string {
}
return string(out)
}
// indexOf returns the index of the first occurrence of pat in b, or -1.
func indexOf(b, pat []byte) int {
for i := 0; i+len(pat) <= len(b); i++ {
j := 0
for j < len(pat) && b[i+j] == pat[j] {
j++
}
if j == len(pat) {
return i
}
}
return -1
}
+342 -89
View File
@@ -10,11 +10,12 @@ import (
"os"
"os/exec"
"path/filepath"
"strings"
"sync"
)
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link
// consumes directly — so gasm-assembled functions drop into a go build
// This file emits GOOBJ, the Go toolchain's object format, which cmd/link
// consumes directly, so gasm-assembled functions drop into a go build
// without the Go assembler. The layout follows cmd/internal/goobj: a
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its
// block offsets, a string table, symbol definitions, the relocation /
@@ -22,11 +23,20 @@ import (
//
// The object carries what the linker requires of an assembly object: the
// functions (non-package symbols, as cmd/asm emits them), the GLOBL data,
// one FuncInfo per function, and the pc-value tables (pcsp, pcfile,
// pcline, pcinline). DWARF and the implicit funcdata symbols are omitted;
// the linker fills their defaults.
// one FuncInfo per function, the per-function DWARF symbols (the
// .debug_line program and the subprogram DIE, which the linker's DWARF
// pass reads verbatim), and the pc-value tables (pcsp, pcfile, pcline,
// pcinline). The implicit funcdata symbols are omitted; the linker fills
// their defaults.
//
// emitGOObject is architecture-agnostic; the per-architecture GOObject*
// methods supply the toolchain preamble, the MinLC (pc-value delta unit)
// and the relocation-type mapping for code relocations.
// GOOBJ block indices (cmd/internal/goobj).
// GOOBJ block indices (cmd/internal/goobj). These MUST match the real
// archive layout: the emitter writes the header offsets per index and the
// reader (groundtruth, goobj_resolve) parses real Go archives with them.
// blkAutolib is unused by the emitter but still defines index 0.
const (
blkAutolib = iota
blkPkgIdx
@@ -51,26 +61,31 @@ const (
// Symbol kinds used by assembly objects (cmd/internal/objabi).
const (
kindSTEXT = 1
kindSRODATA = 3
kindSDATA = 7
kindSTEXT = 1
kindSRODATA = 3
kindSDATA = 7
kindSDWARFFCN = 14
kindSDWARFLINES = 20
)
// Symbol flags (cmd/internal/goobj).
// Symbol flags (cmd/internal/goobj). The linkname flag is set only for
// //go:linkname symbols (and main.main); ordinary assembly symbols carry
// none, matching cmd/asm's output.
const (
symFlagDupok = 0x01
symFlagNoSplit = 0x10
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
symABIStatic = 0xffff
)
// Aux entry types (cmd/internal/goobj).
const (
auxFuncInfo = 1
auxPcsp = 7
auxPcfile = 8
auxPcline = 9
auxPcinline = 10
auxFuncInfo = 1
auxDwarfInfo = 3
auxDwarfLines = 6
auxPcsp = 7
auxPcfile = 8
auxPcline = 9
auxPcinline = 10
)
// FuncInfo flags (internal/abi).
@@ -80,14 +95,78 @@ const (
)
// Relocation types (cmd/internal/objabi).
const relocPCRel = 14
// R_ADDR, R_CALL, R_PCREL and R_TLS_LE are stable across Go versions.
const (
relocAddr = 1 // R_ADDR
relocCall = 7 // R_CALL
relocPCRel = 14 // R_PCREL
relocTLSLE = 15 // R_TLS_LE
)
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
// installed Go toolchain. The value shifted between Go 1.26 (103) and
// Go 1.27 (106) because new LoongArch relocations were inserted before it.
func relocDWTXTADDRU4() uint16 {
if isGo127OrLater() {
return 106
}
return 103
}
var (
goVersionOnce sync.Once
goVersionGT26 bool
)
// isGo127OrLater reports whether the installed Go toolchain is 1.27 or later.
func isGo127OrLater() bool {
goVersionOnce.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
return
}
out, err := exec.Command(goBin, "version").Output()
if err != nil {
return
}
// "go version go1.27rc1 linux/amd64"
s := string(out)
for _, prefix := range []string{"go version go1.27", "go version go1.28", "go version go1.29", "go version go2."} {
if strings.Contains(s, prefix) {
goVersionGT26 = true
return
}
}
})
return goVersionGT26
}
// Special package indices for symbol references.
const (
pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb
pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb
pkgIdxBuiltin = 0x7ffffffc
)
// goobjBuiltinMorestackNoctxt is the index of runtime.morestack_noctxt in
// cmd/internal/goobj/builtinlist.go of the toolchain the object targets
// (246 since Go 1.25; the list is append-only).
const goobjBuiltinMorestackNoctxt = 246
// goobjBuiltinMorestack is the builtin reference the toolchain emits for the
// stack-guard call.
var goobjBuiltinMorestack = "runtime\u00b7morestack_noctxt"
// isCallReloc reports whether k is one of the per-arch call relocations a
// direct branch to a TEXT symbol carries.
func isCallReloc(k RelocKind) bool {
switch k {
case RelCall, RelRISCVJal, RelArm64Branch, RelLoong64Branch:
return true
}
return false
}
const goobjMagic = "\x00go120ld"
// goSym is one symbol definition under construction.
@@ -110,58 +189,51 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte {
return binary.LittleEndian.AppendUint32(b, s.align)
}
// dwarfRelocSet attaches emitter-generated relocations (the DWARF
// lines/info symbols' address references) to a definition index.
type dwarfRelocSet struct {
si int
relocs []goobjReloc
}
// GOObject returns the image as a GOOBJ object file for the given package
// path (the linker qualifies the exported symbols with it, the way cmd/asm
// does with its -p flag). srcPath names the source file recorded in the
// object's file table and line tables. The toolchain's object preamble is
// captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object.
// toolchain it was produced on, exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
pre, err := toolchainObjectPreamble()
if err != nil {
return nil, err
}
// amd64: MinLC 1, R_PCREL for displacements, R_CALL for calls and
// R_TLS_LE for the stack-guard TLS load.
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelCall:
return relocCall, 4
case RelTLSLE:
return relocTLSLE, 4
default:
return relocPCRel, 4
}
})
}
// The symbol tables. Package definitions: the GLOBL symbols, then one
// anonymous FuncInfo symbol per function. Non-package definitions: the
// pc-value tables and the functions themselves, as cmd/asm lays them
// out. defIdx maps a GLOBL's bare name to its definition index for the
// relocations; fnNpIdx maps a function to its non-package index.
var defs []goSym
var defData [][]byte
defIdx := map[string]int{}
for _, d := range img.DataSyms {
name := d.Name
if !d.Static {
name = pkgPath + "." + name
}
typ := uint8(kindSDATA)
if d.Rodata {
typ = kindSRODATA
}
flag := uint8(0)
if d.Dupok {
flag = symFlagDupok
}
abi := uint16(0)
if d.Static {
abi = symABIStatic
}
defIdx[d.Name] = len(defs)
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
for i := range img.Funcs {
data := marshalFuncInfo(img.Funcs[i])
fnFiIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
defData = append(defData, data)
// emitGOObject assembles the GOOBJ payload for any architecture. pre is
// the toolchain's object preamble; minLC is the architecture's minimum
// instruction length, the unit of the pc-value table deltas; relocField
// maps a code relocation to its objabi relocation type and the width of
// the instruction field the linker writes.
func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, relocField func(Reloc) (uint16, uint8)) ([]byte, error) {
if pkgPath == "" {
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
// The non-package definitions first, the DWARF symbols reference the
// functions by these indices: per function the four pc-value tables
// and the function itself, as cmd/asm lays them out.
type npSym struct {
sym goSym
data []byte
@@ -175,10 +247,10 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
data []byte
dst *int
}{
{pcspTable(fn), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size), &pcIdx[i].inl},
{pcspTable(fn, minLC), &pcIdx[i].sp},
{pcValueFlat(0, fn.Size, minLC), &pcIdx[i].file},
{pcValueFlat(int32(fn.Line), fn.Size, minLC), &pcIdx[i].line},
{pcValueFlat(-1, fn.Size, minLC), &pcIdx[i].inl},
}
for _, t := range tables {
*t.dst = len(nps)
@@ -201,46 +273,212 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
fnNpIdx[i] = len(nps)
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
// The linker writes the resolved displacement into the field;
// leave it zero, as cmd/asm's object does.
// Only the amd64 encoder resolves file-local static symbols
// into a disp32 field at assemble time; GOOBJ must leave that
// field zero for the linker to fill. The RISC-V and LoongArch
// encoders emit zero immediates with a relocation instead, and
// their relocations cover whole AUIPC/pcalau12i pairs, so
// zeroing r.Off would erase the opcode/register bits the linker
// preserves when it patches only the immediate.
if r.Kind != RelPCRel32 && r.Kind != RelCall {
continue
}
if r.Off >= 0 && r.Off+4 <= len(code) {
code[r.Off], code[r.Off+1], code[r.Off+2], code[r.Off+3] = 0, 0, 0, 0
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, size: uint32(fn.Size)},
data: code,
})
}
// The package definitions: the GLOBL symbols, then, per function, the
// FuncInfo and the two DWARF symbols (the .debug_line program and the
// subprogram DIE). defIdx maps a GLOBL's bare name to its definition
// index for the code relocations.
var defs []goSym
var defData [][]byte
defIdx := map[string]int{}
for _, d := range img.DataSyms {
name := d.Name
if !d.Static {
name = pkgPath + "." + name
}
typ := uint8(kindSDATA)
if d.Rodata {
typ = kindSRODATA
}
flag := uint8(0)
if d.Dupok {
flag = symFlagDupok
}
abi := uint16(0)
if d.Static {
abi = symABIStatic
}
defIdx[d.Name] = len(defs)
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, size: uint32(d.Size)})
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
fnLinesIdx := make([]int, len(img.Funcs))
fnDIEIdx := make([]int, len(img.Funcs))
var dwarfRelocs []dwarfRelocSet
for i, fn := range img.Funcs {
data := marshalFuncInfo(fn)
fnFiIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))})
defData = append(defData, data)
name := fn.Name
if !fn.Static {
name = pkgPath + "." + name
}
// The DWARF symbols: the .debug_line state-machine program and the
// subprogram DIE, both referencing the function by its non-package
// index (package definitions, like cmd/asm's).
lines, lrel := goobjDwarfLines(fn, fnNpIdx[i])
fnLinesIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDWARFLINES, size: uint32(len(lines))})
defData = append(defData, lines)
die, drel := goobjDwarfInfo(fn, name, fnNpIdx[i])
fnDIEIdx[i] = len(defs)
defs = append(defs, goSym{typ: kindSDWARFFCN, size: uint32(len(die))})
defData = append(defData, die)
dwarfRelocs = append(dwarfRelocs,
dwarfRelocSet{si: fnLinesIdx[i], relocs: lrel},
dwarfRelocSet{si: fnDIEIdx[i], relocs: drel},
)
}
// Index the non-package TEXT definitions by short name for the internal
// call references.
textNpIdx := map[string]int{}
for i, fn := range img.Funcs {
textNpIdx[fn.Name] = fnNpIdx[i]
}
// Resolve external symbol references (cross-package). Build the
// package index table and determine each external symbol's SymIdx
// by reading the target package's export data.
var extPkgTable []string
var extPkgIdx map[string]int
var extSymIdx map[string]int
if len(img.Externals) > 0 {
// The morestack call is a builtin reference, not a resolved external.
var need []string
for _, n := range img.Externals {
if n == goobjBuiltinMorestack {
continue
}
need = append(need, n)
}
if len(need) > 0 {
var err error
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(need)
if err != nil {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
}
}
}
// Relocations, per defined symbol in definition order (package defs,
// then non-package defs). Only file-local GLOBL references resolve;
// external symbols need the import machinery of a later increment.
// then non-package defs).
nsyms := len(defs) + len(nps)
symRelocs := make([][]byte, nsyms) // flat 23-byte records
for i, fn := range img.Funcs {
si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs {
if r.External {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q is not supported yet", r.Name)
typ, size := relocField(r)
if r.Kind == RelTLSLE {
// The TLS load has no symbol: {0, 0} is the nil ref.
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], 0)
binary.LittleEndian.PutUint32(rec[19:], 0)
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
if r.External && r.Name == goobjBuiltinMorestack {
// The stack-guard morestack call uses the toolchain's
// builtin reference.
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
if r.External {
// Split package-qualified name: "runtime·morestack" → runtime, morestack.
pkg, name := splitQualified(r.Name)
if pkg == "" {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
}
pIdx, ok := extPkgIdx[pkg]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
}
sIdx, ok := extSymIdx[pkg+"·"+name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
pkg := uint32(pkgIdxSelf)
di, ok := defIdx[r.Name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
// A call to a TEXT function of the same file references the
// non-package definition table.
ni, isText := textNpIdx[r.Name]
if !isText || !isCallReloc(r.Kind) {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
pkg = pkgIdxNone
di = ni
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = 4 // field width
binary.LittleEndian.PutUint16(rec[5:], relocPCRel)
rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[15:], pkg)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...)
}
}
// The DWARF symbols' own relocations (the function address references).
for _, ds := range dwarfRelocs {
for _, r := range ds.relocs {
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(r.off))
rec[4] = r.siz
binary.LittleEndian.PutUint16(rec[5:], r.typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.add))
binary.LittleEndian.PutUint32(rec[15:], r.pkg)
binary.LittleEndian.PutUint32(rec[19:], r.sym)
symRelocs[ds.si] = append(symRelocs[ds.si], rec[:]...)
}
}
// Aux entries per function: FuncInfo, then the four pc tables.
// References into the non-package table use pkgIdxNone.
// Aux entries per function: FuncInfo, the DWARF symbols, then the four
// pc tables. References into the non-package table use pkgIdxNone.
symAux := make([][]byte, nsyms)
for i := range img.Funcs {
si := len(defs) + fnNpIdx[i]
@@ -252,16 +490,20 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
symAux[si] = append(symAux[si], rec[:]...)
}
aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i]))
aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl))
aux(auxDwarfInfo, pkgIdxSelf, uint32(fnDIEIdx[i]))
aux(auxDwarfLines, pkgIdxSelf, uint32(fnLinesIdx[i]))
// The pc-table references are 0-based within the non-package
// definitions; the loader adds the package-definition count itself.
aux(auxPcsp, pkgIdxNone, uint32(pcIdx[i].sp))
aux(auxPcfile, pkgIdxNone, uint32(pcIdx[i].file))
aux(auxPcline, pkgIdxNone, uint32(pcIdx[i].line))
aux(auxPcinline, pkgIdxNone, uint32(pcIdx[i].inl))
}
// The string table. Absolute offsets: it starts right after the
// 96-byte header (magic, fingerprint, flags, the 19 block offsets).
const headerSize = 8 + 8 + 4 + 4*(blkEnd+1)
strTab := []byte{}
var strTab []byte
strOff := map[string]uint32{}
addStr := func(s string) {
if _, ok := strOff[s]; ok {
@@ -291,7 +533,16 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
for _, s := range nps {
npdefBlk = s.sym.append(npdefBlk, strOff)
}
pkgIdxBlk := stringRef(nil, "") // index 0: the dummy invalid package
// Package index table: index 0 is the dummy invalid package.
// External packages follow, in pkgIdx order.
for _, pkg := range extPkgTable {
addStr(pkg)
}
pkgIdxBlk := stringRef(nil, "") // index 0: dummy
for _, pkg := range extPkgTable {
pkgIdxBlk = stringRef(pkgIdxBlk, pkg)
}
fileBlk := stringRef(nil, srcPath)
var relocBlk, auxBlk, dataBlk []byte
@@ -299,7 +550,7 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
auxIdxBlk := make([]byte, 0, 4*(nsyms+1))
dataIdxBlk := make([]byte, 0, 4*(nsyms+1))
var nr, na, nd uint32
for si := 0; si < nsyms; si++ {
for si := range nsyms {
relocIdxBlk = binary.LittleEndian.AppendUint32(relocIdxBlk, nr)
auxIdxBlk = binary.LittleEndian.AppendUint32(auxIdxBlk, na)
dataIdxBlk = binary.LittleEndian.AppendUint32(dataIdxBlk, nd)
@@ -340,7 +591,7 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
// The fingerprint stays zero, as cmd/asm leaves it.
binary.LittleEndian.PutUint32(payload[16:], 4) // ObjFlagFromAssembly
off := uint32(headerSize + len(strTab))
for i := 0; i < blkEnd; i++ {
for i := range blkEnd {
binary.LittleEndian.PutUint32(payload[20+4*i:], off)
off += uint32(len(blocks[i]))
}
@@ -374,19 +625,21 @@ func marshalFuncInfo(fn FuncLayout) []byte {
}
// pcValueFlat encodes a pc-value table holding v over the whole function.
func pcValueFlat(v int32, size int) []byte {
// The pc deltas are in MinLC units (the runtime scales them by the
// architecture's minimum instruction length).
func pcValueFlat(v int32, size, minLC int) []byte {
// The table is delta-encoded from an implicit value of -1: a varint
// value delta, an unsigned pc delta to the end, and a zero terminator.
out := binary.AppendVarint(nil, int64(v)+1)
out = binary.AppendUvarint(out, uint64(size))
out = binary.AppendUvarint(out, uint64(size/minLC))
return append(out, 0)
}
// pcspTable encodes the stack-adjustment table: the SP delta in effect at
// every pc, from the function's prologue and epilogue boundaries.
func pcspTable(fn FuncLayout) []byte {
func pcspTable(fn FuncLayout, minLC int) []byte {
if len(fn.Spadj) == 0 {
return pcValueFlat(0, fn.Size)
return pcValueFlat(0, fn.Size, minLC)
}
pts := make([]SpadjStep, 0, len(fn.Spadj)+1)
pts = append(pts, SpadjStep{PC: 0, Value: 0})
@@ -394,11 +647,11 @@ func pcspTable(fn FuncLayout) []byte {
out := binary.AppendVarint(nil, int64(pts[0].Value)+1)
cur, old := pts[0].PC, pts[0].Value
for _, p := range pts[1:] {
out = binary.AppendUvarint(out, uint64(p.PC-cur))
out = binary.AppendUvarint(out, uint64((p.PC-cur)/minLC))
out = binary.AppendVarint(out, int64(p.Value-old))
cur, old = p.PC, p.Value
}
out = binary.AppendUvarint(out, uint64(fn.Size-cur))
out = binary.AppendUvarint(out, uint64((fn.Size-cur)/minLC))
return append(out, 0)
}
+185
View File
@@ -0,0 +1,185 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
)
// This file generates the per-function DWARF symbols the linker's DWARF
// pass requires of an assembly object, byte-identical to what cmd/asm
// emits: the .debug_line state-machine program (SDWARFLINES) and the
// subprogram DIE (SDWARFFCN). The linker copies the DIE and line-program
// bytes verbatim into .debug_info and .debug_line, fixing up their
// relocations, so the formats here must match cmd/internal/dwarf's
// DW_ABRV_FUNCTION and generateDebugLinesSymbol exactly.
//
// DWARF5 is assumed throughout (the toolchain's default on Linux and the
// other non-Darwin targets gasm supports).
// Line-program parameters (cmd/internal/obj/dwarf.go).
const (
dwLineBase = -4
dwLineRange = 10
dwOpcodeBase = 11
dwPCRange = (255 - dwOpcodeBase) / dwLineRange
)
// goobjReloc is one relocation attached to an emitter-generated symbol
// (the DWARF lines/info symbols), in goobj's on-disk encoding fields.
type goobjReloc struct {
off int32
siz uint8
typ uint16
add int64
pkg uint32
sym uint32
}
// goobjDwarfLines builds the function's .debug_line state-machine program:
// an LNE_set_address extended opcode establishing the function's start
// address (carrying the R_ADDR relocation), one row per source line
// change across the function's instructions, an advance to the end of the
// function and an end-of-sequence opcode. The linker appends these bytes
// after the unit's line header, so they must start with the address and
// leave the state machine terminated.
func goobjDwarfLines(fn FuncLayout, fnNpIdx int) ([]byte, []goobjReloc) {
// Rows: the prologue, if any, then the body instructions (fn.Lines
// covers the body only). The first body offset > 0 means a prologue
// precedes it; the toolchain reports the prologue on the TEXT line.
pts := make([]LineEntry, 0, len(fn.Lines)+1)
if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 {
pts = append(pts, LineEntry{Offset: 0, Line: fn.Line})
}
pts = append(pts, fn.Lines...)
out := []byte{0, 9, 2, 0, 0, 0, 0, 0, 0, 0, 0} // LNE_set_address, address zeroed
relocs := []goobjReloc{{
off: 3, siz: 8, typ: relocAddr,
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
}}
// The state machine starts at line 1, pc 0 (function-relative); the
// implicit initial pc is the function entry, so the first pc delta is
// against 0.
line := int64(1)
pc := uint64(0)
for _, p := range pts {
if p.Line == 0 || uint64(p.Offset) < pc {
continue
}
// Rows mark source-line changes only; the pc delta is measured from
// the previous row, not the previous instruction.
if int64(p.Line) == line {
continue
}
deltaPC := uint64(p.Offset) - pc
deltaLC := int64(p.Line) - line
out = dwPutPCLCDelta(out, deltaPC, deltaLC)
line, pc = int64(p.Line), uint64(p.Offset)
}
// Cover the rest of the function and close the sequence.
if end := uint64(fn.Size) - pc; end > 0 {
out = append(out, 2) // DW_LNS_advance_pc
out = binary.AppendUvarint(out, end)
}
out = append(out, 0, 1, 1) // LNE_end_sequence
return out, relocs
}
// dwPutPCLCDelta encodes one (pcDelta, lineDelta) step as the shortest
// special opcode plus any standard-opcode remainder, exactly like
// cmd/internal/obj's putpclcdelta.
func dwPutPCLCDelta(b []byte, deltaPC uint64, deltaLC int64) []byte {
opcode := dwSelectOpcode(deltaPC, deltaLC)
deltaPC -= uint64((opcode - dwOpcodeBase) / dwLineRange)
deltaLC -= (opcode-dwOpcodeBase)%dwLineRange + dwLineBase
// The remainder: standard opcodes first, then the special opcode
// (which emits the row).
if deltaPC != 0 {
switch {
case deltaPC <= uint64(dwPCRange):
opcode -= dwLineRange * int64(uint64(dwPCRange)-deltaPC)
b = append(b, 8) // DW_LNS_const_add_pc
case (1<<14) <= deltaPC && deltaPC < (1<<16):
b = append(b, 9) // DW_LNS_fixed_advance_pc
b = binary.LittleEndian.AppendUint16(b, uint16(deltaPC))
default:
b = append(b, 2) // DW_LNS_advance_pc
b = binary.AppendUvarint(b, deltaPC)
}
}
if deltaLC != 0 {
b = append(b, 3) // DW_LNS_advance_line
b = binary.AppendVarint(b, deltaLC)
}
return append(b, byte(opcode))
}
// dwSelectOpcode picks the special opcode for (deltaPC, deltaLC) per
// cmd/internal/obj's putpclcdelta selection logic.
func dwSelectOpcode(deltaPC uint64, deltaLC int64) int64 {
switch {
case deltaLC < dwLineBase:
if deltaPC >= uint64(dwPCRange) {
return dwOpcodeBase + dwLineRange*dwPCRange
}
return dwOpcodeBase + dwLineRange*int64(deltaPC)
case deltaLC < dwLineBase+dwLineRange:
if deltaPC >= uint64(dwPCRange) {
op := int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*dwPCRange
if op > 255 {
op -= dwLineRange
}
return op
}
return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC)
default:
if deltaPC <= uint64(dwPCRange) {
op := min(int64(dwOpcodeBase)+(dwLineRange-1)+dwLineRange*int64(deltaPC), 255)
return op
}
switch deltaPC - uint64(dwPCRange) {
case uint64(dwPCRange), (1 << 7) - 1, (1 << 16) - 1, (1 << 21) - 1,
(1 << 28) - 1, (1 << 35) - 1, (1 << 42) - 1, (1 << 49) - 1,
(1 << 56) - 1, (1 << 63) - 1:
return 255
default:
// 250: the toolchain's "249" comment is stale.
return dwOpcodeBase + dwLineRange*dwPCRange - 1
}
}
}
// goobjDwarfInfo builds the function's DWARF5 subprogram DIE (abbrev
// DW_ABRV_FUNCTION): name, low_pc as a .debug_addr index (the
// R_DWTXTADDR_U4 relocation), high_pc as the size, the call-frame-CFA
// frame base, the decl file/line and the external flag. name is the
// symbol's object name (package-qualified unless static).
func goobjDwarfInfo(fn FuncLayout, name string, fnNpIdx int) ([]byte, []goobjReloc) {
out := []byte{3} // DW_ABRV_FUNCTION
out = append(out, name...)
out = append(out, 0)
addrx := len(out)
out = append(out, 0, 0, 0, 0) // DW_AT_low_pc: addrx slot, zeroed
out = binary.AppendUvarint(out, uint64(fn.Size))
out = append(out, 1, 0x9c) // DW_AT_frame_base: block1, DW_OP_call_frame_cfa
out = binary.LittleEndian.AppendUint32(out, 1)
out = binary.AppendUvarint(out, uint64(fn.Line))
if fn.Static {
out = append(out, 0)
} else {
out = append(out, 1) // DW_AT_external
}
out = append(out, 0) // end of children
relocs := []goobjReloc{{
off: int32(addrx), siz: 4, typ: relocDWTXTADDRU4(),
pkg: pkgIdxNone, sym: uint32(fnNpIdx),
}}
return out, relocs
}
+214
View File
@@ -0,0 +1,214 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"testing"
)
// TestDWSelectOpcode checks the special-opcode selection against
// hand-computed values for the boundary cases: line deltas below, inside
// and above the line range, and pc deltas at and beyond PC_RANGE (24).
func TestDWSelectOpcode(t *testing.T) {
cases := []struct {
deltaPC uint64
deltaLC int64
want int64
}{
{0, 2, 17}, // the common single-instruction step
{0, -4, 11}, // deltaLC == LINE_BASE
{0, -5, 11}, // deltaLC below LINE_BASE: opcode adds nothing
{0, 6, 20}, // deltaLC == LINE_BASE+LINE_RANGE, remainder via advance_line
{4, 1, 56}, // the 4-byte loong64 instruction step
{23, 1, 246}, // deltaPC == PC_RANGE-1
{24, 1, 246}, // deltaPC == PC_RANGE: wraps past 255
{25, 1, 246}, // deltaPC past PC_RANGE (the const_add_pc remainder adjusts it later)
{100, 1, 246},
{151, 10, 255}, // deltaPC-PC_RANGE == (1<<7)-1, large line delta
{100, 10, 250}, // deltaPC-PC_RANGE not on a switch boundary
{23, 10, 250}, // large line delta inside PC_RANGE
}
for _, c := range cases {
if got := dwSelectOpcode(c.deltaPC, c.deltaLC); got != c.want {
t.Errorf("dwSelectOpcode(%d, %d) = %d, want %d", c.deltaPC, c.deltaLC, got, c.want)
}
}
}
// decodeDWLineProgram decodes a .debug_line state-machine program (as
// emitted by goobjDwarfLines) into (pc, line) rows.
func decodeDWLineProgram(t *testing.T, b []byte) (pcs []uint64, lines []int64) {
t.Helper()
pc, line := uint64(0), int64(1)
emit := func() {
if len(pcs) == 0 || pcs[len(pcs)-1] != pc || lines[len(lines)-1] != line {
pcs = append(pcs, pc)
lines = append(lines, line)
}
}
advancePC := func(delta uint64) { pc += delta }
advanceLine := func(delta int64) { line += delta }
for i := 0; i < len(b); {
op := b[i]
i++
switch {
case op == 0: // extended opcode
ln, n := binary.Uvarint(b[i:])
i += n
sub := b[i]
i++
_ = ln
switch sub {
case 2: // DW_LNE_set_address: 8-byte address
pc = binary.LittleEndian.Uint64(b[i:])
i += 8
case 1: // DW_LNE_end_sequence
// terminates the sequence; no new row
}
case op == 2: // DW_LNS_advance_pc
v, n := binary.Uvarint(b[i:])
i += n
advancePC(v)
case op == 3: // DW_LNS_advance_line
v, n := binary.Varint(b[i:])
i += n
advanceLine(v)
case op == 8: // DW_LNS_const_add_pc
advancePC(uint64(dwPCRange))
case op == 9: // DW_LNS_fixed_advance_pc
advancePC(uint64(binary.LittleEndian.Uint16(b[i:])))
i += 2
case op >= dwOpcodeBase: // special opcode
advancePC(uint64((int64(op) - dwOpcodeBase) / dwLineRange))
advanceLine((int64(op)-dwOpcodeBase)%dwLineRange + dwLineBase)
emit()
}
}
return pcs, lines
}
// TestGoobjDwarfLinesRows checks the emitted line program's rows for
// synthetic functions: a zero-frame function with one instruction per
// line, a framed function (the prologue row is prepended on the TEXT
// line), instructions sharing a line, and a function with a large pc gap
// (the const_add_pc remainder path).
func TestGoobjDwarfLinesRows(t *testing.T) {
cases := []struct {
name string
fn FuncLayout
want [][2]int64 // (pc, line)
}{
{
"one instruction per line",
FuncLayout{Size: 20, Line: 2, Lines: []LineEntry{
{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7},
}},
[][2]int64{{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7}},
},
{
"framed: prologue row on the TEXT line",
FuncLayout{Size: 24, Line: 2, Lines: []LineEntry{
{12, 3}, {16, 4},
}},
[][2]int64{{0, 2}, {12, 3}, {16, 4}},
},
{
"instructions sharing a line fold into one row",
FuncLayout{Size: 16, Line: 2, Lines: []LineEntry{
{0, 3}, {4, 3}, {8, 4}, {12, 4},
}},
[][2]int64{{0, 3}, {8, 4}},
},
{
"large gap crosses PC_RANGE",
FuncLayout{Size: 60, Line: 2, Lines: []LineEntry{
{0, 3}, {40, 4},
}},
[][2]int64{{0, 3}, {40, 4}},
},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
prog, relocs := goobjDwarfLines(c.fn, 0)
if len(relocs) != 1 || relocs[0].off != 3 || relocs[0].siz != 8 || relocs[0].typ != relocAddr || relocs[0].sym != 0 {
t.Fatalf("relocs = %+v", relocs)
}
pcs, lines := decodeDWLineProgram(t, prog)
if len(pcs) != len(c.want) {
t.Fatalf("rows = %d (%v / %v), want %d", len(pcs), pcs, lines, len(c.want))
}
for i, w := range c.want {
if pcs[i] != uint64(w[0]) || lines[i] != w[1] {
t.Errorf("row %d = (%d, %d), want (%d, %d)", i, pcs[i], lines[i], w[0], w[1])
}
}
})
}
}
// TestGoobjDwarfInfo checks the subprogram DIE for an exported and a
// static function: the abbrev, name, high_pc, frame base, decl file/line,
// the external flag and the addrx relocation position.
func TestGoobjDwarfInfo(t *testing.T) {
fn := FuncLayout{Size: 20, Line: 2}
die, relocs := goobjDwarfInfo(fn, "pkg.f", 3)
want := []byte{
0x03,
'p', 'k', 'g', '.', 'f', 0,
0, 0, 0, 0, // addrx slot at offset 7
0x14, // high_pc: 20
0x01, 0x9c, // frame_base
0x01, 0, 0, 0, // decl_file 1
0x02, // decl_line 2
0x01, // external
0x00, // end of children
}
if !bytes.Equal(die, want) {
t.Errorf("DIE = %x, want %x", die, want)
}
if len(relocs) != 1 || relocs[0].off != 7 || relocs[0].siz != 4 || relocs[0].typ != relocDWTXTADDRU4() || relocs[0].sym != 3 {
t.Errorf("relocs = %+v", relocs)
}
// A static function carries no external flag and no package prefix.
fn.Static = true
die, _ = goobjDwarfInfo(fn, "f", 1)
if die[len(die)-2] != 0 {
t.Errorf("static external flag = %d, want 0", die[len(die)-2])
}
}
// TestDwPutPCLCDeltaRemainders checks the standard-opcode remainders:
// const_add_pc and fixed_advance_pc after a special opcode.
func TestDwPutPCLCDeltaRemainders(t *testing.T) {
// deltaPC 25 past PC_RANGE: opcode 26 covers (1, 1), const_add_pc
// covers the remaining 23 pc and 0 line.
got := dwPutPCLCDelta(nil, 25, 1)
if !bytes.Equal(got, []byte{8, 26}) {
t.Errorf("25/1 = %x, want [8 1a]", got)
}
// deltaPC 20000: opcode 246 covers 23, fixed_advance_pc covers the
// remaining 19977.
got = dwPutPCLCDelta(nil, 20000, 1)
if got[0] != 9 || binary.LittleEndian.Uint16(got[1:]) != 19977 || got[3] != 246 {
t.Errorf("20000/1 = %x, want fixed_advance_pc 19977 then 246", got)
}
// Line remainder: deltaLC 10 leaves 5 past the opcode's reach, encoded
// as advance_line 5 (zigzag 0x0a) before opcode 250.
got = dwPutPCLCDelta(nil, 23, 10)
if !bytes.Equal(got, []byte{3, 0x0a, 250}) {
t.Errorf("23/10 = %x, want [03 0a fa]", got)
}
// Negative line remainder: deltaLC -5 leaves advance_line -1 (zigzag
// 0x01) after opcode 11.
got = dwPutPCLCDelta(nil, 0, -5)
if !bytes.Equal(got, []byte{3, 1, 11}) {
t.Errorf("0/-5 = %x, want [03 01 0b]", got)
}
}
+370
View File
@@ -0,0 +1,370 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"fmt"
"os"
"os/exec"
"strings"
)
// exportPath returns the export file path for a given import path by running
// "go list -export". The result is cached so repeated calls for the same
// package are fast.
func exportPath(importPath string) (string, error) {
cmd := exec.Command("go", "list", "-json", "-export", importPath)
out, err := cmd.Output()
if err != nil {
return "", fmt.Errorf("go list %s: %w", importPath, err)
}
// Quick JSON extraction: find "Export": "…"
const key = `"Export": "`
i := bytes.Index(out, []byte(key))
if i < 0 {
return "", fmt.Errorf("go list %s: no Export field", importPath)
}
start := i + len(key)
end := bytes.IndexByte(out[start:], '"')
if end < 0 {
return "", fmt.Errorf("go list %s: malformed Export field", importPath)
}
return string(out[start : start+end]), nil
}
// resolveExternalGOOBJ resolves a set of external symbol references into
// (package index, symbol index) pairs suitable for GOOBJ emission.
//
// refs maps package import paths to the symbol names referenced from that
// package. The returned pkgIdx maps each import path to its position in
// the blkPkgIdx table, which reserves index 0 for the dummy invalid
// package (cmd/internal/obj/sym.go: "0 is invalid index"; the loader's
// reader loop starts at 1), so package i sits at block index i+1 and its
// relocations carry i+1. symIdx gives each symbol's index within its
// package.
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
pkgIdx = make(map[string]int, len(refs))
symIdx = make(map[string]int)
// Assign package indices in sorted order for determinism.
packages := sortedPkgRefs(refs)
for i, pkg := range packages {
// Block index 0 is the dummy invalid package; the first real
// package starts at 1.
pkgIdx[pkg.path] = i + 1
exp, err := exportPath(pkg.path)
if err != nil {
return nil, nil, err
}
data, err := os.ReadFile(exp)
if err != nil {
return nil, nil, err
}
gobj, err := extractGOOBJ(data)
if err != nil {
return nil, nil, fmt.Errorf("%s: %w", pkg.path, err)
}
for _, name := range pkg.syms {
idx := gobj.findSymbol(pkg.path, name)
if idx < 0 {
return nil, nil, fmt.Errorf("symbol %s·%s not found in export data of %s", pkg.path, name, pkg.path)
}
symIdx[pkg.path+"·"+name] = idx
}
}
return pkgIdx, symIdx, nil
}
type pkgRef struct {
path string
syms []string
}
func sortedPkgRefs(refs map[string][]string) []pkgRef {
var pkgs []pkgRef
for pkg, syms := range refs {
pkgs = append(pkgs, pkgRef{pkg, syms})
}
// Simple insertion sort, the list is tiny (usually 1-3 packages).
for i := 1; i < len(pkgs); i++ {
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
}
}
return pkgs
}
// extractGOOBJ finds the GOOBJ data in an ar archive and returns a parsed
// goobjFile. The archive member _go_.o contains the "go object …\n!\n"
// preamble followed by the GOOBJ payload; __.PKGDEF is the compiler export
// data (type information) and is not the GOOBJ object.
func extractGOOBJ(data []byte) (*goobjFile, error) {
if len(data) < 8 || string(data[:8]) != "!<arch>\n" {
return nil, fmt.Errorf("not an ar archive")
}
pos := 8
for pos+60 <= len(data) {
hdr := data[pos : pos+60]
pos += 60
// Parse ar header fields.
name := strings.TrimRight(string(hdr[:16]), " /")
size := parseArDecimal(hdr[48:58])
if size < 0 {
return nil, fmt.Errorf("invalid ar header: bad size")
}
if pos+size > len(data) {
return nil, fmt.Errorf("ar entry %q extends past end of file", name)
}
body := data[pos : pos+size]
pos += size
// ar pads to even bytes.
if pos%2 != 0 {
pos++
}
if name == "_go_.o" {
return parseGOOBJ(body)
}
}
return nil, fmt.Errorf("archive contains no _go_.o member")
}
// parseArDecimal parses a decimal number from a space-padded field.
func parseArDecimal(b []byte) int {
v := 0
for _, c := range b {
if c == ' ' {
continue
}
if c < '0' || c > '9' {
return -1
}
v = v*10 + int(c-'0')
}
return v
}
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
// blocks. The hashed blocks are kept raw: their symbols carry no names, only
// the loader needs their counts.
type goobjFile struct {
strTab []byte // string table, at headerSize + n
symdef []byte // blkSymdef raw block
hashed64 []byte // blkHashed64def raw block
hashed []byte // blkHasheddef raw block
npdef []byte // blkNonpkgdef raw block
}
// loaderIndexBase returns the index the first nonpkgdef symbol occupies in the
// loader's per-object symbol array. cmd/link lays the definition blocks out as
// symdef, hashed64def, hasheddef, nonpkgdef, nonpkgref (loader.go: preloadSyms
// fills r.syms in exactly that order, and resolve() indexes PkgIdxNone and
// cross-package SymIdx into it), so a symbol found in blkNonpkgdef carries the
// three leading blocks' symbol counts as its base.
func (f *goobjFile) loaderIndexBase() int {
return len(f.symdef)/recSymSize + len(f.hashed64)/recSymSize + len(f.hashed)/recSymSize
}
// symbols returns the names of the symdef and nonpkgdef blocks in
// definition order. Package definitions (blkSymdef) use fully-qualified
// names like "runtime.morestack"; non-package definitions (blkNonpkgdef)
// use bare names like "morestack". For lookups by index prefer
// findSymbol: it adds the hashed blocks' count the loader's array
// interleaves between the two.
func (f *goobjFile) symbols() []string {
return append(f.defNames(), f.npdefNames()...)
}
// findSymbol returns the index of a symbol within the loader's per-object
// symbol array, or -1 if not found. It first tries the fully-qualified
// name (pkg.name), then the bare name (assembly objects store dotless
// names, e.g. runtime's "gogo", for symbols other packages reach through
// a linkname).
func (f *goobjFile) findSymbol(pkg, name string) int {
base := f.loaderIndexBase()
qualified := pkg + "." + name
for i, s := range f.defNames() {
if s == qualified {
return i
}
}
for i, s := range f.npdefNames() {
if s == qualified {
return base + i
}
}
// Try bare name (for dotless assembly definitions).
for i, s := range f.defNames() {
if s == name {
return i
}
}
for i, s := range f.npdefNames() {
if s == name {
return base + i
}
}
return -1
}
// defNames returns names from blkSymdef only.
func (f *goobjFile) defNames() []string {
return f.readSymNames(f.symdef)
}
// npdefNames returns names from blkNonpkgdef.
func (f *goobjFile) npdefNames() []string {
return f.readSymNames(f.npdef)
}
// recSymSize is the size of one Sym record in the definition blocks
// (goobj.SymSize: stringRefSize + 2 + 1 + 1 + 1 + 4 + 4).
const recSymSize = 21
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
// size (u32), align (u32). nameOff is an absolute offset into the string
// table.
func (f *goobjFile) readSymNames(block []byte) []string {
const recSize = recSymSize
if len(block) < recSize {
return nil
}
n := len(block) / recSize
names := make([]string, 0, n)
for i := range n {
rec := block[i*recSize : (i+1)*recSize]
nameLen := binary.LittleEndian.Uint32(rec[0:4])
nameOff := binary.LittleEndian.Uint32(rec[4:8])
// nameOff is an absolute offset into the GOOBJ payload. The string
// table we have starts at goobjHeaderSize, so we subtract that.
if nameOff < goobjHeaderSize {
continue
}
relOff := nameOff - goobjHeaderSize
if relOff >= uint32(len(f.strTab)) || relOff+nameLen > uint32(len(f.strTab)) {
continue
}
names = append(names, string(f.strTab[relOff:relOff+nameLen]))
}
return names
}
const goobjHeaderSize = 8 + 8 + 4 + 4*(blkEnd+1) // magic + fingerprint + flags + 19 block offsets
// parseGOOBJ parses a raw GOOBJ payload (the data after the "\n!\n" preamble).
func parseGOOBJ(data []byte) (*goobjFile, error) {
// Find the "\n!\n" separator.
sep := []byte("\n!\n")
i := bytes.Index(data, sep)
if i < 0 {
// Maybe the data has no preamble (e.g. a raw .o file).
i = -3 // treat as if preamble starts before the data
}
payload := data[i+len(sep):]
if len(payload) < goobjHeaderSize {
return nil, fmt.Errorf("GOOBJ payload too short (%d bytes)", len(payload))
}
if string(payload[:8]) != goobjMagic {
return nil, fmt.Errorf("bad GOOBJ magic: %q", payload[:8])
}
// Read block offsets. The header layout is:
// [0:8] magic
// [8:16] fingerprint
// [16:20] flags
// [20:96] 19 × uint32 offsets
var offs [blkEnd + 1]uint32
for i := range blkEnd + 1 {
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
}
// The string table lives at headerSize.
strTabStart := uint32(goobjHeaderSize)
f := &goobjFile{
strTab: payload[strTabStart:offs[0]],
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
hashed64: blockSlice(payload, offs, blkHashed64def, blkHashed64def+1),
hashed: blockSlice(payload, offs, blkHasheddef, blkHasheddef+1),
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
}
return f, nil
}
// blockSlice extracts a block from the payload using its offset pair.
func blockSlice(payload []byte, offs [blkEnd + 1]uint32, start, end int) []byte {
if start < 0 || end > blkEnd || offs[end] < offs[start] {
return nil
}
beg := offs[start]
fin := offs[end]
if int(fin) > len(payload) || int(beg) > int(fin) {
return nil
}
return payload[beg:fin]
}
// resolveExternalSymbols is the high-level entry point for GOOBJ emission.
// Given a list of external symbol names (e.g. ["runtime·morestack",
// "runtime·g0"]), it returns the package-index table entries and a map from
// full symbol name to GOOBJ {pkgIdx, symIdx}.
//
// The package table entries should be written into blkPkgIdx, and the
// returned indices should replace pkgIdxSelf / placeholder values in the
// relocation records.
func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap map[string]int, symIdxMap map[string]int, err error) {
// Group references by package.
refs := make(map[string]map[string]bool)
for _, full := range externals {
pkg, name := splitQualified(full)
if refs[pkg] == nil {
refs[pkg] = make(map[string]bool)
}
refs[pkg][name] = true
}
// Convert maps to slices.
r := make(map[string][]string, len(refs))
for pkg, names := range refs {
for name := range names {
r[pkg] = append(r[pkg], name)
}
}
pkgIdx1, symIdx1, err := resolveExternalGOOBJ(r)
if err != nil {
return nil, nil, nil, err
}
// Build the package table in pkgIdx order. The indices are 1-based
// (0 is the dummy invalid package, written by the emitter itself), so
// the table without the dummy is indexed one below.
pkgTable = make([]string, len(pkgIdx1))
for pkg, idx := range pkgIdx1 {
pkgTable[idx-1] = pkg
}
return pkgTable, pkgIdx1, symIdx1, nil
}
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
// package path and local name. The separator is the middle dot (U+00B7),
// whose UTF-8 encoding is two bytes, so the search must be string-based:
// IndexByte would match only the second byte and leave the lead byte on
// the package path. If no separator is found, the symbol is assumed to be
// in the current package (empty pkg).
func splitQualified(full string) (pkg, name string) {
if before, after, ok := strings.Cut(full, "\u00b7"); ok {
return before, after
}
if before, after, ok := strings.Cut(full, "."); ok {
return before, after
}
return "", full
}
+70
View File
@@ -0,0 +1,70 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"os/exec"
"testing"
)
// TestReadRuntimeSymbols verifies the GOOBJ reader can extract and find
// symbols from the runtime package's compiled archive.
func TestReadRuntimeSymbols(t *testing.T) {
exp, err := exportPath("runtime")
if err != nil {
t.Skipf("cannot find runtime export: %v (need Go toolchain)", err)
}
data, err := os.ReadFile(exp)
if err != nil {
t.Skipf("cannot read runtime export: %v", err)
}
gobj, err := extractGOOBJ(data)
if err != nil {
t.Fatalf("extractGOOBJ: %v", err)
}
t.Logf("runtime: %d symbols", len(gobj.symbols()))
// Verify we can find well-known runtime symbols.
for _, tc := range []struct{ pkg, name string }{
{"runtime", "g0"},
{"runtime", "morestack"},
{"runtime", "newstack"},
} {
idx := gobj.findSymbol(tc.pkg, tc.name)
if idx < 0 {
t.Errorf("findSymbol(%q, %q) = -1", tc.pkg, tc.name)
} else {
t.Logf("findSymbol(%q, %q) = %d", tc.pkg, tc.name, idx)
}
}
}
// TestResolveExternalSymbols verifies end-to-end resolution of external
// symbol references.
func TestResolveExternalSymbols(t *testing.T) {
if _, err := exec.LookPath("go"); err != nil {
t.Skip("go toolchain not available")
}
refs := map[string][]string{
"runtime": {"g0"},
}
pkgIdx, symIdx, err := resolveExternalGOOBJ(refs)
if err != nil {
t.Fatalf("resolveExternalGOOBJ: %v", err)
}
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 1 {
// Index 0 is the dummy invalid package in the blkPkgIdx table;
// the loader's reader loop starts at 1 (cmd/link/internal/
// loader/loader.go: "PkgIdx 0 is a dummy invalid package"), so
// the first real package must carry index 1.
t.Errorf("pkgIdx = %v, want runtime→1", pkgIdx)
}
if _, ok := symIdx["runtime·g0"]; !ok {
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
}
t.Logf("runtime·g0 → SymIdx=%d", symIdx["runtime·g0"])
}
+267 -44
View File
@@ -111,17 +111,28 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags)
}
// Package defs: the static GLOBL, then one anonymous FuncInfo per
// function.
// Package defs: the static GLOBL, then per function the FuncInfo and the
// two DWARF symbols (debug_line program, subprogram DIE).
defs := v.syms(blkSymdef)
if len(defs) != 3 {
t.Fatalf("symdefs = %d, want 3", len(defs))
if len(defs) != 7 {
t.Fatalf("symdefs = %d, want 7", len(defs))
}
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
// The linkname flag stays clear: the toolchain sets it only for
// //go:linkname symbols, and an ordinary static GLOBL is not one.
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != 0 {
t.Errorf("mask symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
t.Errorf("funcinfo symbol = %+v", defs[1])
t.Errorf("addq funcinfo symbol = %+v", defs[1])
}
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
t.Errorf("addq lines symbol = %+v", defs[2])
}
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
t.Errorf("addq DIE symbol = %+v", defs[3])
}
if defs[4].name != "" || defs[4].typ != kindSDATA || defs[4].size != 28 {
t.Errorf("loadmask funcinfo symbol = %+v", defs[4])
}
// Non-package defs: four pc tables and the function, per function.
@@ -142,45 +153,62 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
// FuncInfo: args 24, FuncFlag Asm, one file, no inline tree.
le := binary.LittleEndian
data := v.blk(blkData)
didx := v.blk(blkDataIdx)
fi := data[16:44]
if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm ||
le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 {
t.Errorf("funcinfo bytes %x", fi)
}
// pcsp: a flat zero over the whole function (zero-frame NOSPLIT).
if got := data[72:75]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
// The pc-value tables of addq (non-package indices 0-3, so global
// indices 7-10): pcsp a flat zero over the whole function, pcinline a
// flat -1, both with the pc delta in MinLC (1) units.
pcsp := data[le.Uint32(didx[4*7:]):]
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
t.Errorf("pcsp = %x, want 021300", got)
}
// pcinline: a flat -1.
if got := data[81:84]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
pcinl := data[le.Uint32(didx[4*10:]):]
if got := pcinl[:3]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) {
t.Errorf("pcinline = %x, want 001300", got)
}
// The one relocation: R_PCREL, four bytes wide, against the GLOBL,
// with the field in the function code left zero. The loadmask code's
// offset comes from the data index (symbol 3 defs + 9 non-package).
// Relocations: the four DWARF address references (two per function, in
// definition order), then the loadmask code's R_PCREL against the
// GLOBL, with the field in the function code left zero. The loadmask
// code's offset comes from the data index (7 defs + 9 non-package).
relocs := v.blk(blkReloc)
if len(relocs) != 23 {
t.Fatalf("relocs = %d bytes, want one 23-byte entry", len(relocs))
if len(relocs) != 5*23 {
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
}
off := int32(le.Uint32(relocs[0:]))
if off != 4 || relocs[4] != 4 || le.Uint16(relocs[5:]) != relocPCRel ||
le.Uint64(relocs[7:]) != 0 || le.Uint32(relocs[15:]) != pkgIdxSelf || le.Uint32(relocs[19:]) != 0 {
t.Errorf("reloc = %x", relocs)
// addq's DWARF references (defs 2 and 3) against the function, which
// is non-package index 4.
lr := relocs[:23]
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
t.Errorf("addq lines reloc = %x", lr)
}
didx := v.blk(blkDataIdx)
lm := le.Uint32(didx[4*(3+9):])
dr := relocs[23:46]
if dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
t.Errorf("addq DIE reloc = %x", dr)
}
cr := relocs[4*23:]
off := int32(le.Uint32(cr[0:]))
if off != 4 || cr[4] != 4 || le.Uint16(cr[5:]) != relocPCRel ||
le.Uint64(cr[7:]) != 0 || le.Uint32(cr[15:]) != pkgIdxSelf || le.Uint32(cr[19:]) != 0 {
t.Errorf("loadmask reloc = %x", cr)
}
lm := le.Uint32(didx[4*16:])
code := data[lm : lm+18]
if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) {
t.Errorf("relocated field = %x, want zeroed", code[4:8])
}
// Aux wiring: FuncInfo (package symbol), then the four pc tables
// (non-package symbols).
// Aux wiring: FuncInfo, the two DWARF symbols (package symbols), then
// the four pc tables (non-package symbols).
auxs := v.blk(blkAux)
if len(auxs) != 2*5*9 {
t.Fatalf("aux = %d bytes, want 10 entries", len(auxs))
if len(auxs) != 2*7*9 {
t.Fatalf("aux = %d bytes, want 14 entries", len(auxs))
}
wantAux := []struct {
typ uint8
@@ -188,15 +216,19 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
idx uint32
}{
{auxFuncInfo, pkgIdxSelf, 1},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 0)},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 1)},
{auxPcline, pkgIdxNone, uint32(len(defs) + 2)},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 3)},
{auxFuncInfo, pkgIdxSelf, 2},
{auxPcsp, pkgIdxNone, uint32(len(defs) + 5)},
{auxPcfile, pkgIdxNone, uint32(len(defs) + 6)},
{auxPcline, pkgIdxNone, uint32(len(defs) + 7)},
{auxPcinline, pkgIdxNone, uint32(len(defs) + 8)},
{auxDwarfInfo, pkgIdxSelf, 3},
{auxDwarfLines, pkgIdxSelf, 2},
{auxPcsp, pkgIdxNone, 0},
{auxPcfile, pkgIdxNone, 1},
{auxPcline, pkgIdxNone, 2},
{auxPcinline, pkgIdxNone, 3},
{auxFuncInfo, pkgIdxSelf, 4},
{auxDwarfInfo, pkgIdxSelf, 6},
{auxDwarfLines, pkgIdxSelf, 5},
{auxPcsp, pkgIdxNone, 5},
{auxPcfile, pkgIdxNone, 6},
{auxPcline, pkgIdxNone, 7},
{auxPcinline, pkgIdxNone, 8},
}
for i, w := range wantAux {
e := auxs[i*9:]
@@ -253,7 +285,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
t.Fatalf("AssembleFile: %v", err)
}
fn := img.Funcs[0]
pcs, vals := decodePCValues(pcspTable(fn))
pcs, vals := decodePCValues(pcspTable(fn, 1))
// Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change),
// SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds
// ADDQ $8, SP (+8) then POPQ BP (0).
@@ -264,7 +296,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
}
for i := range wantPCs {
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d); all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
}
}
// The last two steps unwind the epilogue to zero.
@@ -303,7 +335,7 @@ TEXT ·useext(SB), NOSPLIT, $0-8
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
// assembly object, link, and run — the output must match the baseline
// assembly object, link, and run; the output must match the baseline
// binary the Go assembler produced. Skipped when no Go toolchain is
// available.
func TestGOObjectLinkAndRun(t *testing.T) {
@@ -349,7 +381,7 @@ func main() {
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.26\n"), 0o644); err != nil {
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module goobjtest\n\ngo 1.27\n"), 0o644); err != nil {
t.Fatal(err)
}
@@ -363,7 +395,7 @@ func main() {
}
var work string
var asmObj, pkgArch, linkLine string
for _, line := range strings.Split(string(buildLog), "\n") {
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
@@ -401,13 +433,12 @@ func main() {
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if err := os.WriteFile(asmObj, obj, 0o644); err != nil {
t.Fatal(err)
}
// Rebuild the package archive with our object in place of the
// toolchain's (go tool pack has no replace-in-place that dedupes, so
// extract, substitute and repack).
// extract, substitute and repack). The archive member holding the
// assembler's output is named after the asm object file, e.g.
// main_amd64.o.
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
@@ -417,6 +448,13 @@ func main() {
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(asmObj))
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
@@ -425,7 +463,7 @@ func main() {
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for _, m := range strings.Fields(string(listOut)) {
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
@@ -475,3 +513,188 @@ func fieldAfter(line, flag string) string {
}
return ""
}
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
// GOOBJ whose code references a real external package symbol (runtime's
// morestack, a plain reference rather than the builtin noctxt form) must
// carry a package index that points past the blkPkgIdx table's dummy entry
// 0, and the object must link against the real runtime. Pre-fix, the
// relocations carried block index 0, which the loader never fills, so the
// reference resolved against whatever object was loaded first and the link
// failed. The binary is not run: morestack returns to the call site's
// stack check, which a hand-written caller has none of.
func TestGOObjectExternalPackageLink(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
const asmSrc = `
#include "textflag.h"
TEXT ·fn(SB), NOSPLIT, $0-0
CALL ·helper(SB)
RET
TEXT ·helper(SB), NOSPLIT, $0-0
RET
`
const mainSrc = `package main
func fn()
func helper()
func main() {
fn()
helper()
}
`
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module extlink\n\ngo 1.27\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the build the toolchain performs and re-run only its link
// step with our object swapped into the package archive, mirroring
// TestGOObjectLinkAndRun.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj, pkgArch string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
// Assemble the source with gasm, then retarget fn's internal call at
// a real external package symbol: the reloc's qualified name drives
// the export-data resolution the way a source-level runtime·sym(SB)
// reference would.
f, errs := parser.Parse("main_amd64.s", asmSrc)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
fn := &img.Funcs[0]
for i := range fn.Relocs {
fn.Relocs[i].Name = "runtime\u00b7morestack"
fn.Relocs[i].External = true
}
img.Externals = []string{"runtime\u00b7morestack"}
obj, err := img.GOObject("main", "main_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
// Structural check: the blkPkgIdx block reserves entry 0 for the
// dummy invalid package and places runtime at entry 1, and fn's call
// relocation carries PkgIdx 1.
v := openGoobj(t, obj)
pkgBlk := v.blk(blkPkgIdx)
if len(pkgBlk) != 2*8 {
t.Fatalf("blkPkgIdx = %d bytes, want two entries", len(pkgBlk))
}
le := binary.LittleEndian
strEntry := func(i int) string {
e := pkgBlk[i*8 : (i+1)*8]
return v.str(le.Uint32(e[4:]), le.Uint32(e[0:]))
}
if s := strEntry(0); s != "" {
t.Errorf("blkPkgIdx[0] = %q, want the dummy empty package", s)
}
if s := strEntry(1); s != "runtime" {
t.Errorf("blkPkgIdx[1] = %q, want runtime", s)
}
relocs := v.blk(blkReloc)
// fn is the last non-package symbol (two functions, four pc tables
// each); its one reloc is the final record.
fnRec := relocs[len(relocs)-23:]
if pIdx := le.Uint32(fnRec[15:]); pIdx != 1 {
t.Errorf("external reloc PkgIdx = %d, want 1 (runtime)", pIdx)
}
// Swap the object into the package archive and link with cmd/link;
// the link line consumes the archive, not the loose object file.
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(asmObj))
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, pkgArch, newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "prog2"))
linkCmd := exec.Command("sh", "-c", "cd "+dir+" && "+linkLine)
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// The call must have resolved to the real runtime symbol.
dump, err := exec.Command(goBin, "tool", "objdump", "-s", "main.fn", filepath.Join(dir, "prog2")).CombinedOutput()
if err != nil {
t.Fatalf("objdump main.fn: %v\n%s", err, dump)
}
if !bytes.Contains(dump, []byte("runtime.morestack")) {
t.Errorf("main.fn does not call runtime.morestack:\n%s", dump)
}
}
+113
View File
@@ -0,0 +1,113 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
// the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and the arm64 relocation types for the ADRP
// pairs and BL calls.
//
// The toolchain records one relocation per ADRP pair: a single R_ADDRARM64
// or R_ARM64_PCREL_LDST64 of Siz 8 at the ADRP word, from which the linker
// patches both instructions of the pair (cmd/internal/obj/arm64/asm7.go,
// the ADRP cases: one AddRel with Off at the pair's pc and Siz 8). gasm's
// assembler records the ADRP+ADD form as two word relocs, so the second
// word's twin is dropped here before emission.
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleAARCH64()
if err != nil {
return nil, err
}
coalesced := *img
coalesced.Funcs = append([]FuncLayout(nil), img.Funcs...)
for i := range coalesced.Funcs {
rs := coalesced.Funcs[i].Relocs
var keep []Reloc
for j := 0; j < len(rs); j++ {
keep = append(keep, rs[j])
if rs[j].Kind == RelArm64Addr && j+1 < len(rs) &&
rs[j+1].Kind == RelArm64Addr && rs[j+1].Off == rs[j].Off+4 {
j++ // the ADD word's twin: the Siz-8 pair reloc covers it
}
}
coalesced.Funcs[i].Relocs = keep
}
return coalesced.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelArm64Branch:
return relocArm64Branch, 4
case RelArm64LDST64:
return relocArm64LDST64, 8
default:
return relocArm64Addr, 8
}
})
}
// arm64 relocation types (cmd/internal/objabi).
const (
relocArm64Addr = 3 // R_ADDRARM64, ADRP+ADD pair
relocArm64Branch = 9 // R_CALLARM64, BL instruction
relocArm64LDST64 = 40 // R_ARM64_PCREL_LDST64, ADRP+LDR/STR pair
)
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
// the installed go tool asm writes for arm64, captured by assembling a
// one-instruction probe.
var (
preambleAARCH64Once sync.Once
preambleAARCH64 []byte
preambleAARCH64Err error
)
func toolchainObjectPreambleAARCH64() ([]byte, error) {
preambleAARCH64Once.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleAARCH64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble-arm64")
if err != nil {
preambleAARCH64Err = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_arm64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleAARCH64Err = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleAARCH64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleAARCH64Err = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleAARCH64Err = fmt.Errorf("unrecognised assembler object layout")
return
}
preambleAARCH64 = data[:i+3]
})
return preambleAARCH64, preambleAARCH64Err
}
+97
View File
@@ -0,0 +1,97 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
// the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays, with the loong64 preamble, the MinLC of 4
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
// the pcalau12i+addi.d address pairs.
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleLOONG64()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
// A pcalau12i+addi.d pair: the high part carries
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO; the guard's
// morestack call carries R_CALLLOONG64.
switch {
case r.Kind == RelLoong64AddrLo:
return relocLoong64AddrLo, 4
case r.Kind == RelLoong64Branch:
return relocCallLoong64, 4
default:
return relocLoong64AddrHi, 4
}
})
}
// Loong64 relocation types (cmd/internal/objabi). R_LOONG64_ADDR_HI
// resolves the high 20 bits of a PC-relative address into pcalau12i;
// R_LOONG64_ADDR_LO the low 12 bits into addi.d/ld/st.
const (
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
relocCallLoong64 = 84 // R_CALLLOONG64
)
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
// the installed go tool asm writes for loong64, captured by assembling a
// one-instruction probe (see toolchainObjectPreamble).
var (
preambleLOONG64Once sync.Once
preambleLOONG64 []byte
preambleLOONG64Err error
)
func toolchainObjectPreambleLOONG64() ([]byte, error) {
preambleLOONG64Once.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleLOONG64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble-loong64")
if err != nil {
preambleLOONG64Err = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_loong64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleLOONG64Err = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleLOONG64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleLOONG64Err = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleLOONG64Err = fmt.Errorf("unrecognised assembler object layout")
return
}
preambleLOONG64 = data[:i+3]
})
return preambleLOONG64, preambleLOONG64Err
}
+95
View File
@@ -0,0 +1,95 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"sync"
)
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
// shared one in goobj.go, the toolchain preamble, the go120ld header with
// its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays, with the RISC-V preamble, the MinLC of 2 for
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
// relocation, not the ELF HI20/LO12 pair).
func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleRISCV()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 2, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelRISCVPCRELSType:
return relocRISCVPcrelStype, 8
case RelRISCVJal:
return relocRISCVJal, 4
default:
return relocRISCVPcrelItype, 8
}
})
}
// RISC-V relocation types (cmd/internal/objabi). The Go linker applies
// R_RISCV_PCREL_ITYPE/STYPE to an AUIPC + I/S-type instruction pair as a
// single 8-byte field; R_RISCV_JAL covers a single 4-byte J-type instruction.
const (
relocRISCVJal = 59 // R_RISCV_JAL
relocRISCVPcrelItype = 62 // R_RISCV_PCREL_ITYPE
relocRISCVPcrelStype = 63 // R_RISCV_PCREL_STYPE
)
// toolchainObjectPreambleRISCV returns the "go object ...\n!\n" header
// the installed go tool asm writes for riscv64, captured by assembling a
// one-instruction probe (see toolchainObjectPreamble).
var (
preambleRISCVOnce sync.Once
preambleRISCV []byte
preambleRISCVErr error
)
func toolchainObjectPreambleRISCV() ([]byte, error) {
preambleRISCVOnce.Do(func() {
goBin, err := exec.LookPath("go")
if err != nil {
preambleRISCVErr = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err)
return
}
dir, err := os.MkdirTemp("", "gasm-preamble-riscv")
if err != nil {
preambleRISCVErr = err
return
}
defer os.RemoveAll(dir)
src := filepath.Join(dir, "probe_riscv64.s")
if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil {
preambleRISCVErr = err
return
}
obj := filepath.Join(dir, "probe.o")
cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src)
cmd.Env = append(os.Environ(), "GOARCH=riscv64")
if out, err := cmd.CombinedOutput(); err != nil {
preambleRISCVErr = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out)
return
}
data, err := os.ReadFile(obj)
if err != nil {
preambleRISCVErr = err
return
}
i := bytes.Index(data, []byte("\n!\n"))
if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) {
preambleRISCVErr = fmt.Errorf("unrecognised assembler object layout")
return
}
preambleRISCV = data[:i+3]
})
return preambleRISCV, preambleRISCVErr
}
+329
View File
@@ -0,0 +1,329 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
// verified with go tool objdump): the stack-split guard classes, the morestack
// block and the auto-NOSPLIT leaf behaviour.
func TestStackGuardBytes(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"554889e54883ec104883c4105dc3"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
// Class 2 with a body long enough that the underflow JB relaxes to
// rel32: its displacement must span the real 6-byte JB, else the
// branch lands 4 bytes past the morestack block, inside the CALL
// displacement field.
{"leafbiglong", "TEXT \u00b7leafbiglong(SB), $8192-0\n" + strings.Repeat("\tMOVQ AX, BX\n", 40) + "\tRET\n",
"644c8b3425000000004989e44981ec881f00000f82960000004d3b66100f868c000000554889e54881ec00200000" + strings.Repeat("4889c3", 40) + "4881c4002000005dc3e800000000e947ffffff"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"554889e54883ec104883c4105dc3"},
} {
f, errs := parser.Parse("g_amd64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
// The toolchain's object leaves every relocation field zero for the
// linker, while the gasm image resolves file-internal references, so
// the comparison masks the patch sites the way verify's ground truth
// does.
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardRelocs checks the guard's patch sites: the TLS slot and the
// morestack call.
func TestStackGuardRelocs(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
relocs := img.Funcs[0].Relocs
if len(relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(relocs))
}
tls, call := relocs[0], relocs[1]
if tls.Kind != RelTLSLE || tls.Off != 5 || tls.Name != "" || tls.External {
t.Errorf("tls reloc = %+v, want RelTLSLE at 5 with no symbol", tls)
}
if call.Kind != RelCall || call.Name != "runtime\u00b7morestack_noctxt" || !call.External {
t.Errorf("call reloc = %+v, want RelCall to runtime.morestack_noctxt", call)
}
}
// TestStackGuardGOObj emissions succeed with the guard's TLS and builtin
// references in play.
func TestStackGuardGOObj(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tCALL \u00b7helper(SB)\n\tRET\nTEXT \u00b7helper(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
obj, err := img.GOObject("testpkg", "g_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if !bytes.Contains(obj, []byte("go120ld")) {
t.Fatal("object lacks the GOOBJ magic")
}
}
// The arm64 stack-split guard, pinned from `go tool asm` (Go 1.27, arm64):
// the guard classes, the auto-NOSPLIT leaf behaviour and the morestack
// block. Relocation fields are masked: the toolchain's object leaves them
// zero for the linker, the gasm image resolves file-internal references.
func TestStackGuardBytesARM64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"900b40f9f14302d13f0210eb09010054f44304d19dfa3fa99f020091fd2300d1fd230491ff430491c0035fd6e3031eaa00000000f3ffff17"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"900b40f91bf283d2f1633beba30100543f0210eb690100541b0284d2f4633bcb9dfa3fa99f020091fd2300d11b0184d2fd633b8b1b0284d2ff633b8bc0035fd6e3031eaa00000000eeffff17"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"900b40f9ff6330eb09010054fe0f1ef8fd831ff8fd2300d100000000fd835ff8fe0742f8c0035fd6e3031eaa00000000f4ffff17"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
} {
f, errs := parser.Parse("g_arm64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardBranchTargetsARM64 checks the class-2 guard's branch
// positions for a frame whose guard constant needs two MOV words: the
// displacements must be computed from byte offsets (8+4*ml and 16+4*ml), so
// both branches land on the morestack block rather than inside the body.
// The frame size makes the toolchain switch its own prologue decomposition,
// so the assertion is on the branch targets, not pinned bytes.
func TestStackGuardBranchTargetsARM64(t *testing.T) {
f, errs := parser.Parse("g_arm64.s", "TEXT \u00b7f(SB), $65664-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if len(code)%4 != 0 {
t.Fatalf("function size %d is not a word multiple", len(code))
}
// autosize = 65680, so the guard materialises 65552 = MOVZ+MOVK: ml = 2
// and the branches sit at bytes 16 and 24 of the guard prefix.
const morestackBlock = 12 // MOVD R30, R3; BL; B back
blockStart := len(code) - morestackBlock
check := func(name string, off int) {
t.Helper()
w := leWord(code[off:])
imm19 := int32(w>>5) & 0x7FFFF
if imm19&(1<<18) != 0 {
imm19 -= 1 << 19
}
if target := off + int(imm19)*4; target != blockStart {
t.Errorf("%s at byte %d targets byte %d, want the morestack block at %d", name, off, target, blockStart)
}
}
check("B.LO", 16)
check("B.LS", 24)
}
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// riscv64): the morestack call sits between the guard and the body, and the
// guard branches forward over it. Relocation fields are masked.
func TestStackGuardBytesRISCV64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"03b30d0163662300000000006ff05fff233411fe211106e08260610167800000"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"03b30d01930381f763667300000000006ff01fff233c11ee130181ef06e082601301811067800000"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"03b30d0189639b8383f863697100f97f9b8f8f07b303f10163667300000000006ff01ffef97f8a9f23bc1ffef97fe13f7e9106e08260896fa12f7e9167800000"},
{"frameless", "TEXT \u00b7frameless(SB), $0-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"03b30d0163662300000000006ff05fff233c11fe611106e0000000008260210167800000"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"233411fe211106e08260610167800000"},
} {
f, errs := parser.Parse("g_riscv64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// The loong64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// loong64): every guard class (including the medium class with the
// materialised constant and the big class with the ORI-less constants), the
// auto-NOSPLIT leaf behaviour, the large-frame R30 prologue/epilogue forms
// and the morestack block. Relocation fields are masked.
func TestStackGuardBytesLOONG64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"61a0ff2963a0ff026100c0296360c0022000004c"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"d442c02878e0fd0294e21200801a004061e0fb2963e0fb026100c0296320c4022000004c3f00150000000000ffd7ff53"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"61a0ff2963a0ff026100c0296360c0022000004c"},
// The LR store leaves the 12-bit store-offset range while the SP
// adjust immediate still fits, and the epilogue adjusts through a
// single ORI.
{"fit2048", "TEXT \u00b7fit2048(SB), $2040-0\n\tRET\n",
"d442c0287800e20294e21200802600401e000014de8f1000c103e0296300e0026100c0291e00a00363f810002000004c3f00150000000000ffcbff53"},
// Medium class at the materialisation boundary (off = 2048 still
// immediate, 2049+ goes through R30).
{"med2048off", "TEXT \u00b7med2048off(SB), $2168-0\n\tRET\n",
"d442c0287800e00294e21200802e0040feffff15de8f1000c103de29feffff15de039e0363f810006100c0291e00a20363f810002000004c3f00150000000000ffc3ff53"},
{"medmat", "TEXT \u00b7medmat(SB), $2176-0\n\tRET\n",
"d442c028feffff15dee39f0378f8100094e21200802e0040feffff15de8f1000c1e3dd29feffff15dee39d0363f810006100c0291e20a20363f810002000004c3f00150000000000ffbbff53"},
// Big class with the rounding-split store and the floor-split adjust.
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"d442c0283e000014de23be0378f8120000470044deffff15dee3810378f8100094e2120080320040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c0295e000014de23800363f810002000004c3f00150000000000ffa7ff53"},
// Zero low 12 bits drop the ORI from the store, the adjust and the
// epilogue materialisation.
{"bigzero", "TEXT \u00b7bigzero(SB), $4088-0\n\tRET\n",
"d442c028feffff15de03820378f8100094e21200802a0040feffff15de8f1000c103c029feffff1563f810006100c0293e00001463f810002000004c3f00150000000000ffbfff53"},
// Big class whose first constant has a zero high part: a single ORI.
{"big3976", "TEXT \u00b7big3976(SB), $4096-0\n\tRET\n",
"d442c0281e20be0378f8120000470044feffff15dee3810378f8100094e2120080320040feffff15de8f1000c1e3ff29deffff15dee3bf0363f810006100c0293e000014de23800363f810002000004c3f00150000000000ffabff53"},
// Big class at a multiple of 4096: both guard constants lose their
// ORI word.
{"giantlo0", "TEXT \u00b7giantlo0(SB), $4216-0\n\tRET\n",
"d442c0283e00001478f8120000430044feffff1578f8100094e2120080320040feffff15de8f1000c103fe29deffff15de03be0363f810006100c0293e000014de03820363f810002000004c3f00150000000000ffafff53"},
// Non-leaf big frame: the body call plus the LR restore epilogue.
{"callbig", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"d442c0283e000014de23be0378f81200004f0044deffff15dee3810378f8100094e21200803a0040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c029000000006100c0285e000014de23800363f810002000004c3f00150000000000ff9fff53"},
} {
f, errs := parser.Parse("g_loong64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardGOObjInternalCall checks that GOOBJ emission succeeds when a
// guarded function calls a TEXT symbol of the same file, for every arch's
// call relocation kind.
func TestStackGuardGOObjInternalCall(t *testing.T) {
for _, tt := range []struct {
src string
assemble func(*ast.File) (*Image, error)
}{
{"g_amd64.s", AssembleFile},
{"g_arm64.s", AssembleFileARM64},
{"g_riscv64.s", AssembleFileRISCV},
{"g_loong64.s", AssembleFileLOONG64},
} {
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.src, errs)
}
img, err := tt.assemble(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.src, err)
}
if _, err := img.GOObject("testpkg", tt.src); err != nil {
t.Errorf("%s: GOObject: %v", tt.src, err)
}
}
}
+395 -52
View File
@@ -21,7 +21,7 @@ var aluOp = map[string]struct {
}
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
// the 0xFE/0xFF group (the short 0x40–0x4F forms are REX prefixes in 64-bit
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
// mode); NEG/NOT use the 0xF6/0xF7 group.
var unaryOp = map[string]struct {
digit int
@@ -33,7 +33,7 @@ var unaryOp = map[string]struct {
"NEG": {3, 0xF7},
}
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0–0xD3 group.
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
var shiftOp = map[string]int{
"SHL": 4,
"SHR": 5,
@@ -48,11 +48,62 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
}
src, dst := ops[0], ops[1]
// Integer scalar XMM moves: MOVQ with an XMM operand is the SSE2
// packed-quadword move, NOT a GPR move: mem→xmm encodes as F3 0F 7E
// (reg = dst, no REX.W, the Go assembler's form), xmm→mem as
// 66 0F D6 (rm = xmm). Register forms against a GPR use the MOVD
// opcodes with REX.W instead: 66 REX.W 0F 6E (gpr→xmm) and
// 66 REX.W 0F 7E (xmm→gpr); the memory opcodes with a register r/m
// would be undefined forms. MOVL is the packed-dword move:
// 66 0F 6E load, 66 0F 7E store, no REX.W. A GPR-move fallback would
// silently emit REX.W 8B with the wrong operand meaning.
_, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst)
if srcVec || dstVec {
if dstVec {
if g, ok := src.(Reg); ok && !g.isVec() {
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: size == 8}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
return e.emit(i)
}
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1}
if size == 4 {
i.prefix = 0x66
i.opcode = []byte{0x0F, 0x6E}
}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
return e.emit(i)
}
srcXMM, srcIsXMM := src.(Reg)
if !srcIsXMM || !srcXMM.isVec() {
return fmt.Errorf("MOV: store needs an XMM source")
}
if g, ok := dst.(Reg); ok && !g.isVec() {
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: size == 8}
if err := setRM(i, srcXMM, dst, 8); err != nil {
return err
}
return e.emit(i)
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if size == 4 {
i.opcode = []byte{0x0F, 0x7E}
}
if err := setRM(i, srcXMM, dst, 8); err != nil {
return err
}
return e.emit(i)
}
dstReg, dstIsReg := dst.(Reg)
switch src := src.(type) {
case Reg:
if dstIsReg {
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst — the form the Go
// MOV r/m, r: 0x88/0x89, reg=src, rm=dst, the form the Go
// assembler emits for register-to-register moves.
i := newInstr(size, []byte{movRM(size)})
if err := setRM(i, src, dst, size); err != nil {
@@ -91,7 +142,28 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
case Imm:
if dstIsReg {
// MOV r, imm: 0xB0+reg (8-bit) / 0xB8+reg (16/32/64, imm64 for Q).
v := int64(src)
// The Go assembler compresses 64-bit moves whose immediate fits
// a signed int32, choosing per sign:
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the
// hardware, REX.B still emitted for R8-R15);
// v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd
// form would zero-extend and corrupt the value).
// Out-of-range immediates keep the B8+rd imm64 form.
if size == 8 && v >= 0 && v <= (1<<31)-1 {
i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)})
i.rexB = dstReg.idx >= 8
i.imm = le32(v)
return e.emit(i)
}
if size == 8 && v < 0 && v >= -(1<<31) {
i := newInstr(8, []byte{0xC7})
if err := setRMDigit(i, 0, dstReg, 8); err != nil {
return err
}
i.imm = le32(v)
return e.emit(i)
}
opBase := byte(0xB8)
if size == 1 {
opBase = 0xB0
@@ -101,7 +173,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
if dstReg.needsREX(size) {
i.rexForced = true
}
i.imm = immediate(int64(src), size, true)
imm, err := immediate(v, size, true)
if err != nil {
return err
}
i.imm = imm
return e.emit(i)
}
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
@@ -113,7 +189,11 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
if err := setRMDigit(i, 0, dst, size); err != nil {
return err
}
i.imm = immediate(int64(src), size, false)
imm, err := immediate(int64(src), size, false)
if err != nil {
return err
}
i.imm = imm
return e.emit(i)
}
return fmt.Errorf("MOV: invalid operands")
@@ -144,12 +224,18 @@ func (e *enc) encodeALU(op struct {
}
src, dst := ops[0], ops[1]
// CMP never takes its immediate first: the Go assembler rejects
// CMPL $0, AX outright (only CMPL AX, $0 is legal, unlike TEST and the
// writing ALU ops whose immediate is naturally the source).
if imm, ok := src.(Imm); ok {
if op.digit == 7 {
return fmt.Errorf("CMP immediate must be the second operand (reg, $imm)")
}
return e.encodeALUImm(op.digit, dst, int64(imm), size)
}
// CMP accepts the immediate in the second position too — CMPL CX, $31 is
// the form the Go assembler itself accepts — and encodes it identically
// CMP accepts the immediate in the second position too, CMPL CX, $31 is
// the form the Go assembler itself accepts, and encodes it identically
// (CMP r/m, imm sets the flags as first − second). No other ALU op takes
// an immediate destination.
if imm, ok := dst.(Imm); ok {
@@ -219,11 +305,15 @@ func (e *enc) encodeALU(op struct {
func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if size == 1 {
immBytes, err := immediate(imm, 1, false)
if err != nil {
return err
}
i := newInstr(1, []byte{0x80})
if err := setRMDigit(i, digit, dst, 1); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
i.imm = immBytes
return e.emit(i)
}
if fits8(imm) {
@@ -235,12 +325,29 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
// 0x81 /digit, imm16/imm32, or the Go assembler's accumulator short
// form (opcode+5, no ModR/M) when the destination is AX/AL, which it
// prefers over the generic form exactly here.
if r, ok := dst.(Reg); ok && r.idx == 0 {
accOp := map[int]byte{0: 0x05, 1: 0x0D, 2: 0x15, 3: 0x1D, 4: 0x25, 5: 0x2D, 6: 0x35, 7: 0x3D}[digit]
i := newInstr(size, []byte{accOp})
immBytes, err := immediate(imm, size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
// 0x81 /digit, imm16/imm32.
i := newInstr(size, []byte{0x81})
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
i.imm = immediate(imm, size, false)
immBytes, err := immediate(imm, size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
@@ -252,7 +359,22 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
}
src, dst := ops[0], ops[1]
if imm, ok := src.(Imm); ok {
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0.
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0, but the Go assembler
// always uses the accumulator forms (A8/A9, no ModR/M) when the
// register operand is AL/AX, whatever the immediate's width.
if r, ok := dst.(Reg); ok && r.idx == 0 {
op := byte(0xA9)
if size == 1 {
op = 0xA8
}
i := newInstr(size, []byte{op})
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
op := byte(0xF7)
if size == 1 {
op = 0xF6
@@ -261,7 +383,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
if err := setRMDigit(i, 0, dst, size); err != nil {
return err
}
i.imm = immediate(int64(imm), size, false)
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
srcReg, ok := src.(Reg)
@@ -359,7 +485,13 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
}
return e.emit(i)
}
// 0xC0 (8-bit) / 0xC1, imm8.
// 0xC0 (8-bit) / 0xC1, imm8. The count is an unsigned byte: go tool asm
// rejects negative and ≥256 counts, and the hardware masks the count, so
// a silent truncation ($300 encoding 44) would shift by a different
// amount than the source states.
if imm < 0 || imm > 255 {
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
}
op := byte(0xC1)
if size == 1 {
op = 0xC0
@@ -368,7 +500,7 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
if err := setRMDigit(i, digit, dst, size); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
i.imm = []byte{byte(imm)}
return e.emit(i)
}
@@ -410,7 +542,11 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
if err := setRM(i, dstReg, ops[1], size); err != nil {
return err
}
i.imm = immediate(int64(imm), size, false)
immBytes, err := immediate(int64(imm), size, false)
if err != nil {
return err
}
i.imm = immBytes
return e.emit(i)
}
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
@@ -418,10 +554,21 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
// --- PUSH / POP -------------------------------------------------------------
func (e *enc) encodePushPop(ops []Operand, push bool) error {
func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
if len(ops) != 1 {
return fmt.Errorf("PUSH/POP expects 1 operand, got %d", len(ops))
}
// In 64-bit mode go tool asm knows the 64-bit push (the default, with or
// without the Q suffix) and the 16-bit W form with its 0x66 operand-size
// prefix, and rejects the B and L spellings outright ("illegal in 64-bit
// mode"); silently widening those would push a different width than the
// source states.
switch size {
case 0, 8, 2:
default:
return fmt.Errorf("PUSH/POP size suffix is illegal in 64-bit mode")
}
w16 := size == 2
switch op := ops[0].(type) {
case Reg:
base := byte(0x50) // PUSH r; POP is 0x58
@@ -429,7 +576,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
base = 0x58
}
// PUSH/POP default to 64-bit in 64-bit mode; no REX.W needed.
i := &instr{opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
i := &instr{opSize16: w16, opcode: []byte{base + byte(op.idx&7)}, modrm: -1, sib: -1}
i.rexB = op.idx >= 8
return e.emit(i)
case Mem:
@@ -439,7 +586,7 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
opc = 0x8F // POP r/m: /0
digit = 0
}
i := &instr{opcode: []byte{opc}, modrm: -1, sib: -1}
i := &instr{opSize16: w16, opcode: []byte{opc}, modrm: -1, sib: -1}
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
return err
}
@@ -449,10 +596,17 @@ func (e *enc) encodePushPop(ops []Operand, push bool) error {
return fmt.Errorf("POP does not take an immediate")
}
if fits8(int64(op)) {
i := &instr{opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
i := &instr{opSize16: w16, opcode: []byte{0x6A}, modrm: -1, sib: -1, imm: []byte{byte(int8(op))}}
return e.emit(i)
}
i := &instr{opSize16: false, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: le32(int64(op))}
// PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
// immediate by the same signed/unsigned 32-bit span as every other
// scalar immediate.
immBytes, err := immediate(int64(op), 8, false)
if err != nil {
return err
}
i := &instr{opSize16: w16, opcode: []byte{0x68}, modrm: -1, sib: -1, imm: immBytes}
return e.emit(i)
}
return fmt.Errorf("PUSH/POP: invalid operand")
@@ -477,6 +631,24 @@ func (e *enc) encodeJmpRel(ops []Operand, opcode []byte) error {
return e.emit(&instr{opcode: opcode, modrm: -1, sib: -1, imm: le32(int64(imm))})
}
// encodeIndirectBranch encodes JMP/CALL through a register or memory operand:
// FF /4 for JMP, FF /2 for CALL. The operand size is fixed at 64 bits in
// 64-bit mode, so no REX.W is emitted; a REX appears only for R8-R15 bases.
func (e *enc) encodeIndirectBranch(mnem string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
digit := 4 // JMP r/m64
if mnem == "CALL" {
digit = 2 // CALL r/m64
}
i := &instr{opcode: []byte{0xFF}, modrm: -1, sib: -1}
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
return err
}
return e.emit(i)
}
// condCode maps a Plan 9 conditional-jump mnemonic to its x86 condition code.
func condCode(upper string) (int, bool) {
if len(upper) < 2 || upper[0] != 'J' || upper == "JMP" {
@@ -523,19 +695,28 @@ func (e *enc) encodeJcc(cc int, ops []Operand) error {
// immediate encodes an immediate of the given operand size. full64 selects the
// 64-bit immediate form (only valid for MOV r64, imm64); otherwise a 32-bit
// sign-extended immediate is used for 64-bit operands.
func immediate(v int64, size int, full64 bool) []byte {
//
// The span mirrors go tool asm: every scalar immediate must fit a signed or
// unsigned 32-bit word, and the narrower fields then take the low bits
// silently (ADDB $256, AL encodes imm8 0, MOVW $65536, AX imm16 0). Only the
// imm64 form may exceed the span; anything wider elsewhere is an error rather
// than a truncation the source never asked for.
func immediate(v int64, size int, full64 bool) ([]byte, error) {
if !(size == 8 && full64) && (v < -(1<<31) || v > (1<<32)-1) {
return nil, fmt.Errorf("immediate $%d does not fit in 32 bits", v)
}
switch size {
case 1:
return []byte{byte(int8(v))}
return []byte{byte(int8(v))}, nil
case 2:
return le16(v)
return le16(v), nil
case 4:
return le32(v)
return le32(v), nil
default: // 8
if full64 {
return le64(v)
return le64(v), nil
}
return le32(v) // sign-extended imm32
return le32(v), nil // sign-extended imm32
}
}
@@ -580,7 +761,7 @@ func (e *enc) encodeCmov(upper string, ops []Operand) error {
}
// encodeSet encodes a conditional byte set: SET + condition (SETNE, SETEQ, …),
// always a byte write — 0F 90+cc /0 into a register or memory operand.
// always a byte write, 0F 90+cc /0 into a register or memory operand.
func (e *enc) encodeSet(upper string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("SETcc expects 1 operand, got %d", len(ops))
@@ -597,46 +778,84 @@ func (e *enc) encodeSet(upper string, ops []Operand) error {
return e.emit(i)
}
// --- LZCNT / TZCNT ----------------------------------------------------------
// --- bit scan / bit count ----------------------------------------------------
// encodeCount encodes LZCNT/TZCNT (leading / trailing zero count): F3 0F BD
// or F3 0F BC, with reg = dst and rm = src. The size suffix selects the
// operand width (LZCNTW/LZCNTL/LZCNTQ).
// countOp maps the bit-scan and bit-count mnemonics to their opcode byte and
// mandatory prefix. TZCNT/LZCNT/POPCNT are the F3-prefixed forms of the
// same map as BSF/BSR's 0F BC/BD; POPCNT is F3 0F B8.
var countOp = map[string]struct {
op byte
prefix byte
}{
"BSF": {0xBC, 0},
"BSR": {0xBD, 0},
"TZCNT": {0xBC, 0xF3},
"LZCNT": {0xBD, 0xF3},
"POPCNT": {0xB8, 0xF3},
}
// encodeCount encodes the bit-scan and bit-count family, BSF (0F BC),
// BSR (0F BD), TZCNT (F3 0F BC), LZCNT (F3 0F BD) and POPCNT (F3 0F B8)
// with reg = dst and rm = src. The size suffix selects the operand width
// (BSFQ, TZCNTL, …). Note BSF/BSR leave the destination undefined when the
// source is zero (unlike their F3-prefixed counterparts); callers must
// guard non-zero inputs themselves.
func (e *enc) encodeCount(base string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
}
op := byte(0xBD)
if base == "TZCNT" {
op = 0xBC
}
spec := countOp[base]
dstReg, ok := ops[1].(Reg)
if !ok {
return fmt.Errorf("%s destination must be a register", base)
}
i := newInstr(size, []byte{0x0F, op})
i.prefix = 0xF3
i := newInstr(size, []byte{0x0F, spec.op})
i.prefix = spec.prefix
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// encodeBswap encodes BSWAP: the single register operand is encoded in the
// opcode byte (0F C8+r), with REX.B for R8-R15 and REX.W for the quad form.
func (e *enc) encodeBswap(ops []Operand, size int) error {
if len(ops) != 1 {
return fmt.Errorf("BSWAP expects 1 operand, got %d", len(ops))
}
reg, ok := ops[0].(Reg)
if !ok {
return fmt.Errorf("BSWAP operand must be a register")
}
i := newInstr(size, []byte{0x0F, 0xC8 + byte(reg.idx&7)})
i.rexB = reg.idx >= 8
return e.emit(i)
}
// --- mixed-width sign/zero-extending moves -----------------------------------
// movExtendOp maps Go's mixed-width move names to their opcode and destination
// width. The source is narrower than the destination, so the plain size-suffix
// convention does not apply to these names.
var movExtendOp = map[string]struct {
op []byte
dst64 bool
op []byte
dstSize int
}{
"MOVBLZX": {[]byte{0x0F, 0xB6}, false}, // byte → long, zero-extend
"MOVBQZX": {[]byte{0x0F, 0xB6}, true}, // byte → quad, zero-extend
"MOVWLZX": {[]byte{0x0F, 0xB7}, false}, // word → long, zero-extend
"MOVWQZX": {[]byte{0x0F, 0xB7}, true}, // word → quad, zero-extend
"MOVWLSX": {[]byte{0x0F, 0xBF}, false}, // word → long, sign-extend
"MOVLQSX": {[]byte{0x63}, true}, // long → quad, sign-extend (MOVSXD)
"MOVBLZX": {[]byte{0x0F, 0xB6}, 4}, // byte → long, zero-extend
"MOVBQZX": {[]byte{0x0F, 0xB6}, 8}, // byte → quad, zero-extend
"MOVWLZX": {[]byte{0x0F, 0xB7}, 4}, // word → long, zero-extend
"MOVWQZX": {[]byte{0x0F, 0xB7}, 8}, // word → quad, zero-extend
"MOVWLSX": {[]byte{0x0F, 0xBF}, 4}, // word → long, sign-extend
"MOVLQSX": {[]byte{0x63}, 8}, // long → quad, sign-extend (MOVSXD)
"MOVBWZX": {[]byte{0x0F, 0xB6}, 2}, // byte → word, zero-extend
"MOVBWSX": {[]byte{0x0F, 0xBE}, 2}, // byte → word, sign-extend
"MOVBLSX": {[]byte{0x0F, 0xBE}, 4}, // byte → long, sign-extend
"MOVBQSX": {[]byte{0x0F, 0xBE}, 8}, // byte → quad, sign-extend
"MOVWQSX": {[]byte{0x0F, 0xBF}, 8}, // word → quad, sign-extend
// A long → quad zero-extend is a plain 32-bit move: every 32-bit
// operation zero-extends its result into the full register, so the
// toolchain lowers MOVLQZX to the plain MOVL encoding.
"MOVLQZX": {[]byte{0x8B}, 4},
}
// encodeMovExtend encodes a mixed-width extending move: reg = dst (the wider
@@ -650,12 +869,30 @@ func (e *enc) encodeMovExtend(base string, ops []Operand) error {
if !ok {
return fmt.Errorf("%s destination must be a register", base)
}
size := 4
if spec.dst64 {
size = 8
i := newInstr(spec.dstSize, spec.op)
if err := setRM(i, dstReg, ops[0], spec.dstSize); err != nil {
return err
}
i := newInstr(size, spec.op)
if err := setRM(i, dstReg, ops[0], size); err != nil {
return e.emit(i)
}
// encodePmovmskb encodes PMOVMSKB, the legacy SSE2 byte mask extract: the
// XMM source's sign bytes pack into a GP destination, 66 0F D7 /r.
func (e *enc) encodePmovmskb(base string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
}
srcReg, srcVec := vecReg(ops[0])
if !srcVec {
return fmt.Errorf("%s source must be an XMM register", base)
}
dstReg, ok := ops[1].(Reg)
if !ok {
return fmt.Errorf("%s destination must be a register", base)
}
i := newInstr(4, []byte{0x0F, 0xD7})
i.prefix = 0x66
if err := setRM(i, dstReg, srcReg, 4); err != nil {
return err
}
return e.emit(i)
@@ -674,8 +911,8 @@ type sseMove struct {
}
var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU — unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA — aligned octa
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
@@ -721,6 +958,112 @@ func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
return e.emit(i)
}
// --- legacy SSE packed binary and shuffles -----------------------------------
// sseBin describes a legacy (non-VEX) SSE packed/scalar binary op: an
// optional mandatory prefix plus the 0F-prefixed opcode (0F38 for the
// SSSE3 integer shuffles). Plan 9 asm lists the source operand first, so
// MULPS X0, X1 computes X1 = X1 * X0.
type sseBin struct {
prefix byte // 0, 0x66, 0xF2 or 0xF3
op byte
map38 bool // opcode lives under 0F38 instead of 0F
}
var sseBinTable = map[string]sseBin{
"ADDPS": {0, 0x58, false}, "ADDPD": {0x66, 0x58, false},
"MULPS": {0, 0x59, false}, "MULPD": {0x66, 0x59, false},
"SUBPS": {0, 0x5C, false}, "SUBPD": {0x66, 0x5C, false},
"DIVPS": {0, 0x5E, false}, "DIVPD": {0x66, 0x5E, false},
"ANDPS": {0, 0x54, false}, "ANDPD": {0x66, 0x54, false},
"ORPS": {0, 0x56, false}, "ORPD": {0x66, 0x56, false},
"XORPS": {0, 0x57, false}, "XORPD": {0x66, 0x57, false},
"MINPS": {0, 0x5D, false}, "MINPD": {0x66, 0x5D, false},
"MAXPS": {0, 0x5F, false}, "MAXPD": {0x66, 0x5F, false},
"ADDSS": {0xF3, 0x58, false}, "ADDSD": {0xF2, 0x58, false},
"MULSS": {0xF3, 0x59, false}, "MULSD": {0xF2, 0x59, false},
"SUBSS": {0xF3, 0x5C, false}, "SUBSD": {0xF2, 0x5C, false},
"DIVSS": {0xF3, 0x5E, false}, "DIVSD": {0xF2, 0x5E, false},
"MINSS": {0xF3, 0x5D, false}, "MINSD": {0xF2, 0x5D, false},
"MAXSS": {0xF3, 0x5F, false}, "MAXSD": {0xF2, 0x5F, false},
"UNPCKLPS": {0, 0x14, false}, "UNPCKHPS": {0, 0x15, false},
"UNPCKLPD": {0x66, 0x14, false}, "UNPCKHPD": {0x66, 0x15, false},
"CVTSS2SD": {0xF3, 0x5A, false}, "CVTSD2SS": {0xF2, 0x5A, false},
"CVTPS2PD": {0, 0x5A, false}, "CVTPD2PS": {0x66, 0x5A, false},
// SSE2 packed integers (reg = reg op rm) and the SSSE3 byte shuffle.
"PXOR": {0x66, 0xEF, false},
"POR": {0x66, 0xEB, false},
"PAND": {0x66, 0xDB, false},
"PANDN": {0x66, 0xDF, false},
"PADDB": {0x66, 0xFC, false}, "PADDW": {0x66, 0xFD, false},
"PADDD": {0x66, 0xFE, false}, "PADDQ": {0x66, 0xD4, false},
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
"PCMPEQD": {0x66, 0x76, false},
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
"PCMPGTD": {0x66, 0x66, false},
"PSHUFB": {0x66, 0x00, true},
}
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
// (PSHUFD/PSHUFHW/PSHUFLW also carry the packed-int 0x66/F3/F2 prefixes).
type sseShuf struct {
prefix byte
op byte
}
var sseShufTable = map[string]sseShuf{
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
}
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("SSE binary expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("SSE binary destination must be a vector register")
}
opcode := []byte{0x0F, m.op}
if m.map38 {
opcode = []byte{0x0F, 0x38, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEShuf encodes an imm8 shuffle: SHUFPS $imm, src, dst.
func (e *enc) encodeSSEShuf(m sseShuf, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("SSE shuffle expects 3 operands, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("SSE shuffle needs an imm8 first operand")
}
if imm < -128 || imm > 255 {
return fmt.Errorf("SSE shuffle imm8 %d out of range", imm)
}
src, dst := ops[1], ops[2]
dstReg, ok2 := dst.(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("SSE shuffle destination must be a vector register")
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
i.imm = []byte{byte(int8(imm))}
return e.emit(i)
}
// --- CVTSL2SD / CVTSQ2SD -----------------------------------------------------
// encodeCvtsi2sd encodes a signed integer to scalar double conversion
+159
View File
@@ -0,0 +1,159 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build integration
// Package asm integration tests against the production go-libraries kernels.
// These are excluded from the default test run (go test ./...) so that the
// coverage numbers are identical locally and in CI, where go-libraries is
// not checked out. Run them explicitly with: go test -tags=integration ./asm/
package asm
import (
"bytes"
"os"
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
// all functions plus the file-local mask24 constant; and checks that every
// static-symbol load resolves to the right bytes in the image.
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
}
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
if len(img.Funcs) != 17 {
t.Errorf("functions = %d, want 17", len(img.Funcs))
}
// mask24 as the DATA directives define it.
mask := []byte{
0x00, 0x01, 0x02, 0x80, 0x03, 0x04, 0x05, 0x80,
0x06, 0x07, 0x08, 0x80, 0x09, 0x0a, 0x0b, 0x80,
}
image := img.Bytes()
if got := image[img.Symbols["mask24"] : img.Symbols["mask24"]+16]; !bytes.Equal(got, mask) {
t.Errorf("mask24 contents %x, want %x", got, mask)
}
// Every VMOVDQU mask24<>(SB), X15 (c5 7a 6f 3d + rel32, i.e. a VMOVDQU
// with a RIP-relative r/m) must land on the mask bytes within the image.
loads := 0
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
for pc := 0; pc < len(code); {
inst, err := x86asm.Decode(code[pc:], 64)
if err != nil {
t.Fatalf("%s: decode at +%d: %v", fn.Name, pc, err)
}
// mod=00, rm=101 → RIP-relative.
if inst.Op == x86asm.VMOVDQU && inst.Len == 8 && code[pc+3]&0xC7 == 0x05 {
rel := int32(uint32(code[pc+4]) | uint32(code[pc+5])<<8 | uint32(code[pc+6])<<16 | uint32(code[pc+7])<<24)
target := fn.Offset + pc + 8 + int(rel)
if !bytes.Equal(image[target:target+16], mask) {
t.Errorf("%s: mask load at +%d lands on %x, want %x", fn.Name, pc, image[target:target+16], mask)
}
loads++
}
pc += inst.Len
}
}
if loads != 2 {
t.Errorf("mask loads found = %d, want 2", loads)
}
}
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
// kernel, all functions plus the file-global idx16 constant, and checks
// that the static-symbol load resolves to the right bytes in the image.
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
}
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
if len(img.Funcs) != 10 {
t.Errorf("functions = %d, want 10", len(img.Funcs))
}
// idx16 as the DATA directives define it: dwords 1..16.
idx := make([]byte, 0, 64)
for i := 1; i <= 16; i++ {
idx = append(idx, byte(i), 0, 0, 0)
}
image := img.Bytes()
base := img.Symbols["idx16"]
if base == 0 {
t.Fatal("idx16 not laid out")
}
if got := image[base : base+64]; hexCompact(got) != hexCompact(idx) {
t.Errorf("idx16 contents %x, want %x", got, idx)
}
// The VMOVDQU32 idx16(SB), Z13 load (62 71 7e 48 6f 2d + rel32) must
// resolve to idx16 within the image.
loads := 0
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
pat := []byte{0x62, 0x71, 0x7e, 0x48, 0x6f, 0x2d}
for pos := 0; ; {
i := indexOf(code[pos:], pat)
if i < 0 {
break
}
i += pos
rel := int32(uint32(code[i+6]) | uint32(code[i+7])<<8 | uint32(code[i+8])<<16 | uint32(code[i+9])<<24)
target := fn.Offset + i + 10 + int(rel)
if target != base {
t.Errorf("%s: idx16 load at +%d targets 0x%x, want 0x%x", fn.Name, i, target, base)
}
loads++
pos = i + 10
}
}
if loads != 1 {
t.Errorf("idx16 loads found = %d, want 1", loads)
}
}
// indexOf returns the index of the first occurrence of pat in b, or -1.
func indexOf(b, pat []byte) int {
for i := 0; i+len(pat) <= len(b); i++ {
j := 0
for j < len(pat) && b[i+j] == pat[j] {
j++
}
if j == len(pat) {
return i
}
}
return -1
}
+326
View File
@@ -0,0 +1,326 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
// the symbol tables, the function code bytes and the relocation wiring.
func TestGOObjectLOONG64Structure(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
GLOBL ·table<>(SB), RODATA, $8
DATA ·table<>+0(SB)/8, $0x1122334455667788
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.GOObjectLOONG64("testpkg", "k_loong64.s")
if err != nil {
t.Fatalf("GOObjectLOONG64: %v", err)
}
v := openGoobj(t, obj)
// Package defs: the static GLOBL, then the FuncInfo and the two DWARF
// symbols (debug_line program, subprogram DIE).
defs := v.syms(blkSymdef)
if len(defs) != 4 {
t.Fatalf("symdefs = %d, want 4", len(defs))
}
if defs[0].name != "table" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
t.Errorf("table symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
t.Errorf("funcinfo symbol = %+v", defs[1])
}
if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 {
t.Errorf("lines symbol = %+v", defs[2])
}
if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 {
t.Errorf("DIE symbol = %+v", defs[3])
}
// Non-package defs: four pc tables and the function.
nps := v.syms(blkNonpkgdef)
if len(nps) != 5 {
t.Fatalf("nonpkgdefs = %d, want 5", len(nps))
}
fn := nps[4]
if fn.name != "testpkg.add" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 20 {
t.Errorf("add symbol = %+v", fn)
}
// The function code: 20 bytes, the ground-truth encoding. It sits
// after the GLOBL, FuncInfo, two DWARF symbols and four pc tables.
dataIdx := v.blk(blkDataIdx)
dataBlk := v.blk(blkData)
le := binary.LittleEndian
dOff := le.Uint32(dataIdx[8*4:])
code := dataBlk[dOff : dOff+20]
want := []byte{
0x64, 0x20, 0xc0, 0x28, // ld.d r4, 8(r3)
0x65, 0x40, 0xc0, 0x28, // ld.d r5, 16(r3)
0x84, 0x94, 0x10, 0x00, // add.d r4, r4, r5
0x64, 0x60, 0xc0, 0x29, // st.d r4, 24(r3)
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
for i := range want {
if code[i] != want[i] {
t.Fatalf("code byte %d = %02x, want %02x", i, code[i], want[i])
}
}
// The debug_line program: LNE_set_address (the R_ADDR relocation
// carries the function address), then one row per line change; the
// TEXT is on line 4 (a leading blank line precedes the include), the
// instructions on lines 5-9; an advance to the 20-byte end and an
// end-of-sequence.
linesOff := le.Uint32(dataIdx[4*2:])
lines := dataBlk[linesOff : linesOff+21]
wantLines := []byte{
0x00, 0x09, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // LNE_set_address
0x13, // pc 0, line 5
0x38, // pc 4, line 6
0x38, // pc 8, line 7
0x38, // pc 12, line 8
0x38, // pc 16, line 9
0x02, 0x04, // advance_pc to 20
0x00, 0x01, 0x01, // end_sequence
}
for i := range wantLines {
if lines[i] != wantLines[i] {
t.Fatalf("lines byte %d = %02x, want %02x", i, lines[i], wantLines[i])
}
}
// The subprogram DIE: abbrev 3 (FUNCTION), the qualified name, the
// addrx low_pc slot (R_DWTXTADDR_U4), the size as high_pc, the
// call-frame-CFA frame base, decl file/line and the external flag.
dieOff := le.Uint32(dataIdx[4*3:])
die := dataBlk[dieOff : dieOff+27]
wantDie := []byte{
0x03,
't', 'e', 's', 't', 'p', 'k', 'g', '.', 'a', 'd', 'd', 0,
0x00, 0x00, 0x00, 0x00, // low_pc: addrx slot
0x14, // high_pc: 20
0x01, 0x9c, // frame_base: DW_OP_call_frame_cfa
0x01, 0x00, 0x00, 0x00, // decl_file: 1
0x04, // decl_line: 4
0x01, // external
0x00, // end of children
}
for i := range wantDie {
if die[i] != wantDie[i] {
t.Fatalf("DIE byte %d = %02x, want %02x", i, die[i], wantDie[i])
}
}
// The DWARF symbols carry the function-address references: R_ADDR for
// the line program's set_address, R_DWTXTADDR_U4 for the DIE's addrx
// slot, both against the function's non-package index. The reloc
// index counts relocations, not bytes.
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
if le.Uint32(relocIdx[4*2:]) != 0 || le.Uint32(relocIdx[4*3:]) != 1 || le.Uint32(relocIdx[4*4:]) != 2 {
t.Fatalf("dwarf reloc index ranges: %d %d %d", le.Uint32(relocIdx[4*2:]), le.Uint32(relocIdx[4*3:]), le.Uint32(relocIdx[4*4:]))
}
lr := relocs[:23]
if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr ||
le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 {
t.Errorf("lines reloc = %x", lr)
}
dr := relocs[23:46]
if int32(le.Uint32(dr[0:])) != 13 || dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4() ||
le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 {
t.Errorf("die reloc = %x", dr)
}
// The pc-value deltas are in MinLC (4) units: the flat pcsp covers
// the whole 20-byte function with a delta of 5.
pcspOff := le.Uint32(dataIdx[4*4:])
if got := dataBlk[pcspOff : pcspOff+3]; !bytes.Equal(got, []byte{0x02, 0x05, 0x00}) {
t.Errorf("pcsp = %x, want 020500", got)
}
}
// TestGOObjectLOONG64Link cross-compiles a Go program with the gasm-produced
// object substituted into the package archive, proving cmd/link accepts the
// emitted GOOBJ. The binary is not executed (no LoongArch host or qemu).
// Skipped when no Go toolchain is available.
func TestGOObjectLOONG64Link(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_loong64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module l64link\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the cross build (GOARCH=loong64): the package archive and the
// link line.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=loong64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var pkgArch, work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_loong64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if pkgArch == "" || linkLine == "" || asmObj == "" {
t.Skip("could not locate the archive, asm output or link line in the build log")
}
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
// The archive member holding the assembler's output is named after the
// asm object file (main_loong64.o), as cmd/go packs it with `pack r`.
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
// Assemble the same source with gasm and swap the object in.
pf, perrs := parser.Parse(filepath.Join(dir, "main_loong64.s"), asmSrc)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs)
}
pimg, err := AssembleFileLOONG64(pf)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := pimg.GOObjectLOONG64("main", filepath.Join(dir, "main_loong64.s"))
if err != nil {
t.Fatalf("GOObjectLOONG64: %v", err)
}
// Extract the archive, substitute the object member, repack.
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
extract.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
// Substitute the gasm object for the assembler's archive member (pack
// extracts members read-only).
member := filepath.Join(membersDir, asmMember)
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listCmd.Env = append(os.Environ(), "GOARCH=loong64")
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
pack.Env = append(os.Environ(), "GOARCH=loong64")
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Re-link with our archive in place of the toolchain's. The link line
// carries a GOROOT assignment and $WORK placeholders; run it through the
// shell with the GOEXPERIMENT and GOARCH the toolchain expects (the
// linker compares the object header against its own, experiments
// included).
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=loong64")
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with gasm object: %v\n%s", err, out)
}
// The binary is not executed: there is no LoongArch host or qemu here.
// The link itself and the symbol table prove cmd/link accepted the gasm
// object and laid out the function.
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
nm.Env = append(os.Environ(), "GOARCH=loong64")
nmOut, err := nm.CombinedOutput()
if err != nil {
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
}
if !strings.Contains(string(nmOut), "main.add") {
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
}
}
+335 -75
View File
@@ -6,6 +6,7 @@ package asm
import (
"fmt"
"sort"
"strconv"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
@@ -15,7 +16,7 @@ import (
// file-local static symbols are encoded RIP-relative and resolved within the
// image, so the raw bytes are self-consistent and executable at any base
// address; references to external symbols are recorded as relocations
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file
// (Funcs[i].Relocs, Externals) and left unresolved, the object-file
// emitters turn them into linker relocations.
type Image struct {
Code []byte // concatenated function bodies
@@ -24,6 +25,10 @@ type Image struct {
Symbols map[string]int // static symbol → byte offset within the image
DataSyms []DataSymbol // GLOBL symbols, in layout order
Externals []string // referenced but undefined symbols, sorted
// SourcePath is the assembled file's path, recorded in the DWARF
// sections in place of a placeholder name. Empty when the image was
// not built from a named file.
SourcePath string
}
// FuncLayout describes one assembled function within an Image.
@@ -41,6 +46,7 @@ type FuncLayout struct {
Labels map[string]int // local labels, function-relative
Relocs []Reloc // static-symbol references, in emission order
Spadj []SpadjStep // stack-adjustment boundaries, ascending by PC
Lines []LineEntry // source-line table: byte offset → source line
}
// SpadjStep is one stack-adjustment boundary: Value is the SP delta from the
@@ -50,17 +56,69 @@ type SpadjStep struct {
Value int
}
// Reloc is one static-symbol reference within a function body: the disp32
// field at Off (function-relative) must reach the symbol plus Addend,
// measured from After, the address just past the instruction. An External
// relocation names a symbol no GLOBL in the file defines; the object-file
// emitters carry it into the output's relocation table.
// LineEntry maps a byte offset (function-relative) to a source line number.
type LineEntry struct {
Offset int
Line int
}
// LineAt returns the source line number for the given function-relative byte
// offset, using a binary search on the line table. Returns 0 if the offset
// is before the first instruction or the table is empty.
func (fl *FuncLayout) LineAt(offset int) int {
if len(fl.Lines) == 0 {
return 0
}
// Binary search: find the last entry with Offset <= offset.
lo, hi := 0, len(fl.Lines)-1
for lo < hi {
mid := (lo + hi + 1) / 2
if fl.Lines[mid].Offset <= offset {
lo = mid
} else {
hi = mid - 1
}
}
if fl.Lines[lo].Offset <= offset {
return fl.Lines[lo].Line
}
return 0
}
// RelocKind discriminates the relocation a static-symbol reference needs;
// the encoders record one per SB reference, and the object-file emitters map
// it to their format's relocation type.
type RelocKind int
const (
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
RelCall // R_CALL: CALL to a function symbol (amd64)
RelTLSLE // R_TLS_LE: local-exec TLS load, no symbol (amd64 guard)
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
RelRISCVJal // R_RISCV_JAL (J-type call)
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
RelArm64Branch // R_CALLARM64 (BL instruction)
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
)
type Reloc struct {
// Off is the function-relative offset of the field the linker patches
// and After the address just past the instruction, the base the
// assembler measures PC-relative displacements from. Name plus
// Addend select the target: the symbol plus the byte offset. An
// External relocation names a symbol no GLOBL in the file defines;
// the object-file emitters carry it into the output's relocation
// table.
Off int
After int
Name string
Addend int64
External bool
Kind RelocKind
}
// DataSymbol describes one GLOBL symbol laid out in the data section.
@@ -86,7 +144,7 @@ func (img *Image) Bytes() []byte {
// reference to a file-local static symbol becomes a RIP-relative load whose
// displacement is resolved against that layout; a reference to a symbol no
// GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero — the object-file emitters resolve it at link
// displacement left zero, the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f)
@@ -99,7 +157,8 @@ func AssembleFile(f *ast.File) (*Image, error) {
}
link := &linkInfo{symbols: known, allowExternal: true}
img := &Image{Symbols: map[string]int{}}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
textOff := map[string]int{}
type asmFunc struct {
name string
patches []sbPatch
@@ -110,7 +169,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
if !ok {
continue
}
code, patches, labels, steps, err := assemble(t, link)
code, patches, labels, steps, lines, err := assemble(t, link)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
@@ -124,6 +183,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
Lines: lines,
}
for _, f := range t.Flags {
switch f {
@@ -136,6 +196,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, s := range steps {
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
}
textOff[t.Name.Name] = len(img.Code)
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
@@ -168,13 +229,27 @@ func AssembleFile(f *ast.File) (*Image, error) {
base := img.Funcs[i].Offset
code := img.Code[base : base+img.Funcs[i].Size]
for _, p := range fn.patches {
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend}
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend, Kind: p.kind}
if p.kind == RelTLSLE {
// The TLS slot has no symbol: the linker fills the offset
// from the runtime's TLS layout.
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
continue
}
if imgOff, ok := img.Symbols[p.name]; ok {
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else if imgOff, ok := textOff[p.name]; ok {
// A CALL to a TEXT function of the same file: resolve the
// displacement against the function's layout position.
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else {
reloc.External = true
externals[p.name] = true
@@ -189,88 +264,273 @@ func AssembleFile(f *ast.File) (*Image, error) {
return img, nil
}
// AssembleFileRISCV assembles every TEXT function of a parsed RISC-V file
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
// code. SB references in the code are encoded as AUIPC pairs with zero
// immediates; the object-file emitters record relocations for the linker.
func AssembleFileRISCV(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
continue
}
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
Static: t.Name.Static,
Offset: len(img.Code),
Size: len(code),
Frame: frameSize(t),
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
Lines: lines,
Spadj: spadj,
Relocs: relocs,
}
for _, f := range t.Flags {
switch f {
case "NOSPLIT":
fl.NoSplit = true
case "SPWRITE":
fl.SPWrite = true
}
}
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
}
// Lay out the data section behind the code, 16-aligned.
dataStart := len(img.Code)
for _, d := range dataSyms {
pos := dataStart + len(img.Data)
for pos%16 != 0 {
img.Data = append(img.Data, 0)
pos++
}
img.Symbols[d.name] = pos
img.Data = append(img.Data, d.buf...)
img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name,
Pkg: d.pkg,
Offset: len(img.Data) - len(d.buf), // relative to the data section
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Dupok: d.dupok,
})
}
markExternals(img, dataSyms)
return img, nil
}
// AssembleFileLOONG64 assembles every TEXT function of a parsed loong64 file
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
// code. SB references in the code are encoded as pcalau12i pairs with zero
// immediates; the object-file emitters record R_LOONG64_ADDR_HI/LO
// relocations for the linker.
func AssembleFileLOONG64(f *ast.File) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
continue
}
code, labels, relocs, lines, spadj, err := assembleLOONG64(t)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
Static: t.Name.Static,
Offset: len(img.Code),
Size: len(code),
Frame: frameSize(t),
Args: argsSize(t),
Line: t.Pos().Line,
Labels: labels,
Lines: lines,
Spadj: spadj,
Relocs: relocs,
}
for _, f := range t.Flags {
switch f {
case "NOSPLIT":
fl.NoSplit = true
case "SPWRITE":
fl.SPWrite = true
}
}
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
}
// Lay out the data section behind the code, 16-aligned.
dataStart := len(img.Code)
for _, d := range dataSyms {
pos := dataStart + len(img.Data)
for pos%16 != 0 {
img.Data = append(img.Data, 0)
pos++
}
img.Symbols[d.name] = pos
img.Data = append(img.Data, d.buf...)
img.DataSyms = append(img.DataSyms, DataSymbol{
Name: d.name,
Pkg: d.pkg,
Offset: len(img.Data) - len(d.buf), // relative to the data section
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Dupok: d.dupok,
})
}
markExternals(img, dataSyms)
return img, nil
}
// markExternals identifies relocations that reference symbols not defined in
// the file (neither a GLOBL/DATA symbol nor a TEXT function) and records them
// as external. The non-amd64 architectures emit relocations for every SB
// reference; this post-processing step distinguishes file-local from external.
func markExternals(img *Image, dataSyms []dataSym) {
known := make(map[string]bool, len(dataSyms)+len(img.Funcs))
for _, d := range dataSyms {
known[d.name] = true
}
for _, fn := range img.Funcs {
known[fn.Name] = true
}
externals := map[string]bool{}
for i := range img.Funcs {
for j := range img.Funcs[i].Relocs {
r := &img.Funcs[i].Relocs[j]
if !known[r.Name] {
r.External = true
externals[r.Name] = true
}
}
}
for name := range externals {
img.Externals = append(img.Externals, name)
}
sort.Strings(img.Externals)
}
// dataSym is one GLOBL symbol and its DATA initialiser.
type dataSym struct {
name string
pkg string
buf []byte
size int
static bool
rodata bool
dupok bool
}
// collectData gathers the file's static symbols (GLOBL) and their initial
// contents (DATA) into byte buffers, in declaration order.
// contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts
// every DATA line before its symbol's GLOBL, so the symbols are registered
// before the initialisers are applied.
func collectData(f *ast.File) ([]dataSym, error) {
index := map[string]int{}
var syms []dataSym
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Globl:
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
name := dd.Name.Name
if _, dup := index[name]; dup {
return nil, fmt.Errorf("duplicate GLOBL %q", name)
}
size := 0
if dd.Size != nil && dd.Size.Imm.HasVal {
size = int(dd.Size.Imm.Val)
}
index[name] = len(syms)
ds := dataSym{
name: name,
pkg: dd.Name.Pkg,
buf: make([]byte, size),
static: dd.Name.Static,
}
for _, f := range dd.Flags {
switch f {
case "RODATA":
ds.rodata = true
case "DUPOK":
ds.dupok = true
case "1":
ds.dupok = true
case "8":
ds.rodata = true
case "9":
ds.dupok = true
ds.rodata = true
gd, ok := d.(*ast.Globl)
if !ok {
continue
}
if gd.Name == nil || gd.Name.Pseudo != "SB" {
continue
}
name := gd.Name.Name
if _, dup := index[name]; dup {
return nil, fmt.Errorf("duplicate GLOBL %q", name)
}
size := 0
if gd.Size != nil && gd.Size.Imm.HasVal {
size = int(gd.Size.Imm.Val)
}
index[name] = len(syms)
ds := dataSym{
name: name,
pkg: gd.Name.Pkg,
buf: make([]byte, size),
size: size,
static: gd.Name.Static,
}
for _, f := range gd.Flags {
switch f {
case "RODATA":
ds.rodata = true
case "DUPOK":
ds.dupok = true
default:
// Legacy numeric flag constants (runtime/textflag.h):
// DUPOK is 2, RODATA is 8; combinations arrive as one
// number (e.g. 10 = RODATA|DUPOK).
if n, err := strconv.Atoi(f); err == nil {
if n&2 != 0 {
ds.dupok = true
}
if n&8 != 0 {
ds.rodata = true
}
}
}
syms = append(syms, ds)
case *ast.Data:
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
i, ok := index[dd.Name.Name]
if !ok {
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
}
if dd.Value == nil || !dd.Value.Imm.HasVal {
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
}
w := dd.Width
switch w {
case 1, 2, 4, 8:
default:
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
}
off := dd.Name.Offset
buf := syms[i].buf
if off < 0 || off+int64(w) > int64(len(buf)) {
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
}
v := dd.Value.Imm.Val
if dd.Value.Imm.Neg {
v = -v
}
for j := 0; j < w; j++ {
buf[off+int64(j)] = byte(v >> (8 * j))
}
}
syms = append(syms, ds)
}
for _, d := range f.Decls {
dd, ok := d.(*ast.Data)
if !ok {
continue
}
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
i, ok := index[dd.Name.Name]
if !ok {
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
}
if dd.Value == nil || !dd.Value.Imm.HasVal {
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
}
w := dd.Width
switch w {
case 1, 2, 4, 8:
default:
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
}
off := dd.Name.Offset
buf := syms[i].buf
if off < 0 || off+int64(w) > int64(len(buf)) {
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
}
v := dd.Value.Imm.Val
if dd.Value.Imm.Neg {
v = -v
}
for j := range w {
buf[off+int64(j)] = byte(v >> (8 * j))
}
}
return syms, nil
+35 -63
View File
@@ -4,18 +4,14 @@
package asm
import (
"bytes"
"os"
"strings"
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestAssembleFileStaticData checks the whole-image layout — code, padding
// and the data section — and that the RIP-relative displacements of static
// TestAssembleFileStaticData checks the whole-image layout; code, padding
// and the data section; and that the RIP-relative displacements of static
// symbol loads resolve to the right bytes.
func TestAssembleFileStaticData(t *testing.T) {
f, errs := parser.Parse("d_amd64.s", `
@@ -133,64 +129,40 @@ DATA x<>+0(SB)/4, $1
}
}
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
// all functions plus the file-local mask24 constant — and checks that every
// static-symbol load resolves to the right bytes in the image. Skipped when
// the sibling repository is not checked out.
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
// TestCollectDataNumericFlags pins the numeric GLOBL flag constants from
// runtime/textflag.h: DUPOK is 2, RODATA is 8, and combinations arrive as
// one number (9 = NOPROF|RODATA, 10 = RODATA|DUPOK).
func TestCollectDataNumericFlags(t *testing.T) {
tests := []struct {
flags string
rodata bool
dupok bool
}{
{"2", false, true},
{"8", true, false},
{"9", true, false}, // NOPROF|RODATA, not DUPOK
{"10", true, true}, // RODATA|DUPOK
{"RODATA", true, false},
{"DUPOK", false, true},
{"RODATA|DUPOK", true, true},
}
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
if len(img.Funcs) != 17 {
t.Errorf("functions = %d, want 17", len(img.Funcs))
}
// mask24 as the DATA directives define it.
mask := []byte{
0x00, 0x01, 0x02, 0x80, 0x03, 0x04, 0x05, 0x80,
0x06, 0x07, 0x08, 0x80, 0x09, 0x0a, 0x0b, 0x80,
}
image := img.Bytes()
if got := image[img.Symbols["mask24"] : img.Symbols["mask24"]+16]; !bytes.Equal(got, mask) {
t.Errorf("mask24 contents %x, want %x", got, mask)
}
// Every VMOVDQU mask24<>(SB), X15 (c5 7a 6f 3d + rel32, i.e. a VMOVDQU
// with a RIP-relative r/m) must land on the mask bytes within the image.
loads := 0
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
for pc := 0; pc < len(code); {
inst, err := x86asm.Decode(code[pc:], 64)
if err != nil {
t.Fatalf("%s: decode at +%d: %v", fn.Name, pc, err)
}
// mod=00, rm=101 → RIP-relative.
if inst.Op == x86asm.VMOVDQU && inst.Len == 8 && code[pc+3]&0xC7 == 0x05 {
rel := int32(uint32(code[pc+4]) | uint32(code[pc+5])<<8 | uint32(code[pc+6])<<16 | uint32(code[pc+7])<<24)
target := fn.Offset + pc + 8 + int(rel)
if !bytes.Equal(image[target:target+16], mask) {
t.Errorf("%s: mask load at +%d lands on %x, want %x", fn.Name, pc, image[target:target+16], mask)
}
loads++
}
pc += inst.Len
for _, tt := range tests {
src := "TEXT \u00b7f(SB), NOSPLIT, $0\n\tRET\nGLOBL sym(SB), " + tt.flags + ", $8\n"
f, errs := parser.Parse("f_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse %q: %v", tt.flags, errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble %q: %v", tt.flags, err)
}
if len(img.DataSyms) != 1 {
t.Fatalf("%q: data syms = %d, want 1", tt.flags, len(img.DataSyms))
}
d := img.DataSyms[0]
if d.Rodata != tt.rodata || d.Dupok != tt.dupok {
t.Errorf("flags %q: rodata=%v dupok=%v, want rodata=%v dupok=%v",
tt.flags, d.Rodata, d.Dupok, tt.rodata, tt.dupok)
}
}
if loads != 2 {
t.Errorf("mask loads found = %d, want 2", loads)
}
}
File diff suppressed because it is too large Load Diff
+597
View File
@@ -0,0 +1,597 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// loong64 (LoongArch) instruction encoding.
//
// The encoder is data-driven: each mnemonic maps to an instruction format and
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own loong64
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
// exactly, the ground-truth oracle for the verify suite.
//
// All LoongArch instructions are 32 bits, little-endian. The formats used
// here (per the LoongArch Volume I specification):
//
// 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0]
// 2R opcode[31:15] | rj[4:0] | rd[4:0]
// 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0]
// 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0]
// 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0]
// 2RI20 opcode[31:25] | si20[19:0] | rd[4:0]
// 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z)
// B/BL opcode[31:26] | offs[25:0]
// 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0]
// IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0]
// 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0]
//
// The opcode constants are pre-positioned (they include the zero bit ranges
// of the immediate and register fields), mirroring the toolchain's OP_*
// helpers, so each l64* function only ORs its fields in.
import "maps"
// loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
// flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's
// assembly uses. Returns -1 for an unrecognised name.
func loong64RegNum(name string) int {
switch name {
case "R0", "ZERO":
return 0
case "R1", "RA", "LINK":
return 1
case "R2", "TP":
return 2
case "R3", "SP":
return 3
case "R4", "A0":
return 4
case "R5", "A1":
return 5
case "R6", "A2":
return 6
case "R7", "A3":
return 7
case "R8", "A4":
return 8
case "R9", "A5":
return 9
case "R10", "A6":
return 10
case "R11", "A7":
return 11
case "R12", "T0":
return 12
case "R13", "T1":
return 13
case "R14", "T2":
return 14
case "R15", "T3":
return 15
case "R16", "T4":
return 16
case "R17", "T5":
return 17
case "R18", "T6":
return 18
case "R19", "T7":
return 19
case "R20", "T8":
return 20
case "R21":
return 21
case "R22", "G", "g", "FP":
return 22
case "R23", "S0":
return 23
case "R24", "S1":
return 24
case "R25", "S2":
return 25
case "R26", "S3":
return 26
case "R27", "S4":
return 27
case "R28", "S5":
return 28
case "R29", "S6", "CTXT":
return 29
case "R30", "S7", "TMP":
return 30
case "R31", "S8":
return 31
}
// F0-F31, FCC0-FCC7, FCSR0-FCSR31.
if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], 31)
}
if len(name) >= 3 && name[:3] == "FCC" {
return loong64RegSpecial(name[3:], 7)
}
if len(name) < 2 {
return -1
}
prefix, digits := name[:1], name[1:]
if digits[0] < '0' || digits[0] > '9' {
return -1
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return -1
}
n = n*10 + int(digits[i]-'0')
}
if prefix == "F" && n <= 31 {
return n
}
return -1
}
// loong64RegSpecial parses a numbered FCC/FCSR register.
func loong64RegSpecial(digits string, max int) int {
if digits == "" {
return -1
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return -1
}
n = n*10 + int(digits[i]-'0')
}
if n <= max {
return n
}
return -1
}
// ---- format helpers ----
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
func l64rrr(op uint32, rk, rj, rd int) uint32 {
return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64rr encodes a 2R instruction: op | rj<<5 | rd.
func l64rr(op uint32, rj, rd int) uint32 {
return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd.
func l64irr(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd.
func l64irr14(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd.
func l64irr16(op uint32, imm, rj, rd int) uint32 {
return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64ir encodes a 2RI20 instruction: op | si20<<5 | rd.
func l64ir(op uint32, imm, rd int) uint32 {
return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f)
}
// l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the
// 4-byte-aligned word distance (the toolchain stores the shifted value).
func l64bbl(op uint32, offs int) uint32 {
return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF
}
// l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF):
// op | si21[15:0]<<10 | rj<<5 | si21[20:16].
func l64ir21(op uint32, offs, rj int) uint32 {
v := uint32(offs)
return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F
}
// l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4.
func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f)
}
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
// The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller
// must have validated them (0..31 for the .w forms, 0..63 for the .d forms,
// lsb <= msb), the same rule the toolchain enforces as "illegal bit number".
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
}
// l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd.
func l64irrr(op uint32, sa, rk, rj, rd int) uint32 {
return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
}
// l64i15 encodes a no-operand system instruction with a 15-bit code field
// (SYSCALL, BREAK, DBAR): op | code[14:0].
func l64i15(op uint32, code int) uint32 {
return op | uint32(code)&0x7FFF
}
// l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint.
func l64irr5i(op uint32, offs, rj, hint int) uint32 {
return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f)
}
// l64wordLE encodes a uint32 as 4 little-endian bytes.
func l64wordLE(w uint32) []byte {
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
}
// l64WordsLE concatenates one or more instruction words as little-endian bytes.
func l64WordsLE(ws ...uint32) []byte {
var out []byte
for _, w := range ws {
out = append(out, l64wordLE(w)...)
}
return out
}
// ---- instruction formats ----
type l64Format uint8
const (
l64Frrr l64Format = iota // 3R (integer and FP arithmetic)
l64Frr // 2R
l64Firr // 2RI12 (arithmetic with 12-bit immediate)
l64Firr14 // 2RI14 (ldptr/stptr)
l64Firr16 // 2RI16 (addu16i.d)
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
l64Firir // bstrins/bstrpick
l64Firrr // alsl
l64Fi15 // syscall/break/dbar
l64Fam // atomic (3R with the AM field order)
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
l64Fshift // 2RI12 with a 5/6-bit shift immediate
l64Fpreld // preld (2RI12 + 5-bit hint)
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
// opcode constant, positioned at its exact bit range.
type l64Enc struct {
format l64Format
op uint32
}
// l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form
// and the 2RI12 immediate form (which is a shift for the shift mnemonics).
type l64DualEnc struct {
rrr uint32 // 3R register form
imm uint32 // 2RI12 immediate form
shift bool // the immediate form is a 5/6-bit shift amount
}
// l64DualTable maps the dual-form arithmetic/logic mnemonics to both
// encodings; the assembler picks by operand kind.
var l64DualTable = map[string]l64DualEnc{}
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
// yet; the base integer, memory and floating-point ISA is complete.
var l64InstrTable = map[string]l64Enc{}
func init() {
// 3R, integer.
rrr := map[string]uint32{
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
"SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15,
"ROTR": 0x36 << 15, "ROTRV": 0x37 << 15,
"MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15,
"MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15,
"MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15,
"DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15,
"DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15,
"DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15,
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
}
// 3R, floating point.
rrr["MULF"] = 0x209 << 15
rrr["MULD"] = 0x20a << 15
rrr["DIVF"] = 0x20d << 15
rrr["DIVD"] = 0x20e << 15
rrr["SUBF"] = 0x205 << 15
rrr["SUBD"] = 0x206 << 15
rrr["ADDF"] = 0x201 << 15
rrr["ADDD"] = 0x202 << 15
rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15
rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15
rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15
rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15
rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15
rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15
rrr["FMINF"] = 0x215 << 15
rrr["FMIND"] = 0x216 << 15
rrr["FMAXF"] = 0x211 << 15
rrr["FMAXD"] = 0x212 << 15
rrr["FMAXAF"] = 0x219 << 15
rrr["FMAXAD"] = 0x21a << 15
rrr["FMINAF"] = 0x21d << 15
rrr["FMINAD"] = 0x21e << 15
rrr["FSCALEBF"] = 0x221 << 15
rrr["FSCALEBD"] = 0x222 << 15
rrr["FCOPYSGF"] = 0x225 << 15
rrr["FCOPYSGD"] = 0x226 << 15
for m, op := range rrr {
l64InstrTable[m] = l64Enc{format: l64Frrr, op: op}
}
// 2R.
rr := map[string]uint32{
"CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10,
"CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10,
"REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10,
"REVH2W": 0x10 << 10, "REVHV": 0x11 << 10,
"BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10,
"EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10,
"TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10,
"MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10,
"MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10,
"FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10,
"MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10,
"ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10,
"MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10,
"NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10,
"SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10,
"FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10,
"FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10,
"FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10,
"FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10,
"FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10,
"FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10,
"FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10,
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
}
for m, op := range rr {
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
}
// RDTIME is a 2R instruction with rd and rj in swapped positions.
l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10}
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate.
maps.Copy(l64DualTable, map[string]l64DualEnc{
"ADD": {rrr: 0x20 << 15, imm: 0x00a << 22},
"ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22},
"ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22},
"ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22},
"AND": {rrr: 0x29 << 15, imm: 0x00d << 22},
"OR": {rrr: 0x2a << 15, imm: 0x00e << 22},
"XOR": {rrr: 0x2b << 15, imm: 0x00f << 22},
"SGT": {rrr: 0x24 << 15, imm: 0x008 << 22},
"SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22},
"SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true},
"SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true},
"SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true},
"ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true},
"SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true},
"SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true},
"SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true},
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
})
// 2RI12, pure immediate arithmetic (LU52ID has no register form).
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
// 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and
// stores (ldptr/stptr), with the offset scaled by 4.
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d
l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d)
// 2RI20.
l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25}
l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25}
l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25}
// LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R, fused multiply-add.
rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
}
for m, op := range rrrr {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
}
// IRIR, bit-field insert/extract.
irir := map[string]uint32{
"BSTRINSW": 0x3<<21 | 0x0<<15,
"BSTRINSV": 0x2 << 22,
"BSTRPICKW": 0x3<<21 | 0x1<<15,
"BSTRPICKV": 0x3 << 22,
}
for m, op := range irir {
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
}
// 3RI2, ALSL.
irrr := map[string]uint32{
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
}
for m, op := range irrr {
l64InstrTable[m] = l64Enc{format: l64Firrr, op: op}
}
// 0-operand system instructions.
l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15}
l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15}
l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15}
// PRELD.
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
"AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15,
"AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15,
"AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15,
"AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15,
"AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15,
"AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15,
"AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15,
"AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15,
"AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15,
"AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15,
"AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15,
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
}
for m, op := range am {
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
}
}
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
// register move between the integer and floating-point register banks, the
// MOVW/MOVV specials the Go assembler accepts.
var l64FpMovTable = map[string]uint32{
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
"MOVV.R.FCC": 0x4536 << 10, // movgr2cf
"MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr
"MOVV.F.R": 0x452e << 10, // movfr2gr.d
"MOVV.F.FCC": 0x4534 << 10, // movfr2cf
"MOVV.FCC.R": 0x4537 << 10, // movcf2gr
"MOVV.FCC.F": 0x4535 << 10, // movcf2fr
"MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr
"MOVW.R.F": 0x4529 << 10, // movgr2fr.w
"MOVW.F.R": 0x452d << 10, // movfr2gr.s
}
// l64branchTable holds the 16-bit branch and jump encodings (2RI16).
var l64branchTable = map[string]uint32{
"BEQ": 0x16 << 26,
"BNE": 0x17 << 26,
"BLT": 0x18 << 26,
"BGE": 0x19 << 26,
"BLTU": 0x1a << 26,
"BGEU": 0x1b << 26,
"JIRL": 0x13 << 26,
}
// l64branch21Table holds the single-register branches with 21-bit offsets:
// the negative opcode constants the toolchain uses for the short forms.
var l64branch21Table = map[string]uint32{
"BEQZ": 0x10 << 26, // beq r0, rj → beqz
"BNEZ": 0x11 << 26, // bne r0, rj → bnez
"BLTZ": 0x18 << 26, // blt rj, r0 → bltz
"BGEZ": 0x19 << 26, // bge rj, r0 → bgez
"BGTZ": 0x18 << 26, // blt r0, rj → bgtz
"BLEZ": 0x19 << 26, // bge r0, rj → blez
"BFPT": 0x12<<26 | 0x1<<8,
"BFPF": 0x12<<26 | 0x0<<8,
}
// l64jumpTable maps the jump pseudo-instructions and their aliases to the
// B/BL opcode constants.
var l64jumpTable = map[string]uint32{
"JMP": 0x14 << 26, // b
"B": 0x14 << 26, // b
"JAL": 0x15 << 26, // bl
"CALL": 0x15 << 26, // bl
"BL": 0x15 << 26, // bl
}
// l64loadStoreTable maps the MOV width mnemonics to their load and store
// 2RI12 opcodes. The load opcode is the negated store opcode, exactly as
// the toolchain derives it.
var l64loadStoreTable = map[string]struct{ ld, st uint32 }{
"MOVB": {0x0a0 << 22, 0x0a4 << 22},
"MOVH": {0x0a1 << 22, 0x0a5 << 22},
"MOVW": {0x0a2 << 22, 0x0a6 << 22},
"MOVV": {0x0a3 << 22, 0x0a7 << 22},
"MOVBU": {0x0a8 << 22, 0x0a4 << 22},
"MOVHU": {0x0a9 << 22, 0x0a5 << 22},
"MOVWU": {0x0aa << 22, 0x0a6 << 22},
"MOVF": {0x0ac << 22, 0x0ad << 22},
"MOVD": {0x0ae << 22, 0x0af << 22},
}
// l64movRegTable maps a register-to-register MOV mnemonic to its expansion,
// matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h,
// MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d
// and are handled separately in the assembler.
type l64MovRegEnc struct {
rr bool // 2R format (ext.w.b/ext.w.h)
op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode
imm int // 2RI12 immediate for MOVBU's andi
}
var l64movRegTable = map[string]l64MovRegEnc{
"MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj
"MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj
"MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0
"MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0
"MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff
}
// l64movFpRegTable maps a floating-point register move mnemonic to its 2R
// opcode (fmov.s / fmov.d), used when both operands are F registers.
var l64movFpRegTable = map[string]uint32{
"MOVF": 0x4525 << 10,
"MOVD": 0x4526 << 10,
}
// l64RegClass discriminates integer (R), floating-point (F) and condition
// (FCC) registers for the MOV pseudo-instruction's register-move encoding.
type l64RegClass int
const (
l64ClsNone l64RegClass = iota
l64ClsGR
l64ClsFP
l64ClsFCC
l64ClsFCSR
)
// loong64RegClass reports the register class of a register operand name.
func loong64RegClass(name string) l64RegClass {
switch {
case name == "":
return l64ClsNone
case len(name) >= 3 && name[:3] == "FCC":
return l64ClsFCC
case len(name) >= 4 && name[:4] == "FCSR":
return l64ClsFCSR
case name[0] == 'F':
return l64ClsFP
default:
return l64ClsGR
}
}
+330
View File
@@ -0,0 +1,330 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
func firstTextLOONG64(t *testing.T, src string) *ast.Text {
t.Helper()
f, errs := parser.Parse("f_loong64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
for _, d := range f.Decls {
if fn, ok := d.(*ast.Text); ok {
return fn
}
}
t.Fatal("no TEXT found")
return nil
}
// assembleLOONG64Helper assembles one TEXT function and returns its bytes.
func assembleLOONG64Helper(t *testing.T, fn *ast.Text) []byte {
t.Helper()
code, _, _, _, _, err := assembleLOONG64(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
return code
}
// wantWords checks that code matches the expected little-endian words.
func wantWords(t *testing.T, code []byte, want ...uint32) {
t.Helper()
got := make([]uint32, 0, len(code)/4)
for i := 0; i+4 <= len(code); i += 4 {
got = append(got, binary.LittleEndian.Uint32(code[i:]))
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
func TestLOONG64_add(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
// 5 instructions: two ld.d, add.d, st.d, jirl r0, r1, 0.
wantWords(t, code,
0x28C02064, // ld.d r4, 8(r3)
0x28C04065, // ld.d r5, 16(r3)
0x00109484, // add.d r4, r4, r5
0x29C06064, // st.d r4, 24(r3)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_arithmetic(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·arith(SB), NOSPLIT, $0
ADDV R4, R5, R6
SUBV R7, R8, R9
MULV R10, R11, R12
DIVV R13, R14, R15
AND R16, R17, R18
OR R18, R19, R20
XOR R20, R21, R2
SLLV R2, R23, R24
SRLV R24, R25, R26
SRAV R26, R27, R28
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x001090A6, // add.d r6, r5, r4
0x00119D09, // sub.d r9, r8, r7
0x001DA96C, // mul.d r12, r11, r10
0x002235CF, // div.d r15, r14, r13
0x0014C232, // and r18, r17, r16
0x00154A74, // or r20, r19, r18
0x0015D2A2, // xor r2, r21, r20
0x00188AF8, // sll.d r24, r23, r2
0x0019633A, // srl.d r26, r25, r24
0x0019EB7C, // sra.d r28, r27, r26
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_immediates(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·imm(SB), NOSPLIT, $0
ADDV $42, R4, R5
ADDV $-8, R6
AND $0xff, R7, R8
OR $1, R9, R10
SGT $100, R13, R14
SLLV $4, R15, R16
MOVV $0x12345, R17
MOVV $0, R18
MOVW $0, R19
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x02C0A885, // addi.d r5, r4, 42
0x02FFE0C6, // addi.d r6, r6, -8
0x0343FCE8, // andi r8, r7, 0xff
0x0380052A, // ori r10, r9, 1
0x020191AE, // slti r14, r13, 100
0x004111F0, // slli.d r16, r15, 4
0x14000251, // lu12i.w r17, 0x12
0x038D1631, // ori r17, r17, 0x345
0x00150012, // or r18, r0, r0
0x00170013, // sll.w r19, r0, r0
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_loadStore(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·mem(SB), NOSPLIT, $0
MOVV (R4), R5
MOVV R5, (R6)
MOVW 8(R7), R8
MOVB R9, -4(R10)
MOVV (R11)(R12), R13
MOVV R14, (R15)(R16)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x28C00085, // ld.d r5, 0(r4)
0x29C000C5, // st.d r5, 0(r6)
0x288020E8, // ld.w r8, 8(r7)
0x293FF149, // st.b r9, -4(r10)
0x380C316D, // ldx.d r13, r11, r12
0x381C41EE, // stx.d r14, r15, r16
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_branches(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·br(SB), NOSPLIT, $0
BEQ R4, R5, done
BNE R6, R7, skip
BLT R8, R9, done
BGE R10, R11, done
BLTU R12, R13, done
BGEU R14, R15, done
skip:
JMP done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
// skip is at 0x18 (6 words), done at 0x1c.
wantWords(t, code,
0x58001C85, // beq r5, r4, +7
0x5C0018C7, // bne r7, r6, +6
0x60001509, // blt r9, r8, +5
0x6400114B, // bge r11, r10, +4
0x68000D8D, // bltu r13, r12, +3
0x6C0009CF, // bgeu r15, r14, +2
0x50000400, // b done (+1, chain-folded through skip)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_frame(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $32-8
MOVV R4, R5
MOVV arg+0(FP), R6
MOVV R7, local-8(SP)
MOVV local-8(SP), R8
MOVV R9, ret+0(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
// autosize = align8(32+8) = 40; prologue stores LR at -40(SP),
// opens the frame, stores LR again at 0(SP). The function is a leaf
// (no calls), so the epilogue skips the LR restore. FP args are at
// autosize+8; SP locals at autosize+offset.
wantWords(t, code,
0x29FF6061, // st.d r1, -40(r3)
0x02FF6063, // addi.d r3, r3, -40
0x29C00061, // st.d r1, 0(r3)
0x00150085, // or r5, r4, r0
0x28C0C066, // ld.d r6, 48(r3) arg+0(FP) → 0+40+8
0x29C08067, // st.d r7, 32(r3) local-8(SP) → 40-8
0x28C08068, // ld.d r8, 32(r3)
0x29C0C069, // st.d r9, 48(r3) ret+0(FP) → 0+40+8
0x02C0A063, // addi.d r3, r3, 40
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_jumpChain(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·jc(SB), NOSPLIT, $0
JMP a
a:
JMP b
b:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x50000800, // b +2 (a, chain-folded to b)
0x50000400, // b +1 (b)
0x4C000020, // jirl r0, r1, 0
)
}
func TestLOONG64_dconClasses(t *testing.T) {
cases := []struct {
v int64
word int // expected word count
}{
{0x123456789, 3}, // lu12i.w + ori + lu32i.d
{-1, 2}, // addi.d + lu52i.d (the MOV path handles -1 earlier)
{0x1000000000000, 2}, // addi.w + lu32i.d
{0x123456789abcdef0, 4}, // full sequence
{0xFFFFFFFFF, 2}, // lu12i.w + ori
{0x1234567800000000, 3}, // addi.w + lu32i.d + lu52i.d
}
for _, c := range cases {
if n := len(l64DconMovWords(0, c.v)); n != c.word {
t.Errorf("0x%x: %d words, want %d", c.v, n, c.word)
}
}
}
func TestLOONG64_regNames(t *testing.T) {
cases := map[string]int{
"R0": 0, "R31": 31, "F0": 0, "F31": 31, "FCC0": 0, "FCC7": 7,
"FCSR0": 0, "FCSR3": 3, "ZERO": 0, "RA": 1, "SP": 3, "g": 22, "G": 22,
"R32": -1, "FCC8": -1, "X0": -1, "R": -1, "TMP": 30, "CTXT": 29,
}
for name, want := range cases {
if got := loong64RegNum(name); got != want {
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
}
}
}
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
// A spot-check that assembleLOONG64 emits the same bytes the Go
// toolchain does for a small kernel (the full comparison lives in
// verify's TestGroundTruthLOONG64).
src := `#include "textflag.h"
TEXT ·k(SB), NOSPLIT, $0-0
ADDV R4, R5, R6
MOVV $0x100000, R7
BEQ R6, R7, done
JMP done
done:
RET
`
fn := firstTextLOONG64(t, src)
code := assembleLOONG64Helper(t, fn)
want := []byte{
0xa6, 0x90, 0x10, 0x00, // add.d r6, r5, r4
0x07, 0x20, 0x00, 0x14, // lu12i.w r7, 0x100
0xc7, 0x08, 0x00, 0x58, // beq r7, r6, +2 (done)
0x00, 0x04, 0x00, 0x50, // b +1 (done)
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestLOONG64IndirectBranch pins the indirect branch encodings: JMP (Rj) and
// JAL (Rj) lower to jirl, and the raw JIRL spelling encodes the written
// offset (the Go loong64 assembler deletes raw JIRL instructions entirely,
// so this form is a gasm-only superset with faithful semantics).
func TestLOONG64IndirectBranch(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
JMP (R4)
JIRL R0, R4, 8
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x4C000080, // jirl r0, r4, 0
0x4C002080, // jirl r0, r4, 8
0x4C000020, // jirl r0, r1, 0 (RET)
)
// JAL (R5) links, so the toolchain gives the function its autosize-8
// prologue and epilogue around the call and the closing RET.
fn = firstTextLOONG64(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
JAL (R5)
RET
`)
code = assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x29FFE061, // st.d r1, -8(r3) (prologue saves RA below the new SP)
0x02FFE063, // addi.d r3, r3, -8 (prologue opens the frame)
0x29C00061, // st.d r1, 0(r3) (prologue saves RA at SP)
0x4C0000A1, // jirl r1, r5, 0
0x28C00061, // ld.d r1, 0(r3) (epilogue restores RA)
0x02C02063, // addi.d r3, r3, 8
0x4C000020, // jirl r0, r1, 0 (RET)
)
}
+340
View File
@@ -0,0 +1,340 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
//
// Go's loong64 functions have no frame pointer: FP and SP are synthetic
// registers resolved against the hardware stack pointer (R3) and the frame
// size. The return address lives in R1 (the link register).
//
// The autosize is the real stack adjustment: the declared local frame plus
// the 8 bytes for the saved link register, rounded up to a multiple of 8
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
// A leaf function (no calls) with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0, small), byte-identical to the toolchain:
//
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
// ADDV $-autosize, R3 // open the frame
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
//
// Large frames (autosize past the 12-bit offset or immediate ranges) expand
// the store and the adjust through REGTMP (R30) exactly as the toolchain's
// assembler does: the store via the rounding LU12IW split, the adjust via
// the floor LU12IW/ORI split.
//
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
// restore; the adjust materialised when the immediate does not fit); the
// RET's jirl r0, r1, 0 follows.
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
type loong64FrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR, aligned)
frame int // the declared $framesize
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
// Stack-split guard state: like amd64 and arm64, a leaf function with a
// small autosize is auto-marked NOSPLIT by the toolchain.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// loong64ComputeFrame derives the frame layout for a TEXT function.
func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
fi := loong64FrameInfo{
frame: frameSize(t),
args: argsSize(t),
}
for _, f := range t.Flags {
if f == "NOSPLIT" {
fi.noSplit = true
}
}
fi.leaf = loong64IsLeaf(t)
if fi.frame != 0 {
fi.autosize = fi.frame + 8 // space for the saved LR
if fi.autosize&4 != 0 {
fi.autosize += 4
}
} else if !fi.leaf {
// A zero-frame non-leaf function still opens an 8-byte frame for LR.
fi.autosize = 8
}
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
default:
fi.needSplit = true
switch {
case fi.autosize <= stackSmall:
fi.splitClass = 0
case fi.autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
// loong64GuardLen returns the byte length of the stack-split guard prefix
// (zero when the function needs no guard). The big class materialises two
// constants through R30; each materialisation shrinks by one word when the
// constant's low 12 bits are zero.
func loong64GuardLen(fi loong64FrameInfo) int {
if !fi.needSplit {
return 0
}
off := int64(fi.autosize - stackSmall)
switch fi.splitClass {
case 0:
return 12
case 1:
if off <= 2048 {
return 16 // ADDV $-off fits the signed 12-bit immediate
}
return 24 // MOVV + LU12IW + ORI + ADDV + SGTU + BEQ
default:
// MOVV + [mat] + SGTU + BNE + [mat] + ADDV + SGTU + BEQ
return (6 + loong64MatLen(off) + loong64MatLen(-off)) * 4
}
}
// loong64MatLen reports the word count of materialising v in R30: a value
// with a zero high part needs only the ORI (the toolchain's MOVW $v, R30),
// one with a zero low part only the LU12IW.
func loong64MatLen(v int64) int {
if v>>12 == 0 || v&0xFFF == 0 {
return 1
}
return 2
}
// loong64MatWords appends the words that materialise v in R30, splitting it
// as v>>12 plus the zero-extended low 12 bits.
func loong64MatWords(ws []uint32, v int64) []uint32 {
hi := v >> 12
lo := v & 0xFFF
if hi == 0 {
return append(ws, l64irr(l64OriOp, int(v), 0, 30))
}
ws = append(ws, l64ir(l64Lu12iwOp, int(hi), 30))
if lo != 0 {
ws = append(ws, l64irr(l64OriOp, int(lo), 30, 30))
}
return ws
}
// The LU12IW and ORI opcode bases (2RI20 and 2RI12 formats); the ORI reads
// and writes rd itself.
const (
l64Lu12iwOp = 0x0a << 25
l64OriOp = 0x0e << 22
)
// loong64Imm12 reports whether v fits a signed 12-bit immediate.
func loong64Imm12(v int64) bool { return v >= -2048 && v <= 2047 }
// loong64GuardBytes emits the stack-split guard prefix. blockStart is the
// function-relative address of the morestack call at the end of the function;
// branch displacements are in instructions and are computed from each
// branch's own position.
func loong64GuardBytes(fi loong64FrameInfo, blockStart int) []byte {
// MOVV 16(g), R20 (g.stackguard0), g = R22.
ws := []uint32{l64irr(l64loadStoreTable["MOVV"].ld, 16, 22, 20)}
off := int64(fi.autosize - stackSmall)
// beq appends BEQ R20, blockStart from the branch's own position.
beq := func() {
ws = append(ws, loong64Beqz(20, int32((blockStart-len(ws)*4)>>2)))
}
switch fi.splitClass {
case 0:
// SGTU SP, R20, R20; BEQ R20, more
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 3, 20, 20))
beq()
case 1:
ws = append(ws, loong64MediumWords(off)...)
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
beq()
default:
// SGTU $off, SP, R24 catches the SP underflow a huge frame would
// cause; BNE jumps to morestack in that case.
ws = append(ws, loong64MatWords(nil, off)...)
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 30, 3, 24))
ws = append(ws, loong64Bnez(24, int32((blockStart-len(ws)*4)>>2)))
ws = append(ws, loong64MatWords(nil, -off)...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
beq()
}
return l64WordsLE(ws...)
}
// loong64MediumWords emits the medium-class stack check for offset off: the
// ADDV immediate when it fits, otherwise the same sequence with the constant
// materialised in R30.
func loong64MediumWords(off int64) []uint32 {
if off <= 2048 {
return []uint32{l64irr(l64DualTable["ADDV"].imm, int(-off), 3, 24)}
}
ws := loong64MatWords(nil, -off)
return append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
}
// loong64Beqz/loong64Bnez build the 21-bit conditional branches against R0
// that the toolchain emits for its guard compares.
func loong64Beqz(rj int, dispInstr int32) uint32 {
return l64ir21(l64branch21Table["BEQZ"], int(dispInstr), rj)
}
func loong64Bnez(rj int, dispInstr int32) uint32 {
return l64ir21(l64branch21Table["BNEZ"], int(dispInstr), rj)
}
// loong64MoreStackBlock emits the trailing block: MOVV R1, R31 (save LR, the
// toolchain's OR R1, R0, R31 expansion), BL runtime.morestack_noctxt, B back
// to the function entry.
func loong64MoreStackBlock(blockStart int) ([]byte, Reloc) {
ws := []uint32{
l64rrr(l64DualTable["OR"].rrr, 0, 1, 31), // MOVV R1, R31 (OR R1, R0, R31)
l64bbl(l64jumpTable["BL"], 0), // BL, patched by the linker
}
disp := (-(blockStart + 8)) >> 2
ws = append(ws, l64bbl(l64jumpTable["B"], int(disp)))
reloc := Reloc{
Off: blockStart + 4,
After: blockStart + 8,
Name: "runtime\u00b7morestack_noctxt",
Kind: RelLoong64Branch,
}
return l64WordsLE(ws...), reloc
}
// loong64IsLeaf reports whether a function contains no call instructions
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
// and the epilogue shape.
func loong64IsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "JAL", "CALL", "BL":
return false
}
}
return true
}
// loong64Prologue returns the prologue bytes for a loong64 function. When
// the LR store offset leaves the toolchain's 12-bit store range ([-2046,
// 2045], BIG_12 = 2046) or the SP adjust immediate its 12-bit immediate
// range, each switches to the R30 materialisation the assembler expands it
// to: the store uses the rounding %hi/%lo split (LU12IW of (v+2048)>>12,
// REGTMP += SP, store at the raw offset), the adjust the floor split
// (LU12IW, ORI when the low part is non-zero, REGTMP += SP).
func loong64Prologue(fi loong64FrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
addiD := l64DualTable["ADDV"].imm
var ws []uint32
storeBase := 3
if fi.autosize > 2046 {
// The store goes through REGTMP: LU12IW of the rounding split,
// REGTMP += SP, then the store at REGTMP with the truncated offset.
v := -int64(fi.autosize)
ws = append(ws, l64ir(l64Lu12iwOp, int((v+2048)>>12), 30))
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 3, 30, 30))
storeBase = 30
}
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, storeBase, 1)) // MOVV R1, -autosize(base)
if loong64Imm12(-int64(fi.autosize)) {
ws = append(ws, l64irr(addiD, -fi.autosize, 3, 3)) // ADDV $-autosize, R3
} else {
ws = append(ws, loong64MatWords(nil, -int64(fi.autosize))...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
}
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1)) // MOVV R1, 0(R3)
return l64WordsLE(ws...)
}
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by jirl r0, r1, 0.
func loong64Return(fi loong64FrameInfo) []byte {
var ws []uint32
if fi.autosize != 0 {
if !fi.leaf {
// MOVV 0(R3), R1, restore the link register.
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
}
// ADDV $autosize, R3, close the frame (materialised when the
// immediate does not fit).
if loong64Imm12(int64(fi.autosize)) {
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
} else {
ws = append(ws, loong64MatWords(nil, int64(fi.autosize))...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
}
}
// jirl r0, r1, 0, return.
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
return l64WordsLE(ws...)
}
// loong64StoreWords reports the prologue word count of the LR store, and
// loong64AdjustWords the word count of an SP adjust of v: the immediate
// forms when they fit, otherwise the R30 materialisation sequences.
func loong64StoreWords(autosize int) int {
if autosize > 2046 {
return 3
}
return 1
}
func loong64AdjustWords(v int64) int {
if loong64Imm12(v) {
return 1
}
return loong64MatLen(v) + 1
}
// loong64EpilogueWords reports the epilogue word count the RET expands to.
func loong64EpilogueWords(fi loong64FrameInfo) int {
n := loong64AdjustWords(int64(fi.autosize))
if !fi.leaf {
n++
}
return n
}
// loong64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
func loong64ResolvePseudo(sym *ast.Symbol, fi loong64FrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
switch sym.Pseudo {
case "FP":
return 3, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
return 3, int32(fi.autosize) + int32(sym.Offset)
case "SB":
return -1, int32(sym.Offset)
}
return -1, 0
}
+440
View File
@@ -0,0 +1,440 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestLOONG64_sys exercises the no-operand system instructions and the
// bare-data pseudo-instructions. The words match `go tool asm`
// (GOARCH=loong64) for the same source.
func TestLOONG64_sys(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·sys(SB), NOSPLIT, $0
NOOP
UNDEF
WORD $0x12345678
SYSCALL $0x10
BREAK $0x20
DBAR $1
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x03400000, // andi r0, r0, 0 (NOOP)
0x002A0000, // break 0 (UNDEF)
0x12345678, // WORD
0x002B0010, // syscall 0x10
0x002A0020, // break 0x20
0x38720001, // dbar 1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_branches21 exercises the single-register branch forms: the
// 21-bit BEQZ/BNEZ/BLTZ/BGEZ and the rd-field BGTZ/BLEZ.
func TestLOONG64_branches21(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·b21(SB), NOSPLIT, $0
BEQZ R4, done
BNEZ R5, done
BLTZ R6, done
BGEZ R7, done
BGTZ R8, done
BLEZ R9, done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x40001880, // beqz r4, +6
0x440014A0, // bnez r5, +5
0x600010C0, // bltz r6, +4
0x64000CE0, // bgez r7, +3
0x60000808, // bgtz r8, +2 (register in the rd field)
0x64000409, // blez r9, +1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_fma exercises the four fused multiply-add forms (4 and 3
// operand spellings).
func TestLOONG64_fma(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·fma(SB), NOSPLIT, $0
FMADDD F0, F1, F2, F3
FMSUBD F4, F5, F6
FNMADDD F7, F8, F9, F10
FNMSUBD F11, F12, F13
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x08200443, // fmadd.d f3, f2, f1, f0
0x086214C6, // fmsub.d f6, f5, f5, f4
0x08A3A12A, // fnmadd.d f10, f9, f8, f7
0x08E5B1AD, // fnmsub.d f13, f12, f12, f11
0x4C000020,
)
}
// TestLOONG64_bitops exercises BSTRINS/BSTRPICK (the 6-bit msb/lsb fields)
// and ALSL (the sa−1 shift field).
func TestLOONG64_bitops(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·bits(SB), NOSPLIT, $0
BSTRINSW $3, R4, $0, R5
BSTRINSV $3, R4, $1, R6
BSTRPICKW $3, R4, $0, R5
BSTRPICKV $6, R7, $0, R8
ALSLW $1, R4, R5, R6
ALSLW $4, R7, R8, R9
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x00630085, // bstrins.w r5, r4, $3, $0
0x00830486, // bstrins.d r6, r4, $3, $1
0x00638085, // bstrpick.w r5, r4, $3, $0
0x00C600E8, // bstrpick.d r8, r7, $6, $0
0x00041486, // alsl.w r6, r5, r4, $1 (sa-1)
0x0005A0E9, // alsl.w r9, r8, r7, $4
0x4C000020,
)
}
// TestLOONG64_ptr exercises the 14-bit-offset memory forms (LL/SC/MOVWP/
// MOVVP with the offset scaled by 4) and PRELD.
func TestLOONG64_ptr(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·ptr(SB), NOSPLIT, $0
LLW 8(R14), R15
SCW R16, -4(R17)
MOVWP 16(R18), R19
MOVVP R20, 24(R21)
PRELD 32(R22), $0
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x200009CF, // ll.w r15, 8(r14)
0x21FFFE30, // sc.w r16, -4(r17)
0x24001253, // ldptr.w r19, 16(r18)
0x27001AB4, // stptr.d r20, 24(r21)
0x2AC082C0, // preld 32(r22), 0
0x4C000020,
)
}
// TestLOONG64_atomics exercises the AM* read-modify-write forms and
// RDTIME, plus the MOVV FP→GP move.
func TestLOONG64_atomics(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·atoms(SB), NOSPLIT, $0
AMADDW R4, (R5), R6
RDTIMED R7, R8
MOVV F1, R2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x386110A6, // amadd.w r6, r5, r4
0x000068E8, // rdtime.d r8, r7
0x0114B822, // movfr2gr.d r2, f1
0x4C000020,
)
}
// TestLOONG64_lu52 exercises the LU52I.D immediate form (a gasm extension
// the toolchain reaches only through its MOVV expansion).
func TestLOONG64_lu52(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·lu52(SB), NOSPLIT, $0
LU52ID $0x345, R10
LU52ID $0x123, R11, R12
ADDV16 $0x10000, R13
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x030D154A, // lu52i.d r10, r10, 0x345
0x03048D6C, // lu52i.d r12, r11, 0x123
0x100005AD, // addu16i.d r13, r13, 0x10000>>16
0x4C000020,
)
}
// TestLOONG64_sbRefs checks the static-symbol reference forms through the
// full file assembly: each pcalau12i+addi.d/ld/st pair carries the
// R_LOONG64_ADDR_HI/LO relocation pair, and the immediate fields are left
// zero for the linker.
func TestLOONG64_sbRefs(t *testing.T) {
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0
MOVV $·table(SB), R4
MOVV ·table+8(SB), R5
MOVV R6, ·table(SB)
RET
GLOBL ·table(SB), RODATA, $8
DATA ·table+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 28 {
t.Fatalf("function size = %d, want 28", fn.Size)
}
var hi, lo int
// The three references: $·table (0), ·table+8 (8), ·table (0).
wantAdd := []int64{0, 0, 8, 8, 0, 0}
for i, r := range fn.Relocs {
wantKind := RelLoong64AddrHi
wantOff := (i / 2) * 8
if i%2 == 1 {
wantKind = RelLoong64AddrLo
wantOff += 4
}
if r.Kind != wantKind || r.Off != wantOff || r.Name != "table" || r.Addend != wantAdd[i] {
t.Errorf("reloc %d = {kind %v off %d name %q addend %d}", i, r.Kind, r.Off, r.Name, r.Addend)
}
if r.Kind == RelLoong64AddrHi {
hi++
} else {
lo++
}
}
if hi != 3 || lo != 3 {
t.Errorf("relocs = %d hi + %d lo, want 3 + 3", hi, lo)
}
// The image carries the zero-immediate pair encodings (the linker
// fills the immediate fields from the relocations).
code := img.Code[fn.Offset : fn.Offset+fn.Size]
wantWords(t, code,
0x1A000004, // pcalau12i r4, 0
0x02C00084, // addi.d r4, r4, 0
0x1A00001E, // pcalau12i r30, 0
0x28C003C5, // ld.d r5, 0(r30)
0x1A00001E, // pcalau12i r30, 0
0x29C003C6, // st.d r6, 0(r30)
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
// invalid register operands and operand-count mismatches.
func TestLOONG64_errors(t *testing.T) {
cases := []string{
`TEXT ·e(SB), NOSPLIT, $0
JMP nowhere
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
BEQZ X0, done
done:
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
ADDV R4
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
FMADDD F0, F1
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
AMADDW R4, R5
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
WORD
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
PRELD 32(R4)
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
ALSLW $5, R4, R5, R6
RET
`,
}
for i, src := range cases {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
// the prologue raises the SP delta by autosize (in effect from the third
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
// in MinLC (4) units; byte-identical to `go tool asm`.
func TestLOONG64_pcsp(t *testing.T) {
cases := []struct {
name string
src string
want []byte
}{
{
"leaf",
`#include "textflag.h"
TEXT ·leaf(SB), NOSPLIT, $8-0
MOVV R4, R5
RET
`,
[]byte{0x02, 0x02, 0x20, 0x03, 0x1f, 0x01, 0x00},
},
{
"nonleaf",
`#include "textflag.h"
TEXT ·nonleaf(SB), NOSPLIT, $8-0
MOVV R4, R5
JAL (R12)
RET
`,
[]byte{0x02, 0x02, 0x20, 0x05, 0x1f, 0x01, 0x00},
},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
f, errs := parser.Parse("pcsp_loong64.s", c.src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
if got := pcspTable(img.Funcs[0], 4); !bytes.Equal(got, c.want) {
t.Errorf("pcsp = % x, want % x", got, c.want)
}
})
}
}
// TestLOONG64_sbRefsUndefined checks that a reference to a symbol no GLOBL
// defines assembles into a relocation and is rejected at object emission.
func TestLOONG64_sbRefsUndefined(t *testing.T) {
f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0
MOVV missing(SB), R4
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
if len(img.Funcs[0].Relocs) != 2 {
t.Fatalf("relocs = %d, want the HI/LO pair", len(img.Funcs[0].Relocs))
}
if _, err := img.GOObjectLOONG64("p", "sb_loong64.s"); err == nil {
t.Error("expected an unknown-symbol error at emission")
}
}
// TestLOONG64_movImmToFp checks the immediate-to-FP move: MOVW $c, Fd is the
// only spelling the toolchain accepts, expanding to ori (or addi.w for the
// negative span) into R30 plus movgr2fr.w. The pinned words are the
// toolchain's own bytes; the other widths and out-of-range constants are
// illegal combinations there and are diagnosed here.
func TestLOONG64_movImmToFp(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·fpmov(SB), NOSPLIT, $0
MOVW $0x1, F0
MOVW $0x2, F4
MOVW $-1, F4
RET
`)
code := assembleLOONG64Helper(t, fn)
want := []byte{
0x1e, 0x04, 0x80, 0x03, // ori r30, r0, 1
0xc0, 0xa7, 0x14, 0x01, // movgr2fr.w f0, r30
0x1e, 0x08, 0x80, 0x03, // ori r30, r0, 2
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
0x1e, 0xfc, 0xbf, 0x02, // addi.w r30, r0, -1
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestLOONG64_movImmToFpErrors checks the immediate-to-FP diagnostics: the
// widths the toolchain rejects as illegal combinations, and constants beyond
// the 12-bit ori/addi.w span (the toolchain never materialises a wider
// constant on this path).
func TestLOONG64_movImmToFpErrors(t *testing.T) {
cases := []string{
"MOVV $1, F0",
"MOVF $2, F4",
"MOVD $2, F4",
"MOVW $100000, F1",
"MOVW $-2049, F1",
"MOVW $4096, F1",
}
for _, src := range cases {
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
// TestLOONG64_branch16Unsigned pins the unsigned two-operand branches: with
// one register BLTU/BGEU keep the register-register form against R0 (never
// taken), the toolchain's encoding, where a beqz would test the wrong
// condition; the three-operand forms are unchanged.
func TestLOONG64_branch16Unsigned(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·u(SB), NOSPLIT, $0
BLTU R4, done
BGEU R5, done
BLTU R6, R7, done
BGEU R8, R9, done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x68001080, // bltu r4, r0, +4
0x6C000CA0, // bgeu r5, r0, +3
0x680008C7, // bltu r6, r7, +2
0x6C000509, // bgeu r8, r9, +1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_bitFieldRange checks the BSTRINS/BSTRPICK bit-number
// validation, mirroring the toolchain's "illegal bit number" rule: 0..31 for
// the .w forms, 0..63 for the .d forms, and lsb <= msb.
func TestLOONG64_bitFieldRange(t *testing.T) {
cases := []string{
"BSTRINSW $32, R4, $0, R5",
"BSTRPICKW $31, R4, $32, R5",
"BSTRINSV $64, R4, $0, R5",
"BSTRPICKV $3, R4, $4, R5",
"BSTRINSW $-1, R4, $0, R5",
}
for _, src := range cases {
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
+42
View File
@@ -0,0 +1,42 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
// relocation offsets of a framed loong64 function: the offsets used to
// exclude the prologue, so every relocation landed on a prologue
// instruction in the GOOBJ/ELF output.
func TestLOONG64RelocOffsetsIncludePrologue(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7f(SB), $16-0\n"+
"\tMOVV $gdata(SB), R4\n"+
"\tRET\n"+
"GLOBL gdata(SB), $8\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
fn := img.Funcs[0]
// Layout: 12-byte prologue (autosize 32), pcalau12i+addi.d (12, 16),
// epilogue with RET.
if len(fn.Relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
}
hi, lo := fn.Relocs[0], fn.Relocs[1]
if hi.Kind != RelLoong64AddrHi || hi.Off != 12 || hi.After != 12 {
t.Errorf("hi reloc = {off %d after %d kind %d}, want {off 12 after 12 kind RelLoong64AddrHi}", hi.Off, hi.After, hi.Kind)
}
if lo.Kind != RelLoong64AddrLo || lo.Off != 16 || lo.After != 16 {
t.Errorf("lo reloc = {off %d after %d kind %d}, want {off 16 after 16 kind RelLoong64AddrLo}", lo.Off, lo.After, lo.Kind)
}
}
-258
View File
@@ -1,258 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
)
// This file emits Mach-O x86-64 objects (MH_OBJECT) from an assembled
// Image, in the shape the Darwin assembler produces: one unnamed segment
// carrying a __TEXT,__text and a __DATA,__data section laid out back to
// back at addresses zero and len(code), a symbol table (locals first, then
// exported definitions, then undefined externals) and one relocation entry
// per static-symbol reference, of type X86_64_RELOC_SIGNED.
//
// The image's own address space carries straight over — the data section
// starts immediately after the code, and the layout padding already lives
// inside Image.Data — so every symbol keeps its image address as its
// n_value, and a local (non-external) relocation leaves the displacement
// the assembler resolved in place: the linker only adjusts it by the
// section's final movement.
// Mach-O constants.
const (
machoMagic64 = 0xfeedfacf
machoCPUamd64 = 0x01000007 // CPU_TYPE_X86_64
machoCPUSubAll = 3 // CPU_SUBTYPE_X86_64_ALL
machoObj = 1 // MH_OBJECT
machoSegment64 = 0x19 // LC_SEGMENT_64
machoSymtab = 0x2 // LC_SYMTAB
machoSectTextFlags = 0x80000400 // S_ATTR_PURE_INSTRUCTIONS | S_ATTR_SOME_INSTRUCTIONS
nUndf = 0x00 // undefined symbol
nSect = 0x0e // defined in section number n_sect
nExt = 0x01 // external (exported or undefined-global) bit
x8664RelocSigned = 1
)
// MachOObject returns the image as a Mach-O x86-64 relocatable object
// (MH_OBJECT), the shape the Darwin toolchain links. Symbol names follow
// the same rules as the ELF output. Every static-symbol reference becomes
// an X86_64_RELOC_SIGNED relocation: external references against their
// undefined symbol, file-local ones against the __DATA section with the
// resolved displacement carried in the instruction bytes.
func (img *Image) MachOObject() ([]byte, error) {
le := binary.LittleEndian
// Section ordinals (1-based, as Mach-O numbers them).
const (
sectText = 1
sectData = 2
)
// Object address space: code at 0, data immediately after (the layout
// padding is already part of img.Data, so image addresses are object
// addresses).
textAddr := uint64(0)
dataAddr := uint64(len(img.Code))
vmsize := dataAddr + uint64(len(img.Data))
// The code, with external displacements primed to addend − 4: the
// linker adds the symbol's address to the field as it stands. Local
// displacements stay as the assembler resolved them.
code := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.External {
// Prime the field to the addend measured from the patch
// site: the assembler records it from the instruction end,
// After − Off bytes past the field.
copy(code[fn.Offset+r.Off:], le32(r.Addend-int64(r.After-r.Off)))
}
}
}
// Symbols: locals first, then exported definitions, then undefined
// externals — the order the classic link editor expects.
type machoSym struct {
name string
typ byte
sect byte
value uint64
}
var locals, globals, undefs []machoSym
for _, fn := range img.Funcs {
s := machoSym{name: objectName(fn.Pkg, fn.Name), typ: nSect, sect: sectText, value: textAddr + uint64(fn.Offset)}
if fn.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, d := range img.DataSyms {
s := machoSym{name: objectName(d.Pkg, d.Name), typ: nSect, sect: sectData, value: dataAddr + uint64(d.Offset)}
if d.Static {
locals = append(locals, s)
} else {
s.typ |= nExt
globals = append(globals, s)
}
}
for _, name := range img.Externals {
undefs = append(undefs, machoSym{name: name, typ: nUndf | nExt})
}
syms := append(append(locals, globals...), undefs...)
symIdx := map[string]int{}
for i, s := range syms {
symIdx[s.name] = i
}
// Relocations, attached to the __text section.
type machoReloc struct {
addr uint32
symnum uint32
extern bool
}
var relocs []machoReloc
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
rel := machoReloc{addr: uint32(fn.Offset + r.Off)}
if r.External {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
rel.symnum = uint32(idx)
rel.extern = true
} else {
// Section-relative: r_symbolnum carries the section number
// and the resolved displacement stays in the bytes.
rel.symnum = sectData
}
relocs = append(relocs, rel)
}
}
// The string table opens with the conventional " \0".
strtab := []byte{' ', 0}
strOff := map[string]int{}
for _, s := range syms {
if _, ok := strOff[s.name]; ok {
continue
}
strOff[s.name] = len(strtab)
strtab = append(strtab, s.name...)
strtab = append(strtab, 0)
}
// File layout: header, the two load commands, section data (code,
// data), the relocation table, the symbol table, the string table.
const (
hdrSize = 32
segCmdSize = 72 + 2*80 // segment command with two sections
symCmdSize = 24
)
sizeofcmds := segCmdSize + symCmdSize
dataOff := hdrSize + sizeofcmds
reloff := dataOff + len(code) + len(img.Data)
symoff := reloff + 8*len(relocs)
stroff := symoff + 16*len(syms)
out := make([]byte, stroff+len(strtab))
// mach_header_64.
le.PutUint32(out[0:], machoMagic64)
le.PutUint32(out[4:], machoCPUamd64)
le.PutUint32(out[8:], machoCPUSubAll)
le.PutUint32(out[12:], machoObj)
le.PutUint32(out[16:], 2) // ncmds
le.PutUint32(out[20:], uint32(sizeofcmds))
le.PutUint32(out[24:], 0) // flags
le.PutUint32(out[28:], 0) // reserved
// LC_SEGMENT_64 with the two sections.
p := hdrSize
le.PutUint32(out[p:], machoSegment64)
le.PutUint32(out[p+4:], segCmdSize)
// segname: the empty string, zero-padded to 16 bytes.
le.PutUint64(out[p+8:], 0)
le.PutUint64(out[p+16:], 0)
le.PutUint64(out[p+24:], 0) // vmaddr
le.PutUint64(out[p+32:], vmsize)
le.PutUint64(out[p+40:], uint64(dataOff))
le.PutUint64(out[p+48:], vmsize)
le.PutUint32(out[p+56:], 7) // maxprot rwx
le.PutUint32(out[p+60:], 7) // initprot rwx
le.PutUint32(out[p+64:], 2) // nsects
le.PutUint32(out[p+68:], 0) // flags
// __TEXT,__text
s := p + 72
copy(out[s:], "__text")
copy(out[s+16:], "__TEXT")
le.PutUint64(out[s+32:], textAddr)
le.PutUint64(out[s+40:], uint64(len(code)))
le.PutUint32(out[s+48:], uint32(dataOff))
le.PutUint32(out[s+52:], 4) // align 2^4
le.PutUint32(out[s+56:], uint32(reloff))
le.PutUint32(out[s+60:], uint32(len(relocs)))
le.PutUint32(out[s+64:], machoSectTextFlags)
// __DATA,__data
s += 80
copy(out[s:], "__data")
copy(out[s+16:], "__DATA")
le.PutUint64(out[s+32:], dataAddr)
le.PutUint64(out[s+40:], uint64(len(img.Data)))
le.PutUint32(out[s+48:], uint32(dataOff+len(code)))
le.PutUint32(out[s+52:], 4) // align 2^4
// LC_SYMTAB.
p = hdrSize + segCmdSize
le.PutUint32(out[p:], machoSymtab)
le.PutUint32(out[p+4:], symCmdSize)
le.PutUint32(out[p+8:], uint32(symoff))
le.PutUint32(out[p+12:], uint32(len(syms)))
le.PutUint32(out[p+16:], uint32(stroff))
le.PutUint32(out[p+20:], uint32(len(strtab)))
// Section data.
copy(out[dataOff:], code)
copy(out[dataOff+len(code):], img.Data)
// Relocation entries.
for i, r := range relocs {
e := out[reloff+i*8:]
le.PutUint32(e[0:], r.addr)
bits := r.symnum & 0x00ffffff
bits |= 1 << 24 // r_pcrel
bits |= 2 << 25 // r_length = 4 bytes
if r.extern {
bits |= 1 << 27 // r_extern
}
bits |= x8664RelocSigned << 28
le.PutUint32(e[4:], bits)
}
// nlist_64 entries.
for i, s := range syms {
e := out[symoff+i*16:]
le.PutUint32(e[0:], uint32(strOff[s.name]))
e[4] = s.typ
e[5] = s.sect
le.PutUint16(e[6:], 0) // n_desc
le.PutUint64(e[8:], s.value)
}
// String table.
copy(out[stroff:], strtab)
return out, nil
}
-127
View File
@@ -1,127 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/macho"
"encoding/binary"
"testing"
)
// TestMachOObject checks the structure of the emitted MH_OBJECT: the two
// sections and their addresses, the symbol table (types, sections, values)
// and the __text relocation entries, parsed back with debug/macho. No
// Darwin toolchain is available on the test hosts, so the check is
// structural — the ELF output carries the end-to-end link-and-run proof of
// the shared symbol and relocation model.
func TestMachOObject(t *testing.T) {
img := elfTestImage(t)
obj, err := img.MachOObject()
if err != nil {
t.Fatalf("MachOObject: %v", err)
}
f, err := macho.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer f.Close()
if f.Type != macho.TypeObj {
t.Errorf("file type = %v, want MH_OBJECT", f.Type)
}
if f.Cpu != macho.CpuAmd64 {
t.Errorf("cpu = %v, want CpuAmd64", f.Cpu)
}
text := f.Section("__text")
data := f.Section("__data")
if text == nil || data == nil {
t.Fatal("missing __text or __data section")
}
if text.Addr != 0 || text.Size != uint64(len(img.Code)) {
t.Errorf("__text addr/size = %#x/%d, want 0/%d", text.Addr, text.Size, len(img.Code))
}
if data.Addr != uint64(len(img.Code)) {
t.Errorf("__data addr = %#x, want %#x", data.Addr, len(img.Code))
}
// Symbol table: locals, exported definitions, undefined externals.
syms := f.Symtab.Syms
byName := map[string]macho.Symbol{}
for _, s := range syms {
byName[s.Name] = s
}
wantSym := func(name string, typ, sect uint8, value uint64) {
t.Helper()
s, ok := byName[name]
if !ok {
t.Errorf("symbol %q not found", name)
return
}
if s.Type != typ || s.Sect != sect || s.Value != value {
t.Errorf("%s: type/sect/value = %#x/%d/%#x, want %#x/%d/%#x",
name, s.Type, s.Sect, s.Value, typ, sect, value)
}
}
const (
defined = nSect | nExt
local = nSect
undefined = nUndf | nExt
)
wantSym("addq", defined, 1, 0)
wantSym("getanswer", defined, 1, 5)
wantSym("useextern", defined, 1, 13)
answer := byName["answer"]
if answer.Type != local || answer.Sect != 2 {
t.Errorf("answer: type/sect = %#x/%d, want %#x/2", answer.Type, answer.Sect, local)
}
wantSym("extvar", undefined, 0, 0)
// Relocations: both X86_64_RELOC_SIGNED, PC-relative, 4 bytes wide.
// The local one carries its section number in Value, the external one
// its symbol number.
if len(text.Relocs) != 2 {
t.Fatalf("__text relocs = %d, want 2", len(text.Relocs))
}
var sawLocal, sawExternal bool
for _, r := range text.Relocs {
if !r.Pcrel || r.Len != 2 || r.Type != x8664RelocSigned {
t.Errorf("reloc at %#x: pcrel/len/type = %v/%d/%d", r.Addr, r.Pcrel, r.Len, r.Type)
}
switch {
case r.Extern:
if name := syms[r.Value].Name; name != "extvar" {
t.Errorf("external reloc at %#x names %q, want extvar", r.Addr, name)
}
sawExternal = true
default:
if r.Value != 2 { // __data, the second section
t.Errorf("local reloc at %#x: section %d, want 2 (__data)", r.Addr, r.Value)
}
sawLocal = true
}
}
if !sawLocal || !sawExternal {
t.Errorf("relocs seen: local=%v external=%v, want both", sawLocal, sawExternal)
}
// The __text bytes are the image code, with the external displacement
// primed to addend − 4 and the local one left resolved.
textData, err := text.Data()
if err != nil {
t.Fatal(err)
}
want := append([]byte(nil), img.Code...)
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.Name == "extvar" {
binary.LittleEndian.PutUint32(want[fn.Offset+r.Off:], 0xfffffffc) // −4
}
}
}
if !bytes.Equal(textData, want) {
t.Errorf("__text bytes %x, want %x", textData, want)
}
}
-5
View File
@@ -37,11 +37,6 @@ func Idx(base, index Reg, scale int, disp int64, size int) Mem {
return Mem{Base: base, Index: index, Scale: scale, Disp: disp, Size: size, HasBase: true, HasIndex: true}
}
// Rip builds a RIP-relative memory operand (RIP)+disp.
func Rip(disp int64, size int) Mem {
return Mem{Disp: disp, Size: size}
}
// sbMem is a memory operand that references a static (SB) symbol. It encodes
// as a RIP-relative reference with a placeholder displacement; the encoder
// records a patch site so the file-level layout can fill in the true rel32
+31 -31
View File
@@ -7,36 +7,38 @@
// by round-tripping through golang.org/x/arch's decoder in the tests.
package asm
import "maps"
import "strings"
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
// are size-agnostic — the instruction suffix (MOVQ vs MOVL) fixes the width —
// are size-agnostic, the instruction suffix (MOVQ vs MOVL) fixes the width
// so the encoder keys off the register's index and lets the mnemonic supply the
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0–K7.
// registers K0-K7.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0–K7 opmask register
mask bool // K0-K7 opmask register
}
// Index returns the register number (0–15 for GPRs, 0–31 for vectors).
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
func (r Reg) Index() int { return r.idx }
// Size returns the width in bytes implied by the register's name.
func (r Reg) Size() int { return r.size }
// IsMask reports whether r is an AVX-512 opmask register (K0–K7).
// IsMask reports whether r is an AVX-512 opmask register (K0-K7).
func (r Reg) IsMask() bool { return r.mask }
func (r Reg) isOperand() {}
// needsREX reports whether this register forces a REX prefix at the given
// operand size: the extended registers R8–R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4–7, not high) do as well.
// operand size: the extended registers R8-R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4-7, not high) do as well.
func (r Reg) needsREX(opSize int) bool {
if r.idx >= 8 {
return true
@@ -63,28 +65,28 @@ var (
CX = Reg{idx: 1, size: 2}
DX = Reg{idx: 2, size: 2}
BX = Reg{idx: 3, size: 2}
SP = Reg{idx: 4, size: 2}
BP = Reg{idx: 5, size: 2}
_ = Reg{idx: 4, size: 2}
_ = Reg{idx: 5, size: 2}
SI = Reg{idx: 6, size: 2}
DI = Reg{idx: 7, size: 2}
EAX = Reg{idx: 0, size: 4}
ECX = Reg{idx: 1, size: 4}
EDX = Reg{idx: 2, size: 4}
EBX = Reg{idx: 3, size: 4}
ESP = Reg{idx: 4, size: 4}
EBP = Reg{idx: 5, size: 4}
ESI = Reg{idx: 6, size: 4}
EDI = Reg{idx: 7, size: 4}
_ = Reg{idx: 0, size: 4}
_ = Reg{idx: 1, size: 4}
_ = Reg{idx: 2, size: 4}
_ = Reg{idx: 3, size: 4}
_ = Reg{idx: 4, size: 4}
_ = Reg{idx: 5, size: 4}
_ = Reg{idx: 6, size: 4}
_ = Reg{idx: 7, size: 4}
RAX = Reg{idx: 0, size: 8}
RCX = Reg{idx: 1, size: 8}
RDX = Reg{idx: 2, size: 8}
RBX = Reg{idx: 3, size: 8}
RSP = Reg{idx: 4, size: 8}
RBP = Reg{idx: 5, size: 8}
RSI = Reg{idx: 6, size: 8}
RDI = Reg{idx: 7, size: 8}
_ = Reg{idx: 0, size: 8}
_ = Reg{idx: 1, size: 8}
_ = Reg{idx: 2, size: 8}
_ = Reg{idx: 3, size: 8}
_ = Reg{idx: 4, size: 8}
_ = Reg{idx: 5, size: 8}
_ = Reg{idx: 6, size: 8}
_ = Reg{idx: 7, size: 8}
)
// regByName maps an assembly register name (case-insensitive) to a Reg.
@@ -121,19 +123,17 @@ func buildRegByName() map[string]Reg {
}
// 8-bit: AL..BH, SPL..DIL, R8B..R15B.
for n, r := range map[string]Reg{
maps.Copy(m, map[string]Reg{
"AL": AL, "CL": CL, "DL": DL, "BL": BL,
"AH": AH, "CH": CH, "DH": DH, "BH": BH,
"SPL": SPL, "BPL": BPL, "SIL": SIL, "DIL": DIL,
} {
m[n] = r
}
})
for i := 8; i <= 15; i++ {
m["R"+itoa(i)+"B"] = Reg{idx: i, size: 1}
}
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX
// Z0..Z31 (512-bit, size 64). Indices 16-31 are only encodable in EVEX
// (AVX-512) instructions; the encoder validates that through its tables.
for i := 0; i <= 31; i++ {
m["X"+itoa(i)] = Reg{idx: i, size: 16}
File diff suppressed because it is too large Load Diff
+568
View File
@@ -0,0 +1,568 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// RISC-V register encoding: maps register names to their 5-bit numbers.
// The Go assembler uses the standard RISC-V ABI naming.
// riscvRegNum returns the 5-bit register number for a RISC-V register name.
// Returns -1 if the register is not recognized.
func riscvRegNum(name string) int {
switch name {
// Numbered integer registers.
case "X0", "ZERO":
return 0
case "X1", "RA", "LR":
return 1
case "X2", "SP":
return 2
case "X3", "GP":
return 3
case "X4", "TP":
return 4
case "X5", "T0":
return 5
case "X6", "T1":
return 6
case "X7", "T2":
return 7
case "X8", "S0", "FP":
return 8
case "X9", "S1":
return 9
case "X10", "A0":
return 10
case "X11", "A1":
return 11
case "X12", "A2":
return 12
case "X13", "A3":
return 13
case "X14", "A4":
return 14
case "X15", "A5":
return 15
case "X16", "A6":
return 16
case "X17", "A7":
return 17
case "X18", "S2":
return 18
case "X19", "S3":
return 19
case "X20", "S4":
return 20
case "X21", "S5":
return 21
case "X22", "S6":
return 22
case "X23", "S7":
return 23
case "X24", "S8":
return 24
case "X25", "S9":
return 25
case "X26", "S10":
return 26
case "X27", "S11", "g":
return 27
case "X28", "T3":
return 28
case "X29", "T4":
return 29
case "X30", "T5":
return 30
case "X31", "T6", "TMP":
return 31
// Floating-point registers (F0-F31).
case "F0", "FT0":
return 0
case "F1", "FT1":
return 1
case "F2", "FT2":
return 2
case "F3", "FT3":
return 3
case "F4", "FT4":
return 4
case "F5", "FT5":
return 5
case "F6", "FT6":
return 6
case "F7", "FT7":
return 7
case "F8", "FS0":
return 8
case "F9", "FS1":
return 9
case "F10", "FA0":
return 10
case "F11", "FA1":
return 11
case "F12", "FA2":
return 12
case "F13", "FA3":
return 13
case "F14", "FA4":
return 14
case "F15", "FA5":
return 15
case "F16", "FA6":
return 16
case "F17", "FA7":
return 17
case "F18", "FS2":
return 18
case "F19", "FS3":
return 19
case "F20", "FS4":
return 20
case "F21", "FS5":
return 21
case "F22", "FS6":
return 22
case "F23", "FS7":
return 23
case "F24", "FS8":
return 24
case "F25", "FS9":
return 25
case "F26", "FS10":
return 26
case "F27", "FS11":
return 27
case "F28", "FT8":
return 28
case "F29", "FT9":
return 29
case "F30", "FT10":
return 30
case "F31", "FT11":
return 31
default:
return -1
}
}
// RISC-V instruction encoding parameters.
type riscvEnc struct {
opcode uint32 // bits [6:0]
funct3 uint32 // bits [14:12]
funct7 uint32 // bits [31:25]
}
// riscvInstrTable maps RISC-V mnemonics to their encoding.
var riscvInstrTable = map[string]riscvEnc{
// RV64I, R-type arithmetic/logic.
"ADD": {0x33, 0x0, 0x00},
"SUB": {0x33, 0x0, 0x20},
"SLL": {0x33, 0x1, 0x00},
"SLT": {0x33, 0x2, 0x00},
"SLTU": {0x33, 0x3, 0x00},
"XOR": {0x33, 0x4, 0x00},
"SRL": {0x33, 0x5, 0x00},
"SRA": {0x33, 0x5, 0x20},
"OR": {0x33, 0x6, 0x00},
"AND": {0x33, 0x7, 0x00},
// RV64I, 32-bit variants (W suffix).
"ADDW": {0x3B, 0x0, 0x00},
"SUBW": {0x3B, 0x0, 0x20},
"SLLW": {0x3B, 0x1, 0x00},
"SRLW": {0x3B, 0x5, 0x00},
"SRAW": {0x3B, 0x5, 0x20},
// RV64I, I-type shift-immediate (shamt in rs2 field).
"SLLI": {0x13, 0x1, 0x00},
"SRLI": {0x13, 0x5, 0x00},
"SRAI": {0x13, 0x5, 0x20},
"SLLIW": {0x1B, 0x1, 0x00},
"SRLIW": {0x1B, 0x5, 0x00},
"SRAIW": {0x1B, 0x5, 0x20},
// RV64M, multiply/divide.
"MUL": {0x33, 0x0, 0x01},
"MULH": {0x33, 0x1, 0x01},
"MULHSU": {0x33, 0x2, 0x01},
"MULHU": {0x33, 0x3, 0x01},
"DIV": {0x33, 0x4, 0x01},
"DIVU": {0x33, 0x5, 0x01},
"REM": {0x33, 0x6, 0x01},
"REMU": {0x33, 0x7, 0x01},
// RV64M, 32-bit variants.
"MULW": {0x3B, 0x0, 0x01},
"DIVW": {0x3B, 0x4, 0x01},
"DIVUW": {0x3B, 0x5, 0x01},
"REMW": {0x3B, 0x6, 0x01},
"REMUW": {0x3B, 0x7, 0x01},
// RV64I, I-type arithmetic.
"ADDI": {0x13, 0x0, 0x00},
"ADDIW": {0x1B, 0x0, 0x00},
"SLTI": {0x13, 0x2, 0x00},
"SLTIU": {0x13, 0x3, 0x00},
"XORI": {0x13, 0x4, 0x00},
"ORI": {0x13, 0x6, 0x00},
"ANDI": {0x13, 0x7, 0x00},
// Loads (I-type).
"LB": {0x03, 0x0, 0x00},
"LH": {0x03, 0x1, 0x00},
"LW": {0x03, 0x2, 0x00},
"LD": {0x03, 0x3, 0x00},
"LBU": {0x03, 0x4, 0x00},
"LHU": {0x03, 0x5, 0x00},
"LWU": {0x03, 0x6, 0x00},
// Stores (S-type).
"SB": {0x23, 0x0, 0x00},
"SH": {0x23, 0x1, 0x00},
"SW": {0x23, 0x2, 0x00},
"SD": {0x23, 0x3, 0x00},
// Branches (B-type).
"BEQ": {0x63, 0x0, 0x00},
"BNE": {0x63, 0x1, 0x00},
"BLT": {0x63, 0x4, 0x00},
"BGE": {0x63, 0x5, 0x00},
"BLTU": {0x63, 0x6, 0x00},
"BGEU": {0x63, 0x7, 0x00},
// U-type.
"LUI": {0x37, 0x0, 0x00},
"AUIPC": {0x17, 0x0, 0x00},
// System.
"ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00},
// JALR, indirect jump/call (I-type).
"JALR": {0x67, 0x0, 0x00},
// RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
"AMOORW": {0x2F, 0x2, 0x06 << 2},
"AMOORD": {0x2F, 0x3, 0x06 << 2},
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
// RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00},
"FSUBS": {0x53, 0x0, 0x04},
"FMULS": {0x53, 0x0, 0x08},
"FDIVS": {0x53, 0x0, 0x0C},
"FADDD": {0x53, 0x0, 0x01},
"FSUBD": {0x53, 0x0, 0x05},
"FMULD": {0x53, 0x0, 0x09},
"FDIVD": {0x53, 0x0, 0x0D},
"FSQRTS": {0x53, 0x0, 0x2C},
"FSQRTD": {0x53, 0x0, 0x2D},
// FP loads/stores.
"FLW": {0x07, 0x2, 0x00},
"FLD": {0x07, 0x3, 0x00},
"FSW": {0x27, 0x2, 0x00},
"FSD": {0x27, 0x3, 0x00},
// FP min/max.
"FMINS": {0x53, 0x0, 0x14},
"FMAXS": {0x53, 0x1, 0x14},
"FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2},
"LRD": {0x2F, 0x3, 0x02 << 2},
"SCW": {0x2F, 0x2, 0x03 << 2},
"SCD": {0x2F, 0x3, 0x03 << 2},
// FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50},
"FLTS": {0x53, 0x1, 0x50},
"FLES": {0x53, 0x0, 0x50},
"FEQD": {0x53, 0x2, 0x51},
"FLTD": {0x53, 0x1, 0x51},
"FLED": {0x53, 0x0, 0x51},
}
// riscvRType encodes an R-type instruction: funct7 | rs2 | rs1 | funct3 | rd | opcode.
func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
}
// riscvAMOType encodes an atomic (AMO) instruction.
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
}
// FP conversion instructions (FCVT, FMV). These use the rs2 field to
// encode the conversion type rather than a register, so they are handled
// separately from the general instruction table.
type riscvCvtEnc struct {
funct7 uint32 // bits [31:25]
rs2 uint32 // conversion-type code in bits [24:20]
opcode uint32 // always 0x53 (OP-FP)
}
var riscvCvtTable = map[string]riscvCvtEnc{
// float → int (rs2 selects the integer width/sign).
"FCVTWS": {0x60, 0x0, 0x53}, // float32 → int32
"FCVTWUS": {0x60, 0x1, 0x53}, // float32 → uint32
"FCVTLS": {0x60, 0x2, 0x53}, // float32 → int64
"FCVTLUS": {0x60, 0x3, 0x53}, // float32 → uint64
"FCVTWD": {0x61, 0x0, 0x53}, // float64 → int32
"FCVTWUD": {0x61, 0x1, 0x53}, // float64 → uint32
"FCVTLD": {0x61, 0x2, 0x53}, // float64 → int64
"FCVTLUD": {0x61, 0x3, 0x53}, // float64 → uint64
// int → float (rs2 selects the integer width/sign).
"FCVTSW": {0x68, 0x0, 0x53}, // int32 → float32
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
"FCVTDLU": {0x69, 0x3, 0x53}, // uint64 → float64
// float → float width conversion.
"FCVTSD": {0x20, 0x1, 0x53}, // float64 → float32
"FCVTDS": {0x21, 0x0, 0x53}, // float32 → float64
// Bit moves between integer and FP registers (no conversion).
"FMVXD": {0x71, 0x0, 0x53}, // float64 → int64 (bit move)
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
}
// riscvCvtType encodes an FP conversion instruction.
// Layout: funct7 | rs2(convtype) | rs1 | funct3(0) | rd | opcode.
func riscvCvtType(enc riscvCvtEnc, rd, rs1 int) uint32 {
return (enc.funct7 << 25) | (enc.rs2 << 20) | (uint32(rs1) << 15) |
(uint32(rd) << 7) | enc.opcode
}
// R4-type fused multiply-add instructions (FMADD/FMSUB/FNMSUB/FNMADD).
// These take 4 register operands: rs1, rs2, rs3, rd.
// Layout: rs3 | fmt | rs2 | rs1 | rm | rd | opcode.
type riscvFmaEnc struct {
fmt uint32 // bits [26:25]: 0x0 = single, 0x1 = double
opcode uint32 // bits [6:0]
}
var riscvFmaTable = map[string]riscvFmaEnc{
"FMADDS": {0x0, 0x43}, // rd = rs1*rs2 + rs3
"FMADDD": {0x1, 0x43},
"FMSUBS": {0x0, 0x47}, // rd = rs1*rs2 - rs3
"FMSUBD": {0x1, 0x47},
"FNMSUBS": {0x0, 0x4B}, // rd = -(rs1*rs2) + rs3
"FNMSUBD": {0x1, 0x4B},
"FNMADDS": {0x0, 0x4F}, // rd = -(rs1*rs2) - rs3
"FNMADDD": {0x1, 0x4F},
}
// riscvFmaType encodes an R4-type fused multiply-add instruction.
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
(uint32(rs1) << 15) | (0x0 << 12) /* rm=RNE */ | (uint32(rd) << 7) | enc.opcode
}
// CSR (Control and Status Register) instructions.
// Format: csr[11:0] | rs1/zimm | funct3 | rd | opcode (0x73).
type riscvCsrEnc struct {
funct3 uint32 // bits [14:12]
imm bool // true for CSRRWI/CSRRSI/CSRRCI (5-bit uimm variant)
}
var riscvCsrTable = map[string]riscvCsrEnc{
"CSRRW": {0x1, false}, // rd=CSR, CSR=rs1
"CSRRS": {0x2, false}, // rd=CSR, CSR |= rs1
"CSRRC": {0x3, false}, // rd=CSR, CSR &= ~rs1
"CSRRWI": {0x5, true}, // rd=CSR, CSR=uimm
"CSRRSI": {0x6, true}, // rd=CSR, CSR |= uimm
"CSRRCI": {0x7, true}, // rd=CSR, CSR &= ~uimm
}
// riscvCsrType encodes a CSR instruction.
// csr is the 12-bit CSR address; src is either a register number or a 5-bit
// unsigned immediate (depending on enc.imm).
func riscvCsrType(enc riscvCsrEnc, rd, src int, csr int32) uint32 {
return (uint32(csr&0xFFF) << 20) | (uint32(src&0x1F) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | 0x73
}
// riscvIType encodes an I-type instruction: imm[11:0] | rs1 | funct3 | rd | opcode.
func riscvIType(enc riscvEnc, rd, rs1 int, imm int32) uint32 {
return (uint32(imm&0xFFF) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
}
// riscvSType encodes an S-type instruction: imm[11:5] | rs2 | rs1 | funct3 | imm[4:0] | opcode.
func riscvSType(enc riscvEnc, rs1, rs2 int, imm int32) uint32 {
immU := uint32(imm) & 0xFFF
return ((immU >> 5) << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | ((immU & 0x1F) << 7) | enc.opcode
}
// riscvBType encodes a B-type instruction (branches).
func riscvBType(enc riscvEnc, rs1, rs2 int, offset int32) uint32 {
imm := uint32(offset) & 0x1FFE // bits [12:1], bit 0 is always 0
return (((imm >> 12) & 1) << 31) | // imm[12]
(((imm >> 5) & 0x3F) << 25) | // imm[10:5]
(uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) |
(((imm >> 1) & 0xF) << 8) | // imm[4:1]
(((imm >> 11) & 1) << 7) | // imm[11]
enc.opcode
}
// riscvUType encodes a U-type instruction: imm[31:12] | rd | opcode.
func riscvUType(enc riscvEnc, rd int, imm int32) uint32 {
return (uint32(imm) & 0xFFFFF000) | (uint32(rd) << 7) | enc.opcode
}
// riscvJType encodes a J-type instruction (JAL).
func riscvJType(rd int, offset int32) uint32 {
imm := uint32(offset) & 0x1FFFFE // bits [20:1]
return (((imm >> 20) & 1) << 31) | // imm[20]
(((imm >> 1) & 0x3FF) << 21) | // imm[10:1]
(((imm >> 11) & 1) << 20) | // imm[11]
(((imm >> 12) & 0xFF) << 12) | // imm[19:12]
(uint32(rd) << 7) |
0x6F // JAL opcode
}
// ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
// prime register field used by compressed instructions (x8-x15).
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
// rvcReg3 returns the 3-bit encoding for registers x8-x15 (0-7).
func rvcReg3(r int) uint32 { return uint32(r - 8) }
// rvcCR encodes a CR-type (register) compressed instruction.
// Format: funct4 | rd/rs1 | rs2 | op=2.
func rvcCR(funct4, rd, rs2 uint32) uint16 {
return uint16((funct4 << 12) | (rd << 7) | (rs2 << 2) | 0x2)
}
// rvcCI encodes a CI-type (immediate) compressed instruction.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW, linear 6-bit immediate.
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
}
// rvcSLLI encodes C.SLLI, which shares funct3=0 with C.ADDI but lives in the
// op=10 quadrant (unlike C.ADDI's op=01).
func rvcSLLI(rd, shamt uint32) uint16 {
return uint16(((shamt>>5)&1)<<12 | (rd << 7) | (shamt&0x1F)<<2 | 0x2)
}
// encodeRVCPattern extracts the bits listed in pattern (MSB first) from imm
// into a packed value, matching cmd/internal/obj/riscv's encodeBitPattern.
func encodeRVCPattern(imm uint32, pattern []int) uint32 {
packed := uint32(0)
for _, bit := range pattern {
packed = packed<<1 | (imm>>bit)&1
}
return packed
}
// rvcLSP encodes a stack-relative compressed load (op=10 quadrant): C.LWSP
// (funct3=2, 4-byte scale), C.LDSP (funct3=3) or C.FLDSP (funct3=1, 8-byte
// scale). offset is the full byte offset.
func rvcLSP(funct3, rd uint32, offset uint32) uint16 {
pattern := []int{5, 4, 3, 8, 7, 6}
if funct3 == 0x2 {
pattern = []int{5, 4, 3, 2, 7, 6}
}
packed := uint32(0)
for i, b := range pattern {
packed |= ((offset >> b) & 1) << (5 - i)
}
return uint16((funct3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x2)
}
// rvcSSP encodes a stack-relative compressed store (op=10 quadrant): C.SWSP
// (funct3=6, 4-byte scale), C.SDSP (funct3=7) or C.FSDSP (funct3=5, 8-byte
// scale). offset is the full byte offset.
func rvcSSP(funct3, rs2 uint32, offset uint32) uint16 {
pattern := []int{5, 4, 3, 8, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 4, 3, 2, 7, 6}
}
packed := uint32(0)
for i, b := range pattern {
packed |= ((offset >> b) & 1) << (5 - i)
}
return uint16((funct3 << 13) | (packed << 7) | (rs2 << 2) | 0x2)
}
// rvcCL encodes a register-relative compressed load (op=00 quadrant): C.LW
// (funct3=2), C.LD (funct3=3) or C.FLD (funct3=1). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CL format.
func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
pattern := []int{5, 4, 3, 7, 6}
if funct3 == 0x2 {
pattern = []int{5, 4, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rd << 2))
}
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CS format, with the
// same five-bit patterns as the load side ({5,4,3,7,6} and {5,4,3,2,6},
// matching the toolchain's encodeCS).
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
pattern := []int{5, 4, 3, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 4, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
}
// rvcCIW encodes a CIW-type compressed immediate wide instruction: C.ADDI4SPN
// (funct3=0). imm is the raw byte offset.
func rvcCIW(funct3, rd uint32, imm uint32) uint16 {
packed := encodeRVCPattern(imm, []int{5, 4, 9, 8, 7, 6, 2, 3})
return uint16((funct3 << 13) | (packed << 5) | (rd << 2))
}
// rvcCA encodes a CA-type (arithmetic) compressed instruction.
// Format: funct6[15:10] | rd'/rs1'[9:7] | funct2[6:5] | rs2'[4:2] | op=01.
func rvcCA(funct6, funct2, rd, rs2 uint32) uint16 {
return uint16((funct6 << 10) | (rd << 7) | (funct2 << 5) | (rs2 << 2) | 0x1)
}
// rvcCBShift encodes a CB-type shift/immediate compressed instruction
// (C.SRLI, C.SRAI, C.ANDI). rd is the 3-bit prime-register index; imm is
// the 6-bit shamt/immediate; funct2 selects the operation (0=SRLI, 1=SRAI,
// 2=ANDI).
func rvcCBShift(funct2, rd, imm uint32) uint16 {
return uint16((0x4 << 13) | ((imm>>5)&1)<<12 | (funct2 << 10) | (rd << 7) | (imm&0x1F)<<2 | 0x1)
}
// rvcADDI16SP encodes C.ADDI16SP: ADDI rd, imm, rd for the stack pointer
// with a 10-bit signed, 16-byte-scaled immediate. imm is the raw byte
// offset; the immediate bits are extracted in the order [9|4|6|8:7|5].
func rvcADDI16SP(rd uint32, imm int32) uint16 {
u := uint32(imm)
packed := uint32(0)
for _, bit := range []uint{9, 4, 6, 8, 7, 5} {
packed = packed<<1 | (u>>bit)&1
}
return uint16((0x3 << 13) | ((packed>>5)&1)<<12 | (rd << 7) | (packed&0x1F)<<2 | 0x1)
}
+973
View File
@@ -0,0 +1,973 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// firstTextRISCV parses assembly source and returns the first TEXT function body.
func firstTextRISCV(t *testing.T, src string) *ast.Text {
t.Helper()
f, errs := parser.Parse("f_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
for _, d := range f.Decls {
if fn, ok := d.(*ast.Text); ok {
return fn
}
}
t.Fatal("no TEXT found")
return nil
}
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
t.Helper()
code, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
return code
}
func TestRISCV_add(t *testing.T) {
// func add(a, b int64) int64
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
`)
code := assembleRISCVHelper(t, fn)
// should be 12 bytes with RVC: C.LDSP + C.LDSP + ADD + C.SDSP + C.JR
_ = code
if len(code) == 0 {
t.Error("empty output")
}
}
func TestRISCV_arithmetic(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·arith(SB), NOSPLIT, $0
ADD X10, X11, X12
SUB X12, X13, X14
MUL X14, X15, X16
DIV X16, X17, X18
REM X18, X19, X20
RET
`)
code := assembleRISCVHelper(t, fn)
// 5 R-type instructions + RET = 5*4 + 4 = 24
if len(code) != 24 {
t.Errorf("expected 24 bytes, got %d", len(code))
}
}
func TestRISCV_loadStore(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·mem(SB), NOSPLIT, $0
LD (X10), X11
SD X11, (X12)
LW (X13), X14
SW X14, (X15)
RET
`)
code := assembleRISCVHelper(t, fn)
// Four register-relative loads/stores compress (2B each) + JALR (4B) = 12.
if len(code) != 12 {
t.Errorf("expected 12 bytes, got %d", len(code))
}
}
func TestRISCV_immediate(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·imm(SB), NOSPLIT, $0
ADDI $42, X10, X11
ANDI $0xFF, X11, X12
ORI $1, X12, X13
XORI $0, X13, X14
RET
`)
code := assembleRISCVHelper(t, fn)
// 4 I-type + JALR = 4*4 + 4 = 20
if len(code) != 20 {
t.Errorf("expected 20 bytes, got %d", len(code))
}
}
func TestRISCV_branches(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·br(SB), NOSPLIT, $0
ADDI $1, X10, X10
loop:
BEQ X10, X11, done
ADDI $1, X10, X10
JMP loop
done:
RET
`)
code := assembleRISCVHelper(t, fn)
_ = code
if len(code) == 0 {
t.Error("empty output")
}
}
func TestRISCV_MOV_imm_small(t *testing.T) {
// MOV $42, rd → ADDI (fits in 12 bits, but not C.LI's 6-bit immediate).
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·small(SB), NOSPLIT, $0
MOV $42, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// ADDI (4B) + JALR (4B) = 8
if len(code) != 8 {
t.Errorf("expected 8 bytes, got %d", len(code))
}
}
func TestRISCV_MOV_imm_large(t *testing.T) {
// MOV $0x12345, rd → C.LUI $18 (2B) + ADDIW $837 (4B).
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·large(SB), NOSPLIT, $0
MOV $0x12345, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LUI (2B) + ADDIW (4B) + JALR (4B) = 10
if len(code) != 10 {
t.Errorf("expected 10 bytes, got %d", len(code))
}
}
func TestRISCV_MOV_reg(t *testing.T) {
// MOV rs, rd → ADDI $0, rs, rd, compresses to C.MV
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·reg(SB), NOSPLIT, $0
MOV X10, X11
RET
`)
code := assembleRISCVHelper(t, fn)
// C.MV (2B) + JALR (4B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d (% x)", len(code), code)
}
}
func TestRISCV_MOV_frame(t *testing.T) {
// MOV name+off(FP), rd → load with frame mapping
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·frame(SB), NOSPLIT, $0-8
MOV a+0(FP), X10
MOV X10, ret+0(FP)
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
if len(code) != 8 {
t.Errorf("expected 8 bytes, got %d", len(code))
}
}
func TestRISCV_RVC_loadStore(t *testing.T) {
// Verify that loads/stores from SP are compressed.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·rvcstore(SB), NOSPLIT, $0
LD 0(SP), X10
SD X10, 8(SP)
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LDSP (2B) + C.SDSP (2B) + JALR (4B) = 8
if len(code) != 8 {
t.Errorf("expected 8 bytes, got %d (% x)", len(code), code)
}
}
func TestRISCV_atomics(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0
AMOADDD X10, (X11), X12
LRD (X13), X14
SCD X15, (X16), X17
RET
`)
code := assembleRISCVHelper(t, fn)
// 3 AMO instructions (4B each) + JALR (4B) = 16
if len(code) != 16 {
t.Errorf("expected 16 bytes, got %d", len(code))
}
}
func TestRISCV_fpArith(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fpadd(SB), NOSPLIT, $0
FADDD F10, F11, F12
FSUBD F12, F13, F14
FMULD F14, F15, F16
FDIVD F16, F17, F18
FSQRTD F18, F19
RET
`)
code := assembleRISCVHelper(t, fn)
// 5 FP instructions (4B each) + JALR (4B) = 24
if len(code) != 24 {
t.Errorf("expected 24 bytes, got %d (%d)", len(code), len(code))
}
}
func TestRISCV_csr(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csrtest(SB), NOSPLIT, $0
CSRRS $0x300, X0, X10
CSRRW $0x305, X10, X11
CSRRSI $0x304, $5, X12
RET
`)
code := assembleRISCVHelper(t, fn)
// 3 CSR instructions (4B each) + JALR (4B) = 16
if len(code) != 16 {
t.Errorf("expected 16 bytes, got %d", len(code))
}
}
func TestRISCV_fma(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fmatest(SB), NOSPLIT, $0
FMADDD F10, F11, F12, F13
FMSUBD F13, F14, F15, F16
FNMSUBD F16, F17, F18, F19
FNMADDD F19, F10, F11, F12
RET
`)
code := assembleRISCVHelper(t, fn)
// 4 FMA instructions (4B each) + JALR (4B) = 20
if len(code) != 20 {
t.Errorf("expected 20 bytes, got %d", len(code))
}
}
func TestRISCV_conversions(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cvt(SB), NOSPLIT, $0
FCVTDL X10, F10
FCVTLD F10, X11
FMVXD F10, X12
FMVDX X12, F11
RET
`)
code := assembleRISCVHelper(t, fn)
// 4 conversion instructions (4B each) + JALR (4B) = 20
if len(code) != 20 {
t.Errorf("expected 20 bytes, got %d", len(code))
}
}
func TestRISCV_fpCmp(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cmp(SB), NOSPLIT, $0
FEQD F10, F11, X10
FLTD F12, F13, X11
FLED F14, F15, X12
RET
`)
code := assembleRISCVHelper(t, fn)
// 3 FP compare (4B each) + JALR (4B) = 16
if len(code) != 16 {
t.Errorf("expected 16 bytes, got %d", len(code))
}
}
func TestRISCV_forwardBranch(t *testing.T) {
// Forward label reference; must not fail.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fwd(SB), NOSPLIT, $0
ADDI $1, X10, X10
BEQ X10, X11, done
ADDI $1, X10, X10
done:
RET
`)
code := assembleRISCVHelper(t, fn)
_ = code
if len(code) == 0 {
t.Error("empty output")
}
}
func TestRISCV_RVC_ADDI(t *testing.T) {
// ADDI where rd=rs1 and small imm → C.ADDI
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·caddi(SB), NOSPLIT, $0
ADDI $5, X10, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// C.ADDI (2B) + JALR (4B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d", len(code))
}
}
func TestRISCV_RVC_LI(t *testing.T) {
// ADDI X0, $imm, rd → C.LI
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cli(SB), NOSPLIT, $0
ADDI $7, X0, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LI (2B) + JALR (4B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d", len(code))
}
}
func TestRISCV_RVC_LUI(t *testing.T) {
// LUI rd, small nonzero imm → C.LUI
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·clui(SB), NOSPLIT, $0
LUI X10, $1
RET
`)
code := assembleRISCVHelper(t, fn)
// C.LUI (2B) + JALR (4B) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes, got %d", len(code))
}
}
func TestRISCV_AssembleFile(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOV a+0(FP), X10
RET
TEXT ·sub(SB), NOSPLIT, $0
SUB X10, X11, X12
RET
`
f, errs := parser.Parse("t_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
if len(img.Funcs) != 2 {
t.Fatalf("expected 2 functions, got %d", len(img.Funcs))
}
// func add: C.LDSP(2) + JALR(4) = 6
if img.Funcs[0].Size != 6 {
t.Errorf("add: expected 6 bytes, got %d", img.Funcs[0].Size)
}
// func sub: SUB(4) + JALR(4) = 8
if img.Funcs[1].Size != 8 {
t.Errorf("sub: expected 8 bytes, got %d", img.Funcs[1].Size)
}
}
func TestRISCV_encodings(t *testing.T) {
// Smoke test that all known RISC-V mnemonics encode successfully.
tests := []struct {
name, src string
wantBytes int
}{
{"ADD", "ADD X10, X11, X12\nRET\n", 8},
{"SUBW", "SUBW X10, X11, X12\nRET\n", 8},
{"MUL", "MUL X10, X11, X12\nRET\n", 8},
{"DIVW", "DIVW X10, X11, X12\nRET\n", 8},
{"REMUW", "REMUW X10, X11, X12\nRET\n", 8},
{"ADDIW", "ADDIW $5, X10, X11\nRET\n", 8},
{"SLLI", "SLLI $3, X10, X11\nRET\n", 8}, // ADDI+SLLI? No, SLLI uses I-type
{"SRLI", "SRLI $2, X10, X11\nRET\n", 8},
{"SRAI", "SRAI $1, X10, X11\nRET\n", 8},
{"LB", "LB (X10), X11\nRET\n", 8},
{"LBU", "LBU (X10), X11\nRET\n", 8},
{"LH", "LH (X10), X11\nRET\n", 8},
{"LHU", "LHU (X10), X11\nRET\n", 8},
{"LWU", "LWU (X10), X11\nRET\n", 8},
{"SB", "SB X10, (X11)\nRET\n", 8},
{"SH", "SH X10, (X11)\nRET\n", 8},
{"SW", "SW X10, (X11)\nRET\n", 6},
{"LUI", "LUI X10, $0x12345\nRET\n", 8},
{"AUIPC", "AUIPC X10, $0\nRET\n", 8},
{"FLW", "FLW (X10), F10\nRET\n", 8},
{"FSW", "FSW F10, (X11)\nRET\n", 8},
{"FADDS", "FADDS F10, F11, F12\nRET\n", 8},
{"FMINS", "FMINS F10, F11, F12\nRET\n", 8},
{"FMAXD", "FMAXD F10, F11, F12\nRET\n", 8},
{"FCVTSD", "FCVTSD F10, F11\nRET\n", 8},
{"FCVTDS", "FCVTDS F10, F11\nRET\n", 8},
{"FMVXW", "FMVXW F10, X10\nRET\n", 8},
{"FMADD_S", "FMADDS F10, F11, F12, F13\nRET\n", 8},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·`+tt.name+`(SB), NOSPLIT, $0
`+tt.src)
code := assembleRISCVHelper(t, fn)
if len(code) != tt.wantBytes {
t.Errorf("expected %d bytes, got %d", tt.wantBytes, len(code))
}
})
}
}
func TestRISCV_RVC_branch(t *testing.T) {
// Branches are never RVC-compressed (no C.BEQZ/C.BNEZ), matching go tool asm.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cbeqz(SB), NOSPLIT, $0
ADDI $1, X10, X10
BEQ X10, X0, done
ADDI $1, X10, X10
done:
RET
`)
code := assembleRISCVHelper(t, fn)
// C.ADDI(2) + BEQ(4) + C.ADDI(2) + JALR(4) = 12
if len(code) != 12 {
t.Errorf("expected 12 bytes with uncompressed BEQ, got %d", len(code))
}
}
func TestRISCV_RVC_CJ(t *testing.T) {
// JMP target → JAL X0 (never compressed to C.J), matching go tool asm.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cj(SB), NOSPLIT, $0
JMP done
done:
RET
`)
code := assembleRISCVHelper(t, fn)
// JAL(4) + JALR(4) = 8
if len(code) != 8 {
t.Errorf("expected 8 bytes with uncompressed JMP, got %d", len(code))
}
}
func TestRISCV_RVC_CADD(t *testing.T) {
// ADD where rd==rs1 and both in prime regs → C.ADD.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cadd(SB), NOSPLIT, $0
ADD X10, X11, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// C.ADD(2) + JALR(4) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.ADD, got %d", len(code))
}
}
func TestRISCV_RVC_CADD_commute(t *testing.T) {
// ADD where rd==rs2 (commutative swap) → C.ADD.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cadd2(SB), NOSPLIT, $0
ADD X11, X10, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// C.ADD(2) + JALR(4) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.ADD (commuted), got %d", len(code))
}
}
func TestRISCV_RVC_CSUB(t *testing.T) {
// SUB rs2, rs1, rd → C.SUB when rd == rs1 and both in prime regs.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csub(SB), NOSPLIT, $0
SUB X11, X10, X10
RET
`)
code := assembleRISCVHelper(t, fn)
// SUB X11, X10, X10 → rs2=X11, rs1=X10, rd=X10; rd==rs1 → C.SUB (2B) + JALR (4B) = 6.
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.SUB, got %d (% x)", len(code), code)
}
}
func TestRISCV_RVC_CXOR(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cxor(SB), NOSPLIT, $0
XOR X10, X11, X10
RET
`)
code := assembleRISCVHelper(t, fn)
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.XOR, got %d", len(code))
}
}
func TestRISCV_RVC_COR(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cor(SB), NOSPLIT, $0
OR X10, X11, X10
RET
`)
code := assembleRISCVHelper(t, fn)
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.OR, got %d", len(code))
}
}
func TestRISCV_RVC_CAND(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cand(SB), NOSPLIT, $0
AND X10, X11, X10
RET
`)
code := assembleRISCVHelper(t, fn)
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.AND, got %d", len(code))
}
}
func TestRISCV_RVC_CFLDSP(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cfldsp(SB), NOSPLIT, $0-8
FLD a+0(FP), F10
RET
`)
code := assembleRISCVHelper(t, fn)
// C.FLDSP(2) + JALR(4) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.FLDSP, got %d", len(code))
}
}
func TestRISCV_RVC_CFSDSP(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·cfsdsp(SB), NOSPLIT, $0-8
FSD F10, ret+0(FP)
RET
`)
code := assembleRISCVHelper(t, fn)
// C.FSDSP(2) + JALR(4) = 6
if len(code) != 6 {
t.Errorf("expected 6 bytes with C.FSDSP, got %d", len(code))
}
}
func TestRISCV_SB_addr(t *testing.T) {
// MOV $sym<>(SB), rd → AUIPC + ADDI (8 bytes for SB).
src := `#include "textflag.h"
TEXT ·sbaddr(SB), NOSPLIT, $0
MOV $answer<>(SB), X10
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`
f, errs := parser.Parse("t_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
// AUIPC(4) + ADDI(4) + JALR(4) = 12
if img.Funcs[0].Size != 12 {
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
}
}
func TestRISCV_SB_store(t *testing.T) {
// MOV rd, sym<>(SB) → AUIPC + SD (8 bytes for SB).
src := `#include "textflag.h"
TEXT ·sbstore(SB), NOSPLIT, $0
MOV X10, result<>(SB)
RET
GLOBL result<>(SB), NOPTR, $8
`
f, errs := parser.Parse("t_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
// AUIPC X31(4) + SD X10,0(X31)(4) + JALR(4) = 12
if img.Funcs[0].Size != 12 {
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
}
}
func TestRISCV_ELF(t *testing.T) {
src := `#include "textflag.h"
TEXT ·simple(SB), NOSPLIT, $0
RET
`
f, errs := parser.Parse("t_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
if len(obj) < 4 || obj[0] != 0x7f || obj[1] != 'E' || obj[2] != 'L' || obj[3] != 'F' {
t.Fatal("not a valid ELF file")
}
if len(obj) >= 20 {
machine := uint16(obj[18]) | uint16(obj[19])<<8
if machine != 243 {
t.Errorf("e_machine = %d, want 243 (EM_RISCV)", machine)
}
}
}
func TestRISCV_ELF_withData(t *testing.T) {
src := `#include "textflag.h"
TEXT ·get(SB), NOSPLIT, $0
RET
GLOBL val<>(SB), RODATA, $4
DATA val<>+0(SB)/4, $7
`
f, errs := parser.Parse("t_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
if len(img.DataSyms) != 1 {
t.Fatalf("expected 1 data symbol, got %d", len(img.DataSyms))
}
if img.DataSyms[0].Name != "val" {
t.Errorf("data symbol name = %q, want val", img.DataSyms[0].Name)
}
if img.DataSyms[0].Size != 4 {
t.Errorf("data symbol size = %d, want 4", img.DataSyms[0].Size)
}
obj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
_ = obj
}
func TestRISCV_SB_load(t *testing.T) {
// MOV sym<>(SB), rd → AUIPC + LD (8 bytes for SB).
src := `#include "textflag.h"
TEXT ·sbload(SB), NOSPLIT, $0
MOV answer<>(SB), X10
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`
f, errs := parser.Parse("t_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
// AUIPC(4) + LD(4) + JALR(4) = 12
if img.Funcs[0].Size != 12 {
t.Errorf("expected 12 bytes, got %d", img.Funcs[0].Size)
}
}
// TestRISCV_RVC_StorePatterns pins the register-relative compressed store
// encodings for offsets with immediate bits 4 and 5 set, byte-identical to
// the toolchain's encodeCS (patterns {5,4,3,7,6} and {5,4,3,2,6}).
// Regression: the store-side patterns dropped imm[4], so every such store
// silently encoded the wrong address while the loads stayed correct.
func TestRISCV_RVC_StorePatterns(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csstores(SB), NOSPLIT, $0
SD X9, 24(X8)
SW X10, 16(X11)
FSD F8, 40(X12)
LD 24(X8), X9
LW 16(X11), X10
FLD 40(X12), F8
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0x04, 0xec, // c.sd x9, 24(x8)
0x88, 0xc9, // c.sw x10, 16(x11)
0x00, 0xb6, // c.fsd f8, 40(x12)
0x04, 0x6c, // c.ld x9, 24(x8)
0x88, 0x49, // c.lw x10, 16(x11)
0x00, 0x36, // c.fld f8, 40(x12)
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestRISCV_FENCE pins the FENCE encoding: the toolchain expands the bare
// mnemonic to fence iorw, iorw (0x0FF0000F), not fence 0,0.
func TestRISCV_FENCE(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fence(SB), NOSPLIT, $0
FENCE
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0x0f, 0x00, 0xf0, 0x0f, // fence iorw, iorw
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestRISCV_RVC_WidthSpellings pins the compression of the GOROOT width
// spellings: MOVW and MOVD lower to their base load/store and compress
// exactly like LW/SW/FLD/FSD would (the toolchain compresses these shapes;
// before the normalisation they stayed 4 bytes).
func TestRISCV_RVC_WidthSpellings(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·widths(SB), NOSPLIT, $0-16
MOVW w+0(FP), X9
MOVW X9, v+4(FP)
MOVD d+0(FP), F8
MOVD F8, r+8(FP)
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0xa2, 0x44, // c.lwsp x9, 8
0x26, 0xc6, // c.swsp x9, 12
0x22, 0x24, // c.fldsp f8, 8
0x22, 0xa8, // c.fsdsp f8, 16
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
func TestRISCV_system_instrs(t *testing.T) {
// Test FENCE, ECALL, EBREAK encoding.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·sys(SB), NOSPLIT, $0
FENCE
ECALL
EBREAK
RET
`)
code := assembleRISCVHelper(t, fn)
// FENCE(4) + ECALL(4) + C.EBREAK(2) + JALR(4) = 14
if len(code) != 14 {
t.Errorf("expected 14 bytes, got %d (% x)", len(code), code)
}
}
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
// MOV $sym(FP), rd should return an error (unsupported).
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·badfp(SB), NOSPLIT, $0
MOV $arg(FP), X10
RET
`)
_, _, _, _, _, err := assembleRISCV(fn)
if err == nil {
t.Error("expected error for MOV $arg(FP), got nil")
}
}
func TestRISCV_CALL(t *testing.T) {
// CALL sym(SB) → JAL X1, sym(SB) with a single R_RISCV_JAL relocation.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·calltest(SB), NOSPLIT, $0
CALL ext(SB)
RET
`)
code, _, relocs, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// prologue (8) + JAL (4) + epilogue+JALR (8) = 20
if len(code) != 20 {
t.Fatalf("expected 20 bytes with CALL sym(SB), got %d", len(code))
}
if len(relocs) != 1 {
t.Fatalf("relocs = %d, want 1", len(relocs))
}
r := relocs[0]
if r.Kind != RelRISCVJal || r.Name != "ext" || r.Off != 8 || r.After != 12 || r.Addend != 0 {
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d}", r.Kind, r.Off, r.After, r.Name, r.Addend)
}
// The JAL instruction itself is JAL X1, 0 at function offset 8.
wantJAL := wordLE(riscvJType(1, 0))
if !bytes.Equal(code[8:12], wantJAL) {
t.Errorf("JAL = % x, want % x", code[8:12], wantJAL)
}
}
func TestRISCV_CALL_local_error(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·calllocal(SB), NOSPLIT, $0
CALL sub
sub:
RET
`)
_, _, _, _, _, err := assembleRISCV(fn)
if err == nil {
t.Error("expected error for CALL to local label, got nil")
}
}
// TestRISCVIndirectBranch pins the indirect branch encodings: JMP (X5) is the
// toolchain's JALR X0, 0(X5), and the trampoline form JALR rd, offset(rs1)
// takes its destination from the first operand (regression: the base
// register was once read as the destination, silently jumping to X0).
func TestRISCVIndirectBranch(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
JMP (X5)
JALR X0, 0(X6)
JALR X28, 0(X9)
RET
`)
code := assembleRISCVHelper(t, fn)
wantWords(t, code,
0x00028067, // jalr x0, 5(x0), 0
0x00030067, // jalr x0, 6(x0), 0
0x00048e67, // jalr x28, 9(x0), 0
0x00008067, // jalr x0, 1(x0), 0 (RET)
)
}
// encodeOneInstrRISCV encodes a single parsed instruction against a synthetic
// offsets map, the smallest honest harness for the branch-range diagnostics:
// the spans are far larger than any source a test would want to spell out.
func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]int) ([]byte, error) {
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
instr := fn.Body[0].(*ast.Instr)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil)
}
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
// [-4096, 4094] and the J-type span [-1048576, 1048574] are diagnosed instead
// of wrapping silently to a wrong target.
func TestRISCVBranchJumpRange(t *testing.T) {
cases := []struct {
name string
src string
off int // the target's function-relative offset (pc 0)
ok bool
}{
{"branch max", "BEQ X10, X11, tgt\nRET\n", 4094, true},
{"branch past max", "BEQ X10, X11, tgt\nRET\n", 4096, false},
{"branch back max", "BEQ X10, X11, tgt\nRET\n", -4096, true},
{"branch back past max", "BEQ X10, X11, tgt\nRET\n", -4098, false},
{"branchz past max", "BEQZ X10, tgt\nRET\n", 4096, false},
{"jump max", "JMP tgt\nRET\n", 1048574, true},
{"jump past max", "JMP tgt\nRET\n", 1048576, false},
{"jump back max", "JMP tgt\nRET\n", -1048576, true},
{"jump back past max", "JMP tgt\nRET\n", -1048578, false},
{"jal past max", "JAL tgt\nRET\n", 1048576, false},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
_, err := encodeOneInstrRISCV(t, "TEXT ·f(SB), NOSPLIT, $0\n\t"+c.src, 0, map[string]int{"tgt": c.off})
if c.ok && err != nil {
t.Fatalf("unexpected error: %v", err)
}
if !c.ok && err == nil {
t.Fatal("expected an out-of-range diagnostic, got none")
}
})
}
}
// TestRISCVBranchFarBody drives the range check through the full two-pass
// assembler: a forward branch over a body larger than the B-type span must
// error rather than wrap.
func TestRISCVBranchFarBody(t *testing.T) {
var sb strings.Builder
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
for range 1100 {
sb.WriteString("\tADD X10, X11, X12\n")
}
sb.WriteString("done:\n\tRET\n")
fn := firstTextRISCV(t, sb.String())
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Error("expected a branch-out-of-range error, got none")
}
}
// TestRISCV_CSRRange checks the CSR address range: the 12-bit field is
// diagnosed rather than masked, so CSRRW $4096 does not silently address
// CSR 0.
func TestRISCV_CSRRange(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csrhi(SB), NOSPLIT, $0
CSRRW $4096, X10, X11
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Error("expected an out-of-range error for CSR $4096, got none")
}
fn = firstTextRISCV(t, `#include "textflag.h"
TEXT ·csrmax(SB), NOSPLIT, $0
CSRRW $4095, X10, X11
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("CSR $4095 must assemble: %v", err)
}
}
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
// span are diagnosed instead of silently truncated to their low 32 bits (the
// toolchain materialises such constants via SLLI expansion, which this
// assembler does not implement).
func TestRISCV_Imm64Rejected(t *testing.T) {
cases := []string{
"MOV $0x123456789, X10",
"ADDI $0x100000000, X10, X11",
"ANDI $-0x800000001, X10, X11",
"SUB $0x100000000, X10, X11",
}
for _, src := range cases {
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Errorf("%s: expected an out-of-range error, got none", src)
}
}
// The full signed 32-bit span still assembles, including the SUB form
// whose negated immediate only just fits.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·edge(SB), NOSPLIT, $0
MOV $2147483647, X10
MOV $-2147483648, X11
SUB $0x80000000, X12, X13
RET
`)
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("int32-span immediates must assemble: %v", err)
}
}
+364
View File
@@ -0,0 +1,364 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
//
// Go's riscv64 functions have no hardware frame pointer: FP and SP are
// synthetic registers resolved against the hardware stack pointer (X2) and
// the frame size. The return address lives in the link register (X1, RA/LR).
//
// The autosize is the real stack adjustment: the declared local frame plus
// the 8 bytes for the saved link register (the toolchain's FixedFrameSize).
// A leaf function with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0), byte-identical to the toolchain:
//
// MOV LR, -autosize(SP) // save LR below the new SP (traceback-safe)
// ADDI $-autosize, SP, SP // open the frame
// MOV LR, 0(SP) // save LR again at SP (signal-safety)
//
// Epilogue (autosize > 0): MOV 0(SP), LR; ADDI $autosize, SP, SP; the RET's
// uncompressed JALR X0, 0(X1) follows. The toolchain restores LR on every
// frame, leaf or not.
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
type riscvFrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR)
// Stack-split guard state: the toolchain emits the check for every
// non-NOSPLIT function whose autosize is nonzero (a zero autosize is
// "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf
// auto-NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// riscvComputeFrame derives the frame layout for a TEXT function.
func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
frame := frameSize(t)
if frame != 0 || !riscvIsLeaf(t) {
// FixedFrameSize = 8: space for the saved link register. A
// zero-frame non-leaf function still opens an 8-byte frame for LR.
autosize := frame + 8
fi := riscvFrameInfo{autosize: autosize}
if !hasNoSplitFlag(t) {
fi.needSplit = true
switch {
case autosize <= stackSmall:
fi.splitClass = 0
case autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
return riscvFrameInfo{}
}
// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT.
func hasNoSplitFlag(t *ast.Text) bool {
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
return true
}
}
return false
}
// riscvIsLeaf reports whether a function contains no call instructions.
// CALL always links; JAL/JALR link only when their destination register is
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
func riscvIsLeaf(t *ast.Text) bool {
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "CALL":
return false
case "JAL":
// JAL rd, target, a call only when rd is the link register.
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
return false
}
case "JALR":
// JALR rd, offset(rs1) links when the destination register (the
// first operand) is X1; JALR rs1, rd links when the second
// register is X1; JALR offset(rs1) always links to X1.
if len(in.Operands) == 1 {
return false
}
if isMemOperand(in.Operands[1]) {
if regFromOperand(in.Operands[0]) == 1 {
return false
}
continue
}
if regFromOperand(in.Operands[1]) == 1 {
return false
}
}
}
return true
}
// riscvPrologue returns the prologue bytes for a RISC-V function, matching
// the toolchain's compression: the SP adjustment compresses to C.ADDI when
// the immediate fits, and the second LR save compresses to C.SDSP.
func riscvPrologue(fi riscvFrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
var out []byte
// MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond
// the imm12 range the toolchain materialises the address in X31.
if fits12(int32(-fi.autosize)) {
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
} else {
out = append(out, riscvAddressInX31(int32(-fi.autosize))...)
lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12)
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...)
}
// ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31
// materialisation beyond imm12).
if fits12(int32(-fi.autosize)) {
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(-fi.autosize))...)
}
// MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0.
c := rvcSSP(0x7, 1, 0)
out = append(out, byte(c), byte(c>>8))
return out
}
func fits12(v int32) bool { return v >= -2048 && v <= 2047 }
// splitHi returns the LUI half of the hi/lo split of v (what remains is the
// sign-extended 12-bit low part).
func splitHi(v int32) int32 {
_, high := splitRISCV32Imm(v)
return high
}
// riscvAddressInX31 materialises hi(v) into X31 against the stack pointer,
// matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi;
// C.ADD (or ADD) X31, SP.
func riscvAddressInX31(v int32) []byte {
return riscvAddressInX31WithBase(v, 2)
}
// riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary
// base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field
// carries the full 5-bit register, so the compressed form is always
// available.
func riscvAddressInX31WithBase(v int32, rs1 int) []byte {
hi := splitHi(v)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
c := rvcCR(0x9, 31, uint32(rs1))
return append(out, byte(c), byte(c>>8))
}
// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry:
// C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form).
func riscvAddToSP(v int32) []byte {
hi := splitHi(v)
lo := v - (hi << 12)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
if lo >= -32 && lo <= 31 {
c := rvcCI(0x1, 31, uint32(lo)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...)
}
c := rvcCR(0x9, 2, 31)
return append(out, byte(c), byte(c>>8))
}
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by the uncompressed JALR X0,
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
func riscvReturn(fi riscvFrameInfo) []byte {
var out []byte
if fi.autosize != 0 {
// MOV 0(SP), LR, LD X1, 0(X2) → C.LDSP X1, 0.
c := rvcLSP(0x3, 1, 0)
out = append(out, byte(c), byte(c>>8))
// ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits).
if fits12(int32(fi.autosize)) {
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(fi.autosize))...)
}
}
// JALR X0, 0(X1).
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
}
// riscvSPAdjust emits an ADDI rd, imm, rd for the stack pointer (rd = rs1 =
// X2), compressed to C.ADDI16SP when the immediate is a nonzero 16-byte
// multiple, else C.ADDI when it fits 6-bit signed.
func riscvSPAdjust(imm int32) []byte {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
c := rvcADDI16SP(2, imm)
return []byte{byte(c), byte(c >> 8)}
}
if riscvFitsCAddi(imm) {
c := rvcCI(0x0, 2, uint32(imm)&0x3F)
return []byte{byte(c), byte(c >> 8)}
}
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 2, 2, imm))
}
// riscvFitsCAddi reports whether imm compresses to C.ADDI (a nonzero 6-bit
// signed immediate).
func riscvFitsCAddi(imm int32) bool {
return imm != 0 && imm >= -32 && imm <= 31
}
// riscvPrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize). It is
// computed from the same expansion functions the prologue emits, so the
// large-frame X31 materialisations are counted: C.LUI + C.ADD before the SD,
// C.LUI + ADDIW + C.ADD for the SP adjust.
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
adj := int32(-fi.autosize)
if fits12(adj) {
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 4 + len(riscvSPAdjust(adj))
}
return len(riscvAddressInX31(adj)) + 4 + len(riscvAddToSP(adj))
}
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR, the point where SP is restored. The
// small frame closes with C.LDSP + ADDI/C.ADDI; the large frame materialises
// the adjustment through X31 (C.LUI + ADDIW + C.ADD).
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
adj := int32(fi.autosize)
if fits12(adj) {
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 2 + len(riscvSPAdjust(adj))
}
return 2 + len(riscvAddToSP(adj))
}
// riscvResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x+N(SP) → (N + autosize)(SP). Returns base = -1 for an unresolvable
// reference (SB: static data, handled by the relocation path).
func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32) {
if sym == nil {
return -1, 0
}
switch sym.Pseudo {
case "FP":
return 2, int32(sym.Offset) + int32(fi.autosize) + 8
case "SP":
return 2, int32(fi.autosize) + int32(sym.Offset)
case "SB":
return -1, int32(sym.Offset)
}
return -1, 0
}
// riscvGuardLen returns the byte length of the stack-split guard prefix
// including the inline morestack call (zero when the function needs no
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
// between the guard and the body: the guard branches forward over it.
func riscvGuardLen(fi riscvFrameInfo) (int, error) {
g, _, err := riscvGuard(fi)
if err != nil {
return 0, err
}
return len(g), nil
}
// riscvGuard emits the stack-split guard prefix with the inline morestack
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
// relative to the guard itself, which sits at function offset 0.
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc, error) {
if !fi.needSplit {
return nil, Reloc{}, nil
}
// MOV 16(g), X6 (g.stackguard0), g = X27.
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
jalBack := func() []byte {
// JAL X0 back to the function start: it sits right after the JAL X5,
// so its displacement is minus the current offset.
return wordLE(riscvJType(0, int32(-len(out))))
}
var reloc Reloc
switch fi.splitClass {
case 0:
// BLTU X6, SP, done (+12: over the CALL and the JMP back)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
case 1:
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+12)
off := int32(fi.autosize - stackSmall)
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
default:
// MOV $(framesize-StackSmall), X7; BLTU SP, X7, call;
// ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call
off := int32(fi.autosize - stackSmall)
mov := encodeRISCVLoadImm(7, off)
out = append(out, mov...)
addiLen := riscvItypeImmediateSize("ADDI", -off)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
if err != nil {
// The ADDI expansion failed: the SP adjustment this class
// depends on is not emittable, and silently dropping it would
// corrupt every stack reference in the body.
return nil, Reloc{}, fmt.Errorf("stack-split guard: %w", err)
}
out = append(out, addi...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
}
return out, reloc, nil
}
+122
View File
@@ -0,0 +1,122 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestRISCVFrameSpadjAndLines checks that a framed function records its
// stack-adjustment boundaries and source-line table, the inputs the GOOBJ
// emitter turns into the pcsp/pcfile/pcline tables.
func TestRISCVFrameSpadjAndLines(t *testing.T) {
f, errs := parser.Parse("frame_riscv64.s", `#include "textflag.h"
TEXT ·framed(SB), NOSPLIT, $16-16
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 24 {
t.Fatalf("size = %d, want 24", fn.Size)
}
// autosize = 16 + 8 = 24; the prologue boundary is just past its C.ADDI
// (SD 4 + C.ADDI 2 = 6), and the RET restores SP just past its C.ADDI
// (RET starts at 16; C.LDSP 2 + C.ADDI 2 = 20).
wantSpadj := []SpadjStep{{PC: 6, Value: 24}, {PC: 20, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
// One line entry per instruction, in emission order.
wantLines := []LineEntry{
{Offset: 8, Line: 4},
{Offset: 10, Line: 5},
{Offset: 12, Line: 6},
{Offset: 14, Line: 7},
{Offset: 16, Line: 8},
}
if len(fn.Lines) != len(wantLines) {
t.Fatalf("lines = %v, want %v", fn.Lines, wantLines)
}
for i := range wantLines {
if fn.Lines[i] != wantLines[i] {
t.Errorf("lines[%d] = %v, want %v", i, fn.Lines[i], wantLines[i])
}
}
}
// TestRISCVFrameSpadjLargeFrame checks the stack-adjustment boundaries of a
// frame past the imm12 range: the prologue materialises the LR-store address
// and the SP adjustment through X31 (C.LUI + C.ADD + SD, then C.LUI + ADDIW +
// C.ADD), so the SP boundary lands at PC 16, and the RET closes with
// C.LDSP plus the same X31 adjustment, 10 bytes. Regression: both helpers
// assumed the small-frame prologue and reported 8 and 6.
func TestRISCVFrameSpadjLargeFrame(t *testing.T) {
f, errs := parser.Parse("bigframe_riscv64.s", `#include "textflag.h"
TEXT ·big(SB), NOSPLIT, $9000-8
MOV a+0(FP), X10
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
// autosize = 9008. Prologue: C.LUI X31 + C.ADD X31,SP (4) + SD (4) +
// C.LUI X31 + ADDIW X31 + C.ADD SP,X31 (8) = 16 bytes to the SP boundary;
// C.SDSP X1 (2) follows, so the body starts at 18.
wantSpadj := []SpadjStep{{PC: 16, Value: 9008}, {PC: 36, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
// The FP load materialises its 9016-byte offset through X31 as well
// (8 bytes), then RET's epilogue (C.LDSP + X31 adjust = 10) plus JALR.
if fn.Size != 18+8+14 {
t.Errorf("size = %d, want %d", fn.Size, 18+8+14)
}
}
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
// defines: LR is the link register (X1) and TMP is the assembler scratch
// register (X31/T6).
func TestRISCVRegAliases(t *testing.T) {
for name, want := range map[string]int{
"X1": 1, "RA": 1, "LR": 1,
"X31": 31, "T6": 31, "TMP": 31,
"X2": 2, "SP": 2,
} {
if got := riscvRegNum(name); got != want {
t.Errorf("riscvRegNum(%q) = %d, want %d", name, got, want)
}
}
}
+429
View File
@@ -0,0 +1,429 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
// instruction carrying an R_RISCV_JAL relocation (4-byte field) in both the
// GOOBJ and ELF object emitters.
func TestGOObjectRISCVCallReloc(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·c(SB), NOSPLIT, $0-0
CALL callee<>(SB)
RET
GLOBL callee<>(SB), RODATA, $8
DATA callee<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if len(fn.Relocs) != 1 {
t.Fatalf("relocs = %d, want 1", len(fn.Relocs))
}
r := fn.Relocs[0]
if r.Kind != RelRISCVJal || r.Off != 8 || r.After != 12 || r.Name != "callee" || r.Addend != 0 || r.External {
t.Errorf("reloc = {kind %v off %d after %d name %q addend %d external %v}", r.Kind, r.Off, r.After, r.Name, r.Addend, r.External)
}
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
v := openGoobj(t, obj)
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
// The function is the last non-package symbol: 4 package defs, then the
// 4 pc tables and the function.
first := int(binary.LittleEndian.Uint32(relocIdx[(4+4)*4:]))
if (first+1)*23 > len(relocs) {
t.Fatalf("reloc block too short: first=%d len=%d", first, len(relocs))
}
e := relocs[first*23:]
le := binary.LittleEndian
if int32(le.Uint32(e[0:])) != 8 || e[4] != 4 || le.Uint16(e[5:]) != relocRISCVJal || le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
t.Errorf("GOOBJ reloc = off %d size %d type %d pkg %d sym %d", int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
}
// The ELF object must carry a single R_RISCV_JAL relocation in .rela.text.
elfObj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
if !hasELFRISCVJAL(t, elfObj) {
t.Error("ELF object missing R_RISCV_JAL relocation")
}
}
// TestELFRISCVPCRELLO12Anchor checks the psABI's LO12 pairing rule: the
// R_RISCV_PCREL_LO12_I/S relocation must reference a symbol whose value is
// the AUIPC site of its HI20 partner (psABI §8.4.9; cmd/link generates one
// local text symbol per AUIPC for exactly this). The emitter pairs each
// HI20 (against the target symbol) with a LO12 against the .text section
// symbol whose addend is the AUIPC's section-relative offset, so S + A is
// the AUIPC address.
func TestELFRISCVPCRELLO12Anchor(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
MOV answer<>(SB), X11
MOV X12, answer<>(SB)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse ELF: %v", err)
}
defer ef.Close()
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efRISCVFloatAbiDouble {
t.Errorf("e_flags = %#x, want %#x (EF_RISCV_FLOAT_ABI_DOUBLE)", flags, efRISCVFloatAbiDouble)
}
rela := ef.Section(".rela.text")
if rela == nil {
t.Fatal("missing .rela.text")
}
b, err := rela.Data()
if err != nil {
t.Fatal(err)
}
if len(b) != 6*24 {
t.Fatalf(".rela.text holds %d entries, want six (three HI20/LO12 pairs)", len(b)/24)
}
le := binary.LittleEndian
wantLo := []uint32{rRISCVPCRELLO12I, rRISCVPCRELLO12I, rRISCVPCRELLO12S}
for p := range 3 {
auipc := 8 * p
hi := b[p*2*24:]
lo := b[(p*2+1)*24:]
if off := le.Uint64(hi[0:]); off != uint64(auipc) {
t.Errorf("pair %d: HI20 r_offset = %d, want %d (the AUIPC)", p, off, auipc)
}
if typ := uint32(le.Uint64(hi[8:])); typ != rRISCVPCRELHI20 {
t.Errorf("pair %d: HI20 type = %d, want %d", p, typ, rRISCVPCRELHI20)
}
if sym := int(le.Uint64(hi[8:]) >> 32); sym == 0 || sym == 1 {
t.Errorf("pair %d: HI20 against symbol %d, want the target", p, sym)
}
if off := le.Uint64(lo[0:]); off != uint64(auipc+4) {
t.Errorf("pair %d: LO12 r_offset = %d, want %d", p, off, auipc+4)
}
if typ := uint32(le.Uint64(lo[8:])); typ != wantLo[p] {
t.Errorf("pair %d: LO12 type = %d, want %d", p, typ, wantLo[p])
}
// The LO12 must denote the AUIPC site: the .text section symbol
// (index 1) plus the AUIPC's section-relative offset as addend.
if sym := int(le.Uint64(lo[8:]) >> 32); sym != 1 {
t.Errorf("pair %d: LO12 against symbol %d, want 1 (the .text section symbol)", p, sym)
}
if add := int64(le.Uint64(lo[16:])); add != int64(auipc) {
t.Errorf("pair %d: LO12 addend = %d, want %d (S + A = the AUIPC address)", p, add, auipc)
}
}
}
func TestGOObjectRISCVStructure(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
MOV answer<>(SB), X11
MOV X12, answer<>(SB)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
if fn.Size != 28 {
t.Fatalf("function size = %d, want 28", fn.Size)
}
if len(fn.Relocs) != 3 {
t.Fatalf("relocs = %d, want 3", len(fn.Relocs))
}
wantKind := []RelocKind{RelRISCVPCRELIType, RelRISCVPCRELIType, RelRISCVPCRELSType}
wantOff := []int{0, 8, 16}
for i, r := range fn.Relocs {
if r.Kind != wantKind[i] || r.Off != wantOff[i] || r.After != r.Off+8 || r.Name != "answer" || r.Addend != 0 {
t.Errorf("reloc %d = {kind %v off %d after %d name %q addend %d}", i, r.Kind, r.Off, r.After, r.Name, r.Addend)
}
}
obj, err := img.GOObjectRISCV("testpkg", "k_riscv64.s")
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
v := openGoobj(t, obj)
// Package defs: the static GLOBL, the FuncInfo, then the two DWARF
// symbols.
defs := v.syms(blkSymdef)
if len(defs) != 4 {
t.Fatalf("symdefs = %d, want 4", len(defs))
}
if defs[0].name != "answer" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 {
t.Errorf("answer symbol = %+v", defs[0])
}
if defs[2].typ != kindSDWARFLINES || defs[3].typ != kindSDWARFFCN {
t.Errorf("dwarf symbols = %+v, %+v", defs[2], defs[3])
}
// The three code relocations, in definition order: ITYPE, ITYPE, STYPE,
// each 8 bytes wide against the GLOBL (package symbol 0).
relocIdx := v.blk(blkRelocIdx)
relocs := v.blk(blkReloc)
if len(relocs) != 5*23 {
t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs))
}
// The function is the last non-package symbol; its relocs start after
// the DWARF symbols' (defs 2 and 3 each carry one).
le := binary.LittleEndian
first := int(le.Uint32(relocIdx[4*(4+4):]))
wantType := []uint16{relocRISCVPcrelItype, relocRISCVPcrelItype, relocRISCVPcrelStype}
wantOffAbs := []int{0, 8, 16}
for i := range 3 {
e := relocs[(first+i)*23:]
if int32(le.Uint32(e[0:])) != int32(wantOffAbs[i]) || e[4] != 8 || le.Uint16(e[5:]) != wantType[i] ||
le.Uint32(e[15:]) != pkgIdxSelf || le.Uint32(e[19:]) != 0 {
t.Errorf("reloc %d = off %d size %d type %d pkg %d sym %d", i, int32(le.Uint32(e[0:])), e[4], le.Uint16(e[5:]), le.Uint32(e[15:]), le.Uint32(e[19:]))
}
}
// The function code: three AUIPC+second-instruction pairs with zero
// immediates, then the uncompressed JALR X0, 0(X1) the toolchain emits
// for RET.
code := img.Code[fn.Offset : fn.Offset+fn.Size]
want := append(wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 10, 0)), wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 10, 10, 0))...)
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 11, 0))...)
want = append(want, wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 11, 11, 0))...)
want = append(want, wordLE(riscvUType(riscvEnc{0x17, 0x0, 0x00}, 31, 0))...)
want = append(want, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 12, 0))...)
want = append(want, 0x67, 0x80, 0x00, 0x00) // JALR X0, 0(X1)
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
// The same bytes must survive into the object's data block intact: the
// linker patches only the immediate fields of the AUIPC pairs, so the
// opcode/register bits of every instruction must not be zeroed.
dataIdx := v.blk(blkDataIdx)
dataBlk := v.blk(blkData)
dOff := int(le.Uint32(dataIdx[8*4:])) // the function is the last symbol
emitted := dataBlk[dOff : dOff+fn.Size]
if !bytes.Equal(emitted, want) {
t.Errorf("emitted data = % x\nwant % x", emitted, want)
}
}
// TestGOObjectRISCVLink cross-compiles a Go program with the gasm-produced
// object substituted into the package archive, proving cmd/link accepts the
// emitted RISC-V GOOBJ. The binary is not executed (no riscv64 host or
// qemu). Skipped when no Go toolchain is available.
func TestGOObjectRISCVLink(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOV a+0(FP), X10
MOV b+8(FP), X11
ADD X11, X10, X10
MOV X10, ret+16(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_riscv64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func add(a, b int64) int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module rvlink\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=riscv64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var pkgArch, work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_riscv64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if pkgArch == "" || linkLine == "" || asmObj == "" {
t.Skip("could not locate the archive, asm output or link line in the build log")
}
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work))
pf, perrs := parser.Parse(filepath.Join(dir, "main_riscv64.s"), asmSrc)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs)
}
pimg, err := AssembleFileRISCV(pf)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := pimg.GOObjectRISCV("main", filepath.Join(dir, "main_riscv64.s"))
if err != nil {
t.Fatalf("GOObjectRISCV: %v", err)
}
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
extract.Env = append(os.Environ(), "GOARCH=riscv64")
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, asmMember)
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listCmd.Env = append(os.Environ(), "GOARCH=riscv64")
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
pack.Env = append(os.Environ(), "GOARCH=riscv64")
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2"))
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=riscv64")
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with gasm object: %v\n%s", err, out)
}
nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2"))
nm.Env = append(os.Environ(), "GOARCH=riscv64")
nmOut, err := nm.CombinedOutput()
if err != nil {
t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut)
}
if !strings.Contains(string(nmOut), "main.add") {
t.Errorf("main.add not found in linked binary:\n%s", nmOut)
}
}
// hasELFRISCVJAL reports whether the ELF object carries an R_RISCV_JAL
// relocation in its .rela.text section.
func hasELFRISCVJAL(t *testing.T, data []byte) bool {
t.Helper()
f, err := elf.NewFile(bytes.NewReader(data))
if err != nil {
t.Fatalf("parse ELF: %v", err)
}
defer f.Close()
rela := f.Section(".rela.text")
if rela == nil {
return false
}
b, err := rela.Data()
if err != nil {
t.Fatalf(".rela.text data: %v", err)
}
const rRISCVJAL = 17
for i := 0; i+24 <= len(b); i += 24 {
info := binary.LittleEndian.Uint64(b[i+8:])
if uint32(info) == rRISCVJAL {
return true
}
}
return false
}
+45 -43
View File
@@ -36,11 +36,11 @@ const (
vexNDS3Imm
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
// source lives in the reg field, the destination in r/m — the PEXTR-style
// source lives in the reg field, the destination in r/m, the PEXTR-style
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m — the layout of the EVEX
// in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
@@ -68,7 +68,7 @@ type vexSpec struct {
// incrementally; every entry is covered by a byte-for-byte ground-truth test
// against the Go assembler.
var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare.
// VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
@@ -82,7 +82,7 @@ var vexTable = map[string]vexSpec{
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
// VEX.256.66.0F38.W0 — dword permute (three-operand NDS form).
// VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG.
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
@@ -90,14 +90,14 @@ var vexTable = map[string]vexSpec{
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
// VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
// VEX.128/256.0F.WIG, packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
@@ -107,7 +107,7 @@ var vexTable = map[string]vexSpec{
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
// VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
// opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
@@ -115,7 +115,7 @@ var vexTable = map[string]vexSpec{
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
// VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
// opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
@@ -123,10 +123,10 @@ var vexTable = map[string]vexSpec{
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src,
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
@@ -141,70 +141,72 @@ var vexTable = map[string]vexSpec{
"VPMOVZXWQ": {2, 0x34, 0, 1, -1, vexRM},
"VPBROADCASTD": {2, 0x58, 0, 1, -1, vexRM},
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG — signed dword to packed single conversion
// VEX.128/256.0F.WIG, signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG — packed single to packed double conversion
// VEX.128/256.0F.WIG, packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes — its machine code is the oracle, not the
// the Go assembler's bytes, its machine code is the oracle, not the
// manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
// VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
// VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
// VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift).
// VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
// VEX.128/256.66.0F.WIG — immediate shuffle (reg=dst, rm=src, imm8).
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1 — qword permute (reg=dst, rm=src, imm8).
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
// VEX.128/256.66.0F.WIG — two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8).
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — permute / insert (same shape; VINSERTI128's rm is
// VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
// the XMM or memory source).
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
// VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
// VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
// VEX.128.0F.W0 — no operands.
// VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
// VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
// VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
// VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
@@ -228,14 +230,14 @@ var vexTable = map[string]vexSpec{
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
// VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F — packed double to packed dword conversions, truncating and
// VEX.F2.0F, packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
@@ -255,7 +257,7 @@ var vexSrcLen = map[string]int{
"VCVTPD2PSY": 1,
}
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
// vexVarShift maps the shift mnemonics to their variable-count opcode, the
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
// an ordinary NDS encoding rather than the /digit immediate form above.
var vexVarShift = map[string]byte{
@@ -286,20 +288,20 @@ type vexMoveSpec struct {
// vexMoveTable maps an upper-case move mnemonic to its encoding.
var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG — unaligned integer move.
// VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG — unaligned packed double move.
// VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0 — 32-bit GPR/memory ↔ XMM.
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
// VMOVQ — 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
// VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
// register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
// VEX.128.F3.0F.WIG, scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256 — aligned packed moves.
// VEX.128/256, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
}
@@ -315,7 +317,7 @@ func isVex(mnemUpper string) bool {
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
// Vector register indices 16–31 exist only in EVEX encodings; fail
// Vector register indices 16-31 exist only in EVEX encodings; fail
// loudly rather than silently truncating the index.
for _, op := range ops {
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
@@ -418,7 +420,7 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
}
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source — fixed
// the destination always XMM and the VEX.L bit following the source, fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
+5 -5
View File
@@ -173,7 +173,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
// Floating point (packed and scalar) and FMA — same NDS form, the pp
// Floating point (packed and scalar) and FMA; same NDS form, the pp
// bits and map select the operation.
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
@@ -217,7 +217,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
// Moves — each direction picks its own opcode and VEX.W.
// Moves; each direction picks its own opcode and VEX.W.
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
@@ -234,7 +234,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
// Packed double arithmetic and unpack — the NDS form, the opcode
// Packed double arithmetic and unpack; the NDS form, the opcode
// selects the operation.
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
@@ -255,12 +255,12 @@ func TestVexGroundTruth(t *testing.T) {
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
// VMOVDDUP; duplicate the low double (reg=dst, rm=src, F2 pp).
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
// prefix — see the table comment), DQ→PD.
// prefix; see the table comment), DQ→PD.
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
+4 -5
View File
@@ -17,7 +17,7 @@ type File struct {
Orphans []Stmt // labels/instructions seen before any TEXT directive
// Macros holds the names introduced by #define directives in this file.
// The linter uses it to avoid flagging macro invocations as unknown
// instructions (macro expansion itself is out of scope — see the docs).
// instructions (macro expansion itself is out of scope, see the docs).
Macros map[string]bool
}
@@ -114,6 +114,7 @@ type Symbol struct {
Pkg string // package prefix before the middle dot ("" = current package)
Name string // identifier without the middle dot or <>
Static bool // the <> marker is present
ABI string // the <NAME> ABI marker, e.g. ABIInternal ("" when absent)
Pseudo string // FP, SP, SB or PC ("" for a bare name)
Offset int64
HasOff bool
@@ -123,10 +124,8 @@ type Symbol struct {
// OpKind classifies an operand syntactically.
type OpKind int
// Operand kinds.
const (
OpInvalid OpKind = iota
OpImmediate // $value
OpImmediate = iota // $value
OpAddr // register, memory reference, symbol or label
)
@@ -158,5 +157,5 @@ type Address struct {
Scale int // index scale; 0 when absent
Offset int64 // leading displacement, from off(base)
HasOff bool // a leading displacement is present
Shift string // verbatim arm64 shift suffix, e.g. "<<2"
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
}
+2 -2
View File
@@ -54,11 +54,11 @@ func TestStmtPositions(t *testing.T) {
// TestInterfaces confirms the node types satisfy their interfaces, so callers
// can range over Decls and Stmts.
func TestInterfaces(t *testing.T) {
var decls []Decl = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
var decls = []Decl{&Include{}, &Preproc{}, &Text{}, &Globl{}, &Data{}}
if len(decls) != 5 {
t.Fatal("decl interface set")
}
var stmts []Stmt = []Stmt{&Label{}, &Instr{}}
var stmts = []Stmt{&Label{}, &Instr{}}
if len(stmts) != 2 {
t.Fatal("stmt interface set")
}
+522
View File
@@ -0,0 +1,522 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"os"
"os/exec"
"path/filepath"
"regexp"
"runtime"
"slices"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
// own assembler, probed black-box: every mnemonic in the gasm table is offered
// to go tool asm in its bare form, and a mnemonic counts as known to Go when
// the error is anything but "unrecognized instruction" (a wrong-shape error
// still proves the mnemonic exists in Go's tables). The audit answers three
// questions at a glance:
//
// - which mnemonics gasm can encode that go tool asm does not know
// (superset encodings, usable only through the gasm goobj path);
// - which mnemonics the architecture table knows but the encoder cannot
// emit yet (the implementation backlog);
// - which mnemonics go tool asm knows that gasm cannot encode (feature
// gaps).
//
// The amd64 derived families (Jcc, CMOVcc, SETcc) exist on both sides by
// construction and are excluded from the diff; the other architectures list
// their conditional branches outright.
func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
Go side is probed black-box one bare mnemonic at a time, so the audit tracks
whatever toolchain `+"`go env GOROOT`"+` provides; the gasm side answers from
the encoder table on amd64 and from trial assembly over a battery of operand
shapes elsewhere. Names go tool asm knows and gasm does not cannot be
enumerated by probing, because Go's table is visible only through names
already in the gasm table; the report closes with a note saying so.
With --corpus the audit changes shape: it assembles every .s file under the
given directory (default GOROOT/src) with the gasm encoder only, no
toolchain probing. A file whose name carries a recognisable _arch suffix is
attempted for that architecture; a file without one is attempted for all
four, exactly as a GOARCH build would compile it. The report gives the
per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order.
`)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
if err := fs.Parse(args); err != nil {
return err
}
if *corpus {
return cmdAuditCorpus(fs.Args())
}
archName := "amd64"
switch n := len(fs.Args()); {
case n > 1:
return &usageError{fmt.Errorf("audit-instructions takes at most one architecture argument")}
case n == 1:
archName = strings.ToLower(fs.Arg(0))
}
a, err := auditArch(archName)
if err != nil {
return err
}
tab := arch.ForArch(a)
var names []string
seen := map[string]bool{}
for _, in := range tab.Instructions() {
name := strings.ToUpper(in.Name)
if a == arch.AMD64 && derivedFamily(name) || seen[name] {
continue
}
seen[name] = true
names = append(names, name)
}
goKnown, err := probeGoAsm(goarchName(a), names)
if err != nil {
return err
}
var superset, backlog, shared []string
for _, name := range names {
switch {
case !gasmEncodable(a, name):
backlog = append(backlog, name)
case !goKnown[name]:
superset = append(superset, name)
default:
shared = append(shared, name)
}
}
// GO-ONLY is not enumerable by probing: Go's table is only visible
// through names we already know, so nothing can be reported there.
slices.Sort(superset)
slices.Sort(backlog)
slices.Sort(shared)
w := os.Stdout
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognised: %d\n", len(shared)+len(superset), countTrue(goKnown))
fmt.Fprintf(w, "shared: %d\n", len(shared))
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
for _, n := range superset {
fmt.Fprintf(w, " %s\n", n)
}
fmt.Fprintf(w, "\nKnown but not encodable (backlog):\n")
for _, n := range backlog {
fmt.Fprintf(w, " %s\n", n)
}
fmt.Fprintf(w, "\nGo-only names cannot be enumerated by probing; extend the gasm\n")
fmt.Fprintf(w, "table from the Go release notes when a new instruction family ships.\n")
return nil
}
// auditArch resolves the audit's architecture argument.
func auditArch(name string) (arch.Arch, error) {
switch strings.ToLower(name) {
case "amd64":
return arch.AMD64, nil
case "arm64":
return arch.ARM64, nil
case "riscv64", "riscv":
return arch.RISCV, nil
case "loong64", "loong":
return arch.LOONG64, nil
}
return arch.Unknown, &usageError{fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)}
}
// goarchName maps an arch identifier onto its GOARCH spelling.
func goarchName(a arch.Arch) string {
switch a {
case arch.ARM64:
return "arm64"
case arch.RISCV:
return "riscv64"
case arch.LOONG64:
return "loong64"
}
return "amd64"
}
func countTrue(m map[string]bool) int {
n := 0
for _, v := range m {
if v {
n++
}
}
return n
}
// derivedFamily reports whether a mnemonic belongs to a family both
// assemblers construct from condition codes rather than list exhaustively
// (JEQ/CMOVLGT/SETNE and friends). Such names never probe cleanly, so
// including them in the diff would be noise. amd64 only: the other
// architectures list their conditional branches outright.
func derivedFamily(name string) bool {
if strings.HasPrefix(name, "J") && name != "JMP" && name != "JMPQ" {
return true
}
if strings.HasPrefix(name, "CMOV") || strings.HasPrefix(name, "SET") {
return true
}
return false
}
var unrecognizedRe = regexp.MustCompile(`unrecognized instruction`)
// probeGoAsm feeds every mnemonic to go tool asm in one generated file and
// classifies the diagnostics. "Unrecognized instruction" is a parse-stage
// verdict on the mnemonic alone, so a single bare-instruction probe per
// mnemonic decides recognition; the combined file still reports every line's
// error even when others fail.
func probeGoAsm(goarch string, names []string) (map[string]bool, error) {
dir, err := os.MkdirTemp("", "gasm-audit")
if err != nil {
return nil, err
}
defer os.RemoveAll(dir)
var sb strings.Builder
sb.WriteString("TEXT ·probe(SB), 4, $0\n\tRET\n")
lineMnemonic := map[int]string{}
line := 3
for _, name := range names {
fmt.Fprintf(&sb, "TEXT ·p%s%d(SB), 4, $0\n", sanitize(name), line)
sb.WriteString("\t" + name + "\n\tRET\n")
lineMnemonic[line+1] = name // the instruction line, after TEXT
line += 3
}
probePath := filepath.Join(dir, "probe.s")
if err := os.WriteFile(probePath, []byte(sb.String()), 0o644); err != nil {
return nil, err
}
toolDir, err := exec.Command("go", "env", "GOTOOLDIR").Output()
if err != nil {
return nil, fmt.Errorf("go env GOTOOLDIR: %w", err)
}
asmBin := filepath.Join(strings.TrimSpace(string(toolDir)), "asm")
if _, err := os.Stat(asmBin); err != nil {
return nil, fmt.Errorf("go tool asm not found at %s", asmBin)
}
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
out, _ := cmd.CombinedOutput()
// The expected failure mode is a non-zero exit with compiler diagnostics
// on stdout; empty output means the probe broke at the exec level (a
// killed child, a tool that would not start), and seeding every name as
// recognized on that silence would fake a clean audit.
if len(out) == 0 {
return nil, fmt.Errorf("go tool asm probe for GOARCH=%s produced no output", goarch)
}
result := map[string]bool{}
for _, name := range names {
result[name] = true // no news = the name parsed fine
}
reParse := regexp.MustCompile(`probe\.s:(\d+):`)
for l := range strings.SplitSeq(string(out), "\n") {
m := reParse.FindStringSubmatch(l)
if m == nil {
continue
}
lineNo, err := strconv.Atoi(m[1])
if err != nil {
continue
}
if name, ok := lineMnemonic[lineNo]; ok && unrecognizedRe.MatchString(l) {
result[name] = false
}
}
return result, nil
}
// probeShapes lists representative operand shapes for the encodability
// probe. The assemblers report an unknown mnemonic and a known mnemonic
// with no supported form alike ("unsupported <arch> instruction"), so only
// a shape that assembles cleanly counts, and the backlog over-approximates:
// a name whose real forms the battery misses lands there. amd64 keeps its
// exact table-driven check.
func probeShapes(a arch.Arch) []string {
switch a {
case arch.ARM64:
return []string{
"X0, X1, X2", "X0, X1", "X0", "$1, X0", "X0, (X1)", "(X0), X1",
"X0, (X1, 8)", "(SP), X0", "F0, F1, F2", "F0, F1", "F0",
"V0.B16, V1.B16, V2.B16", "p2", "X0, p2", "X0, X1, p2",
// The conditional select family spells the condition first
// and takes R register spellings.
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
"GE, F0, F1, F2", "NE, F0, F1, $0",
}
case arch.RISCV:
return []string{
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
"X5, X6, p2", "p2(SB)",
}
case arch.LOONG64:
return []string{
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
}
}
return nil
}
// gasmEncodable reports whether the gasm encoder for a can emit the
// mnemonic, decided by trial assembly over the shape battery.
func gasmEncodable(a arch.Arch, name string) bool {
switch a {
case arch.ARM64, arch.RISCV, arch.LOONG64:
default:
return asm.Encodable(name)
}
for _, shape := range probeShapes(a) {
if gasmAssembles(a, name, shape) {
return true
}
}
return false
}
// gasmAssembles reports whether a one-instruction probe file containing name
// with the given operand shape assembles without error.
func gasmAssembles(a arch.Arch, name, shape string) bool {
src := "TEXT ·p(SB), NOSPLIT, $0\n\t" + name
if shape != "" {
src += " " + shape
}
src += "\n\tRET\np2:\n\tRET\n"
f, errs := parser.Parse("probe.s", src)
if len(errs) > 0 {
return false
}
var err error
switch a {
case arch.ARM64:
_, err = asm.AssembleFileARM64(f)
case arch.RISCV:
_, err = asm.AssembleFileRISCV(f)
case arch.LOONG64:
_, err = asm.AssembleFileLOONG64(f)
}
return err == nil
}
// sanitize makes a mnemonic safe for use in a Go symbol name.
func sanitize(name string) string {
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
}
// --- corpus audit -----------------------------------------------------------
// corpusTarget is one architecture row of the corpus report.
type corpusTarget struct {
a arch.Arch
name string
}
// corpusTally accumulates one architecture's attempts over the corpus.
type corpusTally struct {
attempted int
assembled int
reasons map[string]int // failure reason → count
example map[string]string // failure reason → one representative file
}
func (t *corpusTally) fail(path, reason string) {
t.reasons[reason]++
if t.example[reason] == "" {
t.example[reason] = path
}
}
// cmdAuditCorpus implements audit-instructions --corpus.
func cmdAuditCorpus(args []string) error {
if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
}
root := ""
if len(args) == 1 {
root = args[0]
} else {
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
return fmt.Errorf("locate GOROOT: %w", err)
}
root = filepath.Join(strings.TrimSpace(string(out)), "src")
}
stats, err := runCorpusAudit(root)
if err != nil {
return err
}
printCorpusStats(stats)
return nil
}
// corpusStats is the outcome of one corpus audit run.
type corpusStats struct {
root string
files int
generic int // files attempted for all four architectures
full int // files that assembled for every target architecture
targets []corpusTarget
tallies []*corpusTally
}
// runCorpusAudit assembles every .s file under root and returns the stats.
func runCorpusAudit(root string) (*corpusStats, error) {
files, err := asmFiles(root)
if err != nil {
return nil, err
}
targets := []corpusTarget{
{arch.AMD64, "amd64"},
{arch.ARM64, "arm64"},
{arch.RISCV, "riscv64"},
{arch.LOONG64, "loong64"},
}
tallies := make([]*corpusTally, len(targets))
for i := range tallies {
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
}
// full is the north-star number: a file counts when every architecture
// its name allows assembles it.
full, generic := 0, 0
for _, path := range files {
src, err := readSource(path)
if err != nil {
return nil, err
}
f, errs := parser.Parse(path, src)
var wanted []int // indexes into targets
if a := arch.FromFilename(path); a != arch.Unknown {
for i, tg := range targets {
if tg.a == a {
wanted = append(wanted, i)
}
}
} else {
generic++
for i := range targets {
wanted = append(wanted, i)
}
}
ok := true
for _, i := range wanted {
tg, t := targets[i], tallies[i]
t.attempted++
var err error
if len(errs) > 0 {
err = errs[0] // a parse failure is a failure for every target
} else {
_, err = assembleFile(tg.a, f)
}
if err != nil {
ok = false
t.fail(path, corpusReason(err))
continue
}
t.assembled++
}
if ok && len(wanted) > 0 {
full++
}
}
return &corpusStats{
root: root,
files: len(files),
generic: generic,
full: full,
targets: targets,
tallies: tallies,
}, nil
}
// printCorpusStats renders the corpus audit report.
func printCorpusStats(s *corpusStats) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures)\n", s.root, s.files, s.generic)
fmt.Printf(" assemble for every target architecture: %d (%.1f%%)\n", s.full, 100*float64(s.full)/float64(max(s.files, 1)))
for i, tg := range s.targets {
t := s.tallies[i]
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
for _, r := range topReasons(t) {
fmt.Printf(" %4d %s\n", t.reasons[r], r)
fmt.Printf(" e.g. %s\n", t.example[r])
}
}
}
// corpusReason buckets an assembly or parse failure for the histogram.
func corpusReason(err error) string {
msg := err.Error()
switch {
case strings.Contains(msg, "unsupported"), strings.Contains(msg, "cannot encode"):
return "instruction not encodable"
case strings.Contains(msg, "undefined label"):
return "undefined label"
case strings.Contains(msg, "undefined symbol"), strings.Contains(msg, "external symbol"), strings.Contains(msg, "file-level assembly"):
return "undefined symbol or external"
case strings.Contains(msg, "operand"), strings.Contains(msg, "operand form"):
return "unsupported operand form"
default:
return "other: " + firstLine(msg)
}
}
// topReasons returns at most five reasons, most frequent first.
func topReasons(t *corpusTally) []string {
type kv struct {
k string
n int
}
var kvs []kv
for k, n := range t.reasons {
kvs = append(kvs, kv{k, n})
}
slices.SortFunc(kvs, func(a, b kv) int { return b.n - a.n })
if len(kvs) > 5 {
kvs = kvs[:5]
}
out := make([]string, len(kvs))
for i, kv := range kvs {
out[i] = kv.k
}
return out
}
// firstLine returns the first line of an error message, truncated.
func firstLine(msg string) string {
if i := strings.IndexByte(msg, '\n'); i >= 0 {
msg = msg[:i]
}
if len(msg) > 80 {
msg = msg[:80]
}
return msg
}
+66
View File
@@ -0,0 +1,66 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
)
func TestDerivedFamily(t *testing.T) {
for _, n := range []string{"JEQ", "JLT", "JCC", "CMOVLGT", "SETNE", "SETA"} {
if !derivedFamily(n) {
t.Errorf("derivedFamily(%q) = false, want true", n)
}
}
for _, n := range []string{"JMP", "ADDQ", "VPGATHERDD", "MOVBE", "PSHUFB"} {
if derivedFamily(n) {
t.Errorf("derivedFamily(%q) = true, want false", n)
}
}
}
func TestSanitize(t *testing.T) {
if got := sanitize("VPCMP.UB"); got != "VPCMP_UB" {
t.Errorf("sanitize: got %q", got)
}
}
func TestAuditArch(t *testing.T) {
for in, want := range map[string]arch.Arch{
"amd64": arch.AMD64, "arm64": arch.ARM64,
"riscv64": arch.RISCV, "riscv": arch.RISCV,
"loong64": arch.LOONG64, "LOONG": arch.LOONG64,
} {
got, err := auditArch(in)
if err != nil || got != want {
t.Errorf("auditArch(%q) = %v, %v; want %v", in, got, err, want)
}
}
if _, err := auditArch("mips"); err == nil {
t.Error("auditArch(mips) must fail")
}
}
func TestGasmEncodable(t *testing.T) {
cases := []struct {
a arch.Arch
yes string
no string
}{
{arch.AMD64, "ADDQ", "NOSUCHMNEMONIC"},
{arch.ARM64, "ADD", "NOSUCHMNEMONIC"},
{arch.RISCV, "ADD", "NOSUCHMNEMONIC"},
{arch.LOONG64, "ADDV", "NOSUCHMNEMONIC"},
}
for _, c := range cases {
if !gasmEncodable(c.a, c.yes) {
t.Errorf("%s: %s should be encodable", c.a, c.yes)
}
if gasmEncodable(c.a, c.no) {
t.Errorf("%s: %s should not be encodable", c.a, c.no)
}
}
}
+336
View File
@@ -0,0 +1,336 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package main
import (
"fmt"
"io"
"os"
"sort"
"strings"
"time"
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
func cmdDebug(args []string) int {
fs := newCommand("debug", "gasm debug <file.s> --func <name>", `
Interactive debugger for JIT-assembled functions. Launches the
function in a traced subprocess (ptrace), then provides a REPL for
single-stepping, breakpoints, register and memory inspection.
REPL commands:
break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
set a breakpoint, optionally conditional on a
comparison of one register against a constant,
another register, or the 8-byte word at *addr
delete <label|addr> remove a breakpoint
info break list all breakpoints
step [n], s single-step n instructions (default 1)
next, n step over a CALL
finish, fin run until the function returns
continue, c run until a breakpoint, watchpoint or exit
disas [n], u disassemble n instructions at PC
regs print general-purpose and vector registers
where show source line and nearest label at PC
stack show stack near RSP (return address + ABI0 args)
bt, backtrace backtrace (current frame + return address)
x [addr] [len] hex-dump memory (default: current PC, 64 bytes)
w <addr> <val...> write bytes to memory
set <reg> <value> set a register
watch <addr> [r|w] [size]
set a hardware watchpoint (write by default)
unwatch [<slot>] clear one watchpoint, or all without an argument
labels, l list function labels and offsets
help, h, ? show command help
quit, q kill the debuggee and exit
`)
funcName := fs.String("func", "", "function to debug")
argsFile := fs.String("args", "", "file containing the ABI0 argument block")
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs; a timeout exits 3")
fs.Parse(args)
// --- Debuggee mode (internal, spawned by the debugger) ---
if os.Getenv("GASM_DEBUG_TARGET") != "" {
tmpDir := os.Getenv("GASM_DEBUG_TMP")
if tmpDir == "" || fs.NArg() < 1 || *funcName == "" || *argsFile == "" {
fmt.Fprintln(os.Stderr, "gasm debug: internal debuggee mode")
return 2
}
if err := debug.RunTarget(fs.Arg(0), *funcName, *argsFile, tmpDir); err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
return 0
}
// --- Debugger mode (interactive REPL) ---
if fs.NArg() < 1 || *funcName == "" {
fmt.Fprintln(os.Stderr, "usage: gasm debug <file.s> --func <name>")
return 2
}
path := fs.Arg(0)
// The watchdog is armed before anything can block: ptrace attach and a
// continued kernel loop both hang the run when the environment forbids
// tracing or the kernel loops forever, and neither is interruptible from
// the inside.
if *timeout > 0 {
go func() {
time.Sleep(*timeout)
fmt.Fprintf(os.Stderr, "gasm debug: timeout (%s), killing the debuggee\n", *timeout)
os.Exit(3)
}()
}
// Load the kernel to extract function metadata and labels.
k, err := verify.Load(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
defer k.Close()
fl, err := k.Func(*funcName)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
// Build the label list for the REPL.
var labels []debug.Label
for name, off := range fl.Labels {
labels = append(labels, debug.Label{Name: name, Offset: off})
}
sort.Slice(labels, func(i, j int) bool { return labels[i].Offset < labels[j].Offset })
// Launch the debuggee with the argument block.
var argBlock []byte
var bufAddrs []uint64
var sess *debug.Session
if *bufSpec != "" {
// Parse the function signature to determine argument layout.
src, err := readSource(path)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
sig, ok := verify.ExtractFuncSig(src, *funcName)
if !ok {
fmt.Fprintf(os.Stderr, "gasm debug: no // func signature found for %s\n", *funcName)
return 1
}
layout := verify.ArgLayout(sig)
// Parse the buffer spec to get buffer names.
bufNames := parseBufNames(*bufSpec)
// Allocate buffers in the debuggee.
argBlock = make([]byte, fl.Args)
sess, bufAddrs, err = debug.LaunchWithBuffers("", path, *funcName, argBlock, *bufSpec)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
// Construct the argument block with buffer pointers at the correct positions.
for _, arg := range layout {
if !arg.IsPtr {
continue
}
// Find the buffer that matches this argument.
for i, name := range bufNames {
if i < len(bufAddrs) && (name == arg.Name || strings.HasPrefix(arg.Name, name)) {
addr := bufAddrs[i]
off := arg.Offset
if off+8 <= len(argBlock) {
argBlock[off] = byte(addr)
argBlock[off+1] = byte(addr >> 8)
argBlock[off+2] = byte(addr >> 16)
argBlock[off+3] = byte(addr >> 24)
argBlock[off+4] = byte(addr >> 32)
argBlock[off+5] = byte(addr >> 40)
argBlock[off+6] = byte(addr >> 48)
argBlock[off+7] = byte(addr >> 56)
}
// For slices, also set the length and capacity.
if strings.HasPrefix(arg.Typ, "[]") && off+24 <= len(argBlock) {
// Find the buffer size from the spec.
size := parseBufSize(*bufSpec, name)
// Length at offset+8, capacity at offset+16.
for j := range 8 {
argBlock[off+8+j] = byte(size >> (j * 8))
argBlock[off+16+j] = byte(size >> (j * 8))
}
}
break
}
}
}
} else {
argBlock = make([]byte, fl.Args)
sess, err = debug.Launch("", path, *funcName, argBlock)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
}
defer sess.Kill()
bm := debug.NewBreakpoints(sess)
fmt.Printf("gasm debug: %s in %s (pid %d)\n", *funcName, path, sess.Pid())
// Convert the line table for the command loop.
var srcLines []debug.SourceLine
for _, le := range fl.Lines {
srcLines = append(srcLines, debug.SourceLine{Offset: le.Offset, Line: le.Line})
}
// Coverage mode: pre-register a breakpoint on every instruction (walked
// by length through the function body while the debuggee is stopped) and
// let the kernel run to completion. Each trap counts a hit for that
// instruction, so the final report shows exactly which instructions
// executed and how often, with the label-level view derived from it.
// Expect the run to slow to ptrace speed: one trap per executed
// instruction.
if *cover {
base := sess.CodeBase() + uint64(fl.Offset)
type coverInstr struct {
off uint64
text string
}
var instrs []coverInstr
for off := uint64(0); off < uint64(fl.Size); {
text, ln, err := sess.Disassemble(base + off)
if err != nil || ln == 0 {
break
}
instrs = append(instrs, coverInstr{off: off, text: text})
off += uint64(ln)
}
for _, in := range instrs {
if _, err := bm.SetWithCond(base+in.off, fmt.Sprintf("func+%#x", in.off), nil); err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: cover: %v\n", err)
return 1
}
}
fmt.Printf("gasm debug: coverage run over %d instructions\n", len(instrs))
for {
for _, bp := range bm.All() {
bm.Reinsert(bp.Addr)
}
if err := sess.Continue(); err != nil {
break // debuggee finished or died
}
if sess.Exited() {
break
}
// A genuine signal-delivery-stop (a fault in the kernel): the
// run cannot make progress, because resuming would restart the
// faulting instruction and fault forever. Report and stop.
if sig := sess.LastSignal(); sig != 0 {
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
break
}
regs, rerr := sess.GetRegs()
if rerr != nil {
break
}
// HandleTrap restores the original byte, rewinds PC and counts
// the hit on the breakpoint itself. Single-step over the
// restored instruction so the reinsertion at the top of the
// loop cannot re-trap on the same breakpoint.
if bp := bm.HandleTrap(&regs); bp != nil {
if err := sess.Step(); err != nil {
break
}
}
}
hits := map[uint64]int{}
traps := 0
for _, bp := range bm.All() {
if n := bp.Hits(); n > 0 {
hits[bp.Addr-base] = n
traps += n
}
}
var hit []string
var missed []string
for _, l := range labels {
if hits[uint64(l.Offset)] > 0 {
hit = append(hit, l.Name)
} else {
missed = append(missed, l.Name)
}
}
sort.Strings(hit)
sort.Strings(missed)
fmt.Printf("coverage: %d/%d instructions executed (%d traps)\n", len(hits), len(instrs), traps)
fmt.Printf("coverage: %d/%d labels reached\n", len(hit), len(labels))
for _, l := range hit {
fmt.Printf(" covered %s\n", l)
}
for _, l := range missed {
fmt.Printf(" MISSED %s\n", l)
}
fmt.Println("executed instructions:")
for _, in := range instrs {
if n := hits[in.off]; n > 0 {
fmt.Printf(" func+%#04x %4dx %s\n", in.off, n, in.text)
}
}
return 0
}
// Headless mode: run the script through the normal command loop and
// exit. The watchdog armed above covers launch, continue and step.
var in io.Reader = os.Stdin
if *script != "" {
if *script == "-" {
in = os.Stdin
} else {
f, err := os.Open(*script)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm debug: %v\n", err)
return 1
}
defer f.Close()
in = f
}
}
debug.REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, labels, srcLines, in)
return 0
}
// parseBufNames extracts buffer names from a buffer specification.
// Format: name:size:pattern[,name:size:pattern...]
func parseBufNames(spec string) []string {
var names []string
for part := range strings.SplitSeq(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) >= 1 && fields[0] != "" {
names = append(names, fields[0])
}
}
return names
}
// parseBufSize extracts the size of a named buffer from a buffer specification.
func parseBufSize(spec, name string) int {
for part := range strings.SplitSeq(spec, ",") {
fields := strings.SplitN(part, ":", 3)
if len(fields) >= 2 && fields[0] == name {
var size int
fmt.Sscanf(fields[1], "%d", &size)
return size
}
}
return 0
}
+16
View File
@@ -0,0 +1,16 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build !linux
package main
import (
"fmt"
"os"
)
func cmdDebug(args []string) int {
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux (ptrace)")
return 1
}
+148
View File
@@ -0,0 +1,148 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"os"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// cmdDis disassembles machine code: either a raw binary (standard input with
// "-") whose architecture is given with -a, or a .s file, which is assembled
// first so the listing shows the real function and label layout.
func cmdDis(args []string) int {
fs := newCommand("dis", "gasm dis [-a arch] <file>", `
Disassemble machine code to instruction text (via golang.org/x/arch).
With a .s file, the file is assembled first and the listing follows the
real layout: one block per TEXT function, local labels printed at their
offsets. The architecture comes from the file name suffix, or from -a.
With any other file, or "-" for standard input, the bytes are disassembled
linearly and -a selects the architecture (amd64, arm64, riscv64 or
loong64).
`)
archName := fs.String("a", "", "architecture for raw input: amd64, arm64, riscv64 or loong64")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm dis [-a arch] <file>")
return 2
}
path := fs.Arg(0)
var target arch.Arch
if *archName != "" {
var err error
target, err = auditArch(*archName)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 2
}
}
if strings.HasSuffix(path, ".s") {
if target == arch.Unknown {
target = arch.FromFilename(path)
}
if target == arch.Unknown {
fmt.Fprintln(os.Stderr, "gasm dis: cannot infer the architecture from the file name; use -a")
return 2
}
return disSource(path, target)
}
if target == arch.Unknown {
fmt.Fprintln(os.Stderr, "gasm dis: raw input needs -a (amd64, arm64, riscv64 or loong64)")
return 2
}
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm dis:", err)
return 1
}
printListing(target, []byte(src), 0, nil)
return 0
}
// disSource assembles a .s file and prints one listing block per function.
func disSource(path string, target arch.Arch) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm dis:", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := assembleFile(target, f)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 1
}
if len(img.Funcs) == 0 {
fmt.Fprintln(os.Stderr, "gasm dis: no assemblable TEXT functions found")
return 1
}
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
labels := make(map[int][]string, len(fn.Labels))
for name, off := range fn.Labels {
labels[off] = append(labels[off], name)
}
for off := range labels {
sort.Strings(labels[off])
}
printListing(target, code, uint64(fn.Offset), labels)
}
if len(img.Data) > 0 {
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
}
return 0
}
// printListing decodes code linearly from offset base, printing label lines
// (label name to offset within the block) as they are reached.
func printListing(a arch.Arch, code []byte, base uint64, labels map[int][]string) {
pc := 0
for pc < len(code) {
for _, name := range labels[pc] {
fmt.Printf("%s:\n", name)
}
ins, err := disasm.Decode(a, code[pc:], base+uint64(pc))
if err != nil {
break
}
end := min(pc+ins.Len, len(code))
fmt.Printf(" %04x: %-16s %s\n", base+uint64(pc), hexBytes(code[pc:end]), ins.Text)
if ins.Len <= 0 {
break
}
pc += ins.Len
}
}
// hexBytes renders up to 8 bytes as contiguous hex.
func hexBytes(b []byte) string {
var sb strings.Builder
for i, c := range b {
if i == 8 {
break
}
if i > 0 {
sb.WriteByte(' ')
}
fmt.Fprintf(&sb, "%02x", c)
}
return sb.String()
}
+1241 -116
View File
File diff suppressed because it is too large Load Diff
+225 -2
View File
@@ -7,9 +7,15 @@ import (
"bytes"
"io"
"os"
"os/exec"
"path/filepath"
"runtime"
"strings"
"syscall"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
)
const clean = "#include \"textflag.h\"\n" +
@@ -216,8 +222,9 @@ func TestCmdVersion(t *testing.T) {
if code != 0 {
t.Fatalf("code = %d", code)
}
if !strings.Contains(out, version) {
t.Errorf("version output %q does not mention %q", out, version)
got := version()
if !strings.Contains(out, got) {
t.Errorf("version output %q does not mention %q", out, got)
}
}
@@ -237,3 +244,219 @@ func TestCmdArgErrors(t *testing.T) {
t.Errorf("cmdParse() code = %d, want 2", code)
}
}
// TestUsageExitCodes pins the exit-code contract for the commands whose main
// dispatches on a returned error: a wrong argument set exits 2, the same as
// the commands that count their arguments themselves, while a runtime
// failure (an unreadable file) keeps exit 1.
func TestUsageExitCodes(t *testing.T) {
for name, err := range map[string]error{
"audit-instructions extra argument": cmdAuditInstructions([]string{"amd64", "extra"}),
"audit-instructions unknown arch": cmdAuditInstructions([]string{"mips"}),
"audit-instructions corpus extra": cmdAuditInstructions([]string{"--corpus", "a", "b"}),
"scaffold no arguments": cmdScaffold(nil),
"scaffold extra arguments": cmdScaffold([]string{"differential", "a.s", "b.s"}),
} {
if err == nil {
t.Errorf("%s: expected an error", name)
continue
}
if code := exitCodeFor(err); code != 2 {
t.Errorf("%s: exit code = %d, want 2 (err: %v)", name, code, err)
}
}
if err := cmdScaffold([]string{"differential", "/nonexistent/file.s"}); err == nil {
t.Error("scaffold on a missing file should fail")
} else if code := exitCodeFor(err); code != 1 {
t.Errorf("scaffold on a missing file: exit code = %d, want 1", code)
}
}
// TestCmdAsmFormatValidation checks that an unknown --format exits 2 with
// and without -o, instead of assembling and silently dumping a raw image.
func TestCmdAsmFormatValidation(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out := filepath.Join(t.TempDir(), "f.bin")
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", path}) }); code != 2 {
t.Errorf("asm --format bogus without -o: code = %d, want 2", code)
}
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", "-o", out, path}) }); code != 2 {
t.Errorf("asm --format bogus with -o: code = %d, want 2", code)
}
}
// TestCmdAsmOutputFile pins the documented -o behaviour: the output goes to
// the file and stdout carries no hex dump; without -o the dump is the output.
func TestCmdAsmOutputFile(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out := filepath.Join(t.TempDir(), "f.bin")
stdout, _, code := capture(func() int { return cmdAsm([]string{"-o", out, path}) })
if code != 0 {
t.Fatalf("code = %d", code)
}
if strings.Contains(stdout, "0000:") {
t.Errorf("stdout carries a hex dump despite -o:\n%s", stdout)
}
if !strings.Contains(stdout, "wrote ") {
t.Errorf("stdout misses the wrote line:\n%s", stdout)
}
b, err := os.ReadFile(out)
if err != nil {
t.Fatal(err)
}
if len(b) == 0 {
t.Error("the output file is empty")
}
stdout, _, code = capture(func() int { return cmdAsm([]string{path}) })
if code != 0 {
t.Fatalf("without -o: code = %d", code)
}
if !strings.Contains(stdout, "0000:") {
t.Errorf("without -o the hex dump is missing:\n%s", stdout)
}
}
// TestVerifyNonJITAMD64GroundTruth drives the cross-architecture
// ground-truth path for an amd64 kernel: the path a host of any other
// architecture takes, which must compare against the toolchain rather than
// refuse to run.
func TestVerifyNonJITAMD64GroundTruth(t *testing.T) {
if testing.Short() {
t.Skip("runs go tool asm")
}
path := writeTemp(t, "f_amd64.s", clean)
out, _, code := capture(func() int { return cmdVerifyNonJIT(path, arch.AMD64, true, false) })
if code != 0 {
t.Fatalf("code = %d (%s)", code, out)
}
if !strings.Contains(out, "1/1 matched") {
t.Errorf("output misses the matched report:\n%s", out)
}
}
// TestVerifySmokeCrashIsolation checks that a function faulting on its
// zeroed smoke arguments is reported as CRASH by a child process instead of
// killing `gasm verify` itself.
func TestVerifySmokeCrashIsolation(t *testing.T) {
if testing.Short() {
t.Skip("builds the gasm binary")
}
if runtime.GOARCH != "amd64" {
t.Skip("amd64 JIT only")
}
bin := filepath.Join(t.TempDir(), "gasm")
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
t.Fatalf("build gasm: %v\n%s", err, out)
}
src := filepath.Join(t.TempDir(), "crash_amd64.s")
kernel := "#include \"textflag.h\"\n" +
"\n" +
"// func Fault(x []byte) int\n" +
"TEXT ·Fault(SB), NOSPLIT, $0-32\n" +
"\tMOVQ\tx+0(FP), AX\n" +
"\tMOVQ\t(AX), AX // faults on the zeroed nil pointer\n" +
"\tMOVQ\tAX, ret+24(FP)\n" +
"\tRET\n"
if err := os.WriteFile(src, []byte(kernel), 0o644); err != nil {
t.Fatal(err)
}
cmd := exec.Command(bin, "verify", "-smoke", src)
out, err := cmd.CombinedOutput()
if err == nil {
t.Fatalf("expected a failure report, got success:\n%s", out)
}
if exitErr, ok := err.(*exec.ExitError); ok {
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
t.Fatalf("verify died from %v; the crash was not isolated:\n%s", ws.Signal(), out)
}
}
if !strings.Contains(string(out), "CRASH") {
t.Errorf("output does not report CRASH:\n%s", out)
}
}
func TestSweepCheckLines(t *testing.T) {
out := []byte("crash_amd64.s: 1 functions JIT-loaded\n" +
" Fault: 21 bytes, args=32, frame=0 NOSPLIT\n" +
" smoke: OK\n" +
" abi: clean (10 varied inputs)\n")
want := " smoke: OK\n abi: clean (10 varied inputs)"
if got := sweepCheckLines(out); got != want {
t.Errorf("sweepCheckLines = %q, want %q", got, want)
}
}
// TestRunCorpusAudit drives the corpus audit over a small fixture tree: one
// suffixed amd64 file, one suffixed arm64 file whose body is not arm64, one
// generic file, and one file that does not parse.
func TestRunCorpusAudit(t *testing.T) {
dir := t.TempDir()
write := func(name, src string) {
t.Helper()
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
write("good_amd64.s", "#include \"textflag.h\"\nTEXT ·add(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
write("bad_arm64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
stats, err := runCorpusAudit(dir)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
if stats.files != 4 {
t.Errorf("files = %d, want 4", stats.files)
}
if stats.generic != 2 {
t.Errorf("generic = %d, want 2 (generic.s and broken.s)", stats.generic)
}
// good_amd64 and generic.s assemble everywhere they are attempted.
if stats.full != 2 {
t.Errorf("full = %d, want 2", stats.full)
}
get := func(name string) *corpusTally {
for i, tg := range stats.targets {
if tg.name == name {
return stats.tallies[i]
}
}
t.Fatalf("no tally for %s", name)
return nil
}
// amd64: good_amd64 + generic.s + broken.s; the broken file fails to parse.
if a := get("amd64"); a.attempted != 3 || a.assembled != 2 {
t.Errorf("amd64 = %d/%d, want 2/3", a.assembled, a.attempted)
}
// arm64: bad_arm64 (MOVQ is not arm64) + generic.s + broken.s.
if a := get("arm64"); a.attempted != 3 || a.assembled != 1 {
t.Errorf("arm64 = %d/%d, want 1/3", a.assembled, a.attempted)
}
if r := get("amd64").reasons["instruction not encodable"]; r != 0 {
t.Errorf("amd64 unexpected unencodable reason: %d", r)
}
if r := get("arm64").reasons["instruction not encodable"]; r != 1 {
t.Errorf("arm64 unencodable reasons = %d, want 1", r)
}
}
// TestCompareGroundTruthPadding pins the padding-aware ground-truth
// comparison: the toolchain pads text symbols to 16-byte boundaries, so
// trailing zeros in the reference must not read as a mismatch, while any
// non-zero tail still must.
func TestCompareGroundTruthPadding(t *testing.T) {
code := []byte{0x48, 0x8b, 0x07, 0xc3} // 4 bytes, not a multiple of 16
img := &asm.Image{Code: code, Funcs: []asm.FuncLayout{{Name: "f", Offset: 0, Size: len(code)}}}
padded := append(append([]byte(nil), code...), 0, 0, 0)
matched, total, diffs := compareGroundTruth(img, map[string][]byte{"f": padded})
if matched != 1 || total != 1 || diffs != 0 {
t.Fatalf("zero padding should match: matched=%d total=%d diffs=%d", matched, total, diffs)
}
dirty := append(append([]byte(nil), code...), 0, 0x90, 0)
matched, _, diffs = compareGroundTruth(img, map[string][]byte{"f": dirty})
if matched != 0 || diffs != 1 {
t.Fatalf("non-zero padding must mismatch: matched=%d diffs=%d", matched, diffs)
}
}
+241
View File
@@ -0,0 +1,241 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"os"
"os/exec"
"path/filepath"
"regexp"
"strings"
"testing"
)
// TestManPagesTrackTheCLI builds the binary once, then compares every
// command's live `-h` output with its docs/man/gasm-<command>.1 page: the
// flag sets must agree both ways, and the page's SYNOPSIS line must carry
// the command's usage line. A flag or a usage change that skips the man
// page fails here, so the pages cannot drift from the binary.
func TestManPagesTrackTheCLI(t *testing.T) {
if testing.Short() {
t.Skip("builds the gasm binary")
}
bin := filepath.Join(t.TempDir(), "gasm")
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
t.Fatalf("build gasm: %v\n%s", err, out)
}
for _, cmd := range []string{
"tokens", "parse", "fmt", "lint", "asm", "dis", "verify",
"debug", "diff", "profile", "audit-instructions", "scaffold", "lsp",
} {
t.Run(cmd, func(t *testing.T) {
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm-"+cmd+".1"))
if err != nil {
t.Fatalf("read man page: %v", err)
}
page := string(raw)
out, _ := exec.Command(bin, cmd, "-h").CombinedOutput()
help := string(out)
binFlags := helpFlags(help)
pageFlags := roffFlags(page)
for f := range binFlags {
if !pageFlags[f] {
t.Errorf("flag -%s is in the binary's help but missing from the man page", f)
}
}
for f := range pageFlags {
if !binFlags[f] {
t.Errorf("flag -%s is in the man page but the binary does not accept it", f)
}
}
want := helpUsage(help)
got := roffSynopsis(page)
if want != "" && got != want {
t.Errorf("SYNOPSIS drift:\n page: %s\nbinary: %s", got, want)
}
})
}
}
// TestManCommandsTrackHelp compares the gasm(1) COMMANDS list with the
// top-level help output, so a subcommand added to the binary cannot miss
// its man entry and a stale entry cannot outlive its command.
func TestManCommandsTrackHelp(t *testing.T) {
if testing.Short() {
t.Skip("builds the gasm binary")
}
bin := filepath.Join(t.TempDir(), "gasm")
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
t.Fatalf("build gasm: %v\n%s", err, out)
}
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm.1"))
if err != nil {
t.Fatalf("read man page: %v", err)
}
helpOut, err := exec.Command(bin, "--help").Output()
if err != nil {
t.Fatalf("gasm --help: %v", err)
}
binCmds := helpCommands(string(helpOut))
pageCmds := roffCommands(string(raw))
for c := range binCmds {
if !pageCmds[c] {
t.Errorf("command %q is in the binary's help but missing from gasm(1) COMMANDS", c)
}
}
for c := range pageCmds {
if !binCmds[c] {
t.Errorf("command %q is in gasm(1) COMMANDS but the binary does not list it", c)
}
}
}
// helpFlags extracts the flag names from a `gasm <cmd> -h` output.
func helpFlags(help string) map[string]bool {
m := map[string]bool{}
inFlags := false
for line := range strings.SplitSeq(help, "\n") {
if strings.TrimRight(line, " \t") == "Flags:" {
inFlags = true
continue
}
if !inFlags {
continue
}
if !strings.HasPrefix(line, " -") {
continue
}
token := strings.FieldsFunc(strings.TrimLeft(line, " "), func(r rune) bool {
return r == ' ' || r == '\t'
})
if len(token) == 0 {
continue
}
m[strings.TrimLeft(token[0], "-")] = true
}
return m
}
var roffEscape = regexp.MustCompile(`\\f[BIRP]`)
// roffFlags extracts the flag names from a man page's OPTIONS section.
func roffFlags(page string) map[string]bool {
m := map[string]bool{}
inOptions := false
for line := range strings.SplitSeq(page, "\n") {
if strings.HasPrefix(line, ".SH ") {
inOptions = strings.HasPrefix(line, ".SH OPTIONS")
continue
}
if !inOptions {
continue
}
// Flag entries are written as either `.B \-flag` or `\fB\-flag`.
var body string
switch {
case strings.HasPrefix(line, `.B \-`):
body = line[3:]
case strings.HasPrefix(line, `\fB\-`):
body = line[1:]
default:
continue
}
name := roffEscape.ReplaceAllString(body, "")
name = strings.ReplaceAll(name, `\-`, "-")
name = strings.TrimSpace(name)
if i := strings.IndexAny(name, " \t"); i >= 0 {
name = name[:i]
}
m[strings.TrimLeft(name, "-")] = true
}
return m
}
// helpCommands extracts the command names from the top-level help output's
// Commands section.
func helpCommands(help string) map[string]bool {
m := map[string]bool{}
inCmds := false
for line := range strings.SplitSeq(help, "\n") {
if strings.TrimSpace(line) == "Commands:" {
inCmds = true
continue
}
if !inCmds {
continue
}
t := strings.TrimSpace(line)
if t == "" {
break
}
name, _, _ := strings.Cut(t, " ")
m[name] = true
}
return m
}
// roffCommands extracts the command names from gasm(1)'s COMMANDS section,
// where each entry is written as `.B gasm\-<name>(1)` or `.B gasm <name>`.
func roffCommands(page string) map[string]bool {
m := map[string]bool{}
inCmds := false
for line := range strings.SplitSeq(page, "\n") {
if strings.HasPrefix(line, ".SH ") {
inCmds = strings.HasPrefix(line, ".SH COMMANDS")
continue
}
if !inCmds || !strings.HasPrefix(line, ".B gasm") {
continue
}
entry := strings.ReplaceAll(strings.TrimPrefix(line, ".B "), `\-`, "-")
entry = strings.TrimSuffix(entry, "(1)")
switch {
case strings.HasPrefix(entry, "gasm-"):
m[strings.TrimPrefix(entry, "gasm-")] = true
case strings.HasPrefix(entry, "gasm "):
m[strings.TrimPrefix(entry, "gasm ")] = true
}
}
return m
}
// helpUsage returns the command's usage line without the "Usage: " prefix.
func helpUsage(help string) string {
for line := range strings.SplitSeq(help, "\n") {
if strings.HasPrefix(line, "Usage: ") {
return normaliseUsage(line[len("Usage: "):])
}
}
return ""
}
// roffSynopsis returns the page's SYNOPSIS usage line, unescaped.
func roffSynopsis(page string) string {
inSyn := false
for line := range strings.SplitSeq(page, "\n") {
if strings.HasPrefix(line, ".SH ") {
inSyn = strings.HasPrefix(line, ".SH SYNOPSIS")
continue
}
if !inSyn || !strings.HasPrefix(line, ".B ") {
continue
}
return normaliseUsage(strings.ReplaceAll(line[3:], `\-`, "-"))
}
return ""
}
// normaliseUsage flattens whitespace and drops the roff font escapes so that
// the binary's usage line and the page's SYNOPSIS line compare equal.
func normaliseUsage(s string) string {
s = roffEscape.ReplaceAllString(s, "")
return strings.Join(strings.Fields(s), " ")
}
+345
View File
@@ -0,0 +1,345 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"go/ast"
"go/parser"
"go/token"
"os"
"strings"
gasmast "sourcedock.dev/petrbalvin/gasm-devkit/ast"
gasmparser "sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// cmdScaffold generates a differential test skeleton for every kernel in a
// file: a Go test that seeds random states, drives both the assembly kernel
// and a caller-provided portable reference, and compares the outputs
// byte-for-byte. The lesson this encodes: a pipeline-level fuzz cannot see
// an unwired kernel, only a direct-call differential against the portable
// specification can, so every kernel ships with one.
//
// The generated file follows two conventions the caller fills in:
// - the assembly symbols resolve because the test lives in the kernel's
// own package (the //go:noescape declarations reference them);
// - each kernel gets a <name>Portable Go function the author implements as
// the specification, and the test fails on the first divergent byte.
func cmdScaffold(args []string) error {
fs := newCommand("scaffold", "gasm scaffold differential <file.s>", `
Print a differential test skeleton for every // func signature in FILE.
The test seeds random states, drives the kernel and a portable reference
(<name>Portable), and compares outputs byte-for-byte. Write the reference
bodies, place the file in the kernel's package, and run it in CI.
`)
if err := fs.Parse(args); err != nil {
return err
}
rest := fs.Args()
// The first positional word is the scaffold style; "differential" is the
// only one today.
if len(rest) > 0 && rest[0] == "differential" {
rest = rest[1:]
}
if len(rest) != 1 {
return &usageError{fmt.Errorf("usage: gasm scaffold differential <file.s>")}
}
path := rest[0]
src, err := os.ReadFile(path)
if err != nil {
return err
}
f, errs := gasmparser.Parse(path, string(src))
if len(errs) > 0 {
return fmt.Errorf("parse: %v", errs[0])
}
var out strings.Builder
out.WriteString(headerComment)
out.WriteString("package " + packageName + "\n\n")
out.WriteString("import (\n\t\"bytes\"\n\t\"math/rand\"\n\t\"testing\"\n)\n\n")
out.WriteString(generatedHelpers)
kernels := 0
for _, d := range f.Decls {
txt, ok := d.(*gasmast.Text)
if !ok {
continue
}
params, results, ok := parseSig(txt.Doc)
if !ok || len(params) == 0 {
continue
}
kernels++
name := txt.Name.Name
fmt.Fprintf(&out, "// %sPortable is the specification %s is pinned against:\n", name, name)
fmt.Fprintf(&out, "// fill in a straightforward implementation of the same contract.\n")
fmt.Fprintf(&out, "func %sPortable(%s) (%s) {\n\tpanic(\"implement the portable specification\")\n}\n\n", name, paramDecl(params), resultDecl(results))
fmt.Fprintf(&out, "func Test%sDifferential(t *testing.T) {\n", strings.ToUpper(name[:1])+name[1:])
fmt.Fprintf(&out, "\trng := rand.New(rand.NewSource(1))\n")
fmt.Fprintf(&out, "\tfor range 1000 {\n")
// Seed two independent argument sets per iteration: the kernel runs
// on set A, the portable reference on set B, so in-place writes
// through pointer/slice arguments cannot contaminate the other side.
var sliceNames []string
seen := map[string]bool{}
aArgs := make([]string, 0, len(params))
bArgs := make([]string, 0, len(params))
for _, p := range params {
a, b, slices := genParamSeed(&out, p, seen)
aArgs = append(aArgs, a)
bArgs = append(bArgs, b)
sliceNames = append(sliceNames, slices...)
}
fmt.Fprintf(&out, "\t\tgot := %s(%s)\n", name, strings.Join(aArgs, ", "))
fmt.Fprintf(&out, "\t\twant := %sPortable(%s)\n", name, strings.Join(bArgs, ", "))
fmt.Fprintf(&out, "\t\tif !bytes.Equal(outputBytes(got), outputBytes(want)) {\n")
fmt.Fprintf(&out, "\t\t\tt.Fatalf(\"kernel diverges from the portable spec (seed 1, deterministic)\")\n")
fmt.Fprintf(&out, "\t\t}\n")
for _, s := range sliceNames {
fmt.Fprintf(&out, "\t\tif !bytes.Equal(outputBytes(%sA), outputBytes(%sB)) {\n", s, s)
fmt.Fprintf(&out, "\t\t\tt.Fatalf(\"kernel mutated %%q differently (seed 1, deterministic)\", %q)\n", s)
fmt.Fprintf(&out, "\t\t}\n")
}
fmt.Fprintf(&out, "\t}\n}\n\n")
}
if kernels == 0 {
return fmt.Errorf("%s: no // func signatures found; add one doc comment per kernel", path)
}
os.Stdout.WriteString(out.String())
return nil
}
const packageName = "yourpkg"
const headerComment = `// Code generated by gasm scaffold differential; EDIT THE PANICS.
// Each Test*Differential drives the assembly kernel and its portable
// reference over the same random states and compares the outputs.
// Place this file in the kernel's own package so the symbols resolve.
`
// sigParam is one parsed // func parameter.
type sigParam struct {
Names []string
Type string
}
type sigResult struct {
Names []string
Type string
}
// parseSig parses the // func signature of a doc comment.
func parseSig(doc string) ([]sigParam, []sigResult, bool) {
var line string
for l := range strings.SplitSeq(doc, "\n") {
if t := strings.TrimSpace(l); strings.HasPrefix(t, "func ") {
line = t
break
}
}
if line == "" {
return nil, nil, false
}
fset := token.NewFileSet()
f, err := parser.ParseFile(fset, "sig.go", "package p\n"+line+" {}\n", 0)
if err != nil {
return nil, nil, false
}
fd, ok := f.Decls[0].(*ast.FuncDecl)
if !ok || fd.Type == nil {
return nil, nil, false
}
var params []sigParam
for _, field := range fd.Type.Params.List {
typ := exprString(field.Type)
if len(field.Names) == 0 {
params = append(params, sigParam{Names: []string{""}, Type: typ})
continue
}
// Shared names (`L, result *byte`) expand to one entry per name:
// every name is a separate argument at the call site.
for _, n := range field.Names {
params = append(params, sigParam{Names: []string{n.Name}, Type: typ})
}
}
var results []sigResult
if fd.Type.Results != nil {
for _, field := range fd.Type.Results.List {
results = append(results, sigResult{Names: identNames(field.Names), Type: exprString(field.Type)})
}
}
return params, results, true
}
func identNames(idents []*ast.Ident) []string {
var out []string
for _, id := range idents {
out = append(out, id.Name)
}
return out
}
func exprString(e ast.Expr) string {
switch t := e.(type) {
case *ast.Ident:
return t.Name
case *ast.StarExpr:
return "*" + exprString(t.X)
case *ast.SelectorExpr:
return exprString(t.X) + "." + t.Sel.Name
case *ast.ArrayType:
if t.Len == nil {
return "[]" + exprString(t.Elt)
}
return "[N]" + exprString(t.Elt)
}
return "interface{}"
}
// paramDecl renders a parameter list for the portable reference signature.
func paramDecl(params []sigParam) string {
var parts []string
for _, p := range params {
if len(p.Names) == 0 {
parts = append(parts, p.Type)
continue
}
for _, n := range p.Names {
parts = append(parts, n+" "+p.Type)
}
}
return strings.Join(parts, ", ")
}
// resultDecl renders a result list; unnamed results keep bare types.
func resultDecl(results []sigResult) string {
if len(results) == 0 {
return ""
}
var parts []string
for _, r := range results {
parts = append(parts, r.Type)
}
return strings.Join(parts, ", ")
}
// genParamSeed emits the seeding statements for one parameter and returns
// the kernel-side (A) and reference-side (B) argument expressions, plus the
// names of any slice variables written in place (compared after the calls).
func genParamSeed(out *strings.Builder, p sigParam, seen map[string]bool) (aArg, bArg string, slices []string) {
name := p.Names[0]
elem := strings.TrimPrefix(p.Type, "*")
isSlice := strings.HasPrefix(p.Type, "[]")
if isSlice {
elem = strings.TrimPrefix(p.Type, "[]")
}
switch {
case isSlice:
v := uniqueName(seen, name)
fmt.Fprintf(out, "\t\t%sA := make([]%s, 1+rng.Intn(512))\n", v, elem)
fmt.Fprintf(out, "\t\t%sB := make([]%s, len(%sA))\n", v, elem, v)
fmt.Fprintf(out, "\t\tfor i := range %sA {\n", v)
fmt.Fprintf(out, "\t\t\tw%s := %s(rng.Intn(256))\n", v, goCast(elem))
fmt.Fprintf(out, "\t\t\t%sA[i] = w%s\n", v, v)
fmt.Fprintf(out, "\t\t\t%sB[i] = w%s\n", v, v)
fmt.Fprintf(out, "\t\t}\n")
return v, v, []string{v}
case strings.HasPrefix(p.Type, "*"):
v := uniqueName(seen, name)
fmt.Fprintf(out, "\t\tvar %sA, %sB %s\n", v, v, elem)
fmt.Fprintf(out, "\t\tw%s := %s(rng.Intn(256))\n", v, goCast(elem))
fmt.Fprintf(out, "\t\t%sA = w%s\n", v, v)
fmt.Fprintf(out, "\t\t%sB = w%s\n", v, v)
return "&" + v + "A", "&" + v + "B", nil
default:
v := uniqueName(seen, name)
fmt.Fprintf(out, "\t\tw%s := %s(rng.Intn(512))\n", v, goCast(""))
fmt.Fprintf(out, "\t\tvar %sA, %sB %s = w%s, w%s\n", v, v, p.Type, v, v)
return v + "A", v + "B", nil
}
}
// uniqueName de-duplicates seeded variable names when one kernel takes two
// parameters of the same name (impossible in Go) or a name repeats across
// kernels in one file.
func uniqueName(seen map[string]bool, base string) string {
if base == "" {
base = "arg"
}
if !seen[base] {
seen[base] = true
return base
}
for i := 2; ; i++ {
cand := fmt.Sprintf("%s%d", base, i)
if !seen[cand] {
seen[cand] = true
return cand
}
}
}
// goCast returns the conversion turning rng.Intn into the element type.
func goCast(elem string) string {
switch elem {
case "byte", "uint8":
return "byte"
case "int8":
return "int8"
case "uint16":
return "uint16"
case "int16":
return "int16"
case "uint32":
return "uint32"
case "int32":
return "int32"
case "uint64":
return "uint64"
default:
return "int"
}
}
// generatedHelpers is emitted into every generated test file: outputBytes
// narrows returned slices and scalars to a byte form for the comparison.
// It lives in the template, not in this binary, because only the generated
// file ever calls it.
const generatedHelpers = `// outputBytes narrows a returned slice or scalar to bytes for the
// comparison; extend the switch when a kernel returns a wider type.
func outputBytes(v any) []byte {
switch t := v.(type) {
case []byte:
return t
case []int32:
b := make([]byte, 4*len(t))
for i, x := range t {
b[i*4] = byte(x)
b[i*4+1] = byte(x >> 8)
b[i*4+2] = byte(x >> 16)
b[i*4+3] = byte(x >> 24)
}
return b
case []uint16:
b := make([]byte, 2*len(t))
for i, x := range t {
b[i*2] = byte(x)
b[i*2+1] = byte(x >> 8)
}
return b
case int:
b := make([]byte, 8)
for i := range 8 {
b[i] = byte(uint64(t) >> (8 * i))
}
return b
default:
return nil
}
}
`
+140
View File
@@ -0,0 +1,140 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"slices"
"strings"
)
// unifiedDiff renders a unified diff with three lines of context between the
// two line slices, in the form `gofmt -d` prints. An empty result means the
// inputs are identical.
func unifiedDiff(name string, a, b []string) string {
if slices.Equal(a, b) {
return ""
}
var out strings.Builder
fmt.Fprintf(&out, "--- %s\n+++ %s\n", name, name)
// Longest common subsequence over the lines (assembly files are small
// enough for the quadratic table).
n, m := len(a), len(b)
lcs := make([][]int, n+1)
for i := range lcs {
lcs[i] = make([]int, m+1)
}
for i := n - 1; i >= 0; i-- {
for j := m - 1; j >= 0; j-- {
if a[i] == b[j] {
lcs[i][j] = lcs[i+1][j+1] + 1
} else if lcs[i+1][j] >= lcs[i][j+1] {
lcs[i][j] = lcs[i+1][j]
} else {
lcs[i][j] = lcs[i][j+1]
}
}
}
// Walk the LCS once, assigning every op its absolute position in both
// files (1-based, the position an insertion sits before).
type op struct {
kind byte // ' ', '-' or '+'
aLine, bLine int
text string
}
var ops []op
aPos, bPos := 0, 0
emit := func(kind byte, text string) {
ops = append(ops, op{kind: kind, aLine: aPos + 1, bLine: bPos + 1, text: text})
switch kind {
case ' ':
aPos++
bPos++
case '-':
aPos++
case '+':
bPos++
}
}
i, j := 0, 0
for i < n && j < m {
switch {
case a[i] == b[j]:
emit(' ', a[i])
i++
j++
case lcs[i+1][j] >= lcs[i][j+1]:
emit('-', a[i])
i++
default:
emit('+', b[j])
j++
}
}
for ; i < n; i++ {
emit('-', a[i])
}
for ; j < m; j++ {
emit('+', b[j])
}
// Group the edits into hunks: consecutive changes separated by more than
// twice the context lines start a new hunk.
const context = 3
var changes []int
for k, o := range ops {
if o.kind != ' ' {
changes = append(changes, k)
}
}
for g := 0; g < len(changes); {
last := g
for last+1 < len(changes) && changes[last+1]-changes[last]-1 <= 2*context {
last++
}
lo := max(0, changes[g]-context)
hi := min(len(ops), changes[last]+1+context)
// The header numbers are the first line of each side actually shown:
// the first context, deletion or insertion line. A hunk that shows
// no old lines is a pure insertion and reports the position it sits
// before (0 at the top of the file); the mirror rule holds for a
// pure deletion.
aStart := ops[lo].aLine - 1
bStart := ops[lo].bLine - 1
countA, countB := 0, 0
for _, o := range ops[lo:hi] {
switch o.kind {
case ' ':
countA++
countB++
case '-':
countA++
case '+':
countB++
}
}
for _, o := range ops[lo:hi] {
if o.kind != '+' {
aStart = o.aLine
break
}
}
for _, o := range ops[lo:hi] {
if o.kind != '-' {
bStart = o.bLine
break
}
}
fmt.Fprintf(&out, "@@ -%d,%d +%d,%d @@\n", aStart, countA, bStart, countB)
for _, o := range ops[lo:hi] {
out.WriteByte(o.kind)
out.WriteString(o.text)
out.WriteByte('\n')
}
g = last + 1
}
return out.String()
}
+94
View File
@@ -0,0 +1,94 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"slices"
"strings"
"testing"
)
func lines(ss ...string) []string { return ss }
func TestUnifiedDiffIdentical(t *testing.T) {
if got := unifiedDiff("f", lines("a", "b"), lines("a", "b")); got != "" {
t.Errorf("identical inputs produced %q, want empty", got)
}
}
func TestUnifiedDiffSingleChange(t *testing.T) {
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
b := lines("1", "2", "3!", "4", "5", "6", "7", "8")
want := "--- f\n+++ f\n" +
"@@ -1,6 +1,6 @@\n" +
" 1\n 2\n-3\n+3!\n 4\n 5\n 6\n"
if got := unifiedDiff("f", a, b); got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffInsertAtStart(t *testing.T) {
got := unifiedDiff("f", lines("x"), lines("new", "x"))
// The single existing line is shown as trailing context, so the hunk
// covers it.
want := "--- f\n+++ f\n@@ -1,1 +1,2 @@\n+new\n x\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffDeleteAtEnd(t *testing.T) {
got := unifiedDiff("f", lines("x", "y"), lines("x"))
want := "--- f\n+++ f\n@@ -1,2 +1,1 @@\n x\n-y\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffTwoHunks(t *testing.T) {
var a, b []string
for i := 1; i <= 20; i++ {
a = append(a, itoa(i))
b = append(b, itoa(i))
}
b[1] = "2!"
b[17] = "18!"
got := unifiedDiff("f", a, b)
if !strings.Contains(got, "@@ -1,5 +1,5 @@\n 1\n-2\n+2!\n 3\n 4\n 5\n") {
t.Errorf("first hunk wrong:\n%s", got)
}
if !strings.Contains(got, "@@ -15,6 +15,6 @@\n 15\n 16\n 17\n-18\n+18!\n 19\n 20\n") {
t.Errorf("second hunk wrong:\n%s", got)
}
}
// TestUnifiedDiffAdjacentHunks merges changes separated by exactly twice the
// context into one hunk.
func TestUnifiedDiffAdjacentHunks(t *testing.T) {
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
b := slices.Clone(a)
b[0] = "1!"
b[7] = "8!"
got := unifiedDiff("f", a, b)
want := "--- f\n+++ f\n" +
"@@ -1,8 +1,8 @@\n" +
"-1\n+1!\n 2\n 3\n 4\n 5\n 6\n 7\n-8\n+8!\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func itoa(n int) string {
if n == 0 {
return "0"
}
var buf [4]byte
i := len(buf)
for n > 0 {
i--
buf[i] = byte('0' + n%10)
n /= 10
}
return string(buf[i:])
}
+307
View File
@@ -0,0 +1,307 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package debug
import "strings"
import "fmt"
// Breakpoint is one software breakpoint in the debuggee.
type Breakpoint struct {
Addr uint64 // absolute address in the debuggee
Label string // source label ("" for raw addresses)
Orig []byte // original bytes at Addr (restored on removal)
Enabled bool
Cond *Condition // optional condition (nil = unconditional)
hits int
}
// Condition is a register-comparison condition evaluated when a breakpoint
// is hit. Supports three forms:
// - register vs constant: <reg> <op> <value>
// - register vs register: <reg> <op> <reg2>
// - register vs memory: <reg> <op> *<addr>
type Condition struct {
Reg string // register name (rax, rbx, rip, rsp, ...)
Op string // comparison operator: ==, !=, <, >, <=, >=
Value uint64 // constant value (when Reg2 == "" and MemAddr == 0)
Reg2 string // second register name (for register-register comparison)
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
}
// Eval checks the condition against the current registers. For the
// register-memory form, mem reads an 8-byte little-endian word from the
// debuggee; it may be nil when no reader is available. Anything that cannot
// be decided (unknown register or operator, unreadable memory) does not
// block the breakpoint.
func (c *Condition) Eval(regs *Regs, mem func(addr uint64) (uint64, bool)) bool {
actual, ok := regs.RegValue(c.Reg)
if !ok {
return true // unknown register, don't block
}
var expected uint64
switch {
case c.Reg2 != "":
// Register-register comparison.
v, ok := regs.RegValue(c.Reg2)
if !ok {
return true
}
expected = v
case c.MemAddr != 0:
// Register-memory comparison, resolved in the debuggee at
// evaluation time.
if mem == nil {
return true
}
v, ok := mem(c.MemAddr)
if !ok {
return true
}
expected = v
default:
expected = c.Value
}
switch c.Op {
case "==", "=":
return actual == expected
case "!=":
return actual != expected
case "<":
return actual < expected
case ">":
return actual > expected
case "<=":
return actual <= expected
case ">=":
return actual >= expected
default:
return true
}
}
// String renders the condition for display.
func (c *Condition) String() string {
switch {
case c.Reg2 != "":
return fmt.Sprintf("%s %s %s", c.Reg, c.Op, c.Reg2)
case c.MemAddr != 0:
return fmt.Sprintf("%s %s *%#x", c.Reg, c.Op, c.MemAddr)
default:
return fmt.Sprintf("%s %s %#x", c.Reg, c.Op, c.Value)
}
}
// Breakpoints manages the software breakpoints of one Session.
type Breakpoints struct {
t tracer
bps map[uint64]*Breakpoint
}
// NewBreakpoints creates a new breakpoint manager.
func NewBreakpoints(t tracer) *Breakpoints {
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
}
// breakpointMask is the byte mask of the breakpoint instruction inside a
// peeked word: the low len(breakpointInsn) bytes, because every supported
// architecture is little-endian and patches the instruction at the lowest
// address of the word.
func breakpointMask() uint64 {
var mask uint64
for range breakpointInsn {
mask = (mask << 8) | 0xFF
}
return mask
}
// Set installs a breakpoint at addr (replaces any existing one).
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
return bm.SetWithCond(addr, label, nil)
}
// SetWithCond installs a breakpoint with an optional condition.
func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (*Breakpoint, error) {
if bp, ok := bm.bps[addr]; ok {
bp.Enabled = true
bp.Cond = cond
return bp, nil
}
// Read the original bytes.
word, err := bm.t.Peek(addr)
if err != nil {
return nil, err
}
orig := make([]byte, len(breakpointInsn))
for i := range orig {
orig[i] = byte(word >> (8 * i))
}
// Patch with the breakpoint instruction, preserving the rest of the word.
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
if err := bm.t.Poke(addr, patched); err != nil {
return nil, err
}
bp := &Breakpoint{Addr: addr, Label: label, Orig: orig, Enabled: true, Cond: cond}
bm.bps[addr] = bp
return bp, nil
}
// Info returns a formatted list of all breakpoints.
func (bm *Breakpoints) Info() string {
if len(bm.bps) == 0 {
return "no breakpoints set\n"
}
var result strings.Builder
i := 0
for _, bp := range bm.bps {
i++
status := "enabled"
if !bp.Enabled {
status = "disabled"
}
label := bp.Label
if label == "" {
label = fmt.Sprintf("%#x", bp.Addr)
}
cond := ""
if bp.Cond != nil {
cond = " if " + bp.Cond.String()
}
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
}
return result.String()
}
// restore writes the saved original bytes back over the breakpoint
// instruction, preserving the rest of the peeked word. It reports whether
// both the peek and the poke succeeded.
func (bm *Breakpoints) restore(addr uint64, bp *Breakpoint) bool {
word, err := bm.t.Peek(addr)
if err != nil {
return false
}
orig := uint64(0)
for i, b := range bp.Orig {
orig |= uint64(b) << (8 * i)
}
return bm.t.Poke(addr, (word&^breakpointMask())|orig) == nil
}
// Clear removes the breakpoint at addr, restoring the original bytes.
func (bm *Breakpoints) Clear(addr uint64) error {
bp, ok := bm.bps[addr]
if !ok {
return fmt.Errorf("debug: no breakpoint at %#x", addr)
}
if !bm.restore(addr, bp) {
word, err := bm.t.Peek(addr)
if err != nil {
return err
}
return fmt.Errorf("debug: restore breakpoint at %#x failed, word is %#x", addr, word)
}
delete(bm.bps, addr)
return nil
}
// ClearAll removes all breakpoints.
func (bm *Breakpoints) ClearAll() error {
for addr := range bm.bps {
if err := bm.Clear(addr); err != nil {
return err
}
}
return nil
}
// At returns the breakpoint at addr, if any.
func (bm *Breakpoints) At(addr uint64) *Breakpoint {
return bm.bps[addr]
}
// All returns all breakpoints.
func (bm *Breakpoints) All() []*Breakpoint {
out := make([]*Breakpoint, 0, len(bm.bps))
for _, bp := range bm.bps {
out = append(out, bp)
}
return out
}
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
// whether the trap was caused by one of our breakpoints (PC-adjust matches
// a breakpoint address), restores the original bytes, rewinds PC, and
// returns the breakpoint that was hit (or nil if it was a single-step).
// Hits returns how many times the breakpoint has been hit.
func (bp *Breakpoint) Hits() int { return bp.hits }
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
// On amd64 the kernel reports the trap with RIP past the INT3; on the
// other supported architectures the PC still stands on the trap
// instruction, which breakpointPCAdjust encodes per architecture.
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
bp, ok := bm.bps[trapAddr]
if !ok || !bp.Enabled {
return nil // single-step trap or unknown
}
// Check the condition (if any).
if bp.Cond != nil && !bp.Cond.Eval(regs, bm.peekValue) {
// Condition not met: step the original instruction and re-arm the
// breakpoint, leaving the debuggee stopped just past it, ready to
// resume silently. The PC must be rewound first: on architectures
// that report the trap past the instruction (amd64) it would
// otherwise sit on the second byte of the replaced instruction.
if !bm.restore(trapAddr, bp) {
return nil
}
regs.SetPC(trapAddr)
if err := bm.t.SetRegs(regs); err != nil {
return nil
}
if err := bm.t.Step(); err != nil {
return nil
}
bm.Reinsert(trapAddr)
return nil
}
bp.hits++
// Restore the original bytes and rewind PC to re-execute them.
bm.restore(trapAddr, bp)
regs.SetPC(trapAddr)
bm.t.SetRegs(regs)
return bp
}
// peekValue adapts tracer.Peek to the Condition value reader.
func (bm *Breakpoints) peekValue(addr uint64) (uint64, bool) {
v, err := bm.t.Peek(addr)
return v, err == nil
}
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
// Called after Step() when we want the breakpoint to fire again on the
// next Continue().
func (bm *Breakpoints) Reinsert(addr uint64) error {
bp, ok := bm.bps[addr]
if !ok || !bp.Enabled {
return nil
}
word, err := bm.t.Peek(addr)
if err != nil {
return err
}
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
return bm.t.Poke(addr, patched)
}
// breakpointWord converts the breakpoint instruction bytes to a uint64.
func breakpointWord(insn []byte) uint64 {
var w uint64
for i, b := range insn {
w |= uint64(b) << (i * 8)
}
return w
}
+265
View File
@@ -0,0 +1,265 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package debug
// Architecture-neutral tests: label and line tables, and the breakpoint
// manager against the mock tracer. These do not launch a debuggee, so they
// build on every supported linux architecture.
import (
"strings"
"testing"
)
func TestLineAt(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
{Offset: 15, Line: 8},
}
tests := []struct {
offset int
want int
}{
{0, 5},
{1, 5},
{4, 5},
{5, 6},
{7, 6},
{10, 7},
{12, 7},
{15, 8},
{20, 8},
}
for _, tt := range tests {
got := lineAt(lines, tt.offset)
if got != tt.want {
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
}
}
// Empty table.
if lineAt(nil, 5) != 0 {
t.Error("lineAt(nil, 5) should return 0")
}
}
func TestOffsetForLine(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
}
tests := []struct {
line int
want int
}{
{5, 0},
{6, 5},
{7, 10},
{99, -1}, // not found
{0, -1}, // not found
}
for _, tt := range tests {
got := offsetForLine(lines, tt.line)
if got != tt.want {
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
}
}
}
func TestNearestLabel(t *testing.T) {
labels := []Label{
{Name: "start", Offset: 0},
{Name: "loop", Offset: 10},
{Name: "done", Offset: 20},
}
tests := []struct {
offset int
want string
}{
{0, "start"},
{5, "start"},
{10, "loop"},
{15, "loop"},
{20, "done"},
{25, "done"},
}
for _, tt := range tests {
got := nearestLabel(labels, tt.offset)
if got != tt.want {
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
}
}
}
func TestBreakpointsSetAndClear(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
// Set a breakpoint at address 0x1000.
bp, err := bm.Set(0x1000, "test")
if err != nil {
t.Fatalf("Set: %v", err)
}
if !bp.Enabled {
t.Error("breakpoint not enabled")
}
if bp.Label != "test" {
t.Errorf("label = %q, want test", bp.Label)
}
// Verify Peek was called.
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
}
// Verify Poke wrote the breakpoint instruction's bytes.
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
t.Errorf("pokes = %v", tr.pokes)
}
if got := tr.pokes[0].val & breakpointMask(); got != breakpointWord(breakpointInsn) {
t.Errorf("patched bytes %#x, want %#x", got, breakpointWord(breakpointInsn))
}
// At should find it.
if bm.At(0x1000) == nil {
t.Error("At(0x1000) returned nil")
}
// All should return it.
all := bm.All()
if len(all) != 1 {
t.Errorf("All() = %d breakpoints, want 1", len(all))
}
// Clear it.
if err := bm.Clear(0x1000); err != nil {
t.Fatalf("Clear: %v", err)
}
if bm.At(0x1000) != nil {
t.Error("At(0x1000) after Clear should be nil")
}
}
// TestBreakpointRestoreWidth proves the restore path writes back every
// byte of the breakpoint instruction's width, not just the first byte: on
// arm64, riscv64 and loong64 the instruction is four bytes, and restoring
// one byte would leave three bytes of the trap instruction in place.
func TestBreakpointRestoreWidth(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
tr.mem[0x3000] = 0x11
tr.mem[0x3001] = 0x22
tr.mem[0x3002] = 0x33
tr.mem[0x3003] = 0x44
if _, err := bm.Set(0x3000, "width"); err != nil {
t.Fatalf("Set: %v", err)
}
for i, b := range breakpointInsn {
if tr.mem[0x3000+uint64(i)] != b {
t.Fatalf("byte %d after Set = %#x, want the breakpoint byte %#x", i, tr.mem[0x3000+uint64(i)], b)
}
}
if len(bm.At(0x3000).Orig) != len(breakpointInsn) {
t.Fatalf("Orig holds %d bytes, want %d", len(bm.At(0x3000).Orig), len(breakpointInsn))
}
if err := bm.Clear(0x3000); err != nil {
t.Fatalf("Clear: %v", err)
}
want := []byte{0x11, 0x22, 0x33, 0x44}
for i, b := range want {
if tr.mem[0x3000+uint64(i)] != b {
t.Errorf("byte %d after Clear = %#x, want %#x (restore must cover the full instruction width)", i, tr.mem[0x3000+uint64(i)], b)
}
}
}
func TestBreakpointsSetWithCond(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
if err != nil {
t.Fatalf("SetWithCond: %v", err)
}
if bp.Cond == nil || bp.Cond.Value != 42 {
t.Error("condition not set")
}
// Re-setting the same address should update the condition.
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
if err != nil {
t.Fatalf("SetWithCond (update): %v", err)
}
if bp2.Cond.Value != 100 {
t.Error("condition not updated")
}
// Should have only 1 Peek (first Set), second is update (no Peek needed).
if len(tr.peeks) != 1 {
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
}
}
func TestBreakpointsClearAll(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x1000, "a")
bm.Set(0x2000, "b")
bm.Set(0x3000, "c")
if len(bm.All()) != 3 {
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
}
bm.ClearAll()
if len(bm.All()) != 0 {
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
}
}
func TestBreakpointInfo(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x4000, "info_test")
info := bm.Info()
if info == "" {
t.Error("Info returned empty string")
}
if !strings.Contains(info, "info_test") {
t.Errorf("Info %q does not contain label", info)
}
}
// TestConditionString covers the display of all three condition forms.
func TestConditionString(t *testing.T) {
tests := []struct {
cond Condition
want string
}{
{Condition{Reg: "rax", Op: "==", Value: 42}, "rax == 0x2a"},
{Condition{Reg: "rax", Op: "!=", Reg2: "rbx"}, "rax != rbx"},
{Condition{Reg: "rax", Op: "<", MemAddr: 0x5000}, "rax < *0x5000"},
}
for _, tt := range tests {
if got := tt.cond.String(); got != tt.want {
t.Errorf("Condition.String() = %q, want %q", got, tt.want)
}
}
}
+175
View File
@@ -0,0 +1,175 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"testing"
)
func TestConditionEval(t *testing.T) {
regs := &Regs{
RAX: 42,
RBX: 0,
RCX: 100,
RIP: 0x1000,
RSP: 0x2000,
R8: 8,
R15: 15,
}
tests := []struct {
cond Condition
want bool
}{
{Condition{Reg: "rax", Op: "==", Value: 42}, true},
{Condition{Reg: "rax", Op: "==", Value: 43}, false},
{Condition{Reg: "rax", Op: "!=", Value: 43}, true},
{Condition{Reg: "rax", Op: "!=", Value: 42}, false},
{Condition{Reg: "rax", Op: "<", Value: 50}, true},
{Condition{Reg: "rax", Op: "<", Value: 40}, false},
{Condition{Reg: "rax", Op: ">", Value: 40}, true},
{Condition{Reg: "rax", Op: ">", Value: 50}, false},
{Condition{Reg: "rax", Op: "<=", Value: 42}, true},
{Condition{Reg: "rax", Op: ">=", Value: 42}, true},
{Condition{Reg: "rbx", Op: "==", Value: 0}, true},
{Condition{Reg: "rcx", Op: ">", Value: 50}, true},
{Condition{Reg: "rip", Op: "==", Value: 0x1000}, true},
{Condition{Reg: "rsp", Op: ">", Value: 0x1000}, true},
{Condition{Reg: "r8", Op: "==", Value: 8}, true},
{Condition{Reg: "r15", Op: "==", Value: 15}, true},
{Condition{Reg: "eax", Op: "==", Value: 42}, true}, // 32-bit alias
{Condition{Reg: "ax", Op: "==", Value: 42}, true}, // 16-bit alias
{Condition{Reg: "unknown", Op: "==", Value: 0}, true}, // unknown reg → don't block
{Condition{Reg: "rax", Op: "??", Value: 0}, true}, // unknown op → don't block
}
for _, tt := range tests {
got := tt.cond.Eval(regs, nil)
if got != tt.want {
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
}
}
}
// TestConditionEvalMem covers the register-memory form: the value is read
// through the supplied reader, and a missing or failing reader must not
// block the breakpoint.
func TestConditionEvalMem(t *testing.T) {
regs := &Regs{RAX: 7}
mem := func(addr uint64) (uint64, bool) {
if addr == 0x5000 {
return 7, true
}
return 0, false
}
eq := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
if !eq.Eval(regs, mem) {
t.Error("register-memory comparison with matching word should hold")
}
ne := Condition{Reg: "rax", Op: "!=", MemAddr: 0x5000}
if ne.Eval(regs, mem) {
t.Error("register-memory comparison with mismatching word should not hold")
}
bad := Condition{Reg: "rax", Op: "==", MemAddr: 0x6000}
if !bad.Eval(regs, mem) {
t.Error("unreadable memory must not block the breakpoint")
}
noReader := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
if !noReader.Eval(regs, nil) {
t.Error("missing memory reader must not block the breakpoint")
}
}
func TestDecodeRflags(t *testing.T) {
tests := []struct {
flags uint64
want string
}{
{0x202, "IF"}, // only IF set (bit 9)
{0x246, "PF ZF IF"}, // PF(2) + ZF(6) + IF(9)
{0x001, "CF"}, // carry flag
{0x080, "SF"}, // sign flag
{0x800, "OF"}, // overflow flag
{0x000, "none"}, // no flags
{0x202 | 0x001, "CF IF"}, // CF + IF
{0x3F7, "CF PF AF ZF SF TF IF"}, // all arithmetic flags
}
for _, tt := range tests {
got := decodeRflags(tt.flags)
if got != tt.want {
t.Errorf("decodeRflags(%#x) = %q, want %q", tt.flags, got, tt.want)
}
}
}
func TestWatchpointSlotTracking(t *testing.T) {
s := &Session{} // per-session slots start free
// All four slots are free initially.
for i := range 4 {
if s.IsWatchpointSlotUsed(i) {
t.Errorf("slot %d should be free initially", i)
}
}
if got := s.FindFreeWatchpointSlot(); got != 0 {
t.Errorf("FindFreeWatchpointSlot() = %d, want 0", got)
}
// Manually mark slots 0 and 2 as used (simulating successful SetWatchpoint).
s.wpSlots[0] = true
s.wpSlots[2] = true
if !s.IsWatchpointSlotUsed(0) {
t.Error("slot 0 should be in use")
}
if s.IsWatchpointSlotUsed(1) {
t.Error("slot 1 should be free")
}
if !s.IsWatchpointSlotUsed(2) {
t.Error("slot 2 should be in use")
}
if s.IsWatchpointSlotUsed(3) {
t.Error("slot 3 should be free")
}
if got := s.FindFreeWatchpointSlot(); got != 1 {
t.Errorf("FindFreeWatchpointSlot() = %d, want 1", got)
}
// Out-of-range slot queries return false.
if s.IsWatchpointSlotUsed(-1) {
t.Error("slot -1 should be reported as free (out of range)")
}
if s.IsWatchpointSlotUsed(4) {
t.Error("slot 4 should be reported as free (out of range)")
}
// Mark all slots used: FindFreeWatchpointSlot returns -1.
for i := range 4 {
s.wpSlots[i] = true
}
if got := s.FindFreeWatchpointSlot(); got != -1 {
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
}
}
// TestUnwatchSlotBound checks the bound the REPL parses against: it must
// cover the architecture's whole slot range, not a hardcoded 0-3.
func TestUnwatchSlotBound(t *testing.T) {
max := maxWatchpoints()
if max < 4 {
t.Fatalf("maxWatchpoints() = %d, want at least 4", max)
}
s := &Session{}
if s.IsWatchpointSlotUsed(max - 1) {
t.Errorf("slot %d should be free initially", max-1)
}
if s.IsWatchpointSlotUsed(max) {
t.Errorf("slot %d must be out of range", max)
}
}
+56
View File
@@ -0,0 +1,56 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 15)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.AMD64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr and returns
// them as a formatted string with addresses and byte offsets.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result strings.Builder
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
break
}
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
if length == 0 {
length = 1
}
pc += uint64(length)
}
return result.String()
}
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
// call. The first token must match exactly: a prefix test would also catch
// unrelated mnemonics.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
return strings.ToLower(m) == "call"
}
+59
View File
@@ -0,0 +1,59 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.ARM64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result string
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
break
}
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
if length == 0 {
length = 4
}
pc += uint64(length)
}
return result
}
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
// The first token must match exactly so branches never match.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl":
return true
}
return false
}
+60
View File
@@ -0,0 +1,60 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.LOONG64, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result string
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
break
}
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
if length == 0 {
length = 4
}
pc += uint64(length)
}
return result
}
// isCallInsn reports whether disassembled text (loong64asm.GoSyntax) is a
// call. GoSyntax renders bl and jirl calls as CALL (jirl returns print
// RET); the native mnemonics are accepted too. The first token must match
// exactly: a "bl" prefix would catch bltz and other branches.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl", "jirl":
return true
}
return false
}
+61
View File
@@ -0,0 +1,61 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
mem, err := s.ReadMemory(addr, 4)
if err != nil {
return "", 0, err
}
ins, err := disasm.Decode(arch.RISCV, mem, addr)
if err != nil {
return "", 0, err
}
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
func (s *Session) DisassembleN(addr uint64, n int) string {
var result string
pc := addr
for range n {
text, length, err := s.Disassemble(pc)
if err != nil {
result += fmt.Sprintf(" %#08x: <error: %v>\n", pc, err)
break
}
result += fmt.Sprintf(" %#08x: %s\n", pc, text)
if length == 0 {
length = 4
}
pc += uint64(length)
}
return result
}
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
// are accepted too. The first token must match exactly: a prefix test on
// "bl" would catch branches on other architectures, and jalr as ret prints
// RET, which must not be stepped over.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "jal", "jalr":
return true
}
return false
}
+102
View File
@@ -0,0 +1,102 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (YMM):")
for i := 0; i < 16; i += 2 {
fmt.Printf(" YMM%-2d = ", i)
printYMM(v.YMM[i][:])
fmt.Printf(" YMM%-2d = ", i+1)
printYMM(v.YMM[i+1][:])
fmt.Println()
}
}
func printYMM(b []byte) {
for j := 0; j < 32; j += 4 {
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
fmt.Printf("%08x ", v)
}
}
func decodeRflags(f uint64) string {
var flags string
if f&1 != 0 {
flags += "CF "
}
if f&(1<<2) != 0 {
flags += "PF "
}
if f&(1<<4) != 0 {
flags += "AF "
}
if f&(1<<6) != 0 {
flags += "ZF "
}
if f&(1<<7) != 0 {
flags += "SF "
}
if f&(1<<8) != 0 {
flags += "TF "
}
if f&(1<<9) != 0 {
flags += "IF "
}
if f&(1<<10) != 0 {
flags += "DF "
}
if f&(1<<11) != 0 {
flags += "OF "
}
if flags == "" {
return "none"
}
return flags[:len(flags)-1]
}
// archReturnAddr reads the return address of the current frame (amd64
// ABI0 convention). A function that contains a CALL (or has a frame) is
// assembled with the prologue PUSHQ BP; MOVQ SP, BP, so mid-function the
// word at SP is the saved caller BP, a stack address, and the return
// address sits further up. Walk the stack from SP and take the first word
// that lies in an executable mapping: stack and data words never do, a
// return address always does.
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
ranges := execRanges(s.pid)
for off := uint64(0); off < 512; off += 8 {
word, err := s.Peek(regs.RSP + off)
if err != nil {
break
}
for _, r := range ranges {
if word >= r.lo && word < r.hi {
return word, nil
}
}
}
// No mapping available or nothing code-like on the stack: fall back to
// the raw entry convention, [SP] before any push.
return s.Peek(regs.RSP)
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "RSP" }
+48
View File
@@ -0,0 +1,48 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && arm64
package debug
import (
"encoding/binary"
"fmt"
)
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.SP, regs.X29)
fmt.Printf(" LR = %#016x\n", regs.X30)
fmt.Printf(" X0 = %#016x X1 = %#016x\n", regs.X0, regs.X1)
fmt.Printf(" X2 = %#016x X3 = %#016x\n", regs.X2, regs.X3)
fmt.Printf(" X4 = %#016x X5 = %#016x\n", regs.X4, regs.X5)
fmt.Printf(" X6 = %#016x X7 = %#016x\n", regs.X6, regs.X7)
fmt.Printf(" X8 = %#016x X9 = %#016x\n", regs.X8, regs.X9)
fmt.Printf(" X10 = %#016x X11 = %#016x\n", regs.X10, regs.X11)
fmt.Printf(" X12 = %#016x X13 = %#016x\n", regs.X12, regs.X13)
fmt.Printf(" X14 = %#016x X15 = %#016x\n", regs.X14, regs.X15)
fmt.Printf(" X16 = %#016x X17 = %#016x\n", regs.X16, regs.X17)
fmt.Printf(" X18 = %#016x X19 = %#016x\n", regs.X18, regs.X19)
fmt.Printf(" X20 = %#016x X21 = %#016x\n", regs.X20, regs.X21)
fmt.Printf(" X22 = %#016x X23 = %#016x\n", regs.X22, regs.X23)
fmt.Printf(" X24 = %#016x X25 = %#016x\n", regs.X24, regs.X25)
fmt.Printf(" X26 = %#016x X27 = %#016x\n", regs.X26, regs.X27)
fmt.Printf(" X28 = %#016x PSTATE = %#x\n", regs.X28, regs.PSTATE)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (V0-V31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" V%-2d = %016x%016x\n", i, binary.LittleEndian.Uint64(v.V[i][8:16]), binary.LittleEndian.Uint64(v.V[i][0:8]))
fmt.Printf(" V%-2d = %016x%016x\n", i+1, binary.LittleEndian.Uint64(v.V[i+1][8:16]), binary.LittleEndian.Uint64(v.V[i+1][0:8]))
}
}
// archReturnAddr reads the return address from LR (arm64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.X30, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+44
View File
@@ -0,0 +1,44 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && loong64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.R31, regs.R31-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.R3, regs.R21)
fmt.Printf(" RA = %#016x\n", regs.R1)
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.R4, regs.R5)
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.R6, regs.R7)
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.R8, regs.R9)
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.R10, regs.R11)
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.R12, regs.R13)
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.R14, regs.R15)
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.R16, regs.R17)
fmt.Printf(" T6 = %#016x T7 = %#016x\n", regs.R18, regs.R19)
fmt.Printf(" T8 = %#016x\n", regs.R20)
fmt.Printf(" S0 = %#016x S1 = %#016x\n", regs.R22, regs.R23)
fmt.Printf(" S2 = %#016x S3 = %#016x\n", regs.R24, regs.R25)
fmt.Printf(" S4 = %#016x S5 = %#016x\n", regs.R26, regs.R27)
fmt.Printf(" S6 = %#016x S7 = %#016x\n", regs.R28, regs.R29)
fmt.Printf(" S8 = %#016x\n", regs.R30)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n FP registers (F0-F31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
}
fmt.Printf(" FCC = %#016x FCSR = %#x\n", v.FCC, v.FCSR)
}
// archReturnAddr reads the return address from RA (loong64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.R1, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+44
View File
@@ -0,0 +1,44 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && riscv64
package debug
import "fmt"
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.Sp, regs.S0)
fmt.Printf(" RA = %#016x\n", regs.Ra)
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.A0, regs.A1)
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.A2, regs.A3)
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.A4, regs.A5)
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.A6, regs.A7)
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.T0, regs.T1)
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.T2, regs.T3)
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.T4, regs.T5)
fmt.Printf(" T6 = %#016x\n", regs.T6)
fmt.Printf(" S1 = %#016x S2 = %#016x\n", regs.S1, regs.S2)
fmt.Printf(" S3 = %#016x S4 = %#016x\n", regs.S3, regs.S4)
fmt.Printf(" S5 = %#016x S6 = %#016x\n", regs.S5, regs.S6)
fmt.Printf(" S7 = %#016x S8 = %#016x\n", regs.S7, regs.S8)
fmt.Printf(" S9 = %#016x S10 = %#016x\n", regs.S9, regs.S10)
fmt.Printf(" S11 = %#016x\n", regs.S11)
}
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n FP registers (F0-F31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
}
fmt.Printf(" FCSR = %#x\n", v.FCSR)
}
// archReturnAddr reads the return address from RA (riscv64 convention).
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return regs.Ra, nil
}
// archSPLabel returns the SP register name for display.
func archSPLabel() string { return "SP" }
+427
View File
@@ -0,0 +1,427 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"bytes"
"fmt"
"io"
"os"
"path/filepath"
"runtime"
"strings"
"testing"
"time"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// Integration tests beyond the basic entry breakpoint: hardware watchpoints,
// conditional breakpoints, next/finish over a CALL, faulting kernels and the
// xstate vector-register readout. All drive a real ptrace session, so they
// run on amd64 hosts only.
// writeKernel writes an assembly source to a temporary file with the
// architecture suffix the assembler dispatcher expects.
func writeKernel(t *testing.T, src string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "kernel_amd64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatalf("write kernel: %v", err)
}
return path
}
// launchKernel launches a session for the kernel source and returns the
// session, its breakpoint manager and the function layout.
func launchKernel(t *testing.T, bin, path, funcName string, args []byte) (*Session, *Breakpoints, asm.FuncLayout) {
t.Helper()
k, err := verify.Load(path)
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
fl, err := k.Func(funcName)
if err != nil {
t.Fatalf("Func: %v", err)
}
if len(args) < fl.Args {
padded := make([]byte, fl.Args)
copy(padded, args)
args = padded
}
sess, err := Launch(bin, path, funcName, args)
if err != nil {
t.Fatalf("Launch: %v", err)
}
t.Cleanup(sess.Kill)
bm := NewBreakpoints(sess)
return sess, bm, fl
}
// runToEntry resumes the freshly launched debuggee until the breakpoint at
// the function entry traps, mirroring the REPL continue loop: the debuggee
// SIGSTOPs twice (launch barrier and entry barrier) before entering the JIT
// call.
func runToEntry(t *testing.T, sess *Session, bm *Breakpoints, entry uint64) {
t.Helper()
for range 50 {
for _, bp := range bm.All() {
bm.Reinsert(bp.Addr)
}
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
if sess.Exited() {
t.Fatal("debuggee exited before the entry breakpoint trapped")
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if bm.HandleTrap(&regs) != nil {
return
}
}
t.Fatal("no entry breakpoint trap after 50 resumes")
}
// captureStdout runs fn with os.Stdout redirected to a pipe and returns
// what it printed (the REPL writes its reports to stdout).
func captureStdout(t *testing.T, fn func()) string {
t.Helper()
r, w, err := os.Pipe()
if err != nil {
t.Fatalf("pipe: %v", err)
}
old := os.Stdout
os.Stdout = w
done := make(chan string, 1)
go func() {
b, _ := io.ReadAll(r)
done <- string(b)
}()
defer func() { os.Stdout = old }()
fn()
w.Close()
return <-done
}
// TestWatchpointArmRunHit proves the debug-register offsets: the watchpoint
// must fire on the store, with si_addr naming the watched address. The
// kernel writes its return value to ret+0(FP), which is the 8-byte word
// right above the stack pointer at entry.
func TestWatchpointArmRunHit(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func wpret() int64
TEXT ·wpret(SB), NOSPLIT, $0-8
MOVQ $0x5a5a5a5a5a5a5a5a, AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "wpret", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
watched := regs.RSP + 8 // ret+0(FP): the store target
slot := sess.FindFreeWatchpointSlot()
if slot < 0 {
t.Fatal("no free watchpoint slot")
}
if err := sess.SetWatchpoint(slot, watched, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint: %v (wrong debug-register offsets?)", err)
}
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
reason, addr := sess.StopInfo()
if reason != StopWatchpoint {
t.Fatalf("stop reason = %v, want StopWatchpoint (DR0-DR3/DR7 offsets are wrong)", reason)
}
if addr != watched {
t.Fatalf("watchpoint address = %#x, want %#x", addr, watched)
}
// The watched word holds the stored value: x86 data breakpoints are
// reported with the access complete.
if word, err := sess.Peek(watched); err != nil || word != 0x5a5a5a5a5a5a5a5a {
t.Errorf("watched word = %#x (err %v), want 0x5a5a5a5a5a5a5a5a", word, err)
}
if err := sess.ClearWatchpoint(slot); err != nil {
t.Fatalf("ClearWatchpoint: %v", err)
}
}
// TestConditionalBreakpointFalseThenTrue proves the false-condition path:
// the breakpoint steps over the original instruction, re-arms itself and
// keeps running silently, and the true condition stops exactly once with the
// register in the expected state.
func TestConditionalBreakpointFalseThenTrue(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func countdown(n int64) int64
TEXT ·countdown(SB), NOSPLIT, $0-16
MOVQ n+0(FP), CX
loop:
DECQ CX
CMPQ CX, $0
JNE loop
MOVQ CX, ret+8(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "countdown", []byte{8})
loopAddr := sess.CodeBase() + uint64(fl.Offset) + uint64(fl.Labels["loop"])
// The length of the breakpointed instruction, from a disassembly taken
// before the INT3 is patched in.
_, insnLen, err := sess.Disassemble(loopAddr)
if err != nil || insnLen <= 0 {
t.Fatalf("Disassemble at %#x: len=%d err=%v", loopAddr, insnLen, err)
}
cond := &Condition{Reg: "rcx", Op: "==", Value: 1}
bp, err := bm.SetWithCond(loopAddr, "loop", cond)
if err != nil {
t.Fatalf("SetWithCond: %v", err)
}
hits := 0
exited := false
for range 200 {
for _, b := range bm.All() {
bm.Reinsert(b.Addr)
}
if err := sess.Continue(); err != nil {
exited = true
break // the debuggee finished
}
if sess.Exited() {
exited = true
break
}
if sig := sess.LastSignal(); sig != 0 {
t.Fatalf("unexpected signal stop %v", sig)
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if hit := bm.HandleTrap(&regs); hit != nil {
hits++
if regs.RCX != 1 {
t.Fatalf("hit with RCX=%d, want 1", regs.RCX)
}
// Park after the instruction, as the REPL does.
if err := sess.Step(); err != nil {
t.Fatalf("Step: %v", err)
}
} else {
// A false evaluation must leave the debuggee past the whole
// original instruction: a PC inside it (trapAddr+1 on amd64)
// means the resume happens mid-instruction.
fresh, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if fresh.RIP > loopAddr && fresh.RIP < loopAddr+uint64(insnLen) {
t.Fatalf("false evaluation left the PC at %#x, inside the %d-byte instruction at %#x",
fresh.RIP, insnLen, loopAddr)
}
}
}
if hits != 1 {
t.Fatalf("conditional breakpoint hit %d times, want exactly 1 (false evaluations must run through silently)", hits)
}
if bp.Hits() != 1 {
t.Errorf("bp.Hits() = %d, want 1", bp.Hits())
}
if !exited || !sess.Exited() {
t.Fatal("debuggee did not run to completion after the conditional hit")
}
}
// TestNextAndFinishOverCall proves next and finish evaluate the trap with
// registers fetched after the stop: next lands exactly on the instruction
// after the CALL, and finish stops exactly on the return address.
func TestNextAndFinishOverCall(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func caller(x int64) int64
// The argument travels in AX: FP argument slots of CALL-bearing functions
// are an assembler concern outside this test's scope.
TEXT ·caller(SB), NOSPLIT, $0-16
MOVQ $5, AX
CALL ·bump(SB)
aftercall:
MOVQ AX, ret+8(FP)
RET
// func bump(x int64) int64
TEXT ·bump(SB), NOSPLIT, $0-0
ADDQ $3, AX
RET
`
path := writeKernel(t, kernel)
// next: step the prologue and the constant load (3 instructions), then
// step over the CALL and check the landing address and RAX.
sess, bm, fl := launchKernel(t, bin, path, "caller", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
afterOff := uint64(fl.Labels["aftercall"])
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("step 3\nnext\nregs\nquit\n"))
})
if !strings.Contains(out, fmt.Sprintf("func+%#x", afterOff)) {
t.Errorf("next did not land on the instruction after the CALL (func+%#x); output:\n%s", afterOff, out)
}
if !strings.Contains(out, "RAX = 0x0000000000000008") {
t.Errorf("callee did not run exactly once under next (want RAX=8); output:\n%s", out)
}
// finish: run to the return address read off the stack at entry.
sess2, bm2, fl2 := launchKernel(t, bin, path, "caller", nil)
entry2 := sess2.CodeBase() + uint64(fl2.Offset)
if _, err := bm2.Set(entry2, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess2, bm2, entry2)
regs, err := sess2.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
retAddr, err := sess2.Peek(regs.RSP)
if err != nil {
t.Fatalf("Peek return address: %v", err)
}
out2 := captureStdout(t, func() {
REPL(sess2, bm2, sess2.CodeBase(), fl2.Offset, fl2.Size, fl2.Args, nil, nil,
strings.NewReader("step 1\nfinish\nquit\n"))
})
want := fmt.Sprintf("finished, now at %#x\n", retAddr)
if !strings.Contains(out2, want) {
t.Errorf("finish stopped at the wrong PC; want %q in output:\n%s", want, out2)
}
}
// TestSignalStopSurfaced proves a faulting kernel surfaces as a reported
// stop instead of an infinite fault loop. A regression here hangs, so a
// watchdog fails the run rather than letting CI stall.
func TestSignalStopSurfaced(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func crash() int64
TEXT ·crash(SB), NOSPLIT, $0-8
XORQ AX, AX
MOVQ (AX), AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, _ := launchKernel(t, bin, path, "crash", nil)
timer := time.AfterFunc(time.Minute, func() {
panic("watchdog: the debugger hung on the faulting kernel instead of reporting the signal stop")
})
defer timer.Stop()
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), 0, 0, 0, nil, nil,
strings.NewReader("continue\nquit\n"))
})
if !strings.Contains(out, "stopped on signal") {
t.Errorf("SIGSEGV did not surface as a reported stop; output:\n%s", out)
}
if !sess.Exited() {
t.Error("debuggee should be killed by quit after the signal stop")
}
}
// TestGetVectorRegsXState proves the NT_X86_XSTATE readout: the request
// succeeds on a normal process and the XMM halves agree with
// PTRACE_GETFPREGS.
func TestGetVectorRegsXState(t *testing.T) {
// The FPRegs layout must mirror the kernel's user_fpregs_struct
// exactly: PTRACE_GETFPREGS fills all 512 bytes, so a short struct
// overflows the caller's memory.
if got := unsafe.Sizeof(FPRegs{}); got != 512 {
t.Fatalf("sizeof(FPRegs) = %d, want 512", got)
}
if got := unsafe.Offsetof(FPRegs{}.XMM); got != 160 {
t.Fatalf("offsetof(FPRegs.XMM) = %d, want 160", got)
}
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func vprobe() int64
TEXT ·vprobe(SB), NOSPLIT, $0-8
MOVQ $1, AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "vprobe", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
v, err := sess.GetVectorRegs()
if err != nil {
t.Fatalf("GetVectorRegs: %v", err)
}
fp, err := sess.GetFPRegs()
if err != nil {
t.Fatalf("GetFPRegs: %v", err)
}
for i := range 16 {
if !bytes.Equal(v.YMM[i][:16], fp.XMM[i][:]) {
t.Errorf("YMM%d low half %x, want the FPRegs XMM half %x", i, v.YMM[i][:16], fp.XMM[i][:])
}
}
}
+126
View File
@@ -0,0 +1,126 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"fmt"
"os"
"os/exec"
"path/filepath"
"runtime"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// buildGasm produces the gasm binary the debugger spawns as its debuggee.
func buildGasm(t *testing.T) string {
t.Helper()
if p := os.Getenv("GASM_TEST_BIN"); p != "" {
return p
}
bin := filepath.Join(t.TempDir(), "gasm")
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm")
out, err := cmd.CombinedOutput()
if err != nil {
t.Fatalf("build gasm: %v: %s", err, out)
}
return bin
}
// TestLaunchAndBreakpoint drives a real ptrace session end to end: launch the
// debuggee, break on the first instruction of the function and expect a
// breakpoint trap instead of a clean exit.
func TestLaunchAndBreakpoint(t *testing.T) {
if runtime.GOARCH != "amd64" {
t.Skip("runs only on amd64 hosts")
}
// The tracer is the OS thread that forked the debuggee (PTRACE_TRACEME
// binds the relation to that thread); every ptrace request must come
// from the same thread, so pin the test goroutine to one thread.
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernelPath = "../testdata/verify/basic_amd64.s"
k, err := verify.Load(kernelPath)
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
fl, err := k.Func("wideCopy")
if err != nil {
t.Fatalf("Func: %v", err)
}
sess, err := Launch(bin, kernelPath, "wideCopy", make([]byte, fl.Args))
if err != nil {
t.Fatalf("Launch: %v", err)
}
t.Cleanup(sess.Kill)
bm := NewBreakpoints(sess)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
// The INT3 must be visible in the debuggee's memory.
word, err := sess.Peek(entry)
if err != nil {
t.Fatalf("Peek: %v", err)
}
if b := word & 0xFF; b != 0xCC {
t.Fatalf("int3 not patched: first byte %#02x at %#x", b, entry)
}
// The debuggee raises a second SIGSTOP after the launch barrier (the
// child's RunTarget marks its entry), so like the REPL and the cover
// mode the test keeps resuming until the breakpoint trap arrives.
for range 10 {
if err := sess.Continue(); err != nil {
st, _ := os.ReadFile(fmt.Sprintf("/proc/%d/stat", sess.Pid()))
status, _ := os.ReadFile(fmt.Sprintf("/proc/%d/status", sess.Pid()))
t.Fatalf("Continue: %v\nstate: %s\n%s", err, fieldName(st), statusDump(status))
}
if sess.Exited() {
t.Fatal("debuggee exited instead of trapping on the breakpoint")
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if bp := bm.HandleTrap(&regs); bp != nil {
if bp.Addr != entry {
t.Fatalf("trap at %#x, want %#x", bp.Addr, entry)
}
return // trap on the entry breakpoint: the whole flow works
}
}
t.Fatal("no breakpoint trap after 10 resumes")
}
func fieldName(stat []byte) string {
f := strings.Split(string(stat), " ")
if len(f) > 2 {
return "state=" + f[2]
}
return "no stat"
}
func statusDump(b []byte) string {
var out []string
for l := range strings.SplitSeq(string(b), "\n") {
if strings.HasPrefix(l, "State") || strings.HasPrefix(l, "Pid") ||
strings.HasPrefix(l, "PPid") || strings.HasPrefix(l, "TracerPid") ||
strings.HasPrefix(l, "Threads") || strings.HasPrefix(l, "SigPnd") ||
strings.HasPrefix(l, "SigBlk") || strings.HasPrefix(l, "SigIgn") {
out = append(out, l)
}
}
return strings.Join(out, "\n")
}

Some files were not shown because too many files have changed in this diff Show More