Compare commits

...
130 Commits
Author SHA1 Message Date
petrbalvin cf6bc6987e fix(ci): pass the upload file to curl, not its interpolation
Test / test (push) Successful in 2m26s
Release / gates (push) Successful in 2m25s
Release / build (amd64, linux) (push) Successful in 1m15s
Release / build (arm64, linux) (push) Successful in 1m16s
Release / build (loong64, linux) (push) Successful in 1m16s
Release / build (riscv64, linux) (push) Successful in 1m16s
Release / release (push) Successful in 34s
2026-09-22 01:31:49 +02:00
petrbalvin ff7b1452b1 docs: name 0.35.0 as the supported release
Test / test (push) Successful in 2m33s
Release / gates (push) Successful in 2m29s
Release / build (amd64, linux) (push) Successful in 1m18s
Release / build (arm64, linux) (push) Successful in 1m20s
Release / build (loong64, linux) (push) Successful in 1m17s
Release / build (riscv64, linux) (push) Successful in 1m26s
Release / release (push) Failing after 35s
2026-09-22 00:52:56 +02:00
petrbalvin 517c1cea25 chore: prepare release v0.35.0
Test / test (push) Successful in 2m33s
Release / gates (push) Failing after 46s
Release / build (amd64, linux) (push) Skipped
Release / build (arm64, linux) (push) Skipped
Release / build (loong64, linux) (push) Skipped
Release / build (riscv64, linux) (push) Skipped
Release / release (push) Skipped
2026-09-22 00:44:10 +02:00
petrbalvin a3e3010e0f fix(cmd): resolve the runtime header test GOROOT from the go command
Test / test (push) Successful in 2m39s
2026-09-21 22:46:07 +02:00
petrbalvin 057c4eb545 docs: complete the release delta in the changelog and readme 2026-09-21 22:45:56 +02:00
petrbalvin f720381d43 feat(asm): the segment-absolute and crash-store forms GOROOT writes
Test / test (push) Failing after 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-21 22:19:53 +02:00
petrbalvin 2c9042d62c feat(asm): PCALIGN alignment on amd64
Assisted-by: GLM 5.3 Flash
2026-09-21 22:00:30 +02:00
petrbalvin 82ef289d3a feat(asm): the immediate multiply and arm64 indirect branches GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:50:11 +02:00
petrbalvin 7246b0e002 feat(asm): the TLS access pair in the toolchain's one-instruction form
Assisted-by: GLM 5.3 Flash
2026-09-21 21:35:15 +02:00
petrbalvin 8cfd40aac8 feat(asm): the operand forms and defines GOROOT writes
Assisted-by: GLM 5.3 Flash
2026-09-21 21:17:34 +02:00
petrbalvin 5382c9a8e4 feat(audit): list every corpus failure per architecture 2026-09-21 21:17:34 +02:00
petrbalvin 53de91b2df docs(asm): describe the four target architectures
Test / test (push) Failing after 2m23s
Assisted-by: GLM 5.3 Flash
2026-09-21 20:15:55 +02:00
petrbalvin 8a36af7c7d docs(asm): generate the instruction appendices
Assisted-by: GLM 5.3 Flash
2026-09-21 20:15:55 +02:00
petrbalvin e9789ce3f4 chore(arch): regenerate the instruction tables 2026-09-21 20:15:55 +02:00
petrbalvin 837231c068 docs(asm): open the assembly language reference
Assisted-by: GLM 5.3 Flash
2026-09-21 19:49:04 +02:00
petrbalvin 95025be1bc docs(changelog): describe the encoder entries by content
Test / test (push) Failing after 2m33s
2026-09-21 19:20:01 +02:00
petrbalvin 03a964bb2d docs(goobj): document the GOOBJ object file format 2026-09-21 19:19:53 +02:00
petrbalvin 123a16e346 docs(readme): state the documentation goal 2026-09-21 18:35:27 +02:00
petrbalvin 9701812bee docs: changelog for the completeness waves
Test / test (push) Failing after 3m6s
Assisted-by: GLM 5.3 Flash
2026-09-21 02:04:44 +02:00
petrbalvin 29ac03468e feat(amd64): floating-point immediates through a synthesised pool
Assisted-by: GLM 5.3 Flash
2026-09-21 02:04:44 +02:00
petrbalvin bfb7701db1 feat(amd64): emit the quad-register EVEX families
Assisted-by: GLM 5.3 Flash
2026-09-21 02:02:19 +02:00
petrbalvin e8b6ff5d7c fix(parser): fold a signed parenthesised displacement expression
Test / test (push) Failing after 2m21s
Assisted-by: GLM 5.3 Flash
2026-09-21 00:45:33 +02:00
petrbalvin 1456907000 feat(riscv64,loong64): operand tail, float DATA and honest port classification
Assisted-by: GLM 5.3 Flash
2026-09-21 00:44:47 +02:00
petrbalvin ec1c521187 feat(cmd): GOOS-aware headers, audit battery shapes and semicolon spacing
Test / test (push) Failing after 2m21s
Assisted-by: GLM 5.3 Flash
2026-09-20 22:02:46 +02:00
petrbalvin a7744c24bd fix(parser): substitute macro parameters behind element selectors
Assisted-by: GLM 5.3 Flash
2026-09-20 22:02:19 +02:00
petrbalvin 522e6f2ae8 feat(parser): bracket register ranges, index-only VSIB and bare trailing immediates
Assisted-by: GLM 5.3 Flash
2026-09-20 22:02:19 +02:00
petrbalvin 81d4bd81e4 test(verify): register the wave kernels
Test / test (push) Failing after 2m20s
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:31 +02:00
petrbalvin 687678a2ea feat(elf): emit data relocations on arm64, riscv64 and loong64
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
petrbalvin b0f9071bf5 feat(arm64): whole-vector moves, bookkeeping ops and truncating-move lowering
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
petrbalvin 81e2673923 feat(amd64): encode the AVX-512 and BMI corpus families
Assisted-by: GLM 5.3 Flash
2026-09-20 21:17:20 +02:00
petrbalvin 75e9fd771b feat(parser): split plain statements on semicolons in the raw parse
Test / test (push) Failing after 2m30s
Assisted-by: GLM 5.3 Flash
2026-09-20 19:15:35 +02:00
petrbalvin 863926abd6 test(verify): register the loong64 vector kernels
Assisted-by: GLM 5.3 Flash
2026-09-20 19:15:05 +02:00
petrbalvin 241e7256f6 fix(arm64): reject bare BTI with a diagnostic and accept the full family
Assisted-by: GLM 5.3 Flash
2026-09-20 19:15:05 +02:00
petrbalvin 6556b85abf feat(asm): symbol-valued DATA, division slash in symbols and plain semicolons
Assisted-by: GLM 5.3 Flash
2026-09-20 19:15:05 +02:00
petrbalvin 289cabe993 feat(loong64): encode the full LSX and LASX table
Assisted-by: GLM 5.3 Flash
2026-09-20 19:14:43 +02:00
petrbalvin d6cf7cfa44 fix(format): keep statement separators and canonical macro bodies
Assisted-by: GLM 5.3 Flash
2026-09-20 19:14:43 +02:00
petrbalvin 4cc2f0eba5 feat(cmd): generate go_asm.h for package-context assembly
Assisted-by: GLM 5.3 Flash
2026-09-20 19:14:43 +02:00
petrbalvin 97dfaa7526 docs: changelog for macro expansion and the corrected corpus audit
Test / test (push) Successful in 2m14s
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 66aa4dbc8b test(verify): register the campaign kernels in the ground-truth suites
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin dce5d31462 feat(amd64): LOCK and REP prefixes, literal data pseudo-ops and ADJSP
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 9dc3987e02 feat(riscv64,loong64): PCALIGN, branch relaxation and operand shapes
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 9b238a525a feat(arm64): wide immediates, SIMD compare and system operand forms
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin ad82aac663 feat(parser): macro expansion, conditionals and include splicing with -I
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 0629f5e2df feat(arm64): assemble PCALIGN padding and BYTE literal bytes
Test / test (push) Successful in 2m16s
Assisted-by: GLM 5.3 Flash
2026-09-20 11:49:05 +02:00
petrbalvin ecb203dcf5 fix(lexer): treat trailing CR as line end so comment text is idempotent
Test / test (push) Successful in 2m13s
Assisted-by: GLM 5.3 Flash
2026-09-20 11:40:39 +02:00
petrbalvin 6c672567f3 feat(amd64): assemble the double-shift and static-SB operand shapes
Assisted-by: GLM 5.3 Flash
2026-09-20 11:40:39 +02:00
petrbalvin cc6e416c59 fix(lint): exempt shift counts, SETcc and ABIInternal from false positives
Assisted-by: GLM 5.3 Flash
2026-09-20 11:40:39 +02:00
petrbalvin c66a47973a fix(format): preserve square brackets in SIMD operands
Test / test (push) Successful in 2m15s
Assisted-by: GLM 5.3 Flash
2026-09-20 09:58:25 +02:00
petrbalvin 9629897202 docs: changelog and readme for the instruction wave and the honest corpus rate
Test / test (push) Successful in 2m17s
Assisted-by: GLM 5.3 Flash
2026-09-20 06:45:03 +02:00
petrbalvin 5399a8a724 feat(audit): probe the new operand shapes and measure attemptable files
Assisted-by: GLM 5.3 Flash
2026-09-20 06:45:03 +02:00
petrbalvin de5d9f358e feat(riscv64,loong64): encode AMO atomics, vector slices and bit ops
Assisted-by: GLM 5.3 Flash
2026-09-20 06:44:51 +02:00
petrbalvin ca3fdce0e0 feat(arm64): encode pairs, atomics, crypto, system and NEON slices
Assisted-by: GLM 5.3 Flash
2026-09-20 06:44:51 +02:00
petrbalvin fc2d92eabd feat(amd64): encode the GOROOT instruction families
Assisted-by: GLM 5.3 Flash
2026-09-20 06:44:51 +02:00
petrbalvin 39d2e80145 ci(release): refuse empty assets and verify what the release serves
Test / test (push) Successful in 2m10s
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 02:01:36 +02:00
petrbalvin 9f4f949c1f chore: prepare release v0.34.0
Test / test (push) Successful in 2m11s
Release / gates (push) Successful in 2m11s
Release / build (amd64, linux) (push) Successful in 1m13s
Release / build (arm64, linux) (push) Successful in 1m10s
Release / build (loong64, linux) (push) Successful in 1m12s
Release / build (riscv64, linux) (push) Successful in 1m33s
Release / release (push) Successful in 58s
2026-09-20 01:44:23 +02:00
petrbalvin f0d5238c47 docs: state the validation status and correct claims the material contradicts
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin 2931bbd6b2 ci(release): refuse a tag the security policy does not name
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin 63562a503a test(justfile): run the CLI and debugger tests outside the coverage set
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 01:40:51 +02:00
petrbalvin e836d6150d docs: changelog entry for the loong64 JIT enablement
Test / test (push) Successful in 2m7s
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin 8a51b060da feat(cmd): enable loong64 JIT execution, all trampolines qemu-validated
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin d3d47db727 test(verify): seed the arm64 ABI kernel arguments
Assisted-by: GLM 5.3
2026-09-20 00:57:02 +02:00
petrbalvin 0758556b7d docs: changelog entries for the parity round and corpus number
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin ddb8440340 fix(cmd): padding-aware ground-truth comparison
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin f15ff66fb1 fix(riscv64): accept the g spelling of the goroutine register
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin 187e4856d3 feat(amd64): encode the mixed-width extend family and PMOVMSKB
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin d315a998ce fix(arm64): store-exclusive operand order and large-frame parity
Assisted-by: GLM 5.3
2026-09-20 00:38:24 +02:00
petrbalvin a6f3828c02 docs: changelog entries for the review fixes
Test / test (push) Successful in 2m4s
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin e3b35bb817 style(testdata): canonical gasm formatting for the verify kernels
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin eb0a89e58d ci(release): state the version contract inline
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin dd32d9e66e chore(justfile): one-line install-man comment and long flag forms
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin 3a73acb20a docs: drop process labels and refresh the architecture and manual pages
Assisted-by: GLM 5.3
2026-09-19 23:49:27 +02:00
petrbalvin 7604a9443f fix(cmd): usage exit codes, asm output file and cross-arch ground truth
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin b3908fc43d fix(lsp): parse-error survival, symbol ranges and UTF-16 positions
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin eb8b0cd316 fix(lint): trailing-label CFG guard and the goroutine alias
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin a8bfd54ed2 fix(debug): hardware watchpoints, signal stops and breakpoint restore
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin 375182ef1f fix(verify): arm64 stack save, adaptive canary and host gating
Assisted-by: GLM 5.3
2026-09-19 23:49:19 +02:00
petrbalvin 87b1081c53 fix(goobj): external package and symbol indices and arm64 pair relocations
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin f3c8510a58 fix(elf): relocation records, DWARF tables and per-architecture frame data
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin ebdf14939f fix(loong64): FP immediates through R30 and unsigned branch forms
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin 79a2c16bac fix(riscv64): compressed store offsets, FENCE and branch range checks
Assisted-by: GLM 5.3
2026-09-19 23:49:13 +02:00
petrbalvin 401386956c fix(arm64): encode shifts, divides and multiplies and align sizes with emission
Assisted-by: GLM 5.3
2026-09-19 23:49:07 +02:00
petrbalvin 4258131a3a fix(amd64): correct guard displacements, frameless FP offsets and immediate ranges
Assisted-by: GLM 5.3
2026-09-19 23:49:07 +02:00
petrbalvin 94e09e8070 fix(format): preserve flag separators and normalise CRLF input
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
petrbalvin 7aefe6a42d fix(parser): parse ABI markers and keep TEXT decls usable on errors
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
petrbalvin ac1c05c793 fix(lexer): tokenise the flag separator and handle NUL and invalid UTF-8
Assisted-by: GLM 5.3
2026-09-19 23:48:47 +02:00
petrbalvin 93c47a312a feat(docs): man pages for gasm and every command, guarded against CLI drift
Test / test (push) Successful in 2m4s
Assisted-by: GLM 5.3 Flash
2026-09-19 21:18:43 +02:00
petrbalvin 708d0a0a5e docs: trim the changelog entries to user-visible deltas
Assisted-by: GLM 5.3 Flash
2026-09-19 20:54:56 +02:00
petrbalvin 3c8f7cb411 test(format): pin the fuzz-found crashers as regression seeds
Assisted-by: GLM 5.3 Flash
2026-09-19 20:48:51 +02:00
petrbalvin 7c5b7a1419 docs: add the changelog entries and the corpus number to the readme
Assisted-by: GLM 5.3 Flash
2026-09-19 20:48:51 +02:00
petrbalvin bc3f448738 feat(format): fuzz targets for the parser and formatter
Assisted-by: GLM 5.3 Flash
2026-09-19 20:41:43 +02:00
petrbalvin f37f183577 feat(riscv64): GOROOT instruction shapes, DATA order and offset expressions
Assisted-by: GLM 5.3 Flash
2026-09-19 19:58:43 +02:00
petrbalvin 1e77e58250 feat(gasm): audit a .s corpus with audit-instructions --corpus
Assisted-by: GLM 5.3 Flash
2026-09-19 19:27:30 +02:00
petrbalvin 1d0969ed64 feat(gasm): select the asm and diff architecture with -GOARCH
Assisted-by: GLM 5.3 Flash
2026-09-19 19:20:47 +02:00
petrbalvin 23c001be51 feat(asm): encode indirect JMP and CALL on all four architectures
Assisted-by: GLM 5.3 Flash
2026-09-19 19:17:07 +02:00
petrbalvin 96e81cc98d docs: add the Plan 9 assembly case and real-use note to the README
Test / test (push) Successful in 2m6s
2026-09-19 18:06:18 +02:00
petrbalvin c834d98210 docs: bring the document set into the standard shape
Test / test (push) Successful in 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 03d6d4da54 style: put the repository assembly in gasm fmt canonical form
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 0b42ce7952 style: use one spelling for colour across the CLI
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 288a64ccd2 ci: align the pipelines with the hand-written templates
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin 5fddfa704b build: declare the exact toolchain and the canonical recipes
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:18 +02:00
petrbalvin a2bb5eeb4e chore: drop the stale comment from the ignore list
Assisted-by: GLM 5.3 Flash
2026-09-17 20:33:14 +02:00
petrbalvin 48449b7a7f build: declare the go1.27.1 toolchain
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 3de043c494 docs: add SECURITY.md and record the round in the CHANGELOG
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 0078f7be5c style: purge em dashes from the produced text
Assisted-by: GLM 5.3 Flash
2026-09-16 23:12:31 +02:00
petrbalvin 6a7317d141 chore: trim the ignore list to the convention
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin d08523caa5 docs: move the recipe and version descriptions with the behaviour
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 20e4b8d9c4 ci: align the pipelines with the hand-written templates
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 61f4247cef refactor(gasm): report the toolchain-recorded version
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 049872ddff build: restore the canonical justfile recipe set
Assisted-by: GLM 5.3 Flash
2026-09-16 22:53:01 +02:00
petrbalvin 3669f64ff6 build: install the gasm binary into the user-local bin directory
Test / vet (push) Successful in 46s
Test / test (push) Successful in 2m44s
Test / build (push) Successful in 42s
2026-09-14 23:41:21 +02:00
petrbalvin fff9f75595 chore: prepare release v0.33.0
Release / build (amd64, linux) (push) Successful in 49s
Release / build (arm64, linux) (push) Successful in 43s
Release / build (loong64, linux) (push) Successful in 46s
Release / build (riscv64, linux) (push) Successful in 45s
Test / vet (push) Successful in 47s
Release / release (push) Successful in 18s
Test / test (push) Successful in 2m39s
Test / build (push) Successful in 43s
2026-09-14 23:36:19 +02:00
petrbalvin 40476546df fix(asm): close the oracle parity gaps in frame addressing and calls 2026-09-14 23:25:14 +02:00
petrbalvin 70218e84ba feat(asm): emit the loong64 stack-split guard for big frames 2026-09-14 22:38:58 +02:00
petrbalvin db50b98179 feat(asm): emit the loong64 stack-split guard for small and medium frames 2026-09-14 21:21:58 +02:00
petrbalvin 2e2c0b82a0 feat(asm): emit the riscv64 stack-split guard and fix large-frame addressing 2026-09-14 21:09:13 +02:00
petrbalvin 8dc1e98ca1 feat(asm): emit the arm64 stack-split guard and morestack block 2026-09-14 20:49:03 +02:00
petrbalvin 1d8e68c574 feat(asm): emit the amd64 stack-split guard and morestack block 2026-09-14 20:35:55 +02:00
petrbalvin 89fa6ea15e feat(lsp): resolve definition and references across open documents 2026-09-14 18:50:55 +02:00
petrbalvin edc20ffa97 feat(cmd): add gasm dis and share the decoder with the debugger 2026-09-14 18:47:08 +02:00
petrbalvin 50db6615b2 feat(cmd): add gofmt-style -l and -d modes to gasm fmt 2026-09-14 18:47:08 +02:00
petrbalvin 95f1d6f083 style: replace em dashes in the scaffold comments 2026-09-14 18:22:25 +02:00
petrbalvin f43e791e5a chore: add .qwen to the gitignore metadata block 2026-09-14 18:22:18 +02:00
petrbalvin 1691c81095 style: replace em and en dashes across sources 2026-09-14 18:22:18 +02:00
petrbalvin 2db563be07 refactor(cmd): consolidate cross-arch verify and drop dead code 2026-09-14 18:22:00 +02:00
petrbalvin 4f190ee1a2 refactor(debug): move watchpoint slot state into the session 2026-09-14 18:22:00 +02:00
petrbalvin 909f874797 fix(lsp): recover from handler panics and decode client uris 2026-09-14 18:22:00 +02:00
petrbalvin e307bf830f fix(lint): guard unnamed TEXT and refresh the textflag table 2026-09-14 18:22:00 +02:00
petrbalvin 953c258d6a fix(asm): make arm64 and loong64 relocations match the toolchain 2026-09-14 18:22:00 +02:00
petrbalvin c6f0286732 fix(asm): encode amd64 frame adjustments above 127 bytes with imm32 2026-09-14 18:22:00 +02:00
petrbalvin f5fc22d390 fix(parser): reject malformed TEXT frames and parse signed frame sizes 2026-09-14 18:22:00 +02:00
330 changed files with 47049 additions and 4147 deletions
+37
View File
@@ -0,0 +1,37 @@
# Race, Go. Dispatched by hand, and never a gate on a push or a tag: the release tag is
# cut only after `just gates` has already raced the tree, so this workflow is the
# explicit second opinion, not a step of the release.
#
# The race detector roughly doubles both time and memory, which the shared runner box
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
# task; here it is a decision rather than a routine.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Race
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
race:
runs-on: fedora
timeout-minutes: 20
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install gcc
# The race detector needs cgo and the runner image carries no C compiler.
run: dnf install -y gcc
- name: Race
run: go test -race -count=1 -timeout 10m ./...
+319 -86
View File
@@ -1,73 +1,222 @@
# Release — gasm binaries. Runs on version tags (v0.28.0) pushed to main.
# Release, Go binaries. Runs on version tags (v1.2.3) pushed to main.
#
# The module sits at the repository root: the toolchain records a version only for a root
# module, measured on go1.27.1, so a build of a module in a subdirectory reports (devel)
# even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never pass for
# it. A Go repository is one module at the root.
#
# The version contract these steps implement: nothing is injected. The toolchain records
# the tag into the binary's build information, so the build simply has to happen at the
# tag, which the trigger guarantees.
#
# The gates run in their own job, once, before the matrix, minus the race detector: race
# never runs on a push path or a tag, and the local gate raced this tree before the tag
# was cut. Putting the gates inside the matrix would run the whole suite once per target
# on the box that also hosts the forge. Each job validates the tag for itself rather than
# passing a value between jobs, so no workflow feature has to be trusted for the version
# to reach the file name.
name: Release
on:
push:
tags: ["v*"]
env:
# The box is shared with the forge, so parallelism is bounded on purpose. The gates job
# needs it most; the build jobs inherit it for their parallel compilation.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
gates:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install Perl
# Perl for the steps below. The install is a no-op where the package
# is already present.
run: dnf install -y perl
- name: Validate the tag
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
print qq{tag $v\n};
'
- name: Security policy names this release
# The supported-versions table is the one part of SECURITY.md that
# carries a version, so it goes stale the moment a tag is cut. Fail
# here rather than publish a policy naming the previous release.
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
(my $nv = $v) =~ s/^v//;
open(my $f, q{<}, q{SECURITY.md}) or die qq{SECURITY.md: $!\n};
local $/;
my $t = <$f>;
close $f;
$t =~ m{^\|\s*\Q$nv\E\s*\|\s*yes\s*\|}m
or die qq{ERROR: SECURITY.md does not name $nv as supported; update the table before releasing.\n};
print qq{SECURITY.md names $nv\n};
'
- name: Build
run: go build ./...
- name: Format
run: |
perl -e '
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
'
- name: Vet
run: go vet ./...
- name: Modernise
run: go fix -diff ./...
- name: Tests
# The same command as in test.yml, so the floor is the same number everywhere.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The same command as in test.yml: the CLI's exit codes and manual-page guard,
# and the debugger's architecture-neutral units, run outside the floor.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Coverage floor
run: |
perl -e '
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
'
build:
runs-on: fedora
timeout-minutes: 25
needs: gates
strategy:
fail-fast: false
matrix:
# Portable targets: amd64, arm64, loong64 and riscv64 on Linux, at the toolchain
# default level. No 32-bit, no wasm, no macOS, no Windows. FreeBSD stays out until
# verify/jit.go ports off syscall.Mprotect: the Go syscall package defines no
# Mprotect for freebsd, and verify/jit.go:50 calls it to drop the write bit from
# the JIT mapping, so every freebsd target fails to build with "undefined:
# syscall.Mprotect" (verified for amd64, arm64 and riscv64 on go1.27.1).
include:
- goos: linux
goarch: amd64
- goos: linux
goarch: arm64
- goos: linux
goarch: riscv64
- goos: linux
goarch: loong64
- goos: linux
goarch: riscv64
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version: "1.27"
go-version-file: go.mod
cache: true
- name: Download dependencies
run: go mod download
- name: Install Perl
run: dnf install -y perl
- name: Validate tag and build
id: build
- name: Validate the tag
id: version
env:
VERSION: ${{ gitea.ref_name }}
run: |
set -euo pipefail
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
(my $nv = $v) =~ s{^v}{};
open(my $o, q{>>}, $ENV{GITEA_OUTPUT}) or die qq{GITEA_OUTPUT: $!};
print $o qq{version_no_v=$nv\n};
close($o);
print qq{version $nv\n};
'
if ! echo "$VERSION" | grep -qE '^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$'; then
echo "ERROR: expected a semver tag like v1.2.3, got: '$VERSION'"
exit 1
fi
VERSION_NO_V="${VERSION#v}"
echo "version_no_v=${VERSION_NO_V}" >> "$GITEA_OUTPUT"
mkdir -p bin
GOOS=${{ matrix.goos }} GOARCH=${{ matrix.goarch }} CGO_ENABLED=0 \
go build -ldflags "-s -w -X main.version=${VERSION_NO_V}" \
-o "bin/gasm-${VERSION_NO_V}-${{ matrix.goos }}-${{ matrix.goarch }}" \
./cmd/gasm
- name: Build
env:
VERSION_NO_V: ${{ steps.version.outputs.version_no_v }}
GOOS: ${{ matrix.goos }}
GOARCH: ${{ matrix.goarch }}
CGO_ENABLED: "0"
run: |
# Nothing is injected. The toolchain records the tag into the binary's build
# information, so the version is right because this build happens at the tag, and
# there is no path for anyone to get wrong. -s -w only strips symbols.
go build -ldflags "-s -w" -o "bin/gasm-${VERSION_NO_V}-${GOOS}-${GOARCH}" ./cmd/gasm
# Artifacts stay on v3: v4 and later detect Gitea as GHES and abort.
- name: Upload artifact
uses: actions/upload-artifact@v3
with:
name: gasm-${{ matrix.goos }}-${{ matrix.goarch }}
path: bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
path: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
if-no-files-found: error
- name: Smoke test
# Only a binary matching the runner can be run here. The check is not that --version
# exits cleanly but that it reports the tag and nothing more: a build outside version
# control reports (devel), and a build whose tree was dirty reports +dirty, and both
# would otherwise be published.
if: matrix.goos == 'linux' && matrix.goarch == 'amd64'
env:
TAG: ${{ gitea.ref_name }}
BIN: bin/gasm-${{ steps.version.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
run: |
chmod +x bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }}
./bin/gasm-${{ steps.build.outputs.version_no_v }}-${{ matrix.goos }}-${{ matrix.goarch }} --version
perl -e '
my $want = $ENV{TAG} // die qq{ERROR: no tag\n};
open(my $bin, q{-|}, $ENV{BIN}, q{--version}) or die qq{$ENV{BIN}: $!};
my $got = <$bin>;
close($bin);
$got = defined $got ? $got : q{};
chomp $got;
index($got, $want) >= 0
or die qq{ERROR: the binary printed "$got", which does not contain $want. Version control was disabled, so there is no recorded version.\n};
index($got, q{+dirty}) < 0
or die qq{ERROR: the binary printed "$got". The tree was dirty at build time, which means the checkout was not the tag, or the build artefacts are not ignored.\n};
print qq{$ENV{BIN} reports $got\n};
'
release:
runs-on: fedora
timeout-minutes: 15
needs: build
permissions:
# contents: read is required for the checkout: a job that declares any
# permissions gets a token scoped to exactly those, and releases: write
# alone leaves the fetch with no read access, which Gitea answers with
# a 404 "Repository not found". Verified on the instance 2026-09-16.
contents: read
releases: write
steps:
- uses: actions/checkout@v7
@@ -77,81 +226,165 @@ jobs:
with:
path: dist
- name: Extract CHANGELOG section
- name: Install Perl
run: dnf install -y perl
- name: Extract the CHANGELOG section
env:
VERSION: ${{ gitea.ref_name }}
run: |
set -euo pipefail
VERSION_NO_V="${VERSION#v}"
# Each step derives what it needs from the tag, so no value has to travel between
# jobs.
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ s{^v}{};
open(my $vout, q{>}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
print $vout $v;
close($vout);
open(my $in, q{<}, q{CHANGELOG.md}) or die qq{CHANGELOG.md: $!};
my @lines = <$in>;
close($in);
my ($start, $end) = (-1, scalar @lines);
for my $i (0 .. $#lines) {
if ($start < 0) { $start = $i if $lines[$i] =~ m{^##\s+\[\Q$v\E\]} }
elsif ($lines[$i] =~ m{^##\s+\[}) { $end = $i; last }
}
$start >= 0 or die qq{ERROR: no CHANGELOG section for $v, expected a heading like: ## [$v] - YYYY-MM-DD\n};
my @body = grep { m{\S} } @lines[$start + 1 .. $end - 1];
@body or die qq{ERROR: the CHANGELOG section for $v is empty\n};
open(my $out, q{>}, q{release-body.md}) or die qq{release-body.md: $!};
print $out @body;
close($out);
printf qq{notes for %s: %d lines\n}, $v, scalar @body;
'
sed -n "/^## \[${VERSION_NO_V}\] /,/^## \[/p" CHANGELOG.md \
| sed '$d' \
| tail -n +2 \
> release-body.md
- name: Build the release request
run: |
perl -e '
open(my $vin, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
my $v = <$vin>;
close($vin);
chomp $v;
open(my $in, q{<:raw}, q{release-body.md}) or die qq{release-body.md: $!};
my $body = do { local $/; <$in> };
close($in);
# Byte-oriented escaping: JSON is UTF-8, so non-ASCII passes through and only the
# characters JSON forbids are rewritten.
$body =~ s/([\\"])/\\$1/g;
$body =~ s/\t/\\t/g;
$body =~ s/\r//g;
$body =~ s/\n/\\n/g;
$body =~ s/([\x00-\x08\x0b\x0c\x0e-\x1f])/sprintf(q{\u%04x}, ord($1))/ge;
my $json = sprintf(qq{{"tag_name":"v%s","name":"v%s","body":"%s","draft":false,"prerelease":false}}, $v, $v, $body);
open(my $out, q{>}, q{release.json}) or die qq{release.json: $!};
print $out $json;
close($out);
print qq{release.json written for v$v\n};
'
if [ ! -s release-body.md ]; then
echo "ERROR: no CHANGELOG section found for ${VERSION_NO_V}"
echo "Expected a heading like: ## [${VERSION_NO_V}] — YYYY-MM-DD"
exit 1
fi
- name: Create release
- name: Create the release
env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
GITEA_SERVER_URL: ${{ gitea.server_url }}
GITEA_REPOSITORY: ${{ gitea.repository }}
GITEA_REF_NAME: ${{ gitea.ref_name }}
run: |
set -euo pipefail
BODY=$(sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\t/\\t/g' -e 's/\r//g' release-body.md | sed ':a;N;$!ba;s/\n/\\n/g')
BODY="\"${BODY}\""
response=$(curl -sS -w '\n%{http_code}' \
-H "Authorization: token ${GITEA_TOKEN}" \
-H "Content-Type: application/json" \
-X POST \
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases" \
-d "{\"tag_name\":\"${GITEA_REF_NAME}\",\"name\":\"${GITEA_REF_NAME}\",\"body\":${BODY},\"draft\":false,\"prerelease\":false}")
http_code=$(echo "$response" | tail -1)
payload=$(echo "$response" | sed '$d')
echo "HTTP ${http_code}"
if [ "$http_code" != "201" ]; then
echo "Failed to create release: ${payload}"
exit 1
fi
RELEASE_ID=$(echo "$payload" | grep -oE '"id"[[:space:]]*:[[:space:]]*[0-9]+' | head -1 | grep -oE '[0-9]+')
echo "Created release ID=${RELEASE_ID}"
printf '%s' "${RELEASE_ID}" > release-id.txt
perl -e '
my @cmd = (q{curl}, q{-sS}, q{-o}, q{response.json}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/json},
q{-X}, q{POST},
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases},
q{--data-binary}, q{@release.json});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>;
my $ok = close($curl);
my $exit = $? >> 8;
$code = defined $code ? $code : q{};
$ok or die qq{ERROR: curl failed (exit $exit) calling $ENV{GITEA_SERVER_URL}\n};
open(my $r, q{<:raw}, q{response.json}) or die qq{response.json: $!};
my $body = do { local $/; <$r> };
close($r);
$code eq q{201} or die qq{ERROR: the release was not created, HTTP $code: $body\n};
$body =~ m{"id"\s*:\s*([0-9]+)} or die qq{ERROR: no release id in the response: $body\n};
open(my $o, q{>}, q{release-id.txt}) or die qq{release-id.txt: $!};
print $o $1;
close($o);
print qq{release id $1\n};
'
- name: Upload assets
env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
GITEA_SERVER_URL: ${{ gitea.server_url }}
GITEA_REPOSITORY: ${{ gitea.repository }}
GITEA_REF_NAME: ${{ gitea.ref_name }}
run: |
set -euo pipefail
RELEASE_ID=$(cat release-id.txt)
for binary in dist/gasm-*/gasm-*; do
[ -f "$binary" ] || continue
fname=$(basename "$binary")
echo "Uploading ${fname}..."
http_code=$(curl -sS -o /dev/null -w '%{http_code}' \
-H "Authorization: token ${GITEA_TOKEN}" \
-H "Content-Type: application/octet-stream" \
-X POST \
--data-binary "@${binary}" \
"${GITEA_SERVER_URL}/api/v1/repos/${GITEA_REPOSITORY}/releases/${RELEASE_ID}/assets?name=${fname}")
echo " HTTP ${http_code}"
if [ "$http_code" != "201" ]; then
echo "Failed to upload ${fname}"
exit 1
fi
done
echo "Release ${GITEA_REF_NAME} is live."
perl -e '
open(my $f, q{<}, q{release-id.txt}) or die qq{release-id.txt: $!};
my $id = <$f>;
close($f);
chomp $id;
my @files = grep { -f $_ } glob(q{dist/*/*});
@files or die qq{ERROR: no assets under dist/\n};
# A file that arrived empty from the artifact step would be uploaded as an
# empty attachment, every status would still be 201, and the run would go
# green over a release nobody can install. Refuse it here, before the
# upload, and verify what was stored afterwards.
my %size;
for my $path (@files) {
my $n = -s $path // 0;
(my $name = $path) =~ s{.*/}{};
$n > 0 or die qq{ERROR: $path is empty, so there is nothing to upload\n};
$size{$name} = $n;
}
my $bad = 0;
for my $path (@files) {
(my $name = $path) =~ s{.*/}{};
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
q{-H}, q{Content-Type: application/octet-stream},
# The @ must not sit inside a qq{} string: there it starts an
# array interpolation and the upload body collapses to empty,
# which Gitea stores as a 201-created zero-byte attachment.
q{-X}, q{POST}, q{--data-binary}, q{@} . $path,
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
my $code = <$curl>;
my $ok = close($curl);
my $exit = $? >> 8;
$code = defined $code ? $code : q{};
unless ($ok) {
printf qq{%s: curl failed (exit %d)\n}, $name, $exit;
$bad = 1;
next;
}
printf qq{%s: HTTP %s\n}, $name, $code;
$bad = 1 if $code ne q{201};
}
# Read every asset back through the release download route and require the
# served length to be the file that was sent: stored but empty is a broken
# release however green the run looks.
open(my $v, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
my $v = <$v>;
close($v);
chomp $v;
for my $name (sort keys %size) {
my $url = qq{$ENV{GITEA_SERVER_URL}/$ENV{GITEA_REPOSITORY}/releases/download/v$v/$name};
my @head = (q{curl}, q{-sS}, q{-I}, q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}}, $url);
open(my $h, q{-|}, @head) or die qq{curl: $!};
my $len;
my $status;
while (my $l = <$h>) {
$status = $1 if $l =~ m{^HTTP/\S+\s+(\d+)};
$len = $1 if $l =~ m{^content-length:\s*(\d+)}i;
}
my $ok = close($h);
$len = defined $len ? $len : 0;
if (!$ok || $status != 200 || $len != $size{$name}) {
printf qq{ERROR: %s serves %s bytes, expected %d\n}, $name, $len, $size{$name};
$bad = 1;
next;
}
printf qq{%s: serves %d bytes\n}, $name, $len;
}
exit($bad ? 1 : 0);
'
+90 -76
View File
@@ -1,4 +1,17 @@
# Test — gasm-devkit. Runs on push and pull request to development.
# Test, Go. Push and pull request to development. Never on main.
#
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
# runner box cannot afford the race detector on every push, so it lives in race.yml.
# The box is one core and 2 GB beside Gitea, so parallelism is bounded on purpose and
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
# the dependency download three times without buying any parallelism.
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
# on ash, and it is one language instead of two. The Perl uses builtins only, because
# Fedora packages the Perl modules separately and nothing beyond `perl` itself may be
# assumed present.
name: Test
on:
@@ -7,90 +20,91 @@ on:
pull_request:
branches: [development]
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
# A superseded run of the same ref is cancelled instead of queueing behind one that
# no longer matters. Verified on Gitea 1.27.1 on 2026-09-17: a queued run whose ref
# moved on is cancelled before it ever reaches the runner, while a run already
# dispatched there runs to completion.
concurrency:
group: ${{ gitea.workflow }}-${{ gitea.ref }}
cancel-in-progress: true
jobs:
vet:
runs-on: fedora
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version: "1.27"
- name: Download dependencies
run: go mod download
- name: gofmt
run: |
set -euo pipefail
unformatted=$(gofmt -l .)
if [ -n "$unformatted" ]; then
echo "These files need gofmt:"
echo "$unformatted"
exit 1
fi
- name: go vet
run: go vet ./...
test:
runs-on: fedora
needs: vet
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version: "1.27"
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
cache: true
- name: Download dependencies
run: go mod download
- name: Install gcc
run: dnf install -y gcc
- name: go test -race
run: go test -race -count=1 ./...
- name: Coverage gate — 80 % minimum
run: |
set -euo pipefail
# Exclude packages inherently untestable without hardware:
# debug — interactive ptrace, requires a live process
# cmd/gasm — CLI glue, covered by integration tests
go test -coverprofile=coverage.out \
sourcedock.dev/petrbalvin/gasm-devkit/arch \
sourcedock.dev/petrbalvin/gasm-devkit/asm \
sourcedock.dev/petrbalvin/gasm-devkit/ast \
sourcedock.dev/petrbalvin/gasm-devkit/format \
sourcedock.dev/petrbalvin/gasm-devkit/lexer \
sourcedock.dev/petrbalvin/gasm-devkit/lint \
sourcedock.dev/petrbalvin/gasm-devkit/lsp \
sourcedock.dev/petrbalvin/gasm-devkit/parser \
sourcedock.dev/petrbalvin/gasm-devkit/token \
sourcedock.dev/petrbalvin/gasm-devkit/verify
coverage=$(go tool cover -func=coverage.out | awk '/^total:/ { gsub("%", "", $3); print $3 }')
echo "Total coverage: ${coverage}%"
if awk -v c="$coverage" 'BEGIN { exit !(c+0 < 80) }'; then
echo "ERROR: coverage ${coverage}% is below the 80% threshold"
exit 1
fi
build:
runs-on: fedora
needs: test
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version: "1.27"
- name: Download dependencies
run: go mod download
- name: Install Perl
# The runner images are minimal and Perl is not guaranteed. The install is a
# no-op where it is already present; drop this step once verified on the box.
run: dnf install -y perl
# The steps follow the `gates` order of the justfile contract: build, format,
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
- name: Build
run: go build -ldflags="-s -w" -o bin/gasm ./cmd/gasm
run: go build ./...
- name: Smoke test
run: ./bin/gasm --version
- name: Format
run: |
perl -e '
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
'
- name: Vet
run: go vet ./...
- name: Modernise
# Exits non-zero when it has something to rewrite, so it needs no output capture.
run: go fix -diff ./...
- name: Tests
# The suite must be fast: a push pipeline that cannot finish in a few minutes moves
# its heavy part behind a dispatch. The inner timeout matches the job's, so a
# hanging test reports its own goroutine dump rather than a silent job kill.
# The pattern is `packages` in the project's justfile: the logic packages, since a
# thin cmd/ would drag the total under the floor. release.yml runs the same
# command, so the floor is the same number everywhere. ./verify/... carries the
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth
# suites); the runner's Go setup provides both the tool and GOROOT.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The CLI and the debugger sit outside `packages` because a thin main and a
# ptrace-bound package pull the total under the floor, but their tests guard
# shipped surfaces: the command exit codes, the manual pages against the
# binary's own help, and the debugger's architecture-neutral units. They run
# here so the floor stays a product measure and nothing is left untested.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Oracle parity
# Re-run the live go-tool-asm comparison as its own step so that a parity
# regression names the gate that failed instead of hiding inside the suite.
run: go test -count=1 -timeout 10m -run 'TestGroundTruth' ./verify/...
- name: Coverage floor
run: |
perl -e '
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
'
+3 -14
View File
@@ -1,24 +1,13 @@
# Metadata (always first, per repo convention)
.idea/
.zcode/
.mimocode/
# Binaries
/gasm
# Build output
/bin/
*.exe
# Test and coverage artefacts
/gasm
coverage.out
*.test
# Crash dumps
# Crash dumps from the emulator runs
core
core.*
*.core
# Scratch / temporary work
_scratch/
# ZCode workspace
.zcode
+650 -147
View File
File diff suppressed because it is too large Load Diff
+105 -78
View File
@@ -1,107 +1,134 @@
# Contributing to gasm-devkit
# Contributing
Thanks for contributing to gasm-devkit.
Contributions to **gasm-devkit** are governed by the Contributor terms
below; submitting one means you accept them.
## Contributor terms
1. This project belongs to its owner alone. The owner decides what is
accepted, in what form and when; the decision is final and needs no
justification.
2. By submitting a contribution you assign to Petr Balvín
<opensource@petrbalvin.org> all present and future copyright and
related rights in it, worldwide, for the full term of the rights,
with the right to relicense and sublicense without restriction,
including under proprietary terms.
3. Where that assignment is not effective, it counts as a perpetual,
irrevocable, royalty-free licence with the same scope.
4. To the fullest extent permitted by law, you waive any right of
attribution and integrity in the contribution. The project names no
contributors and keeps no credits list.
5. By submitting you represent that the work is yours and that you
hold the rights to assign it as above.
## Development setup
Requirements: Go 1.27 or later, the [just](https://github.com/casey/just)
command runner, and a Linux host on amd64, arm64, riscv64 or loong64.
Requirements: Go 1.27.1, the exact version the `go` directive in `go.mod`
declares, [just](https://github.com/casey/just) for the recipes, and a C
compiler (gcc), because `just gates` includes `just race` and the race
detector needs cgo.
```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
cd gasm-devkit
just install # download module dependencies
just build # go vet + gofmt check
just test # full suite, race detector, 80 % coverage gate
just build
just gates
```
## Workflow
1. Branch from `development`; never commit directly to `main` (`main` is
release-only: merge from `development`, then tag).
2. Commit with [Conventional Commits](https://www.conventionalcommits.org/):
`type(scope): description`: subject line only, imperative mood,
lowercase after the colon, no trailing dot. Allowed types: `feat`,
`fix`, `docs`, `style`, `refactor`, `perf`, `test`, `chore`, `ci`,
`build`, `revert`. The only line after the subject is the trailer:
`Assisted-by: <model-name>`. No `Co-Authored-By`, no `Signed-off-by`,
no other trailers.
3. Record every user-visible change in `CHANGELOG.md` under
`## [development]` (categories: Added, Changed, Fixed, Removed,
Security).
4. Add or update tests; coverage must stay **at or above 80 %** (hard
gate, enforced by CI).
5. Update the documentation when behaviour, flags or the public surface
change.
6. Open a pull request against `development`.
1. Branch from `development`. Never commit directly to `main`, which is release-only.
2. Commit in [Conventional Commits](https://www.conventionalcommits.org/) form:
`type(scope): description`, subject line only, imperative mood, lowercase after the
colon, no trailing full stop. Allowed types: `feat`, `fix`, `docs`, `style`,
`refactor`, `perf`, `test`, `chore`, `ci`, `build`, `revert`.
3. One logical change per commit. A refactor, a behaviour change and a formatting pass
are three commits, never one.
4. Record every user-visible change in `CHANGELOG.md` under `## [development]`.
5. Add or update tests. Coverage stays at 80 percent or more; it is a hard gate.
6. Update the documentation when the public API, the configuration or the behaviour
changes.
7. Open a pull request against `development`.
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`;
CI builds and publishes the binaries for all four architectures.
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The release
workflow builds the assets and publishes the release and its notes.
## Code style
`gofmt` and `go vet` via `just fmt` / `just build`; both must pass with
zero output; `go fix -diff ./...` must report nothing on touched packages.
`gofmt` and `go vet` run through `just fmt` and `just vet`, with zero diff and zero
warnings tolerated. `just vet` is two gates, `go vet ./...` and `go fix -diff ./...`,
so the modernisation rewrites are enforced too. `just gates` is the definition of done in
one command, and the recipe file names what it contains. Errors are checked explicitly,
wrapped as `fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. The
recipe file holds the commands, and the language and standard-library surface is the one
the `go` directive in `go.mod` pins.
- Standard library only in production code; `golang.org/x/arch` is used
in tests only (round-trip decoding) and is never linked into the `gasm`
binary.
- No cgo, no C, no external toolchains at runtime.
- Explicit `if err != nil`; errors wrapped with
`fmt.Errorf("context: %w", err)`; no panics outside `main`.
- The parser, lexer and formatter are hand-written; the `arch` instruction
tables are generated only via `_gen/gen.go` (`just gen`), never edited.
- `golang.org/x/arch` is the one module dependency, and it is linked into the binary:
`gasm dis` and the debugger's listings decode through it. Everything else is the
standard library.
- No cgo and no C. The standalone encoder paths (`gasm asm --format raw` and `--format
elf`) need no Go installation; `gasm verify --ground-truth`, `gasm verify --fuzz`,
`gasm audit-instructions` and `gasm asm --format goobj` resolve through the installed
Go toolchain.
- The parser, lexer and formatter are hand-written; the `arch` instruction tables are
generated only by `_gen/gen.go` (`just gen`) and never edited by hand.
- Assembly committed to the repository goes through `gasm fmt` and `gasm lint`, so a
`.s` file that `gasm fmt -l .` lists is unfinished.
## Running a single test
New source files open with the project's two-line licence header, whose SPDX
identifier matches `LICENSE`. Configuration files, workflows and dotfiles do not carry
it.
```sh
go test -run TestVexGroundTruth ./asm/
go test -run TestGroundTruthBasic ./verify/
go test -run TestGOObjectLinkAndRun ./asm/
go test -run TestFuzzWideCopy ./verify/
```
## AI contribution policy
The interactive debugger (`gasm debug`) requires a compiled binary on
`$PATH`; `go run` does not work for the traced child process. Install
first with `just install-bin`.
AI tools are welcome as productivity aids and are a normal part of modern software
development. What matters is that the contribution stays understandable, reviewable and
genuinely useful.
## CI (Gitea Actions)
- **Disclose the assistance.** If AI helped draft any part of a commit, issue, pull
request or review, say so.
- **Commit messages carry exactly one trailer**, as a git trailer on the line after a
blank line that closes the subject:
Workflows live in `.gitea/workflows/` and run on self-hosted runners:
```
Assisted-by: MODEL
```
Name the model that did the work, spelled the way its maker spells it, for example
`GLM 5.3`, `DeepSeek V4.1 Flash` or `Qwen 3.8 Flash`. No `Co-Authored-By`, no `Signed-off-by`,
no other trailers, and no prose: the trailer is the disclosure.
- **Issues and pull requests** attribute the assistance in a comment, for example
`_Assisted-by: GLM 5.3_`. It does not belong in the pull request description.
- **Take responsibility.** You are accountable for the accuracy, completeness and
intent of everything you submit, whether or not AI produced it.
- **Review before marking ready.** Read the diff carefully, run it locally, and add the
tests it needs. Do not mark a pull request ready until you can defend every change in
it.
- **Quality over quantity.** Contributions that look like un-reviewed output, or whose
author cannot engage substantively during review, may be closed.
- **Preferred models.** Prefer open-weight models with transparent training data and
minimal output filtering.
AI assists. It does not replace judgement.
## Continuous integration
Workflows live in `.gitea/workflows/` and run on the project's own runners:
| Workflow | Trigger | What it does |
|----------|---------|--------------|
| Test | push / PR to `development` | gofmt check, `go vet`, `go test -race`, 80 % coverage gate |
| Release | tag `v*` | cross-compiles binaries for linux/{amd64,arm64,riscv64,loong64} and publishes the Gitea release |
|---|---|---|
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor, the CLI and debugger tests outside the profile, then the oracle-parity rerun against `go tool asm` |
| Release | a `v*` tag | the same gates as Test minus the oracle-parity step, then the matrix build, the version smoke test and the release itself; the race detector runs locally in `just gates` before the tag is cut |
The Definition of Done (`just build` + `just test` + `just fmt`) must
still pass locally before pushing.
## AI Contribution Policy
AI tools are welcome as productivity aids. What matters is that
contributions remain understandable, reviewable, and genuinely useful.
- **Disclose AI use.** If you used AI to draft or generate any part of a
commit, issue, pull request, or code review, say so clearly.
- **Commit messages:** end every commit with exactly one trailer:
`Assisted-by: <model-name>` (e.g. `Assisted-by: GLM 5.3`).
- **Pull requests and issues:** attribute AI assistance in one trailing
line, e.g. `_Assisted-by: GLM 5.3_`. Do not paste it into the PR
description as a section.
- **Take responsibility.** You remain accountable for the accuracy,
completeness, and intent of everything you submit.
- **Review before marking ready.** Read AI-generated diffs carefully, run
them locally, and add or update tests where appropriate.
- **Preferred models.** Prefer open-weight models with transparent
training data: **GLM**, **DeepSeek**, and **MiMo**.
The local equivalent is `just gates`, which is the same set plus the race detector. The
race detector also has its own workflow, dispatched by hand; it never runs on a push or a
tag, where it would double the time and the memory a shared runner cannot spare.
## Reporting bugs
Open an issue at
[sourcedock.dev/petrbalvin/gasm-devkit](https://sourcedock.dev/petrbalvin/gasm-devkit/issues)
with the version (`gasm --version`), OS and architecture, the exact
command, the full output, and the expected versus actual behaviour.
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the
version, the operating system and architecture, the exact command, the full output,
and the expected against the actual behaviour.
**Security issues:** email **opensource@petrbalvin.org** instead of opening
a public issue.
**Security issues do not go in the issue tracker.** Report them as
[SECURITY.md](SECURITY.md) describes, to **opensource@petrbalvin.org**.
+208 -36
View File
@@ -1,13 +1,65 @@
# gasm-devkit
# Plan 9 assembly tooling, inside and outside Go
Developer tooling for **GAsm**, Go's built-in Plan 9 assembler.
> **Warning: this is an experiment.** gasm-devkit is under active
> development and is not stable. The version is 0.x.x: commands, flags,
> output formats and behaviour can change without warning at any time.
> A 1.0.0 release is light years away. Nothing in this document is a
> stability promise. For all of that, this is not a paper project: gasm
> is already in active use and is tested on real assembly work. Only
> amd64 is validated on real hardware; the other three architectures run
> under emulation ([Validation status](#validation-status)).
Go ships an assembler but no tooling for it: there is no syntax highlighting,
no autocomplete, no linter, no static analyser, no formatter, no standalone
assembler and no debugger for `.s` files. Developers write assembly blind,
validate it by benchmark, and debug it by print statement. gasm-devkit is the
missing toolkit: a single, self-contained binary, `gasm`, that brings proper
developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
**GAsm** is Go's Plan 9 assembler, and Go ships it without tooling:
there is no formatter, no linter and no debugger for `.s` files, and no
assembler that works without a Go installation. Developers write
assembly blind, validate it by benchmark, and debug it by print
statement. gasm-devkit is the missing toolkit: a single, self-contained
binary, `gasm`, that serves both purposes.
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
dynamic verification, a source-level debugger and a language server,
for `.s` files in Go programs.
- **Use Plan 9 assembly outside the Go toolchain.** `gasm asm` encodes
on its own and writes raw images or linkable ELF objects with DWARF5
debug sections, with no Go installation in the loop; the Go
toolchain's own GOOBJ format, which `go build` consumes in place of
the toolchain's output, needs the installed toolchain.
## Why Plan 9 assembly
Plan 9 assembly is the quiet triumph of the field. One syntax across
every architecture Go builds for: the same source-first operand order,
the same four pseudo-registers, the same frame convention, whether the
target is x86, ARM, RISC-V or LoongArch. Learn it once and you can
read a kernel on any of them.
Compare the alternatives. Intel syntax and AT&T syntax disagree on the
one question every instruction answers, which operand is the source
and which is the destination, so half the world writes it one way,
half the other, and every assembly programmer carries both in their
head forever. GNU as settles the argument with directives that switch
dialects mid-file (`.intel_syntax noprefix`), a percent sign on every
register and a dollar on every immediate: punctuation that carries
nothing the operand order did not already say. And the x86 family
fragments again underneath: NASM is not MASM is not GAS, each with its
own directive zoo and macro language, so every project picks a dialect
and every reader learns a different one by accident.
Plan 9 assembly has none of it. Registers are bare names. Memory is
one notation, `offset(base)`, extended by an index and a scale when
the instruction needs it. Arguments arrive named and offset-checked:
`x+0(FP)` is the argument x, on every architecture, and `go vet`
polices the offsets against the Go prototype.
```text
AT&T (GNU as): movq %rax, -16(%rbp)
Plan 9 (Go): MOVQ AX, total-16(SP)
```
The same lines, but only one of them tells you what the number is for.
The syntax is uppercase, regular and boring, which is the highest
compliment a language for machine code can earn. gasm-devkit exists
to give that syntax the tooling it deserves.
## Features
@@ -16,68 +68,179 @@ developer tooling to Plan 9 assembly on amd64, arm64, riscv64 and loong64.
directly.
- **Formatter.** `gasm fmt` canonicalises indentation, operand spacing,
per-function mnemonic alignment and blank-line layout: `gofmt` for assembly,
operating recursively on directories the way `go fmt` does.
operating recursively on directories the way `go fmt` does. `-l` lists
files whose formatting differs and `-d` prints a unified diff.
- **Linter.** `gasm lint` runs 18 conservative static checks, among them
`undefined-label`, `abi-argsize` (declared frame vs the `// func` signature),
`register-clobber` (Go ABI register liveness over the control-flow graph),
`stack-imbalance`, `abi0-register-args` and `unencodable-instruction`.
`undefined-label`, `abi-argsize` (declared argument area vs the `// func`
signature), `register-clobber` (Go ABI register liveness over the
control-flow graph), `stack-imbalance`, `abi0-register-args` and
`unencodable-instruction`.
- **Standalone assembler.** `gasm asm` encodes all four architectures without
the Go toolchain and writes raw images, linkable ELF objects (with DWARF5
debug sections) or the Go toolchain's own GOOBJ format, which `go build`
consumes in place of the toolchain's output.
the Go toolchain and writes raw images or linkable ELF objects (with DWARF5
debug sections) with no Go installation needed, or the Go toolchain's own
GOOBJ format, which needs the installed toolchain and which `go build`
consumes in place of the toolchain's output. Framed functions get the
stack-split guard and the morestack block, byte-identical to the
toolchain's, so split functions link too. The assembler preprocesses
like the toolchain (`#define`, `#include` with `-I`, `#ifdef`), generates
`go_asm.h` from the package's Go files, and carries `PCALIGN`, the
`LOCK`/`REP` prefixes and the literal-data pseudo-ops.
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
offsets after assembling, or disassembles raw bytes from a file or stdin.
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
executable memory: smoke calls, ABI checks (sentinel registers, red-zone
canary), differential fuzzing against the `go tool asm` build, and
byte-for-byte ground-truth comparison of the machine code.
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
breakpoints (optionally conditional), hardware watchpoints, register and
memory inspection, and headless script runs with label-level coverage.
memory inspection, and headless script runs that report instruction and
label coverage.
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature
help, document highlights, workspace symbol search, #include document
links and folding ranges over stdio.
links and folding ranges over stdio; definition, references and rename
work across every open document.
- **Comparators and audits.** `gasm diff` compares the machine code of two
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
`gasm audit-instructions` diffs the encoder against the installed toolchain,
and `gasm scaffold` generates a differential test skeleton for a kernel.
- **Complete instruction coverage.** The instruction tables are generated
from the Go toolchain's own assembler source, so the toolkit recognises
every mnemonic the real assembler accepts; `just gen` refreshes them.
### Architecture support
Four architectures, the four that matter in practice:
| Architecture | GOARCH | File suffix | Instructions recognised |
|--------------|-------------|--------------|---------------------------------------------|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
| ARM64 | `arm64` | `_arm64.s` | 645 + common opcodes |
| RISC-V | `riscv64` | `_riscv64.s` | 992 + common opcodes |
| LoongArch | `loong64` | `_loong64.s` | 808 + common opcodes |
"Common opcodes" are the instructions shared by every architecture (`RET`,
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
carries the traditional conditional-jump spellings (`JZ`, `JNZ`, `JA`, `JC`,
...) that the assembler accepts as aliases. Regenerating the tables is one
command (`just gen`) and requires only a Go installation; the committed output
has no runtime dependency on the toolchain.
...) that the assembler accepts as aliases. The tables are generated from
the Go toolchain's own assembler source (`just gen` refreshes them), so
every mnemonic the real assembler accepts is recognised; what the encoder
can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 291 of 353 attemptable files
(82.4 %) assembling for every target architecture today (files named for
other Go ports are counted but never attempted), with the top failure
reasons per architecture; the number moves with every release.
### Validation status
**Only amd64 is validated on real hardware.** The other three
architectures are validated under qemu-user emulation, because the
project owns no arm64, riscv64 or loong64 machine, and emulation is the
only substitute available for the hardware. The distinction matters and
is stated rather than implied: everything below is a claim about what has
actually been executed.
| Layer | amd64 | arm64, riscv64, loong64 |
|---|---|---|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more |
Consequences, stated plainly. An emulator is a model of a CPU, not the
CPU: instruction semantics are implemented in software and can differ
from silicon in ways a test suite does not reveal. A kernel that passes
under qemu-user is therefore not proven correct on real hardware, and a
discrepancy found on real hardware is a defect in gasm, reported like any
other. Encoding parity is the exception: the byte comparison against the
toolchain runs on the host for every architecture, so no emulator stands
between the claim and the evidence. The debugger is the weakest case: on
the three emulated architectures its per-architecture ptrace code has
been compiled and read, never executed. Its architecture-neutral units
run under `go test ./...`, which the race workflow and a manual run
perform; the default `just test` gate does not sweep `./debug/...`.
## The documentation goal
The toolkit is the primary goal. The secondary one is documentation: a
specification of the Plan 9 assembly language and of the GOOBJ object
format that is 100 % complete, detailed enough to implement against,
and written to a professional standard. These are the two subjects this
project works with every day, and they are the two for which no usable
documentation exists.
Go documents the language on a single page, "A Quick Guide to Go's
Assembler", which carries no section for loong64, one of the four
architectures gasm supports, and covers a fraction of what each
assembler accepts. What exists beyond it lives as comments inside the
toolchain's internal source: per-architecture reference manuals for
arm64, ppc64, riscv64 and loong64, written for the toolchain's own
maintainers rather than for an outside reader, and none at all for
amd64. GOOBJ fares worst of all. The format that `go build` consumes
has no specification anywhere: it is described by a comment in an
internal package, it is not a stable interface, and it can change with
any toolchain release.
The gap is therefore filled the only way it can be filled: by reverse
engineering the toolchain itself, the same work the encoders already
perform. Most of the documentation can come from nowhere else, and it
is written as that knowledge is produced during development. It is
verified the way the code is verified: an encoding documented here is
one that differential tests against `go tool asm` confirm
byte-for-byte, and a format field documented here is one the linker
demonstrably reads. The work has begun: [docs/GOOBJ.md](docs/GOOBJ.md)
specifies the object file format completely, and
[docs/asm/README.md](docs/asm/README.md) opens the language reference
with its common core. The per-architecture pages follow.
## Direction
The plan, in the order it is being worked:
- **Extended instruction support.** Two layers. First, encoding
coverage for every mnemonic the Go toolchain itself accepts, closed in
order of how often real code needs each instruction;
`gasm audit-instructions` measures the gap. Second, the larger work:
an extended instruction set the toolchain does not know at all. The
toolchain-derived tables stay generated and untouched; only the
extended instructions are hand-maintained, with their own spellings
and encoders, verified by execution (on real hardware for amd64, under
emulation for the rest, per the validation status above) because the
toolchain offers no ground truth to compare against. The gaps exist
on every architecture, amd64 included.
- **Full GOOBJ and ELF compilation.** The destination is a complete,
standalone compilation path: linkable ELF objects for consumers outside
Go, and GOOBJ objects that `go build` links directly. Through GOOBJ, a
Go program will be able to use machine instructions that the Go
toolchain itself does not support; through ELF, Plan 9 assembly becomes
usable outside Go entirely.
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
architectures and is where the binary builds. FreeBSD follows: the
JIT's executable-memory mapping and the ptrace debugger layer are the
two pieces of porting work. Other unix systems may follow those two.
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
No others are planned.
## Install
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
linux/loong64 are on the
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
From source (Go 1.27 or later):
From source (Go 1.27.1):
```sh
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
```
Or from a repository checkout, with the development version stamped:
Or from a repository checkout:
```sh
just install-bin
just install
```
The installed binary reports the version the toolchain recorded: the tag
on a tagged checkout, a pseudo-version naming the commit below one.
## Quick start
```sh
@@ -102,14 +265,18 @@ gasm verify --call add --args a=2,b=3 hello_amd64.s # JIT-call it with argumen
```sh
gasm fmt # reformat every .s below here, like go fmt
gasm fmt -w kernel_amd64.s # canonicalise one file in place
gasm fmt -l *.s # list files whose formatting differs
gasm fmt -d kernel_amd64.s # print a unified diff instead
gasm lint *.s # static checks
gasm asm --format elf -o k.o k.s # assemble to a linkable ELF object
gasm asm --format goobj -p pkg/path -o k.o k.s # Go object, consumed by go build
gasm dis k.s # assemble, then list each function
gasm dis -a amd64 - < dump.bin # disassemble raw bytes from stdin
gasm verify --ground-truth k.s # byte-for-byte vs go tool asm
gasm verify --fuzz k.s # differential fuzz vs the go tool asm build
gasm debug --func name k.s # interactive debugger
gasm debug --func name --script cmds.txt --timeout 30s k.s # headless run
gasm debug --func name --cover k.s # which labels did execution reach?
gasm debug --func name --cover k.s # instruction and label coverage
gasm diff a.s b.s # compare machine code byte-for-byte
gasm diff --map wideCopyAVX2=wideCopyAVX512 avx2.s avx512.s
gasm profile k.s # show basic-block structure
@@ -132,9 +299,9 @@ infers the target architecture from the file-name suffix
## Development
```sh
just install # download module dependencies
just build # go vet + gofmt check, zero errors and zero warnings
just test # full suite, race detector, 80 % coverage gate
just build # compile, zero errors and zero warnings
just test # the suite, no cache, the 80 % coverage floor
just gates # build, fmt-check, vet, test, race: the definition of done
just fmt # gofmt the tree
just gen # regenerate the instruction tables from the Go toolchain
```
@@ -145,14 +312,19 @@ recipe.
## Documentation
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/CLI.md](docs/CLI.md): full command reference
- man pages: `just install-man` installs gasm(1) and one page per command
except `version`, which is documented inside gasm(1) instead, into
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
them
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/GOOBJ.md](docs/GOOBJ.md): the GOOBJ object file format specification
- [docs/asm/](docs/asm/README.md): the Plan 9 assembly language reference
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
- [docs/DECISIONS.md](docs/DECISIONS.md): deferred design decisions
- [CHANGELOG.md](CHANGELOG.md): release history
## Licence
BSD-3-Clause — see [LICENSE](LICENSE).
BSD-3-Clause; see [LICENSE](LICENSE).
Copyright © 2026 [Petr Balvín](https://petrbalvin.org)
+41
View File
@@ -0,0 +1,41 @@
# Security policy
## Supported versions
Security fixes go to the newest release and to the `development` branch. Older
releases do not receive them.
| Version | Supported |
|---|---|
| 0.35.0 | yes |
| older releases | no |
## Reporting a vulnerability
**Do not open a public issue for a security problem.** A public report tells everyone
about the flaw before there is a fix. Report it privately to
**opensource@petrbalvin.org**.
Include:
- the version or commit you tested, and the platform
- what the problem is, and what an attacker gains from it
- the smallest reproducer you have, ideally a test or a single command
- a suggested fix, if you have one
## What to expect
- A human reads the report, and you get an acknowledgement.
- You are kept informed while the fix is being made, and told when it ships.
- The fix is released before the details are published, and the timing is agreed with
you.
- The fix ships without naming you: the project keeps no credits list, so the release
notes, the changelog and the commits name no reporter.
## Out of scope
- Findings that require the attacker to already run code as the user, or to have local
access.
- Missing hardening with no demonstrated impact.
- Flaws in a third-party dependency: report them to that project, and to this one only
when this project's use of it makes them reachable.
+101 -3
View File
@@ -8,6 +8,10 @@
// names so gasm-devkit supports every instruction the real assembler does,
// with no hand-maintained (and therefore inevitably incomplete) lists.
//
// The same data feeds the generated instruction appendices of the assembly
// language reference, docs/asm/INSTRUCTIONS-<ARCH>.md, so that the reference
// cannot drift from the tables it documents.
//
// Usage (via the justfile):
//
// just gen
@@ -26,6 +30,9 @@ import (
"path/filepath"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
)
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
@@ -39,11 +46,30 @@ var archDirs = []struct {
{"loong64", "loong64"},
}
// docPages maps an architecture to its generated appendix in the language
// reference. The amd64 page carries a per-mnemonic encodability column,
// decided by asm.Encodable, which mirrors the encoder's own dispatch; the
// other targets have no single cheap predicate, so their pages carry the
// inventory and point at the live measurement instead.
var docPages = []struct {
arch arch.Arch
title string
file string
anames string
encodable bool
}{
{arch.AMD64, "AMD64", "INSTRUCTIONS-AMD64.md", "cmd/internal/obj/x86/anames.go", true},
{arch.ARM64, "ARM64", "INSTRUCTIONS-ARM64.md", "cmd/internal/obj/arm64/anames.go", false},
{arch.RISCV, "RISC-V 64", "INSTRUCTIONS-RISCV64.md", "cmd/internal/obj/riscv/anames.go", false},
{arch.LOONG64, "LoongArch 64", "INSTRUCTIONS-LOONG64.md", "cmd/internal/obj/loong64/anames.go", false},
}
func main() {
goroot := strings.TrimSpace(runGoEnvGOROOT())
if goroot == "" {
fatal("could not determine GOROOT")
}
version := strings.TrimSpace(runGoEnv("GOVERSION"))
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
@@ -57,16 +83,24 @@ func main() {
}
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
names := map[string][]string{}
for _, a := range archDirs {
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
names, err := extractInstrs(path)
names[a.arch], err = extractInstrs(path)
if err != nil {
fatal("extract %s: %v", a.arch, err)
}
if err := writeGen(a.arch, a.sub, names); err != nil {
if err := writeGen(a.arch, a.sub, names[a.arch]); err != nil {
fatal("write %s: %v", a.arch, err)
}
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch)
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names[a.arch]), a.arch)
}
for _, p := range docPages {
if err := writeDocPage(p.arch, p.title, p.file, p.anames, version, p.encodable); err != nil {
fatal("write %s: %v", p.file, err)
}
fmt.Printf("%-8s -> docs/asm/%s\n", p.arch, p.file)
}
}
@@ -172,6 +206,61 @@ func writeGen(arch, sub string, names []string) error {
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
}
// writeDocPage emits docs/asm/<file>, the generated instruction appendix of
// the language reference for one architecture: every mnemonic the toolchain
// accepts, with the curated summary where the architecture table carries one
// and, on amd64, a per-mnemonic encodability column.
func writeDocPage(a arch.Arch, title, file, anames, version string, encodable bool) error {
table := arch.ForArch(a)
instrs := table.Instructions()
var b strings.Builder
b.WriteString("# " + title + ": instruction inventory\n\n")
b.WriteString("Generated by gasm-devkit's `_gen` from the Go toolchain's instruction table\n")
b.WriteString("(`" + anames + "`, " + version + "); DO NOT EDIT. This page lists every mnemonic\n")
b.WriteString("`go tool asm` accepts on this target, which is the upper bound of the\n")
b.WriteString("language on it: a name absent here is not an instruction of the target,\n")
b.WriteString("and a name present here may still be one gasm's encoder cannot emit yet.\n\n")
encodableCount := 0
if encodable {
b.WriteString("The `gasm encodes` column reports whether gasm's encoder can emit the\n")
b.WriteString("mnemonic today; the gap is the encoder backlog, measured live by\n")
b.WriteString("`gasm audit-instructions`.\n\n")
b.WriteString("| Mnemonic | gasm encodes | Notes |\n")
b.WriteString("|---|---|---|\n")
for _, in := range instrs {
ok := asm.Encodable(in.Name)
if ok {
encodableCount++
}
b.WriteString("| `" + in.Name + "` | " + yesNo(ok) + " | " + in.Summary + " |\n")
}
b.WriteString("\n")
fmt.Fprintf(&b, "Recognised: %d mnemonics. gasm encodes: %d.\n", len(instrs), encodableCount)
} else {
b.WriteString("The inventory carries no per-mnemonic encoder column: on this target\n")
b.WriteString("encodability is decided per operand shape, and the live measured\n")
b.WriteString("coverage is reported by `gasm audit-instructions`.\n\n")
b.WriteString("| Mnemonic | Notes |\n")
b.WriteString("|---|---|\n")
for _, in := range instrs {
b.WriteString("| `" + in.Name + "` | " + in.Summary + " |\n")
}
b.WriteString("\n")
fmt.Fprintf(&b, "Recognised: %d mnemonics.\n", len(instrs))
}
return os.WriteFile(filepath.Join("docs", "asm", file), []byte(b.String()), 0o644)
}
// yesNo renders a boolean as the word the appendix tables use.
func yesNo(v bool) string {
if v {
return "yes"
}
return "no"
}
func runGoEnvGOROOT() string {
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
@@ -180,6 +269,15 @@ func runGoEnvGOROOT() string {
return string(out)
}
// runGoEnv runs `go env` for a single variable.
func runGoEnv(name string) string {
out, err := exec.Command("go", "env", name).Output()
if err != nil {
return ""
}
return string(out)
}
func fatal(format string, args ...any) {
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
os.Exit(1)
+4
View File
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
}
// x87 stack registers (FMOVD and the other x87 moves).
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
}
return regs
}
+3
View File
@@ -55,6 +55,7 @@ const (
Mask // AVX-512 mask register (K)
Float // arm64 floating-point register (F)
VecARM // arm64 SIMD/vector register (V)
VecSIMD // architecture-neutral SIMD/vector register (LoongArch LSX/LASX)
Special // architecture-special register
)
@@ -73,6 +74,8 @@ func (c RegClass) String() string {
return "float"
case VecARM:
return "vector (arm64)"
case VecSIMD:
return "vector"
case Special:
return "special"
default:
+30 -2
View File
@@ -29,10 +29,11 @@ func arm64Registers() []Register {
regs = append(regs, Register{Name: name, Class: class, Desc: desc})
}
// General-purpose integer registers R0–R30.
// General-purpose integer registers R0-R30.
for i := 0; i <= 30; i++ {
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
}
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
add("ZR", Special, "zero register (reads as 0)")
add("SP", Special, "stack pointer")
add("LR", Special, "link register (alias of R30)")
@@ -145,11 +146,38 @@ func arm64Curated() []Instr {
for _, op := range []string{
"LDAXR", "LDAXRB", "LDAXRH", "LDAXRW", "STXR", "STXRB", "STXRH", "STXRW",
"LDAR", "LDARB", "LDARH", "LDARW", "STLR", "STLRB", "STLRH", "STLRW",
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL", "CASAL",
"LDADD", "LDCLR", "LDEOR", "LDSET", "SWP", "CAS", "CASAL", "CASL",
} {
t = append(t, i(op, "Atomic memory operation"))
}
// Register-pair loads and stores.
for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} {
t = append(t, ic(op, "Register-pair load or store", 2, 2))
}
// Cache maintenance and prefetch.
t = append(t, i("DC", "Data cache maintenance"))
t = append(t, i("PRFM", "Memory prefetch"))
for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} {
t = append(t, i(op, "Atomic memory operation with acquire and release semantics"))
}
// Cryptographic extensions.
for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} {
t = append(t, i(op, "AES round"))
}
for _, op := range []string{
"SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1",
"SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1",
"SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1",
} {
t = append(t, i(op, "SHA round"))
}
for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} {
t = append(t, i(op, "Three-way XOR / rotate crypto vector operation"))
}
// Floating-point scalar.
for _, op := range []string{
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
+107
View File
@@ -364,6 +364,8 @@ var arm64GeneratedInstrs = []string{
"REVW",
"ROR",
"RORW",
"RPRFM",
"SB",
"SBC",
"SBCS",
"SBCSW",
@@ -477,23 +479,68 @@ var arm64GeneratedInstrs = []string{
"UXTH",
"UXTHW",
"UXTW",
"VABS",
"VADD",
"VADDP",
"VADDV",
"VAND",
"VBCAX",
"VBIC",
"VBIF",
"VBIT",
"VBSL",
"VCLS",
"VCLZ",
"VCMEQ",
"VCMGE",
"VCMGT",
"VCMHI",
"VCMHS",
"VCMLE",
"VCMLT",
"VCMTST",
"VCNT",
"VDUP",
"VEOR",
"VEOR3",
"VEXT",
"VFABS",
"VFADD",
"VFADDP",
"VFCMEQ",
"VFCMGE",
"VFCMGT",
"VFCMLE",
"VFCMLT",
"VFCVTL",
"VFCVTL2",
"VFCVTN",
"VFCVTN2",
"VFCVTZS",
"VFCVTZU",
"VFDIV",
"VFMAX",
"VFMAXNM",
"VFMAXNMP",
"VFMAXNMV",
"VFMAXP",
"VFMAXV",
"VFMIN",
"VFMINNM",
"VFMINNMP",
"VFMINNMV",
"VFMINP",
"VFMINV",
"VFMLA",
"VFMLS",
"VFMUL",
"VFNEG",
"VFRINTM",
"VFRINTN",
"VFRINTP",
"VFRINTZ",
"VFSQRT",
"VFSUB",
"VLD1",
"VLD1R",
"VLD2",
@@ -502,11 +549,17 @@ var arm64GeneratedInstrs = []string{
"VLD3R",
"VLD4",
"VLD4R",
"VMLA",
"VMLS",
"VMOV",
"VMOVD",
"VMOVI",
"VMOVQ",
"VMOVS",
"VMUL",
"VNEG",
"VNOT",
"VORN",
"VORR",
"VPMULL",
"VPMULL2",
@@ -515,14 +568,47 @@ var arm64GeneratedInstrs = []string{
"VREV16",
"VREV32",
"VREV64",
"VSCVTF",
"VSHADD",
"VSHL",
"VSHRN",
"VSHRN2",
"VSLI",
"VSMAX",
"VSMAXP",
"VSMAXV",
"VSMIN",
"VSMINP",
"VSMINV",
"VSMLAL",
"VSMLAL2",
"VSMLSL",
"VSMLSL2",
"VSMULL",
"VSMULL2",
"VSQABS",
"VSQADD",
"VSQNEG",
"VSQSHL",
"VSQSUB",
"VSQXTN",
"VSQXTN2",
"VSQXTUN",
"VSQXTUN2",
"VSRHADD",
"VSRI",
"VSRSHR",
"VSSHL",
"VSSHLL",
"VSSHLL2",
"VSSHR",
"VST1",
"VST2",
"VST3",
"VST4",
"VSUB",
"VSXTL",
"VSXTL2",
"VTBL",
"VTBX",
"VTRN1",
@@ -530,8 +616,27 @@ var arm64GeneratedInstrs = []string{
"VUADDLV",
"VUADDW",
"VUADDW2",
"VUCVTF",
"VUHADD",
"VUMAX",
"VUMAXP",
"VUMAXV",
"VUMIN",
"VUMINP",
"VUMINV",
"VUMLAL",
"VUMLAL2",
"VUMLSL",
"VUMLSL2",
"VUMULL",
"VUMULL2",
"VUQADD",
"VUQSHL",
"VUQSUB",
"VUQXTN",
"VUQXTN2",
"VURHADD",
"VUSHL",
"VUSHLL",
"VUSHLL2",
"VUSHR",
@@ -541,6 +646,8 @@ var arm64GeneratedInstrs = []string{
"VUZP1",
"VUZP2",
"VXAR",
"VXTN",
"VXTN2",
"VZIP1",
"VZIP2",
"WFE",
+2 -2
View File
@@ -31,10 +31,10 @@ func loong64Registers() []Register {
add(fmt.Sprintf("F%d", i), Float, "floating-point register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("V%d", i), VecARM, "LSX 128-bit vector register")
add(fmt.Sprintf("V%d", i), VecSIMD, "LSX 128-bit vector register")
}
for i := 0; i <= 31; i++ {
add(fmt.Sprintf("X%d", i), VecARM, "LASX 256-bit vector register")
add(fmt.Sprintf("X%d", i), VecSIMD, "LASX 256-bit vector register")
}
return regs
}
+9
View File
@@ -152,6 +152,8 @@ var loong64GeneratedInstrs = []string{
"FNMADDF",
"FNMSUBD",
"FNMSUBF",
"FRINTD",
"FRINTF",
"FSCALEBD",
"FSCALEBF",
"FSEL",
@@ -177,7 +179,10 @@ var loong64GeneratedInstrs = []string{
"FTINTWF",
"JIRL",
"LL",
"LLACQV",
"LLACQW",
"LLV",
"LLW",
"LU12IW",
"LU32ID",
"LU52ID",
@@ -248,7 +253,11 @@ var loong64GeneratedInstrs = []string{
"ROTR",
"ROTRV",
"SC",
"SCQ",
"SCRELV",
"SCRELW",
"SCV",
"SCW",
"SGT",
"SGTU",
"SLL",
+31
View File
@@ -81,6 +81,9 @@ var riscvGeneratedInstrs = []string{
"CLD",
"CLDSP",
"CLI",
"CLMUL",
"CLMULH",
"CLMULR",
"CLUI",
"CLW",
"CLWSP",
@@ -95,13 +98,20 @@ var riscvGeneratedInstrs = []string{
"CSDSP",
"CSLLI",
"CSRAI",
"CSRC",
"CSRCI",
"CSRLI",
"CSRR",
"CSRRC",
"CSRRCI",
"CSRRS",
"CSRRSI",
"CSRRW",
"CSRRWI",
"CSRS",
"CSRSI",
"CSRW",
"CSRWI",
"CSUB",
"CSUBW",
"CSW",
@@ -259,6 +269,7 @@ var riscvGeneratedInstrs = []string{
"ORCB",
"ORI",
"ORN",
"PAUSE",
"RDCYCLE",
"RDINSTRET",
"RDTIME",
@@ -322,6 +333,8 @@ var riscvGeneratedInstrs = []string{
"VADDVI",
"VADDVV",
"VADDVX",
"VANDNVV",
"VANDNVX",
"VANDVI",
"VANDVV",
"VANDVX",
@@ -329,8 +342,17 @@ var riscvGeneratedInstrs = []string{
"VASUBUVX",
"VASUBVV",
"VASUBVX",
"VBREV8V",
"VBREVV",
"VCLMULHVV",
"VCLMULHVX",
"VCLMULVV",
"VCLMULVX",
"VCLZV",
"VCOMPRESSVM",
"VCPOPM",
"VCPOPV",
"VCTZV",
"VDIVUVV",
"VDIVUVX",
"VDIVVV",
@@ -743,10 +765,16 @@ var riscvGeneratedInstrs = []string{
"VREMUVX",
"VREMVV",
"VREMVX",
"VREV8V",
"VRGATHEREI16VV",
"VRGATHERVI",
"VRGATHERVV",
"VRGATHERVX",
"VROLVV",
"VROLVX",
"VRORVI",
"VRORVV",
"VRORVX",
"VRSUBVI",
"VRSUBVX",
"VS1RV",
@@ -950,6 +978,9 @@ var riscvGeneratedInstrs = []string{
"VWMULVX",
"VWREDSUMUVS",
"VWREDSUMVS",
"VWSLLVI",
"VWSLLVV",
"VWSLLVX",
"VWSUBUVV",
"VWSUBUVX",
"VWSUBUWV",
+170
View File
@@ -4,6 +4,7 @@
package asm
import (
"encoding/binary"
"os"
"os/exec"
"path/filepath"
@@ -61,6 +62,62 @@ TEXT ·add(SB), NOSPLIT, $0-24
}
}
// TestGOObjectAARCH64PairReloc pins the ADRP-pair relocation shape against
// the toolchain's own object for the same source: exactly one R_ADDRARM64
// of Siz 8 at the ADRP word (cmd/internal/obj/arm64/asm7.go adds a single
// Siz-8 relocation per pair and the linker patches both instructions from
// it). gasm's assembler records the ADRP+ADD form as two word relocs; the
// emitter must coalesce them, not emit two Siz-4 records.
func TestGOObjectAARCH64PairReloc(t *testing.T) {
f, errs := parser.Parse("gv_arm64.s", `
#include "textflag.h"
TEXT ·getv(SB), NOSPLIT, $0-8
MOVD $v<>(SB), R4
MOVD R4, ret+0(FP)
RET
GLOBL v<>(SB), RODATA, $8
DATA v<>+0(SB)/8, $7
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.GOObjectAARCH64("main", "gv_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
v := openGoobj(t, obj)
relocs := v.blk(blkReloc)
le := binary.LittleEndian
// Two DWARF relocs on the lines/DIE symbols, then the code's one pair
// relocation.
if len(relocs) != 3*23 {
t.Fatalf("relocs = %d bytes, want three entries", len(relocs))
}
cr := relocs[2*23:]
if off := int32(le.Uint32(cr[0:])); off != 0 {
t.Errorf("pair reloc off = %d, want 0 (the ADRP word)", off)
}
if siz := cr[4]; siz != 8 {
t.Errorf("pair reloc siz = %d, want 8", siz)
}
if typ := le.Uint16(cr[5:]); typ != relocArm64Addr {
t.Errorf("pair reloc type = %d, want %d (R_ADDRARM64)", typ, relocArm64Addr)
}
if pkg := le.Uint32(cr[15:]); pkg != pkgIdxSelf {
t.Errorf("pair reloc PkgIdx = %#x, want pkgIdxSelf", pkg)
}
// The GLOBL is the first package definition.
if sym := le.Uint32(cr[19:]); sym != 0 {
t.Errorf("pair reloc SymIdx = %d, want 0 (the GLOBL definition)", sym)
}
}
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
// Go program for arm64, substitutes the gasm-produced object into the package
// archive, re-links with cmd/link, and verifies the symbol appears in the
@@ -79,6 +136,14 @@ TEXT ·add(SB), NOSPLIT, $0-24
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
TEXT ·getv(SB), NOSPLIT, $0-8
MOVD $v<>(SB), R4
MOVD R4, ret+0(FP)
RET
GLOBL v<>(SB), RODATA, $8
DATA v<>+0(SB)/8, $7
`
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
@@ -86,11 +151,15 @@ TEXT ·add(SB), NOSPLIT, $0-24
mainSrc := `package main
func add(a, b int64) int64
func getv() *int64
func main() {
if add(20, 22) != 42 {
panic("bad add")
}
if getv() == nil {
panic("bad getv")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
@@ -176,3 +245,104 @@ func main() {
t.Error("binary does not contain expected symbol")
}
}
// TestGOObjectAARCH64DataSymbolLink does for symbol-valued DATA fields what
// the rt0 files do ("DATA _rt0…lib+0(SB)/8, $_rt0…lib(SB)"): the gasm object
// carries an R_ADDR against the file's own TEXT symbol, the toolchain links
// it, and the binary is checked for the symbol (no arm64 host to run it).
func TestGOObjectAARCH64DataSymbolLink(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
GLOBL entry(SB), NOPTR, $8
DATA entry+0(SB)/8, $·keepme(SB)
TEXT ·keepme(SB), NOSPLIT, $0-0
RET
TEXT ·entryptr(SB), NOSPLIT, $0-8
MOVD entry+0(SB), R4
MOVD R4, ret+0(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
func keepme()
func entryptr() uintptr
func main() {
if entryptr() == 0 {
panic("the entry word is empty")
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64dlink\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
build.Env = append(os.Environ(), "GOARCH=arm64")
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse("main_arm64.s", string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64dlink", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
binData, err := os.ReadFile(filepath.Join(dir, "prog"))
if err != nil {
t.Fatal(err)
}
if !strings.Contains(string(binData), "keepme") {
t.Error("binary does not contain the keepme symbol")
}
}
+3183 -232
View File
File diff suppressed because it is too large Load Diff
+834 -97
View File
File diff suppressed because it is too large Load Diff
+1140 -5
View File
File diff suppressed because it is too large Load Diff
+256 -19
View File
@@ -63,6 +63,12 @@ type arm64FrameInfo struct {
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
// Stack-split guard state: needSplit mirrors the toolchain, which skips
// the check for NOSPLIT functions and auto-marks leaf functions with an
// autosize below StackSmall as NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// arm64ComputeFrame derives the frame layout for a TEXT function.
@@ -80,15 +86,68 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
if fi.frame != 0 || !fi.leaf {
fi.autosize = fi.frame + 8 // space for the saved LR
if fi.autosize%16 != 0 {
// The toolchain aligns to 16: if autosize%16 == 8, add 8;
// otherwise add whatever is needed.
// The toolchain always adds an extrasize: 8 when the total leaves a
// 16-byte alignment gap, another 16 when already aligned.
switch fi.autosize % 16 {
case 8:
fi.autosize += 8
case 0:
fi.autosize += 16
default:
// The toolchain rejects unaligned frames; round up so such
// sources still assemble.
fi.autosize += 16 - (fi.autosize % 16)
}
}
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
default:
fi.needSplit = true
switch {
case fi.autosize <= stackSmall:
fi.splitClass = 0
case fi.autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
// arm64GuardLen returns the byte length of the stack-split guard prefix
// (zero when the function needs no guard). The big class materialises
// framesize-StackSmall into REGTMP, whose MOVZ/MOVK sequence length varies.
func arm64GuardLen(fi arm64FrameInfo) int {
if !fi.needSplit {
return 0
}
switch fi.splitClass {
case 0:
return 12
case 1:
return 16
default:
n, err := arm64LoadImmLen(int64(fi.autosize - stackSmall))
if err != nil {
return 0
}
return 4 + n + 4 + 4 + 4 + 4
}
}
// arm64LoadImmLen returns the byte length of the MOVZ/MOVK sequence that
// loads v into a register.
func arm64LoadImmLen(v int64) (int, error) {
b, err := encodeARM64LoadImm(27, v, "MOVD")
if err != nil {
return 0, err
}
return len(b), nil
}
// arm64IsLeaf reports whether a function contains no call instructions
// (BL/CALL), matching the toolchain's LEAF mark.
func arm64IsLeaf(t *ast.Text) bool {
@@ -119,12 +178,99 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
)
}
// Large frame: SUB $autosize, SP, R20; STP (FP,LR), -8(R20); ADD $0, R20, SP; SUB $8, SP, FP
return a64WordsLE(
a64AddSub(1, 1, 0, 0, uint32(fi.autosize), 31, 20), // SUB $autosize, SP, R20
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
ws := arm64SubImmWords(uint32(fi.autosize), 20)
ws = append(ws,
a64LSP(2, 0, 0, -1, 30, 20, 29), // STP FP, LR, [R20, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, 0, 20, 31), // ADD $0, R20, SP (= MOV R20, SP)
a64AddSub(1, 1, 0, 0, 8, 31, 29), // SUB $8, SP, FP (op=1 for SUB)
)
return a64WordsLE(ws...)
}
// arm64SplitImm12 reports whether the toolchain decomposes ADD/SUB $imm into
// two imm12 instructions instead of materialising it into REGTMP
// (asm7.go case 48, the C_ADDCON2 class): the value must fit 24 bits
// unsigned and be neither encodable as one imm12 (checked by the callers
// first), nor loadable into a register in a single MOVZ/MOVN word, nor a
// logical immediate, because conclass tests all three before C_ADDCON2.
func arm64SplitImm12(imm uint32) bool {
if imm > 0xFFFFFF {
return false
}
if _, _, _, ok := arm64Bitmask(uint64(imm), 1); ok {
return false
}
return arm64Movcon(int64(imm)) < 0 && arm64Movcon(^int64(imm)) < 0
}
// arm64SubImmWords emits SUB $imm, SP, Rd with the toolchain's ladder for an
// ADD/SUB constant (asm7.go conclass and cases 2, 48, 62 and 13): the
// immediate form when the value fits imm12 (plain, or shifted left by 12
// when it is a multiple of 4096); a value with a single 16-bit chunk, a
// logical immediate, or one wider than 24 bits is materialised into REGTMP
// (R27) and subtracted in the extended-register form; everything else up to
// 0xFFFFFF is split into two imm12 instructions:
//
// SUB $(imm&0xfff), SP, Rd
// SUB $((imm&0xfff000)>>12)<<12, Rd, Rd
func arm64SubImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 1, 0, 0, imm, 31, rd)}
}
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 1, 0, 1, imm>>12, 31, rd)}
}
if !arm64SplitImm12(imm) {
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpSub, 27, 31, rd))
}
return []uint32{
a64AddSub(1, 1, 0, 0, imm&0xFFF, 31, rd),
a64AddSub(1, 1, 0, 1, (imm&0xFFF000)>>12, rd, rd),
}
}
// arm64AddImmWords emits ADD $imm, SP, Rd with the same imm12, shifted-imm12,
// split and REGTMP ladder as arm64SubImmWords.
func arm64AddImmWords(imm uint32, rd uint32) []uint32 {
if imm <= 0xFFF {
return []uint32{a64AddSub(1, 0, 0, 0, imm, 31, rd)}
}
if imm <= 4095<<12 && imm&0xFFF == 0 {
return []uint32{a64AddSub(1, 0, 0, 1, imm>>12, 31, rd)}
}
if !arm64SplitImm12(imm) {
mov, err := encodeARM64LoadImm(27, int64(imm), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, rd))
}
return []uint32{
a64AddSub(1, 0, 0, 0, imm&0xFFF, 31, rd),
a64AddSub(1, 0, 0, 1, (imm&0xFFF000)>>12, rd, rd),
}
}
// arm64RetAddWords emits the frame deallocation of a non-leaf RET with a
// large frame. The toolchain adds the frame back with a single instruction:
// a plain imm12 ADD when autosize fits 12 bits, otherwise the value is
// materialised into REGTMP and added as a register, so the epilogue never
// leaves a partially deallocated frame (obj7.go ARET, issue 73259). The
// shifted-imm12 and split-imm12 forms are therefore never used here, unlike
// the leaf epilogue's plain ADD instructions.
func arm64RetAddWords(autosize uint32) []uint32 {
if autosize < 1<<12 {
return []uint32{a64AddSub(1, 0, 0, 0, autosize, 31, 31)}
}
mov, err := encodeARM64LoadImm(27, int64(autosize), "MOVD")
if err != nil {
mov = nil
}
return append(wordsOf(mov), arm64DPExtWords(arm64OpAdd, 27, 31, 31))
}
// arm64Return returns the bytes for a RET: the epilogue (restore FP/LR and
@@ -134,10 +280,8 @@ func arm64Return(fi arm64FrameInfo) []byte {
if fi.autosize != 0 {
if fi.leaf {
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
ws = append(ws,
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
)
ws = append(ws, arm64AddImmWords(uint32(fi.autosize-8), 29)...)
ws = append(ws, arm64AddImmWords(uint32(fi.autosize), 31)...)
} else if fi.autosize <= 0xf0 {
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
ws = append(ws,
@@ -145,11 +289,11 @@ func arm64Return(fi arm64FrameInfo) []byte {
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
)
} else {
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
// Large frame: LDP -8(SP), (FP, LR), then deallocate.
ws = append(ws,
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
a64LSP(2, 0, 1, -1, 30, 31, 29), // LDP FP, LR, [SP, #-8] (opc=2 for 64-bit pair)
)
ws = append(ws, arm64RetAddWords(uint32(fi.autosize))...)
}
}
// RET: BR LR (0xd65f03c0)
@@ -166,22 +310,32 @@ func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
if fi.autosize <= 0xf0 {
return 4 // MOVD.W instruction decrements SP
}
return 8 // SUB + STP + MOVD (3 instructions, SP updated at the MOVD)
// Large frame: [SUB words][STP][ADD R20, SP]; SP moves at the ADD, whose
// position depends on how many words the SUB itself took (immediate,
// shifted immediate, the two-word imm12 split, or a materialised REGTMP
// sequence).
return 4 * (len(arm64SubImmWords(uint32(fi.autosize), 20)) + 1)
}
// arm64ReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final RET instruction.
// (but not including) the final RET instruction. The lengths are read from
// the same word-emitting helpers the epilogue uses rather than assumed: the
// leaf path shares the prologue's immediate ladder, and a materialised
// autosize costs its MOV words plus the ADD itself.
func arm64ReturnEpilogueLen(fi arm64FrameInfo) int {
if fi.autosize == 0 {
return 0
}
if fi.leaf {
return 8 // ADD + ADD
return 4 * (len(arm64AddImmWords(uint32(fi.autosize-8), 29)) +
len(arm64AddImmWords(uint32(fi.autosize), 31)))
}
if fi.autosize <= 0xf0 {
return 8 // LDR + LDR.P
}
return 8 // LDP + ADD
// LDP + the deallocation emitted by arm64RetAddWords, so the length
// tracks whatever the MOVD ladder needs.
return 4 + 4*len(arm64RetAddWords(uint32(fi.autosize)))
}
// arm64ResolvePseudo translates a pseudo-register memory reference into a
@@ -235,3 +389,86 @@ func arm64PostLoad(size, V int, imm9 int32, rn, rt int) uint32 {
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | 1<<22 |
1<<10 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
}
// Data-processing (shifted register) base opcodes for the guard blocks.
const (
arm64OpAdd = 1<<31 | 0<<30 | 0<<29 | 0x0b<<24
arm64OpSub = 1<<31 | 1<<30 | 0<<29 | 0x0b<<24
arm64OpSubs = 1<<31 | 1<<30 | 1<<29 | 0x0b<<24
)
// arm64DPSRWords builds one data-processing (shifted register) word:
// OP Rm, Rn, Rd in the Go assembler's operand order.
func arm64DPSRWords(base uint32, rm, rn, rd uint32) uint32 {
return base | rm<<16 | rn<<5 | rd
}
// arm64DPExtWords builds one data-processing (extended register) word, the
// form the toolchain picks when a large immediate was materialised into
// REGTMP before the operation: base | 1<<21 | Rm<<16 | UXTX<<13 | Rn<<5 | Rd.
func arm64DPExtWords(base, rm, rn, rd uint32) uint32 {
return base | 1<<21 | rm<<16 | 3<<13 | rn<<5 | rd
}
// wordsOf converts little-endian instruction bytes back to words.
func wordsOf(b []byte) []uint32 {
ws := make([]uint32, 0, len(b)/4)
for i := 0; i+4 <= len(b); i += 4 {
ws = append(ws, uint32(b[i])|uint32(b[i+1])<<8|uint32(b[i+2])<<16|uint32(b[i+3])<<24)
}
return ws
}
// arm64GuardBytes emits the stack-split guard prefix; blockStart is the
// function-relative byte address of the morestack block the branches target.
func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
// MOVD 16(R28), R16 (g.stackguard0)
ws := []uint32{a64LSU(3, 0, 1, 2, 28, 16)}
br := func(from int, cond uint32) uint32 {
return a64BranchCond(int32((blockStart-from)>>2), cond)
}
switch fi.splitClass {
case 0:
// CMP R16, RSP in the exact encoding go tool asm emits for it.
ws = append(ws, 0xeb3063ff)
ws = append(ws, br(8, a64CondLS))
case 1:
ws = append(ws, a64AddSub(1, 1, 0, 0, uint32(fi.autosize-stackSmall), 31, 17))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(12, a64CondLS))
default:
mov, err := encodeARM64LoadImm(27, int64(fi.autosize-stackSmall), "MOVD")
if err != nil {
mov = nil
}
ws = append(ws, wordsOf(mov)...)
ml := len(mov) / 4
ws = append(ws, arm64DPExtWords(arm64OpSubs, 27, 31, 17)) // SUBS R17, RSP, R27
// The branches sit at fixed byte offsets in the guard prefix: after
// the LDR (4), the ml MOV words (4*ml) and the SUBS (4) for B.LO,
// then a further B.LO word and the CMP for B.LS.
ws = append(ws, br(8+4*ml, a64CondLO))
ws = append(ws, arm64DPSRWords(arm64OpSubs, 16, 17, 31)) // CMP R16, R17
ws = append(ws, br(16+4*ml, a64CondLS))
}
return a64WordsLE(ws...)
}
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
// the R_CALLARM64 relocation.
func arm64MoreStackBlock(blockStart int) ([]byte, Reloc) {
ws := []uint32{
1<<31 | 1<<29 | 0x0a<<24 | 30<<16 | 31<<5 | 3, // MOVD R30, R3
a64Branch(1, 0), // BL, patched by the linker
}
bPC := blockStart + 8
ws = append(ws, a64Branch(0, int32(-bPC>>2))) // B back to the entry
reloc := Reloc{
Off: blockStart + 4,
After: blockStart + 8,
Name: "runtime\u00b7morestack_noctxt",
Kind: RelArm64Branch,
}
return a64WordsLE(ws...), reloc
}
+126
View File
@@ -0,0 +1,126 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// parseArm64File is a helper assembling one arm64 source file.
func parseArm64File(t *testing.T, src string) *Image {
t.Helper()
f, errs := parser.Parse("k_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
return img
}
// TestArm64RelocOffsetsIncludePrologue pins the function-relative relocation
// offsets of a framed function: the offsets used to exclude the prologue, so
// every relocation landed on a prologue instruction in the GOOBJ/ELF output.
// The function calls an external, so it is a non-leaf and carries the
// stack-split guard (12 bytes, small class) before the prologue.
func TestArm64RelocOffsetsIncludePrologue(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7f(SB), $16-0\n"+
"\tBL ext\u00b7foo(SB)\n"+
"\tMOVD $gdata(SB), R5\n"+
"\tMOVD $extsym(SB), R6\n"+
"\tRET\n"+
"GLOBL gdata(SB), $8\n")
fn := img.Funcs[0]
// Layout: 12-byte guard, 12-byte prologue, BL (24), ADRP+ADD (28, 32),
// ADRP+ADD (36, 40), 12-byte epilogue with RET, 12-byte morestack block.
want := []struct {
off int
after int
name string
kind RelocKind
external bool
}{
{24, 28, "foo", RelArm64Branch, true},
{28, 28, "gdata", RelArm64Addr, false},
{32, 32, "gdata", RelArm64Addr, false},
{36, 36, "extsym", RelArm64Addr, true},
{40, 40, "extsym", RelArm64Addr, true},
{60, 64, "runtime\u00b7morestack_noctxt", RelArm64Branch, true},
}
if len(fn.Relocs) != len(want) {
t.Fatalf("relocs = %d, want %d", len(fn.Relocs), len(want))
}
for i, w := range want {
r := fn.Relocs[i]
if r.Off != w.off || r.After != w.after || r.Name != w.name || r.Kind != w.kind || r.External != w.external {
t.Errorf("reloc %d = {off %d after %d name %q kind %d ext %v}, want {off %d after %d name %q kind %d ext %v}",
i, r.Off, r.After, r.Name, r.Kind, r.External, w.off, w.after, w.name, w.kind, w.external)
}
}
// The BL with a zero offset sits exactly at the first reloc site.
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if w := binary.LittleEndian.Uint32(code[24:28]); w != 0x94000000 {
t.Errorf("BL word = %08x, want 94000000", w)
}
}
// TestArm64SBLoadStoreMatchesToolchain pins the ADRP scratch register
// (REGTMP, R27) and the LDST64 relocation kind for sym loads and stores,
// against the bytes go tool asm emits for MOVD sym(SB), R5.
func TestArm64SBLoadStoreMatchesToolchain(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
"\tMOVD sym(SB), R5\n"+
"\tMOVD R5, sym(SB)\n"+
"\tRET\n"+
"GLOBL sym(SB), $8\n")
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
// go tool asm: ADRP 0(PC), R27 (9000001b); MOVD (R27), R5 (f9400365);
// ADRP 0(PC), R27; MOVD R5, (R27) (f9000365).
for off, want := range map[int]uint32{0: 0x9000001b, 4: 0xf9400365, 8: 0x9000001b, 12: 0xf9000365} {
if got := binary.LittleEndian.Uint32(code[off : off+4]); got != want {
t.Errorf("word at %d = %08x, want %08x", off, got, want)
}
}
if len(fn.Relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
}
for i, w := range []struct{ off, after int }{{0, 8}, {8, 16}} {
r := fn.Relocs[i]
if r.Kind != RelArm64LDST64 {
t.Errorf("reloc %d kind = %d, want RelArm64LDST64 (%d)", i, r.Kind, RelArm64LDST64)
}
if r.Off != w.off || r.After != w.after {
t.Errorf("reloc %d = {off %d after %d}, want {off %d after %d}", i, r.Off, r.After, w.off, w.after)
}
}
}
// TestArm64GOObjRelocTypes checks that GOOBJ emission succeeds with the new
// relocation kinds in play; the detailed layout is covered by the goobj tests.
func TestArm64GOObjRelocTypes(t *testing.T) {
img := parseArm64File(t, "TEXT \u00b7ld(SB), NOSPLIT, $0\n"+
"\tMOVD sym(SB), R5\n"+
"\tMOVD R5, sym(SB)\n"+
"\tRET\n"+
"GLOBL sym(SB), $8\n")
obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
if len(obj) == 0 {
t.Fatal("empty object")
}
// The detailed layout is covered by the goobj tests; here we only pin
// that emission succeeds with the new relocation kinds in play.
}
+910 -47
View File
File diff suppressed because it is too large Load Diff
+244 -2
View File
@@ -4,6 +4,7 @@
package asm
import (
"bytes"
"strings"
"testing"
@@ -159,6 +160,57 @@ TEXT ·loadarg(SB), NOSPLIT, $0-24
}
}
// TestAssembleFramelessCall verifies the forced base-pointer frame a $0-frame
// function containing a CALL receives: the PUSHQ BP prologue with no stack
// adjustment and the x+N(FP) → (N+16)(SP) translation, against the bytes the
// Go assembler produces. The push is the frame, so the offset must not count
// it twice.
func TestAssembleFramelessCall(t *testing.T) {
f, errs := parser.Parse("frameless_call_amd64.s", `
#include "textflag.h"
TEXT ·withcall(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
CALL ·other(SB)
MOVQ AX, ret+8(FP)
RET
TEXT ·other(SB), NOSPLIT, $0-0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
code := append([]byte(nil), img.Code[img.Funcs[0].Offset:img.Funcs[0].Offset+img.Funcs[0].Size]...)
for _, r := range img.Funcs[0].Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
// From `go tool objdump` of the Go-assembled function:
// PUSHQ BP 55
// MOVQ SP, BP 4889e5
// MOVQ 0x10(SP), AX 488b442410
// CALL other e800000000
// MOVQ AX, 0x18(SP) 4889442418
// POPQ BP 5d
// RET c3
want := []byte{
0x55,
0x48, 0x89, 0xe5,
0x48, 0x8b, 0x44, 0x24, 0x10,
0xe8, 0x00, 0x00, 0x00, 0x00,
0x48, 0x89, 0x44, 0x24, 0x18,
0x5d,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("frameless CALL FP translation mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleFrame verifies a function with a non-zero frame: the Go-style
// prologue/epilogue and the x+N(FP) → (N+frame+16)(SP) translation, against
// the bytes the Go assembler produces.
@@ -201,8 +253,8 @@ TEXT ·withframe(SB), NOSPLIT, $16-16
}
// TestAssembleVexKernel assembles the horizontal-sum reduction the go-flac
// kernels end with — exercising the VEX moves, shuffle and extract forms
// through the full parser → encoder path — and checks the output is
// kernels end with; exercising the VEX moves, shuffle and extract forms
// through the full parser → encoder path; and checks the output is
// byte-identical to the Go assembler's.
func TestAssembleVexKernel(t *testing.T) {
fn := firstText(t, `
@@ -351,3 +403,193 @@ TEXT ·pf(SB), NOSPLIT, $0
t.Errorf("PREFETCHT0 bytes: got %s, want 0f 18 0b", hex)
}
}
// TestAssembleBareJump checks that a zero-operand jump (which parses, because
// the parser does not arity-check mnemonics) is rejected with an error rather
// than panicking in the layout loop, which indexes Operands[0] before the
// emission pass gets a chance to diagnose the arity.
func TestAssembleBareJump(t *testing.T) {
for _, mnem := range []string{"JE", "JMP", "JLT", "CALL"} {
fn := firstText(t, "TEXT ·bare(SB), $16-0\n\t"+mnem+"\n")
if _, _, err := Assemble(fn); err == nil {
t.Errorf("%s with no operand: expected an error, got none", mnem)
}
}
}
// TestSubSPEncodings pins the prologue SUB against the bytes go tool asm
// emits for SUBQ $size, SP: imm8 for -128..127, the imm32 form for anything
// larger. The intermediate 129..255 range used to encode an ADD with a
// truncated immediate, moving SP the wrong way.
func TestSubSPEncodings(t *testing.T) {
for _, tt := range []struct {
size int
want []byte
}{
{8, []byte{0x48, 0x83, 0xEC, 0x08}},
{127, []byte{0x48, 0x83, 0xEC, 0x7F}},
{128, []byte{0x48, 0x81, 0xEC, 0x80, 0x00, 0x00, 0x00}},
{200, []byte{0x48, 0x81, 0xEC, 0xC8, 0x00, 0x00, 0x00}},
{255, []byte{0x48, 0x81, 0xEC, 0xFF, 0x00, 0x00, 0x00}},
{4096, []byte{0x48, 0x81, 0xEC, 0x00, 0x10, 0x00, 0x00}},
} {
got := subSP(tt.size)
if !bytes.Equal(got, tt.want) {
t.Errorf("subSP(%d) = %x, want %x", tt.size, got, tt.want)
}
}
}
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
// asserts the three behaviours the toolchain shows: each prefix statement is
// a standalone byte with a PC of its own (so a label placed on the LOCK
// points at the F0), the data pseudo-ops write their literal bytes inline,
// and END terminates nothing (the statements after it still belong to the
// function and carry no trace of it).
func TestAssemblePseudoStatements(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·pseudo(SB), NOSPLIT, $0-0
pfx:
LOCK
CMPXCHGQ AX, (BX)
REP
MOVSQ
BYTE $0x0f
BYTE $0x1f
WORD $0x1234
END
BYTE $0x02
RET
`)
code, labels, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
want := []byte{
0xf0,
0x48, 0x0f, 0xb1, 0x03,
0xf3, 0x48, 0xa5,
0x0f, 0x1f, 0x34, 0x12,
0x02, 0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
// The label sits on the LOCK byte, exactly where the toolchain's PC
// listing puts it.
if off := labels["pfx"]; off != 0 {
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
}
// The trailing BYTE lands where the layout says: after the 8 bytes of
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
// none.
if code[12] != 0x02 {
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
}
}
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
// ADJSP: the straight-line sum of the adjustments must be zero at each
// RET, branches in between counting for nothing (verified against go tool
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
// $16/$-16 pair with a JMP in between assembles).
func TestAssembleAdjspBalance(t *testing.T) {
// Balanced pair with a branch in between, bytes pinned from go tool asm.
fn := firstText(t, `
#include "textflag.h"
TEXT ·adjsp(SB), NOSPLIT, $0-0
ADJSP $16
JMP body
body:
ADJSP $-16
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
if hexBytes(code) != hexBytes(want) {
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
// Unbalanced at the RET: the toolchain diagnoses, so must we.
_, _, err = Assemble(firstText(t, `
#include "textflag.h"
TEXT ·unbalanced(SB), NOSPLIT, $0-0
ADJSP $16
RET
`))
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
}
// The check runs per RET: a closed pair before the first RET does not
// excuse an open adjustment before the second.
_, _, err = Assemble(firstText(t, `
#include "textflag.h"
TEXT ·tworet(SB), NOSPLIT, $0-0
ADJSP $8
ADJSP $-8
RET
mid:
ADJSP $8
RET
`))
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
}
// A framed function: the assembler's own prologue and epilogue
// contribute matching deltas, so the pair in the body still balances,
// and the bytes match go tool asm end to end.
fn = firstText(t, `
#include "textflag.h"
TEXT ·framed(SB), $16-8
ADJSP $8
ADJSP $-8
RET
`)
code, _, err = Assemble(fn)
if err != nil {
t.Fatalf("Assemble framed: %v", err)
}
want = []byte{
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
0xC3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
// TestAssembleRegRange pins the bracketed register range at the statement
// level: exactly four consecutive same-width vector registers assemble, the
// toolchain's rejected shapes all report an error.
func TestAssembleRegRange(t *testing.T) {
asm := func(t *testing.T, op string) ([]byte, error) {
t.Helper()
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse %s: %v", op, errs)
}
code, _, err := Assemble(f.Decls[0].(*ast.Text))
return code, err
}
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
if _, err := asm(t, op); err != nil {
t.Errorf("%s: %v", op, err)
}
}
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
if _, err := asm(t, op); err == nil {
t.Errorf("%s: assembled, want an error", op)
}
}
}
+124 -17
View File
@@ -12,7 +12,7 @@ import (
// Image: a .text section holding the function bodies, a .data section
// holding the GLOBL initialisers, a symbol table with one symbol per TEXT
// and GLOBL (file-local <> symbols are STB_LOCAL, the rest STB_GLOBAL), and
// a .rela.text relocation table — one R_X86_64_PC32 entry per static-symbol
// a .rela.text relocation table, one R_X86_64_PC32 entry per static-symbol
// reference, internal references resolving against the local data symbols
// and external ones against undefined globals. The output links with the
// system toolchain (cc/ld) the way a hand-assembled .o would.
@@ -43,6 +43,12 @@ const (
stInfoShift = 4
rX8664PC32 = 2
// R_X86_64_32 (debug/elf): the absolute 32-bit address of a symbol, the
// R_ADDR shape a 4-byte DATA field carries.
rX8664Abs32 = 10
// R_X86_64_TPOFF32 (debug/elf): the local-exec TLS offset the stack
// guard loads from FS. 20 is R_X86_64_TLSLD, a different relocation.
rX8664TPOFF32 = 23
)
// elfSym is one symbol-table entry in construction.
@@ -71,7 +77,7 @@ func (img *Image) ELFObject() ([]byte, error) {
// Build the symbol table: the null entry and the two section symbols
// come first, then the local symbols (static TEXT and GLOBL), then the
// globals (exported TEXT and GLOBL, and the undefined externals) — ELF
// globals (exported TEXT and GLOBL, and the undefined externals), ELF
// requires every local to precede every global, and sh_info records the
// boundary. symIdx maps a symbol name to its index for the relocations.
var locals, globals []elfSym
@@ -125,11 +131,19 @@ func (img *Image) ELFObject() ([]byte, error) {
type elfRela struct {
off uint64
sym int
typ uint32
addend int64
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
var typ uint32 = rX8664PC32
if r.Kind == RelTLSLE {
// R_X86_64_TPOFF32 resolves to the local-exec TLS offset and
// carries no symbol.
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), sym: 0, typ: rX8664TPOFF32})
continue
}
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
@@ -137,6 +151,7 @@ func (img *Image) ELFObject() ([]byte, error) {
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
sym: idx,
typ: typ,
// R_X86_64_PC32 computes S + A − P with P the patch site; the
// assembler measures the symbol from the instruction end,
// After − Off bytes past the field, so the addend carries
@@ -146,6 +161,50 @@ func (img *Image) ELFObject() ([]byte, error) {
}
}
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rX8664Abs64
case 4:
typ = rX8664Abs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Serialise the string tables.
stNames := newElfStrtab()
for _, s := range syms {
@@ -155,19 +214,13 @@ func (img *Image) ELFObject() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
// Section presence: .rela.text only when there are relocations.
hasRela := len(relas) > 0
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Lay the file out: header, section data, section headers.
var out []byte
out = append(out, make([]byte, 64)...) // ELF header, filled last
@@ -202,14 +255,25 @@ func (img *Image) ELFObject() ([]byte, error) {
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
var relaOff, relaDataOff int
if hasRela {
align(8)
relaOff = len(out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|rX8664PC32)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
@@ -218,15 +282,32 @@ func (img *Image) ELFObject() ([]byte, error) {
shstrOff := len(out)
out = append(out, stSections.bytes()...)
// DWARF debug sections (no relocations — the linker resolves DWARF fixups).
// DWARF debug sections; the address placeholders they leave are carried
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiAMD64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
nSections += 4 // .debug_abbrev, .debug_info, .debug_line, .debug_line_str
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rX8664Abs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
@@ -255,16 +336,42 @@ func (img *Image) ELFObject() ([]byte, error) {
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers.
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
+162 -75
View File
@@ -12,43 +12,74 @@ import (
// self-contained sections because the system linker only performs fixup
// relocations, not assembly.
// DWARF5 attribute, form and line-table constants (the values the
// toolchain uses, cmd/internal/dwarf/dwarf_defs.go; the DIE streams below
// are written against these forms).
const (
dwAtName = 0x03 // DW_AT_name
dwAtStmtList = 0x10 // DW_AT_stmt_list
dwAtLowPC = 0x11 // DW_AT_low_pc
dwAtHighPC = 0x12 // DW_AT_high_pc
dwAtDeclFile = 0x3a // DW_AT_decl_file
dwAtDeclLine = 0x3b // DW_AT_decl_line
dwAtExternal = 0x3f // DW_AT_external
dwAtFrameBase = 0x40 // DW_AT_frame_base
dwTagSubprog = 0x2e // DW_TAG_subprogram
dwTagCompUnit = 0x11 // DW_TAG_compile_unit
dwFormAddr = 0x01 // DW_FORM_addr
dwFormData8 = 0x07 // DW_FORM_data8
dwFormString = 0x08 // DW_FORM_string
dwFormData1 = 0x0b // DW_FORM_data1
dwFormUdata = 0x0f // DW_FORM_udata
dwFormSecOff = 0x17 // DW_FORM_sec_offset
dwFormExprloc = 0x18 // DW_FORM_exprloc
dwFormLineStrp = 0x1f // DW_FORM_line_strp
dwLnctPath = 0x01 // DW_LNCT_path
dwLnctDirIndex = 0x02 // DW_LNCT_directory_index
)
// dwarfAbbrevTable returns the .debug_abbrev content: a single compilation
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries.
// unit with DW_TAG_compile_unit and DW_TAG_subprogram entries. The
// attribute/form pairs must match the DIE streams dwarfBuildInfoSection
// writes byte for byte, in the same order, or every consumer's parse of
// .debug_info desynchronises.
func dwarfAbbrevTable() []byte {
var b []byte
// Abbrev 1: DW_TAG_compile_unit
b = append(b, 1) // abbreviation code
b = append(b, 0x11) // DW_TAG_compile_unit
b = append(b, 1) // DW_CHILDREN_yes
b = appendUleb(b, 0x1b) // DW_AT_low_pc
b = appendUleb(b, 0x01) // DW_FORM_addr
b = appendUleb(b, 0x29) // DW_AT_high_pc
b = appendUleb(b, 0x07) // DW_FORM_data8
b = appendUleb(b, 0x10) // DW_AT_stmt_list
b = appendUleb(b, 0x25) // DW_FORM_sec_offset
b = appendUleb(b, 0x01) // DW_AT_name
b = appendUleb(b, 0x08) // DW_FORM_string
b = appendUleb(b, 0) // end of attributes
// Abbrev 1: DW_TAG_compile_unit.
b = append(b, 1) // abbreviation code
b = appendUleb(b, dwTagCompUnit) // DW_TAG_compile_unit
b = append(b, 1) // DW_CHILDREN_yes
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
b = appendUleb(b, dwFormData8) // DW_FORM_data8
b = appendUleb(b, dwAtStmtList) // DW_AT_stmt_list
b = appendUleb(b, dwFormSecOff) // DW_FORM_sec_offset (4 bytes here)
b = appendUleb(b, dwAtName) // DW_AT_name
b = appendUleb(b, dwFormString) // DW_FORM_string
b = appendUleb(b, 0) // end of attributes: attr 0
b = appendUleb(b, 0) // ... paired with form 0
// Abbrev 2: DW_TAG_subprogram
b = append(b, 2) // abbreviation code
b = append(b, 0x2e) // DW_TAG_subprogram
b = append(b, 0) // DW_CHILDREN_no
b = appendUleb(b, 0x03) // DW_AT_name
b = appendUleb(b, 0x08) // DW_FORM_string
b = appendUleb(b, 0x11) // DW_AT_low_pc
b = appendUleb(b, 0x01) // DW_FORM_addr
b = appendUleb(b, 0x29) // DW_AT_high_pc
b = appendUleb(b, 0x07) // DW_FORM_data8
b = appendUleb(b, 0x3f) // DW_AT_frame_base
b = appendUleb(b, 0x18) // DW_FORM_exprloc
b = appendUleb(b, 0x3b) // DW_AT_decl_file
b = appendUleb(b, 0x0b) // DW_FORM_data1
b = appendUleb(b, 0x37) // DW_AT_decl_line
b = appendUleb(b, 0x0b) // DW_FORM_data1
b = appendUleb(b, 0x63) // DW_AT_external
b = appendUleb(b, 0x0b) // DW_FORM_flag
b = appendUleb(b, 0) // end of attributes
// Abbrev 2: DW_TAG_subprogram.
b = append(b, 2) // abbreviation code
b = appendUleb(b, dwTagSubprog) // DW_TAG_subprogram
b = append(b, 0) // DW_CHILDREN_no
b = appendUleb(b, dwAtName) // DW_AT_name
b = appendUleb(b, dwFormString) // DW_FORM_string
b = appendUleb(b, dwAtLowPC) // DW_AT_low_pc
b = appendUleb(b, dwFormAddr) // DW_FORM_addr
b = appendUleb(b, dwAtHighPC) // DW_AT_high_pc
b = appendUleb(b, dwFormData8) // DW_FORM_data8
b = appendUleb(b, dwAtFrameBase) // DW_AT_frame_base
b = appendUleb(b, dwFormExprloc) // DW_FORM_exprloc
b = appendUleb(b, dwAtDeclFile) // DW_AT_decl_file
b = appendUleb(b, dwFormData1) // DW_FORM_data1
b = appendUleb(b, dwAtDeclLine) // DW_AT_decl_line
b = appendUleb(b, dwFormData1) // DW_FORM_data1
b = appendUleb(b, dwAtExternal) // DW_AT_external
b = appendUleb(b, 0x0c) // DW_FORM_flag (one byte, 0 or 1)
b = appendUleb(b, 0) // end of attributes: attr 0
b = appendUleb(b, 0) // ... paired with form 0
// End of table.
b = append(b, 0)
@@ -68,6 +99,9 @@ type dwarfSections struct {
infoRelocs []dwarfReloc
// Relocations for .debug_line: (offset, symbol name, addend).
lineRelocs []dwarfReloc
// Relocations for .debug_frame: (offset, symbol name, addend), one per
// FDE initial_location.
frameRelocs []dwarfReloc
}
type dwarfReloc struct {
@@ -76,8 +110,9 @@ type dwarfReloc struct {
addend int64
}
// emitDWARF generates complete DWARF5 sections for the image.
func emitDWARF(img *Image, srcFile string) *dwarfSections {
// emitDWARF generates complete DWARF5 sections for the image. cfi carries
// the architecture's .debug_frame register conventions.
func emitDWARF(img *Image, srcFile string, cfi cfiArch) *dwarfSections {
ds := &dwarfSections{}
ds.debugAbbrev = dwarfAbbrevTable()
@@ -86,19 +121,21 @@ func emitDWARF(img *Image, srcFile string) *dwarfSections {
lineStr.add(srcFile)
ds.debugLineStr = lineStr.bytes()
// Build .debug_line.
ds.debugLine = dwarfBuildLineSection(img, ds)
// Build .debug_line; the file table references the source name through
// its offset in .debug_line_str.
ds.debugLine = dwarfBuildLineSection(img, uint32(lineStr.at(srcFile)), ds)
// Build .debug_info.
ds.debugInfo = dwarfBuildInfoSection(img, srcFile, ds)
// Build .debug_frame.
ds.debugFrame = dwarfBuildFrameSection(img)
ds.debugFrame = dwarfBuildFrameSection(img, cfi, ds)
return ds
}
// dwarfBuildLineSection builds a complete .debug_line section.
func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
// dwarfBuildLineSection builds a complete .debug_line section. srcStrOff is
// the source file name's offset in .debug_line_str.
func dwarfBuildLineSection(img *Image, srcStrOff uint32, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
@@ -120,15 +157,25 @@ func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
// Standard opcode lengths (opcode 1..opcode_base-1).
b = append(b, 0, 1, 1, 1, 1, 0, 0, 0, 1, 0)
// Directory table (DWARF5 format).
b = append(b, 0) // one directory entry (index 0 = empty)
// File table.
b = appendUleb(b, 1) // file count
// File 1: name index into .debug_line_str, dir index, time, size.
b = appendUleb(b, 0) // name (index 0 in line_str)
b = appendUleb(b, 0) // directory index
b = appendUleb(b, 0) // last modification time
b = appendUleb(b, 0) // file size
// Directory table (DWARF5 §6.2.4): entry format descriptors followed by
// the entries. One directory, the compilation directory, whose path is
// the empty string at .debug_line_str offset 0.
b = append(b, 1) // directory_entry_format_count
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
b = appendUleb(b, 1) // directories_count
b = le.AppendUint32(b, 0) // .debug_line_str offset of ""
// File table (DWARF5 §6.2.5). v5 indexes files from 0, so the source
// file is entry 0, matching the DW_AT_decl_file value 0 the DIEs carry.
b = append(b, 2) // file_name_entry_format_count
b = appendUleb(b, dwLnctPath) // DW_LNCT_path
b = appendUleb(b, dwFormLineStrp) // DW_FORM_line_strp
b = appendUleb(b, dwLnctDirIndex) // DW_LNCT_directory_index
b = appendUleb(b, dwFormUdata) // DW_FORM_udata
b = appendUleb(b, 1) // file_names_count
b = le.AppendUint32(b, srcStrOff) // .debug_line_str offset of the source name
b = appendUleb(b, 0) // directory index 0 (the compilation directory)
headerEnd := len(b)
@@ -176,8 +223,11 @@ func dwarfBuildLineSection(img *Image, ds *dwarfSections) []byte {
// Patch unit_length.
le.PutUint32(b[headerStart:], uint32(len(b)-headerStart-4))
// Patch header_length.
le.PutUint32(b[headerStart+6:], uint32(headerEnd-headerStart-10))
// Patch header_length. In the v5 header it follows the one-byte
// address_size and segment_selector_size (offset 8, not the DWARF2-4
// offset 6), and counts from just past itself to the first program
// byte.
le.PutUint32(b[headerStart+8:], uint32(headerEnd-headerStart-12))
return b
}
@@ -195,14 +245,16 @@ func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte
// DW_TAG_compile_unit (abbrev 1).
b = append(b, 1) // abbreviation code
// DW_AT_low_pc: address of .text start.
infoRelocBase := len(b)
// DW_AT_low_pc: address of .text start. A data-only image has no
// functions to relocate against; its CU covers no code, so the base
// stays zero (the DWARF "no base address" value) with no relocation.
b = le.AppendUint64(b, 0) // placeholder
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
off: uint64(infoRelocBase),
name: img.Funcs[0].Name,
addend: 0,
})
if len(img.Funcs) > 0 {
ds.infoRelocs = append(ds.infoRelocs, dwarfReloc{
off: uint64(len(b) - 8),
name: img.Funcs[0].Name,
})
}
// DW_AT_high_pc: size of .text.
b = le.AppendUint64(b, uint64(len(img.Code)))
// DW_AT_stmt_list: offset into .debug_line (0).
@@ -229,8 +281,9 @@ func dwarfBuildInfoSection(img *Image, srcFile string, ds *dwarfSections) []byte
b = le.AppendUint64(b, uint64(fn.Size))
// DW_AT_frame_base: DW_OP_call_frame_cfa.
b = append(b, 1, 0x9c)
// DW_AT_decl_file: file index 1.
b = append(b, 1)
// DW_AT_decl_file: the single file-table entry, index 0 (v5 indexes
// files from 0).
b = append(b, 0)
// DW_AT_decl_line.
b = append(b, uint8(fn.Line))
// DW_AT_external.
@@ -253,41 +306,75 @@ func appendUleb(b []byte, v uint64) []byte {
return binary.AppendUvarint(b, v)
}
// appendSleb appends v in signed LEB128, the encoding DWARF specifies:
// two's-complement sign extension, which is NOT Go's zigzag varint
// (binary.AppendVarint(-8) encodes 15, where DWARF wants 0x78).
func appendSleb(b []byte, v int64) []byte {
return binary.AppendVarint(b, v)
for {
c := byte(v & 0x7f)
v >>= 7
if (v == 0 && c&0x40 == 0) || (v == -1 && c&0x40 != 0) {
return append(b, c)
}
b = append(b, c|0x80)
}
}
// cfiArch carries the .debug_frame CIE parameters that differ per
// architecture: the DWARF register numbers of the stack pointer the initial
// CFA rule names and of the return address. The values are the ones the Go
// linker writes into its own CIE (cmd/link/internal/ld/dwarf.go uses
// Dwarfregsp and Dwarfreglr; the per-architecture constants live in
// cmd/link/internal/<arch>/l.go).
type cfiArch struct {
name string
cfaReg byte // the stack-pointer register the initial CFA rule names
raReg byte // the return-address register
}
var (
cfiAMD64 = cfiArch{"amd64", 7, 16} // RSP, RIP
cfiARM64 = cfiArch{"arm64", 31, 30} // SP (X31), LR (X30)
cfiRISCV64 = cfiArch{"riscv64", 2, 1} // X2 (sp), X1 (ra)
cfiLOONG64 = cfiArch{"loong64", 3, 1} // $r3 (sp), $r1 (ra)
)
// dwarfBuildFrameSection builds a .debug_frame section with CFI for stack
// unwinding. It emits one CIE and one FDE per function, encoding the
// CFA (Canonical Frame Address) rule changes at each stack-adjustment
// boundary recorded in FuncLayout.Spadj.
func dwarfBuildFrameSection(img *Image) []byte {
func dwarfBuildFrameSection(img *Image, cfi cfiArch, ds *dwarfSections) []byte {
var b []byte
le := binary.LittleEndian
// CIE (Common Information Entry).
cieStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
b = append(b, 3) // version (DWARF3, widely supported)
b = append(b, 0) // augmentation (empty)
b = appendUleb(b, 1) // code alignment
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
b = appendUleb(b, 16) // return address register (LR on arm64, RIP on amd64)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, 0xFFFFFFFF) // CIE marker
b = append(b, 3) // version (DWARF3, widely supported)
b = append(b, 0) // augmentation (empty)
b = appendUleb(b, 1) // code alignment
b = appendSleb(b, -8) // data alignment (-8 for 64-bit)
b = appendUleb(b, uint64(cfi.raReg)) // return address register
// Initial CFA rule: DW_CFA_def_cfa (SP, 0)
b = append(b, 0x0c) // DW_CFA_def_cfa
b = appendUleb(b, 31) // register: SP (RSP=7 on amd64, SP=31 on arm64)
b = appendUleb(b, 0) // offset: 0
b = append(b, 0) // DW_CFA_nop (padding)
b = append(b, 0x0c) // DW_CFA_def_cfa
b = appendUleb(b, uint64(cfi.cfaReg)) // the architecture's stack pointer
b = appendUleb(b, 0) // offset: 0
b = append(b, 0) // DW_CFA_nop (padding)
// Patch CIE length.
le.PutUint32(b[cieStart:], uint32(len(b)-cieStart-4))
// FDEs (Frame Description Entries) — one per function.
// FDEs (Frame Description Entries), one per function.
for _, fn := range img.Funcs {
fdeStart := len(b)
b = append(b, 0, 0, 0, 0) // length (placeholder)
b = le.AppendUint32(b, uint32(cieStart)) // CIE pointer (offset from start)
// Initial location: function offset in .text (relocated by linker).
// Initial location: function offset in .text, referenced through
// the function's symbol so the linker relocates it.
ds.frameRelocs = append(ds.frameRelocs, dwarfReloc{
off: uint64(fdeStart + 8),
name: fn.Name,
})
b = le.AppendUint64(b, uint64(fn.Offset))
// Address range: function size.
b = le.AppendUint64(b, uint64(fn.Size))
+84 -24
View File
@@ -3,6 +3,17 @@
package asm
import "encoding/binary"
// Absolute 64-bit relocation types for the DWARF address fixups, one per
// supported architecture (the numbers debug/elf carries).
const (
rX8664Abs64 = 1 // R_X86_64_64
rAARCH64Abs64 = 257 // R_AARCH64_ABS64
rRISCVAbs64 = 2 // R_RISCV_64
rLarchAbs64 = 2 // R_LARCH_64
)
// dwarfELFSections holds the laid-out DWARF sections ready for inclusion
// in an ELF file.
type dwarfELFSections struct {
@@ -11,15 +22,23 @@ type dwarfELFSections struct {
lineOff, lineSize int
lineStrOff, lineStrSize int
frameOff, frameSize int
// Relocations for .debug_info address references.
// .rela.debug_info and .rela.debug_line contents: file offsets and
// entry counts (zero count: the section is absent).
infoRelaOff, infoRelaCount int
lineRelaOff, lineRelaCount int
frameRelaOff, frameRelaCount int
// Relocations for .debug_info address references, offsets relative to
// the section start (what an r_offset in .rela.debug_info means).
infoRelocs []elfDwarfReloc
// Relocations for .debug_line address references.
// Relocations for .debug_line address references, section-relative.
lineRelocs []elfDwarfReloc
// Relocations for .debug_frame FDE initial locations, section-relative.
frameRelocs []elfDwarfReloc
}
type elfDwarfReloc struct {
off uint64
sym int // symbol index in .symtab
off uint64 // offset within the target section
sym int // symbol index in .symtab
addend int64
}
@@ -29,8 +48,9 @@ type elfDwarfReloc struct {
//
// symIdx maps function names to their .symtab indices (needed for relocations
// against .text symbols). The map uses objectName format (pkg.name); the
// DWARF code uses bare function names, so we build a reverse lookup.
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int)) *dwarfELFSections {
// DWARF code uses bare function names, so we build a reverse lookup. cfi
// carries the architecture's .debug_frame register conventions.
func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[string]int, align func(int), cfi cfiArch) *dwarfELFSections {
// Build a lookup from bare function name to symbol index.
nameToIdx := make(map[string]int, len(symIdx))
for name, idx := range symIdx {
@@ -45,7 +65,7 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
}
nameToIdx[name] = idx
}
ds := emitDWARF(img, srcFile)
ds := emitDWARF(img, srcFile, cfi)
if ds == nil || len(ds.debugAbbrev) == 0 {
return nil
}
@@ -68,15 +88,11 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
align(1)
result.lineOff = len(*out)
result.lineSize = len(ds.debugLine)
lineBase := len(*out)
*out = append(*out, ds.debugLine...)
// Patch .debug_line relocations: replace placeholder addresses with
// actual .text offsets via symbol lookup.
for _, dr := range ds.lineRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.lineRelocs = append(result.lineRelocs, elfDwarfReloc{
off: uint64(lineBase) + dr.off,
off: dr.off,
sym: idx,
addend: dr.addend,
})
@@ -87,33 +103,77 @@ func appendDWARFSections(out *[]byte, img *Image, srcFile string, symIdx map[str
align(1)
result.infoOff = len(*out)
result.infoSize = len(ds.debugInfo)
infoBase := len(*out)
*out = append(*out, ds.debugInfo...)
// .debug_frame
if len(ds.debugFrame) > 0 {
align(1)
result.frameOff = len(*out)
result.frameSize = len(ds.debugFrame)
*out = append(*out, ds.debugFrame...)
}
// Patch .debug_info relocations.
for _, dr := range ds.infoRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.infoRelocs = append(result.infoRelocs, elfDwarfReloc{
off: uint64(infoBase) + dr.off,
off: dr.off,
sym: idx,
addend: dr.addend,
})
}
}
// .debug_frame: the section header declares alignment 8, so the data is
// padded to 8, matching it.
if len(ds.debugFrame) > 0 {
align(8)
result.frameOff = len(*out)
result.frameSize = len(ds.debugFrame)
*out = append(*out, ds.debugFrame...)
for _, dr := range ds.frameRelocs {
if idx, ok := nameToIdx[dr.name]; ok {
result.frameRelocs = append(result.frameRelocs, elfDwarfReloc{
off: dr.off,
sym: idx,
addend: dr.addend,
})
}
}
}
return result
}
// appendDWARFRelas writes the .rela.debug_info and .rela.debug_line section
// bodies from the relocations appendDWARFSections recorded, with the
// architecture's absolute 64-bit relocation type, and records their file
// offsets and entry counts on dw. Called after the DWARF sections
// themselves so the r_offsets (section-relative) need no adjustment.
func appendDWARFRelas(out *[]byte, dw *dwarfELFSections, abs64 uint32, align func(int)) {
le := binary.LittleEndian
write := func(relas []elfDwarfReloc) (off, count int) {
if len(relas) == 0 {
return 0, 0
}
align(8)
off = len(*out)
for _, r := range relas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(abs64))
le.PutUint64(b[16:], uint64(r.addend))
*out = append(*out, b[:]...)
}
return off, len(relas)
}
dw.infoRelaOff, dw.infoRelaCount = write(dw.infoRelocs)
dw.lineRelaOff, dw.lineRelaCount = write(dw.lineRelocs)
dw.frameRelaOff, dw.frameRelaCount = write(dw.frameRelocs)
}
// dwarfSourceName returns the source name the DWARF sections record: the
// image's source path when the assembler captured one, "gasm.s" otherwise.
func dwarfSourceName(img *Image) string {
if img.SourcePath != "" {
return img.SourcePath
}
return "gasm.s"
}
// dwarfSectionNames returns the DWARF section names for the string table.
var dwarfSectionNames = []string{
".debug_abbrev", ".debug_info", ".debug_line", ".debug_line_str",
".debug_frame", ".rela.debug_info", ".rela.debug_line",
".rela.debug_frame",
}
+303 -12
View File
@@ -4,11 +4,313 @@
package asm
import (
"bytes"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
// encoding.
type ulebIter struct {
b []byte
i int
}
func (r *ulebIter) uleb(t *testing.T) uint64 {
t.Helper()
v, n := binary.Uvarint(r.b[r.i:])
if n <= 0 {
t.Fatalf("bad ULEB at %d", r.i)
}
r.i += n
return v
}
func (r *ulebIter) byteAt(t *testing.T) byte {
t.Helper()
if r.i >= len(r.b) {
t.Fatalf("read past end at %d", r.i)
}
c := r.b[r.i]
r.i++
return c
}
func (r *ulebIter) uint32At(t *testing.T) uint32 {
t.Helper()
v := binary.LittleEndian.Uint32(r.b[r.i:])
r.i += 4
return v
}
// sleb reads a signed LEB128, the DWARF encoding (sign-extended two's
// complement, not Go's zigzag varint).
func (r *ulebIter) sleb(t *testing.T) int64 {
t.Helper()
var v int64
var shift uint
for {
c := r.byteAt(t)
v |= int64(c&0x7f) << shift
shift += 7
if c&0x80 == 0 {
if c&0x40 != 0 {
v |= -1 << shift
}
return v
}
}
}
// dwarfAttr is one attribute/form pair of an abbreviation.
type dwarfAttr struct{ attr, form uint64 }
// dwarfAbbrev is one parsed abbreviation declaration.
type dwarfAbbrev struct {
code uint64
tag uint64
children bool
attrs []dwarfAttr
}
// parseAbbrevs walks a .debug_abbrev table: abbreviation code, tag,
// children flag, then attr/form ULEB pairs terminated by a double zero.
func parseAbbrevs(t *testing.T, b []byte) map[uint64]dwarfAbbrev {
t.Helper()
out := map[uint64]dwarfAbbrev{}
r := &ulebIter{b: b}
for {
code := r.uleb(t)
if code == 0 {
return out
}
ab := dwarfAbbrev{code: code, tag: r.uleb(t)}
ab.children = r.byteAt(t) == 1
for {
attr := r.uleb(t)
form := r.uleb(t)
if attr == 0 && form == 0 {
break
}
if attr == 0 || form == 0 {
t.Fatalf("abbrev %d: half-terminated attr/form pair (%d, %d)", code, attr, form)
}
ab.attrs = append(ab.attrs, dwarfAttr{attr, form})
}
out[code] = ab
}
}
func eqAttrs(t *testing.T, ab dwarfAbbrev, want []dwarfAttr) {
t.Helper()
if len(ab.attrs) != len(want) {
t.Fatalf("abbrev %d attrs = %v, want %v", ab.code, ab.attrs, want)
}
for i, w := range want {
if ab.attrs[i] != w {
t.Fatalf("abbrev %d attr %d = (%#x, %#x), want (%#x, %#x)", ab.code, i, ab.attrs[i].attr, ab.attrs[i].form, w.attr, w.form)
}
}
}
// TestDwarfAbbrevTable walks the abbreviation table as a consumer does and
// checks the attribute/form sets against the constants the toolchain uses
// (cmd/internal/dwarf/dwarf_defs.go). A wrong constant here renames an
// attribute (0x1b is comp_dir, not low_pc; 0x29 and 0x37 are bounds and
// count) and a wrong form desynchronises the DIE parse: 0x25 is strx1, one
// byte, where the writer emits four for a section offset.
func TestDwarfAbbrevTable(t *testing.T) {
abbrev := dwarfAbbrevTable()
if len(abbrev) == 0 {
t.Fatal("empty abbrev table")
}
// Must end with a zero byte (end of table).
if abbrev[len(abbrev)-1] != 0 {
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
}
abs := parseAbbrevs(t, abbrev)
if len(abs) != 2 {
t.Fatalf("abbreviations = %d, want 2", len(abs))
}
cu, ok := abs[1]
if !ok {
t.Fatal("missing abbreviation 1 (compile unit)")
}
if cu.tag != dwTagCompUnit || !cu.children {
t.Errorf("abbrev 1: tag %#x children %v, want compile unit with children", cu.tag, cu.children)
}
eqAttrs(t, cu, []dwarfAttr{
{dwAtLowPC, dwFormAddr},
{dwAtHighPC, dwFormData8},
{dwAtStmtList, dwFormSecOff},
{dwAtName, dwFormString},
})
sp, ok := abs[2]
if !ok {
t.Fatal("missing abbreviation 2 (subprogram)")
}
if sp.tag != dwTagSubprog || sp.children {
t.Errorf("abbrev 2: tag %#x children %v, want subprogram without children", sp.tag, sp.children)
}
eqAttrs(t, sp, []dwarfAttr{
{dwAtName, dwFormString},
{dwAtLowPC, dwFormAddr},
{dwAtHighPC, dwFormData8},
{dwAtFrameBase, dwFormExprloc},
{dwAtDeclFile, dwFormData1},
{dwAtDeclLine, dwFormData1},
{dwAtExternal, 0x0c}, // DW_FORM_flag
})
}
// TestDwarfLineHeaderV5 parses the .debug_line header under DWARF5 rules:
// the directory and file tables are format-descriptor lists, not the
// DWARF2-4 shape of null-terminated strings, and the file entry references
// the source name through .debug_line_str.
func TestDwarfLineHeaderV5(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
MOVQ AX, ret+16(FP)
RET
`
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
r := &ulebIter{b: ds.debugLine}
r.uint32At(t) // unit_length
if v := binary.LittleEndian.Uint16(ds.debugLine[4:]); v != 5 {
t.Fatalf("version = %d, want 5", v)
}
r.i = 6
r.byteAt(t) // address_size
r.byteAt(t) // segment_selector_size
r.uint32At(t) // header_length
r.byteAt(t) // minimum_instruction_length
r.byteAt(t) // maximum_ops_per_instruction
r.byteAt(t) // default_is_stmt
r.byteAt(t) // line_base
r.byteAt(t) // line_range
opcodeBase := r.byteAt(t)
for range int(opcodeBase) - 1 {
r.byteAt(t) // standard opcode lengths
}
// Directory table (DWARF5 §6.2.4).
if n := r.byteAt(t); n != 1 {
t.Fatalf("directory_entry_format_count = %d, want 1", n)
}
if lnct := r.uleb(t); lnct != dwLnctPath {
t.Errorf("directory content type = %#x, want DW_LNCT_path", lnct)
}
if form := r.uleb(t); form != dwFormLineStrp {
t.Errorf("directory form = %#x, want DW_FORM_line_strp", form)
}
if n := r.uleb(t); n != 1 {
t.Fatalf("directories_count = %d, want 1", n)
}
if off := r.uint32At(t); off != 0 {
t.Errorf("compilation directory line_strp = %d, want 0 (the empty string)", off)
}
// File table (DWARF5 §6.2.5).
if n := r.byteAt(t); n != 2 {
t.Fatalf("file_name_entry_format_count = %d, want 2", n)
}
if lnct := r.uleb(t); lnct != dwLnctPath {
t.Errorf("file content type = %#x, want DW_LNCT_path", lnct)
}
if form := r.uleb(t); form != dwFormLineStrp {
t.Errorf("file path form = %#x, want DW_FORM_line_strp", form)
}
if lnct := r.uleb(t); lnct != dwLnctDirIndex {
t.Errorf("file content type = %#x, want DW_LNCT_directory_index", lnct)
}
if form := r.uleb(t); form != dwFormUdata {
t.Errorf("file dir-index form = %#x, want DW_FORM_udata", form)
}
if n := r.uleb(t); n != 1 {
t.Fatalf("file_names_count = %d, want 1", n)
}
strOff := r.uint32At(t)
if dirIdx := r.uleb(t); dirIdx != 0 {
t.Errorf("file directory index = %d, want 0", dirIdx)
}
// The file entry's line_strp must resolve to the source name.
end := int(strOff) + len("test_amd64.s")
if int(strOff) >= len(ds.debugLineStr) || !bytes.Equal(ds.debugLineStr[strOff:end], []byte("test_amd64.s")) {
t.Errorf("file entry line_strp %d does not name the source: %q", strOff, ds.debugLineStr)
}
// The fixed header fields: address_size 8 and a header_length that
// points just past the file table (the patch site is offset 8 in the
// v5 header, and the field counts from its own end).
if ds.debugLine[6] != 8 || ds.debugLine[7] != 0 {
t.Errorf("address_size/segment_selector = %d/%d, want 8/0", ds.debugLine[6], ds.debugLine[7])
}
if hl := binary.LittleEndian.Uint32(ds.debugLine[8:]); hl != uint32(r.i-12) {
t.Errorf("header_length = %d, want %d (the byte after the file table is %d)", hl, r.i-12, r.i)
}
}
// TestDwarfFrameCIEArch checks the shared CIE carries each architecture's
// stack-pointer and return-address registers: the values the Go linker
// writes (cmd/link/internal/<arch>/l.go dwarfRegSP/dwarfRegLR).
func TestDwarfFrameCIEArch(t *testing.T) {
for _, tc := range []struct {
name string
cfi cfiArch
}{
{"amd64", cfiAMD64},
{"arm64", cfiARM64},
{"riscv64", cfiRISCV64},
{"loong64", cfiLOONG64},
} {
frame := dwarfBuildFrameSection(&Image{}, tc.cfi, &dwarfSections{})
r := &ulebIter{b: frame}
r.uint32At(t) // length
if cid := r.uint32At(t); cid != 0xFFFFFFFF {
t.Errorf("%s: CIE id = %#x, want 0xffffffff", tc.name, cid)
}
if v := r.byteAt(t); v != 3 {
t.Errorf("%s: CIE version = %d, want 3", tc.name, v)
}
if aug := r.byteAt(t); aug != 0 {
t.Errorf("%s: CIE augmentation = %d, want 0", tc.name, aug)
}
if ca := r.uleb(t); ca != 1 {
t.Errorf("%s: code alignment = %d, want 1", tc.name, ca)
}
if da := r.sleb(t); da != -8 {
t.Errorf("%s: data alignment = %d, want -8 (signed LEB128, not zigzag)", tc.name, da)
}
if ra := r.uleb(t); ra != uint64(tc.cfi.raReg) {
t.Errorf("%s: return-address register = %d, want %d", tc.name, ra, tc.cfi.raReg)
}
if op := r.byteAt(t); op != 0x0c {
t.Errorf("%s: expected DW_CFA_def_cfa, got opcode %#x", tc.name, op)
}
if cfa := r.uleb(t); cfa != uint64(tc.cfi.cfaReg) {
t.Errorf("%s: CFA register = %d, want %d", tc.name, cfa, tc.cfi.cfaReg)
}
if off := r.uleb(t); off != 0 {
t.Errorf("%s: CFA offset = %d, want 0", tc.name, off)
}
}
}
func TestEmitDWARF(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
@@ -27,7 +329,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s")
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
// .debug_abbrev must not be empty and must start with abbrev code 1.
if len(ds.debugAbbrev) == 0 {
@@ -68,14 +370,3 @@ TEXT ·add(SB), NOSPLIT, $0-24
t.Fatal("no .debug_info relocations")
}
}
func TestDwarfAbbrevTable(t *testing.T) {
abbrev := dwarfAbbrevTable()
if len(abbrev) == 0 {
t.Fatal("empty abbrev table")
}
// Must end with a zero byte (end of table).
if abbrev[len(abbrev)-1] != 0 {
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
}
}
+551 -1
View File
@@ -12,6 +12,7 @@ import (
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
@@ -52,7 +53,7 @@ func elfTestImage(t *testing.T) *Image {
}
// TestAssembleFileExternals checks that a reference to a symbol no GLOBL
// defines is recorded as an external relocation instead of failing — the
// defines is recorded as an external relocation instead of failing; the
// raw image leaves the displacement zero, the object emitters carry it.
func TestAssembleFileExternals(t *testing.T) {
img := elfTestImage(t)
@@ -211,6 +212,75 @@ func TestELFObject(t *testing.T) {
}
}
// TestELFObjectTLSGuardReloc checks that a non-NOSPLIT function's stack
// guard carries an R_X86_64_TPOFF32 relocation against the null symbol in
// .rela.text. The serialisation must honour the record's type field: a
// hardcoded R_X86_64_PC32 mislinks the TLS load as an ordinary
// PC-relative reference.
func TestELFObjectTLSGuardReloc(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", `
#include "textflag.h"
TEXT ·grow(SB), $0
CALL ·other(SB)
RET
TEXT ·other(SB), NOSPLIT, $0
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
var haveTLS bool
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
if r.Kind == RelTLSLE {
haveTLS = true
}
}
}
if !haveTLS {
t.Fatal("test source produced no RelTLSLE relocation")
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
found := false
for i := 0; i+24 <= len(raw); i += 24 {
e := raw[i:]
info := binary.LittleEndian.Uint64(e[8:])
typ := info & 0xffffffff
sym := int(info >> 32)
if typ == uint64(elf.R_X86_64_TPOFF32) {
found = true
if sym != 0 {
t.Errorf("TPOFF32 relocation against symbol %d, want 0 (the null symbol)", sym)
}
}
}
if !found {
t.Errorf("no R_X86_64_TPOFF32 relocation in .rela.text (%d bytes)", len(raw))
}
}
// TestELFObjectNoRelocations checks a file with no static-symbol references
// emits a valid object without a .rela.text section.
func TestELFObjectNoRelocations(t *testing.T) {
@@ -253,6 +323,238 @@ TEXT ·nop(SB), NOSPLIT, $0
}
}
// elfSectionHeaderCount returns the e_shnum the ELF header declares.
func elfSectionHeaderCount(t *testing.T, obj []byte) int {
t.Helper()
return int(binary.LittleEndian.Uint16(obj[60:]))
}
// checkELFSectionAccounting verifies the number of section headers the
// writer physically laid out equals e_shnum: every DWARF section written
// after .shstrtab must be counted, or the last ones (always .debug_frame)
// are invisible to every consumer, debug/elf included.
func checkELFSectionAccounting(t *testing.T, obj []byte) {
t.Helper()
shoff := int(binary.LittleEndian.Uint64(obj[40:]))
shentsize := int(binary.LittleEndian.Uint16(obj[58:]))
shnum := elfSectionHeaderCount(t, obj)
if shentsize != 64 {
t.Fatalf("e_shentsize = %d, want 64", shentsize)
}
if (len(obj)-shoff)%shentsize != 0 {
t.Fatalf("section header table is not a whole number of entries: shoff=%d len=%d", shoff, len(obj))
}
if present := (len(obj) - shoff) / shentsize; present != shnum {
t.Errorf("e_shnum = %d but %d section headers are laid out", shnum, present)
}
}
// TestELFDWARFSectionAccounting runs the header accounting check over all
// four architecture emitters, and additionally checks the .debug_frame
// section is visible (its data aligned as its header declares).
func TestELFDWARFSectionAccounting(t *testing.T) {
parse := func(name, src string) *ast.File {
f, errs := parser.Parse(name, src)
if len(errs) > 0 {
t.Fatalf("parse %s: %v", name, errs)
}
return f
}
cases := []struct {
name string
img *Image
emit func(*Image) ([]byte, error)
}{
{"amd64", elfTestImage(t), (*Image).ELFObject},
{"arm64", mustImage(t, func() (*Image, error) {
return AssembleFileARM64(parse("k_arm64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVD a+0(FP), R4
MOVD b+8(FP), R5
ADD R5, R4, R4
MOVD R4, ret+16(FP)
RET
`))
}), (*Image).ELFAARCH64Object},
{"riscv64", mustImage(t, func() (*Image, error) {
return AssembleFileRISCV(parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`))
}), (*Image).ELFRISCVObject},
{"loong64", mustImage(t, func() (*Image, error) {
return AssembleFileLOONG64(parse("k_loong64.s", `
#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVV a+0(FP), R4
MOVV b+8(FP), R5
ADDV R5, R4, R4
MOVV R4, ret+16(FP)
RET
`))
}), (*Image).ELFLOONG64Object},
}
for _, tc := range cases {
obj, err := tc.emit(tc.img)
if err != nil {
t.Fatalf("%s: emit: %v", tc.name, err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("%s: parse emitted object: %v", tc.name, err)
}
frame := ef.Section(".debug_frame")
if frame == nil {
t.Errorf("%s: .debug_frame invisible to debug/elf (e_shnum too small?)", tc.name)
ef.Close()
continue
}
if frame.Offset%8 != 0 || frame.Addralign != 8 {
t.Errorf("%s: .debug_frame offset %d align %d, want offset%%8==0 align 8", tc.name, frame.Offset, frame.Addralign)
}
ef.Close()
}
}
func mustImage(t *testing.T, f func() (*Image, error)) *Image {
t.Helper()
img, err := f()
if err != nil {
t.Fatal(err)
}
return img
}
// TestELFDWARFRelocations checks the .rela.debug_info and .rela.debug_line
// sections exist and carry absolute 64-bit relocations against the
// function symbols, with r_offsets inside their target sections.
func TestELFDWARFRelocations(t *testing.T) {
img := elfTestImage(t)
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
// The DWARF must record the assembled file's path (threaded through
// Image.SourcePath), not a placeholder name.
info, err := ef.Section(".debug_info").Data()
if err != nil {
t.Fatal(err)
}
if img.SourcePath != "t_amd64.s" || !bytes.Contains(info, []byte(img.SourcePath)) {
t.Errorf("DWARF compilation unit does not name the source %q", img.SourcePath)
}
for _, tc := range []struct {
rela string
target string
want uint32
}{
{".rela.debug_info", ".debug_info", rX8664Abs64},
{".rela.debug_line", ".debug_line", rX8664Abs64},
{".rela.debug_frame", ".debug_frame", rX8664Abs64},
} {
rs := ef.Section(tc.rela)
if rs == nil {
t.Fatalf("missing %s", tc.rela)
}
if rs.Type != elf.SHT_RELA {
t.Errorf("%s: type %v, want SHT_RELA", tc.rela, rs.Type)
}
target := ef.Section(tc.target)
if target == nil {
t.Fatalf("missing %s", tc.target)
}
if rs.Link == 0 || ef.Sections[rs.Info] != target {
t.Errorf("%s: link %d info %d, want the symtab and %s", tc.rela, rs.Link, rs.Info, tc.target)
}
b, err := rs.Data()
if err != nil {
t.Fatal(err)
}
// .debug_line has one address per function; .debug_info adds the
// compile unit's own low_pc.
want := len(img.Funcs)
if tc.target == ".debug_info" {
want++
}
if len(b)/24 != want {
t.Errorf("%s: %d entries, want %d", tc.rela, len(b)/24, want)
}
for i := 0; i+24 <= len(b); i += 24 {
r_offset := binary.LittleEndian.Uint64(b[i:])
info := binary.LittleEndian.Uint64(b[i+8:])
typ := uint32(info)
sym := int(info >> 32)
if typ != tc.want {
t.Errorf("%s entry %d: type %d, want R_X86_64_64 (%d)", tc.rela, i/24, typ, tc.want)
}
if r_offset >= uint64(target.Size) {
t.Errorf("%s entry %d: r_offset %d outside %s (%d bytes)", tc.rela, i/24, r_offset, tc.target, target.Size)
}
if sym == 0 {
t.Errorf("%s entry %d: against the null symbol", tc.rela, i/24)
}
}
}
}
// TestELFDataOnly checks a source with GLOBL data and no TEXT emits a valid
// ELF object: the DWARF compilation unit of a code-less image has no
// function to relocate against and must not reach for one.
func TestELFDataOnly(t *testing.T) {
f, errs := parser.Parse("d0_amd64.s", `
GLOBL table<>(SB), RODATA, $8
DATA table<>+0(SB)/8, $12345
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
found := false
for _, s := range syms {
if s.Name == "table" && s.Size == 8 {
found = true
}
}
if !found {
t.Errorf("data symbol table missing: %v", syms)
}
if ef.Section(".rela.debug_info") != nil || ef.Section(".rela.debug_line") != nil {
t.Error("data-only image must not emit DWARF address relocations")
}
}
// TestELFLinkAndRun is the end-to-end check: assemble the test functions,
// link the emitted object with a C driver that defines the external symbol,
// and run the result. Skipped when no C compiler is available.
@@ -307,4 +609,252 @@ int main(void) {
if got := string(run); got != "42 42 7\n" {
t.Errorf("output %q, want \"42 42 7\\n\"", got)
}
// The DWARF addresses must have resolved at link time: the .debug_info
// placeholders were carried by .rela.debug_info, so every subprogram's
// low_pc must now equal its linked symbol address.
bin, err := os.ReadFile(appPath)
if err != nil {
t.Fatal(err)
}
lef, err := elf.NewFile(bytes.NewReader(bin))
if err != nil {
t.Fatalf("parse linked binary: %v", err)
}
defer lef.Close()
syms, err := lef.Symbols()
if err != nil {
t.Fatal(err)
}
addrByName := map[string]uint64{}
for _, s := range syms {
if elf.ST_TYPE(s.Info) == elf.STT_FUNC && s.Value != 0 {
addrByName[s.Name] = s.Value
}
}
lowPCs := dwarfSubprogramLowPCs(t, lef)
if len(lowPCs) == 0 {
t.Fatal("no subprogram DW_AT_low_pc parsed from the linked binary")
}
for name, pc := range lowPCs {
addr, ok := addrByName[name]
if !ok {
t.Errorf("subprogram %q not in the linked symbol table", name)
continue
}
if pc != addr {
t.Errorf("subprogram %q: DW_AT_low_pc = %#x, linked address %#x (DWARF relocation unresolved)", name, pc, addr)
}
}
}
// dwarfSubprogramLowPCs walks the linked binary's .debug_info with its own
// .debug_abbrev and returns each DW_TAG_subprogram's DW_AT_low_pc by name.
func dwarfSubprogramLowPCs(t *testing.T, ef *elf.File) map[string]uint64 {
t.Helper()
abbrevSec := ef.Section(".debug_abbrev")
infoSec := ef.Section(".debug_info")
if abbrevSec == nil || infoSec == nil {
t.Fatal("linked binary lacks .debug_abbrev or .debug_info")
}
abbrev, err := abbrevSec.Data()
if err != nil {
t.Fatal(err)
}
info, err := infoSec.Data()
if err != nil {
t.Fatal(err)
}
abs := parseAbbrevs(t, abbrev)
le := binary.LittleEndian
out := map[string]uint64{}
r := &ulebIter{b: info}
r.uint32At(t) // unit_length
if v := le.Uint16(info[4:]); v != 5 {
t.Fatalf(".debug_info version %d, want 5", v)
}
r.i = 6
r.byteAt(t) // unit_type
r.byteAt(t) // address_size
r.uint32At(t) // debug_abbrev_offset
var name string
var lowPC uint64
for r.i < len(r.b) {
code := r.uleb(t)
if code == 0 {
continue // end of the CU's children
}
ab, ok := abs[code]
if !ok {
t.Fatalf("unknown abbreviation code %d", code)
}
name, lowPC = "", 0
for _, a := range ab.attrs {
switch a.attr {
case dwAtName:
readFormKeep(t, r, a.form, &name, nil)
case dwAtLowPC:
readFormKeep(t, r, a.form, nil, &lowPC)
default:
readFormSkip(t, r, a.form)
}
}
if ab.tag == dwTagSubprog && name != "" {
out[name] = lowPC
}
}
return out
}
// readFormKeep reads one DIE attribute value, keeping a string or an
// address into the pointer it was given (nil keeps nothing).
func readFormKeep(t *testing.T, r *ulebIter, form uint64, name *string, addr *uint64) {
t.Helper()
switch form {
case dwFormString:
end := r.i
for end < len(r.b) && r.b[end] != 0 {
end++
}
if name != nil {
*name = string(r.b[r.i:end])
}
r.i = end + 1
case dwFormAddr:
if addr != nil {
*addr = binary.LittleEndian.Uint64(r.b[r.i:])
}
r.i += 8
default:
readFormSkip(t, r, form)
}
}
func readFormSkip(t *testing.T, r *ulebIter, form uint64) {
t.Helper()
switch form {
case dwFormString:
for r.i < len(r.b) && r.b[r.i] != 0 {
r.i++
}
r.i++
case dwFormAddr, dwFormData8:
r.i += 8
case dwFormSecOff:
r.i += 4
case dwFormExprloc:
r.i += int(r.uleb(t))
case dwFormData1, 0x0c:
r.i++
default:
t.Fatalf("unsupported form %#x", form)
}
}
// TestELFObjectDataRelocation checks that a symbol-valued DATA field ("DATA
// s+0(SB)/8, $other(SB)") reaches the ELF object as a .rela.data entry: an
// absolute 64-bit relocation at the field's offset within .data, against
// the named symbol, external targets included.
func TestELFObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_amd64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-8
RET
GLOBL holder(SB), NOPTR, $24
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
obj, err := img.ELFObject()
if err != nil {
t.Fatalf("ELFObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_X86_64_64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_X86_64_64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_X86_64_64), addend: 0, target: "extvar"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+147 -29
View File
@@ -15,14 +15,19 @@ const (
// AArch64 relocation types (the ELF psABI).
rArm64PrelPgHi21 = 275 // R_AARCH64_ADR_PREL_PG_HI21 (ADRP page)
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD/STR/LDR page offset)
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
// R_AARCH64_ABS32 (debug/elf 258): the absolute 32-bit address of a
// symbol, the R_ADDR shape a 4-byte DATA field carries. ABS64 (257)
// lives with the DWARF fixup constants as rAARCH64Abs64.
rArm64Abs32 = 258
)
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
// optional .rela.text and an optional .rela.data.
func (img *Image) ELFAARCH64Object() ([]byte, error) {
le := binary.LittleEndian
@@ -80,8 +85,19 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
}
// Build relocations. Each SB reference is an ADRP pair:
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21
// ADD/LDR/STR → R_AARCH64_ADD_ABS_LO12_NC
// ADRP Rd, 0 → R_AARCH64_ADR_PREL_PG_HI21 at the ADRP
// ADD → R_AARCH64_ADD_ABS_LO12_NC at the ADD word
// LDR/STR X → R_AARCH64_LDST64_ABS_LO12_NC at the LDR/STR word
// BL → R_AARCH64_CALL26
// cmd/link's own conversion emits the HI21 at sectoff and the LO12 at
// sectoff+4 (cmd/link/internal/arm64/asm.go), so the ADD or load word
// carries the page-offset relocation, never a second HI21. The
// assembler records two RelArm64Addr relocs per ADRP+ADD pair (one per
// word), so the second of the pair is consumed here.
// Addends stay raw: ADR_PREL_PG_HI21 and the ABS_LO12_NC forms resolve
// against S+A, and CALL26 branches take the branch instruction's own
// place as the PC-relative base, so subtracting the field width (the
// amd64 R_PCREL convention) would misplace every branch by 4 bytes.
type elfRela struct {
off uint64
typ uint32
@@ -90,29 +106,81 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
}
var relas []elfRela
for _, fn := range img.Funcs {
for _, r := range fn.Relocs {
for i := 0; i < len(fn.Relocs); i++ {
r := fn.Relocs[i]
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch {
case r.Kind == RelArm64Branch:
typ = rArm64Call26
case r.Kind == RelArm64Addr && r.Off%4 == 4:
typ = rArm64AddAbsLo12NC
switch r.Kind {
case RelArm64Branch:
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off), typ: rArm64Call26, sym: idx, addend: r.Addend,
})
case RelArm64Addr:
// ADRP+ADD: the pair's second reloc (at Off+4) is the
// assembler's twin of the same pair; skip it.
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64AddAbsLo12NC, sym: idx, addend: r.Addend},
)
i++
case RelArm64LDST64:
// ADRP+LDR/STR: one assembler reloc covers the pair.
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64Ldst64Lo12NC, sym: idx, addend: r.Addend},
)
default:
typ = rArm64PrelPgHi21
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
}
}
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rAARCH64Abs64
case 4:
typ = rArm64Abs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
@@ -122,18 +190,13 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
@@ -168,7 +231,7 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
var relaOff, relaDataOff int
if hasRela {
align(8)
relaOff = len(out)
@@ -180,19 +243,47 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
out = append(out, b[:]...)
}
}
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
// DWARF debug sections.
// DWARF debug sections; the address placeholders they leave are carried
// as .rela.debug_info/.rela.debug_line entries the system linker applies.
dwAlign := func(n int) {
for len(out)%n != 0 {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiARM64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
nSections += 4
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rAARCH64Abs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
@@ -220,14 +311,41 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
+173 -1
View File
@@ -6,6 +6,7 @@ package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
@@ -28,6 +29,7 @@ TEXT ·add(SB), NOSPLIT, $0-24
TEXT ·getanswer(SB), NOSPLIT, $0-8
MOVD answer<>(SB), R4
MOVD $answer<>(SB), R5
MOVD R4, ret+0(FP)
RET
@@ -102,8 +104,63 @@ DATA answer<>+0(SB)/8, $42
// Check that .rela.text exists (getanswer has SB reference).
relaText := ef.Section(".rela.text")
if relaText == nil {
t.Error("missing .rela.text section")
t.Fatal("missing .rela.text section")
}
// The SB references of getanswer form two ADRP pairs: the load
// (MOVD answer<>(SB), R4) is ADRP+LDR carrying HI21 at the ADRP and
// LDST64_ABS_LO12_NC at the LDR word, and the address-of
// (MOVD $answer<>(SB), R5) is ADRP+ADD carrying HI21 and
// ADD_ABS_LO12_NC. cmd/link's own conversion emits exactly this
// sectoff / sectoff+4 pairing; a second HI21 at the ADD or LDR word
// corrupts the pair.
raw, err := relaText.Data()
if err != nil {
t.Fatal(err)
}
if len(raw)%24 != 0 || len(raw)/24 != 4 {
t.Fatalf(".rela.text has %d bytes, want four 24-byte entries", len(raw))
}
wantRela := []struct {
typ elf.R_AARCH64
off uint64 // relative to the getanswer function start
}{
{elf.R_AARCH64_ADR_PREL_PG_HI21, 0},
{elf.R_AARCH64_LDST64_ABS_LO12_NC, 4},
{elf.R_AARCH64_ADR_PREL_PG_HI21, 8},
{elf.R_AARCH64_ADD_ABS_LO12_NC, 12},
}
getanswer := byNameElf(t, ef, "getanswer")
for i, w := range wantRela {
e := raw[i*24 : (i+1)*24]
off := binary.LittleEndian.Uint64(e[0:])
info := binary.LittleEndian.Uint64(e[8:])
typ := elf.R_AARCH64(info & 0xffffffff)
sym := int(info >> 32)
if typ != w.typ || off != getanswer.Value+w.off {
t.Errorf("reloc %d: type %v off %d, want %v at %d", i, typ, off, w.typ, getanswer.Value+w.off)
}
if sym != 3 { // NULL, .text, .data, then the first local: answer
t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym)
}
}
}
// byNameElf returns the symbol table entry for name from the raw .symtab,
// which carries every entry including the null and section symbols in order.
func byNameElf(t *testing.T, ef *elf.File, name string) elf.Symbol {
t.Helper()
syms, err := ef.Symbols()
if err != nil {
t.Fatalf("symbols: %v", err)
}
for _, s := range syms {
if s.Name == name {
return s
}
}
t.Fatalf("symbol %q not found", name)
return elf.Symbol{}
}
// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no
@@ -140,3 +197,118 @@ TEXT ·add(SB), NOSPLIT, $0-24
t.Error("unexpected .rela.text section when there are no relocations")
}
}
// TestELFAARCH64ObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the AArch64 ELF object as a
// .rela.data entry: an R_AARCH64_ABS64 (ABS32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFAARCH64ObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_arm64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
obj, err := img.ELFAARCH64Object()
if err != nil {
t.Fatalf("ELFAARCH64Object: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_AARCH64_ABS64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_AARCH64_ABS32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+123 -16
View File
@@ -13,15 +13,26 @@ import (
const (
emLOONGARCH = 258 // EM_LOONGARCH
// EF_LOONGARCH_ABI_DOUBLE_FLOAT | EF_LOONGARCH_OBJABI_V1: the flags the
// Go toolchain writes (cmd/link/internal/ld/elf.go: Flags = 0x43 for
// Loong64). System linkers refuse to merge ET_REL objects whose float
// ABI differs, so 0 (soft-float) would make the object unlinkable.
efLarchAbiDoubleObjV1 = 0x43
// LoongArch relocation types (the ELF psABI).
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
// R_LARCH_32 (debug/elf 1): the absolute 32-bit address of a symbol,
// the R_ADDR shape a 4-byte DATA field carries. R_LARCH_64 (2) lives
// with the DWARF fixup constants as rLarchAbs64.
rLarchAbs32 = 1
)
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
// optional .rela.text.
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
// optional .rela.text and an optional .rela.data.
func (img *Image) ELFLOONG64Object() ([]byte, error) {
le := binary.LittleEndian
@@ -95,18 +106,65 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
typ := uint32(rLarchPCALAHI20)
if r.Kind == RelLoong64AddrLo {
switch r.Kind {
case RelLoong64AddrLo:
typ = rLarchPCALALO12
case RelLoong64Branch:
typ = rLarchB26
}
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off),
typ: typ,
sym: idx,
addend: r.Addend - int64(r.After-r.Off),
addend: r.Addend,
})
}
}
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rLarchAbs64
case 4:
typ = rLarchAbs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
@@ -116,18 +174,13 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
@@ -162,7 +215,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
var relaOff, relaDataOff int
if hasRela {
align(8)
relaOff = len(out)
@@ -174,6 +227,17 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
out = append(out, b[:]...)
}
}
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
@@ -183,9 +247,25 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiLOONG64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
nSections += 4
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rLarchAbs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
@@ -213,14 +293,41 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
@@ -233,7 +340,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint32(hdr[48:], efLarchAbiDoubleObjV1)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
+162
View File
@@ -55,6 +55,11 @@ DATA answer<>+0(SB)/8, $42
if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH {
t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine)
}
// The double-float ABI plus OBJABI_V1 flags the Go toolchain writes;
// system linkers refuse ABI-mismatched merges.
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efLarchAbiDoubleObjV1 {
t.Errorf("e_flags = %#x, want %#x (double-float, OBJABI_V1)", flags, efLarchAbiDoubleObjV1)
}
text := ef.Section(".text")
data := ef.Section(".data")
@@ -198,3 +203,160 @@ TEXT ·nop(SB), NOSPLIT, $0
t.Error("function symbol nop not found")
}
}
// TestELFLOONG64BranchRelocation checks that the morestack call and an
// internal CALL both carry R_LARCH_B26 in the emitted object, matching the
// Go linker's mapping of its call relocation.
func TestELFLOONG64BranchRelocation(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaSec := ef.Section(".rela.text")
if relaSec == nil {
t.Fatal("missing .rela.text")
}
raw, err := relaSec.Data()
if err != nil {
t.Fatal(err)
}
// The guard's morestack call plus the body's CALL to other.
if len(raw)%24 != 0 || len(raw)/24 != 2 {
t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw))
}
le := binary.LittleEndian
for i := range 2 {
info := le.Uint64(raw[i*24+8:])
if elf.R_LARCH(info&0xffffffff) != elf.R_LARCH_B26 {
t.Errorf("relocation %d type = %v, want R_LARCH_B26", i, elf.R_LARCH(info&0xffffffff))
}
}
}
// TestELFLOONG64ObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the LoongArch ELF object as a
// .rela.data entry: an R_LARCH_64 (R_LARCH_32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFLOONG64ObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_loong64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
obj, err := img.ELFLOONG64Object()
if err != nil {
t.Fatalf("ELFLOONG64Object: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_LARCH_64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_LARCH_64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_LARCH_64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_LARCH_32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+128 -21
View File
@@ -13,17 +13,27 @@ import (
const (
emRISCV = 243 // EM_RISCV
// EF_RISCV_FLOAT_ABI_DOUBLE: the double-precision float ABI the Go
// toolchain targets (cmd/link/internal/ld/elf.go writes Flags = 0x4 for
// RISCV64). System linkers refuse to merge ET_REL objects whose float
// ABI differs, so 0 (soft-float) would make the object unlinkable.
efRISCVFloatAbiDouble = 0x4
// RISC-V relocation types.
rRISCV32 = 1
rRISCVJAL = 17 // R_RISCV_JAL
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
// R_RISCV_32 (debug/elf 1): the absolute 32-bit address of a symbol,
// the R_ADDR shape a 4-byte DATA field carries. R_RISCV_64 (2) lives
// with the DWARF fixup constants as rRISCVAbs64.
rRISVCAbs32 = 1
)
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
// ELF emission: .text, .data, .symtab, .strtab, an optional .rela.text and
// an optional .rela.data.
func (img *Image) ELFRISCVObject() ([]byte, error) {
le := binary.LittleEndian
@@ -83,9 +93,14 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
// Build relocations. Each SB reference is an AUIPC + second-instruction
// pair carrying a single relocation kind; the ELF writer expands it into
// the R_RISCV_PCREL_HI20 + R_RISCV_PCREL_LO12_I/S pair the psABI expects.
// The HI20 carries the symbol addend; the LO12 addend is zero, matching
// cmd/link's own ELF conversion (the LO12 resolves against the HI20's
// AUIPC location).
// The HI20 carries the symbol and its addend. The LO12's symbol must
// denote the AUIPC site the HI20 relocates (psABI §8.4.9: the pair is
// resolved against the label of the AUIPC, not the target symbol;
// cmd/link generates one local text symbol per AUIPC for exactly this,
// cmd/link/internal/riscv64/asm.go). The .text section symbol with the
// AUIPC's section-relative offset as addend gives S + A = the AUIPC
// address, which is that label.
const secSymText = 1 // syms[1], the .text section symbol
type elfRela struct {
off uint64
typ uint32
@@ -99,27 +114,70 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
if !ok {
return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name)
}
auipc := int64(fn.Offset + r.Off)
switch r.Kind {
case RelRISCVPCRELIType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: idx, addend: 0},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12I, sym: secSymText, addend: auipc},
)
case RelRISCVPCRELSType:
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVPCRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: idx, addend: 0},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVPCRELLO12S, sym: secSymText, addend: auipc},
)
case RelRISCVJal:
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
case RelPCRelAbs:
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCV32, sym: idx, addend: r.Addend})
default:
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
}
}
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
// $other(SB)") become .rela.data entries: an absolute relocation of the
// DATA line's width at the field's data-section offset, S + A with no
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
// cannot hold an address, so they are refused rather than truncated.
var dataRelas []elfRela
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
idx, ok := symIdx[r.Name]
if !ok {
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
}
var typ uint32
switch r.Siz {
case 8:
typ = rRISCVAbs64
case 4:
typ = rRISVCAbs32
default:
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
}
dataRelas = append(dataRelas, elfRela{
off: uint64(d.Offset + r.Off),
sym: idx,
typ: typ,
addend: r.Addend,
})
}
}
// Section presence: .rela.text only when there are code relocations,
// .rela.data only when a DATA line holds a symbol value.
hasRela := len(relas) > 0
hasDataRela := len(dataRelas) > 0
nSections := 6
if hasRela {
nSections++
}
if hasDataRela {
nSections++
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// String tables.
stNames := newElfStrtab()
for _, s := range syms {
@@ -129,18 +187,13 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
stSections.add(n)
}
if hasDataRela {
stSections.add(".rela.data")
}
for _, n := range dwarfSectionNames {
stSections.add(n)
}
hasRela := len(relas) > 0
nSections := 6
if hasRela {
nSections = 7
}
secSymtab, secStrtab := 3, 4
secShstr := nSections - 1
// Layout.
var out []byte
out = append(out, make([]byte, 64)...)
@@ -175,7 +228,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
strtabOff := len(out)
out = append(out, stNames.bytes()...)
var relaOff int
var relaOff, relaDataOff int
if hasRela {
align(8)
relaOff = len(out)
@@ -187,6 +240,17 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
out = append(out, b[:]...)
}
}
if hasDataRela {
align(8)
relaDataOff = len(out)
for _, r := range dataRelas {
var b [24]byte
le.PutUint64(b[0:], r.off)
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
le.PutUint64(b[16:], uint64(r.addend))
out = append(out, b[:]...)
}
}
shstrOff := len(out)
out = append(out, stSections.bytes()...)
@@ -196,9 +260,25 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
out = append(out, 0)
}
}
dw := appendDWARFSections(&out, img, "gasm.s", symIdx, dwAlign)
dw := appendDWARFSections(&out, img, dwarfSourceName(img), symIdx, dwAlign, cfiRISCV64)
dwarfStart := 0 // section index of .debug_abbrev, set when DWARF is present
if dw != nil {
nSections += 4
// Five DWARF sections: .debug_abbrev, .debug_info, .debug_line,
// .debug_line_str and .debug_frame (the CIE is unconditional, so
// the frame section is always present), plus the relocation
// sections below when they carry entries.
dwarfStart = nSections
nSections += 5
appendDWARFRelas(&out, dw, rRISCVAbs64, dwAlign)
if dw.infoRelaCount > 0 {
nSections++
}
if dw.lineRelaCount > 0 {
nSections++
}
if dw.frameRelaCount > 0 {
nSections++
}
}
align(8)
@@ -226,14 +306,41 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
if hasRela {
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
}
if hasDataRela {
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
}
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
// DWARF section headers; their indices follow the write order.
if dw != nil {
// secIdx is a running section index: each putSh below emits the
// next header, and the sh_info of a .rela section names the index
// of the section it relocates.
secIdx := dwarfStart
putSh(".debug_abbrev", shtProgbits, 0, dw.abbrevOff, dw.abbrevSize, 0, 0, 1, 0)
secIdx++
putSh(".debug_info", shtProgbits, 0, dw.infoOff, dw.infoSize, 0, 0, 1, 0)
secInfoIdx := secIdx
secIdx++
if dw.infoRelaCount > 0 {
putSh(".rela.debug_info", shtRela, 0, dw.infoRelaOff, 24*dw.infoRelaCount, secSymtab, secInfoIdx, 8, 24)
secIdx++
}
putSh(".debug_line", shtProgbits, 0, dw.lineOff, dw.lineSize, 0, 0, 1, 0)
secLineIdx := secIdx
secIdx++
if dw.lineRelaCount > 0 {
putSh(".rela.debug_line", shtRela, 0, dw.lineRelaOff, 24*dw.lineRelaCount, secSymtab, secLineIdx, 8, 24)
secIdx++
}
putSh(".debug_line_str", shtProgbits, 0, dw.lineStrOff, dw.lineStrSize, 0, 0, 1, 0)
secIdx++
if dw.frameSize > 0 {
putSh(".debug_frame", shtProgbits, 0, dw.frameOff, dw.frameSize, 0, 0, 8, 0)
secFrameIdx := secIdx
secIdx++
if dw.frameRelaCount > 0 {
putSh(".rela.debug_frame", shtRela, 0, dw.frameRelaOff, 24*dw.frameRelaCount, secSymtab, secFrameIdx, 8, 24)
}
}
}
@@ -246,7 +353,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
le.PutUint64(hdr[24:], 0)
le.PutUint64(hdr[32:], 0)
le.PutUint64(hdr[40:], uint64(shoff))
le.PutUint32(hdr[48:], 0)
le.PutUint32(hdr[48:], efRISCVFloatAbiDouble)
le.PutUint16(hdr[52:], 64)
le.PutUint16(hdr[54:], 0)
le.PutUint16(hdr[56:], 0)
+128
View File
@@ -0,0 +1,128 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"debug/elf"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
// ("DATA s+0(SB)/8, $other(SB)") reaches the RISC-V ELF object as a
// .rela.data entry: an R_RISCV_64 (R_RISCV_32 for a width-4 field) at the
// field's offset within .data, against the named symbol, external targets
// included.
func TestELFRISCVObjectDataRelocation(t *testing.T) {
f, errs := parser.Parse("t_riscv64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-0
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep+5(SB)
DATA holder+8(SB)/8, $holder(SB)
DATA holder+16(SB)/8, $extvar(SB)
DATA holder+24(SB)/4, $Keep(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
checkELFSectionAccounting(t, obj)
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse emitted object: %v", err)
}
defer ef.Close()
relaData := ef.Section(".rela.data")
if relaData == nil {
t.Fatal("missing .rela.data section")
}
if relaData.Type != elf.SHT_RELA {
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
}
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
}
if ef.Sections[relaData.Info].Name != ".data" {
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
}
relas, err := relaData.Data()
if err != nil {
t.Fatal(err)
}
var got []struct {
off uint64
sym uint32
typ uint32
addend int64
}
for i := 0; i+24 <= len(relas); i += 24 {
got = append(got, struct {
off uint64
sym uint32
typ uint32
addend int64
}{
off: binary.LittleEndian.Uint64(relas[i:]),
// r_info packs the type in the low dword and the symbol index
// in the high dword.
typ: binary.LittleEndian.Uint32(relas[i+8:]),
sym: binary.LittleEndian.Uint32(relas[i+12:]),
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
})
}
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
syms, err := ef.Symbols()
if err != nil {
t.Fatal(err)
}
name := func(idx uint32) string {
if idx >= 1 && int(idx) <= len(syms) {
return syms[idx-1].Name
}
return ""
}
// The offsets are data-section-relative: the field's DATA offset plus
// the symbol's position in .data (the layout aligns each symbol to 16).
base := uint64(0)
for _, d := range img.DataSyms {
if d.Name == "holder" {
base = uint64(d.Offset)
}
}
want := []struct {
off uint64
typ uint32
addend int64
target string
}{
{off: base + 0, typ: uint32(elf.R_RISCV_64), addend: 5, target: "Keep"},
{off: base + 8, typ: uint32(elf.R_RISCV_64), addend: 0, target: "holder"},
{off: base + 16, typ: uint32(elf.R_RISCV_64), addend: 0, target: "extvar"},
{off: base + 24, typ: uint32(elf.R_RISCV_32), addend: 0, target: "Keep"},
}
if len(got) != len(want) {
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
}
for i, w := range want {
g := got[i]
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
}
if n := name(g.sym); n != w.target {
t.Errorf("entry %d names %q, want %q", i, n, w.target)
}
}
}
+50 -11
View File
@@ -18,7 +18,14 @@ func Encodable(mnemonic string) bool {
// Fixed-name instructions (no size suffix).
switch upper {
case "RET", "NOP", "CALL", "JMP":
case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The literal-data pseudo-ops, the accepted-and-ignored END and
// bookkeeping statements, and the SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
return true
}
if _, ok := noOperandTable[upper]; ok {
return true
}
if _, ok := condCode(upper); ok {
@@ -31,15 +38,20 @@ func Encodable(mnemonic string) bool {
return false
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" {
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return true
}
// CMOV carries size then condition (CMOVLGT); SET carries the condition
// alone (SETNE).
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
// so a spelling like CMOVBGT is not reported encodable when Encode
// would reject it.
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
if _, ok := jccMap[rest[1:]]; ok {
return true
switch rest[0] {
case 'W', 'L', 'Q':
if _, ok := jccMap[rest[1:]]; ok {
return true
}
}
}
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
@@ -48,13 +60,28 @@ func Encodable(mnemonic string) bool {
}
}
// Legacy SSE shuffles and packed binaries dispatch on the full name.
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
// do the imm8-controlled instructions, the lane extracts and inserts and
// the packed integer shifts (their trailing width letters belong to the
// mnemonic).
if _, ok := sseShufTable[upper]; ok {
return true
}
if _, ok := sseBinTable[upper]; ok {
return true
}
if _, ok := sseImm3Table[upper]; ok {
return true
}
if _, ok := sseExtractTable[upper]; ok {
return true
}
if _, ok := sseInsertTable[upper]; ok {
return true
}
if _, ok := sseShiftImm[upper]; ok {
return true
}
// The size-suffix split: retry the tables and the scalar switch on the
// base.
@@ -69,20 +96,32 @@ func Encodable(mnemonic string) bool {
}
}
switch base2 {
case "MOV",
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
case "MOV", "MOVD",
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
"TEST",
"LEA",
"INC", "DEC", "NEG", "NOT",
"SHL", "SHR", "SAR",
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
"BT", "BTS", "BTR", "BTC",
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
"MOVS", "STOS",
"IMUL", "IMUL3",
"PUSH", "POP",
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
"BSWAP",
"PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2",
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
"CVTSL2SD", "CVTSQ2SD",
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
"FMOVD",
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return true
}
// Full-name dispatches the size split would eat (a trailing width
// letter that is part of the mnemonic).
switch upper {
case "PMOVMSKB":
return true
}
return false
+417 -16
View File
@@ -5,6 +5,8 @@ package asm
import (
"fmt"
"math"
"strconv"
"strings"
)
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
type enc struct {
out []byte
patches []encPatch // disp32 fields awaiting static-symbol resolution
// FloatPool collects the pooled constants the floating-point
// immediates reference, in first-use order.
floatPool []floatPoolEntry
floatPoolSeen map[string]bool
}
// floatPoolEntry is one pooled floating-point constant: the symbol name
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
type floatPoolEntry struct {
name string
data []byte
}
// addFloatPool records a pooled constant, deduplicated by symbol name.
func (e *enc) addFloatPool(name string, bits uint64, width int) {
if e.floatPoolSeen == nil {
e.floatPoolSeen = map[string]bool{}
}
if e.floatPoolSeen[name] {
return
}
e.floatPoolSeen[name] = true
data := make([]byte, width)
for i := range width {
data[i] = byte(bits >> (8 * i))
}
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
}
// floatPoolList returns the pooled constants in first-use order.
func (e *enc) floatPoolList() []floatPoolEntry {
return e.floatPool
}
// encPatch marks a 4-byte displacement field in enc.out that must receive the
@@ -29,6 +64,7 @@ type encPatch struct {
off int
name string
addend int64
tls bool // a TLS slot offset: the patch is R_TLSLE with no symbol
}
func (e *enc) encode(mnem string, ops []Operand) error {
@@ -40,14 +76,74 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeRet()
case upper == "NOP":
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
case upper == "CALL":
return e.encodeJmpRel(ops, []byte{0xE8})
case upper == "JMP":
return e.encodeJmpRel(ops, []byte{0xE9})
case upper == "CALL" || upper == "JMP":
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
// Anything else is a rel32 against a label resolved by the assembler.
if len(ops) == 1 {
switch ops[0].(type) {
case Reg, Mem:
return e.encodeIndirectBranch(upper, ops)
}
}
opcode := []byte{0xE8}
if upper == "JMP" {
opcode = []byte{0xE9}
}
return e.encodeJmpRel(ops, opcode)
}
if cc, ok := condCode(upper); ok {
return e.encodeJcc(cc, ops)
}
// No-operand system and string-control instructions (CPUID, RDTSC,
// SYSCALL, the fences, UNDEF, …).
if op, ok := noOperandTable[upper]; ok {
if len(ops) != 0 {
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
}
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper {
case "POPFQ":
if len(ops) != 0 {
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
}
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
case "PUSHFQ":
if len(ops) != 0 {
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
}
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT":
return e.encodeInt(ops)
case "LDMXCSR":
return e.encodeMxcsr(2, ops)
case "STMXCSR":
return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm).
case "CMPSD":
return e.encodeCmpsd(ops)
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2":
return e.encodeSha256rnds2(ops)
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
// the SUBQ and ADDQ forms.
case "BYTE", "WORD", "LONG", "QUAD":
return e.encodeData(upper, ops)
case "END":
return e.encodeEnd(ops)
case "ADJSP":
return e.encodeAdjsp(ops)
// The runtime's bookkeeping statements carry no text bytes: go tool asm
// records FUNCDATA and PCDATA in the program list only, so the encoded
// body shows nothing, on every architecture.
case "FUNCDATA", "PCDATA":
return e.encodeFuncdata(upper, ops)
}
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
@@ -57,7 +153,9 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if err != nil {
return err
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
isEvexPrefGather(base) ||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return e.encodeVec(base, ops, sfx)
}
if sfx.any() {
@@ -84,43 +182,100 @@ func (e *enc) encode(mnem string, ops []Operand) error {
}
// Legacy SSE packed binaries dispatch on the full name: the packed
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
// which the size split must not eat.
// which the size split must not eat. A floating-point immediate
// rewrites into a pooled-constant read on the scalar members.
if m, ok := sseBinTable[upper]; ok {
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatBin(upper, m, f, ops)
}
return e.encodeSSEBin(m, ops)
}
if m, ok := sseBinTable[base]; ok {
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatBin(upper, m, f, ops)
}
return e.encodeSSEBin(m, ops)
}
// The imm8-controlled legacy instructions, the lane extracts and inserts
// and the packed integer shifts all dispatch on the full name: a trailing
// width letter here belongs to the mnemonic, not to the size split.
if m, ok := sseImm3Table[upper]; ok {
return e.encodeSSEImm3(m, ops)
}
if m, ok := sseExtractTable[upper]; ok {
return e.encodeSSEExtract(m, ops)
}
if m, ok := sseInsertTable[upper]; ok {
return e.encodeSSEInsert(m, ops)
}
if _, ok := sseShiftImm[upper]; ok {
return e.encodeSSEShift(upper, ops)
}
// PMOVMSKB ends in a width letter the size split would eat, so it
// dispatches on the full name like the packed binaries above.
if upper == "PMOVMSKB" {
return e.encodePmovmskb(upper, ops)
}
switch base {
case "MOV":
return e.encodeMov(ops, size)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
// REX.W and all.
case "MOVD":
return e.encodeMov(ops, 8)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
return e.encodeALU(aluOp[base], ops, size)
case "TEST":
return e.encodeTest(ops, size)
case "LEA":
return e.encodeLea(ops, size)
case "INC", "DEC", "NEG", "NOT":
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
return e.encodeUnary(unaryOp[base], ops, size)
case "SHL", "SHR", "SAR":
return e.encodeShift(shiftOp[base], ops, size)
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
return e.encodeShift(base, ops, size)
case "BT", "BTS", "BTR", "BTC":
return e.encodeBitTest(base, ops, size)
case "XCHG":
return e.encodeExchange(ops, size)
case "CMPXCHG":
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
case "XADD":
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
case "CRC32":
return e.encodeCrc32(ops, size)
case "ADCX":
return e.encodeCarryExt(0x66, ops, size)
case "ADOX":
return e.encodeCarryExt(0xF3, ops, size)
case "MOVS", "STOS":
return e.encodeStringOp(base, ops, size)
case "IMUL", "IMUL3":
return e.encodeImul(ops, size)
case "PUSH":
return e.encodePushPop(ops, true)
return e.encodePushPop(ops, size, true)
case "POP":
return e.encodePushPop(ops, false)
return e.encodePushPop(ops, size, false)
case "BSF", "BSR", "LZCNT", "TZCNT", "POPCNT":
return e.encodeCount(base, ops, size)
case "BSWAP":
return e.encodeBswap(ops, size)
case "PREFETCHNTA", "PREFETCHT0", "PREFETCHT1", "PREFETCHT2":
return e.encodePrefetch(base, ops)
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX":
case "MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX":
return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD":
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
return e.encodeCvtInt(base, ops, size)
case "FMOVD":
return e.encodeFmov(ops)
case "MOVSD", "MOVSS":
if f, isFloat := floatImmOperand(ops); isFloat {
return e.encodeSSEFloatMove(upper, f, ops)
}
return e.encodeSSEMove(sseMoveTable[base], ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
return e.encodeSSEMove(sseMoveTable[base], ops)
}
return fmt.Errorf("unsupported instruction %q", mnem)
@@ -150,6 +305,214 @@ var prefetchVariant = map[string]int{
"PREFETCHT2": 3,
}
// dataWidth is the literal byte count of each data-emission pseudo-op.
var dataWidth = map[string]int{
"BYTE": 1,
"WORD": 2,
"LONG": 4,
"QUAD": 8,
}
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
// little-endian, with no opcode lookup. The value is truncated to the
// width rather than range-checked, exactly as go tool asm behaves (BYTE
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
// exactly one immediate is accepted: the toolchain rejects a list such as
// BYTE $1, $2, $3.
func (e *enc) encodeData(mnem string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("%s requires an integer immediate", mnem)
}
width := dataWidth[mnem]
out := make([]byte, width)
u := uint64(imm)
for i := range width {
out[i] = byte(u >> (8 * i))
}
e.out = append(e.out, out...)
return nil
}
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
// for either (the entries live in the object's ancillary tables, not the
// function body), and the operand shapes it takes are exactly these: an
// integer count first, then a symbol reference for FUNCDATA and an integer
// value for PCDATA. The other architectures accept-and-ignore the same
// statements; amd64 now matches.
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
if _, ok := ops[0].(Imm); !ok {
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
}
switch upper {
case "FUNCDATA":
if _, ok := ops[1].(sbMem); !ok {
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
}
case "PCDATA":
if _, ok := ops[1].(Imm); !ok {
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
}
}
return nil
}
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
// entirely: the AEND Prog is skipped when the program list is flushed, so
// the statements after an END still belong to the same function and the
// encoded body carries no trace of it, whatever operands follow the name
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
func (e *enc) encodeEnd(ops []Operand) error {
return nil
}
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
// picks (the same selection subSP and addSP make for the frame). go tool
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
// statement's effect on the SP balance is checked by the function-level
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
func (e *enc) encodeAdjsp(ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("ADJSP requires an integer immediate")
}
switch v := int(imm); {
case v > 0:
e.out = append(e.out, subSP(v)...)
case v < 0:
e.out = append(e.out, addSP(-v)...)
default:
return fmt.Errorf("ADJSP $0 has no encoding")
}
return nil
}
// --- floating-point immediates ----------------------------------------------
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
// immediate, the set go tool asm rewrites into a pooled-constant read: the
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
// reject the immediate in the toolchain and are absent here on purpose.
var sseFloatImm = map[string]bool{
"MOVSD": true, "MOVSS": true,
"ADDSD": true, "ADDSS": true,
"SUBSD": true, "SUBSS": true,
"MULSD": true, "MULSS": true,
"DIVSD": true, "DIVSS": true,
"COMISD": true, "COMISS": true,
"UCOMISD": true, "UCOMISS": true,
}
// floatImmOperand reports whether the operand list opens with a
// floating-point immediate in the two-operand spelling (imm, dst).
func floatImmOperand(ops []Operand) (FloatImm, bool) {
if len(ops) != 2 {
return FloatImm{}, false
}
f, ok := ops[0].(FloatImm)
return f, ok
}
// floatPoolValue evaluates a floating-point immediate at the width its
// mnemonic encodes and names the pool constant the toolchain synthesises:
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
// rounding of the parsed value). The name carries the IEEE-754 bits; the
// section holds them little-endian.
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
v, err := strconv.ParseFloat(f.Text, 64)
if err != nil {
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
}
if f.Neg {
v = -v
}
if strings.HasSuffix(mnem, "D") {
bits = math.Float64bits(v)
return bits, fmt.Sprintf("$f64.%016x", bits), nil
}
bits = uint64(math.Float32bits(float32(v)))
return bits, fmt.Sprintf("$f32.%08x", bits), nil
}
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
// source. A positive zero needs no memory read: the toolchain emits
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
// file-level layout or the linker resolves.
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
if !sseFloatImm[mnem] {
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
}
dst, ok := ops[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
bits, name, err := floatPoolValue(mnem, f)
if err != nil {
return err
}
e.addFloatPool(name, bits, mwidth(mnem))
if bits == 0 {
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
if err := setRM(i, dst, dst, 8); err != nil {
return err
}
return e.emit(i)
}
m := sseMoveTable[mnem]
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
// a floating-point immediate source: the constant is read from the pool into
// the instruction's r/m side (reg = destination), the rewrite go tool asm
// performs at the source level.
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
if !sseFloatImm[mnem] {
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
}
dst, ok := ops[1].(Reg)
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", mnem)
}
bits, name, err := floatPoolValue(mnem, f)
if err != nil {
return err
}
e.addFloatPool(name, bits, mwidth(mnem))
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
return err
}
return e.emit(i)
}
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
// spellings end in D, the single spellings in S.
func mwidth(mnem string) int {
if strings.HasSuffix(mnem, "D") {
return 8
}
return 4
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) {
if upper == "" {
@@ -179,7 +542,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
if ss, ok := scatterTable[upper]; ok {
return e.encodeScatter(upper, ss, ops, sfx)
}
if upper == "KMOVW" || upper == "KMOVQ" {
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
@@ -216,6 +579,7 @@ type instr struct {
disp []byte
imm []byte
sb *sbRef // static-symbol displacement in disp, awaiting resolution
tls bool // the displacement is a TLS slot offset, patched R_TLSLE
}
// sbRef records that an instruction's displacement refers to a static symbol
@@ -258,6 +622,9 @@ func (e *enc) emit(i *instr) error {
if i.sb != nil {
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
}
if i.tls {
e.patches = append(e.patches, encPatch{off: len(e.out), tls: true})
}
e.out = append(e.out, i.disp...)
e.out = append(e.out, i.imm...)
return nil
@@ -284,7 +651,7 @@ func setRM(i *instr, reg Reg, rm Operand, opSize int) error {
}
// setRMDigit fills in the ModR/M for an instruction whose reg field is an
// opcode /digit extension (0–7), which carries none of the register REX rules.
// opcode /digit extension (0-7), which carries none of the register REX rules.
func setRMDigit(i *instr, digit int, rm Operand, opSize int) error {
return setRMReg(i, digit, false, false, rm, opSize)
}
@@ -312,12 +679,30 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
i.disp = le32(0)
i.sb = &sbRef{name: r.name, addend: r.addend}
return nil
case TLSMem:
// off(TLS): the segment-prefixed absolute access, mod=00 with the
// SIB escape's disp32 absolute form. The displacement is the TLS
// slot offset, patched by the linker's TLS relocation.
i.prefix = r.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(r.Disp)
i.tls = true
return nil
case SegAbs:
// 0x30(GS): the segment override with the SIB escape's disp32
// absolute form, no relocation.
setSegAbs(i, regField, r)
return nil
default:
return fmt.Errorf("invalid r/m operand %T", rm)
}
}
func setMem(i *instr, regField int, m Mem) error {
if m.Seg != 0 {
i.prefix = m.Seg
}
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
if err != nil {
return err
@@ -330,11 +715,27 @@ func setMem(i *instr, regField int, m Mem) error {
return nil
}
// setSegAbs assembles a segment-absolute operand, 0x30(GS): the segment
// override with the mod=00 SIB escape's disp32 absolute form and no
// relocation.
func setSegAbs(i *instr, regField int, m SegAbs) {
i.prefix = m.Seg
i.modrm = 0x04 | regField<<3
i.sib = 0x25
i.disp = le32(m.Disp)
}
// memComponents computes the ModR/M byte (with the given reg field), the SIB
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
func memComponents(regField int, m Mem) (modrm, sib int, disp []byte, xBit, bBit int, err error) {
sib = -1
// A displacement wider than int32 fits no encoding form; truncating it
// would address a different location, and go tool asm reports "offset
// too large" for the same operand.
if m.Disp < -(1<<31) || m.Disp > (1<<31)-1 {
return 0, -1, nil, 0, 0, fmt.Errorf("displacement %d does not fit in 32 bits", m.Disp)
}
// RIP-relative: neither base nor index.
if !m.HasBase && !m.HasIndex {
return regField<<3 | 0x05, -1, le32(m.Disp), 0, 0, nil // mod=00, rm=101
+778 -2
View File
@@ -9,6 +9,9 @@ import (
"testing"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// decode encodes an instruction and decodes it back, returning the decoded
@@ -149,6 +152,45 @@ func TestPushPop(t *testing.T) {
checkSyntax(t, "push rbx", "PUSHQ", BX)
checkSyntax(t, "pop r12", "POPQ", Reg{idx: 12, size: 8})
checkSyntax(t, "push 0x5", "PUSHQ", Imm(5))
// The W spelling carries the 0x66 operand-size prefix, byte for byte
// with go tool asm; the L and B spellings are illegal in 64-bit mode
// there and rejected here rather than silently widened.
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"PUSHW AX", "PUSHW", []Operand{AX}, "6650"},
{"POPW AX", "POPW", []Operand{AX}, "6658"},
{"PUSHW $5", "PUSHW", []Operand{Imm(5)}, "666a05"},
{"PUSHW (AX)", "PUSHW", []Operand{Ptr(AX, 0, 2)}, "66ff30"},
{"PUSHQ AX", "PUSHQ", []Operand{AX}, "50"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"PUSHL AX", "PUSHL", []Operand{AX}},
{"PUSHL R8", "PUSHL", []Operand{Reg{idx: 8, size: 8}}},
{"POPL BX", "POPL", []Operand{BX}},
{"PUSHB AX", "PUSHB", []Operand{AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
func TestUnary(t *testing.T) {
@@ -161,10 +203,60 @@ func TestUnary(t *testing.T) {
func TestShift(t *testing.T) {
checkSyntax(t, "shl rdx, 0x2", "SHLQ", Imm(2), DX)
checkSyntax(t, "shl rdx, cl", "SHLQ", CL, DX)
checkSyntax(t, "shl rdx, cl", "SHLQ", CX, DX)
checkSyntax(t, "shl rdx, 0x1", "SHLQ", Imm(1), DX)
checkSyntax(t, "sar rcx, 0x1f", "SARQ", Imm(31), CX)
}
// TestDoubleShift pins the three-operand SHL/SHR form, which encodes as
// SHLD/SHRD: go tool asm accepts it for SHL/SHR at W/L/Q widths and rejects
// it for SAR, SAL, the rotates and the B width. The byte pins mirror the
// oracle's objdump output (48 0f a4 fe 0d for the first case, and so on).
func TestDoubleShift(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string // hex encoding
}{
{"SHLQ imm", "SHLQ", []Operand{Imm(0x0d), DI, SI}, "480fa4fe0d"},
{"SHLQ CX high regs", "SHLQ", []Operand{CX, Reg{idx: 8, size: 8}, Reg{idx: 9, size: 8}}, "4d0fa5c1"},
{"SHRQ imm", "SHRQ", []Operand{Imm(1), AX, CX}, "480facc101"},
{"SHLW imm", "SHLW", []Operand{Imm(1), AX, CX}, "660fa4c101"},
{"SHRD CL", "SHRQ", []Operand{CL, AX, CX}, "480fadc1"},
{"SHLD imm high regs", "SHLQ", []Operand{Imm(2), Reg{idx: 10, size: 8}, Reg{idx: 11, size: 8}}, "4d0fa4d302"},
{"SHRD imm max", "SHRQ", []Operand{Imm(63), Reg{idx: 9, size: 8}, Reg{idx: 15, size: 8}}, "4d0faccf3f"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
// Rejected forms: the oracle rejects every one of these.
rejected := []struct {
name string
mnem string
ops []Operand
}{
{"SARQ three operands", "SARQ", []Operand{Imm(1), AX, CX}},
{"SALQ three operands", "SALQ", []Operand{Imm(1), AX, CX}},
{"ROLQ three operands", "ROLQ", []Operand{Imm(1), AX, CX}},
{"SHLB three operands", "SHLB", []Operand{Imm(1), AL, CL}},
{"SHRQ memory source", "SHRQ", []Operand{Imm(1), Ptr(AX, 0, 8), CX}},
{"SHRQ ECX count", "SHRQ", []Operand{Reg{idx: 1, size: 4}, AX, CX}},
}
for _, c := range rejected {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: Encode succeeded, want rejection", c.name)
}
}
}
func TestImul(t *testing.T) {
checkSyntax(t, "imul rdx, rcx", "IMULQ", CX, DX)
checkSyntax(t, "imul edx, edx, 0x3", "IMULL", Imm(3), DX, DX)
@@ -181,6 +273,39 @@ func TestControl(t *testing.T) {
checkOp(t, x86asm.JBE, "JLS", Imm(0))
}
// TestIndirectControlFlow pins the indirect JMP/CALL forms: FF /4 for JMP and
// FF /2 for CALL through a register or memory. A REX appears only for the
// extended registers, never REX.W: the branch operand size is fixed at 64
// bits in long mode.
func TestIndirectControlFlow(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"JMP AX", "JMP", []Operand{AX}, "ffe0"},
{"CALL AX", "CALL", []Operand{AX}, "ffd0"},
{"JMP (BX)", "JMP", []Operand{Ptr(BX, 0, 8)}, "ff23"},
{"CALL (BX)", "CALL", []Operand{Ptr(BX, 0, 8)}, "ff13"},
{"JMP 8(BX)", "JMP", []Operand{Ptr(BX, 8, 8)}, "ff6308"},
{"CALL -16(BX)", "CALL", []Operand{Ptr(BX, -16, 8)}, "ff53f0"},
{"JMP R8", "JMP", []Operand{Reg{idx: 8, size: 2}}, "41ffe0"},
{"CALL R9", "CALL", []Operand{Reg{idx: 9, size: 2}}, "41ffd1"},
{"JMP R15", "JMP", []Operand{Reg{idx: 15, size: 2}}, "41ffe7"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestSSEMoveGroundTruth checks the legacy (non-VEX) SSE moves byte for byte
// against the Go assembler. wantOp is the decoder's name, which differs from
// the Plan 9 spelling for the octa moves (MOVOU = MOVDQU, MOVO = MOVDQA).
@@ -205,6 +330,12 @@ func TestSSEMoveGroundTruth(t *testing.T) {
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
// Static-symbol (SB) references: the GOROOT crypto kernels load and
// store octa constants by name (MOVOU bswapMask<>+0(SB), X0).
{"MOVOU sym,X0", "MOVOU", []Operand{sbMem{size: 16, name: "bswapMask"}, vreg(t, "X0")}, "f30f6f0500000000", "MOVDQU"},
{"MOVOU X0,sym+8", "MOVOU", []Operand{vreg(t, "X0"), sbMem{size: 16, name: "bswapMask", addend: 8}}, "f30f7f0500000000", "MOVDQU"},
{"MOVO sym,X1", "MOVO", []Operand{sbMem{size: 16, name: "gcmPoly"}, vreg(t, "X1")}, "660f6f0d00000000", "MOVDQA"},
{"MOVO X2,sym", "MOVO", []Operand{vreg(t, "X2"), sbMem{size: 16, name: "gcmPoly"}}, "660f7f1500000000", "MOVDQA"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
@@ -230,7 +361,7 @@ func TestSSEMoveGroundTruth(t *testing.T) {
// TestGoFlacScalarTail encodes the scalar tail of an analyze kernel to confirm
// the encoder handles a realistic instruction sequence.
func TestGoFlacScalarTail(t *testing.T) {
// MOVQ swin_base+0(FP), SI — modelled as MOVQ disp(reg), reg.
// MOVQ swin_base+0(FP), SI; modelled as MOVQ disp(reg), reg.
checkSyntax(t, "mov rsi, qword ptr [rax+0x10]", "MOVQ", Ptr(AX, 0x10, 8), SI)
checkSyntax(t, "lea r9, ptr [rsi+4*rbx]", "LEAQ", Idx(SI, BX, 4, 0, 8), Reg{idx: 9, size: 8})
checkSyntax(t, "and r10, -0x8", "ANDQ", Imm(-8), Reg{idx: 10, size: 8})
@@ -285,6 +416,18 @@ func TestScalarGroundTruth(t *testing.T) {
{"MOVBQZX AL,R8", "MOVBQZX", []Operand{AL, r8}, "4c0fb6c0", "MOVZX"},
{"MOVWLZX AX,CX", "MOVWLZX", []Operand{AX, CX}, "0fb7c8", "MOVZX"},
{"MOVWQZX AX,R8", "MOVWQZX", []Operand{AX, r8}, "4c0fb7c0", "MOVZX"},
// The width pairs the toolchain accepts and GOROOT uses; bytes
// pinned from go tool asm (see testdata/verify/widen_amd64.s).
{"MOVBWZX (BX),R11W", "MOVBWZX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fb61b", "MOVZX"},
{"MOVBWSX (BX),R11W", "MOVBWSX", []Operand{Ptr(BX, 0, 1), Reg{idx: 11, size: 2}}, "66440fbe1b", "MOVSX"},
{"MOVBLSX (BX),AX", "MOVBLSX", []Operand{Ptr(BX, 0, 1), AX}, "0fbe03", "MOVSX"},
{"MOVBQSX (BX),R8", "MOVBQSX", []Operand{Ptr(BX, 0, 1), r8}, "4c0fbe03", "MOVSX"},
{"MOVWQSX (BX),R9", "MOVWQSX", []Operand{Ptr(BX, 0, 2), r9}, "4c0fbf0b", "MOVSX"},
// A long to quad zero-extend is a plain 32-bit move.
{"MOVLQZX (BX),DX", "MOVLQZX", []Operand{Ptr(BX, 0, 4), DX}, "8b13", "MOV"},
{"MOVLQZX AX,DX", "MOVLQZX", []Operand{AX, DX}, "8bd0", "MOV"},
{"PMOVMSKB X1,AX", "PMOVMSKB", []Operand{vreg(t, "X1"), AX}, "660fd7c1", "PMOVMSKB"},
{"PMOVMSKB X11,CX", "PMOVMSKB", []Operand{vreg(t, "X11"), CX}, "66410fd7cb", "PMOVMSKB"},
{"CVTSL2SD R8,X13", "CVTSL2SD", []Operand{r8, vreg(t, "X13")}, "f2450f2ae8", "CVTSI2SD"},
{"CVTSL2SD AX,X0", "CVTSL2SD", []Operand{AX, vreg(t, "X0")}, "f20f2ac0", "CVTSI2SD"},
{"CVTSQ2SD R8,X13", "CVTSQ2SD", []Operand{r8, vreg(t, "X13")}, "f24d0f2ae8", "CVTSI2SD"},
@@ -364,6 +507,346 @@ func TestScalarErrors(t *testing.T) {
}
}
// TestImmediateOutOfRange pins the go-tool-asm parity of the immediate and
// displacement spans: a scalar immediate must fit a signed or unsigned 32-bit
// word (only MOVQ reg, $imm takes the full int64), a scalar shift count must
// be an unsigned byte, and a displacement must fit int32. Every rejected
// shape here is rejected by `go tool asm` too; every accepted one encodes the
// same bytes.
func TestImmediateOutOfRange(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
}{
{"SHLQ count 300", "SHLQ", []Operand{Imm(300), AX}},
{"SHLQ count -1", "SHLQ", []Operand{Imm(-1), AX}},
{"SHLW count 256", "SHLW", []Operand{Imm(256), DX}},
{"SHLB count 300", "SHLB", []Operand{Imm(300), BL}},
{"MOVL imm32+", "MOVL", []Operand{Imm(4294967296), AX}},
{"MOVL imm32-", "MOVL", []Operand{Imm(-2147483649), AX}},
{"MOVW imm32+", "MOVW", []Operand{Imm(4294967296), AX}},
{"MOVB imm32+", "MOVB", []Operand{Imm(4294967296), AL}},
{"ADDB imm32+", "ADDB", []Operand{Imm(4294967296), AL}},
{"ADDL imm32+", "ADDL", []Operand{Imm(4294967296), AX}},
{"ADDQ imm32+", "ADDQ", []Operand{Imm(8589934592), AX}},
{"CMPQ imm32+", "CMPQ", []Operand{AX, Imm(4294967296)}},
{"CMPQ imm32-", "CMPQ", []Operand{AX, Imm(-2147483649)}},
{"TESTL imm32+", "TESTL", []Operand{Imm(4294967296), AX}},
{"IMUL3L imm32+", "IMUL3L", []Operand{Imm(4294967296), CX, DX}},
{"PUSHQ imm32+", "PUSHQ", []Operand{Imm(4294967296)}},
{"MOVQ mem imm32+", "MOVQ", []Operand{Imm(4294967296), Ptr(AX, 0, 8)}},
{"disp32+", "MOVQ", []Operand{Ptr(AX, 4294967296, 8), BX}},
{"disp32+ max", "MOVQ", []Operand{Ptr(AX, 2147483648, 8), BX}},
{"disp32-", "MOVQ", []Operand{Ptr(AX, -2147483649, 8), BX}},
{"VEX disp32+", "VMOVDQU", []Operand{Ptr(AX, 4294967296, 32), vreg(t, "Y1")}},
{"EVEX disp32+", "VMOVDQU32", []Operand{Ptr(AX, 4294967296, 64), vreg(t, "Z1")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestImmediateTruncation pins the toolchain-matching truncations inside the
// accepted 32-bit span: the narrower fields take the low bits silently, byte
// for byte with `go tool asm` (which rejects none of these).
func TestImmediateTruncation(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ADDB $256,BL", "ADDB", []Operand{Imm(256), BL}, "80c300"},
{"ADDB $1000,BL", "ADDB", []Operand{Imm(1000), BL}, "80c3e8"},
{"MOVB $256,AL", "MOVB", []Operand{Imm(256), AL}, "b000"},
{"MOVB $-129,AL", "MOVB", []Operand{Imm(-129), AL}, "b07f"},
{"MOVW $65536,AX", "MOVW", []Operand{Imm(65536), AX}, "66b80000"},
{"MOVW $65535,AX", "MOVW", []Operand{Imm(65535), AX}, "66b8ffff"},
{"MOVW $-32769,AX", "MOVW", []Operand{Imm(-32769), AX}, "66b8ff7f"},
{"MOVL $4294967295,AX", "MOVL", []Operand{Imm(4294967295), AX}, "b8ffffffff"},
{"ADDQ $4294967295,AX", "ADDQ", []Operand{Imm(4294967295), AX}, "4805ffffffff"},
{"CMPB BL,$255", "CMPB", []Operand{BL, Imm(255)}, "80fbff"},
{"CMPQ AX,$4294967295", "CMPQ", []Operand{AX, Imm(4294967295)}, "483dffffffff"},
{"MOVQ $4294967295,0(AX)", "MOVQ", []Operand{Imm(4294967295), Ptr(AX, 0, 8)}, "48c700ffffffff"},
{"SHLQ $255,AX", "SHLQ", []Operand{Imm(255), AX}, "48c1e0ff"},
{"SHLQ $0,AX", "SHLQ", []Operand{Imm(0), AX}, "48c1e000"},
// The one form beyond the 32-bit span: the imm64 MOVQ register move.
{"MOVQ $4294967296,AX", "MOVQ", []Operand{Imm(4294967296), AX}, "48b80000000001000000"},
{"MOVQ disp32 max", "MOVQ", []Operand{Ptr(AX, 2147483647, 8), BX}, "488b98ffffff7f"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must
// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size
// letter) is not reported as encodable.
func TestEncodableCmovSize(t *testing.T) {
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} {
if Encodable(m) {
t.Errorf("Encodable(%q) = true, want false", m)
}
}
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} {
if !Encodable(m) {
t.Errorf("Encodable(%q) = false, want true", m)
}
}
}
// TestCarryShiftMulGroundTruth pins the carry-flag ALU family (ADC/SBB with
// their accumulator immediate forms), the rotate family, MUL/DIV/IDIV and the
// bit-test family byte for byte against go tool asm (see
// testdata/verify/scalar_amd64.s).
func TestCarryShiftMulGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ADCQ AX,BX", "ADCQ", []Operand{AX, BX}, "4811c3"},
{"ADCL AX,BX", "ADCL", []Operand{AX, BX}, "11c3"},
{"ADCB AL,BL", "ADCB", []Operand{AL, BL}, "10c3"},
{"ADCW AX,BX", "ADCW", []Operand{AX, BX}, "6611c3"},
{"SBBQ AX,BX", "SBBQ", []Operand{AX, BX}, "4819c3"},
{"ADCQ $5,BX", "ADCQ", []Operand{Imm(5), BX}, "4883d305"},
{"ADCQ $300,BX", "ADCQ", []Operand{Imm(300), BX}, "4881d32c010000"},
{"ADCQ $300,AX", "ADCQ", []Operand{Imm(300), AX}, "48152c010000"},
{"ADCB $5,AL", "ADCB", []Operand{Imm(5), AL}, "1405"},
{"SBBQ $300,AX", "SBBQ", []Operand{Imm(300), AX}, "481d2c010000"},
{"ADCQ AX,(BX)", "ADCQ", []Operand{AX, Ptr(BX, 0, 8)}, "481103"},
{"ROLQ $3,AX", "ROLQ", []Operand{Imm(3), AX}, "48c1c003"},
{"ROLL CX,BX", "ROLL", []Operand{CL, BX}, "d3c3"},
{"RORQ CL,AX", "RORQ", []Operand{CL, AX}, "48d3c8"},
{"RCRQ $1,BX", "RCRQ", []Operand{Imm(1), BX}, "48d1db"},
{"RCLQ $3,AX", "RCLQ", []Operand{Imm(3), AX}, "48c1d003"},
{"RORB CL,BL", "RORB", []Operand{CL, BL}, "d2cb"},
{"SALQ $2,AX", "SALQ", []Operand{Imm(2), AX}, "48c1e002"},
{"ROLW $1,AX", "ROLW", []Operand{Imm(1), AX}, "66d1c0"},
{"MULQ CX", "MULQ", []Operand{CX}, "48f7e1"},
{"MULL CX", "MULL", []Operand{CX}, "f7e1"},
{"MULB CL", "MULB", []Operand{CL}, "f6e1"},
{"DIVL CX", "DIVL", []Operand{CX}, "f7f1"},
{"IDIVQ CX", "IDIVQ", []Operand{CX}, "48f7f9"},
{"MULW CX", "MULW", []Operand{CX}, "66f7e1"},
{"BTQ AX,DX", "BTQ", []Operand{AX, DX}, "480fa3c2"},
{"BTL AX,DX", "BTL", []Operand{AX, DX}, "0fa3c2"},
{"BTW AX,DX", "BTW", []Operand{AX, DX}, "660fa3c2"},
{"BTQ $3,BX", "BTQ", []Operand{Imm(3), BX}, "480fbae303"},
{"BTQ $3,(AX)", "BTQ", []Operand{Imm(3), Ptr(AX, 0, 8)}, "480fba2003"},
{"BTSQ $5,BX", "BTSQ", []Operand{Imm(5), BX}, "480fbaeb05"},
{"BTCQ AX,BX", "BTCQ", []Operand{AX, BX}, "480fbbc3"},
{"BTRQ $7,BX", "BTRQ", []Operand{Imm(7), BX}, "480fbaf307"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// The bit-test immediate is an unsigned bit index with the negative
// spelling accepted, the shuffle convention: BTQ $300 must be rejected.
if _, err := Encode("BTQ", Imm(300), AX); err == nil {
t.Errorf("BTQ $300: expected an error, got none")
}
}
// TestAtomicSystemGroundTruth pins the exchange/compare-exchange/accumulate
// family, the string primitives, the flag and system instructions, the MXCSR
// pair, the scalar float-to-int conversions and the x87 FMOVD byte for byte
// against go tool asm (see testdata/verify/atomics_amd64.s and
// testdata/verify/system_amd64.s).
func TestAtomicSystemGroundTruth(t *testing.T) {
r8 := Reg{idx: 8, size: 8}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"XCHGQ AX,BX", "XCHGQ", []Operand{AX, BX}, "4893"},
{"XCHGQ BX,AX", "XCHGQ", []Operand{BX, AX}, "4893"},
{"XCHGL AX,BX", "XCHGL", []Operand{AX, BX}, "93"},
{"XCHGB AL,BL", "XCHGB", []Operand{AL, BL}, "86c3"},
{"XCHGW AX,BX", "XCHGW", []Operand{AX, BX}, "6693"},
{"XCHGQ R8,R9", "XCHGQ", []Operand{r8, Reg{idx: 9, size: 8}}, "4d87c1"},
{"XCHGQ BX,(AX)", "XCHGQ", []Operand{BX, Ptr(AX, 0, 8)}, "488718"},
{"XCHGQ (AX),BX", "XCHGQ", []Operand{Ptr(AX, 0, 8), BX}, "488718"},
{"XCHGQ AX,(BX)", "XCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "488703"},
{"CMPXCHGL AX,BX", "CMPXCHGL", []Operand{AX, BX}, "0fb1c3"},
{"CMPXCHGQ AX,(BX)", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
{"CMPXCHGB AL,(BX)", "CMPXCHGB", []Operand{AL, Ptr(BX, 0, 1)}, "0fb003"},
{"CMPXCHGW AX,BX", "CMPXCHGW", []Operand{AX, BX}, "660fb1c3"},
{"XADDL AX,BX", "XADDL", []Operand{AX, BX}, "0fc1c3"},
{"XADDQ AX,(BX)", "XADDQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fc103"},
{"XADDB AL,(BX)", "XADDB", []Operand{AL, Ptr(BX, 0, 1)}, "0fc003"},
{"XADDW AX,BX", "XADDW", []Operand{AX, BX}, "660fc1c3"},
{"ADCXL AX,CX", "ADCXL", []Operand{AX, CX}, "660f38f6c8"},
{"ADCXQ AX,CX", "ADCXQ", []Operand{AX, CX}, "66480f38f6c8"},
{"ADOXL AX,CX", "ADOXL", []Operand{AX, CX}, "f30f38f6c8"},
{"ADOXQ AX,CX", "ADOXQ", []Operand{AX, CX}, "f3480f38f6c8"},
{"CRC32B AX,CX", "CRC32B", []Operand{AX, CX}, "f20f38f0c8"},
{"CRC32W AX,CX", "CRC32W", []Operand{AX, CX}, "66f20f38f1c8"},
{"CRC32L AX,CX", "CRC32L", []Operand{AX, CX}, "f20f38f1c8"},
{"CRC32Q AX,CX", "CRC32Q", []Operand{AX, CX}, "f2480f38f1c8"},
{"CRC32L (AX),CX", "CRC32L", []Operand{Ptr(AX, 0, 4), CX}, "f20f38f108"},
{"MOVSQ", "MOVSQ", []Operand{}, "48a5"},
{"MOVSL", "MOVSL", []Operand{}, "a5"},
{"MOVSB", "MOVSB", []Operand{}, "a4"},
{"MOVSW", "MOVSW", []Operand{}, "66a5"},
{"STOSB", "STOSB", []Operand{}, "aa"},
{"STOSQ", "STOSQ", []Operand{}, "48ab"},
{"STOSL", "STOSL", []Operand{}, "ab"},
{"STOSW", "STOSW", []Operand{}, "66ab"},
{"CLD", "CLD", []Operand{}, "fc"},
{"STD", "STD", []Operand{}, "fd"},
{"POPFQ", "POPFQ", []Operand{}, "9d"},
{"PUSHFQ", "PUSHFQ", []Operand{}, "9c"},
{"CPUID", "CPUID", []Operand{}, "0fa2"},
{"RDTSC", "RDTSC", []Operand{}, "0f31"},
{"RDTSCP", "RDTSCP", []Operand{}, "0f01f9"},
{"SYSCALL", "SYSCALL", []Operand{}, "0f05"},
{"XGETBV", "XGETBV", []Operand{}, "0f01d0"},
{"PAUSE", "PAUSE", []Operand{}, "f390"},
{"LFENCE", "LFENCE", []Operand{}, "0faee8"},
{"MFENCE", "MFENCE", []Operand{}, "0faef0"},
{"SFENCE", "SFENCE", []Operand{}, "0faef8"},
{"UNDEF", "UNDEF", []Operand{}, "0f0b"},
{"INT $3", "INT", []Operand{Imm(3)}, "cd03"},
{"LDMXCSR (AX)", "LDMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae10"},
{"STMXCSR (AX)", "STMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae18"},
{"CVTSD2SL X0,AX", "CVTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2dc0"},
{"CVTTSD2SQ X0,AX", "CVTTSD2SQ", []Operand{vreg(t, "X0"), AX}, "f2480f2cc0"},
{"CVTTSD2SL X0,AX", "CVTTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2cc0"},
{"CVTSS2SQ X0,AX", "CVTSS2SQ", []Operand{vreg(t, "X0"), AX}, "f3480f2dc0"},
{"FMOVD (AX),F0", "FMOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "F0")}, "dd00"},
{"FMOVD F0,(AX)", "FMOVD", []Operand{vreg(t, "F0"), Ptr(AX, 0, 8)}, "dd10"},
{"FMOVD F0,F1", "FMOVD", []Operand{vreg(t, "F0"), vreg(t, "F1")}, "ddd1"},
{"MOVD AX,X0", "MOVD", []Operand{AX, vreg(t, "X0")}, "66480f6ec0"},
{"MOVD X0,AX", "MOVD", []Operand{vreg(t, "X0"), AX}, "66480f7ec0"},
{"MOVD X0,X1", "MOVD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f30f7ec8"},
{"MOVD (AX),X0", "MOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "X0")}, "f30f7e00"},
{"MOVD X0,(AX)", "MOVD", []Operand{vreg(t, "X0"), Ptr(AX, 0, 8)}, "660fd600"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// LDMXCSR/STMXCSR take a memory operand only.
if _, err := Encode("LDMXCSR", AX); err == nil {
t.Errorf("LDMXCSR AX: expected an error, got none")
}
}
// TestSSEGapsGroundTruth pins the legacy SSE gap families: the scalar
// compare and square root, the Plan 9 packed spellings, the imm8-controlled
// shuffles, the lane extracts and inserts, the packed integer shifts and the
// AES/SHA round instructions, byte for byte against go tool asm (see
// testdata/verify/crypto_amd64.s and testdata/verify/sse_amd64.s).
func TestSSEGapsGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ANDNPD X0,X1", "ANDNPD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f55c8"},
{"ANDNPS X0,X1", "ANDNPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f55c8"},
{"COMISD X0,X1", "COMISD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f2fc8"},
{"SQRTSD X0,X1", "SQRTSD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f20f51c8"},
{"PSHUFL $3,X0,X1", "PSHUFL", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f70c803"},
{"PALIGNR $2,X0,X1", "PALIGNR", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0fc802"},
{"PBLENDW $3,X0,X1", "PBLENDW", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0ec803"},
{"PCMPESTRI $1,X0,X1", "PCMPESTRI", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}, "660f3a61c801"},
{"PCLMULQDQ $0,X0,X1", "PCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3a44c800"},
{"PCLMULQDQ $0,(AX),X1", "PCLMULQDQ", []Operand{Imm(0), Ptr(AX, 0, 16), vreg(t, "X1")}, "660f3a440800"},
{"PEXTRB $1,X0,AX", "PEXTRB", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a14c001"},
{"PEXTRD $1,X0,AX", "PEXTRD", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a16c001"},
{"PEXTRQ $1,X0,AX", "PEXTRQ", []Operand{Imm(1), vreg(t, "X0"), AX}, "66480f3a16c001"},
{"PEXTRW $1,X0,AX", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), AX}, "660fc5c001"},
{"PEXTRW $1,X0,(AX)", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), Ptr(AX, 0, 2)}, "660f3a150001"},
{"PINSRB $1,AX,X0", "PINSRB", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a20c001"},
{"PINSRD $1,AX,X0", "PINSRD", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a22c001"},
{"PINSRQ $1,AX,X0", "PINSRQ", []Operand{Imm(1), AX, vreg(t, "X0")}, "66480f3a22c001"},
{"PINSRW $1,AX,X0", "PINSRW", []Operand{Imm(1), AX, vreg(t, "X0")}, "660fc4c001"},
{"PINSRW $1,(AX),X0", "PINSRW", []Operand{Imm(1), Ptr(AX, 0, 2), vreg(t, "X0")}, "660fc40001"},
{"PSLLL $2,X0", "PSLLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72f002"},
{"PSRAL $2,X0", "PSRAL", []Operand{Imm(2), vreg(t, "X0")}, "660f72e002"},
{"PSRLL $2,X0", "PSRLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72d002"},
{"PSRLQ $2,X0", "PSRLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d002"},
{"PSLLQ $2,X0", "PSLLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f002"},
{"PSLLW $2,X0", "PSLLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71f002"},
{"PSRLW $2,X0", "PSRLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71d002"},
{"PSRAW $2,X0", "PSRAW", []Operand{Imm(2), vreg(t, "X0")}, "660f71e002"},
{"PSLLDQ $2,X0", "PSLLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f802"},
{"PSRLDQ $2,X0", "PSRLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d802"},
{"PSLLL X0,X1", "PSLLL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ff2c8"},
{"PSRLQ X0,X1", "PSRLQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660fd3c8"},
{"PSLLL (AX),X1", "PSLLL", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660ff208"},
{"PSUBL X0,X1", "PSUBL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffac8"},
{"PADDL X0,X1", "PADDL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffec8"},
{"PCMPEQL X0,X1", "PCMPEQL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f76c8"},
{"PUNPCKLBW X0,X1", "PUNPCKLBW", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f60c8"},
{"MOVOA X0,X1", "MOVOA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f6fc8"},
{"MOVOA (AX),X1", "MOVOA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f6f08"},
{"MOVOA X0,(AX)", "MOVOA", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "660f7f00"},
{"AESIMC X0,X1", "AESIMC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dbc8"},
{"AESIMC (AX),X1", "AESIMC", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f38db08"},
{"AESENC X0,X1", "AESENC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dcc8"},
{"AESENCLAST X0,X1", "AESENCLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38ddc8"},
{"AESDEC X0,X1", "AESDEC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dec8"},
{"AESDECLAST X0,X1", "AESDECLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dfc8"},
{"AESKEYGENASSIST $0,X0,X1", "AESKEYGENASSIST", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3adfc800"},
{"SHA1MSG1 X0,X1", "SHA1MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c9c8"},
{"SHA1MSG2 X0,X1", "SHA1MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cac8"},
{"SHA1NEXTE X0,X1", "SHA1NEXTE", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c8c8"},
{"SHA1RNDS4 $0,X0,X1", "SHA1RNDS4", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "0f3accc800"},
{"SHA256MSG1 X0,X1", "SHA256MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38ccc8"},
{"SHA256MSG2 X0,X1", "SHA256MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cdc8"},
{"SHA256RNDS2 X0,X1,X2", "SHA256RNDS2", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "0f38cbd1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// SHA256RNDS2's first operand must be the literal X0.
if _, err := Encode("SHA256RNDS2", vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")); err == nil {
t.Errorf("SHA256RNDS2 X1,...: expected an error, got none")
}
// PSLLDQ has no variable-count form.
if _, err := Encode("PSLLDQ", vreg(t, "X0"), vreg(t, "X1")); err == nil {
t.Errorf("PSLLDQ X0,X1: expected an error, got none")
}
}
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
// byte for byte (no prefix / 66 / F2 / F3 variants).
func TestSSEBinGroundTruth(t *testing.T) {
@@ -421,7 +904,7 @@ func TestSSEShuffleGroundTruth(t *testing.T) {
}
// TestMOVQXMMGroundTruth pins the SSE2 packed-quadword move encodings:
// loads and register moves on F3 0F 7E, stores on 66 0F D6 — the forms
// loads and register moves on F3 0F 7E, stores on 66 0F D6; the forms
// the GPR-move fallback silently corrupted.
func TestMOVQXMMGroundTruth(t *testing.T) {
cases := []struct {
@@ -445,3 +928,296 @@ func TestMOVQXMMGroundTruth(t *testing.T) {
}
}
}
// TestPrefixStatements pins LOCK, REP and REPN. go tool asm encodes each as
// a standalone one-byte instruction with a PC of its own (F0, F3, F2), not a
// prefix field merged into the following instruction, and it validates
// nothing about the pairing (LOCK before NOP assembles). The prefixed
// atomic and string shapes are the bytes the runtime's own kernels need.
func TestPrefixStatements(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"LOCK", "LOCK", nil, "f0"},
{"REP", "REP", nil, "f3"},
{"REPN", "REPN", nil, "f2"},
// LOCK; CMPXCHGQ AX, (BX)
{"LOCK CMPXCHGQ", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
// REP; MOVSQ
{"REP MOVSQ", "MOVSQ", nil, "48a5"},
// REPN; MOVSB
{"REPN MOVSB", "MOVSB", nil, "a4"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// The prefix statements take no operands, as the toolchain reports for
// LOCK AX.
if _, err := Encode("LOCK", AX); err == nil {
t.Error("LOCK AX assembled, want an error")
}
if _, err := Encode("REP", Imm(1)); err == nil {
t.Error("REP $1 assembled, want an error")
}
}
// TestDataEmission pins BYTE, WORD, LONG and QUAD: the immediate lands in
// the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
// lookup, truncated to the width rather than range-checked (go tool asm
// emits FF for BYTE $0x1FF and 45 23 for WORD $0x12345, both silently).
func TestDataEmission(t *testing.T) {
cases := []struct {
name string
mnem string
imm Imm
want string
}{
{"BYTE", "BYTE", 0x0f, "0f"},
{"BYTE negative", "BYTE", -1, "ff"},
{"BYTE truncated", "BYTE", 0x1ff, "ff"},
{"WORD", "WORD", 0x1234, "3412"},
{"WORD negative", "WORD", -1, "ffff"},
{"WORD truncated", "WORD", 0x12345, "4523"},
{"LONG", "LONG", 0x11223344, "44332211"},
{"LONG negative", "LONG", -1, "ffffffff"},
{"QUAD", "QUAD", 0x1122334455667788, "8877665544332211"},
{"QUAD negative", "QUAD", -2, "feffffffffffffff"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.imm)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// Exactly one immediate: the toolchain rejects BYTE $1, $2, $3, and a
// register or a missing operand is no immediate at all.
if _, err := Encode("BYTE"); err == nil {
t.Error("BYTE with no operand assembled, want an error")
}
if _, err := Encode("BYTE", Imm(1), Imm(2)); err == nil {
t.Error("BYTE $1, $2 assembled, want an error")
}
if _, err := Encode("WORD", AX); err == nil {
t.Error("WORD AX assembled, want an error")
}
}
// TestEndIgnored pins END: go tool asm drops the statement entirely, so it
// encodes to zero bytes and takes any operands without complaint (the
// toolchain accepts END $0 and END AX alike).
func TestEndIgnored(t *testing.T) {
for _, ops := range [][]Operand{nil, {Imm(0)}, {AX}} {
code, err := Encode("END", ops...)
if err != nil {
t.Errorf("END: %v", err)
continue
}
if len(code) != 0 {
t.Errorf("END = %x, want no bytes", code)
}
}
}
// TestAdjsp pins ADJSP: a positive immediate is SUBQ $imm, SP, a negative
// one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude picks; $0
// has no encoding (go tool asm refuses ADJSP $0 outright).
func TestAdjsp(t *testing.T) {
cases := []struct {
name string
imm Imm
want string
}{
{"imm8", 112, "4883ec70"},
{"imm8 negative", -112, "4883c470"},
{"imm32", 200, "4881ecc8000000"},
{"imm32 negative", -200, "4881c4c8000000"},
{"small", 8, "4883ec08"},
}
for _, c := range cases {
code, err := Encode("ADJSP", c.imm)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("ADJSP %d = %s, want %s", int64(c.imm), got, c.want)
}
}
if _, err := Encode("ADJSP", Imm(0)); err == nil {
t.Error("ADJSP $0 assembled, want an error")
}
if _, err := Encode("ADJSP"); err == nil {
t.Error("ADJSP with no operand assembled, want an error")
}
if _, err := Encode("ADJSP", AX); err == nil {
t.Error("ADJSP AX assembled, want an error")
}
}
// TestFloatImmediateGroundTruth pins the floating-point immediate rewrite
// byte for byte against go tool asm: the scalar moves and the scalar
// arithmetic read the constant from a synthesised read-only pool symbol
// ($f64.<hex>, $f32.<hex>) RIP-relative with the displacement left to the
// relocation, and a positive zero on the moves collapses to XORPS dst, dst.
func TestFloatImmediateGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"MOVSD -1.0", "MOVSD", []Operand{FloatImm{Text: "1.0", Neg: true}, vreg(t, "X2")}, "f20f101500000000"},
{"MOVSD 1.5", "MOVSD", []Operand{FloatImm{Text: "1.5"}, vreg(t, "X3")}, "f20f101d00000000"},
{"MOVSS 2.5", "MOVSS", []Operand{FloatImm{Text: "2.5"}, vreg(t, "X4")}, "f30f102500000000"},
{"MOVSS -0.5", "MOVSS", []Operand{FloatImm{Text: "0.5", Neg: true}, vreg(t, "X5")}, "f30f102d00000000"},
{"MOVSS +0.0 is XORPS", "MOVSS", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X10")}, "450f57d2"},
{"MOVSD +0.0 is XORPS", "MOVSD", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X6")}, "0f57f6"},
{"ADDSD 1.0", "ADDSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f580500000000"},
{"ADDSS 0.5", "ADDSS", []Operand{FloatImm{Text: "0.5"}, vreg(t, "X1")}, "f30f580d00000000"},
{"SUBSD 2.0", "SUBSD", []Operand{FloatImm{Text: "2.0"}, vreg(t, "X3")}, "f20f5c1d00000000"},
{"MULSD -2.5", "MULSD", []Operand{FloatImm{Text: "2.5", Neg: true}, vreg(t, "X3")}, "f20f591d00000000"},
{"DIVSD 1.0", "DIVSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f5e0500000000"},
{"COMISD 1.0", "COMISD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "660f2f0500000000"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
// The pool names carry the IEEE-754 bits, the float32 narrowing for the
// single spellings; negative zero keeps its sign bit and never takes the
// XORPS shortcut.
for _, c := range []struct {
mnem string
imm FloatImm
want string
}{
{"MOVSD", FloatImm{Text: "1.0", Neg: true}, "$f64.bff0000000000000"},
{"MOVSD", FloatImm{Text: "0.5"}, "$f64.3fe0000000000000"},
{"MOVSS", FloatImm{Text: "2.5"}, "$f32.40200000"},
{"MOVSS", FloatImm{Text: "0.5", Neg: true}, "$f32.bf000000"},
{"MOVSD", FloatImm{Text: "0.0", Neg: true}, "$f64.8000000000000000"},
} {
_, name, err := floatPoolValue(c.mnem, c.imm)
if err != nil {
t.Errorf("%s %s: %v", c.mnem, c.imm.Text, err)
continue
}
if name != c.want {
t.Errorf("%s $%s: pool name %s, want %s", c.mnem, c.imm.Text, name, c.want)
}
}
// The shapes the toolchain's parser rejects: the packed and uniform
// forms, a non-vector destination, and the integer spellings.
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"MAXSD rejects the immediate", "MAXSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"MINSD rejects the immediate", "MINSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"SQRTSD rejects the immediate", "SQRTSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
{"integer destination", "MOVSD", []Operand{FloatImm{Text: "1.0"}, AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
// TestBookkeepingGroundTruth pins FUNCDATA and PCDATA as accept-and-ignore:
// go tool asm emits no text bytes for either, on every architecture.
func TestBookkeepingGroundTruth(t *testing.T) {
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"FUNCDATA", "FUNCDATA", []Operand{Imm(3), sbMem{name: "\u00b7f.arginfo0"}}},
{"PCDATA", "PCDATA", []Operand{Imm(1), Imm(-1)}},
} {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if len(code) != 0 {
t.Errorf("%s: emitted %x, want no bytes", c.name, code)
}
}
for _, c := range []struct {
name string
mnem string
ops []Operand
}{
{"FUNCDATA arity", "FUNCDATA", []Operand{Imm(3)}},
{"FUNCDATA missing the count", "FUNCDATA", []Operand{sbMem{name: "x"}}},
{"FUNCDATA integer value", "FUNCDATA", []Operand{Imm(3), Imm(4)}},
{"PCDATA arity", "PCDATA", []Operand{Imm(1)}},
{"PCDATA register value", "PCDATA", []Operand{Imm(1), AX}},
} {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
// At the statement level the bookkeeping lines sit between real
// instructions and contribute nothing to the body, symbol reference
// included: the FUNCDATA operand never needs file-level resolution.
f, errs := parser.Parse("t_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tNOP\n\tFUNCDATA $3, \u00b7f.arginfo0(SB)\n\tPCDATA $1, $-1\n\tFUNCDATA $0, x<>(SB)\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
want := "90c3"
if got := hexCompact(img.Code); got != want {
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
}
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tFUNCDATA $1, X0\n\tRET\n")); err == nil {
t.Error("FUNCDATA $1, X0 assembled, want an error")
}
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tPCDATA $1, X0\n\tRET\n")); err == nil {
t.Error("PCDATA $1, X0 assembled, want an error")
}
// Encodable mirrors Encode for the names this work touched.
for _, mnem := range []string{"FUNCDATA", "PCDATA", "V4FMADDPS", "V4FMADDSS", "V4FNMADDPS", "V4FNMADDSS", "VP4DPWSSD", "VP4DPWSSDS"} {
if !Encodable(mnem) {
t.Errorf("Encodable(%s) = false, want true", mnem)
}
}
}
// mustParse parses src or fails the test.
func mustParse(t *testing.T, src string) *ast.File {
t.Helper()
f, errs := parser.Parse("t_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
return f
}
+735 -132
View File
File diff suppressed because it is too large Load Diff
+254 -13
View File
@@ -16,7 +16,7 @@ import (
// kernels use: NDS arithmetic, immediate and variable shifts, shuffles with
// an immediate, lane extracts, narrowing stores, broadcasts from a GPR or
// memory, mask destinations, mask moves, disp8×N compression and the 5-bit
// register fields (X/Y 16–31, Z 0–31).
// register fields (X/Y 16-31, Z 0-31).
func TestEvexGroundTruth(t *testing.T) {
cases := []struct {
name string
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
// The qword OR spelling always encodes through EVEX.
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
// Byte permute and population count.
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
// Align (NDS + imm8).
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
@@ -52,13 +61,23 @@ func TestEvexGroundTruth(t *testing.T) {
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
// Moves, incl. disp8×N (64 for a 512-bit operand).
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
{"VMOVDQU32 16(SI)(R15*4),Z4", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 16, 64), vreg(t, "Z4")}, "62b17e486fa4be10000000"},
{"VMOVDQU32 Z0,4(SI)(AX*1)", "VMOVDQU32", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f17e487f840604000000"},
{"VMOVDQU32 Z3,(DI)(R15*4)", "VMOVDQU32", []Operand{vreg(t, "Z3"), Idx(DI, vreg(t, "R15"), 4, 0, 64)}, "62b17e487f1cbf"},
// VMOVDQU64 — the W1 qword variant.
// VMOVDQU64; the W1 qword variant.
{"VMOVDQU64 (SI)(R15*4),Z3", "VMOVDQU64", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b1fe486f1cbe"},
{"VMOVDQU64 Z0,4(SI)(AX*1)", "VMOVDQU64", []Operand{vreg(t, "Z0"), Idx(SI, AX, 1, 4, 64)}, "62f1fe487f840604000000"},
{"VMOVDQU64 Z1,Z2", "VMOVDQU64", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1fe487fca"},
@@ -77,7 +96,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VPSHUFB Z1,Z2,Z3", "VPSHUFB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d4800d9"},
{"VMOVDQU8 Z1,Z2", "VMOVDQU8", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f17f487fca"},
{"VMOVDQU16 Z1,Z2", "VMOVDQU16", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff487fca"},
// Indices 16–31: rm[4] rides in X̄ for register operands.
// Indices 16-31: rm[4] rides in X̄ for register operands.
{"VPSHUFD $1,X16,X17", "VPSHUFD", []Operand{Imm(1), vreg(t, "X16"), vreg(t, "X17")}, "62a17d0870c801"},
{"VMOVUPD (DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 0, 64), vreg(t, "Z14")}, "6271fd481037"},
{"VMOVUPD 64(DI),Z14", "VMOVUPD", []Operand{Ptr(DI, 64, 64), vreg(t, "Z14")}, "6271fd48107701"},
@@ -96,7 +115,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VPBROADCASTD 4(SI),Z10", "VPBROADCASTD", []Operand{Ptr(SI, 4, 4), vreg(t, "Z10")}, "62727d48585601"},
{"VPBROADCASTQ R8,X31", "VPBROADCASTQ", []Operand{vreg(t, "R8"), vreg(t, "X31")}, "6242fd087cf8"},
{"VPBROADCASTQ AX,Z9", "VPBROADCASTQ", []Operand{AX, vreg(t, "Z9")}, "6272fd487cc8"},
// Register indices 16–31 exist only in EVEX encodings.
// Register indices 16-31 exist only in EVEX encodings.
{"VPBROADCASTD AX,Y30", "VPBROADCASTD", []Operand{AX, vreg(t, "Y30")}, "62627d287cf0"},
// Packed double arithmetic / unpack (EVEX forms carry W=1).
{"VSUBPD Z1,Z2,Z3", "VSUBPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed485cd9"},
@@ -107,7 +126,7 @@ func TestEvexGroundTruth(t *testing.T) {
{"VUNPCKHPD Z1,Z2,Z3", "VUNPCKHPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed4815d9"},
{"VSUBPD 64(AX),Z1,Z2", "VSUBPD", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5485c5001"},
{"VSUBPD Z17,Z18,Z19", "VSUBPD", []Operand{vreg(t, "Z17"), vreg(t, "Z18"), vreg(t, "Z19")}, "62a1ed405cd9"},
// VMOVDDUP — duplicate the low double; disp8×N = 64 at 512 bits, and
// VMOVDDUP; duplicate the low double; disp8×N = 64 at 512 bits, and
// X16/X17 force EVEX (the mod=11 rm[4] extension rides in X̄).
{"VMOVDDUP Z1,Z2", "VMOVDDUP", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}, "62f1ff4812d1"},
{"VMOVDDUP 64(AX),Z1", "VMOVDDUP", []Operand{Ptr(AX, 64, 64), vreg(t, "Z1")}, "62f1ff48124801"},
@@ -150,7 +169,7 @@ func TestEvexGroundTruth(t *testing.T) {
}
}
// TestEvexMasking checks the AVX-512 mask operand (K1–K7, placed freely among
// TestEvexMasking checks the AVX-512 mask operand (K1-K7, placed freely among
// the operands) and the .Z zeroing suffix, byte for byte against the Go
// assembler.
func TestEvexMasking(t *testing.T) {
@@ -241,11 +260,11 @@ func TestEvexMasking(t *testing.T) {
}
}
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set — ternary
// TestEvexExtendedGroundTruth covers the wider EVEX/AVX-512 set; ternary
// logic, lane shuffles/inserts/extracts, compares with a K destination,
// permutes, the wider integer families, expand/compress, broadcasts,
// rotates and word shifts, the opmask instructions, the EVEX suffixes
// (rounding/SAE/broadcast) and the aligned/scalar moves — byte for byte
// (rounding/SAE/broadcast) and the aligned/scalar moves; byte for byte
// against the Go assembler.
func TestEvexExtendedGroundTruth(t *testing.T) {
mem64 := func(base Reg) Operand { return Ptr(base, 0, 64) }
@@ -275,7 +294,7 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
{"VMULPD.RZ_SAE.Z", "VMULPD.RZ_SAE.Z", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f1edf959d9"},
{"VMAXPD.SAE", "VMAXPD.SAE", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f1ed585fd9"},
{"VADDPD.BCST", "VADDPD.BCST", []Operand{mem64(AX), vreg(t, "Z1"), vreg(t, "Z2")}, "62f1f5585810"},
// Packed single arithmetic (same opcodes, no mandatory prefix) —
// Packed single arithmetic (same opcodes, no mandatory prefix);
// ZMM, YMM and XMM widths, rounding and broadcast.
{"VADDPS", "VADDPS", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f16c4858d9"},
{"VMULPS", "VMULPS", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ec59d9"},
@@ -399,8 +418,8 @@ func TestEvexExtendedGroundTruth(t *testing.T) {
}
// TestEvexHelperGroundTruth covers the floating-point helper and conversion
// tail of the EVEX set — reciprocals, rsqrt, getexp/getmant, scalef,
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions —
// tail of the EVEX set; reciprocals, rsqrt, getexp/getmant, scalef,
// rndscale, reduce, fixupimm, range, fpclass, the remaining conversions;
// plus gather/scatter with VSIB addressing, byte for byte against the Go
// assembler.
func TestEvexHelperGroundTruth(t *testing.T) {
@@ -502,9 +521,9 @@ func TestEvexHelperGroundTruth(t *testing.T) {
}
// TestEvexGprGroundTruth covers the scalar conversions between vector and
// general-purpose registers — the signed and truncated VCVT{,T}S{D,S}2SI
// general-purpose registers; the signed and truncated VCVT{,T}S{D,S}2SI
// forms (VEX and EVEX), the unsigned EVEX-only forms, and the GPR-to-vector
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv — byte
// VCVTSI2*/VCVTUSI2* forms with the preserved vector source in vvvv; byte
// for byte against the Go assembler, including memory sources and extended
// GPRs.
func TestEvexGprGroundTruth(t *testing.T) {
@@ -675,6 +694,15 @@ func TestEvexErrors(t *testing.T) {
{"align arity", "VALIGND", []Operand{Imm(1), vreg(t, "Z0"), vreg(t, "Z1")}},
// VEX-only mnemonics reject registers only EVEX can encode.
{"VMOVMSKPS X16", "VMOVMSKPS", []Operand{vreg(t, "X16"), AX}},
// The scalar EVEX move matches its VEX twin and the Go assembler:
// XMM↔memory only, never reg-reg and never a wider register (the
// toolchain rejects every one of these shapes).
{"VMOVSS X1,X2", "VMOVSS", []Operand{vreg(t, "X1"), vreg(t, "X2")}},
{"VMOVSS X16,X2", "VMOVSS", []Operand{vreg(t, "X16"), vreg(t, "X2")}},
{"VMOVSS Y1,(AX)", "VMOVSS", []Operand{vreg(t, "Y1"), Ptr(AX, 0, 4)}},
{"VMOVSS Z1,Z2", "VMOVSS", []Operand{vreg(t, "Z1"), vreg(t, "Z2")}},
{"VMOVSS Z1,(AX)", "VMOVSS", []Operand{vreg(t, "Z1"), Ptr(AX, 0, 4)}},
{"VMOVSS (AX),Z2", "VMOVSS", []Operand{Ptr(AX, 0, 4), vreg(t, "Z2")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
@@ -693,3 +721,216 @@ func hexCompact(b []byte) string {
}
return string(out)
}
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
// families the toolchain's avx512enc corpus exercises: the bytes are the
// go tool asm output for exactly these operands, and the same families are
// covered end to end by the avx512_amd64.s differential kernel.
func TestAvx512CorpusFamilies(t *testing.T) {
vsib := func(base, idx string, scale int) Operand {
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
// AES rounds (EVEX NDS, VEX twin routed by operand width).
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
// Integer VNNI and the bit algorithm group.
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
// Permutations: immediate and register counts.
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
// Shifts: immediate, register-count and memory-count forms; the
// count source carries its own XMM tuple width.
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
// Conversions and shuffles with the F2 prefix and no prefix.
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
// Gather and scatter prefetch hints (memory-only, /digit in reg).
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
// Opmask broadcasts and the K logic.
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
// Lane extracts to general registers (EVEX and VEX routes).
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
// Moves: masked unaligned, masked scalar register form, half moves
// and non-temporal stores.
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
// Scalar compares with and without the 66 prefix.
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
// Floating point helpers.
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestEvexQuadRegisterGroundTruth pins the quad-register instructions (the
// 4FMAPS and 4VNNIW families) byte for byte against go tool asm: the memory
// source keeps r/m, the bracketed list's LOW register travels the inverted
// 5-bit V'VVVV field, the destination sits in reg, the opmask rides aaa and
// the vector length follows the destination (L'L=512 for the ZMM forms,
// 128 for the scalar ones) while the disp8×N multiplier stays 16 for every
// member. The x86 decoder has no view of these forms, so no decode check
// runs.
func TestEvexQuadRegisterGroundTruth(t *testing.T) {
sp := vreg(t, "RSP")
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"V4FMADDPS 17(SP) [Z0-Z3] K2 Z0", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a9a842411000000"},
{"V4FMADDPS [Z10-Z13]", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z10"), vreg(t, "Z13")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f22f4a9a842411000000"},
{"V4FMADDPS [Z20-Z23]", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z20"), vreg(t, "Z23")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f25f429a842411000000"},
{"V4FMADDPS Z8 dst", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z8")},
"62727f4a9a842411000000"},
{"V4FMADDPS disp8x16", "V4FMADDPS",
[]Operand{Ptr(sp, 64, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a9a442404"},
{"V4FMADDPS unmasked", "V4FMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
"62f27f489a842411000000"},
{"V4FMADDSS 7(AX) [X0-X3] K5 X22", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0d9bb007000000"},
{"V4FMADDSS (DI)", "V4FMADDSS",
[]Operand{Ptr(DI, 0, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0d9b37"},
{"V4FMADDSS [X10-X13]", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X10"), vreg(t, "X13")}, vreg(t, "K5"), vreg(t, "X22")},
"62e22f0d9bb007000000"},
{"V4FMADDSS [X20-X23]", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X22")},
"62e25f059bb007000000"},
{"V4FMADDSS X30 dst", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X30")},
"62627f0d9bb007000000"},
{"V4FMADDSS X3 dst", "V4FMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X3")},
"62f27f0d9b9807000000"},
{"V4FMADDSS disp8x16", "V4FMADDSS",
[]Operand{Ptr(AX, 16, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X30")},
"62625f059b7001"},
{"V4FNMADDPS", "V4FNMADDPS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4aaa842411000000"},
{"V4FNMADDSS", "V4FNMADDSS",
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
"62e27f0dabb007000000"},
{"VP4DPWSSD", "VP4DPWSSD",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
"62f27f4a52842411000000"},
{"VP4DPWSSDS unmasked", "VP4DPWSSDS",
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
"62f27f4853842411000000"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
}
}
}
// TestEvexQuadRegisterErrors pins the operand shapes the toolchain rejects:
// the register class the list and the destination take is fixed per
// instruction, the source is memory only, the opmask slot is positional and
// the list's low register owns V'VVVV.
func TestEvexQuadRegisterErrors(t *testing.T) {
sp := vreg(t, "RSP")
list := func(lo, hi string) RegList {
return RegList{vreg(t, lo), vreg(t, hi)}
}
cases := []struct {
name string
mnem string
ops []Operand
}{
{"X list on the PS form", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("X0", "X3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"Z list on the SS form", "V4FMADDSS",
[]Operand{Ptr(AX, 0, 8), list("Z0", "Z3"), vreg(t, "K5"), vreg(t, "X22")}},
{"Y destination", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Y0")}},
{"register source", "V4FMADDPS",
[]Operand{vreg(t, "Z1"), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"non-mask third operand", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z4"), vreg(t, "Z0")}},
{"k0 mask", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K0"), vreg(t, "Z0")}},
{"K after the destination", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0"), vreg(t, "K2")}},
{"zeroing without a mask", "V4FMADDPS.Z",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0")}},
{"SAE suffix", "V4FMADDPS.SAE",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"high index source", "VP4DPWSSD",
[]Operand{Idx(DI, vreg(t, "X16"), 1, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
{"short operand list", "V4FMADDPS",
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3")}},
}
for _, c := range cases {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: expected an error, got none", c.name)
}
}
}
+159 -21
View File
@@ -14,8 +14,8 @@ import (
"sync"
)
// This file emits GOOBJ — the Go toolchain's object format, which cmd/link
// consumes directly — so gasm-assembled functions drop into a go build
// This file emits GOOBJ, the Go toolchain's object format, which cmd/link
// consumes directly, so gasm-assembled functions drop into a go build
// without the Go assembler. The layout follows cmd/internal/goobj: a
// toolchain preamble ("go object ...\n!\n"), the go120ld header with its
// block offsets, a string table, symbol definitions, the relocation /
@@ -68,11 +68,12 @@ const (
kindSDWARFLINES = 20
)
// Symbol flags (cmd/internal/goobj).
// Symbol flags (cmd/internal/goobj). The linkname flag is set only for
// //go:linkname symbols (and main.main); ordinary assembly symbols carry
// none, matching cmd/asm's output.
const (
symFlagDupok = 0x01
symFlagNoSplit = 0x10
symFlag2Link = 0x10 // asm objects flag every named symbol as linkname
symABIStatic = 0xffff
)
@@ -94,10 +95,12 @@ const (
)
// Relocation types (cmd/internal/objabi).
// R_PCREL and R_ADDR are stable across Go versions.
// R_ADDR, R_CALL, R_PCREL and R_TLS_LE are stable across Go versions.
const (
relocPCRel = 14 // R_PCREL
relocAddr = 1 // R_ADDR
relocCall = 7 // R_CALL
relocPCRel = 14 // R_PCREL
relocTLSLE = 15 // R_TLS_LE
)
// relocDWTXTADDRU4 returns the R_DWTXTADDR_U4 relocation type for the
@@ -140,10 +143,30 @@ func isGo127OrLater() bool {
// Special package indices for symbol references.
const (
pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb
pkgIdxNone = 0x7fffffff
pkgIdxSelf = 0x7ffffffb
pkgIdxBuiltin = 0x7ffffffc
)
// goobjBuiltinMorestackNoctxt is the index of runtime.morestack_noctxt in
// cmd/internal/goobj/builtinlist.go of the toolchain the object targets
// (246 since Go 1.25; the list is append-only).
const goobjBuiltinMorestackNoctxt = 246
// goobjBuiltinMorestack is the builtin reference the toolchain emits for the
// stack-guard call.
var goobjBuiltinMorestack = "runtime\u00b7morestack_noctxt"
// isCallReloc reports whether k is one of the per-arch call relocations a
// direct branch to a TEXT symbol carries.
func isCallReloc(k RelocKind) bool {
switch k {
case RelCall, RelRISCVJal, RelArm64Branch, RelLoong64Branch:
return true
}
return false
}
const goobjMagic = "\x00go120ld"
// goSym is one symbol definition under construction.
@@ -178,14 +201,24 @@ type dwarfRelocSet struct {
// does with its -p flag). srcPath names the source file recorded in the
// object's file table and line tables. The toolchain's object preamble is
// captured from the installed go tool asm, so the output links with the
// toolchain it was produced on — exactly like a real assembly object.
// toolchain it was produced on, exactly like a real assembly object.
func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreamble()
if err != nil {
return nil, err
}
// amd64: MinLC 1, R_PCREL for the code relocations.
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) (uint16, uint8) { return relocPCRel, 4 })
// amd64: MinLC 1, R_PCREL for displacements, R_CALL for calls and
// R_TLS_LE for the stack-guard TLS load.
return img.emitGOObject(pkgPath, srcPath, pre, 1, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelCall:
return relocCall, 4
case RelTLSLE:
return relocTLSLE, 4
default:
return relocPCRel, 4
}
})
}
// emitGOObject assembles the GOOBJ payload for any architecture. pre is
@@ -198,7 +231,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)")
}
// The non-package definitions first — the DWARF symbols reference the
// The non-package definitions first, the DWARF symbols reference the
// functions by these indices: per function the four pc-value tables
// and the function itself, as cmd/asm lays them out.
type npSym struct {
@@ -247,7 +280,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
// their relocations cover whole AUIPC/pcalau12i pairs, so
// zeroing r.Off would erase the opcode/register bits the linker
// preserves when it patches only the immediate.
if r.Kind != RelPCRel32 {
if r.Kind != RelPCRel32 && r.Kind != RelCall {
continue
}
if r.Off >= 0 && r.Off+4 <= len(code) {
@@ -255,7 +288,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
}
}
nps = append(nps, npSym{
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, flag2: symFlag2Link, size: uint32(fn.Size)},
sym: goSym{name: name, abi: abi, typ: kindSTEXT, flag: flag, size: uint32(fn.Size)},
data: code,
})
}
@@ -285,7 +318,7 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
abi = symABIStatic
}
defIdx[d.Name] = len(defs)
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)})
defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, size: uint32(d.Size)})
defData = append(defData, img.Data[d.Offset:d.Offset+d.Size])
}
fnFiIdx := make([]int, len(img.Funcs))
@@ -320,6 +353,13 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
)
}
// Index the non-package TEXT definitions by short name for the internal
// call references.
textNpIdx := map[string]int{}
for i, fn := range img.Funcs {
textNpIdx[fn.Name] = fnNpIdx[i]
}
// Resolve external symbol references (cross-package). Build the
// package index table and determine each external symbol's SymIdx
// by reading the target package's export data.
@@ -327,10 +367,20 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
var extPkgIdx map[string]int
var extSymIdx map[string]int
if len(img.Externals) > 0 {
var err error
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(img.Externals)
if err != nil {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
// The morestack call is a builtin reference, not a resolved external.
var need []string
for _, n := range img.Externals {
if n == goobjBuiltinMorestack {
continue
}
need = append(need, n)
}
if len(need) > 0 {
var err error
extPkgTable, extPkgIdx, extSymIdx, err = resolveExternalSymbols(need)
if err != nil {
return nil, fmt.Errorf("GOOBJ emission: resolving external symbols: %w", err)
}
}
}
@@ -342,6 +392,31 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
si := len(defs) + fnNpIdx[i]
for _, r := range fn.Relocs {
typ, size := relocField(r)
if r.Kind == RelTLSLE {
// The TLS load has no symbol: {0, 0} is the nil ref.
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], 0)
binary.LittleEndian.PutUint32(rec[19:], 0)
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
if r.External && r.Name == goobjBuiltinMorestack {
// The stack-guard morestack call uses the toolchain's
// builtin reference.
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
if r.External {
// Split package-qualified name: "runtime·morestack" → runtime, morestack.
pkg, name := splitQualified(r.Name)
@@ -366,20 +441,83 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
symRelocs[si] = append(symRelocs[si], rec[:]...)
continue
}
pkg := uint32(pkgIdxSelf)
di, ok := defIdx[r.Name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
// A call to a TEXT function of the same file references the
// non-package definition table.
ni, isText := textNpIdx[r.Name]
if !isText || !isCallReloc(r.Kind) {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
pkg = pkgIdxNone
di = ni
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = size // field width
binary.LittleEndian.PutUint16(rec[5:], typ)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[15:], pkg)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
symRelocs[si] = append(symRelocs[si], rec[:]...)
}
}
// The data symbols' own relocations: the symbol-valued DATA fields
// ("DATA s+0(SB)/8, $other(SB)"). The toolchain patches each field
// with the target's absolute address through an R_ADDR of the DATA
// line's width, on every architecture (the code relocations are
// per-architecture PC-relative shapes; a data pointer word is not), so
// this mapping bypasses relocField. The definitions were appended in
// DataSyms order, so data symbol i is definition index i.
for i, d := range img.DataSyms {
for _, r := range d.Relocs {
if r.Kind != RelAddr {
return nil, fmt.Errorf("GOOBJ emission: data symbol %q carries a non-data relocation", d.Name)
}
var rec [23]byte
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
rec[4] = r.Siz
binary.LittleEndian.PutUint16(rec[5:], relocAddr)
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
switch {
case r.External && r.Name == goobjBuiltinMorestack:
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
case r.External:
pkg, name := splitQualified(r.Name)
if pkg == "" {
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
}
pIdx, ok := extPkgIdx[pkg]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
}
sIdx, ok := extSymIdx[pkg+"·"+name]
if !ok {
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
}
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
default:
if di, ok := defIdx[r.Name]; ok {
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
break
}
// A DATA field may hold the address of a TEXT function of
// the same file (the rt0 lib entry spelling), which is a
// non-package definition.
ni, isText := textNpIdx[r.Name]
if !isText {
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
}
binary.LittleEndian.PutUint32(rec[15:], pkgIdxNone)
binary.LittleEndian.PutUint32(rec[19:], uint32(ni))
}
symRelocs[i] = append(symRelocs[i], rec[:]...)
}
}
// The DWARF symbols' own relocations (the function address references).
for _, ds := range dwarfRelocs {
for _, r := range ds.relocs {
+73 -33
View File
@@ -40,8 +40,11 @@ func exportPath(importPath string) (string, error) {
//
// refs maps package import paths to the symbol names referenced from that
// package. The returned pkgIdx maps each import path to its position in
// the blkPkgIdx table (0-based), and symIdx gives each symbol's index within
// its package.
// the blkPkgIdx table, which reserves index 0 for the dummy invalid
// package (cmd/internal/obj/sym.go: "0 is invalid index"; the loader's
// reader loop starts at 1), so package i sits at block index i+1 and its
// relocations carry i+1. symIdx gives each symbol's index within its
// package.
func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symIdx map[string]int, err error) {
pkgIdx = make(map[string]int, len(refs))
symIdx = make(map[string]int)
@@ -50,7 +53,9 @@ func resolveExternalGOOBJ(refs map[string][]string) (pkgIdx map[string]int, symI
packages := sortedPkgRefs(refs)
for i, pkg := range packages {
pkgIdx[pkg.path] = i
// Block index 0 is the dummy invalid package; the first real
// package starts at 1.
pkgIdx[pkg.path] = i + 1
exp, err := exportPath(pkg.path)
if err != nil {
return nil, nil, err
@@ -84,7 +89,7 @@ func sortedPkgRefs(refs map[string][]string) []pkgRef {
for pkg, syms := range refs {
pkgs = append(pkgs, pkgRef{pkg, syms})
}
// Simple insertion sort — the list is tiny (usually 1–3 packages).
// Simple insertion sort, the list is tiny (usually 1-3 packages).
for i := 1; i < len(pkgs); i++ {
for j := i; j > 0 && pkgs[j-1].path > pkgs[j].path; j-- {
pkgs[j-1], pkgs[j] = pkgs[j], pkgs[j-1]
@@ -145,40 +150,65 @@ func parseArDecimal(b []byte) int {
}
// goobjFile is a parsed GOOBJ file: the string table and the symbol-definition
// block.
// blocks. The hashed blocks are kept raw: their symbols carry no names, only
// the loader needs their counts.
type goobjFile struct {
strTab []byte // string table, at headerSize + n
symdef []byte // blkSymdef raw block
npdef []byte // blkNonpkgdef raw block
strTab []byte // string table, at headerSize + n
symdef []byte // blkSymdef raw block
hashed64 []byte // blkHashed64def raw block
hashed []byte // blkHasheddef raw block
npdef []byte // blkNonpkgdef raw block
}
// symbols returns all symbol names in definition order by scanning the
// symdef and nonpkgdef blocks and resolving each name through the string
// table. Package definitions (blkSymdef) use fully-qualified names like
// "runtime.morestack"; non-package definitions (blkNonpkgdef) use bare
// names like "morestack". This combined list matches the index the
// linker expects for cross-package references.
// loaderIndexBase returns the index the first nonpkgdef symbol occupies in the
// loader's per-object symbol array. cmd/link lays the definition blocks out as
// symdef, hashed64def, hasheddef, nonpkgdef, nonpkgref (loader.go: preloadSyms
// fills r.syms in exactly that order, and resolve() indexes PkgIdxNone and
// cross-package SymIdx into it), so a symbol found in blkNonpkgdef carries the
// three leading blocks' symbol counts as its base.
func (f *goobjFile) loaderIndexBase() int {
return len(f.symdef)/recSymSize + len(f.hashed64)/recSymSize + len(f.hashed)/recSymSize
}
// symbols returns the names of the symdef and nonpkgdef blocks in
// definition order. Package definitions (blkSymdef) use fully-qualified
// names like "runtime.morestack"; non-package definitions (blkNonpkgdef)
// use bare names like "morestack". For lookups by index prefer
// findSymbol: it adds the hashed blocks' count the loader's array
// interleaves between the two.
func (f *goobjFile) symbols() []string {
return append(f.defNames(), f.npdefNames()...)
}
// findSymbol returns the index of a symbol within the combined symbol list,
// or -1 if not found. It first tries the fully-qualified name (pkg.name),
// then the bare name.
// findSymbol returns the index of a symbol within the loader's per-object
// symbol array, or -1 if not found. It first tries the fully-qualified
// name (pkg.name), then the bare name (assembly objects store dotless
// names, e.g. runtime's "gogo", for symbols other packages reach through
// a linkname).
func (f *goobjFile) findSymbol(pkg, name string) int {
base := f.loaderIndexBase()
qualified := pkg + "." + name
syms := f.symbols()
for i, s := range syms {
for i, s := range f.defNames() {
if s == qualified {
return i
}
}
// Try bare name (for non-package definitions).
for i, s := range syms {
for i, s := range f.npdefNames() {
if s == qualified {
return base + i
}
}
// Try bare name (for dotless assembly definitions).
for i, s := range f.defNames() {
if s == name {
return i
}
}
for i, s := range f.npdefNames() {
if s == name {
return base + i
}
}
return -1
}
@@ -192,12 +222,16 @@ func (f *goobjFile) npdefNames() []string {
return f.readSymNames(f.npdef)
}
// recSymSize is the size of one Sym record in the definition blocks
// (goobj.SymSize: stringRefSize + 2 + 1 + 1 + 1 + 4 + 4).
const recSymSize = 21
// readSymNames reads symbol names from a symdef/nonpkgdef block. Each record
// is 21 bytes: nameLen (u32), nameOff (u32), abi (u16), typ, flag, flag2,
// size (u32), align (u32). nameOff is an absolute offset into the string
// table.
func (f *goobjFile) readSymNames(block []byte) []string {
const recSize = 21
const recSize = recSymSize
if len(block) < recSize {
return nil
}
@@ -247,16 +281,18 @@ func parseGOOBJ(data []byte) (*goobjFile, error) {
// [16:20] flags
// [20:96] 19 × uint32 offsets
var offs [blkEnd + 1]uint32
for i := 0; i <= blkEnd; i++ {
for i := range blkEnd + 1 {
offs[i] = binary.LittleEndian.Uint32(payload[20+4*i:])
}
// The string table lives at headerSize.
strTabStart := uint32(goobjHeaderSize)
f := &goobjFile{
strTab: payload[strTabStart:offs[0]],
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
strTab: payload[strTabStart:offs[0]],
symdef: blockSlice(payload, offs, blkSymdef, blkSymdef+1),
hashed64: blockSlice(payload, offs, blkHashed64def, blkHashed64def+1),
hashed: blockSlice(payload, offs, blkHasheddef, blkHasheddef+1),
npdef: blockSlice(payload, offs, blkNonpkgdef, blkNonpkgdef+1),
}
return f, nil
}
@@ -306,22 +342,26 @@ func resolveExternalSymbols(externals []string) (pkgTable []string, pkgIdxMap ma
return nil, nil, nil, err
}
// Build the package table in pkgIdx order.
// Build the package table in pkgIdx order. The indices are 1-based
// (0 is the dummy invalid package, written by the emitter itself), so
// the table without the dummy is indexed one below.
pkgTable = make([]string, len(pkgIdx1))
for pkg, idx := range pkgIdx1 {
pkgTable[idx] = pkg
pkgTable[idx-1] = pkg
}
return pkgTable, pkgIdx1, symIdx1, nil
}
// splitQualified splits a qualified Go symbol name (pkgpath·name) into its
// package path and local name. The separator is the middle dot (U+00B7).
// If no separator is found, the symbol is assumed to be in the current
// package (empty pkg).
// package path and local name. The separator is the middle dot (U+00B7),
// whose UTF-8 encoding is two bytes, so the search must be string-based:
// IndexByte would match only the second byte and leave the lead byte on
// the package path. If no separator is found, the symbol is assumed to be
// in the current package (empty pkg).
func splitQualified(full string) (pkg, name string) {
if idx := strings.IndexByte(full, '\u00b7'); idx >= 0 {
return full[:idx], full[idx+len("\u00b7"):]
if before, after, ok := strings.Cut(full, "\u00b7"); ok {
return before, after
}
if before, after, ok := strings.Cut(full, "."); ok {
return before, after
+6 -2
View File
@@ -56,8 +56,12 @@ func TestResolveExternalSymbols(t *testing.T) {
if err != nil {
t.Fatalf("resolveExternalGOOBJ: %v", err)
}
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 0 {
t.Errorf("pkgIdx = %v, want runtime→0", pkgIdx)
if len(pkgIdx) != 1 || pkgIdx["runtime"] != 1 {
// Index 0 is the dummy invalid package in the blkPkgIdx table;
// the loader's reader loop starts at 1 (cmd/link/internal/
// loader/loader.go: "PkgIdx 0 is a dummy invalid package"), so
// the first real package must carry index 1.
t.Errorf("pkgIdx = %v, want runtime→1", pkgIdx)
}
if _, ok := symIdx["runtime·g0"]; !ok {
t.Errorf("symIdx missing runtime·g0, got %v", symIdx)
+192 -5
View File
@@ -117,7 +117,9 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
if len(defs) != 7 {
t.Fatalf("symdefs = %d, want 7", len(defs))
}
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link {
// The linkname flag stays clear: the toolchain sets it only for
// //go:linkname symbols, and an ordinary static GLOBL is not one.
if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != 0 {
t.Errorf("mask symbol = %+v", defs[0])
}
if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 {
@@ -158,8 +160,8 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09
t.Errorf("funcinfo bytes %x", fi)
}
// The pc-value tables of addq (non-package indices 0–3, so global
// indices 7–10): pcsp a flat zero over the whole function, pcinline a
// The pc-value tables of addq (non-package indices 0-3, so global
// indices 7-10): pcsp a flat zero over the whole function, pcinline a
// flat -1, both with the pc delta in MinLC (1) units.
pcsp := data[le.Uint32(didx[4*7:]):]
if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) {
@@ -294,7 +296,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0
}
for i := range wantPCs {
if pcs[i] != wantPCs[i] || vals[i] != wantVals[i] {
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d) — all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
t.Errorf("pcsp[%d] = (%d,%d), want (%d,%d); all: %v %v", i, pcs[i], vals[i], wantPCs[i], wantVals[i], pcs, vals)
}
}
// The last two steps unwind the epilogue to zero.
@@ -333,7 +335,7 @@ TEXT ·useext(SB), NOSPLIT, $0-8
// TestGOObjectLinkAndRun is the end-to-end check: assemble the test
// functions to a GOOBJ, swap it into a go build in place of the toolchain's
// assembly object, link, and run — the output must match the baseline
// assembly object, link, and run; the output must match the baseline
// binary the Go assembler produced. Skipped when no Go toolchain is
// available.
func TestGOObjectLinkAndRun(t *testing.T) {
@@ -511,3 +513,188 @@ func fieldAfter(line, flag string) string {
}
return ""
}
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
// GOOBJ whose code references a real external package symbol (runtime's
// morestack, a plain reference rather than the builtin noctxt form) must
// carry a package index that points past the blkPkgIdx table's dummy entry
// 0, and the object must link against the real runtime. Pre-fix, the
// relocations carried block index 0, which the loader never fills, so the
// reference resolved against whatever object was loaded first and the link
// failed. The binary is not run: morestack returns to the call site's
// stack check, which a hand-written caller has none of.
func TestGOObjectExternalPackageLink(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
const asmSrc = `
#include "textflag.h"
TEXT ·fn(SB), NOSPLIT, $0-0
CALL ·helper(SB)
RET
TEXT ·helper(SB), NOSPLIT, $0-0
RET
`
const mainSrc = `package main
func fn()
func helper()
func main() {
fn()
helper()
}
`
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module extlink\n\ngo 1.27\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the build the toolchain performs and re-run only its link
// step with our object swapped into the package archive, mirroring
// TestGOObjectLinkAndRun.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj, pkgArch string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || pkgArch == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work)
// Assemble the source with gasm, then retarget fn's internal call at
// a real external package symbol: the reloc's qualified name drives
// the export-data resolution the way a source-level runtime·sym(SB)
// reference would.
f, errs := parser.Parse("main_amd64.s", asmSrc)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
fn := &img.Funcs[0]
for i := range fn.Relocs {
fn.Relocs[i].Name = "runtime\u00b7morestack"
fn.Relocs[i].External = true
}
img.Externals = []string{"runtime\u00b7morestack"}
obj, err := img.GOObject("main", "main_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
// Structural check: the blkPkgIdx block reserves entry 0 for the
// dummy invalid package and places runtime at entry 1, and fn's call
// relocation carries PkgIdx 1.
v := openGoobj(t, obj)
pkgBlk := v.blk(blkPkgIdx)
if len(pkgBlk) != 2*8 {
t.Fatalf("blkPkgIdx = %d bytes, want two entries", len(pkgBlk))
}
le := binary.LittleEndian
strEntry := func(i int) string {
e := pkgBlk[i*8 : (i+1)*8]
return v.str(le.Uint32(e[4:]), le.Uint32(e[0:]))
}
if s := strEntry(0); s != "" {
t.Errorf("blkPkgIdx[0] = %q, want the dummy empty package", s)
}
if s := strEntry(1); s != "runtime" {
t.Errorf("blkPkgIdx[1] = %q, want runtime", s)
}
relocs := v.blk(blkReloc)
// fn is the last non-package symbol (two functions, four pc tables
// each); its one reloc is the final record.
fnRec := relocs[len(relocs)-23:]
if pIdx := le.Uint32(fnRec[15:]); pIdx != 1 {
t.Errorf("external reloc PkgIdx = %d, want 1 (runtime)", pIdx)
}
// Swap the object into the package archive and link with cmd/link;
// the link line consumes the archive, not the loose object file.
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", pkgArch)
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(asmObj))
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, obj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, filepath.Join(membersDir, m))
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, pkgArch, newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "prog2"))
linkCmd := exec.Command("sh", "-c", "cd "+dir+" && "+linkLine)
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// The call must have resolved to the real runtime symbol.
dump, err := exec.Command(goBin, "tool", "objdump", "-s", "main.fn", filepath.Join(dir, "prog2")).CombinedOutput()
if err != nil {
t.Fatalf("objdump main.fn: %v\n%s", err, dump)
}
if !bytes.Contains(dump, []byte("runtime.morestack")) {
t.Errorf("main.fn does not call runtime.morestack:\n%s", dump)
}
}
+35 -9
View File
@@ -13,28 +13,54 @@ import (
)
// GOObjectAARCH64 emits a GOOBJ object file for AArch64. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header
// the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and R_ADDRARM64 relocation types for the
// ADRP+ADD/LDR/STR address pairs.
// reloc/aux/data index arrays, with the arm64 preamble, the MinLC of 4
// for the pc-value deltas, and the arm64 relocation types for the ADRP
// pairs and BL calls.
//
// The toolchain records one relocation per ADRP pair: a single R_ADDRARM64
// or R_ARM64_PCREL_LDST64 of Siz 8 at the ADRP word, from which the linker
// patches both instructions of the pair (cmd/internal/obj/arm64/asm7.go,
// the ADRP cases: one AddRel with Off at the pair's pc and Siz 8). gasm's
// assembler records the ADRP+ADD form as two word relocs, so the second
// word's twin is dropped here before emission.
func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
pre, err := toolchainObjectPreambleAARCH64()
if err != nil {
return nil, err
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
if r.Kind == RelArm64Branch {
coalesced := *img
coalesced.Funcs = append([]FuncLayout(nil), img.Funcs...)
for i := range coalesced.Funcs {
rs := coalesced.Funcs[i].Relocs
var keep []Reloc
for j := 0; j < len(rs); j++ {
keep = append(keep, rs[j])
if rs[j].Kind == RelArm64Addr && j+1 < len(rs) &&
rs[j+1].Kind == RelArm64Addr && rs[j+1].Off == rs[j].Off+4 {
j++ // the ADD word's twin: the Siz-8 pair reloc covers it
}
}
coalesced.Funcs[i].Relocs = keep
}
return coalesced.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
switch r.Kind {
case RelArm64Branch:
return relocArm64Branch, 4
case RelArm64LDST64:
return relocArm64LDST64, 8
default:
return relocArm64Addr, 8
}
return relocArm64Addr, 4
})
}
// arm64 relocation types (cmd/internal/objabi).
const (
relocArm64Addr = 3 // R_ADDRARM64 — ADRP+ADD/LDR/STR pair
relocArm64Branch = 9 // R_CALLARM64 — BL instruction
relocArm64Addr = 3 // R_ADDRARM64, ADRP+ADD pair
relocArm64Branch = 9 // R_CALLARM64, BL instruction
relocArm64LDST64 = 40 // R_ARM64_PCREL_LDST64, ADRP+LDR/STR pair
)
// toolchainObjectPreambleAARCH64 returns the "go object ...\n!\n" header
+11 -5
View File
@@ -13,9 +13,9 @@ import (
)
// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is
// the shared one in goobj.go — the toolchain preamble, the go120ld header
// the shared one in goobj.go, the toolchain preamble, the go120ld header
// with its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4
// reloc/aux/data index arrays, with the loong64 preamble, the MinLC of 4
// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for
// the pcalau12i+addi.d address pairs.
func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
@@ -25,11 +25,16 @@ func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
}
return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) (uint16, uint8) {
// A pcalau12i+addi.d pair: the high part carries
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO.
if r.Kind == RelLoong64AddrLo {
// R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO; the guard's
// morestack call carries R_CALLLOONG64.
switch {
case r.Kind == RelLoong64AddrLo:
return relocLoong64AddrLo, 4
case r.Kind == RelLoong64Branch:
return relocCallLoong64, 4
default:
return relocLoong64AddrHi, 4
}
return relocLoong64AddrHi, 4
})
}
@@ -39,6 +44,7 @@ func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) {
const (
relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI
relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO
relocCallLoong64 = 84 // R_CALLLOONG64
)
// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header
+2 -2
View File
@@ -13,9 +13,9 @@ import (
)
// GOObjectRISCV emits a GOOBJ object file for RISC-V. The layout is the
// shared one in goobj.go — the toolchain preamble, the go120ld header with
// shared one in goobj.go, the toolchain preamble, the go120ld header with
// its block offsets, the string table, the symbol definitions and the
// reloc/aux/data index arrays — with the RISC-V preamble, the MinLC of 2 for
// reloc/aux/data index arrays, with the RISC-V preamble, the MinLC of 2 for
// the pc-value deltas, and the single R_RISCV_PCREL_ITYPE/STYPE relocation
// per AUIPC pair, matching `go tool asm`'s model (each pair is one 8-byte
// relocation, not the ELF HI20/LO12 pair).
+329
View File
@@ -0,0 +1,329 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
// verified with go tool objdump): the stack-split guard classes, the morestack
// block and the auto-NOSPLIT leaf behaviour.
func TestStackGuardBytes(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"554889e54883ec104883c4105dc3"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"644c8b3425000000004c8da42478ffffff4d3b66107614554889e54881ec000100004881c4000100005dc3e800000000ebce"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"644c8b3425000000004989e44981ec881f0000721a4d3b66107614554889e54881ec002000004881c4002000005dc3e800000000ebca"},
// Class 2 with a body long enough that the underflow JB relaxes to
// rel32: its displacement must span the real 6-byte JB, else the
// branch lands 4 bytes past the morestack block, inside the CALL
// displacement field.
{"leafbiglong", "TEXT \u00b7leafbiglong(SB), $8192-0\n" + strings.Repeat("\tMOVQ AX, BX\n", 40) + "\tRET\n",
"644c8b3425000000004989e44981ec881f00000f82960000004d3b66100f868c000000554889e54881ec00200000" + strings.Repeat("4889c3", 40) + "4881c4002000005dc3e800000000e947ffffff"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"644c8b342500000000493b66107613554889e54883ec10e8000000004883c4105dc3e800000000ebd7"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"554889e54883ec104883c4105dc3"},
} {
f, errs := parser.Parse("g_amd64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
// The toolchain's object leaves every relocation field zero for the
// linker, while the gasm image resolves file-internal references, so
// the comparison masks the patch sites the way verify's ground truth
// does.
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardRelocs checks the guard's patch sites: the TLS slot and the
// morestack call.
func TestStackGuardRelocs(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
relocs := img.Funcs[0].Relocs
if len(relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(relocs))
}
tls, call := relocs[0], relocs[1]
if tls.Kind != RelTLSLE || tls.Off != 5 || tls.Name != "" || tls.External {
t.Errorf("tls reloc = %+v, want RelTLSLE at 5 with no symbol", tls)
}
if call.Kind != RelCall || call.Name != "runtime\u00b7morestack_noctxt" || !call.External {
t.Errorf("call reloc = %+v, want RelCall to runtime.morestack_noctxt", call)
}
}
// TestStackGuardGOObj emissions succeed with the guard's TLS and builtin
// references in play.
func TestStackGuardGOObj(t *testing.T) {
f, errs := parser.Parse("g_amd64.s", "TEXT \u00b7f(SB), $256-0\n\tCALL \u00b7helper(SB)\n\tRET\nTEXT \u00b7helper(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
obj, err := img.GOObject("testpkg", "g_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if !bytes.Contains(obj, []byte("go120ld")) {
t.Fatal("object lacks the GOOBJ magic")
}
}
// The arm64 stack-split guard, pinned from `go tool asm` (Go 1.27, arm64):
// the guard classes, the auto-NOSPLIT leaf behaviour and the morestack
// block. Relocation fields are masked: the toolchain's object leaves them
// zero for the linker, the gasm image resolves file-internal references.
func TestStackGuardBytesARM64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"900b40f9f14302d13f0210eb09010054f44304d19dfa3fa99f020091fd2300d1fd230491ff430491c0035fd6e3031eaa00000000f3ffff17"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"900b40f91bf283d2f1633beba30100543f0210eb690100541b0284d2f4633bcb9dfa3fa99f020091fd2300d11b0184d2fd633b8b1b0284d2ff633b8bc0035fd6e3031eaa00000000eeffff17"},
{"callsmall", "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"900b40f9ff6330eb09010054fe0f1ef8fd831ff8fd2300d100000000fd835ff8fe0742f8c0035fd6e3031eaa00000000f4ffff17"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"fe0f1ef8fd831ff8fd2300d1fd630091ff830091c0035fd6"},
} {
f, errs := parser.Parse("g_arm64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardBranchTargetsARM64 checks the class-2 guard's branch
// positions for a frame whose guard constant needs two MOV words: the
// displacements must be computed from byte offsets (8+4*ml and 16+4*ml), so
// both branches land on the morestack block rather than inside the body.
// The frame size makes the toolchain switch its own prologue decomposition,
// so the assertion is on the branch targets, not pinned bytes.
func TestStackGuardBranchTargetsARM64(t *testing.T) {
f, errs := parser.Parse("g_arm64.s", "TEXT \u00b7f(SB), $65664-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if len(code)%4 != 0 {
t.Fatalf("function size %d is not a word multiple", len(code))
}
// autosize = 65680, so the guard materialises 65552 = MOVZ+MOVK: ml = 2
// and the branches sit at bytes 16 and 24 of the guard prefix.
const morestackBlock = 12 // MOVD R30, R3; BL; B back
blockStart := len(code) - morestackBlock
check := func(name string, off int) {
t.Helper()
w := leWord(code[off:])
imm19 := int32(w>>5) & 0x7FFFF
if imm19&(1<<18) != 0 {
imm19 -= 1 << 19
}
if target := off + int(imm19)*4; target != blockStart {
t.Errorf("%s at byte %d targets byte %d, want the morestack block at %d", name, off, target, blockStart)
}
}
check("B.LO", 16)
check("B.LS", 24)
}
// The riscv64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// riscv64): the morestack call sits between the guard and the body, and the
// guard branches forward over it. Relocation fields are masked.
func TestStackGuardBytesRISCV64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"03b30d0163662300000000006ff05fff233411fe211106e08260610167800000"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"03b30d01930381f763667300000000006ff01fff233c11ee130181ef06e082601301811067800000"},
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"03b30d0189639b8383f863697100f97f9b8f8f07b303f10163667300000000006ff01ffef97f8a9f23bc1ffef97fe13f7e9106e08260896fa12f7e9167800000"},
{"frameless", "TEXT \u00b7frameless(SB), $0-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"03b30d0163662300000000006ff05fff233c11fe611106e0000000008260210167800000"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"233411fe211106e08260610167800000"},
} {
f, errs := parser.Parse("g_riscv64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// The loong64 stack-split guard, pinned from `go tool asm` (Go 1.27,
// loong64): every guard class (including the medium class with the
// materialised constant and the big class with the ORI-less constants), the
// auto-NOSPLIT leaf behaviour, the large-frame R30 prologue/epilogue forms
// and the morestack block. Relocation fields are masked.
func TestStackGuardBytesLOONG64(t *testing.T) {
for _, tt := range []struct {
name string
src string
want string
}{
{"leafsmall", "TEXT \u00b7leafsmall(SB), $16-0\n\tRET\n",
"61a0ff2963a0ff026100c0296360c0022000004c"},
{"leafmed", "TEXT \u00b7leafmed(SB), $256-0\n\tRET\n",
"d442c02878e0fd0294e21200801a004061e0fb2963e0fb026100c0296320c4022000004c3f00150000000000ffd7ff53"},
{"nosplit", "TEXT \u00b7nosplit(SB), NOSPLIT, $16-0\n\tRET\n",
"61a0ff2963a0ff026100c0296360c0022000004c"},
// The LR store leaves the 12-bit store-offset range while the SP
// adjust immediate still fits, and the epilogue adjusts through a
// single ORI.
{"fit2048", "TEXT \u00b7fit2048(SB), $2040-0\n\tRET\n",
"d442c0287800e20294e21200802600401e000014de8f1000c103e0296300e0026100c0291e00a00363f810002000004c3f00150000000000ffcbff53"},
// Medium class at the materialisation boundary (off = 2048 still
// immediate, 2049+ goes through R30).
{"med2048off", "TEXT \u00b7med2048off(SB), $2168-0\n\tRET\n",
"d442c0287800e00294e21200802e0040feffff15de8f1000c103de29feffff15de039e0363f810006100c0291e00a20363f810002000004c3f00150000000000ffc3ff53"},
{"medmat", "TEXT \u00b7medmat(SB), $2176-0\n\tRET\n",
"d442c028feffff15dee39f0378f8100094e21200802e0040feffff15de8f1000c1e3dd29feffff15dee39d0363f810006100c0291e20a20363f810002000004c3f00150000000000ffbbff53"},
// Big class with the rounding-split store and the floor-split adjust.
{"leafbig", "TEXT \u00b7leafbig(SB), $8192-0\n\tRET\n",
"d442c0283e000014de23be0378f8120000470044deffff15dee3810378f8100094e2120080320040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c0295e000014de23800363f810002000004c3f00150000000000ffa7ff53"},
// Zero low 12 bits drop the ORI from the store, the adjust and the
// epilogue materialisation.
{"bigzero", "TEXT \u00b7bigzero(SB), $4088-0\n\tRET\n",
"d442c028feffff15de03820378f8100094e21200802a0040feffff15de8f1000c103c029feffff1563f810006100c0293e00001463f810002000004c3f00150000000000ffbfff53"},
// Big class whose first constant has a zero high part: a single ORI.
{"big3976", "TEXT \u00b7big3976(SB), $4096-0\n\tRET\n",
"d442c0281e20be0378f8120000470044feffff15dee3810378f8100094e2120080320040feffff15de8f1000c1e3ff29deffff15dee3bf0363f810006100c0293e000014de23800363f810002000004c3f00150000000000ffabff53"},
// Big class at a multiple of 4096: both guard constants lose their
// ORI word.
{"giantlo0", "TEXT \u00b7giantlo0(SB), $4216-0\n\tRET\n",
"d442c0283e00001478f8120000430044feffff1578f8100094e2120080320040feffff15de8f1000c103fe29deffff15de03be0363f810006100c0293e000014de03820363f810002000004c3f00150000000000ffafff53"},
// Non-leaf big frame: the body call plus the LR restore epilogue.
{"callbig", "TEXT \u00b7callbig(SB), $8192-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n",
"d442c0283e000014de23be0378f81200004f0044deffff15dee3810378f8100094e21200803a0040deffff15de8f1000c1e3ff29beffff15dee3bf0363f810006100c029000000006100c0285e000014de23800363f810002000004c3f00150000000000ff9fff53"},
} {
f, errs := parser.Parse("g_loong64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.name, errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.name, err)
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
for _, r := range fn.Relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
got := hex.EncodeToString(code)
if got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.name, got, tt.want)
}
}
}
// TestStackGuardGOObjInternalCall checks that GOOBJ emission succeeds when a
// guarded function calls a TEXT symbol of the same file, for every arch's
// call relocation kind.
func TestStackGuardGOObjInternalCall(t *testing.T) {
for _, tt := range []struct {
src string
assemble func(*ast.File, ...AssembleOption) (*Image, error)
}{
{"g_amd64.s", AssembleFile},
{"g_arm64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileARM64(f) }},
{"g_riscv64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileRISCV(f) }},
{"g_loong64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileLOONG64(f) }},
} {
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("%s: parse: %v", tt.src, errs)
}
img, err := tt.assemble(f)
if err != nil {
t.Fatalf("%s: assemble: %v", tt.src, err)
}
if _, err := img.GOObject("testpkg", tt.src); err != nil {
t.Errorf("%s: GOObject: %v", tt.src, err)
}
}
}
+931 -76
View File
File diff suppressed because it is too large Load Diff
+3 -3
View File
@@ -19,8 +19,8 @@ import (
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel —
// all functions plus the file-local mask24 constant — and checks that every
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
// all functions plus the file-local mask24 constant; and checks that every
// static-symbol load resolves to the right bytes in the image.
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s"
@@ -81,7 +81,7 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
}
// TestAssembleGoFlacAVX512Kernel assembles the whole production AVX-512
// kernel — all functions plus the file-global idx16 constant — and checks
// kernel, all functions plus the file-global idx16 constant, and checks
// that the static-symbol load resolves to the right bytes in the image.
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s"
+219
View File
@@ -0,0 +1,219 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"runtime"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// The differential kernels for the DATA-path and front-end gaps are kept in
// testdata/verify beside the campaign's other kernels; the verify package's
// suites are not open to the asm package, so this test is their runner: each
// kernel assembles through gasm and through go tool asm, and the functions'
// bytes must agree with the relocation sites masked on both sides.
// toolAsmObject assembles path with the installed toolchain's assembler for
// goarch ("" = the host) and returns the object bytes.
func toolAsmObject(t *testing.T, path, goarch string) []byte {
t.Helper()
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
out, err := exec.Command(goBin, "env", "GOROOT").Output()
if err != nil {
t.Fatalf("go env GOROOT: %v", err)
}
includeDir := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
pkg := strings.TrimSuffix(filepath.Base(path), ".s")
pkg = strings.TrimSuffix(pkg, "_amd64")
pkg = strings.TrimSuffix(pkg, "_arm64")
objPath := filepath.Join(t.TempDir(), "oracle.o")
cmd := exec.Command(goBin, "tool", "asm", "-I", includeDir, "-p", pkg, "-o", objPath, path)
if goarch != "" {
environ := os.Environ()
env := make([]string, 0, len(environ)+1)
for _, e := range environ {
if !strings.HasPrefix(e, "GOARCH=") {
env = append(env, e)
}
}
cmd.Env = append(env, "GOARCH="+goarch)
}
if out, err := cmd.CombinedOutput(); err != nil {
t.Fatalf("go tool asm %s: %v\n%s", filepath.Base(path), err, out)
}
obj, err := os.ReadFile(objPath)
if err != nil {
t.Fatal(err)
}
return obj
}
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a
// toolchain object, keyed by the name the object records (pkg.name). Each
// function's span is its own symbol size: a toolchain object that follows
// the text with data symbols (the synthesised float-constant pool) would
// otherwise fold them into the last function's bytes.
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
t.Helper()
v := openGoobj(t, obj)
le := binary.LittleEndian
const symSize = 21
nps := v.syms(blkNonpkgdef)
data := v.blk(blkData)
didx := v.blk(blkDataIdx)
preceding := 0
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
preceding += len(v.blk(bi)) / symSize
}
out := make(map[string][]byte, len(nps))
for i, s := range nps {
if s.typ != kindSTEXT {
continue
}
start := le.Uint32(didx[4*(preceding+i):])
out[s.name] = data[start : start+s.size]
}
return out
}
// maskCode zeroes every relocation field, the way the toolchain's object
// leaves them for the linker.
func maskCode(code []byte, relocs []Reloc) []byte {
for _, r := range relocs {
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
code[j] = 0
}
}
return code
}
// code assembles src for amd64 and returns the image's code bytes.
func code(path, src string) []byte {
f, errs := parser.Parse(path, src)
if len(errs) > 0 {
return nil
}
img, err := AssembleFile(f)
if err != nil {
return nil
}
return img.Code
}
// TestDifferentialKernels pins the new kernels against the oracle.
func TestDifferentialKernels(t *testing.T) {
if runtime.GOARCH != "amd64" {
t.Skip("the amd64 kernels assume an amd64 host assembler default")
}
for _, k := range []struct {
path string
goarch string
arm64 bool
}{
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
} {
t.Run(filepath.Base(k.path), func(t *testing.T) {
src, err := os.ReadFile(k.path)
if err != nil {
t.Fatalf("read: %v", err)
}
f, errs := parser.Parse(k.path, string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
var img *Image
if k.arm64 {
img, err = AssembleFileARM64(f)
} else {
img, err = AssembleFile(f)
}
if err != nil {
t.Fatalf("assemble: %v", err)
}
gt := oracleFuncCode(t, toolAsmObject(t, k.path, k.goarch))
// The oracle keys its functions by the qualified object name
// (pkg.name); match on the local part.
byLocal := make(map[string][]byte, len(gt))
for name, code := range gt {
if _, after, ok := strings.Cut(name, "."); ok {
name = after
}
byLocal[name] = code
}
matched := 0
for _, fn := range img.Funcs {
gasmCode := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
goCode, ok := byLocal[fn.Name]
if !ok {
t.Errorf("%s: not in ground truth (%d functions: %v)", fn.Name, len(gt), keysOf(byLocal))
continue
}
goCode = maskCode(append([]byte(nil), goCode...), fn.Relocs)
cmpLen := min(len(goCode), len(gasmCode))
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\ngasm %x\ngo %x", fn.Name, len(gasmCode), len(goCode), gasmCode, goCode)
continue
}
for _, b := range goCode[len(gasmCode):] {
if b != 0 {
t.Errorf("%s: non-zero trailing bytes in go tool asm output", fn.Name)
break
}
}
matched++
t.Logf("%s: MATCH (%d bytes)", fn.Name, len(gasmCode))
}
if matched == 0 {
t.Fatal("no functions matched")
}
})
}
}
func keysOf(m map[string][]byte) []string {
out := make([]string, 0, len(m))
for k := range m {
out = append(out, k)
}
return out
}
// TestSemicolonSpellingParity pins that the ';' statement separator changes
// nothing about the encoding: the one-line spelling assembles to exactly the
// bytes of the same statements written one per line.
func TestSemicolonSpellingParity(t *testing.T) {
for _, tt := range []struct{ one, two string }{
{"\tROLQ $3, DI; ROLQ $13, DI\n", "\tROLQ $3, DI\n\tROLQ $13, DI\n"},
{"\tREP; MOVSQ\n", "\tREP\n\tMOVSQ\n"},
{"\tXORQ AX, AX; XORQ CX, CX\n", "\tXORQ AX, AX\n\tXORQ CX, CX\n"},
} {
one := code("t.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n"+tt.one+"\tRET\n")
two := code("t.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n"+tt.two+"\tRET\n")
if !bytes.Equal(one, two) {
t.Errorf("semicolon spelling %q: %x, want the two-line bytes %x", tt.one, one, two)
}
}
}
+2 -2
View File
@@ -94,9 +94,9 @@ DATA ·table<>+0(SB)/8, $0x1122334455667788
}
// The debug_line program: LNE_set_address (the R_ADDR relocation
// carries the function address), then one row per line change — the
// carries the function address), then one row per line change; the
// TEXT is on line 4 (a leading blank line precedes the include), the
// instructions on lines 5–9 — an advance to the 20-byte end and an
// instructions on lines 5-9; an advance to the 20-byte end and an
// end-of-sequence.
linesOff := le.Uint32(dataIdx[4*2:])
lines := dataBlk[linesOff : linesOff+21]
+280 -75
View File
@@ -5,7 +5,9 @@ package asm
import (
"fmt"
"math"
"sort"
"strconv"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
@@ -15,7 +17,7 @@ import (
// file-local static symbols are encoded RIP-relative and resolved within the
// image, so the raw bytes are self-consistent and executable at any base
// address; references to external symbols are recorded as relocations
// (Funcs[i].Relocs, Externals) and left unresolved — the object-file
// (Funcs[i].Relocs, Externals) and left unresolved, the object-file
// emitters turn them into linker relocations.
type Image struct {
Code []byte // concatenated function bodies
@@ -24,6 +26,10 @@ type Image struct {
Symbols map[string]int // static symbol → byte offset within the image
DataSyms []DataSymbol // GLOBL symbols, in layout order
Externals []string // referenced but undefined symbols, sorted
// SourcePath is the assembled file's path, recorded in the DWARF
// sections in place of a placeholder name. Empty when the image was
// not built from a named file.
SourcePath string
}
// FuncLayout describes one assembled function within an Image.
@@ -80,33 +86,45 @@ func (fl *FuncLayout) LineAt(offset int) int {
return 0
}
// RelocKind Reloc is one static-symbol reference within a function body: the disp32
// field at Off (function-relative) must reach the symbol plus Addend,
// measured from After, the address just past the instruction. An External
// relocation names a symbol no GLOBL in the file defines; the object-file
// emitters carry it into the output's relocation table.
// RelocKind discriminates the type of relocation needed.
// RelocKind discriminates the relocation a static-symbol reference needs;
// the encoders record one per SB reference, and the object-file emitters map
// it to their format's relocation type.
type RelocKind int
const (
RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64)
RelCall // R_CALL: CALL to a function symbol (amd64)
RelTLSLE // R_TLS_LE: local-exec TLS load, no symbol (amd64 guard)
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
RelRISCVJal // R_RISCV_JAL (J-type call)
RelPCRelAbs // 32-bit absolute (R_RISCV_32)
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD/LDR/STR pair)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
RelArm64Branch // R_CALLARM64 (BL instruction)
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
RelAddr // R_ADDR: the absolute address of a symbol held in a DATA field
)
type Reloc struct {
// Off is the function-relative offset of the field the linker patches
// and After the address just past the instruction, the base the
// assembler measures PC-relative displacements from. Name plus
// Addend select the target: the symbol plus the byte offset. An
// External relocation names a symbol no GLOBL in the file defines;
// the object-file emitters carry it into the output's relocation
// table. Siz is the width of the patched field and is set only for
// data-field relocations (RelAddr, Off relative to the data symbol),
// whose width is the DATA line's; code relocations take their width
// from the architecture's instruction encoding.
Off int
After int
Name string
Addend int64
External bool
Kind RelocKind
Siz uint8
}
// DataSymbol describes one GLOBL symbol laid out in the data section.
@@ -118,6 +136,11 @@ type DataSymbol struct {
Static bool // the <> marker: file-local, not exported
Rodata bool // the RODATA flag: read-only data
Dupok bool // the DUPOK flag: duplicate-OK
// Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8,
// $other(SB)"): fields of this symbol's data that hold another symbol's
// address, resolved by the linker. Off is relative to the symbol's
// data start.
Relocs []Reloc
}
// Bytes returns the whole image: code, then data.
@@ -127,14 +150,26 @@ func (img *Image) Bytes() []byte {
return append(out, img.Data...)
}
// AssembleOption adjusts the file-level assembly context.
type AssembleOption func(*linkInfo)
// WithGOOS selects the target operating system for the forms that depend on
// it, the TLS access shape above all: linux and freebsd take the
// one-instruction form, windows and plan9 keep the two-instruction load.
func WithGOOS(goos string) AssembleOption {
return func(l *linkInfo) {
l.goos = goos
}
}
// AssembleFile assembles every TEXT function of a parsed file and lays out
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
// reference to a file-local static symbol becomes a RIP-relative load whose
// displacement is resolved against that layout; a reference to a symbol no
// GLOBL defines is recorded as an external relocation (Externals) with its
// displacement left zero — the object-file emitters resolve it at link
// displacement left zero, the object-file emitters resolve it at link
// time, while the raw image (Bytes) cannot represent it.
func AssembleFile(f *ast.File) (*Image, error) {
func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
dataSyms, err := collectData(f)
if err != nil {
return nil, err
@@ -143,9 +178,21 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, d := range dataSyms {
known[d.name] = true
}
// TEXT symbols are file-level definitions too: a symbol immediate
// ($fn(SB)) may name one, exactly as a data reference names a GLOBL.
for _, d := range f.Decls {
if t, ok := d.(*ast.Text); ok {
known[t.Name.Name] = true
}
}
link := &linkInfo{symbols: known, allowExternal: true}
for _, o := range opts {
o(link)
}
poolSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
textOff := map[string]int{}
type asmFunc struct {
name string
patches []sbPatch
@@ -156,7 +203,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
if !ok {
continue
}
code, patches, labels, steps, lines, err := assemble(t, link)
code, patches, labels, steps, lines, pool, err := assemble(t, link)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
// The pooled floating-point constants join the declared data as
// read-only symbols, deduplicated across the file (the toolchain
// synthesises the same symbols into its rodata).
for _, entry := range pool {
if poolSeen[entry.name] {
continue
}
poolSeen[entry.name] = true
dataSyms = append(dataSyms, dataSym{
name: entry.name,
buf: entry.data,
size: len(entry.data),
rodata: true,
dupok: true,
})
}
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
@@ -183,6 +249,7 @@ func AssembleFile(f *ast.File) (*Image, error) {
for _, s := range steps {
fl.Spadj = append(fl.Spadj, SpadjStep{PC: s.pc, Value: s.value})
}
textOff[t.Name.Name] = len(img.Code)
img.Funcs = append(img.Funcs, fl)
img.Code = append(img.Code, code...)
funcs = append(funcs, asmFunc{name: t.Name.Name, patches: patches})
@@ -215,13 +282,27 @@ func AssembleFile(f *ast.File) (*Image, error) {
base := img.Funcs[i].Offset
code := img.Code[base : base+img.Funcs[i].Size]
for _, p := range fn.patches {
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend}
reloc := Reloc{Off: p.off, After: p.after, Name: p.name, Addend: p.addend, Kind: p.kind}
if p.kind == RelTLSLE {
// The TLS slot has no symbol: the linker fills the offset
// from the runtime's TLS layout.
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
continue
}
if imgOff, ok := img.Symbols[p.name]; ok {
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else if imgOff, ok := textOff[p.name]; ok {
// A CALL to a TEXT function of the same file: resolve the
// displacement against the function's layout position.
rel := int64(imgOff) + p.addend - int64(base+p.after)
if rel < -1<<31 || rel >= 1<<31 {
return nil, fmt.Errorf("%s: displacement to %q out of rel32 range", fn.name, p.name)
}
copy(code[p.off:p.off+4], le32(rel))
} else {
reloc.External = true
externals[p.name] = true
@@ -229,6 +310,22 @@ func AssembleFile(f *ast.File) (*Image, error) {
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
}
}
// The data symbols' symbol-valued DATA fields resolve the same way the
// code references do: a name the file defines (GLOBL or TEXT) stays an
// internal reference the emitters resolve, anything else is external.
// img.DataSyms was laid out in dataSyms order, so the indexes line up.
for i := range img.DataSyms {
for _, r := range dataSyms[i].relocs {
reloc := r
if _, ok := img.Symbols[reloc.Name]; !ok {
if _, ok := textOff[reloc.Name]; !ok {
reloc.External = true
externals[reloc.Name] = true
}
}
img.DataSyms[i].Relocs = append(img.DataSyms[i].Relocs, reloc)
}
}
for name := range externals {
img.Externals = append(img.Externals, name)
}
@@ -245,17 +342,34 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
if err != nil {
return nil, err
}
// The pooled $i64 constants the wide MOV immediate loads refer to join
// the declared data as read-only symbols, deduplicated across the file
// (the toolchain synthesises the same symbols into its rodata).
litSeen := map[string]bool{}
img := &Image{Symbols: map[string]int{}}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
continue
}
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
code, labels, relocs, lines, spadj, lits, err := assembleRISCV(t)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
for _, lit := range lits {
if litSeen[lit.Name] {
continue
}
litSeen[lit.Name] = true
dataSyms = append(dataSyms, dataSym{
name: lit.Name,
buf: lit.Data,
size: len(lit.Data),
rodata: true,
dupok: true,
})
}
fl := FuncLayout{
Name: t.Name.Name,
Pkg: t.Name.Pkg,
@@ -318,7 +432,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
return nil, err
}
img := &Image{Symbols: map[string]int{}}
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
@@ -401,6 +515,23 @@ func markExternals(img *Image, dataSyms []dataSym) {
}
}
}
// The declared data symbols carry the file's own relocations (the
// symbol-valued DATA fields); the layouts appended img.DataSyms in
// dataSyms order, so the indexes line up. The trailing entries (the
// pooled arm64 literals) have no source relocations.
for i := range img.DataSyms {
if i >= len(dataSyms) {
break
}
for _, r := range dataSyms[i].relocs {
reloc := r
if !known[reloc.Name] {
reloc.External = true
externals[reloc.Name] = true
}
img.DataSyms[i].Relocs = append(img.DataSyms[i].Relocs, reloc)
}
}
for name := range externals {
img.Externals = append(img.Externals, name)
}
@@ -416,81 +547,155 @@ type dataSym struct {
static bool
rodata bool
dupok bool
// relocs are the symbol-valued DATA fields, in declaration order; Off
// is relative to the symbol's data start.
relocs []Reloc
}
// collectData gathers the file's static symbols (GLOBL) and their initial
// contents (DATA) into byte buffers, in declaration order.
// contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts
// every DATA line before its symbol's GLOBL, so the symbols are registered
// before the initialisers are applied.
func collectData(f *ast.File) ([]dataSym, error) {
index := map[string]int{}
var syms []dataSym
for _, d := range f.Decls {
switch dd := d.(type) {
case *ast.Globl:
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
name := dd.Name.Name
if _, dup := index[name]; dup {
return nil, fmt.Errorf("duplicate GLOBL %q", name)
}
size := 0
if dd.Size != nil && dd.Size.Imm.HasVal {
size = int(dd.Size.Imm.Val)
}
index[name] = len(syms)
ds := dataSym{
name: name,
pkg: dd.Name.Pkg,
buf: make([]byte, size),
size: size,
static: dd.Name.Static,
}
for _, f := range dd.Flags {
switch f {
case "RODATA":
ds.rodata = true
case "DUPOK":
ds.dupok = true
case "1":
ds.dupok = true
case "8":
ds.rodata = true
case "9":
ds.dupok = true
ds.rodata = true
gd, ok := d.(*ast.Globl)
if !ok {
continue
}
if gd.Name == nil || gd.Name.Pseudo != "SB" {
continue
}
name := gd.Name.Name
if _, dup := index[name]; dup {
return nil, fmt.Errorf("duplicate GLOBL %q", name)
}
size := 0
if gd.Size != nil && gd.Size.Imm.HasVal {
size = int(gd.Size.Imm.Val)
}
index[name] = len(syms)
ds := dataSym{
name: name,
pkg: gd.Name.Pkg,
buf: make([]byte, size),
size: size,
static: gd.Name.Static,
}
for _, f := range gd.Flags {
switch f {
case "RODATA":
ds.rodata = true
case "DUPOK":
ds.dupok = true
default:
// Legacy numeric flag constants (runtime/textflag.h):
// DUPOK is 2, RODATA is 8; combinations arrive as one
// number (e.g. 10 = RODATA|DUPOK).
if n, err := strconv.Atoi(f); err == nil {
if n&2 != 0 {
ds.dupok = true
}
if n&8 != 0 {
ds.rodata = true
}
}
}
syms = append(syms, ds)
case *ast.Data:
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
syms = append(syms, ds)
}
for _, d := range f.Decls {
dd, ok := d.(*ast.Data)
if !ok {
continue
}
if dd.Name == nil || dd.Name.Pseudo != "SB" {
continue
}
i, ok := index[dd.Name.Name]
if !ok {
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
}
if dd.Value == nil {
return nil, fmt.Errorf("DATA %q: missing value", dd.Name.Name)
}
w := dd.Width
off := dd.Name.Offset
buf := syms[i].buf
if off < 0 || off+int64(w) > int64(len(buf)) {
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
}
// A symbol value ("DATA s+0(SB)/8, $other(SB)", the rt0 spelling)
// leaves the field zero and records a relocation against the named
// symbol: the linker patches the absolute address at this data
// offset. The toolchain emits the same shape, an R_ADDR of the
// DATA width with the value's offset as the addend, on every
// architecture.
if sym := dd.Value.Imm.Sym; !dd.Value.Imm.HasVal && sym != nil {
syms[i].relocs = append(syms[i].relocs, Reloc{
Off: int(off),
Name: sym.Name,
Addend: sym.Offset,
Kind: RelAddr,
Siz: uint8(w),
})
continue
}
// A string or rune value ("DATA s+0(SB)/20, $"text"") writes its
// bytes into the field and leaves the rest zero, the toolchain's
// WriteString: the declared width must hold every byte, and any
// width is legal.
if s := dd.Value.Imm.Str; s != "" && !dd.Value.Imm.HasVal {
text, err := strconv.Unquote(s)
if err != nil {
return nil, fmt.Errorf("DATA %q: invalid string value %s", dd.Name.Name, s)
}
i, ok := index[dd.Name.Name]
if !ok {
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
if len(text) > w {
return nil, fmt.Errorf("DATA %q: string of %d bytes does not fit width %d", dd.Name.Name, len(text), w)
}
if dd.Value == nil || !dd.Value.Imm.HasVal {
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
copy(buf[off:], text)
continue
}
// A floating-point value stores its IEEE-754 bits: /4 the float32
// rounding of the parsed double, /8 the full 64 bits, the
// toolchain's WriteFloat32 and WriteFloat64.
if f := dd.Value.Imm.Float; f != "" && !dd.Value.Imm.HasVal {
num, err := strconv.ParseFloat(f, 64)
if err != nil {
return nil, fmt.Errorf("DATA %q: invalid floating-point value %q", dd.Name.Name, f)
}
w := dd.Width
switch w {
case 1, 2, 4, 8:
default:
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
}
off := dd.Name.Offset
buf := syms[i].buf
if off < 0 || off+int64(w) > int64(len(buf)) {
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
}
v := dd.Value.Imm.Val
if dd.Value.Imm.Neg {
v = -v
num = -num
}
var v uint64
switch w {
case 4:
v = uint64(math.Float32bits(float32(num)))
case 8:
v = math.Float64bits(num)
default:
return nil, fmt.Errorf("DATA %q: invalid width %d for a float (want 4 or 8)", dd.Name.Name, w)
}
for j := range w {
buf[off+int64(j)] = byte(v >> (8 * j))
}
continue
}
if !dd.Value.Imm.HasVal {
return nil, fmt.Errorf("DATA %q: value must be an integer immediate or a symbol address", dd.Name.Name)
}
switch w {
case 1, 2, 4, 8:
default:
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
}
v := dd.Value.Imm.Val
if dd.Value.Imm.Neg {
v = -v
}
for j := range w {
buf[off+int64(j)] = byte(v >> (8 * j))
}
}
return syms, nil
+365 -2
View File
@@ -4,14 +4,19 @@
package asm
import (
"encoding/binary"
"fmt"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestAssembleFileStaticData checks the whole-image layout — code, padding
// and the data section — and that the RIP-relative displacements of static
// TestAssembleFileStaticData checks the whole-image layout; code, padding
// and the data section; and that the RIP-relative displacements of static
// symbol loads resolve to the right bytes.
func TestAssembleFileStaticData(t *testing.T) {
f, errs := parser.Parse("d_amd64.s", `
@@ -128,3 +133,361 @@ DATA x<>+0(SB)/4, $1
t.Errorf("single-function SB: error %v, want a file-level-assembly error", err)
}
}
// TestCollectDataNumericFlags pins the numeric GLOBL flag constants from
// runtime/textflag.h: DUPOK is 2, RODATA is 8, and combinations arrive as
// one number (9 = NOPROF|RODATA, 10 = RODATA|DUPOK).
func TestCollectDataNumericFlags(t *testing.T) {
tests := []struct {
flags string
rodata bool
dupok bool
}{
{"2", false, true},
{"8", true, false},
{"9", true, false}, // NOPROF|RODATA, not DUPOK
{"10", true, true}, // RODATA|DUPOK
{"RODATA", true, false},
{"DUPOK", false, true},
{"RODATA|DUPOK", true, true},
}
for _, tt := range tests {
src := "TEXT \u00b7f(SB), NOSPLIT, $0\n\tRET\nGLOBL sym(SB), " + tt.flags + ", $8\n"
f, errs := parser.Parse("f_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse %q: %v", tt.flags, errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble %q: %v", tt.flags, err)
}
if len(img.DataSyms) != 1 {
t.Fatalf("%q: data syms = %d, want 1", tt.flags, len(img.DataSyms))
}
d := img.DataSyms[0]
if d.Rodata != tt.rodata || d.Dupok != tt.dupok {
t.Errorf("flags %q: rodata=%v dupok=%v, want rodata=%v dupok=%v",
tt.flags, d.Rodata, d.Dupok, tt.rodata, tt.dupok)
}
}
}
// TestCollectDataSymbolValue covers the symbol-valued DATA field ("DATA
// s+0(SB)/8, $other(SB)", the rt0 spelling): the field stays zero in the
// image and the relocation is recorded against the named symbol, whatever
// the file defines (a TEXT function, a GLOBL) or leaves external.
func TestCollectDataSymbolValue(t *testing.T) {
src := `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-8
MOVQ target+0(FP), AX
RET
GLOBL holder(SB), NOPTR, $32
DATA holder+0(SB)/8, $·Keep(SB)
DATA holder+8(SB)/8, $·Keep+5(SB)
DATA holder+16(SB)/8, $holder(SB)
GLOBL spare(SB), NOPTR, $8
DATA spare+0(SB)/8, $extvar(SB)
`
f, errs := parser.Parse("f_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
byName := map[string]DataSymbol{}
for _, d := range img.DataSyms {
byName[d.Name] = d
}
want := []struct {
sym string
off int
name string
addend int64
ext bool
}{
{"holder", 0, "Keep", 0, false},
{"holder", 8, "Keep", 5, false},
{"holder", 16, "holder", 0, false},
{"spare", 0, "extvar", 0, true},
}
var flat []struct {
sym string
r Reloc
}
for _, d := range img.DataSyms {
for _, r := range d.Relocs {
flat = append(flat, struct {
sym string
r Reloc
}{d.Name, r})
}
}
if len(flat) != len(want) {
t.Fatalf("data relocations = %d, want %d", len(flat), len(want))
}
for i, w := range want {
g := flat[i]
r := g.r
if g.sym != w.sym {
t.Errorf("relocation %d sits on %q, want %q", i, g.sym, w.sym)
continue
}
if r.Off != w.off || r.Name != w.name || r.Addend != w.addend || r.External != w.ext {
t.Errorf("relocation %d = {+%d %q addend %d ext %v}, want {+%d %q addend %d ext %v}",
i, r.Off, r.Name, r.Addend, r.External, w.off, w.name, w.addend, w.ext)
}
if r.Kind != RelAddr {
t.Errorf("relocation %d kind = %v, want RelAddr", i, r.Kind)
}
if r.Siz != 8 {
t.Errorf("relocation %d siz = %d, want 8", i, r.Siz)
}
}
// The fields themselves stay zero: only the linker fills them.
for _, b := range img.Data {
if b != 0 {
t.Fatal("data section is not all zero before relocation")
}
}
if len(img.Externals) != 1 || img.Externals[0] != "extvar" {
t.Errorf("Externals = %v, want [extvar]", img.Externals)
}
}
// TestGOObjectDataSymbolReloc pins the GOOBJ record a symbol-valued DATA
// field produces, against the shape the toolchain emits for the same
// source: an R_ADDR of the DATA width at the field offset, pkgIdxNone plus
// the non-package definition index when the target is the file's own TEXT
// function (the rt0 lib entry spelling).
func TestGOObjectDataSymbolReloc(t *testing.T) {
f, errs := parser.Parse("f_amd64.s", `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-8
RET
GLOBL holder(SB), NOPTR, $16
DATA holder+0(SB)/8, $·Keep+5(SB)
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
obj, err := img.GOObject("main", "f_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
v := openGoobj(t, obj)
// Walk every relocation record; the data record is the one of Siz 8
// and type R_ADDR.
var off, add int64
var pkg, sym uint32
found := false
for data := v.blk(blkReloc); len(data) >= 23; data = data[23:] {
if data[4] != 8 || binary.LittleEndian.Uint16(data[5:]) != relocAddr {
continue
}
found = true
off = int64(int32(binary.LittleEndian.Uint32(data[0:])))
add = int64(binary.LittleEndian.Uint64(data[7:]))
pkg = binary.LittleEndian.Uint32(data[15:])
sym = binary.LittleEndian.Uint32(data[19:])
break
}
if !found {
t.Fatal("no data relocation record in the object")
}
if off != 0 || add != 5 {
t.Errorf("data reloc = {off %d addend %d}, want {off 0 addend 5}", off, add)
}
if pkg != pkgIdxNone {
t.Errorf("data reloc pkg = %#x, want pkgIdxNone (the TEXT function)", pkg)
}
// The function's non-package definition index: the four pc tables
// precede it, so index 4.
if sym != 4 {
t.Errorf("data reloc sym = %d, want 4", sym)
}
}
// TestGOObjectDataSymbolLink is the end-to-end proof for symbol-valued DATA
// fields: the gasm object is substituted for the toolchain's and re-linked,
// then executed, and the linked data word must hold the real address of the
// function the DATA line named (runtime.FuncForPC identifies it).
func TestGOObjectDataSymbolLink(t *testing.T) {
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
dir := t.TempDir()
asmSrc := `#include "textflag.h"
GLOBL entry(SB), NOPTR, $8
DATA entry+0(SB)/8, $·keepme(SB)
TEXT ·keepme(SB), NOSPLIT, $0-0
RET
TEXT ·entryptr(SB), NOSPLIT, $0-8
MOVQ entry+0(SB), AX
MOVQ AX, ret+0(FP)
RET
`
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
t.Fatal(err)
}
mainSrc := `package main
import "runtime"
func keepme()
func entryptr() uintptr
func main() {
pc := entryptr()
fn := runtime.FuncForPC(pc)
if fn == nil {
panic("the entry word does not point at a function")
}
if fn.Name() != "main.keepme" {
panic("the entry word points at " + fn.Name())
}
}
`
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module dlink\n\ngo 1.21\n"), 0o644); err != nil {
t.Fatal(err)
}
// Capture the build: the package archive's asm object and the link line.
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
build.Dir = dir
buildLog, err := build.CombinedOutput()
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
// Assemble the same source with gasm and substitute the object.
src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s"))
if err != nil {
t.Fatal(err)
}
f, errs := parser.Parse("main_amd64.s", string(src))
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
gasmObj, err := img.GOObject("dlink", "main_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
// The linked program must run and find the right function behind the
// data word.
out, err := exec.Command(filepath.Join(dir, "prog")).CombinedOutput()
if err != nil {
t.Fatalf("linked program failed: %v\n%s", err, out)
}
}
// TestCollectDataFloatAndStringValues covers the non-integer DATA values the
// runtime's math and asm files use: floating-point initialisers store their
// IEEE-754 bits (/4 the float32 rounding, /8 the full double) and string
// initialisers write their bytes zero-padded within the declared width.
func TestCollectDataFloatAndStringValues(t *testing.T) {
src := `#include "textflag.h"
TEXT ·Keep(SB), NOSPLIT, $0-8
RET
GLOBL vals<>(SB), RODATA, $44
DATA vals<>+0(SB)/8, $0.5
DATA vals<>+8(SB)/8, $-1.0
DATA vals<>+16(SB)/4, $1.5
DATA vals<>+20(SB)/16, $"call frame too "
DATA vals<>+36(SB)/4, $"hi"
`
f, errs := parser.Parse("fvals_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
byName := map[string]DataSymbol{}
for _, d := range img.DataSyms {
byName[d.Name] = d
}
d := byName["vals"]
if d.Size != 44 {
t.Fatalf("vals size = %d, want 44", d.Size)
}
buf := img.Data[d.Offset : d.Offset+44]
// 0.5 = 0x3FE0000000000000, -1.0 = 0xBFF0000000000000 (float64);
// 1.5 = 0x3FC00000 (float32).
for _, c := range []struct {
off int
want []byte
}{
{0, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xE0, 0x3F}},
{8, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xF0, 0xBF}},
{16, []byte{0x00, 0x00, 0xC0, 0x3F}},
{20, []byte("call frame too ")},
{36, []byte{'h', 'i', 0x00, 0x00}},
} {
if string(buf[c.off:c.off+len(c.want)]) != string(c.want) {
t.Errorf("vals+%d: got % x, want % x", c.off, buf[c.off:c.off+len(c.want)], c.want)
}
}
}
// TestCollectDataValueErrors pins the value-kind width rules: a float needs
// width 4 or 8, a string must fit its declared width, and a bad float
// literal is diagnosed rather than stored.
func TestCollectDataValueErrors(t *testing.T) {
cases := []string{
`GLOBL v<>(SB), RODATA, $4
DATA v<>+0(SB)/1, $0.5`,
`GLOBL v<>(SB), RODATA, $2
DATA v<>+0(SB)/2, $"toolarge"`,
}
for i, src := range cases {
full := "#include \"textflag.h\"\nTEXT ·Keep(SB), NOSPLIT, $0-8\n\tRET\n" + src
f, errs := parser.Parse(fmt.Sprintf("verr%d_amd64.s", i), full)
if len(errs) > 0 {
t.Fatalf("case %d parse: %v", i, errs)
}
if _, err := AssembleFile(f); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
+1010 -74
View File
File diff suppressed because it is too large Load Diff
+622 -18
View File
@@ -9,7 +9,7 @@ package asm
// an opcode constant, and the format selects the bit layout. The opcode
// constants and formats are transcribed from the Go toolchain's own loong64
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
// exactly — the ground-truth oracle for the verify suite.
// exactly, the ground-truth oracle for the verify suite.
//
// All LoongArch instructions are 32 bits, little-endian. The formats used
// here (per the LoongArch Volume I specification):
@@ -30,11 +30,14 @@ package asm
// of the immediate and register fields), mirroring the toolchain's OP_*
// helpers, so each l64* function only ORs its fields in.
import "maps"
import (
"maps"
"strings"
)
// loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition
// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
// flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's
// assembly uses. Returns -1 for an unrecognised name.
func loong64RegNum(name string) int {
switch name {
@@ -103,7 +106,12 @@ func loong64RegNum(name string) int {
case "R31", "S8":
return 31
}
// F0–F31, FCC0–FCC7, FCSR0–FCSR31.
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
// X0-X31) are deliberately NOT accepted here: they are a separate
// register class, and the toolchain rejects V/X names wherever an
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
// resolved only through loong64VecRegNum.
if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], 31)
}
@@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int {
return -1
}
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
// of their own: the toolchain accepts them only in the vector operands of the
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
// spellings never reach the integer/FP resolver.
func loong64VecRegNum(name string) int {
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
return -1
}
return loong64RegSpecial(name[1:], 31)
}
// ---- format helpers ----
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
@@ -199,7 +220,9 @@ func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
}
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller.
// The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller
// must have validated them (0..31 for the .w forms, 0..63 for the .d forms,
// lsb <= msb), the same rule the toolchain enforces as "illegal bit number".
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
}
@@ -245,7 +268,7 @@ const (
l64Firr14 // 2RI14 (ldptr/stptr)
l64Firr16 // 2RI16 (addu16i.d)
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
l64Firir // bstrins/bstrpick
l64Firrr // alsl
l64Fi15 // syscall/break/dbar
@@ -253,6 +276,9 @@ const (
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
l64Fshift // 2RI12 with a 5/6-bit shift immediate
l64Fpreld // preld (2RI12 + 5-bit hint)
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -275,12 +301,74 @@ type l64DualEnc struct {
var l64DualTable = map[string]l64DualEnc{}
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
// yet; the base integer, memory and floating-point ISA is complete.
// to their encoding.
var l64InstrTable = map[string]l64Enc{}
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
type l64Vec3Enc struct {
op uint32
lasx bool
}
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
// the opcode, the bank, the accepted immediate range, the bias the toolchain
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
type l64VecImmEnc struct {
op uint32
lasx bool
min, max int
bias int
mask int
}
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
// each accepts; presence in the map routes the mnemonic through the vector
// dispatcher rather than the integer/FP formats.
var l64VecBank = map[string]bool{}
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
var l64VecImmInfo = map[string]l64VecImmEnc{}
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
// vpcnt.v).
var l64Vec2R = map[string]bool{}
// l64Vec4R marks the four-operand vector mnemonics (INSTR va, vk, vj, vd,
// such as vshuf.b).
var l64Vec4R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
pickS, pickU uint32 // vpickve2gr.{,u} element extract
ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move
}
var l64VmovqTable = map[bool]l64VmovqEnc{
false: { // VMOVQ, the LSX (V) bank
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
},
true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
},
}
func init() {
// 3R — integer.
// 3R, integer.
rrr := map[string]uint32{
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
@@ -300,7 +388,7 @@ func init() {
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
}
// 3R — floating point.
// 3R, floating point.
rrr["MULF"] = 0x209 << 15
rrr["MULD"] = 0x20a << 15
rrr["DIVF"] = 0x20d << 15
@@ -358,6 +446,18 @@ func init() {
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
// LSX: convert a 64-bit integer lane to a double float. The operand
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
// so the entry stays on the 2R integer/FP format.
"FFINTDV": 0x474a << 10,
// The rest of the scalar conversions (all F-bank, 2R).
"FFINTFW": 0x4744 << 10, // ffint.s.w
"FFINTFV": 0x4746 << 10, // ffint.s.l
"FFINTDW": 0x4748 << 10, // ffint.d.w
"FTINTWF": 0x46c1 << 10, // ftint.w.s
"FTINTWD": 0x46c2 << 10, // ftint.w.d
"FTINTVF": 0x46c9 << 10, // ftint.l.s
"FTINTVD": 0x46ca << 10, // ftint.l.d
}
for m, op := range rr {
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
@@ -390,12 +490,12 @@ func init() {
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
})
// 2RI12 — pure immediate arithmetic (LU52ID has no register form).
// 2RI12, pure immediate arithmetic (LU52ID has no register form).
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
// 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and
// 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and
// stores (ldptr/stptr), with the offset scaled by 4.
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
@@ -414,18 +514,20 @@ func init() {
// LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R — fused multiply-add.
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
// condition flag, the layout matches the 4R shape).
rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
"FSEL": 0x340 << 18,
}
for m, op := range rrrr {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
}
// IRIR — bit-field insert/extract.
// IRIR, bit-field insert/extract.
irir := map[string]uint32{
"BSTRINSW": 0x3<<21 | 0x0<<15,
"BSTRINSV": 0x2 << 22,
@@ -436,7 +538,7 @@ func init() {
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
}
// 3RI2 — ALSL.
// 3RI2, ALSL.
irrr := map[string]uint32{
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
}
@@ -452,7 +554,11 @@ func init() {
// PRELD.
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics — 3R with the AM field order (rk=value, rj=address, rd=result).
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
// is rejected by the oracle.
am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
@@ -470,14 +576,512 @@ func init() {
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
// The _dbar (acquire/release) add, and, or variants: opcodes read off
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
// The remaining _dbar exchange variants (loong64enc1.s).
"AMXORDBW": 0x070DA << 15, "AMXORDBV": 0x070DB << 15,
"AMMAXDBW": 0x070DC << 15, "AMMAXDBV": 0x070DD << 15,
"AMMINDBW": 0x070DE << 15, "AMMINDBV": 0x070DF << 15,
"AMMAXDBWU": 0x070E0 << 15, "AMMAXDBVU": 0x070E1 << 15,
"AMMINDBWU": 0x070E2 << 15, "AMMINDBVU": 0x070E3 << 15,
}
for m, op := range am {
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
}
// ---- LSX/LASX (V*/XV*) ----
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
// most of them), not assumed from the LoongArch manual.
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
vec3 := map[string]l64Vec3Enc{
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
"XVADDV": {0xE817 << 15, true},
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
}
// The integer and FP add/subtract families: [X]VADD and [X]VSUB by lane
// width, plus the [X]VSADD/[X]VSSUB saturating pairs.
// Opcodes transcribed from the toolchain's loong64enc1.s.
addsub := map[string]l64Vec3Enc{
"VADDB": {0xE014 << 15, false}, "VADDH": {0xE015 << 15, false},
"VADDD": {0xE262 << 15, false}, "VADDF": {0xE261 << 15, false},
"VADDQ": {0xE25A << 15, false},
"VSUBB": {0xE018 << 15, false}, "VSUBH": {0xE019 << 15, false},
"VSUBW": {0xE01A << 15, false}, "VSUBV": {0xE01B << 15, false},
"VSUBQ": {0xE25B << 15, false},
"VSUBF": {0xE265 << 15, false}, "VSUBD": {0xE266 << 15, false},
"VSADDB": {0xE08C << 15, false}, "VSADDH": {0xE08D << 15, false},
"VSADDW": {0xE08E << 15, false}, "VSADDV": {0xE08F << 15, false},
"VSADDBU": {0xE094 << 15, false}, "VSADDHU": {0xE095 << 15, false},
"VSADDWU": {0xE096 << 15, false}, "VSADDVU": {0xE097 << 15, false},
"VSSUBB": {0xE090 << 15, false}, "VSSUBH": {0xE091 << 15, false},
"VSSUBW": {0xE092 << 15, false}, "VSSUBV": {0xE093 << 15, false},
"VSSUBBU": {0xE098 << 15, false}, "VSSUBHU": {0xE099 << 15, false},
"VSSUBWU": {0xE09A << 15, false}, "VSSUBVU": {0xE09B << 15, false},
"XVADDB": {0xE814 << 15, true}, "XVADDH": {0xE815 << 15, true},
"XVADDW": {0xE816 << 15, true},
"XVADDD": {0xEA62 << 15, true}, "XVADDF": {0xEA61 << 15, true},
"XVADDQ": {0xEA5A << 15, true},
"XVSUBB": {0xE818 << 15, true}, "XVSUBH": {0xE819 << 15, true},
"XVSUBW": {0xE81A << 15, true}, "XVSUBV": {0xE81B << 15, true},
"XVSUBQ": {0xEA5B << 15, true},
"XVSUBF": {0xEA65 << 15, true}, "XVSUBD": {0xEA66 << 15, true},
"XVSADDB": {0xE88C << 15, true}, "XVSADDH": {0xE88D << 15, true},
"XVSADDW": {0xE88E << 15, true}, "XVSADDV": {0xE88F << 15, true},
"XVSADDBU": {0xE894 << 15, true}, "XVSADDHU": {0xE895 << 15, true},
"XVSADDWU": {0xE896 << 15, true}, "XVSADDVU": {0xE897 << 15, true},
"XVSSUBB": {0xE890 << 15, true}, "XVSSUBH": {0xE891 << 15, true},
"XVSSUBW": {0xE892 << 15, true}, "XVSSUBV": {0xE893 << 15, true},
"XVSSUBBU": {0xE898 << 15, true}, "XVSSUBHU": {0xE899 << 15, true},
"XVSSUBWU": {0xE89A << 15, true}, "XVSSUBVU": {0xE89B << 15, true},
}
// The multiply families: plain and high-half [X]VMUL/[X]VMUH, the
// widening [X]VMULW{EV,OD} ladder and its accumulating [X]VMADDW twins,
// plus the [X]VMADD/[X]VMSUB fused multiply-add and the [X]VDIV/[X]VMOD
// divide and modulo pairs.
muldiv := map[string]l64Vec3Enc{
"VMULB": {0xE108 << 15, false}, "VMULH": {0xE109 << 15, false},
"VMULW": {0xE10A << 15, false}, "VMULV": {0xE10B << 15, false},
"VMUHB": {0xE10C << 15, false}, "VMUHH": {0xE10D << 15, false},
"VMUHW": {0xE10E << 15, false}, "VMUHV": {0xE10F << 15, false},
"VMUHBU": {0xE110 << 15, false}, "VMUHHU": {0xE111 << 15, false},
"VMUHWU": {0xE112 << 15, false}, "VMUHVU": {0xE113 << 15, false},
"VMULWEVHB": {0xE120 << 15, false}, "VMULWEVWH": {0xE121 << 15, false},
"VMULWEVVW": {0xE122 << 15, false}, "VMULWEVQV": {0xE123 << 15, false},
"VMULWODHB": {0xE124 << 15, false}, "VMULWODWH": {0xE125 << 15, false},
"VMULWODVW": {0xE126 << 15, false}, "VMULWODQV": {0xE127 << 15, false},
"VMULWEVHBU": {0xE130 << 15, false}, "VMULWEVWHU": {0xE131 << 15, false},
"VMULWEVVWU": {0xE132 << 15, false}, "VMULWEVQVU": {0xE133 << 15, false},
"VMULWODHBU": {0xE134 << 15, false}, "VMULWODWHU": {0xE135 << 15, false},
"VMULWODVWU": {0xE136 << 15, false}, "VMULWODQVU": {0xE137 << 15, false},
"VMULWEVHBUB": {0xE140 << 15, false}, "VMULWEVWHUH": {0xE141 << 15, false},
"VMULWEVVWUW": {0xE142 << 15, false}, "VMULWEVQVUV": {0xE143 << 15, false},
"VMULWODHBUB": {0xE144 << 15, false}, "VMULWODWHUH": {0xE145 << 15, false},
"VMULWODVWUW": {0xE146 << 15, false}, "VMULWODQVUV": {0xE147 << 15, false},
"VMADDB": {0xE150 << 15, false}, "VMADDH": {0xE151 << 15, false},
"VMADDW": {0xE152 << 15, false}, "VMADDV": {0xE153 << 15, false},
"VMSUBB": {0xE154 << 15, false}, "VMSUBH": {0xE155 << 15, false},
"VMSUBW": {0xE156 << 15, false}, "VMSUBV": {0xE157 << 15, false},
"VMADDWEVHB": {0xE158 << 15, false}, "VMADDWEVWH": {0xE159 << 15, false},
"VMADDWEVVW": {0xE15A << 15, false}, "VMADDWEVQV": {0xE15B << 15, false},
"VMADDWODHB": {0xE15C << 15, false}, "VMADDWODWH": {0xE15D << 15, false},
"VMADDWODVW": {0xE15E << 15, false}, "VMADDWODQV": {0xE15F << 15, false},
"VMADDWEVHBU": {0xE168 << 15, false}, "VMADDWEVWHU": {0xE169 << 15, false},
"VMADDWEVVWU": {0xE16A << 15, false}, "VMADDWEVQVU": {0xE16B << 15, false},
"VMADDWODHBU": {0xE16C << 15, false}, "VMADDWODWHU": {0xE16D << 15, false},
"VMADDWODVWU": {0xE16E << 15, false}, "VMADDWODQVU": {0xE16F << 15, false},
"VMADDWEVHBUB": {0xE178 << 15, false}, "VMADDWEVWHUH": {0xE179 << 15, false},
"VMADDWEVVWUW": {0xE17A << 15, false}, "VMADDWEVQVUV": {0xE17B << 15, false},
"VMADDWODHBUB": {0xE17C << 15, false}, "VMADDWODWHUH": {0xE17D << 15, false},
"VMADDWODVWUW": {0xE17E << 15, false}, "VMADDWODQVUV": {0xE17F << 15, false},
"VDIVB": {0xE1C0 << 15, false}, "VDIVH": {0xE1C1 << 15, false},
"VDIVW": {0xE1C2 << 15, false}, "VDIVV": {0xE1C3 << 15, false},
"VMODB": {0xE1C4 << 15, false}, "VMODH": {0xE1C5 << 15, false},
"VMODW": {0xE1C6 << 15, false}, "VMODV": {0xE1C7 << 15, false},
"VDIVBU": {0xE1C8 << 15, false}, "VDIVHU": {0xE1C9 << 15, false},
"VDIVWU": {0xE1CA << 15, false}, "VDIVVU": {0xE1CB << 15, false},
"VMODBU": {0xE1CC << 15, false}, "VMODHU": {0xE1CD << 15, false},
"VMODWU": {0xE1CE << 15, false}, "VMODVU": {0xE1CF << 15, false},
"VMULF": {0xE271 << 15, false}, "VMULD": {0xE272 << 15, false},
"VDIVF": {0xE275 << 15, false}, "VDIVD": {0xE276 << 15, false},
"XVMULB": {0xE908 << 15, true}, "XVMULH": {0xE909 << 15, true},
"XVMULW": {0xE90A << 15, true}, "XVMULV": {0xE90B << 15, true},
"XVMUHB": {0xE90C << 15, true}, "XVMUHH": {0xE90D << 15, true},
"XVMUHW": {0xE90E << 15, true}, "XVMUHV": {0xE90F << 15, true},
"XVMUHBU": {0xE910 << 15, true}, "XVMUHHU": {0xE911 << 15, true},
"XVMUHWU": {0xE912 << 15, true}, "XVMUHVU": {0xE913 << 15, true},
"XVMULWEVHB": {0xE920 << 15, true}, "XVMULWEVWH": {0xE921 << 15, true},
"XVMULWEVVW": {0xE922 << 15, true}, "XVMULWEVQV": {0xE923 << 15, true},
"XVMULWODHB": {0xE924 << 15, true}, "XVMULWODWH": {0xE925 << 15, true},
"XVMULWODVW": {0xE926 << 15, true}, "XVMULWODQV": {0xE927 << 15, true},
"XVMULWEVHBU": {0xE930 << 15, true}, "XVMULWEVWHU": {0xE931 << 15, true},
"XVMULWEVVWU": {0xE932 << 15, true}, "XVMULWEVQVU": {0xE933 << 15, true},
"XVMULWODHBU": {0xE934 << 15, true}, "XVMULWODWHU": {0xE935 << 15, true},
"XVMULWODVWU": {0xE936 << 15, true}, "XVMULWODQVU": {0xE937 << 15, true},
"XVMULWEVHBUB": {0xE940 << 15, true}, "XVMULWEVWHUH": {0xE941 << 15, true},
"XVMULWEVVWUW": {0xE942 << 15, true}, "XVMULWEVQVUV": {0xE943 << 15, true},
"XVMULWODHBUB": {0xE944 << 15, true}, "XVMULWODWHUH": {0xE945 << 15, true},
"XVMULWODVWUW": {0xE946 << 15, true}, "XVMULWODQVUV": {0xE947 << 15, true},
"XVMADDB": {0xE950 << 15, true}, "XVMADDH": {0xE951 << 15, true},
"XVMADDW": {0xE952 << 15, true}, "XVMADDV": {0xE953 << 15, true},
"XVMSUBB": {0xE954 << 15, true}, "XVMSUBH": {0xE955 << 15, true},
"XVMSUBW": {0xE956 << 15, true}, "XVMSUBV": {0xE957 << 15, true},
"XVMADDWEVHB": {0xE958 << 15, true}, "XVMADDWEVWH": {0xE959 << 15, true},
"XVMADDWEVVW": {0xE95A << 15, true}, "XVMADDWEVQV": {0xE95B << 15, true},
"XVMADDWODHB": {0xE95C << 15, true}, "XVMADDWODWH": {0xE95D << 15, true},
"XVMADDWODVW": {0xE95E << 15, true}, "XVMADDWODQV": {0xE95F << 15, true},
"XVMADDWEVHBU": {0xE968 << 15, true}, "XVMADDWEVWHU": {0xE969 << 15, true},
"XVMADDWEVVWU": {0xE96A << 15, true}, "XVMADDWEVQVU": {0xE96B << 15, true},
"XVMADDWODHBU": {0xE96C << 15, true}, "XVMADDWODWHU": {0xE96D << 15, true},
"XVMADDWODVWU": {0xE96E << 15, true}, "XVMADDWODQVU": {0xE96F << 15, true},
"XVMADDWEVHBUB": {0xE978 << 15, true}, "XVMADDWEVWHUH": {0xE979 << 15, true},
"XVMADDWEVVWUW": {0xE97A << 15, true}, "XVMADDWEVQVUV": {0xE97B << 15, true},
"XVMADDWODHBUB": {0xE97C << 15, true}, "XVMADDWODWHUH": {0xE97D << 15, true},
"XVMADDWODVWUW": {0xE97E << 15, true}, "XVMADDWODQVUV": {0xE97F << 15, true},
"XVDIVB": {0xE9C0 << 15, true}, "XVDIVH": {0xE9C1 << 15, true},
"XVDIVW": {0xE9C2 << 15, true}, "XVDIVV": {0xE9C3 << 15, true},
"XVMODB": {0xE9C4 << 15, true}, "XVMODH": {0xE9C5 << 15, true},
"XVMODW": {0xE9C6 << 15, true}, "XVMODV": {0xE9C7 << 15, true},
"XVDIVBU": {0xE9C8 << 15, true}, "XVDIVHU": {0xE9C9 << 15, true},
"XVDIVWU": {0xE9CA << 15, true}, "XVDIVVU": {0xE9CB << 15, true},
"XVMODBU": {0xE9CC << 15, true}, "XVMODHU": {0xE9CD << 15, true},
"XVMODWU": {0xE9CE << 15, true}, "XVMODVU": {0xE9CF << 15, true},
"XVMULF": {0xEA71 << 15, true}, "XVMULD": {0xEA72 << 15, true},
"XVDIVF": {0xEA75 << 15, true}, "XVDIVD": {0xEA76 << 15, true},
}
// The lane-wise shifts and rotates (three-register forms; the immediate
// forms live in l64VecImmInfo), the interleave families, the bit
// clear/set/rev register forms, the remaining logic and compare
// spellings, the widening add/subtract ladder and the vector FP
// arithmetic.
vecmisc := map[string]l64Vec3Enc{
"VSLLB": {0xE1D0 << 15, false}, "VSLLH": {0xE1D1 << 15, false},
"VSLLW": {0xE1D2 << 15, false}, "VSLLV": {0xE1D3 << 15, false},
"VSRLB": {0xE1D4 << 15, false}, "VSRLH": {0xE1D5 << 15, false},
"VSRLW": {0xE1D6 << 15, false}, "VSRLV": {0xE1D7 << 15, false},
"VSRAH": {0xE1D9 << 15, false}, "VSRAW": {0xE1DA << 15, false},
"VSRAV": {0xE1DB << 15, false},
"VROTRB": {0xE1DC << 15, false}, "VROTRH": {0xE1DD << 15, false},
"VROTRV": {0xE1DF << 15, false},
"VILVLB": {0xE234 << 15, false}, "VILVLH": {0xE235 << 15, false},
"VILVLW": {0xE236 << 15, false}, "VILVLV": {0xE237 << 15, false},
"VILVHB": {0xE238 << 15, false}, "VILVHH": {0xE239 << 15, false},
"VILVHW": {0xE23A << 15, false}, "VILVHV": {0xE23B << 15, false},
"VBITCLRB": {0xE218 << 15, false}, "VBITCLRH": {0xE219 << 15, false},
"VBITCLRW": {0xE21A << 15, false}, "VBITCLRV": {0xE21B << 15, false},
"VBITSETB": {0xE21C << 15, false}, "VBITSETH": {0xE21D << 15, false},
"VBITSETW": {0xE21E << 15, false}, "VBITSETV": {0xE21F << 15, false},
"VBITREVB": {0xE220 << 15, false}, "VBITREVH": {0xE221 << 15, false},
"VBITREVW": {0xE222 << 15, false}, "VBITREVV": {0xE223 << 15, false},
"VORV": {0xE24D << 15, false}, "VNORV": {0xE24F << 15, false},
"VANDNV": {0xE250 << 15, false}, "VORNV": {0xE251 << 15, false},
"VSEQH": {0xE001 << 15, false}, "VSEQW": {0xE002 << 15, false},
"VSLTB": {0xE00C << 15, false}, "VSLTH": {0xE00D << 15, false},
"VSLTW": {0xE00E << 15, false}, "VSLTV": {0xE00F << 15, false},
"VSLTBU": {0xE010 << 15, false}, "VSLTHU": {0xE011 << 15, false},
"VSLTWU": {0xE012 << 15, false}, "VSLTVU": {0xE013 << 15, false},
"VADDWEVHB": {0xE03C << 15, false}, "VADDWEVWH": {0xE03D << 15, false},
"VADDWEVVW": {0xE03E << 15, false}, "VADDWEVQV": {0xE03F << 15, false},
"VSUBWEVHB": {0xE040 << 15, false}, "VSUBWEVWH": {0xE041 << 15, false},
"VSUBWEVVW": {0xE042 << 15, false}, "VSUBWEVQV": {0xE043 << 15, false},
"VADDWODHB": {0xE044 << 15, false}, "VADDWODWH": {0xE045 << 15, false},
"VADDWODVW": {0xE046 << 15, false}, "VADDWODQV": {0xE047 << 15, false},
"VSUBWODHB": {0xE048 << 15, false}, "VSUBWODWH": {0xE049 << 15, false},
"VSUBWODVW": {0xE04A << 15, false}, "VSUBWODQV": {0xE04B << 15, false},
"VSUBWEVHBU": {0xE060 << 15, false}, "VSUBWEVWHU": {0xE061 << 15, false},
"VSUBWEVVWU": {0xE062 << 15, false}, "VSUBWEVQVU": {0xE063 << 15, false},
"VADDWEVHBU": {0xE05C << 15, false}, "VADDWEVWHU": {0xE05D << 15, false},
"VADDWEVVWU": {0xE05E << 15, false}, "VADDWEVQVU": {0xE05F << 15, false},
"VADDWODHBU": {0xE064 << 15, false}, "VADDWODWHU": {0xE065 << 15, false},
"VADDWODVWU": {0xE066 << 15, false}, "VADDWODQVU": {0xE067 << 15, false},
"VSUBWODHBU": {0xE068 << 15, false}, "VSUBWODWHU": {0xE069 << 15, false},
"VSUBWODVWU": {0xE06A << 15, false}, "VSUBWODQVU": {0xE06B << 15, false},
"VSHUFH": {0xE2F5 << 15, false}, "VSHUFW": {0xE2F6 << 15, false},
"VSHUFV": {0xE2F7 << 15, false},
"XVSLLB": {0xE9D0 << 15, true}, "XVSLLH": {0xE9D1 << 15, true},
"XVSLLW": {0xE9D2 << 15, true}, "XVSLLV": {0xE9D3 << 15, true},
"XVSRLB": {0xE9D4 << 15, true}, "XVSRLH": {0xE9D5 << 15, true},
"XVSRLW": {0xE9D6 << 15, true}, "XVSRLV": {0xE9D7 << 15, true},
"XVSRAB": {0xE9D8 << 15, true}, "XVSRAH": {0xE9D9 << 15, true},
"XVSRAW": {0xE9DA << 15, true}, "XVSRAV": {0xE9DB << 15, true},
"XVROTRB": {0xE9DC << 15, true}, "XVROTRH": {0xE9DD << 15, true},
"XVROTRW": {0xE9DE << 15, true}, "XVROTRV": {0xE9DF << 15, true},
"XVILVLB": {0xEA34 << 15, true}, "XVILVLH": {0xEA35 << 15, true},
"XVILVLW": {0xEA36 << 15, true}, "XVILVLV": {0xEA37 << 15, true},
"XVILVHB": {0xEA38 << 15, true}, "XVILVHH": {0xEA39 << 15, true},
"XVILVHW": {0xEA3A << 15, true}, "XVILVHV": {0xEA3B << 15, true},
"XVBITCLRB": {0xEA18 << 15, true}, "XVBITCLRH": {0xEA19 << 15, true},
"XVBITCLRW": {0xEA1A << 15, true}, "XVBITCLRV": {0xEA1B << 15, true},
"XVBITSETB": {0xEA1C << 15, true}, "XVBITSETH": {0xEA1D << 15, true},
"XVBITSETW": {0xEA1E << 15, true}, "XVBITSETV": {0xEA1F << 15, true},
"XVBITREVB": {0xEA20 << 15, true}, "XVBITREVH": {0xEA21 << 15, true},
"XVBITREVW": {0xEA22 << 15, true}, "XVBITREVV": {0xEA23 << 15, true},
"XVORV": {0xEA4D << 15, true}, "XVNORV": {0xEA4F << 15, true},
"XVANDNV": {0xEA50 << 15, true}, "XVORNV": {0xEA51 << 15, true},
"XVSEQH": {0xE801 << 15, true}, "XVSEQW": {0xE802 << 15, true},
"XVSLTB": {0xE80C << 15, true}, "XVSLTH": {0xE80D << 15, true},
"XVSLTW": {0xE80E << 15, true}, "XVSLTV": {0xE80F << 15, true},
"XVSLTBU": {0xE810 << 15, true}, "XVSLTHU": {0xE811 << 15, true},
"XVSLTWU": {0xE812 << 15, true}, "XVSLTVU": {0xE813 << 15, true},
"XVADDWEVHB": {0xE83C << 15, true}, "XVADDWEVWH": {0xE83D << 15, true},
"XVADDWEVVW": {0xE83E << 15, true}, "XVADDWEVQV": {0xE83F << 15, true},
"XVSUBWEVHB": {0xE840 << 15, true}, "XVSUBWEVWH": {0xE841 << 15, true},
"XVSUBWEVVW": {0xE842 << 15, true}, "XVSUBWEVQV": {0xE843 << 15, true},
"XVADDWODHB": {0xE844 << 15, true}, "XVADDWODWH": {0xE845 << 15, true},
"XVADDWODVW": {0xE846 << 15, true}, "XVADDWODQV": {0xE847 << 15, true},
"XVSUBWODHB": {0xE848 << 15, true}, "XVSUBWODWH": {0xE849 << 15, true},
"XVSUBWODVW": {0xE84A << 15, true}, "XVSUBWODQV": {0xE84B << 15, true},
"XVADDWEVHBU": {0xE85C << 15, true}, "XVADDWEVWHU": {0xE85D << 15, true},
"XVADDWEVVWU": {0xE85E << 15, true}, "XVADDWEVQVU": {0xE85F << 15, true},
"XVSUBWEVHBU": {0xE860 << 15, true}, "XVSUBWEVWHU": {0xE861 << 15, true},
"XVSUBWEVVWU": {0xE862 << 15, true}, "XVSUBWEVQVU": {0xE863 << 15, true},
"XVADDWODHBU": {0xE864 << 15, true}, "XVADDWODWHU": {0xE865 << 15, true},
"XVADDWODVWU": {0xE866 << 15, true}, "XVADDWODQVU": {0xE867 << 15, true},
"XVSUBWODHBU": {0xE868 << 15, true}, "XVSUBWODWHU": {0xE869 << 15, true},
"XVSUBWODVWU": {0xE86A << 15, true}, "XVSUBWODQVU": {0xE86B << 15, true},
"XVSHUFH": {0xEAF5 << 15, true}, "XVSHUFW": {0xEAF6 << 15, true},
"XVSHUFV": {0xEAF7 << 15, true},
}
for _, tab := range []map[string]l64Vec3Enc{addsub, muldiv, vecmisc} {
for m, e := range tab {
if _, dup := vec3[m]; dup {
panic("loong64: duplicate vector mnemonic " + m)
}
vec3[m] = e
}
}
for m, e := range vec3 {
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
l64VecBank[m] = e.lasx
}
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
// range, bias and field mask are the ones the toolchain encodes: vandi.b
// stores the raw 8-bit constant, vsrari.b stores imm+8 (lane-width
// bias), the si5 compares store 5-bit two's-complement values and vseqi.d
// a 7-bit field the toolchain range-checks down to si5.
// The mnemonics that also have a register form (the shifts, the bit
// clear/set/rev families, VSEQ and the logic immediates) keep their
// three-register entry in l64InstrTable; the dispatcher picks the
// immediate opcode from l64VecImmInfo by operand kind, so the immediate
// entries must not overwrite the table.
vecImm := map[string]l64VecImmEnc{
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
"VORB": {0xE7A8 << 15, false, 0, 255, 0, 0xFF},
"XVORB": {0xEFA8 << 15, true, 0, 255, 0, 0xFF},
"VXORB": {0xE7B0 << 15, false, 0, 255, 0, 0xFF},
"XVXORB": {0xEFB0 << 15, true, 0, 255, 0, 0xFF},
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
// 7-bit field, but the toolchain range-checks it down to si5 as well
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
"VSEQH": {0xE501 << 15, false, -16, 15, 0, 0x1F},
"XVSEQH": {0xED01 << 15, true, -16, 15, 0, 0x1F},
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F},
// vslti compares against a signed (or, in the U spellings, unsigned)
// si5/ui5 constant.
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
"XVSLTB": {0xED0C << 15, true, -16, 15, 0, 0x1F},
"VSLTH": {0xE50D << 15, false, -16, 15, 0, 0x1F},
"XVSLTH": {0xED0D << 15, true, -16, 15, 0, 0x1F},
"VSLTW": {0xE50E << 15, false, -16, 15, 0, 0x1F},
"XVSLTW": {0xED0E << 15, true, -16, 15, 0, 0x1F},
"VSLTV": {0xE50F << 15, false, -16, 15, 0, 0x1F},
"XVSLTV": {0xED0F << 15, true, -16, 15, 0, 0x1F},
"VSLTBU": {0xE510 << 15, false, 0, 31, 0, 0x1F},
"XVSLTBU": {0xED10 << 15, true, 0, 31, 0, 0x1F},
"VSLTHU": {0xE511 << 15, false, 0, 31, 0, 0x1F},
"XVSLTHU": {0xED11 << 15, true, 0, 31, 0, 0x1F},
"VSLTWU": {0xE512 << 15, false, 0, 31, 0, 0x1F},
"XVSLTWU": {0xED12 << 15, true, 0, 31, 0, 0x1F},
"VSLTVU": {0xE513 << 15, false, 0, 31, 0, 0x1F},
"XVSLTVU": {0xED13 << 15, true, 0, 31, 0, 0x1F},
// vaddi/vsubi take ui5 constants for every width on this toolchain
// (VADDVU $32 is rejected by the oracle although the field is ui8).
"VADDBU": {0xE514 << 15, false, 0, 31, 0, 0x1F},
"XVADDBU": {0xED14 << 15, true, 0, 31, 0, 0x1F},
"VADDHU": {0xE515 << 15, false, 0, 31, 0, 0x1F},
"XVADDHU": {0xED15 << 15, true, 0, 31, 0, 0x1F},
"VADDWU": {0xE516 << 15, false, 0, 31, 0, 0x1F},
"XVADDWU": {0xED16 << 15, true, 0, 31, 0, 0x1F},
"VADDVU": {0xE517 << 15, false, 0, 31, 0, 0x1F},
"XVADDVU": {0xED17 << 15, true, 0, 31, 0, 0x1F},
"VSUBBU": {0xE518 << 15, false, 0, 31, 0, 0x1F},
"XVSUBBU": {0xED18 << 15, true, 0, 31, 0, 0x1F},
"VSUBHU": {0xE519 << 15, false, 0, 31, 0, 0x1F},
"XVSUBHU": {0xED19 << 15, true, 0, 31, 0, 0x1F},
"VSUBWU": {0xE51A << 15, false, 0, 31, 0, 0x1F},
"XVSUBWU": {0xED1A << 15, true, 0, 31, 0, 0x1F},
"VSUBVU": {0xE51B << 15, false, 0, 31, 0, 0x1F},
"XVSUBVU": {0xED1B << 15, true, 0, 31, 0, 0x1F},
// The shift/rotate immediates ride in a width-sized field whose upper
// bits carry the lane-width code: vslli.b stores ui3 at [12:0] with
// bits [14:13] inside the opcode, vslli.h ui4 under a 4 bit mask, and
// the .w/.d spellings a raw ui5/ui6.
"VSLLB": {0x732C2000, false, 0, 7, 0, 0x7},
"XVSLLB": {0x772C2000, true, 0, 7, 0, 0x7},
"VSLLH": {0x732C4000, false, 0, 15, 0, 0xF},
"XVSLLH": {0x772C4000, true, 0, 15, 0, 0xF},
"VSLLW": {0xE659 << 15, false, 0, 31, 0, 0x1F},
"XVSLLW": {0xEE59 << 15, true, 0, 31, 0, 0x1F},
"VSLLV": {0xE65A << 15, false, 0, 63, 0, 0x3F},
"XVSLLV": {0xEE5A << 15, true, 0, 63, 0, 0x3F},
"VSRLB": {0x73302000, false, 0, 7, 0, 0x7},
"XVSRLB": {0x77302000, true, 0, 7, 0, 0x7},
"VSRLH": {0x73304000, false, 0, 15, 0, 0xF},
"XVSRLH": {0x77304000, true, 0, 15, 0, 0xF},
"VSRLW": {0xE661 << 15, false, 0, 31, 0, 0x1F},
"XVSRLW": {0xEE61 << 15, true, 0, 31, 0, 0x1F},
"VSRLV": {0xE662 << 15, false, 0, 63, 0, 0x3F},
"XVSRLV": {0xEE62 << 15, true, 0, 63, 0, 0x3F},
// vsrari/vrotri bias the field so the lane-width code rides above the
// shift amount (.b adds 8, .h 16, .w 32; .d is a raw ui6).
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
"XVSRAB": {0xEE68 << 15, true, 0, 7, 8, 0x1F},
"VSRAH": {0x73344000, false, 0, 15, 0, 0xF},
"XVSRAH": {0x77344000, true, 0, 15, 0, 0xF},
"VSRAW": {0xE669 << 15, false, 0, 31, 0, 0x1F},
"XVSRAW": {0xEE69 << 15, true, 0, 31, 0, 0x1F},
"VSRAV": {0xE66A << 15, false, 0, 63, 0, 0x3F},
"XVSRAV": {0xEE6A << 15, true, 0, 63, 0, 0x3F},
"VROTRB": {0x72A02000, false, 0, 7, 0, 0x7},
"XVROTRB": {0x76A02000, true, 0, 7, 0, 0x7},
"VROTRH": {0x72A04000, false, 0, 15, 0, 0xF},
"XVROTRH": {0x76A04000, true, 0, 15, 0, 0xF},
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
"XVROTRW": {0xED41 << 15, true, 0, 31, 0, 0x1F},
"VROTRV": {0xE542 << 15, false, 0, 63, 0, 0x3F},
"XVROTRV": {0xED42 << 15, true, 0, 63, 0, 0x3F},
// vbitclri/vbitseti/vbitrevi follow the same width-coded layout.
"VBITCLRB": {0x73102000, false, 0, 7, 0, 0x7},
"XVBITCLRB": {0x77102000, true, 0, 7, 0, 0x7},
"VBITCLRH": {0x73104000, false, 0, 15, 0, 0xF},
"XVBITCLRH": {0x77104000, true, 0, 15, 0, 0xF},
"VBITCLRW": {0xE621 << 15, false, 0, 31, 0, 0x1F},
"XVBITCLRW": {0xEE21 << 15, true, 0, 31, 0, 0x1F},
"VBITCLRV": {0xE622 << 15, false, 0, 63, 0, 0x3F},
"XVBITCLRV": {0xEE22 << 15, true, 0, 63, 0, 0x3F},
"VBITSETB": {0x73142000, false, 0, 7, 0, 0x7},
"XVBITSETB": {0x77142000, true, 0, 7, 0, 0x7},
"VBITSETH": {0x73144000, false, 0, 15, 0, 0xF},
"XVBITSETH": {0x77144000, true, 0, 15, 0, 0xF},
"VBITSETW": {0xE629 << 15, false, 0, 31, 0, 0x1F},
"XVBITSETW": {0xEE29 << 15, true, 0, 31, 0, 0x1F},
"VBITSETV": {0xE62A << 15, false, 0, 63, 0, 0x3F},
"XVBITSETV": {0xEE2A << 15, true, 0, 63, 0, 0x3F},
"VBITREVB": {0x73182000, false, 0, 7, 0, 0x7},
"XVBITREVB": {0x77182000, true, 0, 7, 0, 0x7},
"VBITREVH": {0x73184000, false, 0, 15, 0, 0xF},
"XVBITREVH": {0x77184000, true, 0, 15, 0, 0xF},
"VBITREVW": {0xE631 << 15, false, 0, 31, 0, 0x1F},
"XVBITREVW": {0xEE31 << 15, true, 0, 31, 0, 0x1F},
"VBITREVV": {0xE632 << 15, false, 0, 63, 0, 0x3F},
"XVBITREVV": {0xEE32 << 15, true, 0, 63, 0, 0x3F},
// The 4-bit-select shuffles and the byte-extract/insert permutations
// take ui8 (the .d shuffle ui4 range-checked to 0..15 by the
// toolchain) packing both position nibbles.
"VSHUF4IB": {0xE720 << 15, false, 0, 255, 0, 0xFF},
"XVSHUF4IB": {0xEF20 << 15, true, 0, 255, 0, 0xFF},
"VSHUF4IH": {0xE728 << 15, false, 0, 255, 0, 0xFF},
"XVSHUF4IH": {0xEF28 << 15, true, 0, 255, 0, 0xFF},
"VSHUF4IW": {0xE730 << 15, false, 0, 255, 0, 0xFF},
"XVSHUF4IW": {0xEF30 << 15, true, 0, 255, 0, 0xFF},
"VSHUF4IV": {0xE738 << 15, false, 0, 15, 0, 0xFF},
"XVSHUF4IV": {0xEF38 << 15, true, 0, 15, 0, 0xFF},
"VPERMIW": {0xE7C8 << 15, false, 0, 255, 0, 0xFF},
"XVPERMIW": {0xEFC8 << 15, true, 0, 255, 0, 0xFF},
"XVPERMIV": {0xEFD0 << 15, true, 0, 255, 0, 0xFF},
"XVPERMIQ": {0xEFD8 << 15, true, 0, 255, 0, 0xFF},
"VEXTRINSB": {0xE718 << 15, false, 0, 255, 0, 0xFF},
"XVEXTRINSB": {0xEF18 << 15, true, 0, 255, 0, 0xFF},
"VEXTRINSH": {0xE710 << 15, false, 0, 255, 0, 0xFF},
"XVEXTRINSH": {0xEF10 << 15, true, 0, 255, 0, 0xFF},
"VEXTRINSW": {0xE708 << 15, false, 0, 255, 0, 0xFF},
"XVEXTRINSW": {0xEF08 << 15, true, 0, 255, 0, 0xFF},
"VEXTRINSV": {0xE700 << 15, false, 0, 255, 0, 0xFF},
"XVEXTRINSV": {0xEF00 << 15, true, 0, 255, 0, 0xFF},
}
for m, e := range vecImm {
l64VecImmInfo[m] = e
l64VecBank[m] = e.lasx
}
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
// vsetallnez.*): the sub-op rides in the rk field.
vecCf := map[string]uint32{
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
"VSETEQV": 0xE539<<15 | 6<<10, "XVSETEQV": 0xED39<<15 | 6<<10,
"VSETANYEQH": 0xE539<<15 | 9<<10, "XVSETANYEQH": 0xED39<<15 | 9<<10,
"VSETANYEQW": 0xE539<<15 | 10<<10, "XVSETANYEQW": 0xED39<<15 | 10<<10,
"VSETALLNEB": 0xE539<<15 | 12<<10, "XVSETALLNEB": 0xED39<<15 | 12<<10,
"VSETALLNEH": 0xE539<<15 | 13<<10, "XVSETALLNEH": 0xED39<<15 | 13<<10,
"VSETALLNEW": 0xE539<<15 | 14<<10, "XVSETALLNEW": 0xED39<<15 | 14<<10,
}
for m, op := range vecCf {
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
l64VecBank[m] = strings.HasPrefix(m, "XV")
}
// Lane popcount and the two-operand vector FP/unary spellings: INSTR vj,
// vd (the 2R layout with the opcode extending over the unused vk field;
// the low byte of each constant is the instruction's own sub-op).
vec2r := map[string]l64Vec3Enc{
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
}
// The rest of the lane popcounts, the vector negations and the vector FP
// unary conversions (loong64enc1.s).
vec2rMore := map[string]l64Vec3Enc{
"VPCNTB": {0x1CA708 << 10, false}, "VPCNTH": {0x1CA709 << 10, false},
"VPCNTW": {0x1CA70A << 10, false},
"VNEGB": {0x1CA70C << 10, false}, "VNEGH": {0x1CA70D << 10, false},
"VNEGW": {0x1CA70E << 10, false}, "VNEGV": {0x1CA70F << 10, false},
"VFCLASSF": {0x1CA735 << 10, false}, "VFCLASSD": {0x1CA736 << 10, false},
"VFSQRTF": {0x1CA739 << 10, false}, "VFSQRTD": {0x1CA73A << 10, false},
"VFRECIPF": {0x1CA73D << 10, false}, "VFRECIPD": {0x1CA73E << 10, false},
"VFRSQRTF": {0x1CA741 << 10, false}, "VFRSQRTD": {0x1CA742 << 10, false},
"VFRINTF": {0x1CA74D << 10, false}, "VFRINTD": {0x1CA74E << 10, false},
"VFRINTRMF": {0x1CA751 << 10, false}, "VFRINTRMD": {0x1CA752 << 10, false},
"VFRINTRPF": {0x1CA755 << 10, false}, "VFRINTRPD": {0x1CA756 << 10, false},
"VFRINTRZF": {0x1CA759 << 10, false}, "VFRINTRZD": {0x1CA75A << 10, false},
"VFRINTRNEF": {0x1CA75D << 10, false}, "VFRINTRNED": {0x1CA75E << 10, false},
"XVPCNTB": {0x1DA708 << 10, true}, "XVPCNTH": {0x1DA709 << 10, true},
"XVPCNTW": {0x1DA70A << 10, true},
"XVNEGB": {0x1DA70C << 10, true}, "XVNEGH": {0x1DA70D << 10, true},
"XVNEGW": {0x1DA70E << 10, true}, "XVNEGV": {0x1DA70F << 10, true},
"XVFCLASSF": {0x1DA735 << 10, true}, "XVFCLASSD": {0x1DA736 << 10, true},
"XVFSQRTF": {0x1DA739 << 10, true}, "XVFSQRTD": {0x1DA73A << 10, true},
"XVFRECIPF": {0x1DA73D << 10, true}, "XVFRECIPD": {0x1DA73E << 10, true},
"XVFRSQRTF": {0x1DA741 << 10, true}, "XVFRSQRTD": {0x1DA742 << 10, true},
"XVFRINTF": {0x1DA74D << 10, true}, "XVFRINTD": {0x1DA74E << 10, true},
"XVFRINTRMF": {0x1DA751 << 10, true}, "XVFRINTRMD": {0x1DA752 << 10, true},
"XVFRINTRPF": {0x1DA755 << 10, true}, "XVFRINTRPD": {0x1DA756 << 10, true},
"XVFRINTRZF": {0x1DA759 << 10, true}, "XVFRINTRZD": {0x1DA75A << 10, true},
"XVFRINTRNEF": {0x1DA75D << 10, true}, "XVFRINTRNED": {0x1DA75E << 10, true},
}
maps.Copy(vec2r, vec2rMore)
for m, e := range vec2r {
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
l64VecBank[m] = e.lasx
l64Vec2R[m] = true
}
// The four-register byte shuffle: INSTR va, vk, vj, vd (the operand the
// table reads in each field position, va at bits [19:15]).
vec4r := map[string]l64Vec3Enc{
"VSHUFB": {0x0D50 << 16, false}, "XVSHUFB": {0x0D60 << 16, true},
}
for m, e := range vec4r {
l64InstrTable[m] = l64Enc{format: l64Fvvvv, op: e.op}
l64VecBank[m] = e.lasx
l64Vec4R[m] = true
}
}
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
// register move between the integer and floating-point register banks — the
// register move between the integer and floating-point register banks, the
// MOVW/MOVV specials the Go assembler accepts.
var l64FpMovTable = map[string]uint32{
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
+540
View File
@@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) {
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
}
}
// The X/V spellings name the LSX/LASX vector banks, a register class of
// their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0`
// with "unrecognized instruction" while assembling `VADDV V0, V1, V2`
// and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector
// operands resolve through loong64VecRegNum only.
vecCases := map[string]int{
"V0": 0, "V31": 31, "X0": 0, "X31": 31,
"R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1,
}
for name, want := range vecCases {
if got := loong64VecRegNum(name); got != want {
t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want)
}
}
}
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
@@ -291,3 +305,529 @@ done:
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestLOONG64IndirectBranch pins the indirect branch encodings: JMP (Rj) and
// JAL (Rj) lower to jirl, and the raw JIRL spelling encodes the written
// offset (the Go loong64 assembler deletes raw JIRL instructions entirely,
// so this form is a gasm-only superset with faithful semantics).
func TestLOONG64IndirectBranch(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
JMP (R4)
JIRL R0, R4, 8
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x4C000080, // jirl r0, r4, 0
0x4C002080, // jirl r0, r4, 8
0x4C000020, // jirl r0, r1, 0 (RET)
)
// JAL (R5) links, so the toolchain gives the function its autosize-8
// prologue and epilogue around the call and the closing RET.
fn = firstTextLOONG64(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
JAL (R5)
RET
`)
code = assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x29FFE061, // st.d r1, -8(r3) (prologue saves RA below the new SP)
0x02FFE063, // addi.d r3, r3, -8 (prologue opens the frame)
0x29C00061, // st.d r1, 0(r3) (prologue saves RA at SP)
0x4C0000A1, // jirl r1, r5, 0
0x28C00061, // ld.d r1, 0(r3) (epilogue restores RA)
0x02C02063, // addi.d r3, r3, 8
0x4C000020, // jirl r0, r1, 0 (RET)
)
}
// TestLOONG64_vector pins the LSX/LASX slice against words read off
// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own
// loong64enc1.s): the three-register forms, the immediate forms with their
// biases, the vector-to-condition forms, lane popcount, the FP conversion,
// FSEL and the VMOVQ move family.
func TestLOONG64_vector(t *testing.T) {
t.Run("three-register and immediate forms", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VADDV V1, V2, V3
VADDW V1, V2, V3
VADDV V2, V1
VANDV V1, V2
VXORV V1, V2, V3
VSEQB V1, V2, V3
VSEQV V1, V2, V3
VSRAB V1, V2, V3
VROTRW V1, V2, V3
VANDB $0, V2, V3
VANDB $255, V2
VSEQB $3, V2, V3
VSEQV $15, V2, V3
VSEQV $-15, V2, V3
VSRAB $7, V1, V2
VROTRW $16, V1, V2
VPCNTV V1, V2
XVADDV X1, X2, X3
XVXORV X1, X2, X3
XVSEQB X1, X2, X3
XVPCNTV X1, X2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x700B8443, // vadd.v v3, v2, v1
0x700B0443, // vadd.w
0x700B8821, // vadd.v v1, v1, v2 (two-operand form)
0x71260442, // vand.v v2, v2, v1
0x71270443, // vxor.v
0x70000443, // vseq.b
0x70018443, // vseq.d
0x70EC0443, // vsra.b
0x70EF0443, // vrotr.w
0x73D00043, // vandi.b v3, v2, 0
0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form)
0x72800C43, // vseqi.b v3, v2, 3
0x7281BC43, // vseqi.d v3, v2, 15
0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement)
0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8)
0x72A0C022, // vrotri.w v2, v1, 16
0x729C2C22, // vpcnt.d v2, v1
0x740B8443, // xvadd.d x3, x2, x1
0x75270443, // xvxor.d
0x74000443, // xvseq.b
0x769C2C22, // xvpcnt.d x2, x1
0x4C000020,
)
})
t.Run("vector-to-condition", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETNEV V1, FCC0
VSETANYEQB V1, FCC0
VSETANYEQV V2, FCC0
VSETALLNEV V0, FCC0
XVSETNEV X1, FCC0
XVSETALLNEV X1, FCC0
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x729C9C20, // vsetnez.d fcc0, v1
0x729CA020, // vsetanyeqz.b
0x729CAC40, // vsetanyeqz.d
0x729CBC00, // vsetallnez.d
0x769C9C20, // xvsetnez.d
0x769CBC20, // xvsetallnez.d
0x4C000020,
)
})
t.Run("FP convert and FSEL", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
FFINTDV F0, F1
FSEL FCC0, F3, F4, F3
FSEL FCC1, F1, F2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x011D2801, // ffint.d.v f1, f0
0x0D000C83, // fsel f3, f4, f3, fcc0
0x0D008442, // fsel f2, f2, f1, fcc1
0x4C000020,
)
})
t.Run("VMOVQ move family", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMOVQ V1, V9
VMOVQ (R4), V2
VMOVQ 16(R4), V2
VMOVQ V0, (R4)
VMOVQ V0, 32(R4)
VMOVQ (R4)(R7), V3
VMOVQ V3, (R4)(R7)
VMOVQ R6, V0.B16
VMOVQ R6, V12.W4
VMOVQ (R4), V4.W4
XVMOVQ X3, X7
XVMOVQ (R4), X2
XVMOVQ X0, (R4)
XVMOVQ (R4)(R7), X4
XVMOVQ X0, (R4)(R7)
XVMOVQ R6, X0.B32
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x732D0029, // vori.b v9, v1, 0 (register move)
0x2C000082, // vld v2, r4, 0
0x2C004082, // vld v2, r4, 16
0x2C400080, // vst v0, r4, 0
0x2C408080, // vst v0, r4, 32
0x38401C83, // vldx v3, r4, r7
0x38441C83, // vstx v3, r4, r7
0x729F00C0, // vreplgr2vr.b v0, r6
0x729F08CC, // vreplgr2vr.w v12, r6
0x30200084, // vldrepl.w v4, r4, 0
0x772D0067, // xvori.b x7, x3, 0
0x2C800082, // xvld x2, r4, 0
0x2CC00080, // xvst x0, r4, 0
0x38481C84, // xvldx x4, r4, r7
0x384C1C80, // xvstx x0, r4, r7
0x769F00C0, // xvreplgr2vr.b x0, r6
0x4C000020,
)
})
t.Run("element extract and insert", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMOVQ V0.V[0], R10
VMOVQ V6.V[1], R8
VMOVQ R9, V1.V[0]
XVMOVQ X0.V[0], R10
XVMOVQ X5.W[7], R7
XVMOVQ R4, X7.V[3]
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72EFF00A, // vpickve2gr.d r10, v0, 0
0x72EFF4C8, // vpickve2gr.d r8, v6, 1
0x72EBF121, // vinsgr2vr.d v1, r9, 0
0x76EFE00A, // xvpickve2gr.d r10, x0, 0
0x76EFDCA7, // xvpickve2gr.w r7, x5, 7
0x76EBEC87, // xvinsgr2vr.d x7, r4, 3
0x4C000020,
)
})
// The integer and FP add/subtract families with their saturating pairs
// and immediate spellings (loong64enc1.s words).
t.Run("add and subtract families", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VADDB V1, V2, V3
VADDF V1, V2, V3
VADDD V1, V2, V3
VSUBD V1, V2, V3
VSADDV V1, V2, V3
VSSUBVU V1, V2, V3
VADDBU $1, V2, V1
VADDBU $1, V2
VSUBVU $31, V2
XVSADDV X3, X2, X1
XVSUBD X1, X2, X3
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x700A0443, // vadd.b
0x71308443, // vadd.f
0x71310443, // vadd.d
0x71330443, // vsub.d
0x70478443, // vsadd.v
0x704D8443, // vssub.u.d
0x728A0441, // vaddi.bu v1, v2, 1
0x728A0442, // vaddi.bu v2, v2, 1 (two-operand form)
0x728DFC42, // vsubi.du v2, v2, 31 (two-operand form)
0x74478C41, // xvsadd.d x1, x2, x3
0x75330443, // xvsub.d x3, x2, x1
0x4C000020,
)
})
// The multiply, divide and accumulate families.
t.Run("multiply and divide families", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMULV V1, V2, V3
VMUHHU V1, V2, V3
VDIVBU V1, V2, V3
VMODV V1, V2, V3
VMADDB V1, V2, V3
VMSUBV V1, V2, V3
VMULWEVHB V1, V2, V3
VMULWODQV V1, V2, V3
VMADDWEVHBUB V1, V2, V3
XVDIVD X1, X2, X3
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x70858443, // vmul.v
0x70888443, // vmuh.u.d
0x70E40443, // vdiv.u.b
0x70E38443, // vmod.d
0x70A80443, // vmadd.b
0x70AB8443, // vmsub.d
0x70900443, // vmulwev.h.b
0x70938443, // vmulwod.q.d
0x70BC0443, // vmaddwev.h.bu.b
0x753B0443, // xvdiv.d
0x4C000020,
)
})
// The shift, bit and interleave families in register and immediate
// spellings, with the width-coded shift immediates.
t.Run("shift, bit and interleave families", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSLLV V1, V2, V3
VROTRB V1, V2, V3
VBITCLRV V1, V2, V3
VBITSETW V1, V2, V3
VBITREVV V1, V2, V3
VILVLB V1, V2, V3
VILVHV V1, V2, V3
VSLLB $7, V1, V2
VSLLB $5, V1
VSRLH $15, V1, V2
VSRAW $31, V1, V2
VSRAV $63, V1, V2
VROTRV $63, V1, V2
VBITCLRB $7, V2, V3
VBITREVV $63, V2, V3
VSEQH $-16, V2, V3
VSLTB $1, V2, V3
VSLTHU $31, V2, V3
XVILVLV X3, X2, X1
XVSLLB $7, X2, X1
XVSRAV $63, X2, X1
XVBITREVV $63, X2, X1
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x70E98443, // vsll.d
0x70EE0443, // vrotr.b
0x710D8443, // vbitclr.d
0x710F0443, // vbitset.w
0x71118443, // vbitrev.d
0x711A0443, // vilvl.b
0x711D8443, // vilvh.d
0x732C3C22, // vslli.b v2, v1, 7
0x732C3421, // vslli.b v1, v1, 5 (two-operand form)
0x73307C22, // vsrli.h v2, v1, 15
0x7334FC22, // vsrai.w v2, v1, 31
0x7335FC22, // vsrai.d v2, v1, 63
0x72A1FC22, // vrotri.d v2, v1, 63
0x73103C43, // vbitclri.b v3, v2, 7
0x7319FC43, // vbitrevi.d v3, v2, 63
0x7280C043, // vseqi.h v3, v2, -16
0x72860443, // vslti.b v3, v2, 1
0x7288FC43, // vslti.hu v3, v2, 31
0x751B8C41, // xvilvl.d x1, x2, x3
0x772C3C41, // xvslli.b x1, x2, 7
0x7735FC41, // xvsrai.d x1, x2, 63
0x7719FC41, // xvbitrevi.d x1, x2, 63
0x4C000020,
)
})
// The shuffle, select and permutation families, including the
// four-register byte shuffle.
t.Run("shuffle and permutation families", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSHUFH V1, V2, V3
VSHUFW V1, V2, V3
VSHUFV V1, V2, V3
VSHUFB V1, V2, V3, V4
XVSHUFB X1, X2, X3, X4
VSHUF4IB $255, V2, V1
VSHUF4IV $15, V2, V1
XVSHUF4IV $15, X1, X2
VEXTRINSB $0x18, V1, V2
XVEXTRINSV $0x81, X1, X2
VPERMIW $0x1B, V1, V2
XVPERMIQ $0x4B, X1, X2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x717A8443, // vshuf.h
0x717B0443, // vshuf.w
0x717B8443, // vshuf.d
0x0D508864, // vshuf.b v4, v3, v2, v1
0x0D608864, // xvshuf.b
0x7393FC41, // vshuf4i.b v1, v2, 255
0x739C3C41, // vshuf4i.d v1, v2, 15
0x779C3C22, // xvshuf4i.d x2, x1, 15
0x738C6022, // vextrins.b v2, v1, 0x18
0x77820422, // xvextrins.d x2, x1, 0x81
0x73E46C22, // vpermi.w v2, v1, 0x1b
0x77ED2C22, // xvpermi.q x2, x1, 0x4b
0x4C000020,
)
})
// The vector FP families, the unary spellings, the compare-to-flag
// additions and the scalar int/float conversions.
t.Run("FP and conversion families", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VADDF V1, V2, V3
VMULF V1, V2, V3
VFCLASSD V1, V2
VFSQRTF V1, V2
VFRECIPD V1, V2
VFRSQRTF V1, V2
VFRINTF V1, V2
VFRINTRNED V1, V2
VNEGB V1, V2
VPCNTB V1, V2
XVNEGV X2, X1
XVPCNTW X3, X2
XVFRINTRNEF X1, X2
VSETEQV V1, FCC0
VSETANYEQH V1, FCC0
VSETALLNEB V1, FCC0
XVSETALLNEW X1, FCC0
FFINTFW F0, F1
FTINTVD F0, F1
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x71308443, // vfadd.s
0x71388443, // vfmul.s
0x729CD822, // vfclass.d
0x729CE422, // vfsqrt.s
0x729CF822, // vfrecip.d
0x729D0422, // vfrsqrt.s
0x729D3422, // vfrint.s
0x729D7822, // vfrintne.s
0x729C3022, // vneg.b
0x729C2022, // vpcnt.b
0x769C3C41, // xvneg.d x1, x2
0x769C2862, // xvpcnt.w x2, x3
0x769D7422, // xvfrintne.s x2, x1
0x729C9820, // vseteqz.d fcc0, v1
0x729CA420, // vsetanyeqz.h
0x729CB020, // vsetallnez.b
0x769CB820, // xvsetallnez.w
0x011D1001, // ffint.s.w f1, f0
0x011B2801, // ftint.l.d f1, f0
0x4C000020,
)
})
}
// TestLOONG64_vectorErrors pins the register-class and range diagnostics of
// the vector slice; each shape is rejected by the oracle as well
// (GOARCH=loong64 go tool asm).
func TestLOONG64_vectorErrors(t *testing.T) {
cases := []string{
// Integer registers in vector positions.
`TEXT ·e(SB), NOSPLIT, $0
VADDV R4, R5, R6
RET
`,
// Crossed banks: LSX spellings take V, LASX spellings X.
`TEXT ·e(SB), NOSPLIT, $0
VADDV X1, X2, X3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVADDV V1, V2, V3
RET
`,
// The LASX bank has no .b/.h element forms.
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ R4, X2.B[0]
RET
`,
// Immediate ranges.
`TEXT ·e(SB), NOSPLIT, $0
VANDB $256, V2
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VSEQB $16, V2, V3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VROTRW $32, V1, V2
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VADDVU $32, V2
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VSEQV $32, V2, V3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VSHUF4IV $16, V2, V1
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VEXTRINSB $256, V1, V2
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VSLTV $-17, V2, V3
RET
`,
// VSHUFB wants four vector registers.
`TEXT ·e(SB), NOSPLIT, $0
VSHUFB V1, V2, V3
RET
`,
// The FCC forms still refuse vector registers.
`TEXT ·e(SB), NOSPLIT, $0
VSETEQV V1, V2
RET
`,
// VSET* wants an FCC flag, not a vector register.
`TEXT ·e(SB), NOSPLIT, $0
VSETNEV V1, V2
RET
`,
}
for i, src := range cases {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants.
// The oracle words come from GOARCH=loong64 go tool objdump of kernels
// assembled with go tool asm, and match the toolchain's loong64enc1.s.
func TestLOONG64_dbarAtomics(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·atoms(SB), NOSPLIT, $0
AMADDDBW R14, (R13), R12
AMADDDBV R14, (R13), R12
AMANDDBW R5, (R4), R6
AMANDDBV R5, (R4), R6
AMORDBW R5, (R4), R0
AMORDBV R5, (R4), R6
AMSWAPDBW R5, (R4), R6
AMCASDBV R6, (R4), R5
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x386A39AC, // amadd_db.w r12, r13, r14
0x386AB9AC, // amadd_db.d
0x386B1486, // amand_db.w r6, r4, r5
0x386B9486, // amand_db.d
0x386C1480, // amor_db.w r0, r4, r5
0x386C9486, // amor_db.d
0x38691486, // amswap_db.w
0x385B9885, // amcas_db.w
0x4C000020,
)
}
+223 -12
View File
@@ -20,14 +20,20 @@ import (
// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`).
// A leaf function (no calls) with a zero frame gets no prologue at all.
//
// Prologue (autosize > 0), byte-identical to the toolchain:
// Prologue (autosize > 0, small), byte-identical to the toolchain:
//
// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe)
// ADDV $-autosize, R3 // open the frame
// MOVV R1, 0(R3) // save LR again at SP (signal-safety)
//
// Large frames (autosize past the 12-bit offset or immediate ranges) expand
// the store and the adjust through REGTMP (R30) exactly as the toolchain's
// assembler does: the store via the rounding LU12IW split, the adjust via
// the floor LU12IW/ORI split.
//
// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR
// restore); the RET's jirl r0, r1, 0 follows.
// restore; the adjust materialised when the immediate does not fit); the
// RET's jirl r0, r1, 0 follows.
// loong64FrameInfo holds the frame layout derived from a TEXT directive.
type loong64FrameInfo struct {
@@ -36,6 +42,11 @@ type loong64FrameInfo struct {
args int // the declared -argsize
noSplit bool // the NOSPLIT flag
leaf bool // no call instructions in the body
// Stack-split guard state: like amd64 and arm64, a leaf function with a
// small autosize is auto-marked NOSPLIT by the toolchain.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// loong64ComputeFrame derives the frame layout for a TEXT function.
@@ -59,9 +70,157 @@ func loong64ComputeFrame(t *ast.Text) loong64FrameInfo {
// A zero-frame non-leaf function still opens an 8-byte frame for LR.
fi.autosize = 8
}
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
// Auto-NOSPLIT, as the toolchain's leaf mark concludes.
default:
fi.needSplit = true
switch {
case fi.autosize <= stackSmall:
fi.splitClass = 0
case fi.autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
// loong64GuardLen returns the byte length of the stack-split guard prefix
// (zero when the function needs no guard). The big class materialises two
// constants through R30; each materialisation shrinks by one word when the
// constant's low 12 bits are zero.
func loong64GuardLen(fi loong64FrameInfo) int {
if !fi.needSplit {
return 0
}
off := int64(fi.autosize - stackSmall)
switch fi.splitClass {
case 0:
return 12
case 1:
if off <= 2048 {
return 16 // ADDV $-off fits the signed 12-bit immediate
}
return 24 // MOVV + LU12IW + ORI + ADDV + SGTU + BEQ
default:
// MOVV + [mat] + SGTU + BNE + [mat] + ADDV + SGTU + BEQ
return (6 + loong64MatLen(off) + loong64MatLen(-off)) * 4
}
}
// loong64MatLen reports the word count of materialising v in R30: a value
// with a zero high part needs only the ORI (the toolchain's MOVW $v, R30),
// one with a zero low part only the LU12IW.
func loong64MatLen(v int64) int {
if v>>12 == 0 || v&0xFFF == 0 {
return 1
}
return 2
}
// loong64MatWords appends the words that materialise v in R30, splitting it
// as v>>12 plus the zero-extended low 12 bits.
func loong64MatWords(ws []uint32, v int64) []uint32 {
hi := v >> 12
lo := v & 0xFFF
if hi == 0 {
return append(ws, l64irr(l64OriOp, int(v), 0, 30))
}
ws = append(ws, l64ir(l64Lu12iwOp, int(hi), 30))
if lo != 0 {
ws = append(ws, l64irr(l64OriOp, int(lo), 30, 30))
}
return ws
}
// The LU12IW and ORI opcode bases (2RI20 and 2RI12 formats); the ORI reads
// and writes rd itself.
const (
l64Lu12iwOp = 0x0a << 25
l64OriOp = 0x0e << 22
)
// loong64Imm12 reports whether v fits a signed 12-bit immediate.
func loong64Imm12(v int64) bool { return v >= -2048 && v <= 2047 }
// loong64GuardBytes emits the stack-split guard prefix. blockStart is the
// function-relative address of the morestack call at the end of the function;
// branch displacements are in instructions and are computed from each
// branch's own position.
func loong64GuardBytes(fi loong64FrameInfo, blockStart int) []byte {
// MOVV 16(g), R20 (g.stackguard0), g = R22.
ws := []uint32{l64irr(l64loadStoreTable["MOVV"].ld, 16, 22, 20)}
off := int64(fi.autosize - stackSmall)
// beq appends BEQ R20, blockStart from the branch's own position.
beq := func() {
ws = append(ws, loong64Beqz(20, int32((blockStart-len(ws)*4)>>2)))
}
switch fi.splitClass {
case 0:
// SGTU SP, R20, R20; BEQ R20, more
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 3, 20, 20))
beq()
case 1:
ws = append(ws, loong64MediumWords(off)...)
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
beq()
default:
// SGTU $off, SP, R24 catches the SP underflow a huge frame would
// cause; BNE jumps to morestack in that case.
ws = append(ws, loong64MatWords(nil, off)...)
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 30, 3, 24))
ws = append(ws, loong64Bnez(24, int32((blockStart-len(ws)*4)>>2)))
ws = append(ws, loong64MatWords(nil, -off)...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
ws = append(ws, l64rrr(l64DualTable["SGTU"].rrr, 24, 20, 20))
beq()
}
return l64WordsLE(ws...)
}
// loong64MediumWords emits the medium-class stack check for offset off: the
// ADDV immediate when it fits, otherwise the same sequence with the constant
// materialised in R30.
func loong64MediumWords(off int64) []uint32 {
if off <= 2048 {
return []uint32{l64irr(l64DualTable["ADDV"].imm, int(-off), 3, 24)}
}
ws := loong64MatWords(nil, -off)
return append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 24))
}
// loong64Beqz/loong64Bnez build the 21-bit conditional branches against R0
// that the toolchain emits for its guard compares.
func loong64Beqz(rj int, dispInstr int32) uint32 {
return l64ir21(l64branch21Table["BEQZ"], int(dispInstr), rj)
}
func loong64Bnez(rj int, dispInstr int32) uint32 {
return l64ir21(l64branch21Table["BNEZ"], int(dispInstr), rj)
}
// loong64MoreStackBlock emits the trailing block: MOVV R1, R31 (save LR, the
// toolchain's OR R1, R0, R31 expansion), BL runtime.morestack_noctxt, B back
// to the function entry.
func loong64MoreStackBlock(blockStart int) ([]byte, Reloc) {
ws := []uint32{
l64rrr(l64DualTable["OR"].rrr, 0, 1, 31), // MOVV R1, R31 (OR R1, R0, R31)
l64bbl(l64jumpTable["BL"], 0), // BL, patched by the linker
}
disp := (-(blockStart + 8)) >> 2
ws = append(ws, l64bbl(l64jumpTable["B"], int(disp)))
reloc := Reloc{
Off: blockStart + 4,
After: blockStart + 8,
Name: "runtime\u00b7morestack_noctxt",
Kind: RelLoong64Branch,
}
return l64WordsLE(ws...), reloc
}
// loong64IsLeaf reports whether a function contains no call instructions
// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame
// and the epilogue shape.
@@ -79,17 +238,37 @@ func loong64IsLeaf(t *ast.Text) bool {
return true
}
// loong64Prologue returns the prologue bytes for a loong64 function.
// loong64Prologue returns the prologue bytes for a loong64 function. When
// the LR store offset leaves the toolchain's 12-bit store range ([-2046,
// 2045], BIG_12 = 2046) or the SP adjust immediate its 12-bit immediate
// range, each switches to the R30 materialisation the assembler expands it
// to: the store uses the rounding %hi/%lo split (LU12IW of (v+2048)>>12,
// REGTMP += SP, store at the raw offset), the adjust the floor split
// (LU12IW, ORI when the low part is non-zero, REGTMP += SP).
func loong64Prologue(fi loong64FrameInfo) []byte {
if fi.autosize == 0 {
return nil
}
addiD := l64DualTable["ADDV"].imm
return l64WordsLE(
l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3)
l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3
l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3)
)
var ws []uint32
storeBase := 3
if fi.autosize > 2046 {
// The store goes through REGTMP: LU12IW of the rounding split,
// REGTMP += SP, then the store at REGTMP with the truncated offset.
v := -int64(fi.autosize)
ws = append(ws, l64ir(l64Lu12iwOp, int((v+2048)>>12), 30))
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 3, 30, 30))
storeBase = 30
}
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, storeBase, 1)) // MOVV R1, -autosize(base)
if loong64Imm12(-int64(fi.autosize)) {
ws = append(ws, l64irr(addiD, -fi.autosize, 3, 3)) // ADDV $-autosize, R3
} else {
ws = append(ws, loong64MatWords(nil, -int64(fi.autosize))...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
}
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1)) // MOVV R1, 0(R3)
return l64WordsLE(ws...)
}
// loong64Return returns the bytes for a RET: the epilogue (restore LR and
@@ -98,17 +277,49 @@ func loong64Return(fi loong64FrameInfo) []byte {
var ws []uint32
if fi.autosize != 0 {
if !fi.leaf {
// MOVV 0(R3), R1 — restore the link register.
// MOVV 0(R3), R1, restore the link register.
ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1))
}
// ADDV $autosize, R3 — close the frame.
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
// ADDV $autosize, R3, close the frame (materialised when the
// immediate does not fit).
if loong64Imm12(int64(fi.autosize)) {
ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3))
} else {
ws = append(ws, loong64MatWords(nil, int64(fi.autosize))...)
ws = append(ws, l64rrr(l64DualTable["ADDV"].rrr, 30, 3, 3))
}
}
// jirl r0, r1, 0 — return.
// jirl r0, r1, 0, return.
ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0))
return l64WordsLE(ws...)
}
// loong64StoreWords reports the prologue word count of the LR store, and
// loong64AdjustWords the word count of an SP adjust of v: the immediate
// forms when they fit, otherwise the R30 materialisation sequences.
func loong64StoreWords(autosize int) int {
if autosize > 2046 {
return 3
}
return 1
}
func loong64AdjustWords(v int64) int {
if loong64Imm12(v) {
return 1
}
return loong64MatLen(v) + 1
}
// loong64EpilogueWords reports the epilogue word count the RET expands to.
func loong64EpilogueWords(fi loong64FrameInfo) int {
n := loong64AdjustWords(int64(fi.autosize))
if !fi.leaf {
n++
}
return n
}
// loong64ResolvePseudo translates a pseudo-register memory reference into a
// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP);
// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable
+82 -6
View File
@@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42
}
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
// invalid register operands and operand-count mismatches.
// invalid register operands and operand-count mismatches. The X0 and
// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects
// `BEQZ X0` (the X bank is not an integer register) and the two-register
// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`).
func TestLOONG64_errors(t *testing.T) {
cases := []string{
`TEXT ·e(SB), NOSPLIT, $0
@@ -280,7 +283,7 @@ done:
// TestLOONG64_pcsp checks the stack-adjustment table of a framed function:
// the prologue raises the SP delta by autosize (in effect from the third
// instruction) and the RET's epilogue restores it to zero, with the pc deltas
// in MinLC (4) units — byte-identical to `go tool asm`.
// in MinLC (4) units; byte-identical to `go tool asm`.
func TestLOONG64_pcsp(t *testing.T) {
cases := []struct {
name string
@@ -347,21 +350,94 @@ TEXT ·sb(SB), NOSPLIT, $0
}
}
// TestLOONG64_movImmToFp checks the immediate-to-FP move forms.
// TestLOONG64_movImmToFp checks the immediate-to-FP move: MOVW $c, Fd is the
// only spelling the toolchain accepts, expanding to ori (or addi.w for the
// negative span) into R30 plus movgr2fr.w. The pinned words are the
// toolchain's own bytes; the other widths and out-of-range constants are
// illegal combinations there and are diagnosed here.
func TestLOONG64_movImmToFp(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·fpmov(SB), NOSPLIT, $0
MOVV $0x1, F0
MOVW $0x1, F0
MOVW $0x2, F4
MOVW $-1, F4
RET
`)
code := assembleLOONG64Helper(t, fn)
want := []byte{
0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1
0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2
0x1e, 0x04, 0x80, 0x03, // ori r30, r0, 1
0xc0, 0xa7, 0x14, 0x01, // movgr2fr.w f0, r30
0x1e, 0x08, 0x80, 0x03, // ori r30, r0, 2
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
0x1e, 0xfc, 0xbf, 0x02, // addi.w r30, r0, -1
0xc4, 0xa7, 0x14, 0x01, // movgr2fr.w f4, r30
0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestLOONG64_movImmToFpErrors checks the immediate-to-FP diagnostics: the
// widths the toolchain rejects as illegal combinations, and constants beyond
// the 12-bit ori/addi.w span (the toolchain never materialises a wider
// constant on this path).
func TestLOONG64_movImmToFpErrors(t *testing.T) {
cases := []string{
"MOVV $1, F0",
"MOVF $2, F4",
"MOVD $2, F4",
"MOVW $100000, F1",
"MOVW $-2049, F1",
"MOVW $4096, F1",
}
for _, src := range cases {
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
// TestLOONG64_branch16Unsigned pins the unsigned two-operand branches: with
// one register BLTU/BGEU keep the register-register form against R0 (never
// taken), the toolchain's encoding, where a beqz would test the wrong
// condition; the three-operand forms are unchanged.
func TestLOONG64_branch16Unsigned(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·u(SB), NOSPLIT, $0
BLTU R4, done
BGEU R5, done
BLTU R6, R7, done
BGEU R8, R9, done
done:
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x68001080, // bltu r4, r0, +4
0x6C000CA0, // bgeu r5, r0, +3
0x680008C7, // bltu r6, r7, +2
0x6C000509, // bgeu r8, r9, +1
0x4C000020, // jirl r0, r1, 0
)
}
// TestLOONG64_bitFieldRange checks the BSTRINS/BSTRPICK bit-number
// validation, mirroring the toolchain's "illegal bit number" rule: 0..31 for
// the .w forms, 0..63 for the .d forms, and lsb <= msb.
func TestLOONG64_bitFieldRange(t *testing.T) {
cases := []string{
"BSTRINSW $32, R4, $0, R5",
"BSTRPICKW $31, R4, $32, R5",
"BSTRINSV $64, R4, $0, R5",
"BSTRPICKV $3, R4, $4, R5",
"BSTRINSW $-1, R4, $0, R5",
}
for _, src := range cases {
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
+42
View File
@@ -0,0 +1,42 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
// relocation offsets of a framed loong64 function: the offsets used to
// exclude the prologue, so every relocation landed on a prologue
// instruction in the GOOBJ/ELF output.
func TestLOONG64RelocOffsetsIncludePrologue(t *testing.T) {
f, errs := parser.Parse("k_loong64.s", "TEXT \u00b7f(SB), $16-0\n"+
"\tMOVV $gdata(SB), R4\n"+
"\tRET\n"+
"GLOBL gdata(SB), $8\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
fn := img.Funcs[0]
// Layout: 12-byte prologue (autosize 32), pcalau12i+addi.d (12, 16),
// epilogue with RET.
if len(fn.Relocs) != 2 {
t.Fatalf("relocs = %d, want 2", len(fn.Relocs))
}
hi, lo := fn.Relocs[0], fn.Relocs[1]
if hi.Kind != RelLoong64AddrHi || hi.Off != 12 || hi.After != 12 {
t.Errorf("hi reloc = {off %d after %d kind %d}, want {off 12 after 12 kind RelLoong64AddrHi}", hi.Off, hi.After, hi.Kind)
}
if lo.Kind != RelLoong64AddrLo || lo.Off != 16 || lo.After != 16 {
t.Errorf("lo reloc = {off %d after %d kind %d}, want {off 16 after 16 kind RelLoong64AddrLo}", lo.Off, lo.After, lo.Kind)
}
}
+59
View File
@@ -14,6 +14,51 @@ type Imm int64
func (Imm) isOperand() {}
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
// low register through the inverted 5-bit V'VVVV field; the three higher
// registers are implied by the instruction, so only the pair travels here.
type RegList struct {
Lo Reg
Hi Reg // implied by the encoding; Lo.idx+3 by construction
}
func (RegList) isOperand() {}
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
// encoding takes an XMM/memory source at that position rewrite it as a read
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
// own behaviour; every other instruction rejects it.
type FloatImm struct {
Text string // the numeric text as written, sign excluded
Neg bool // a leading minus
}
func (FloatImm) isOperand() {}
// TLSMem is a thread-local access, the source form off(base)(TLS*1) with the
// base dropped: the toolchain's one-instruction TLS rewrite assembles it as
// the segment-prefixed absolute whose disp32 carries an R_TLS_LE patch site
// (the linker fills the TLS slot offset).
type TLSMem struct {
Disp int64
Size int
Seg byte // the segment override: FS (0x64) or GS (0x65) on windows
}
func (TLSMem) isOperand() {}
// SegAbs is a segment-absolute access, 0x30(GS): the segment override
// prefixes a disp32 absolute reference with no relocation. The base
// register spellings GS and FS produce it.
type SegAbs struct {
Disp int64
Size int
Seg byte // 0x64 FS, 0x65 GS
}
func (SegAbs) isOperand() {}
// Mem is a memory operand of the form disp(base)(index*scale).
type Mem struct {
Base Reg
@@ -23,6 +68,7 @@ type Mem struct {
Size int // operand width in bytes
HasBase bool
HasIndex bool
Seg byte // segment override prefix (0x64 FS, 0x65 GS); 0 = none
}
func (Mem) isOperand() {}
@@ -48,3 +94,16 @@ type sbMem struct {
}
func (sbMem) isOperand() {}
// isX86Mem reports whether the operand is an amd64 memory reference: a base
// or indexed Mem, or an SB-relative sbMem. Encoders that gate on "memory in
// this position" must accept both; the r/m emitters distinguish the two
// themselves.
func isX86Mem(o Operand) bool {
switch o.(type) {
case Mem, sbMem:
return true
default:
return false
}
}
+14 -9
View File
@@ -12,33 +12,34 @@ import "maps"
import "strings"
// Reg is an x86-64 register. In Plan 9 assembly the classic names (AX, BX, …)
// are size-agnostic — the instruction suffix (MOVQ vs MOVL) fixes the width —
// are size-agnostic, the instruction suffix (MOVQ vs MOVL) fixes the width
// so the encoder keys off the register's index and lets the mnemonic supply the
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4–7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0–K7.
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0–K7 opmask register
mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register
}
// Index returns the register number (0–15 for GPRs, 0–31 for vectors).
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
func (r Reg) Index() int { return r.idx }
// Size returns the width in bytes implied by the register's name.
func (r Reg) Size() int { return r.size }
// IsMask reports whether r is an AVX-512 opmask register (K0–K7).
// IsMask reports whether r is an AVX-512 opmask register (K0-K7).
func (r Reg) IsMask() bool { return r.mask }
func (r Reg) isOperand() {}
// needsREX reports whether this register forces a REX prefix at the given
// operand size: the extended registers R8–R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4–7, not high) do as well.
// operand size: the extended registers R8-R15 always do, and at byte size the
// low registers SPL/BPL/SIL/DIL (indices 4-7, not high) do as well.
func (r Reg) needsREX(opSize int) bool {
if r.idx >= 8 {
return true
@@ -133,7 +134,7 @@ func buildRegByName() map[string]Reg {
}
// Vector: X0..X31 (128-bit, size 16), Y0..Y31 (256-bit, size 32),
// Z0..Z31 (512-bit, size 64). Indices 16–31 are only encodable in EVEX
// Z0..Z31 (512-bit, size 64). Indices 16-31 are only encodable in EVEX
// (AVX-512) instructions; the encoder validates that through its tables.
for i := 0; i <= 31; i++ {
m["X"+itoa(i)] = Reg{idx: i, size: 16}
@@ -144,6 +145,10 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ {
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
}
// x87 stack: F0..F7.
for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
}
return m
}
+1771 -87
View File
File diff suppressed because it is too large Load Diff
+174 -50
View File
@@ -63,9 +63,9 @@ func riscvRegNum(name string) int {
return 24
case "X25", "S9":
return 25
case "X26", "S10":
case "X26", "S10", "CTXT":
return 26
case "X27", "S11":
case "X27", "S11", "g":
return 27
case "X28", "T3":
return 28
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
case "F31", "FT11":
return 31
default:
// Vector registers V0-V31 (the "V" extension). They share the
// register numbering with the integer file: a bare number 0-31.
if len(name) >= 2 && name[0] == 'V' {
if n, ok := parseRegDigits(name[1:], 31); ok {
return n
}
}
return -1
}
}
// parseRegDigits parses a decimal register suffix and reports whether it is
// within [0, max].
func parseRegDigits(digits string, max int) (int, bool) {
if digits == "" {
return 0, false
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return 0, false
}
n = n*10 + int(digits[i]-'0')
if n > max {
return 0, false
}
}
return n, true
}
// RISC-V instruction encoding parameters.
type riscvEnc struct {
opcode uint32 // bits [6:0]
@@ -154,7 +180,7 @@ type riscvEnc struct {
// riscvInstrTable maps RISC-V mnemonics to their encoding.
var riscvInstrTable = map[string]riscvEnc{
// RV64I — R-type arithmetic/logic.
// RV64I, R-type arithmetic/logic.
"ADD": {0x33, 0x0, 0x00},
"SUB": {0x33, 0x0, 0x20},
"SLL": {0x33, 0x1, 0x00},
@@ -165,20 +191,20 @@ var riscvInstrTable = map[string]riscvEnc{
"SRA": {0x33, 0x5, 0x20},
"OR": {0x33, 0x6, 0x00},
"AND": {0x33, 0x7, 0x00},
// RV64I — 32-bit variants (W suffix).
// RV64I, 32-bit variants (W suffix).
"ADDW": {0x3B, 0x0, 0x00},
"SUBW": {0x3B, 0x0, 0x20},
"SLLW": {0x3B, 0x1, 0x00},
"SRLW": {0x3B, 0x5, 0x00},
"SRAW": {0x3B, 0x5, 0x20},
// RV64I — I-type shift-immediate (shamt in rs2 field).
// RV64I, I-type shift-immediate (shamt in rs2 field).
"SLLI": {0x13, 0x1, 0x00},
"SRLI": {0x13, 0x5, 0x00},
"SRAI": {0x13, 0x5, 0x20},
"SLLIW": {0x1B, 0x1, 0x00},
"SRLIW": {0x1B, 0x5, 0x00},
"SRAIW": {0x1B, 0x5, 0x20},
// RV64M — multiply/divide.
// RV64M, multiply/divide.
"MUL": {0x33, 0x0, 0x01},
"MULH": {0x33, 0x1, 0x01},
"MULHSU": {0x33, 0x2, 0x01},
@@ -187,13 +213,16 @@ var riscvInstrTable = map[string]riscvEnc{
"DIVU": {0x33, 0x5, 0x01},
"REM": {0x33, 0x6, 0x01},
"REMU": {0x33, 0x7, 0x01},
// RV64M — 32-bit variants.
// RV64M, 32-bit variants.
"MULW": {0x3B, 0x0, 0x01},
"DIVW": {0x3B, 0x4, 0x01},
"DIVUW": {0x3B, 0x5, 0x01},
"REMW": {0x3B, 0x6, 0x01},
"REMUW": {0x3B, 0x7, 0x01},
// RV64I — I-type arithmetic.
// Zicond conditional zeroing.
"CZEROEQZ": {0x33, 0x5, 0x07},
"CZERONEZ": {0x33, 0x7, 0x07},
// RV64I, I-type arithmetic.
"ADDI": {0x13, 0x0, 0x00},
"ADDIW": {0x1B, 0x0, 0x00},
"SLTI": {0x13, 0x2, 0x00},
@@ -221,38 +250,49 @@ var riscvInstrTable = map[string]riscvEnc{
"BGE": {0x63, 0x5, 0x00},
"BLTU": {0x63, 0x6, 0x00},
"BGEU": {0x63, 0x7, 0x00},
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
// with the register operands swapped.
"BGT": {0x63, 0x4, 0x00},
"BLE": {0x63, 0x5, 0x00},
"BGTU": {0x63, 0x6, 0x00},
"BLEU": {0x63, 0x7, 0x00},
// U-type.
"LUI": {0x37, 0x0, 0x00},
"AUIPC": {0x17, 0x0, 0x00},
// System.
"ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00},
// JALR — indirect jump/call (I-type).
"ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00},
"FENCE.TSO": {0x0F, 0x0, 0x00},
"PAUSE": {0x0F, 0x0, 0x00},
// JALR, indirect jump/call (I-type).
"JALR": {0x67, 0x0, 0x00},
// RV64A — atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
"AMOORW": {0x2F, 0x2, 0x06 << 2},
"AMOORD": {0x2F, 0x3, 0x06 << 2},
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
// RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
// the lower two, exactly as the toolchain writes them: every AMO sets both
// aq and rl (funct7 |= 3).
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
// RV64F/D — floating-point arithmetic.
// RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00},
"FSUBS": {0x53, 0x0, 0x04},
"FMULS": {0x53, 0x0, 0x08},
@@ -273,14 +313,25 @@ var riscvInstrTable = map[string]riscvEnc{
"FMAXS": {0x53, 0x1, 0x14},
"FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15},
// FP sign injection (double): rs2 carries the sign source.
"FSGNJD": {0x53, 0x0, 0x11},
"FSGNJS": {0x53, 0x0, 0x10},
"FSGNJX": {0x53, 0x0, 0x14},
"FSGNJXD": {0x53, 0x0, 0x15},
"FSGNJXS": {0x53, 0x0, 0x14},
"FSGNJND": {0x53, 0x1, 0x11},
"FSGNJNS": {0x53, 0x1, 0x10},
"FSGNJNX": {0x53, 0x1, 0x14},
// RV64A — load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2},
"LRD": {0x2F, 0x3, 0x02 << 2},
"SCW": {0x2F, 0x2, 0x03 << 2},
"SCD": {0x2F, 0x3, 0x03 << 2},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
// The toolchain gives LR acquire ordering (aq = 1) and SC release
// ordering (rl = 1).
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
// FP compare — result in integer register (funct7 0x50/0x51).
// FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50},
"FLTS": {0x53, 0x1, 0x50},
"FLES": {0x53, 0x0, 0x50},
@@ -296,11 +347,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
}
// riscvAMOType encodes an atomic (AMO) instruction.
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
// (the table stores the full field, so the word needs no reassembly).
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
}
@@ -328,6 +379,8 @@ var riscvCvtTable = map[string]riscvCvtEnc{
"FCVTSWU": {0x68, 0x1, 0x53}, // uint32 → float32
"FCVTSL": {0x68, 0x2, 0x53}, // int64 → float32
"FCVTSLU": {0x68, 0x3, 0x53}, // uint64 → float32
"FCLASSS": {0x70, 0x0, 0x53}, // classify float32 → GPR mask
"FCLASSD": {0x70, 0x0, 0x53}, // classify float64 → GPR mask
"FCVTDW": {0x69, 0x0, 0x53}, // int32 → float64
"FCVTDWU": {0x69, 0x1, 0x53}, // uint32 → float64
"FCVTDL": {0x69, 0x2, 0x53}, // int64 → float64
@@ -340,6 +393,10 @@ var riscvCvtTable = map[string]riscvCvtEnc{
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
// The toolchain's W/D suffix spellings of the same moves.
"FMVXS": {0x70, 0x0, 0x53},
"FMVFS": {0x78, 0x0, 0x53},
"FMVSX": {0x79, 0x0, 0x53},
}
// riscvCvtType encodes an FP conversion instruction.
@@ -371,7 +428,7 @@ var riscvFmaTable = map[string]riscvFmaEnc{
// riscvFmaType encodes an R4-type fused multiply-add instruction.
func riscvFmaType(enc riscvFmaEnc, rd, rs1, rs2, rs3 int) uint32 {
return (uint32(rs3) << 27) | (enc.fmt << 25) | (uint32(rs2) << 20) |
(uint32(rs1) << 15) | (0x0 << 12) /* rm=dynamic */ | (uint32(rd) << 7) | enc.opcode
(uint32(rs1) << 15) | (0x0 << 12) /* rm=RNE */ | (uint32(rd) << 7) | enc.opcode
}
// CSR (Control and Status Register) instructions.
@@ -439,13 +496,78 @@ func riscvJType(rd int, offset int32) uint32 {
0x6F // JAL opcode
}
// ---- RVV ("V" extension) encoding helpers ----
// The OP-V major opcode and its funct3 subclasses.
const (
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
riscvVf3VV = 0x0 // vector-vector
riscvVf3MV = 0x2 // vector mask
riscvVf3VI = 0x3 // vector-immediate
riscvVf3VX = 0x4 // vector-scalar
riscvVf3Cfg = 0x7 // vsetvli
)
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
// group multiplier in [2:0], the selected element width in [5:3] and the
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
func riscvVType(vsew, vlmul, vta, vma int) int {
return vlmul | vsew<<3 | vta<<6 | vma<<7
}
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
// the toolchain writes above its 10-bit vtype).
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
imm := vtype & 0x3FF
if vsetivli {
imm |= 0xC00
}
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
uint32(rd)<<7 | riscvOpV
}
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
// segment count in bits [31:29], the addressing mode in bits [28:26], the
// unmasked bit at 25 and the width in funct3. width follows the load
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
uint32(rd)<<7 | op
}
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
// the standard positions. vs1 may name an integer register for the *VX forms
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
}
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
// fields live where the second source register would be: rs1Field and vs2 are
// written verbatim (the oracle writes fixed non-zero constants there for some
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
}
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
func riscvSegNF(n int) int32 { return int32(n - 1) }
// ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
// prime register field used by compressed instructions (x8–x15).
// prime register field used by compressed instructions (x8-x15).
func isRVCIntReg(r int) bool { return r >= 8 && r <= 15 }
// rvcReg3 returns the 3-bit encoding for registers x8–x15 (0–7).
// rvcReg3 returns the 3-bit encoding for registers x8-x15 (0-7).
func rvcReg3(r int) uint32 { return uint32(r - 8) }
// rvcCR encodes a CR-type (register) compressed instruction.
@@ -455,7 +577,7 @@ func rvcCR(funct4, rd, rs2 uint32) uint16 {
}
// rvcCI encodes a CI-type (immediate) compressed instruction.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW — linear 6-bit immediate.
// Used for C.ADDI, C.LI, C.LUI, C.ADDIW, linear 6-bit immediate.
func rvcCI(funct3, rd uint32, imm uint32) uint16 {
return uint16((funct3 << 13) | ((imm>>5)&1)<<12 | (rd << 7) | (imm&0x1F)<<2 | 0x1)
}
@@ -520,11 +642,13 @@ func rvcCL(funct3, rd, rs1 uint32, imm uint32) uint16 {
// rvcCS encodes a register-relative compressed store (op=00 quadrant): C.SW
// (funct3=6), C.SD (funct3=7) or C.FSD (funct3=5). imm is the full byte
// offset; the immediate bits are extracted per the RISC-V CS format.
// offset; the immediate bits are extracted per the RISC-V CS format, with the
// same five-bit patterns as the load side ({5,4,3,7,6} and {5,4,3,2,6},
// matching the toolchain's encodeCS).
func rvcCS(funct3, rs2, rs1 uint32, imm uint32) uint16 {
pattern := []int{5, 3, 7, 6}
pattern := []int{5, 4, 3, 7, 6}
if funct3 == 0x6 {
pattern = []int{5, 3, 2, 6}
pattern = []int{5, 4, 3, 2, 6}
}
packed := encodeRVCPattern(imm, pattern)
return uint16((funct3 << 13) | ((packed>>2)&0x7)<<10 | (rs1 << 7) | ((packed & 0x3) << 5) | (rs2 << 2))
+589 -10
View File
@@ -5,6 +5,9 @@ package asm
import (
"bytes"
"encoding/binary"
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -30,7 +33,7 @@ func firstTextRISCV(t *testing.T, src string) *ast.Text {
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
t.Helper()
code, _, _, _, _, err := assembleRISCV(fn)
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -289,7 +292,7 @@ TEXT ·cmp(SB), NOSPLIT, $0
}
func TestRISCV_forwardBranch(t *testing.T) {
// Forward label reference — must not fail.
// Forward label reference; must not fail.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fwd(SB), NOSPLIT, $0
ADDI $1, X10, X10
@@ -691,6 +694,81 @@ DATA answer<>+0(SB)/8, $42
}
}
// TestRISCV_RVC_StorePatterns pins the register-relative compressed store
// encodings for offsets with immediate bits 4 and 5 set, byte-identical to
// the toolchain's encodeCS (patterns {5,4,3,7,6} and {5,4,3,2,6}).
// Regression: the store-side patterns dropped imm[4], so every such store
// silently encoded the wrong address while the loads stayed correct.
func TestRISCV_RVC_StorePatterns(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csstores(SB), NOSPLIT, $0
SD X9, 24(X8)
SW X10, 16(X11)
FSD F8, 40(X12)
LD 24(X8), X9
LW 16(X11), X10
FLD 40(X12), F8
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0x04, 0xec, // c.sd x9, 24(x8)
0x88, 0xc9, // c.sw x10, 16(x11)
0x00, 0xb6, // c.fsd f8, 40(x12)
0x04, 0x6c, // c.ld x9, 24(x8)
0x88, 0x49, // c.lw x10, 16(x11)
0x00, 0x36, // c.fld f8, 40(x12)
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestRISCV_FENCE pins the FENCE encoding: the toolchain expands the bare
// mnemonic to fence iorw, iorw (0x0FF0000F), not fence 0,0.
func TestRISCV_FENCE(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·fence(SB), NOSPLIT, $0
FENCE
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0x0f, 0x00, 0xf0, 0x0f, // fence iorw, iorw
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
// TestRISCV_RVC_WidthSpellings pins the compression of the GOROOT width
// spellings: MOVW and MOVD lower to their base load/store and compress
// exactly like LW/SW/FLD/FSD would (the toolchain compresses these shapes;
// before the normalisation they stayed 4 bytes).
func TestRISCV_RVC_WidthSpellings(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·widths(SB), NOSPLIT, $0-16
MOVW w+0(FP), X9
MOVW X9, v+4(FP)
MOVD d+0(FP), F8
MOVD F8, r+8(FP)
RET
`)
code := assembleRISCVHelper(t, fn)
want := []byte{
0xa2, 0x44, // c.lwsp x9, 8
0x26, 0xc6, // c.swsp x9, 12
0x22, 0x24, // c.fldsp f8, 8
0x22, 0xa8, // c.fsdsp f8, 16
0x67, 0x80, 0x00, 0x00, // jalr x0, 0(x1)
}
if !bytes.Equal(code, want) {
t.Errorf("code = % x\nwant % x", code, want)
}
}
func TestRISCV_system_instrs(t *testing.T) {
// Test FENCE, ECALL, EBREAK encoding.
fn := firstTextRISCV(t, `#include "textflag.h"
@@ -707,16 +785,140 @@ TEXT ·sys(SB), NOSPLIT, $0
}
}
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
// MOV $sym(FP), rd should return an error (unsupported).
func TestRISCV_MOV_sym_FP(t *testing.T) {
// MOV $sym(FP), rd lowers to the frame-adjusted ADDI against SP: the
// toolchain's argframe spelling. A zero frame leaves the offset at the
// 8-byte link slot, compressed to C.ADDI4SPN.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·badfp(SB), NOSPLIT, $0
TEXT ·argfp(SB), NOSPLIT, $0
MOV $arg(FP), X10
RET
`)
_, _, _, _, _, err := assembleRISCV(fn)
if err == nil {
t.Error("expected error for MOV $arg(FP), got nil")
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// prologue (0: leaf, zero frame) + C.ADDI4SPN (2) + RET (4) = 6
want := []byte{0x28, 0x00, 0x67, 0x80, 0x00, 0x00}
if string(code) != string(want) {
t.Errorf("got % x, want % x", code, want)
}
}
func TestRISCV_Bookkeeping(t *testing.T) {
// FUNCDATA and PCDATA contribute no bytes; UNDEF is the toolchain's
// ebreak, compressed to C.EBREAK under RVC.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·book(SB), NOSPLIT, $0-8
FUNCDATA $0, marks<>(SB)
PCDATA $1, $1
UNDEF
MOV $1, X10
MOV X10, ret+0(FP)
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// C.EBREAK (2) + C.LI X10, 1 (2) + C.SWSP (2) + RET (4) = 10: the
// FUNCDATA and PCDATA statements contribute nothing.
want := []byte{0x02, 0x90, 0x05, 0x45, 0x2a, 0xe4, 0x67, 0x80, 0x00, 0x00}
if string(code) != string(want) {
t.Errorf("got % x, want % x", code, want)
}
}
func TestRISCV_JMPPCRel(t *testing.T) {
// JMP N(PC): the displacement tracks the instruction N source slots
// away in the final layout (0 the jump itself, negative backwards).
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·slots(SB), NOSPLIT, $0-0
JMP 2(PC)
MOV $1, X11
MOV $2, X12
MOV X12, X11
JMP -3(PC)
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// JMP 2(PC) lands on the C.MV six bytes ahead; JMP -3(PC) lands back on
// the first C.LI, six bytes behind.
want := []byte{
0x6f, 0x00, 0x60, 0x00, // JAL X0, 6
0x85, 0x45, // C.LI X11, 1
0x09, 0x46, // C.LI X12, 2
0xb2, 0x85, // C.MV X11, X12
0x6f, 0xf0, 0xbf, 0xff, // JAL X0, -6
0x67, 0x80, 0x00, 0x00, // RET
}
if string(code) != string(want) {
t.Errorf("got % x, want % x", code, want)
}
}
func TestRISCV_MOVWideImm(t *testing.T) {
// Shift-sequence constants compress like the toolchain's expansion.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·wide(SB), NOSPLIT, $0-0
MOV $0x8000000000000000, X5
MOV $0x100000000, X5
MOV $0x000fffffffffffda, X5
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
// C.LI -1, C.SLLI 63; C.LI 1, C.SLLI 32; C.LI -19, C.SLLI 13, SRLI 12.
want := []byte{
0xfd, 0x52, 0xfe, 0x12,
0x85, 0x42, 0x82, 0x12,
0xb5, 0x52, 0xb6, 0x02, 0x93, 0xd2, 0xc2, 0x00,
0x67, 0x80, 0x00, 0x00,
}
if string(code) != string(want) {
t.Errorf("got % x, want % x", code, want)
}
}
func TestRISCV_MOVImmPool(t *testing.T) {
// A constant outside the shift shapes loads from the pooled $i64 data
// symbol via AUIPC+LD, named like the toolchain's pool.
src := `#include "textflag.h"
TEXT ·pool(SB), NOSPLIT, $0-8
MOV $0x0101010101010101, X16
MOV X16, ret+0(FP)
RET
`
f, errs := parser.Parse("pool_riscv64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
// AUIPC X16, 0 + LD X16, 0(X16): the relocation pair carries the symbol.
wantCode := []byte{0x17, 0x08, 0x00, 0x00, 0x03, 0x38, 0x08, 0x00}
if string(img.Code[0:8]) != string(wantCode) {
t.Errorf("pool load: got % x", img.Code[0:8])
}
var lit *DataSymbol
for i := range img.DataSyms {
if img.DataSyms[i].Name == "$i64.0101010101010101" {
lit = &img.DataSyms[i]
}
}
if lit == nil {
t.Fatalf("pool symbol missing: %v", img.DataSyms)
}
wantData := []byte{0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01}
if string(img.Data[lit.Offset:lit.Offset+8]) != string(wantData) {
t.Errorf("pool bytes: got % x", img.Data[lit.Offset:lit.Offset+8])
}
}
@@ -727,7 +929,7 @@ TEXT ·calltest(SB), NOSPLIT, $0
CALL ext(SB)
RET
`)
code, _, relocs, _, _, err := assembleRISCV(fn)
code, _, relocs, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -756,8 +958,385 @@ TEXT ·calllocal(SB), NOSPLIT, $0
sub:
RET
`)
_, _, _, _, _, err := assembleRISCV(fn)
_, _, _, _, _, _, err := assembleRISCV(fn)
if err == nil {
t.Error("expected error for CALL to local label, got nil")
}
}
// TestRISCVIndirectBranch pins the indirect branch encodings: JMP (X5) is the
// toolchain's JALR X0, 0(X5), and the trampoline form JALR rd, offset(rs1)
// takes its destination from the first operand (regression: the base
// register was once read as the destination, silently jumping to X0).
func TestRISCVIndirectBranch(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0-0
JMP (X5)
JALR X0, 0(X6)
JALR X28, 0(X9)
RET
`)
code := assembleRISCVHelper(t, fn)
wantWords(t, code,
0x00028067, // jalr x0, 5(x0), 0
0x00030067, // jalr x0, 6(x0), 0
0x00048e67, // jalr x28, 9(x0), 0
0x00008067, // jalr x0, 1(x0), 0 (RET)
)
}
// encodeOneInstrRISCV encodes a single parsed instruction against a synthetic
// offsets map, the smallest honest harness for the branch-range diagnostics:
// the spans are far larger than any source a test would want to spell out.
func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]int) ([]byte, error) {
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
instr := fn.Body[0].(*ast.Instr)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil, nil)
}
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
// [-4096, 4094] and the J-type span [-1048576, 1048574] are diagnosed instead
// of wrapping silently to a wrong target.
func TestRISCVBranchJumpRange(t *testing.T) {
cases := []struct {
name string
src string
off int // the target's function-relative offset (pc 0)
ok bool
}{
{"branch max", "BEQ X10, X11, tgt\nRET\n", 4094, true},
{"branch past max", "BEQ X10, X11, tgt\nRET\n", 4096, false},
{"branch back max", "BEQ X10, X11, tgt\nRET\n", -4096, true},
{"branch back past max", "BEQ X10, X11, tgt\nRET\n", -4098, false},
{"branchz past max", "BEQZ X10, tgt\nRET\n", 4096, false},
{"jump max", "JMP tgt\nRET\n", 1048574, true},
{"jump past max", "JMP tgt\nRET\n", 1048576, false},
{"jump back max", "JMP tgt\nRET\n", -1048576, true},
{"jump back past max", "JMP tgt\nRET\n", -1048578, false},
{"jal past max", "JAL tgt\nRET\n", 1048576, false},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
_, err := encodeOneInstrRISCV(t, "TEXT ·f(SB), NOSPLIT, $0\n\t"+c.src, 0, map[string]int{"tgt": c.off})
if c.ok && err != nil {
t.Fatalf("unexpected error: %v", err)
}
if !c.ok && err == nil {
t.Fatal("expected an out-of-range diagnostic, got none")
}
})
}
}
// TestRISCVBranchFarBody drives the relaxation pass through the full
// assembler: a forward branch over a body larger than the B-type span is
// rewritten as an inverted branch over an inserted JMP, the same layout the
// toolchain produces, instead of wrapping to a wrong target.
func TestRISCVBranchFarBody(t *testing.T) {
var sb strings.Builder
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
for range 1100 {
sb.WriteString("\tADD X10, X11, X12\n")
}
sb.WriteString("done:\n\tRET\n")
fn := firstTextRISCV(t, sb.String())
out, _, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
// The relaxed branch at offset 0 targets the inserted JMP at 4 (bne
// x10, x11, +4); the JMP at 4 carries the far forward displacement.
wantBranch := wordLE(riscvBType(riscvEnc{0x63, 0x1, 0x00}, 10, 11, 4))
if !bytes.Equal(out[0:4], wantBranch) {
t.Errorf("relaxed branch = %x, want %x", out[0:4], wantBranch)
}
// done sits after 1100 ADDs: 4 + 4400, i.e. offset 4404 from the JMP at 4.
wantJmp := wordLE(riscvJType(0, 4404))
if !bytes.Equal(out[4:8], wantJmp) {
t.Errorf("inserted JMP = %x, want %x", out[4:8], wantJmp)
}
}
// TestRISCV_CSRRange checks the CSR address range: the 12-bit field is
// diagnosed rather than masked, so CSRRW $4096 does not silently address
// CSR 0.
func TestRISCV_CSRRange(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·csrhi(SB), NOSPLIT, $0
CSRRW $4096, X10, X11
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Error("expected an out-of-range error for CSR $4096, got none")
}
fn = firstTextRISCV(t, `#include "textflag.h"
TEXT ·csrmax(SB), NOSPLIT, $0
CSRRW $4095, X10, X11
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("CSR $4095 must assemble: %v", err)
}
}
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
// span are diagnosed instead of silently truncated to their low 32 bits for
// the I-type arithmetic; the MOV forms materialise the wide constant instead
// (shift sequence or pooled load), like the toolchain.
func TestRISCV_Imm64Rejected(t *testing.T) {
cases := []string{
"ADDI $0x100000000, X10, X11",
"ANDI $-0x800000001, X10, X11",
"SUB $0x100000000, X10, X11",
}
for _, src := range cases {
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Errorf("%s: expected an out-of-range error, got none", src)
}
}
// The full signed 32-bit span still assembles, including the SUB form
// whose negated immediate only just fits.
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·edge(SB), NOSPLIT, $0
MOV $2147483647, X10
MOV $-2147483648, X11
SUB $0x80000000, X12, X13
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("int32-span immediates must assemble: %v", err)
}
// Beyond the span the MOV forms materialise the constant like the
// toolchain instead of diagnosing it.
fn = firstTextRISCV(t, `#include "textflag.h"
TEXT ·pool(SB), NOSPLIT, $0
MOV $0x123456789, X10
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
t.Errorf("MOV with a 64-bit immediate must assemble: %v", err)
}
}
// riscvWants decodes code as little-endian words and pins each one; the
// expected values below were read off GOARCH=riscv64 go tool objdump of
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWants(t *testing.T, code []byte, want ...uint32) {
t.Helper()
got := make([]uint32, 0, len(code)/4)
for i := 0; i+4 <= len(code); i += 4 {
got = append(got, binary.LittleEndian.Uint32(code[i:]))
}
if len(got) < len(want) {
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
}
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// riscvWantsHex pins the exact hex encoding of a function's instruction
// bytes, including any 2-byte compressed instructions in the stream; the
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
// assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
t.Helper()
got := hex.EncodeToString(code)
if got != wantHex {
t.Errorf("code = %s, want %s", got, wantHex)
}
}
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
// the time CSR).
func TestRISCV_extendedPseudos(t *testing.T) {
t.Run("logic and minmax", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·l(SB), NOSPLIT, $0
ANDN X19, X20, X21
ANDN X19, X20
ORN X20, X19
MAX X26, X28, X29
MIN X29, X30, X5
MAX X5, X5
MAX X5, X5, X6
SEQZ X5, X6
NEG X5, X6
NOT X5
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
// then SEQZ, NEG, NOT and RDTIME.
riscvWantsHex(t, code,
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
"9682"+"1683"+
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
})
t.Run("rotate", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·r(SB), NOSPLIT, $0
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORIW $31, X13, X14
RORIW $1, X14, X15
RORIW $3, X14
RORW X15, X16, X17
RORW $31, X13
RET
`)
code := assembleRISCVHelper(t, fn)
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
riscvWantsHex(t, code,
"b30fa040b39ff50133d6a50033e6cf00"+
"b30fa040b39ff501b3d5a500b3e5bf00"+
"93dff5038605b3e5bf00"+
"9bdff6011b97160033e7ef00"+
"9b5f17009b17f701b3e7ff00"+
"9b5f37001b17d70133e7ef00"+
"b30ff040bb1ff801bb58f800b3e81f01"+
"9bdff6019b961600b3e6df00"+"67800000")
})
t.Run("fp and branches", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
FABSD F1, F2
FSGNJD F1, F0, F2
FMADDD F1, F2, F3, F4
FMSUBD F1, F2, F3, F4
FNMSUBD F1, F2, F3, F4
BGT X5, X6, tgt
BLE X5, X6, tgt
BGTU X5, X6, tgt
BLEU X5, X6, tgt
tgt:
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWantsHex(t, code,
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
"f32210c0"+"67800000")
})
}
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
// exactly as GOARCH=riscv64 go tool asm encodes them.
func TestRISCV_amoWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0
AMOSWAPW X5, (X6), X7
AMOSWAPD X5, (X6), X7
AMOADDW X5, (X6), X7
AMOADDD X5, (X6), X7
AMOANDW X5, (X6), X7
AMOANDD X5, (X6), X7
AMOORW X5, (X6), X7
AMOORD X5, (X6), X7
AMOXORW X5, (X6), X7
AMOXORD X5, (X6), X7
AMOMAXW X5, (X6), X7
AMOMAXD X5, (X6), X7
AMOMAXUW X5, (X6), X7
AMOMAXUD X5, (X6), X7
AMOMINUW X5, (X6), X7
AMOMINUD X5, (X6), X7
LRW (X5), X6
LRD (X5), X6
SCW X5, (X6), X7
SCD X5, (X6), X7
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0E5323AF, // amoswap.w
0x0E5333AF, // amoswap.d
0x065323AF, // amoaddd.w
0x065333AF, // amoadd.d
0x665323AF, // amoand.w
0x665333AF, // amoand.d
0x465323AF, // amoor.w
0x465333AF, // amoor.d
0x265323AF, // amoxor.w
0x265333AF, // amoxor.d
0xA65323AF, // amomax.w
0xA65333AF, // amomax.d
0xE65323AF, // amomaxu.w
0xE65333AF, // amomaxu.d
0xC65323AF, // amominu.w
0xC65333AF, // amominu.d
0x1402A32F, // lr.w (aq)
0x1402B32F, // lr.d
0x1A5323AF, // sc.w (rl)
0x1A5333AF, // sc.d
)
}
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
// toolchain canonicalises an immediate avl to vsetivli even under the
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
// identical), which is what the 0xC00 bit of the first word carries.
func TestRISCV_vectorWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETVLI X5, E8, M8, TA, MA, X6
VSETIVLI $4, E32, M1, TA, MA, X0
VSETVLI $15, E32, M1, TA, MA, X12
VADDVV V1, V2, V3
VADDVX X12, V12, V12
VXORVV V8, V16, V24
VMSEQVX X12, V8, V0
VMSNEVV V8, V16, V0
VSLLVI $8, V28, V30
VSRLVI $25, V29, V29
VFIRSTM V0, X6
VIDV V12
VMV4RV V8, V24
VLE8V (X10), V8
VSE8V V24, (X10)
VSE32V V9, (X11)
VLSSEG4E32V (X14), X0, V0
VLSSEG8E32V (X10), X0, V4
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
0xCD027057, // vsetivli x0, 4
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
0x022081D7, // vadd.vv v3, v2, v1
0x02C64657, // vadd.vx v12, v12, x12
0x2F040C57, // vxor.vv v24, v16, v8
0x62864057, // vmseq.vx v0, v8, x12
0x67040057, // vmsne.vv v0, v16, v8
0x97C43F57, // vsll.vi v30, v28, 8
0xA3DCBED7, // vsrl.vi v29, v29, 25
0x4208A357, // vmfirst.m x6, v0
0x5208A657, // vid.v v12
0x9E81BC57, // vmv4r.v v24, v8
0x02050407, // vle8.v v8, (x10)
0x02050C27, // vse8.v v24, (x10)
0x0205E4A7, // vse32.v v9, (x11)
0x6A076007, // vlsseg4e32.v v0, (x14), x0
0xEA056207, // vlsseg8e32.v v4, (x10), x0
)
}
+211 -29
View File
@@ -4,6 +4,7 @@
package asm
import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -32,6 +33,13 @@ import (
// riscvFrameInfo holds the frame layout derived from a TEXT directive.
type riscvFrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR)
// Stack-split guard state: the toolchain emits the check for every
// non-NOSPLIT function whose autosize is nonzero (a zero autosize is
// "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf
// auto-NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
}
// riscvComputeFrame derives the frame layout for a TEXT function.
@@ -40,11 +48,34 @@ func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
if frame != 0 || !riscvIsLeaf(t) {
// FixedFrameSize = 8: space for the saved link register. A
// zero-frame non-leaf function still opens an 8-byte frame for LR.
return riscvFrameInfo{autosize: frame + 8}
autosize := frame + 8
fi := riscvFrameInfo{autosize: autosize}
if !hasNoSplitFlag(t) {
fi.needSplit = true
switch {
case autosize <= stackSmall:
fi.splitClass = 0
case autosize <= stackBig:
fi.splitClass = 1
default:
fi.splitClass = 2
}
}
return fi
}
return riscvFrameInfo{}
}
// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT.
func hasNoSplitFlag(t *ast.Text) bool {
for _, f := range t.Flags {
if strings.EqualFold(f, "NOSPLIT") {
return true
}
}
return false
}
// riscvIsLeaf reports whether a function contains no call instructions.
// CALL always links; JAL/JALR link only when their destination register is
// the link register (X1), matching cmd/internal/obj/riscv's containsCall.
@@ -58,17 +89,24 @@ func riscvIsLeaf(t *ast.Text) bool {
case "CALL":
return false
case "JAL":
// JAL rd, target — a call only when rd is the link register.
// JAL rd, target, a call only when rd is the link register.
if len(in.Operands) >= 2 && regFromOperand(in.Operands[0]) == 1 {
return false
}
case "JALR":
// JALR rs1, rd — a call when rd is X1; JALR offset(rs1) always
// links to X1.
// JALR rd, offset(rs1) links when the destination register (the
// first operand) is X1; JALR rs1, rd links when the second
// register is X1; JALR offset(rs1) always links to X1.
if len(in.Operands) == 1 {
return false
}
if len(in.Operands) >= 2 && regFromOperand(in.Operands[1]) == 1 {
if isMemOperand(in.Operands[1]) {
if regFromOperand(in.Operands[0]) == 1 {
return false
}
continue
}
if regFromOperand(in.Operands[1]) == 1 {
return false
}
}
@@ -84,28 +122,99 @@ func riscvPrologue(fi riscvFrameInfo) []byte {
return nil
}
var out []byte
// MOV LR, -autosize(SP) — SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes.
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
// ADDI $-autosize, SP, SP — open the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
// MOV LR, 0(SP) — SD X1, 0(X2) → C.SDSP X1, 0.
// MOV LR, -autosize(SP), SD X1, -autosize(X2). The negative offset is
// not compressible to C.SDSP (unsigned), so it stays 4 bytes. Beyond
// the imm12 range the toolchain materialises the address in X31.
if fits12(int32(-fi.autosize)) {
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 2, 1, int32(-fi.autosize)))...)
} else {
out = append(out, riscvAddressInX31(int32(-fi.autosize))...)
lo := int32(-fi.autosize) - (splitHi(int32(-fi.autosize)) << 12)
out = append(out, wordLE(riscvSType(riscvEnc{0x23, 0x3, 0x00}, 31, 1, lo))...)
}
// ADDI $-autosize, SP, SP, open the frame (C.ADDI when it fits; X31
// materialisation beyond imm12).
if fits12(int32(-fi.autosize)) {
out = append(out, riscvSPAdjust(int32(-fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(-fi.autosize))...)
}
// MOV LR, 0(SP), SD X1, 0(X2) → C.SDSP X1, 0.
c := rvcSSP(0x7, 1, 0)
out = append(out, byte(c), byte(c>>8))
return out
}
func fits12(v int32) bool { return v >= -2048 && v <= 2047 }
// splitHi returns the LUI half of the hi/lo split of v (what remains is the
// sign-extended 12-bit low part).
func splitHi(v int32) int32 {
_, high := splitRISCV32Imm(v)
return high
}
// riscvAddressInX31 materialises hi(v) into X31 against the stack pointer,
// matching the toolchain's large-frame addressing: C.LUI (or LUI) X31, hi;
// C.ADD (or ADD) X31, SP.
func riscvAddressInX31(v int32) []byte {
return riscvAddressInX31WithBase(v, 2)
}
// riscvAddressInX31WithBase materialises hi(v) into X31 against an arbitrary
// base register: LUI (or C.LUI) X31, hi; C.ADD X31, rs1. The CR rs2 field
// carries the full 5-bit register, so the compressed form is always
// available.
func riscvAddressInX31WithBase(v int32, rs1 int) []byte {
hi := splitHi(v)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
c := rvcCR(0x9, 31, uint32(rs1))
return append(out, byte(c), byte(c>>8))
}
// riscvAddToSP adds v to SP through X31 for the values imm12 cannot carry:
// C.LUI X31, hi; C.ADDIW X31, lo; C.ADD SP, X31 (the toolchain's form).
func riscvAddToSP(v int32) []byte {
hi := splitHi(v)
lo := v - (hi << 12)
var out []byte
if hi >= -32 && hi <= 31 {
c := rvcCI(0x3, 31, uint32(hi)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvUType(riscvEnc{0x37, 0x0, 0x00}, 31, hi<<12))...)
}
if lo >= -32 && lo <= 31 {
c := rvcCI(0x1, 31, uint32(lo)&0x3F)
out = append(out, byte(c), byte(c>>8))
} else {
out = append(out, wordLE(riscvIType(riscvEnc{0x1b, 0x0, 0x00}, 31, 31, lo))...)
}
c := rvcCR(0x9, 2, 31)
return append(out, byte(c), byte(c>>8))
}
// riscvReturn returns the bytes for a RET: the epilogue (restore LR and
// deallocate the frame when present) followed by the uncompressed JALR X0,
// 0(X1) the toolchain emits for RET (it never compresses RET to C.JR).
func riscvReturn(fi riscvFrameInfo) []byte {
var out []byte
if fi.autosize != 0 {
// MOV 0(SP), LR — LD X1, 0(X2) → C.LDSP X1, 0.
// MOV 0(SP), LR, LD X1, 0(X2) → C.LDSP X1, 0.
c := rvcLSP(0x3, 1, 0)
out = append(out, byte(c), byte(c>>8))
// ADDI $autosize, SP, SP — close the frame (C.ADDI when it fits).
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
// ADDI $autosize, SP, SP, close the frame (C.ADDI when it fits).
if fits12(int32(fi.autosize)) {
out = append(out, riscvSPAdjust(int32(fi.autosize))...)
} else {
out = append(out, riscvAddToSP(int32(fi.autosize))...)
}
}
// JALR X0, 0(X1).
return append(out, wordLE(riscvIType(riscvEnc{0x67, 0x0, 0x00}, 0, 1, 0))...)
@@ -133,33 +242,36 @@ func riscvFitsCAddi(imm int32) bool {
}
// riscvPrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
// prologue has finished decrementing SP (the delta becomes autosize). It is
// computed from the same expansion functions the prologue emits, so the
// large-frame X31 materialisations are counted: C.LUI + C.ADD before the SD,
// C.LUI + ADDIW + C.ADD for the SP adjust.
func riscvPrologueSpadjPC(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 4 + riscvSPAdjustLen(int32(-fi.autosize))
adj := int32(-fi.autosize)
if fits12(adj) {
// SD (4 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 4 + len(riscvSPAdjust(adj))
}
return len(riscvAddressInX31(adj)) + 4 + len(riscvAddToSP(adj))
}
// riscvReturnEpilogueLen returns the byte length of the RET's epilogue up to
// (but not including) the final JALR — the point where SP is restored.
// (but not including) the final JALR, the point where SP is restored. The
// small frame closes with C.LDSP + ADDI/C.ADDI; the large frame materialises
// the adjustment through X31 (C.LUI + ADDIW + C.ADD).
func riscvReturnEpilogueLen(fi riscvFrameInfo) int {
if fi.autosize == 0 {
return 0
}
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 2 + riscvSPAdjustLen(int32(fi.autosize))
}
func riscvSPAdjustLen(imm int32) int {
if imm != 0 && imm%16 == 0 && imm >= -512 && imm <= 511 {
return 2
adj := int32(fi.autosize)
if fits12(adj) {
// C.LDSP (2 bytes) + ADDI/C.ADDI (2 or 4 bytes).
return 2 + len(riscvSPAdjust(adj))
}
if riscvFitsCAddi(imm) {
return 2
}
return 4
return 2 + len(riscvAddToSP(adj))
}
// riscvResolvePseudo translates a pseudo-register memory reference into a
@@ -180,3 +292,73 @@ func riscvResolvePseudo(sym *ast.Symbol, fi riscvFrameInfo) (base int, off int32
}
return -1, 0
}
// riscvGuardLen returns the byte length of the stack-split guard prefix
// including the inline morestack call (zero when the function needs no
// guard). Unlike amd64 and arm64, the toolchain places the morestack call
// between the guard and the body: the guard branches forward over it.
func riscvGuardLen(fi riscvFrameInfo) (int, error) {
g, _, err := riscvGuard(fi)
if err != nil {
return 0, err
}
return len(g), nil
}
// riscvGuard emits the stack-split guard prefix with the inline morestack
// call: the branch skips forward over JAL X5 and JAL X0 straight into the
// body; the JAL X5 carries the R_RISCV_JAL relocation. All offsets are
// relative to the guard itself, which sits at function offset 0.
func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc, error) {
if !fi.needSplit {
return nil, Reloc{}, nil
}
// MOV 16(g), X6 (g.stackguard0), g = X27.
out := wordLE(riscvIType(riscvEnc{0x03, 0x3, 0x00}, 6, 27, 16))
jalBack := func() []byte {
// JAL X0 back to the function start: it sits right after the JAL X5,
// so its displacement is minus the current offset.
return wordLE(riscvJType(0, int32(-len(out))))
}
var reloc Reloc
switch fi.splitClass {
case 0:
// BLTU X6, SP, done (+12: over the CALL and the JMP back)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 2, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
case 1:
// ADDI $-(framesize-StackSmall), SP, X7; BLTU X6, X7, done (+12)
off := int32(fi.autosize - stackSmall)
out = append(out, wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off))...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
default:
// MOV $(framesize-StackSmall), X7; BLTU SP, X7, call;
// ADD $-(framesize-StackSmall), SP, X7; BLTU X6, X7, call
off := int32(fi.autosize - stackSmall)
mov := encodeRISCVLoadImm(7, off)
out = append(out, mov...)
addiLen := riscvItypeImmediateSize("ADDI", -off)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
if err != nil {
// The ADDI expansion failed: the SP adjustment this class
// depends on is not emittable, and silently dropping it would
// corrupt every stack reference in the body.
return nil, Reloc{}, fmt.Errorf("stack-split guard: %w", err)
}
out = append(out, addi...)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 6, 7, 12))...)
call := len(out)
reloc = Reloc{Off: call, After: call + 4, Name: "runtime\u00b7morestack_noctxt", Kind: RelRISCVJal}
out = append(out, wordLE(riscvJType(5, 0))...)
out = append(out, jalBack()...)
}
return out, reloc, nil
}
+41
View File
@@ -65,6 +65,47 @@ TEXT ·framed(SB), NOSPLIT, $16-16
}
}
// TestRISCVFrameSpadjLargeFrame checks the stack-adjustment boundaries of a
// frame past the imm12 range: the prologue materialises the LR-store address
// and the SP adjustment through X31 (C.LUI + C.ADD + SD, then C.LUI + ADDIW +
// C.ADD), so the SP boundary lands at PC 16, and the RET closes with
// C.LDSP plus the same X31 adjustment, 10 bytes. Regression: both helpers
// assumed the small-frame prologue and reported 8 and 6.
func TestRISCVFrameSpadjLargeFrame(t *testing.T) {
f, errs := parser.Parse("bigframe_riscv64.s", `#include "textflag.h"
TEXT ·big(SB), NOSPLIT, $9000-8
MOV a+0(FP), X10
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
fn := img.Funcs[0]
// autosize = 9008. Prologue: C.LUI X31 + C.ADD X31,SP (4) + SD (4) +
// C.LUI X31 + ADDIW X31 + C.ADD SP,X31 (8) = 16 bytes to the SP boundary;
// C.SDSP X1 (2) follows, so the body starts at 18.
wantSpadj := []SpadjStep{{PC: 16, Value: 9008}, {PC: 36, Value: 0}}
if len(fn.Spadj) != len(wantSpadj) {
t.Fatalf("spadj = %v, want %v", fn.Spadj, wantSpadj)
}
for i := range wantSpadj {
if fn.Spadj[i] != wantSpadj[i] {
t.Errorf("spadj[%d] = %v, want %v", i, fn.Spadj[i], wantSpadj[i])
}
}
// The FP load materialises its 9016-byte offset through X31 as well
// (8 bytes), then RET's epilogue (C.LDSP + X31 adjust = 10) plus JALR.
if fn.Size != 18+8+14 {
t.Errorf("size = %d, want %d", fn.Size, 18+8+14)
}
}
// TestRISCVRegAliases checks the Go ABI register aliases that the toolchain
// defines: LR is the link register (X1) and TMP is the assembler scratch
// register (X31/T6).
+83
View File
@@ -74,6 +74,89 @@ DATA callee<>+0(SB)/8, $42
t.Error("ELF object missing R_RISCV_JAL relocation")
}
}
// TestELFRISCVPCRELLO12Anchor checks the psABI's LO12 pairing rule: the
// R_RISCV_PCREL_LO12_I/S relocation must reference a symbol whose value is
// the AUIPC site of its HI20 partner (psABI §8.4.9; cmd/link generates one
// local text symbol per AUIPC for exactly this). The emitter pairs each
// HI20 (against the target symbol) with a LO12 against the .text section
// symbol whose addend is the AUIPC's section-relative offset, so S + A is
// the AUIPC address.
func TestELFRISCVPCRELLO12Anchor(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
TEXT ·sb(SB), NOSPLIT, $0-0
MOV $answer<>(SB), X10
MOV answer<>(SB), X11
MOV X12, answer<>(SB)
RET
GLOBL answer<>(SB), RODATA, $8
DATA answer<>+0(SB)/8, $42
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
obj, err := img.ELFRISCVObject()
if err != nil {
t.Fatalf("ELFRISCVObject: %v", err)
}
ef, err := elf.NewFile(bytes.NewReader(obj))
if err != nil {
t.Fatalf("parse ELF: %v", err)
}
defer ef.Close()
if flags := binary.LittleEndian.Uint32(obj[48:]); flags != efRISCVFloatAbiDouble {
t.Errorf("e_flags = %#x, want %#x (EF_RISCV_FLOAT_ABI_DOUBLE)", flags, efRISCVFloatAbiDouble)
}
rela := ef.Section(".rela.text")
if rela == nil {
t.Fatal("missing .rela.text")
}
b, err := rela.Data()
if err != nil {
t.Fatal(err)
}
if len(b) != 6*24 {
t.Fatalf(".rela.text holds %d entries, want six (three HI20/LO12 pairs)", len(b)/24)
}
le := binary.LittleEndian
wantLo := []uint32{rRISCVPCRELLO12I, rRISCVPCRELLO12I, rRISCVPCRELLO12S}
for p := range 3 {
auipc := 8 * p
hi := b[p*2*24:]
lo := b[(p*2+1)*24:]
if off := le.Uint64(hi[0:]); off != uint64(auipc) {
t.Errorf("pair %d: HI20 r_offset = %d, want %d (the AUIPC)", p, off, auipc)
}
if typ := uint32(le.Uint64(hi[8:])); typ != rRISCVPCRELHI20 {
t.Errorf("pair %d: HI20 type = %d, want %d", p, typ, rRISCVPCRELHI20)
}
if sym := int(le.Uint64(hi[8:]) >> 32); sym == 0 || sym == 1 {
t.Errorf("pair %d: HI20 against symbol %d, want the target", p, sym)
}
if off := le.Uint64(lo[0:]); off != uint64(auipc+4) {
t.Errorf("pair %d: LO12 r_offset = %d, want %d", p, off, auipc+4)
}
if typ := uint32(le.Uint64(lo[8:])); typ != wantLo[p] {
t.Errorf("pair %d: LO12 type = %d, want %d", p, typ, wantLo[p])
}
// The LO12 must denote the AUIPC site: the .text section symbol
// (index 1) plus the AUIPC's section-relative offset as addend.
if sym := int(le.Uint64(lo[8:]) >> 32); sym != 1 {
t.Errorf("pair %d: LO12 against symbol %d, want 1 (the .text section symbol)", p, sym)
}
if add := int64(le.Uint64(lo[16:])); add != int64(auipc) {
t.Errorf("pair %d: LO12 addend = %d, want %d (S + A = the AUIPC address)", p, add, auipc)
}
}
}
func TestGOObjectRISCVStructure(t *testing.T) {
f, errs := parser.Parse("k_riscv64.s", `
#include "textflag.h"
+405 -48
View File
@@ -36,12 +36,12 @@ const (
vexNDS3Imm
// vexExtract is the lane-extract form `OP $imm, ysrc, xdst`: ModRM.reg =
// ysrc (op1), ModRM.rm = xdst or memory (op2), imm8 = op0. The YMM
// source lives in the reg field, the destination in r/m — the PEXTR-style
// source lives in the reg field, the destination in r/m, the PEXTR-style
// layout. VEXTRACTI128 and VEXTRACTF128 use this shape.
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m — the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
// in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
@@ -52,6 +52,30 @@ const (
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
// vexZeroAll is the no-operand form that zeroes the full upper state
// (VZEROALL, the L = 1 twin of VZEROUPPER).
vexZeroAll
// vexNDS3GPR is the three-operand NDS form over general-purpose
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
vexNDS3GPR
// vexImmRMGPR is the immediate form over general-purpose registers
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
vexImmRMGPR
// vexRMOpGPR is the two-operand /digit form over general-purpose
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
// (op0), VEX.vvvv = dst (op1), L = 0.
vexRMOpGPR
// vexCountGPR is the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
// family, with reg = dst (op2), L = 0.
vexCountGPR
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
// carries a vector length, so the register the L'L field follows is the
// XMM source.
vexExtractGPR
)
// vexSpec describes one VEX instruction's encoding parameters.
@@ -68,7 +92,7 @@ type vexSpec struct {
// incrementally; every entry is covered by a byte-for-byte ground-truth test
// against the Go assembler.
var vexTable = map[string]vexSpec{
// VEX.128/256.66.0F.WIG — integer arithmetic / logic / compare.
// VEX.128/256.66.0F.WIG, integer arithmetic / logic / compare.
"VPADDD": {1, 0xFE, 0, 1, -1, vexNDS3},
"VPADDQ": {1, 0xD4, 0, 1, -1, vexNDS3},
"VPSUBD": {1, 0xFA, 0, 1, -1, vexNDS3},
@@ -82,7 +106,7 @@ var vexTable = map[string]vexSpec{
"VPUNPCKHDQ": {1, 0x6A, 0, 1, -1, vexNDS3},
"VPUNPCKLQDQ": {1, 0x6C, 0, 1, -1, vexNDS3},
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3},
// VEX.256.66.0F38.W0 — dword permute (three-operand NDS form).
// VEX.256.66.0F38.W0, dword permute (three-operand NDS form).
"VPERMD": {2, 0x36, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG.
"VPMULLD": {2, 0x40, 0, 1, -1, vexNDS3},
@@ -90,14 +114,14 @@ var vexTable = map[string]vexSpec{
"VPSHUFB": {2, 0x00, 0, 1, -1, vexNDS3},
"VPCMPGTQ": {2, 0x37, 0, 1, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — packed double-precision arithmetic / logic.
// VEX.128/256.66.0F.WIG, packed double-precision arithmetic / logic.
"VADDPD": {1, 0x58, 0, 1, -1, vexNDS3},
"VMULPD": {1, 0x59, 0, 1, -1, vexNDS3},
"VSUBPD": {1, 0x5C, 0, 1, -1, vexNDS3},
"VDIVPD": {1, 0x5E, 0, 1, -1, vexNDS3},
"VMINPD": {1, 0x5D, 0, 1, -1, vexNDS3},
"VMAXPD": {1, 0x5F, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG — packed single-precision arithmetic.
// VEX.128/256.0F.WIG, packed single-precision arithmetic.
"VADDPS": {1, 0x58, 0, 0, -1, vexNDS3},
"VMULPS": {1, 0x59, 0, 0, -1, vexNDS3},
"VSUBPS": {1, 0x5C, 0, 0, -1, vexNDS3},
@@ -107,7 +131,7 @@ var vexTable = map[string]vexSpec{
"VXORPD": {1, 0x57, 0, 1, -1, vexNDS3},
"VUNPCKHPD": {1, 0x15, 0, 1, -1, vexNDS3},
"VUNPCKLPD": {1, 0x14, 0, 1, -1, vexNDS3},
// VEX.128.F2.0F.WIG — scalar double-precision arithmetic (the packed
// VEX.128.F2.0F.WIG, scalar double-precision arithmetic (the packed
// opcodes with an F2 pp).
"VADDSD": {1, 0x58, 0, 3, -1, vexNDS3},
"VSUBSD": {1, 0x5C, 0, 3, -1, vexNDS3},
@@ -115,7 +139,7 @@ var vexTable = map[string]vexSpec{
"VDIVSD": {1, 0x5E, 0, 3, -1, vexNDS3},
"VMINSD": {1, 0x5D, 0, 3, -1, vexNDS3},
"VMAXSD": {1, 0x5F, 0, 3, -1, vexNDS3},
// VEX.128.F3.0F.WIG — scalar single-precision arithmetic (the packed
// VEX.128.F3.0F.WIG, scalar single-precision arithmetic (the packed
// opcodes with an F3 pp).
"VADDSS": {1, 0x58, 0, 2, -1, vexNDS3},
"VSUBSS": {1, 0x5C, 0, 2, -1, vexNDS3},
@@ -123,10 +147,16 @@ var vexTable = map[string]vexSpec{
"VDIVSS": {1, 0x5E, 0, 2, -1, vexNDS3},
"VMINSS": {1, 0x5D, 0, 2, -1, vexNDS3},
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1 — fused multiply-add (NDS form).
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// Scalar fused multiply-add (NDS form). The Go assembler carries the
// same 66 prefix as the packed forms on every FMA row, and W1 on the
// double-precision spellings, so SD shares PD's prefix/W pair and the
// scalar width rides on the W bit.
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG — sign/zero extend and broadcast (reg=dst, rm=src,
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
"VPMOVSXWD": {2, 0x23, 0, 1, -1, vexRM},
"VPMOVSXDQ": {2, 0x25, 0, 1, -1, vexRM},
@@ -143,70 +173,129 @@ var vexTable = map[string]vexSpec{
"VPBROADCASTQ": {2, 0x59, 0, 1, -1, vexRM},
"VPBROADCASTB": {2, 0x78, 0, 1, -1, vexRM},
"VPBROADCASTW": {2, 0x79, 0, 1, -1, vexRM},
// VEX.128/256.F3.0F.WIG — signed dword to packed double conversion
// VEX.128/256.F3.0F.WIG, signed dword to packed double conversion
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VCVTDQ2PD": {1, 0xE6, 0, 2, -1, vexRM},
// VEX.128/256.0F.WIG — signed dword to packed single conversion
// VEX.128/256.0F.WIG, signed dword to packed single conversion
// (reg=dst, rm=src, no vvvv, no mandatory prefix).
"VCVTDQ2PS": {1, 0x5B, 0, 0, -1, vexRM},
// VEX.128/256.0F.WIG — packed single to packed double conversion
// VEX.128/256.0F.WIG, packed single to packed double conversion
// (reg=dst, rm=src; the destination is the wide operand and sets the
// length). Intel's maps prescribe the F3 prefix here (VEX.pp = 10), but
// the Go assembler emits the instruction with pp = 00, and gasm follows
// the Go assembler's bytes — its machine code is the oracle, not the
// the Go assembler's bytes, its machine code is the oracle, not the
// manual.
"VCVTPS2PD": {1, 0x5A, 0, 0, -1, vexRM},
// VEX.128.F2.0F.WIG — duplicate the low double of each 128-bit lane
// VEX.128.F2.0F.WIG, duplicate the low double of each 128-bit lane
// (reg=dst, rm=src, no vvvv; the length follows the destination).
"VMOVDDUP": {1, 0x12, 0, 3, -1, vexRM},
// VEX.128/256.66.0F.WIG — move mask to a GPR (reg=gpr dst, rm=vec src).
// VEX.128/256.66.0F.WIG, move mask to a GPR (reg=gpr dst, rm=vec src).
"VPMOVMSKB": {1, 0xD7, 0, 1, -1, vexRM},
"VMOVMSKPS": {1, 0x50, 0, 0, -1, vexRM}, // no 66 prefix (that would be VMOVMSKPD)
// VEX.128/256.66.0F.WIG — immediate shifts (opdigit selects the shift).
// VEX.128/256.66.0F.WIG, immediate shifts (opdigit selects the shift).
"VPSLLD": {1, 0x72, 0, 1, 6, vexShiftImm},
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm},
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm},
"VPSRLQ": {1, 0x73, 0, 1, 2, vexShiftImm},
"VPSLLQ": {1, 0x73, 0, 1, 6, vexShiftImm},
// VEX.128/256.66.0F.WIG — immediate shuffle (reg=dst, rm=src, imm8).
// VEX.128/256.66.0F.WIG, immediate shuffle (reg=dst, rm=src, imm8).
"VPSHUFD": {1, 0x70, 0, 1, -1, vexImmRM},
// VEX.256.66.0F3A.W1 — qword permute (reg=dst, rm=src, imm8).
// VEX.256.66.0F3A.W1, qword permute (reg=dst, rm=src, imm8).
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM},
// VEX.128/256.66.0F.WIG — two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// VEX.128/256.66.0F.WIG, two-source shuffle (reg=dst, vvvv=src1, rm=src2,
// imm8).
"VSHUFPD": {1, 0xC6, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — permute / insert (same shape; VINSERTI128's rm is
// VEX.256.66.0F3A.W0, permute / insert (same shape; VINSERTI128's rm is
// the XMM or memory source).
"VPERM2I128": {3, 0x46, 0, 1, -1, vexNDS3Imm},
"VINSERTI128": {3, 0x38, 0, 1, -1, vexNDS3Imm},
// VEX.256.66.0F3A.W0 — lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
// VEX.256.66.0F3A.W0, lane extract (reg=YMM src, rm=XMM/memory dst, imm8).
"VEXTRACTI128": {3, 0x39, 0, 1, -1, vexExtract},
"VEXTRACTF128": {3, 0x19, 0, 1, -1, vexExtract},
// VEX.128/256.66.0F3A.W0 — half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8 — the extract layout).
// VEX.128/256.66.0F3A.W0, half-precision convert back ($imm, src, dst:
// reg=src, rm=XMM/memory dst, imm8, the extract layout).
"VCVTPS2PH": {3, 0x1D, 0, 1, -1, vexExtract},
// VEX.128.0F.W0 — no operands.
// VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
// falls through to the EVEX table.
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
// share the F7/F5 opcodes across W, the variable shifts carry their
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
// in F2/F3.
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
// rm = src, vvvv = dst).
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
// VEX.128.0F.W0 — mask-register test (KTESTW k1, k2: reg = dst, rm = src).
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
// VEX.66.0F38.W0 — broadcast a single/double to all lanes (reg=dst,
// VEX.66.0F38.W0, broadcast a single/double to all lanes (reg=dst,
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.66.0F38.W0 — half-precision convert (reg=dst, rm=half-width
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
// YMM (the encoder rejects an XMM destination, as go tool asm does).
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
// memory destination in rm).
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
// VEX.F3.0F.WIG — replicate even/odd singles (reg=dst, rm=src).
// VEX.F3.0F.WIG, replicate even/odd singles (reg=dst, rm=src).
"VMOVSLDUP": {1, 0x12, 0, 2, -1, vexRM},
"VMOVSHDUP": {1, 0x16, 0, 2, -1, vexRM},
// VEX.66.0F.WIG — packed double to packed single conversion, the X/Y
// VEX.66.0F.WIG, packed double to packed single conversion, the X/Y
// spellings: the destination is always XMM and the spelling fixes the
// source length (X = 128, Y = 256).
"VCVTPD2PSX": {1, 0x5A, 0, 1, -1, vexRMSrcLen},
@@ -230,18 +319,140 @@ var vexTable = map[string]vexSpec{
"VCVTSI2SSL": {1, 0x2A, 0, 2, -1, vexNDS3},
"VCVTSI2SSQ": {1, 0x2A, 1, 2, -1, vexNDS3},
// VEX.128/256.66.0F.WIG — word shifts (opdigit selects the shift).
// VEX.128/256.66.0F.WIG, word shifts (opdigit selects the shift).
"VPSRLW": {1, 0x71, 0, 1, 2, vexShiftImm},
"VPSRAW": {1, 0x71, 0, 1, 4, vexShiftImm},
"VPSLLW": {1, 0x71, 0, 1, 6, vexShiftImm},
// VEX.F2.0F — packed double to packed dword conversions, truncating and
// VEX.F2.0F, packed double to packed dword conversions, truncating and
// non-truncating. The destination is always XMM; the X/Y spellings fix
// the source length (XMM/YMM), and VEX.L follows it — see vexSrcLen.
// the source length (XMM/YMM), and VEX.L follows it, see vexSrcLen.
"VCVTPD2DQX": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
// spellings, read off the toolchain opcode tables ---
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
// spellings (no 66 prefix; WIG, so W = 0).
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
// VEX.128.66.0F, the scalar and packed compare forms.
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
}
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
@@ -257,7 +468,7 @@ var vexSrcLen = map[string]int{
"VCVTPD2PSY": 1,
}
// vexVarShift maps the shift mnemonics to their variable-count opcode — the
// vexVarShift maps the shift mnemonics to their variable-count opcode, the
// form whose count comes from an XMM register or memory (VPSRLQ X0, Y8, Y8),
// an ordinary NDS encoding rather than the /digit immediate form above.
var vexVarShift = map[string]byte{
@@ -288,20 +499,22 @@ type vexMoveSpec struct {
// vexMoveTable maps an upper-case move mnemonic to its encoding.
var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG — unaligned integer move.
// VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG — unaligned packed double move.
// VEX.128/256.66.0F.WIG, aligned integer move.
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0 — 32-bit GPR/memory ↔ XMM.
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
"VMOVD": {1, 1, 0x6E, 0x7E, 0, 0, 0, 0, false, true, true},
// VMOVQ — 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
// VMOVQ, 66 6E W1 (r/m→xmm), 66 7E W1 (xmm→r/m), 66 D6 W0 (xmm→xmm).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, 1, 0xD6, 0, true, true, true},
// VEX.128.F2.0F.WIG — scalar double move, memory operands only (the
// VEX.128.F2.0F.WIG, scalar double move, memory operands only (the
// register form takes three operands and is not supported yet).
"VMOVSD": {1, 3, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128.F3.0F.WIG — scalar single move, memory operands only.
// VEX.128.F3.0F.WIG, scalar single move, memory operands only.
"VMOVSS": {1, 2, 0x10, 0x11, 0, 0, 0, 0, false, false, true},
// VEX.128/256 — aligned packed moves.
// VEX.128/256, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 0, 0, 0, 0, true, false, false},
}
@@ -317,13 +530,21 @@ func isVex(mnemUpper string) bool {
// encodeVex encodes a VEX instruction with operands in Plan 9 order.
func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
// Vector register indices 16–31 exist only in EVEX encodings; fail
// Vector register indices 16-31 exist only in EVEX encodings; fail
// loudly rather than silently truncating the index.
for _, op := range ops {
if r, ok := op.(Reg); ok && r.isVec() && r.idx >= 16 {
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
}
}
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
// only; an XMM destination is rejected exactly as go tool asm does.
if mnemUpper == "VBROADCASTI128" {
dstReg, ok := ops[len(ops)-1].(Reg)
if len(ops) != 2 || !ok || dstReg.size != 32 {
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
}
}
if ms, ok := vexMoveTable[mnemUpper]; ok {
return e.encodeVexMove(mnemUpper, ms, ops)
}
@@ -356,6 +577,18 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
case vexZeroAll:
return e.encodeVexZeroAll(mnemUpper, spec, ops)
case vexNDS3GPR:
return e.encodeVexNDS3GPR(spec, ops)
case vexImmRMGPR:
return e.encodeVexImmRMGPR(spec, ops)
case vexRMOpGPR:
return e.encodeVexRMOpGPR(spec, ops)
case vexCountGPR:
return e.encodeVexCountGPR(spec, ops)
case vexRMRev:
return e.encodeVexRMRev(spec, ops)
}
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
}
@@ -420,7 +653,7 @@ func (e *enc) encodeVexRM(spec vexSpec, ops []Operand) error {
}
// encodeVexRMSrcLen encodes a length-narrowing conversion: OP src, dst with
// the destination always XMM and the VEX.L bit following the source — fixed
// the destination always XMM and the VEX.L bit following the source, fixed
// by the mnemonic's spelling (VCVTPD2DQX = 128, VCVTPD2DQY = 256) even when
// the source is memory.
func (e *enc) encodeVexRMSrcLen(mnem string, spec vexSpec, ops []Operand) error {
@@ -457,9 +690,10 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
if !ok {
return fmt.Errorf("shift count must be an immediate")
}
srcReg, ok := src.(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("shift source must be a vector register")
// The count source is a vector register or memory; the VEX length
// follows the destination register either way.
if !vecOrMem(src) {
return fmt.Errorf("shift source must be a vector register or memory")
}
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
@@ -467,7 +701,7 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
}
vvvvBar := 15 - (dstReg.idx & 15)
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
return err
}
immByte, err := imm8(int64(immVal))
@@ -607,6 +841,129 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
return nil
}
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
// twin of VZEROUPPER.
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 0 {
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
}
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
return nil
}
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
// rm = src2 and L = 0.
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
vvvvReg, ok := src1.(Reg)
if !ok || vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
}
// encodeVexImmRMGPR encodes the immediate form over general-purpose
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
// ModRM.rm = src and VEX.vvvv = dst.
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
}
// encodeVexCountGPR encodes the three-operand count form over general-purpose
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
}
src, count, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
countReg, ok := count.(Reg)
if !ok || countReg.isVec() {
return fmt.Errorf("VEX count operand must be a general-purpose register")
}
srcReg, ok := src.(Reg)
if !ok || srcReg.isVec() {
return fmt.Errorf("VEX count source must be a general-purpose register")
}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
}
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
// a store with no register-destination form).
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("store source must be a vector register")
}
if !memOperand(ops[1]) {
return fmt.Errorf("store destination must be memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
+121 -5
View File
@@ -19,6 +19,65 @@ func vreg(t *testing.T, name string) Reg {
return r
}
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
// go tool asm in the ground-truth tests instead.
var x86asmUnrecognised = map[string]bool{
"ANDNL": true,
"ANDNQ": true,
"MULXL": true,
"MULXQ": true,
"RORXL": true,
"RORXQ": true,
"VFMADD213SD": true,
"VFNMADD231SD": true,
// The scalar FMA spellings the decoder's tables lack entirely.
"VFMADD132SD": true,
"VFMADD132SS": true,
"VFMADD213SS": true,
"VFMADD231SD": true,
"VFMADD231SS": true,
"VFMSUB132SD": true,
"VFMSUB132SS": true,
"VFMSUB213SD": true,
"VFMSUB213SS": true,
"VFMSUB231SD": true,
"VFMSUB231SS": true,
"VFNMADD132SD": true,
"VFNMADD132SS": true,
"VFNMADD213SD": true,
"VFNMADD213SS": true,
"VFNMADD231SS": true,
"VFNMSUB132SD": true,
"VFNMSUB132SS": true,
"VFNMSUB213SD": true,
"VFNMSUB213SS": true,
"VFNMSUB231SD": true,
"VFNMSUB231SS": true,
// The BMI1 unary bit ops the decoder's AVX tables lack.
"BLSIL": true,
"BLSIQ": true,
"BLSMSKL": true,
"BLSMSKQ": true,
"BLSRL": true,
"BLSRQ": true,
// The BMI2 bit ops whose W1/LZ rows the decoder misses.
"BEXTRL": true,
"BEXTRQ": true,
"BZHIL": true,
"BZHIQ": true,
"PDEPL": true,
"PDEPQ": true,
"PEXTL": true,
"PEXTQ": true,
"SARXL": true,
"SARXQ": true,
"SHLXL": true,
"SHLXQ": true,
"SHRXL": true,
"SHRXQ": true,
}
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
// instruction and verifies it round-trips through the x86 decoder to the same
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
@@ -37,8 +96,15 @@ func TestVexNDS3(t *testing.T) {
t.Errorf("%s: Encode: %v", mnem, err)
continue
}
// The x86 decoder's table lacks a handful of rows the Go assembler
// emits (the scalar 213/231 FMA spellings among them); those are
// pinned byte for byte against go tool asm in TestVexGroundTruth
// instead of round-tripped here.
inst, err := x86asm.Decode(code, 64)
if err != nil {
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
continue
}
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
continue
}
@@ -173,7 +239,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VPMULLD Y1,Y2,Y3", "VPMULLD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d40d9", ""},
{"VPUNPCKLDQ Y4,Y3,Y5", "VPUNPCKLDQ", []Operand{vreg(t, "Y4"), vreg(t, "Y3"), vreg(t, "Y5")}, "c5e562ec", ""},
{"VPERMD Y1,Y2,Y3", "VPERMD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c4e26d36d9", ""},
// Floating point (packed and scalar) and FMA — same NDS form, the pp
// Floating point (packed and scalar) and FMA; same NDS form, the pp
// bits and map select the operation.
{"VADDPD Y9,Y8,Y8", "VADDPD", []Operand{vreg(t, "Y9"), vreg(t, "Y8"), vreg(t, "Y8")}, "c4413d58c1", ""},
{"VADDPD X1,X2,X3", "VADDPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e958d9", ""},
@@ -184,6 +250,50 @@ func TestVexGroundTruth(t *testing.T) {
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
// Packed single XOR and byte compare (NDS form).
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
// Octa byte shifts (vvvv carries the destination).
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
// Two-operand test and the non-temporal and broadcast stores.
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
// Aligned integer move and the full zeroing form.
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
// BMI1/BMI2 general-register VEX forms.
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
// BMI2 variable shifts and bit ops (three general registers).
{"SHLXL AX,CX,R15", "SHLXL", []Operand{AX, CX, vreg(t, "R15")}, "c46279f7f9", ""},
{"SHRXQ R8,DX,AX", "SHRXQ", []Operand{vreg(t, "R8"), DX, AX}, "c4e2bbf7c2", ""},
{"SARXQ AX,DX,R9", "SARXQ", []Operand{AX, DX, vreg(t, "R9")}, "c462faf7ca", ""},
{"BEXTRL AX,CX,R15", "BEXTRL", []Operand{AX, CX, vreg(t, "R15")}, "c46278f7f9", ""},
{"BZHIQ AX,CX,R15", "BZHIQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f8f5f9", ""},
{"PDEPQ AX,CX,R15", "PDEPQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f3f5f8", ""},
{"PEXTQ AX,CX,R15", "PEXTQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f2f5f8", ""},
// BMI1 unary bit ops (src, dst: /digit in ModRM.reg, dst in vvvv).
{"BLSIL AX,CX", "BLSIL", []Operand{AX, CX}, "c4e270f3d8", ""},
{"BLSRQ AX,CX", "BLSRQ", []Operand{AX, CX}, "c4e2f0f3c8", ""},
{"BLSMSKQ AX,CX", "BLSMSKQ", []Operand{AX, CX}, "c4e2f0f3d0", ""},
// Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
@@ -217,7 +327,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VEXTRACTI128 $1,Y8,X9", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d39c101", ""},
{"VEXTRACTI128 $1,Y8,(DI)", "VEXTRACTI128", []Operand{Imm(1), vreg(t, "Y8"), Ptr(DI, 0, 16)}, "c4637d390701", ""},
{"VEXTRACTF128 $1,Y8,X9", "VEXTRACTF128", []Operand{Imm(1), vreg(t, "Y8"), vreg(t, "X9")}, "c4437d19c101", ""},
// Moves — each direction picks its own opcode and VEX.W.
// Moves; each direction picks its own opcode and VEX.W.
{"VMOVDQU (SI),Y1", "VMOVDQU", []Operand{Ptr(SI, 0, 32), vreg(t, "Y1")}, "c5fe6f0e", ""},
{"VMOVDQU Y3,(DI)", "VMOVDQU", []Operand{vreg(t, "Y3"), Ptr(DI, 0, 32)}, "c5fe7f1f", ""},
{"VMOVDQU X1,X2", "VMOVDQU", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fa7fca", ""},
@@ -234,7 +344,7 @@ func TestVexGroundTruth(t *testing.T) {
{"VMOVD AX,X0", "VMOVD", []Operand{AX, vreg(t, "X0")}, "c5f96ec0", ""},
{"VMOVSD (SI),X8", "VMOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X8")}, "c57b1006", ""},
{"VMOVSD X8,(SI)", "VMOVSD", []Operand{vreg(t, "X8"), Ptr(SI, 0, 8)}, "c57b1106", ""},
// Packed double arithmetic and unpack — the NDS form, the opcode
// Packed double arithmetic and unpack; the NDS form, the opcode
// selects the operation.
{"VSUBPD Y1,Y2,Y3", "VSUBPD", []Operand{vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y3")}, "c5ed5cd9", ""},
{"VDIVPD X1,X2,X3", "VDIVPD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5e95ed9", ""},
@@ -255,12 +365,12 @@ func TestVexGroundTruth(t *testing.T) {
{"VMINSS X6,X7,X8", "VMINSS", []Operand{vreg(t, "X6"), vreg(t, "X7"), vreg(t, "X8")}, "c5425dc6", ""},
{"VMAXSS X1,X2,X3", "VMAXSS", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")}, "c5ea5fd9", ""},
{"VADDSD 8(AX),X1,X2", "VADDSD", []Operand{Ptr(AX, 8, 8), vreg(t, "X1"), vreg(t, "X2")}, "c5f3585008", ""},
// VMOVDDUP — duplicate the low double (reg=dst, rm=src, F2 pp).
// VMOVDDUP; duplicate the low double (reg=dst, rm=src, F2 pp).
{"VMOVDDUP X1,X2", "VMOVDDUP", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5fb12d1", ""},
{"VMOVDDUP Y1,Y2", "VMOVDDUP", []Operand{vreg(t, "Y1"), vreg(t, "Y2")}, "c5ff12d1", ""},
{"VMOVDDUP 8(AX),X1", "VMOVDDUP", []Operand{Ptr(AX, 8, 8), vreg(t, "X1")}, "c5fb124808", ""},
// Conversions: DQ→PS (no prefix), PS→PD (Go emits it without the F3
// prefix — see the table comment), DQ→PD.
// prefix; see the table comment), DQ→PD.
{"VCVTDQ2PS X1,X2", "VCVTDQ2PS", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85bd1", ""},
{"VCVTDQ2PS Y3,Y4", "VCVTDQ2PS", []Operand{vreg(t, "Y3"), vreg(t, "Y4")}, "c5fc5be3", ""},
{"VCVTPS2PD X1,X2", "VCVTPS2PD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "c5f85ad1", ""},
@@ -287,6 +397,12 @@ func TestVexGroundTruth(t *testing.T) {
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
// The decoder's AVX/BMI table lacks a few rows the Go
// assembler emits (the GPR VEX forms and the scalar FMA
// spellings); their bytes are the ground truth here.
if x86asmUnrecognised[c.mnem] {
continue
}
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
continue
}
+19 -8
View File
@@ -17,7 +17,7 @@ type File struct {
Orphans []Stmt // labels/instructions seen before any TEXT directive
// Macros holds the names introduced by #define directives in this file.
// The linter uses it to avoid flagging macro invocations as unknown
// instructions (macro expansion itself is out of scope — see the docs).
// instructions (macro expansion itself is out of scope, see the docs).
Macros map[string]bool
}
@@ -114,6 +114,7 @@ type Symbol struct {
Pkg string // package prefix before the middle dot ("" = current package)
Name string // identifier without the middle dot or <>
Static bool // the <> marker is present
ABI string // the <NAME> ABI marker, e.g. ABIInternal ("" when absent)
Pseudo string // FP, SP, SB or PC ("" for a bare name)
Offset int64
HasOff bool
@@ -150,11 +151,21 @@ type Immediate struct {
// Address is a non-immediate operand: a register, a memory reference, a symbol
// reference or a label. Fields are populated best-effort from the syntax.
type Address struct {
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
Base string // base register, from (base)
Index string // index register, from (index*scale)
Scale int // index scale; 0 when absent
Offset int64 // leading displacement, from off(base)
HasOff bool // a leading displacement is present
Shift string // verbatim arm64 shift suffix, e.g. "<<2"
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
Base string // base register, from (base)
Index string // index register, from (index*scale)
Scale int // index scale; 0 when absent
Offset int64 // leading displacement, from off(base)
HasOff bool // a leading displacement is present
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
Range *RegRange // bracketed register range; nil for every other form
}
// RegRange is a bracketed register range, [Z0-Z3]: the amd64 spelling of
// the four-register source of the 4FMAPS/4VNNIW families. Lo and Hi carry
// the verbatim register spellings; the range is inclusive at both ends.
type RegRange struct {
Lo string
Hi string
Pos token.Position
}
+367
View File
@@ -0,0 +1,367 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"errors"
"fmt"
"go/ast"
"go/build"
"go/constant"
"go/parser"
"go/token"
"go/types"
"os"
"path/filepath"
"regexp"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
)
// go_asm.h is the header the Go compiler writes for every package that
// carries assembly (the compiler's -asmhdr output): "#define const_NAME
// value" for each package constant, and for each named struct type
// "#define TYPE__size size" plus one "#define TYPE_field offset" per field.
// GOROOT assembly includes it, and a standalone assembler has no compiler
// to have produced it, so gasm generates the equivalent itself: the package
// the .s file lives in is parsed and type-checked here, with the target
// architecture's own sizes, and the same defines are written out. The
// type-checking GOOS is selected by the caller: a GOOS-specific file
// (sys_darwin_arm64.s) needs its platform's defines, which a header from
// the ambient GOOS silently omits.
//
// The emitter mirrors cmd/compile's dumpasmhdr exactly: constants come out
// as "const_NAME", struct entries as "NAME__size" followed by the fields in
// declaration order, blank names are skipped, and float and complex
// constants are omitted (the assembler carries integers, bools and strings
// only). Aliases to structs are emitted, generic types are not: they have
// no fixed size. A define the assembly references but this header does not
// carry surfaces later as the assembler's own "undefined" diagnostic naming
// the define, which is the honest failure.
// goAsmInclude matches the #include "go_asm.h" directive, tolerant of
// whitespace, so the wiring knows which files need a generated header
// before the preprocessor runs and would report the header as missing.
var goAsmInclude = regexp.MustCompile(`(?m)^\s*#\s*include\s+"go_asm\.h"`)
// needsGoAsmHeader reports whether src includes go_asm.h.
func needsGoAsmHeader(src string) bool {
return goAsmInclude.MatchString(src)
}
// goAsmHeaderResolved reports whether the include of go_asm.h from a file in
// asmDir already resolves: to a header in the package directory itself, or
// in one of the -I directories, the way the preprocessor searches. Only an
// unresolved include is generated for; a header someone placed by hand is
// the tool the author chose, and it also wins the preprocessor's own search
// order, so generating a second copy would be dead weight at best.
func goAsmHeaderResolved(asmDir string, dirs []string) bool {
candidates := []string{filepath.Join(asmDir, "go_asm.h")}
for _, d := range dirs {
candidates = append(candidates, filepath.Join(d, "go_asm.h"))
}
for _, candidate := range candidates {
if st, err := os.Stat(candidate); err == nil && !st.IsDir() {
return true
}
}
return false
}
// generateGoAsmHeader type-checks the Go package in pkgDir for goos and
// goarch, writes its go_asm.h equivalent into dir, and returns dir. An
// empty goos means the ambient one. The caller owns the directory and its
// removal.
func generateGoAsmHeader(pkgDir, goos, goarch, dir string) (string, error) {
if goos == "" {
goos = build.Default.GOOS
}
imp := newSourceImporter(goos, goarch)
if imp.sizes == nil {
return "", fmt.Errorf("go_asm.h: unknown GOARCH %q", goarch)
}
bp, err := imp.ctxt.ImportDir(pkgDir, 0)
if err != nil {
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
}
files, errs := imp.parse(bp)
if len(errs) > 0 {
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %s", goarch, pkgDir, errorList(errs))
}
_, info, errs := imp.checkPackage(bp, files)
if len(errs) > 0 {
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: package does not type-check: %s", goarch, pkgDir, errorList(errs))
}
var b strings.Builder
fmt.Fprintf(&b, "// generated by gasm from package %s (GOOS %s, GOARCH %s)\n\n", bp.Name, goos, goarch)
// Files in the build's own order and declarations in source order: the
// same walk the compiler's reader makes, so the header reads the same
// way the toolchain's does. Order carries no meaning to the assembler
// (defines form a table), only to a human diffing against one.
for _, f := range files {
for _, decl := range f.Decls {
gd, ok := decl.(*ast.GenDecl)
if !ok {
continue
}
for _, spec := range gd.Specs {
switch gd.Tok {
case token.CONST:
vs, ok := spec.(*ast.ValueSpec)
if !ok {
continue
}
for _, name := range vs.Names {
emitConst(&b, info.Defs[name], name.Name)
}
case token.TYPE:
ts, ok := spec.(*ast.TypeSpec)
if !ok {
continue
}
emitStruct(&b, imp.sizes, info.Defs[ts.Name], ts.Name.Name)
}
}
}
}
if err := os.MkdirAll(dir, 0o755); err != nil {
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
}
out := filepath.Join(dir, "go_asm.h")
if err := os.WriteFile(out, []byte(b.String()), 0o644); err != nil {
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
}
return dir, nil
}
// emitConst writes one const define, skipping what the toolchain skips:
// blank names, and float and complex values the assembler has no syntax for.
func emitConst(b *strings.Builder, obj types.Object, name string) {
c, ok := obj.(*types.Const)
if !ok || name == "_" {
return
}
switch c.Val().Kind() {
case constant.Float, constant.Complex, constant.Unknown:
return
}
fmt.Fprintf(b, "#define const_%s %s\n", name, c.Val().ExactString())
}
// emitStruct writes one named struct type's size and field offsets,
// skipping what the toolchain skips: blank names, non-struct types, and
// generic types, whose size depends on their instantiation.
func emitStruct(b *strings.Builder, sizes types.Sizes, obj types.Object, name string) {
tn, ok := obj.(*types.TypeName)
if !ok || name == "_" {
return
}
t := types.Unalias(tn.Type())
// Generic types are spelled *types.Named with a type-parameter list;
// a plain struct type or an instantiated one carries none.
if named, ok := t.(*types.Named); ok && named.TypeParams().Len() > 0 {
return
}
st, ok := t.Underlying().(*types.Struct)
if !ok {
return
}
fmt.Fprintf(b, "#define %s__size %d\n", name, sizes.Sizeof(t))
fields := make([]*types.Var, st.NumFields())
for i := range st.NumFields() {
fields[i] = st.Field(i)
}
for i, off := range sizes.Offsetsof(fields) {
fld := fields[i]
if fld.Name() == "_" {
continue
}
fmt.Fprintf(b, "#define %s_%s %d\n", name, fld.Name(), off)
}
}
// errorList renders at most three errors, enough to say what is wrong
// without burying the diagnostic the caller actually reads.
func errorList(errs []error) string {
if len(errs) > 3 {
errs = errs[:3]
}
msgs := make([]string, len(errs))
for i, err := range errs {
msgs[i] = err.Error()
}
return strings.Join(msgs, "; ")
}
// sourceImporter type-checks imported packages from source with the target
// architecture's sizes. go/importer's "source" importer pins the host
// GOARCH, which would lay out imported types (internal/cpu, internal/abi)
// for the wrong target on a cross-architecture header, so the recursion is
// carried here with one build context and one sizes instance per
// architecture.
type sourceImporter struct {
fset *token.FileSet
ctxt *build.Context
sizes types.Sizes
pkgs map[string]*types.Package
}
// newSourceImporter returns the importer for one target GOOS and GOARCH.
// Cgo is disabled so the file set is deterministic and independent of the
// host's C toolchain: cgo-tagged files drop out of the build exactly as
// they do from a CGO_ENABLED=0 build, whose assembly is what gasm targets.
func newSourceImporter(goos, goarch string) *sourceImporter {
ctxt := new(build.Context)
*ctxt = build.Default
ctxt.GOOS = goos
ctxt.GOARCH = goarch
ctxt.CgoEnabled = false
return &sourceImporter{
fset: token.NewFileSet(),
ctxt: ctxt,
sizes: types.SizesFor("gc", goarch),
pkgs: map[string]*types.Package{},
}
}
// Import type-checks one imported package and memoises it. "unsafe" must
// resolve to go/types' own package, never to the source in GOROOT/src/unsafe:
// the source declares Sizeof and Offsetof as ordinary functions over
// ArbitraryType, and checking against that signature rejects half the
// unsafe arithmetic the gc compiler accepts, which is exactly the divergence
// srcimporter guards against the same way.
func (im *sourceImporter) Import(path string) (*types.Package, error) {
if path == "unsafe" {
return types.Unsafe, nil
}
if p, ok := im.pkgs[path]; ok {
return p, nil
}
bp, err := im.ctxt.Import(path, "", 0)
if err != nil {
return nil, err
}
files, errs := im.parse(bp)
if len(errs) > 0 {
return nil, errors.New(errorList(errs))
}
pkg, _, _ := im.checkPackage(bp, files)
im.pkgs[path] = pkg
return pkg, nil
}
// parse reads the build package's Go files. Import-level failures (no Go
// files for the target, unreadable files) come back as errors, and the
// type-check decides the rest.
func (im *sourceImporter) parse(bp *build.Package) ([]*ast.File, []error) {
if len(bp.GoFiles) == 0 {
return nil, []error{fmt.Errorf("no Go source files for GOOS=%s GOARCH=%s", im.ctxt.GOOS, im.ctxt.GOARCH)}
}
var (
files []*ast.File
errs []error
)
for _, name := range bp.GoFiles {
f, err := parser.ParseFile(im.fset, filepath.Join(bp.Dir, name), nil, parser.SkipObjectResolution)
if err != nil {
errs = append(errs, err)
continue
}
files = append(files, f)
}
return files, errs
}
// checkPackage type-checks one package's files with the importer's sizes,
// recording every error: a header from a package that does not type-check
// could silently mis-state an offset, so the caller refuses the header
// rather than trusting it. The returned Defs map backs the root package's
// emission walk; imports only need the checked package itself.
func (im *sourceImporter) checkPackage(bp *build.Package, files []*ast.File) (*types.Package, *types.Info, []error) {
var errs []error
conf := &types.Config{
Importer: im,
Sizes: im.sizes,
Error: func(err error) { errs = append(errs, err) },
}
info := &types.Info{Defs: map[*ast.Ident]types.Object{}}
pkg, _ := conf.Check(bp.ImportPath, im.fset, files, info)
return pkg, info, errs
}
// asmhdrCache generates one go_asm.h per package directory and target
// architecture under one temp root, for callers that assemble many files
// (the corpus audit). Failures are cached too: a package that does not
// type-check must not be re-checked once per file.
type asmhdrCache struct {
root string
dirs map[string]string // "pkgDir\x00goos\x00goarch" -> directory holding go_asm.h
errs map[string]error
}
func newAsmhdrCache() (*asmhdrCache, error) {
root, err := os.MkdirTemp("", "gasm-asmhdr")
if err != nil {
return nil, err
}
return &asmhdrCache{root: root, dirs: map[string]string{}, errs: map[string]error{}}, nil
}
// dirFor returns the directory holding the generated go_asm.h for pkgDir
// under goos and goarch, generating it on first use. An empty goos means
// the ambient one, resolved here so that one package cannot generate twice
// under an explicit and an implicit spelling of the same GOOS.
func (c *asmhdrCache) dirFor(pkgDir, goos, goarch string) (string, error) {
if goos == "" {
goos = build.Default.GOOS
}
key := pkgDir + "\x00" + goos + "\x00" + goarch
if dir, ok := c.dirs[key]; ok {
return dir, nil
}
if err, ok := c.errs[key]; ok {
return "", err
}
dir := filepath.Join(c.root, fmt.Sprintf("h%d_%s_%s", len(c.dirs), goos, goarch))
if _, err := generateGoAsmHeader(pkgDir, goos, goarch, dir); err != nil {
c.errs[key] = err
return "", err
}
c.dirs[key] = dir
return dir, nil
}
// close removes the temp root.
func (c *asmhdrCache) close() { os.RemoveAll(c.root) }
// ensureGoAsmHeader prepares the include directory a file that includes
// go_asm.h needs: the generated header for the package in path's directory,
// for the file's target GOOS and architecture. It reports a usage error
// when the architecture cannot be determined, and passes through the
// generator's diagnostics, which name the package.
func ensureGoAsmHeader(path string, target arch.Arch, goos string, cache *asmhdrCache) (string, func(), error) {
if path == "-" {
return "", nil, errors.New("cannot generate go_asm.h for standard input (no package directory)")
}
if target == arch.Unknown {
return "", nil, errors.New("a file that includes go_asm.h needs a target architecture: name the file _<arch>.s or pass -GOARCH")
}
if cache != nil {
dir, err := cache.dirFor(filepath.Dir(path), goos, goarchName(target))
return dir, func() {}, err
}
root, err := os.MkdirTemp("", "gasm-asmhdr")
if err != nil {
return "", nil, err
}
dir, err := generateGoAsmHeader(filepath.Dir(path), goos, goarchName(target), root)
if err != nil {
os.RemoveAll(root)
return "", nil, err
}
return dir, func() { os.RemoveAll(root) }, nil
}
+430
View File
@@ -0,0 +1,430 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
)
// writePkg lays out a minimal Go package in a temp directory.
func writePkg(t *testing.T, files map[string]string) string {
t.Helper()
dir := t.TempDir()
for name, src := range files {
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
return dir
}
// generateFor generates the header for dir and returns its text. An empty
// goos means the ambient one.
func generateFor(t *testing.T, dir, goos, goarch string) string {
t.Helper()
hdrDir, err := generateGoAsmHeader(dir, goos, goarch, t.TempDir())
if err != nil {
t.Fatalf("generateGoAsmHeader(%q, %s, %s): %v", dir, goos, goarch, err)
}
b, err := os.ReadFile(filepath.Join(hdrDir, "go_asm.h"))
if err != nil {
t.Fatal(err)
}
return string(b)
}
func TestGenerateGoAsmHeaderShape(t *testing.T) {
dir := writePkg(t, map[string]string{"sample.go": `package sample
const bufSize = 1024
const (
a = iota * 8
b
c
)
const (
strConst = "hello"
boolConst = true
floatConst = 1.5
_ = "the blank identifier is skipped"
)
const shift = 1 << 20
type reader struct {
r int64
w int64
_ [4]byte
name string
}
type scalar int
type aliased struct {
k uint32
v uint32
}
type alias = aliased
`})
hdr := generateFor(t, dir, "", "amd64")
want := []string{
"#define const_bufSize 1024",
// iota resolves through go/types, one define per name.
"#define const_a 0",
"#define const_b 8",
"#define const_c 16",
`#define const_strConst "hello"`,
"#define const_boolConst true",
// Floats are the toolchain's own skip, as are blank names.
"#define const_shift 1048576",
// The blank field still occupies its bytes: the pad after w runs to
// the string's 8-byte alignment.
"#define reader__size 40",
"#define reader_r 0",
"#define reader_w 8",
"#define reader_name 24",
// Non-struct named types carry no defines; aliases to structs do.
"#define aliased__size 8",
"#define aliased_k 0",
"#define aliased_v 4",
"#define alias__size 8",
"#define alias_k 0",
"#define alias_v 4",
}
for _, w := range want {
if !strings.Contains(hdr, w+"\n") {
t.Errorf("header misses %q\ngot:\n%s", w, hdr)
}
}
for _, banned := range []string{"#define const_floatConst", "#define _ ", "#define scalar"} {
if strings.Contains(hdr, banned) {
t.Errorf("header must not carry %s\ngot:\n%s", banned, hdr)
}
}
}
func TestGenerateGoAsmHeaderPerArch(t *testing.T) {
dir := writePkg(t, map[string]string{
"common.go": `package perarch
type layout struct {
a int32
p uintptr
}
`,
// The build-tagged file set is part of the contract: a per-arch
// package is exactly how internal/cpu declares its layouts.
"const_amd64.go": `//go:build amd64
package perarch
const flavour = 1
`,
"const_arm64.go": `//go:build arm64
package perarch
const flavour = 2
`,
})
amd64 := generateFor(t, dir, "", "amd64")
arm64 := generateFor(t, dir, "", "arm64")
if !strings.Contains(amd64, "#define const_flavour 1\n") {
t.Errorf("amd64 header misses const_flavour 1:\n%s", amd64)
}
if !strings.Contains(arm64, "#define const_flavour 2\n") {
t.Errorf("arm64 header misses const_flavour 2:\n%s", arm64)
}
if strings.Contains(arm64, "#define const_flavour 1\n") {
t.Errorf("arm64 header must not carry the amd64 file's value")
}
// SizesFor makes the layout the target's: uintptr is 4 bytes wide on
// 386 and 8 on amd64, which must move p and grow the struct.
if !strings.Contains(amd64, "#define layout__size 16\n") || !strings.Contains(amd64, "#define layout_p 8\n") {
t.Errorf("amd64 layout wrong:\n%s", amd64)
}
w386 := generateFor(t, dir, "", "386")
if !strings.Contains(w386, "#define layout__size 8\n") || !strings.Contains(w386, "#define layout_p 4\n") {
t.Errorf("386 layout wrong:\n%s", w386)
}
}
// TestGenerateGoAsmHeaderGOOS pins the GOOS half of the target: only the
// platform's own files type-check into the header, which is why
// sys_darwin_arm64.s cannot assemble against a linux-generated one.
func TestGenerateGoAsmHeaderGOOS(t *testing.T) {
dir := writePkg(t, map[string]string{
"common.go": `package goosaware
type shared struct {
a int32
}
`,
"plat_darwin.go": `//go:build darwin
package goosaware
type platform struct {
trampoline_numer int64
}
`,
"plat_windows.go": `//go:build windows
package goosaware
type platform struct {
callbackArgs__size int32
}
`,
})
darwin := generateFor(t, dir, "darwin", "arm64")
if !strings.Contains(darwin, "#define platform__size 8\n") || !strings.Contains(darwin, "#define platform_trampoline_numer 0\n") {
t.Errorf("darwin header misses the darwin layout:\n%s", darwin)
}
if strings.Contains(darwin, "callbackArgs") {
t.Errorf("darwin header must not carry the windows layout:\n%s", darwin)
}
windows := generateFor(t, dir, "windows", "arm64")
if !strings.Contains(windows, "#define platform_callbackArgs__size 0\n") {
t.Errorf("windows header misses the windows layout:\n%s", windows)
}
if strings.Contains(windows, "trampoline_numer") {
t.Errorf("windows header must not carry the darwin layout:\n%s", windows)
}
// The ambient GOOS is neither of the two, so only shared's defines are
// emitted; the shared type keeps its layout there.
ambient := generateFor(t, dir, "", "arm64")
if !strings.Contains(ambient, "#define shared__size 4\n") {
t.Errorf("ambient header misses the shared layout:\n%s", ambient)
}
if strings.Contains(ambient, "#define platform_") {
t.Errorf("ambient header must not carry either platform layout:\n%s", ambient)
}
}
func TestGoosFromFilename(t *testing.T) {
for path, want := range map[string]string{
"/x/sys_darwin_arm64.s": "darwin",
"/x/sys_windows_arm64.s": "windows",
"/x/asm_linux_amd64.s": "linux",
"/x/rt0_darwin_arm64.s": "darwin",
"/x/vgetrandom_zos_s390x.s": "zos",
"/x/rt0_js_wasm.s": "js",
"/x/memmove_amd64.s": "",
"/x/vlop_arm.s": "",
"/x/stubs.s": "",
} {
if got := goosFromFilename(path); got != want {
t.Errorf("goosFromFilename(%q) = %q, want %q", path, got, want)
}
}
}
func TestGenerateGoAsmHeaderErrors(t *testing.T) {
t.Run("type error", func(t *testing.T) {
dir := writePkg(t, map[string]string{"bad.go": `package bad
const x = undefinedIdent
`})
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
if err == nil {
t.Fatal("generation must fail for a package that does not type-check")
}
if !strings.Contains(err.Error(), dir) {
t.Errorf("error must name the package directory: %v", err)
}
if !strings.Contains(err.Error(), "type-check") {
t.Errorf("error must say the package does not type-check: %v", err)
}
})
t.Run("no go files", func(t *testing.T) {
dir := t.TempDir()
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
if err == nil {
t.Fatal("generation must fail without Go files")
}
if !strings.Contains(err.Error(), dir) {
t.Errorf("error must name the package directory: %v", err)
}
})
}
func TestNeedsGoAsmHeader(t *testing.T) {
yes := "#include \"go_asm.h\"\n#include \"textflag.h\"\n"
no := "#include \"textflag.h\"\n#include \"funcdata.h\"\n"
if !needsGoAsmHeader(yes) {
t.Error("needsGoAsmHeader(missing on a go_asm.h include)")
}
if needsGoAsmHeader(no) {
t.Error("needsGoAsmHeader claims other headers need generation")
}
}
func TestGoAsmHeaderResolved(t *testing.T) {
dir := t.TempDir()
if goAsmHeaderResolved(dir, nil) {
t.Error("resolved with no header anywhere")
}
other := t.TempDir()
if goAsmHeaderResolved(dir, []string{other}) {
t.Error("resolved with an empty -I directory")
}
if err := os.WriteFile(filepath.Join(dir, "go_asm.h"), nil, 0o644); err != nil {
t.Fatal(err)
}
if !goAsmHeaderResolved(dir, nil) {
t.Error("not resolved with the header in the package directory")
}
}
func TestOtherGOOSFile(t *testing.T) {
for path, want := range map[string]bool{
"/x/sys_windows_amd64.s": true,
"/x/rt0_js_wasm.s": true,
"/x/sys_darwin_arm64.s": true,
"/x/sys_linux_amd64.s": false,
"/x/time_linux_amd64.s": false,
"/x/memmove_amd64.s": false,
"/x/generic.s": false,
} {
if got := otherGOOSFile(path); got != want {
t.Errorf("otherGOOSFile(%q) = %v, want %v", path, got, want)
}
}
}
// TestRunCorpusAuditGoAsm covers the audit wiring end to end: a package
// beside its kernel, the kernel living off the generated defines, and the
// histogram recording a generation failure as its own reason.
func TestRunCorpusAuditGoAsm(t *testing.T) {
dir := t.TempDir()
write := func(name, src string) {
t.Helper()
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
write("pkg.go", `package corpus
const pageSize = 4096
type header struct {
magic uint64
flags uint64
}
`)
write("kern_amd64.s", "#include \"go_asm.h\"\nTEXT \xc2\xb7f(SB), NOSPLIT, $0-16\n\tMOVQ\t$const_pageSize, AX\n\tMOVQ\t$header__size, BX\n\tRET\n")
// The defines live in the file's own package; a kernel in a directory
// without Go files has no package to generate from.
if err := os.MkdirAll(filepath.Join(dir, "sub"), 0o755); err != nil {
t.Fatal(err)
}
write(filepath.Join("sub", "lonely_arm64.s"), "#include \"go_asm.h\"\nTEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n\tRET\n")
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
get := func(name string) *corpusTally {
for i, tg := range stats.targets {
if tg.name == name {
return stats.tallies[i]
}
}
t.Fatalf("no tally for %s", name)
return nil
}
if a := get("amd64"); a.attempted != 1 || a.assembled != 1 {
t.Errorf("amd64 = %d/%d, want 1/1", a.assembled, a.attempted)
}
// lonely_arm64.s is an arm64 file whose package cannot be generated.
if a := get("arm64"); a.attempted != 1 || a.assembled != 0 {
t.Errorf("arm64 = %d/%d, want 0/1", a.assembled, a.attempted)
}
if r := get("arm64").reasons["go_asm.h generation failed"]; r != 1 {
t.Errorf("arm64 go_asm.h failure count = %d, want 1", r)
}
}
// TestRunCorpusAuditGOOS covers the filename-derived GOOS end to end: a
// kernel whose name names darwin must have its header type-checked with
// GOOS=darwin, so the darwin-only constant it offsets with is defined. The
// operand mirrors sys_darwin_arm64.s's trampoline, where a missing define
// leaves an unexpanded symbol in the offset and fails.
func TestRunCorpusAuditGOOS(t *testing.T) {
dir := t.TempDir()
write := func(name, src string) {
t.Helper()
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
write("pkg.go", "package corpus\n")
write("plat_darwin.go", "//go:build darwin\n\npackage corpus\n\nconst trampolineNumer = 8\n")
write("kern_darwin_arm64.s", "#include \"go_asm.h\"\n"+
"GLOBL timebase<>(SB), NOPTR, $16\n"+
"TEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n"+
"\tMOVD\ttimebase<>+const_trampolineNumer(SB), R0\n"+
"\tRET\n")
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
var arm *corpusTally
for i, tg := range stats.targets {
if tg.name == "arm64" {
arm = stats.tallies[i]
}
}
if arm == nil {
t.Fatal("no arm64 tally")
}
if arm.attempted != 1 || arm.assembled != 1 {
t.Errorf("arm64 = %d/%d, want 1/1; reasons: %v", arm.assembled, arm.attempted, arm.reasons)
}
}
// TestGenerateGoAsmHeaderRuntime pins the generator against the real thing:
// the runtime package of the ambient toolchain, whose header the toolchain's
// own -asmhdr output was sampled from. Skipped in short mode: it type-checks
// the whole package. The GOROOT comes from the go command itself, so the
// test follows whatever toolchain the host provides.
func TestGenerateGoAsmHeaderRuntime(t *testing.T) {
if testing.Short() {
t.Skip("type-checks the whole runtime package")
}
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
t.Skipf("no Go toolchain: %v", err)
}
runtimeDir := filepath.Join(strings.TrimSpace(string(out)), "src", "runtime")
dir, err := generateGoAsmHeader(runtimeDir, "", "amd64", t.TempDir())
if err != nil {
t.Fatalf("generateGoAsmHeader(runtime): %v", err)
}
b, err := os.ReadFile(dir + "/go_asm.h")
if err != nil {
t.Fatal(err)
}
hdr := string(b)
for _, want := range []string{
"#define const_hashSize 8\n",
"#define const_avxSupported 1\n",
"#define const_pageSize 8192\n",
"#define g_stackguard0 16\n",
"#define m__size ",
} {
if !strings.Contains(hdr, want) {
t.Errorf("runtime header misses %q", want)
}
}
}
+505 -8
View File
@@ -5,6 +5,7 @@ package main
import (
"fmt"
"maps"
"os"
"os/exec"
"path/filepath"
@@ -37,21 +38,41 @@ import (
// construction and are excluded from the diff; the other architectures list
// their conditional branches outright.
func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [amd64|arm64|riscv64|loong64]", `
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj), known-but-unencodable names (the backlog) and go-
only names (feature gaps). The Go side is probed black-box with a battery
of bare mnemonics, so the audit tracks whatever toolchain `+"`go env GOROOT`"+`
provides.
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
Go side is probed black-box one bare mnemonic at a time, so the audit tracks
whatever toolchain `+"`go env GOROOT`"+` provides; the gasm side answers from
the encoder table on amd64 and from trial assembly over a battery of operand
shapes elsewhere. Names go tool asm knows and gasm does not cannot be
enumerated by probing, because Go's table is visible only through names
already in the gasm table; the report closes with a note saying so.
With --corpus the audit changes shape: it assembles every .s file under the
given directory (default GOROOT/src) with the gasm encoder only, no
toolchain probing. A file whose name carries a recognisable _arch suffix is
attempted for that architecture; a file without one is attempted for all
four, exactly as a GOARCH build would compile it. The report gives the
per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order. With
-list the report also prints every failing file with its reason, per
architecture.
`)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
list := fs.Bool("list", false, "with --corpus, list every failing file with its reason, per architecture")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
if err := fs.Parse(args); err != nil {
return err
}
if *corpus {
return cmdAuditCorpus(fs.Args(), dirs, *list)
}
archName := "amd64"
switch n := len(fs.Args()); {
case n > 1:
return fmt.Errorf("audit-instructions takes at most one architecture argument")
return &usageError{fmt.Errorf("audit-instructions takes at most one architecture argument")}
case n == 1:
archName = strings.ToLower(fs.Arg(0))
}
@@ -97,7 +118,7 @@ provides.
w := os.Stdout
fmt.Fprintf(w, "gasm table (%s, families excluded): %d mnemonics\n", archName, len(names))
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognized: %d\n", len(shared)+len(superset), countTrue(goKnown))
fmt.Fprintf(w, "gasm encodable: %d go tool asm recognised: %d\n", len(shared)+len(superset), countTrue(goKnown))
fmt.Fprintf(w, "shared: %d\n", len(shared))
fmt.Fprintf(w, "\nSuperset encodings (gasm-only; ship via gasm asm --format goobj):\n")
for _, n := range superset {
@@ -124,7 +145,7 @@ func auditArch(name string) (arch.Arch, error) {
case "loong64", "loong":
return arch.LOONG64, nil
}
return arch.Unknown, fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)
return arch.Unknown, &usageError{fmt.Errorf("unknown architecture %q: want amd64, arm64, riscv64 or loong64", name)}
}
// goarchName maps an arch identifier onto its GOARCH spelling.
@@ -205,6 +226,13 @@ func probeGoAsm(goarch string, names []string) (map[string]bool, error) {
cmd := exec.Command(asmBin, "-p", "probe", "-o", filepath.Join(dir, "probe.o"), probePath)
cmd.Env = append(os.Environ(), "GOARCH="+goarch, "GOOS="+runtime.GOOS)
out, _ := cmd.CombinedOutput()
// The expected failure mode is a non-zero exit with compiler diagnostics
// on stdout; empty output means the probe broke at the exec level (a
// killed child, a tool that would not start), and seeding every name as
// recognized on that silence would fake a clean audit.
if len(out) == 0 {
return nil, fmt.Errorf("go tool asm probe for GOARCH=%s produced no output", goarch)
}
result := map[string]bool{}
for _, name := range names {
@@ -244,18 +272,76 @@ func probeShapes(a arch.Arch) []string {
// and takes R register spellings.
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
"GE, F0, F1, F2", "NE, F0, F1, $0",
// Pairs, acquire/release and exclusive atomics, LSE-AL forms.
"(R0), R1", "R0, (R1)", "R1, (R2), R3", "(R2, R3), 8(R1)",
"8(R1), (R2, R3)", "R1, R2, (R3)", "(R0)",
// System operations and their register/operand names.
"$4, R1, p2", "$35943", "$1", "$1, SPSel", "SPSel, R0",
"IVAC, R0", "(R0), PLDL1KEEP", "R1, R2, R3, R4",
// SIMD element, structure and literal-pool forms.
"(R0), [V1.B16]", "[V1.B16], (R0)", "V13.S[0], R1",
"R1, V2.B[3]", "$4, V1.B16, V2.B16", "V1.B16, (R0)",
"(R0), V1.B16", "",
// The spellings GOROOT's own kernels use, from the
// differential kernels this table was proven against.
"R0, p2", "R0, R1", "F0, F1, F2, F3", "$4, V1.B16, V2.B16, V3.B16, V4.B16",
"(R0), [V0.B8, V1.B8, V2.B8, V3.B8]", "$1, $2, V1",
"R0, R1, p2", "p2, R1", "$1234, R1", "DCZID_EL0, R1",
"$0", "R1, $4, EQ", "$33, R1, $25, R2", "$4, R1, p2",
"$4, V1.B8, V2.B8, V3.B8", "$63, V1.D2, V2.D2, V3.D2",
"V1.B16, [V2.B16], V3.B16", "V1.B8, [V2.B16, V3.B16], V4.B8",
"$4, V1.B16, V2.B16, V3.B16", "$15, V1", "V1, V2, p2",
"R0, R1, $1, $4, p2",
// The landing-pad kind, the compiler's PCDATA
// bookkeeping and the four-operand bitfield
// insert/extract family, as the toolchain's own
// testdata spells them.
"C", "$1, $0", "$0, R1, $1, R2",
}
case arch.RISCV:
return []string{
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
"X5, X6, p2", "p2(SB)",
// AMO atomics: destination, base, source.
"R5, (R4), R6", "X5, (X4), X6",
// Segment stores take the first vector register aligned
// to the segment count, as the toolchain requires.
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
// The FP multiply-add family takes four registers.
"F0, F1, F2, F3",
// The RVV slice: register, vector-register and vtype forms.
"V1, V2, V3", "V1, X5, V2", "V1", "V1, (X5)", "(X5), V1",
"$15, V1", "$15", "V1, V2", "V1, X5",
"X5, X6, p2", "R5, R6, p2",
"X5, E8, M8, TA, MA, X6", "$4, E32, M1, TA, MA, X1",
"(X5), X6, V1, V2",
// The CSR immediate forms the toolchain's testdata spells:
// immediate, CSR name, destination.
"$2, TIME, X5",
"",
}
case arch.LOONG64:
return []string{
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
// AMO atomics: destination, base, source.
"R5, (R4), R6", "X5, (X4), X6",
// Segment stores take the first vector register aligned
// to the segment count, as the toolchain requires.
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
// The LSX and LASX banks share the 5-bit numbering with F.
"V1, V2, V3", "X1, X2, X3", "V1, V2", "X1, X2", "V1", "X1",
// The vector compare-to-flag forms land in an FCC register.
"V1, FCC0", "X1, FCC0",
// The compiler's bookkeeping pair and the raw spellings the
// toolchain's own testdata carries: JIRL rd, rj, offset (the
// form RET lowers to), the prefetch with a 32-bit address and
// hint, and the byte-shuffle quads.
"$1, $0", "R1, R5, 0", "0(R7), $5, $0", "(R7), $5, $0",
"V1, V2, V3, V4", "X1, X2, X3, X4",
"",
}
}
return nil
@@ -305,3 +391,414 @@ func gasmAssembles(a arch.Arch, name, shape string) bool {
func sanitize(name string) string {
return strings.NewReplacer(".", "_", "$", "_").Replace(name)
}
// --- corpus audit -----------------------------------------------------------
// corpusTarget is one architecture row of the corpus report.
type corpusTarget struct {
a arch.Arch
name string
}
// corpusTally accumulates one architecture's attempts over the corpus.
type corpusTally struct {
attempted int
assembled int
reasons map[string]int // failure reason → count
example map[string]string // failure reason → one representative file
fails []corpusFailure // every failure, in file order, for --list
}
// corpusFailure is one failed attempt, recorded for the --list report.
type corpusFailure struct {
path string
reason string
detail string
}
func (t *corpusTally) fail(path string, err error) {
reason := corpusReason(err)
t.reasons[reason]++
if t.example[reason] == "" {
t.example[reason] = path
}
t.fails = append(t.fails, corpusFailure{path: path, reason: reason, detail: firstLine(err.Error())})
}
// cmdAuditCorpus implements audit-instructions --corpus. The include
// directories carry #include resolution over a corpus whose files refer to
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
// needs.
func cmdAuditCorpus(args []string, dirs includeDirs, list bool) error {
if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
}
root := ""
if len(args) == 1 {
root = args[0]
} else {
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
return fmt.Errorf("locate GOROOT: %w", err)
}
root = filepath.Join(strings.TrimSpace(string(out)), "src")
}
// The toolchain's shipped headers (funcdata.h and friends) define the
// macros GOROOT files include; a corpus audit measures those files, so
// the header directory joins the search path automatically. go_asm.h
// is compiler-generated per package, so it is not resolved from here:
// files that include it get one generated per target architecture,
// which runCorpusAudit arranges.
if out, err := exec.Command("go", "env", "GOROOT").Output(); err == nil {
pkgInclude := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
if fi, err := os.Stat(pkgInclude); err == nil && fi.IsDir() {
seen := false
for _, d := range dirs {
if d == pkgInclude {
seen = true
}
}
if !seen {
dirs = append(dirs, pkgInclude)
}
}
}
stats, err := runCorpusAudit(root, dirs)
if err != nil {
return err
}
printCorpusStats(stats, list)
return nil
}
// corpusStats is the outcome of one corpus audit run.
type corpusStats struct {
root string
files int
generic int // files attempted for all four architectures
otherPort int // files named for another Go port: never attempted
full int // files that assembled for every target architecture
targets []corpusTarget
tallies []*corpusTally
}
// runCorpusAudit assembles every .s file under root and returns the stats.
// goPortSuffixes lists every architecture the Go project ports to. A file
// named for one of them belongs to that port's build, not to the generic
// set, even when gasm does not support the architecture.
var goPortSuffixes = []string{
"386", "amd64", "arm", "arm64", "loong64", "mips", "mips64",
"mips64le", "mipsle", "mips64x", "mipsx", "ppc64", "ppc64le",
"ppc64x", "riscv", "riscv64", "s390x", "wasm",
}
// otherPortFile reports whether the file belongs to a build no supported
// target ever compiles: either its name carries a Go-architecture suffix
// gasm does not support, or, for a file with no architecture suffix at all,
// it names another GOOS, which go/build drops from the file set
// (rt0_js_wasm.s is a javascript build, not a generic one).
func otherPortFile(path string) bool {
if otherGOOSFile(path) {
return true
}
base := path
if i := strings.LastIndexByte(base, '/'); i >= 0 {
base = base[i+1:]
}
for _, sfx := range goPortSuffixes {
if strings.HasSuffix(base, "_"+sfx+".s") {
return true
}
}
return false
}
// goOSNames are the GOOS values go/build recognises in file names.
var goOSNames = map[string]bool{
"aix": true, "android": true, "darwin": true, "dragonfly": true,
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
"js": true, "linux": true, "nacl": true, "netbsd": true,
"openbsd": true, "plan9": true, "solaris": true, "wasip1": true,
"windows": true, "zos": true,
}
// resolveGOOS validates a -GOOS flag value, mirroring the architecture
// check's surface: a usage error naming what the tool accepts.
func resolveGOOS(name string) (string, error) {
lower := strings.ToLower(name)
if goOSNames[lower] {
return lower, nil
}
return "", &usageError{fmt.Errorf("unknown GOOS %q: want one of %s", name, strings.Join(slices.Sorted(maps.Keys(goOSNames)), ", "))}
}
// goosFromFilename returns the GOOS the file's name carries, by go/build's
// goodOSArchFile rule: the GOOS segment sits last, or last before the
// architecture segment (sys_darwin_arm64.s, vlop_arm.s carries none). An
// empty result means the name names no GOOS and the ambient one applies.
func goosFromFilename(path string) string {
base := path
if i := strings.LastIndexByte(base, '/'); i >= 0 {
base = base[i+1:]
}
base = strings.TrimSuffix(base, ".s")
// go/build ignores everything before the first underscore, so a GOOS
// segment is only ever looked for from there on.
i := strings.IndexByte(base, '_')
if i < 0 {
return ""
}
segs := strings.Split(base[i:], "_")
if n := len(segs); n >= 2 && goOSNames[segs[n-2]] && slices.Contains(goPortSuffixes, segs[n-1]) {
return segs[n-2]
}
if goOSNames[segs[len(segs)-1]] {
return segs[len(segs)-1]
}
return ""
}
// otherGOOSFile reports whether the file's name names a GOOS other than the
// host's, by go/build's file-name rules.
func otherGOOSFile(path string) bool {
base := path
if i := strings.LastIndexByte(base, '/'); i >= 0 {
base = base[i+1:]
}
for seg := range strings.SplitSeq(strings.TrimSuffix(base, ".s"), "_") {
if goOSNames[seg] && seg != runtime.GOOS {
return true
}
}
return false
}
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
files, err := asmFiles(root)
if err != nil {
return nil, err
}
targets := []corpusTarget{
{arch.AMD64, "amd64"},
{arch.ARM64, "arm64"},
{arch.RISCV, "riscv64"},
{arch.LOONG64, "loong64"},
}
tallies := make([]*corpusTally, len(targets))
for i := range tallies {
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
}
// full is the north-star number: a file counts when every architecture
// its name allows assembles it.
full, generic, otherPort := 0, 0, 0
// Header generation is created on first use, so a corpus with no
// go_asm.h includes never pays for a temp directory.
var hdr *asmhdrCache
defer func() {
if hdr != nil {
hdr.close()
}
}()
for _, path := range files {
src, err := readSource(path)
if err != nil {
return nil, err
}
// The GOOS the header generation type-checks under follows the
// file's name when the name carries one; the ambient GOOS is the
// honest guess otherwise (a build tag naming another GOOS is
// invisible to a file-name rule).
goos := goosFromFilename(path)
var wanted []int // indexes into targets
if a := arch.FromFilename(path); a != arch.Unknown {
for i, tg := range targets {
if tg.a == a {
wanted = append(wanted, i)
}
}
} else if otherPortFile(path) {
// A file named for a Go port gasm does not support (arm,
// 386, s390x, ...) or for another GOOS is compiled by no
// supported-arch build, so it is neither generic nor a
// per-arch attempt: counting it as generic would make the
// headline unreachably low for reasons no supported target
// can fix.
otherPort++
} else {
generic++
for i := range targets {
wanted = append(wanted, i)
}
}
// A file that includes go_asm.h parses against a per-target header:
// the defines differ per architecture (internal/cpu's layout, for
// one) and per GOOS (sys_darwin_arm64.s's trampoline constants,
// for another), so the parse cannot be shared the way a
// header-free file's can. A generation failure is a failure for
// every target, named for the package rather than a bare "include
// not found". A header already resolvable in the package
// directory or the -I list is left alone.
if len(wanted) > 0 && needsGoAsmHeader(src) && !goAsmHeaderResolved(filepath.Dir(path), dirs) {
if hdr == nil {
if hdr, err = newAsmhdrCache(); err != nil {
return nil, err
}
}
pkgDir := filepath.Dir(path)
ok := true
for _, i := range wanted {
tg, t := targets[i], tallies[i]
t.attempted++
hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a))
if err != nil {
ok = false
t.fail(path, err)
continue
}
f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true,
IncludeDirs: append(slices.Clone(dirs), hdrDir),
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
})
if len(errs) > 0 {
ok = false
t.fail(path, errs[0])
continue
}
if _, err := assembleFile(tg.a, f, goos); err != nil {
ok = false
t.fail(path, err)
continue
}
t.assembled++
}
if ok && len(wanted) > 0 {
full++
}
continue
}
ok := true
for _, i := range wanted {
tg, t := targets[i], tallies[i]
t.attempted++
// The parse carries the target's platform predefines, so it
// cannot be shared across targets the way a header-free file's
// could: a #ifdef GOARCH_arm block must be live on arm64 and
// dead everywhere else.
f, errs := parser.ParseWithOptions(path, src, parser.Options{
Expand: true,
IncludeDirs: dirs,
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
})
var err error
if len(errs) > 0 {
err = errs[0] // a parse failure is a failure for every target
} else {
_, err = assembleFile(tg.a, f, goos)
}
if err != nil {
ok = false
t.fail(path, err)
continue
}
t.assembled++
}
if ok && len(wanted) > 0 {
full++
}
}
return &corpusStats{
root: root,
files: len(files),
generic: generic,
otherPort: otherPort,
full: full,
targets: targets,
tallies: tallies,
}, nil
}
// printCorpusStats renders the corpus audit report.
func printCorpusStats(s *corpusStats, list bool) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort)
// The rate is over the files a supported build would attempt: the
// other ports' files sit in the count for completeness but can never
// assemble, so counting them in the denominator would report the gap
// of architectures gasm deliberately does not target.
attemptable := max(s.files-s.otherPort, 1)
fmt.Printf(" assemble for every target architecture: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
for i, tg := range s.targets {
t := s.tallies[i]
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
for _, r := range topReasons(t) {
fmt.Printf(" %4d %s\n", t.reasons[r], r)
fmt.Printf(" e.g. %s\n", t.example[r])
}
if !list {
continue
}
for _, f := range t.fails {
fmt.Printf(" FAIL %s\n", f.path)
fmt.Printf(" %s: %s\n", f.reason, f.detail)
}
}
}
// corpusReason buckets an assembly or parse failure for the histogram.
func corpusReason(err error) string {
msg := err.Error()
switch {
case strings.Contains(msg, "go_asm.h for GOARCH"):
return "go_asm.h generation failed"
case strings.Contains(msg, "unsupported"), strings.Contains(msg, "cannot encode"):
return "instruction not encodable"
case strings.Contains(msg, "undefined label"):
return "undefined label"
case strings.Contains(msg, "undefined symbol"), strings.Contains(msg, "external symbol"), strings.Contains(msg, "file-level assembly"):
return "undefined symbol or external"
case strings.Contains(msg, "operand"), strings.Contains(msg, "operand form"):
return "unsupported operand form"
default:
return "other: " + firstLine(msg)
}
}
// topReasons returns at most five reasons, most frequent first.
func topReasons(t *corpusTally) []string {
type kv struct {
k string
n int
}
var kvs []kv
for k, n := range t.reasons {
kvs = append(kvs, kv{k, n})
}
slices.SortFunc(kvs, func(a, b kv) int { return b.n - a.n })
if len(kvs) > 5 {
kvs = kvs[:5]
}
out := make([]string, len(kvs))
for i, kv := range kvs {
out[i] = kv.k
}
return out
}
// firstLine returns the first line of an error message, truncated.
func firstLine(msg string) string {
if i := strings.IndexByte(msg, '\n'); i >= 0 {
msg = msg[:i]
}
if len(msg) > 80 {
msg = msg[:80]
}
return msg
}
+12 -7
View File
@@ -24,9 +24,10 @@ function in a traced subprocess (ptrace), then provides a REPL for
single-stepping, breakpoints, register and memory inspection.
REPL commands:
break <label|addr> [if <reg> <op> <val>]
break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
set a breakpoint, optionally conditional on a
register comparison (reg-reg or reg-immediate)
comparison of one register against a constant,
another register, or the 8-byte word at *addr
delete <label|addr> remove a breakpoint
info break list all breakpoints
step [n], s single-step n instructions (default 1)
@@ -53,7 +54,7 @@ REPL commands:
bufSpec := fs.String("buf", "", "buffer specification: name:size:pattern[,name:size:pattern...] where pattern is zero, ones, seq, or hex")
script := fs.String("script", "", "run REPL commands from a file (one per line) and exit; '-' reads stdin")
cover := fs.Bool("cover", false, "run to completion with a breakpoint on every instruction and report which executed and how often")
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs")
timeout := fs.Duration("timeout", 0, "kill the debuggee after this duration (e.g. 30s); for headless --script runs; a timeout exits 3")
fs.Parse(args)
// --- Debuggee mode (internal, spawned by the debugger) ---
@@ -84,7 +85,7 @@ REPL commands:
if *timeout > 0 {
go func() {
time.Sleep(*timeout)
fmt.Fprintf(os.Stderr, "gasm debug: timeout (%s) — killing the debuggee\n", *timeout)
fmt.Fprintf(os.Stderr, "gasm debug: timeout (%s), killing the debuggee\n", *timeout)
os.Exit(3)
}()
}
@@ -140,7 +141,6 @@ REPL commands:
}
// Construct the argument block with buffer pointers at the correct positions.
bufIdx := 0
for _, arg := range layout {
if !arg.IsPtr {
continue
@@ -170,12 +170,10 @@ REPL commands:
argBlock[off+16+j] = byte(size >> (j * 8))
}
}
bufIdx++
break
}
}
}
_ = bufIdx
} else {
argBlock = make([]byte, fl.Args)
sess, err = debug.Launch("", path, *funcName, argBlock)
@@ -234,6 +232,13 @@ REPL commands:
if sess.Exited() {
break
}
// A genuine signal-delivery-stop (a fault in the kernel): the
// run cannot make progress, because resuming would restart the
// faulting instruction and fault forever. Report and stop.
if sig := sess.LastSignal(); sig != 0 {
fmt.Printf("gasm debug: cover: stopped on signal %v\n", sig)
break
}
regs, rerr := sess.GetRegs()
if rerr != nil {
break
+148
View File
@@ -0,0 +1,148 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"os"
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// cmdDis disassembles machine code: either a raw binary (standard input with
// "-") whose architecture is given with -a, or a .s file, which is assembled
// first so the listing shows the real function and label layout.
func cmdDis(args []string) int {
fs := newCommand("dis", "gasm dis [-a arch] <file>", `
Disassemble machine code to instruction text (via golang.org/x/arch).
With a .s file, the file is assembled first and the listing follows the
real layout: one block per TEXT function, local labels printed at their
offsets. The architecture comes from the file name suffix, or from -a.
With any other file, or "-" for standard input, the bytes are disassembled
linearly and -a selects the architecture (amd64, arm64, riscv64 or
loong64).
`)
archName := fs.String("a", "", "architecture for raw input: amd64, arm64, riscv64 or loong64")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm dis [-a arch] <file>")
return 2
}
path := fs.Arg(0)
var target arch.Arch
if *archName != "" {
var err error
target, err = auditArch(*archName)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 2
}
}
if strings.HasSuffix(path, ".s") {
if target == arch.Unknown {
target = arch.FromFilename(path)
}
if target == arch.Unknown {
fmt.Fprintln(os.Stderr, "gasm dis: cannot infer the architecture from the file name; use -a")
return 2
}
return disSource(path, target)
}
if target == arch.Unknown {
fmt.Fprintln(os.Stderr, "gasm dis: raw input needs -a (amd64, arm64, riscv64 or loong64)")
return 2
}
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm dis:", err)
return 1
}
printListing(target, []byte(src), 0, nil)
return 0
}
// disSource assembles a .s file and prints one listing block per function.
func disSource(path string, target arch.Arch) int {
src, err := readSource(path)
if err != nil {
fmt.Fprintln(os.Stderr, "gasm dis:", err)
return 1
}
f, errs := parser.Parse(path, src)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
if len(errs) > 0 {
return 1
}
img, err := assembleFile(target, f, "")
if err != nil {
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
return 1
}
if len(img.Funcs) == 0 {
fmt.Fprintln(os.Stderr, "gasm dis: no assemblable TEXT functions found")
return 1
}
for _, fn := range img.Funcs {
code := img.Code[fn.Offset : fn.Offset+fn.Size]
fmt.Printf("%s: %d bytes\n", fn.Name, fn.Size)
labels := make(map[int][]string, len(fn.Labels))
for name, off := range fn.Labels {
labels[off] = append(labels[off], name)
}
for off := range labels {
sort.Strings(labels[off])
}
printListing(target, code, uint64(fn.Offset), labels)
}
if len(img.Data) > 0 {
fmt.Printf("data: %d bytes at 0x%x\n", len(img.Data), len(img.Code))
}
return 0
}
// printListing decodes code linearly from offset base, printing label lines
// (label name to offset within the block) as they are reached.
func printListing(a arch.Arch, code []byte, base uint64, labels map[int][]string) {
pc := 0
for pc < len(code) {
for _, name := range labels[pc] {
fmt.Printf("%s:\n", name)
}
ins, err := disasm.Decode(a, code[pc:], base+uint64(pc))
if err != nil {
break
}
end := min(pc+ins.Len, len(code))
fmt.Printf(" %04x: %-16s %s\n", base+uint64(pc), hexBytes(code[pc:end]), ins.Text)
if ins.Len <= 0 {
break
}
pc += ins.Len
}
}
// hexBytes renders up to 8 bytes as contiguous hex.
func hexBytes(b []byte) string {
var sb strings.Builder
for i, c := range b {
if i == 8 {
break
}
if i > 0 {
sb.WriteByte(' ')
}
fmt.Fprintf(&sb, "%02x", c)
}
return sb.String()
}
+376 -381
View File
File diff suppressed because it is too large Load Diff
+171 -3
View File
@@ -13,6 +13,9 @@ import (
"strings"
"syscall"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
)
const clean = "#include \"textflag.h\"\n" +
@@ -219,8 +222,9 @@ func TestCmdVersion(t *testing.T) {
if code != 0 {
t.Fatalf("code = %d", code)
}
if !strings.Contains(out, version) {
t.Errorf("version output %q does not mention %q", out, version)
got := version()
if !strings.Contains(out, got) {
t.Errorf("version output %q does not mention %q", out, got)
}
}
@@ -241,6 +245,96 @@ func TestCmdArgErrors(t *testing.T) {
}
}
// TestUsageExitCodes pins the exit-code contract for the commands whose main
// dispatches on a returned error: a wrong argument set exits 2, the same as
// the commands that count their arguments themselves, while a runtime
// failure (an unreadable file) keeps exit 1.
func TestUsageExitCodes(t *testing.T) {
for name, err := range map[string]error{
"audit-instructions extra argument": cmdAuditInstructions([]string{"amd64", "extra"}),
"audit-instructions unknown arch": cmdAuditInstructions([]string{"mips"}),
"audit-instructions corpus extra": cmdAuditInstructions([]string{"--corpus", "a", "b"}),
"scaffold no arguments": cmdScaffold(nil),
"scaffold extra arguments": cmdScaffold([]string{"differential", "a.s", "b.s"}),
} {
if err == nil {
t.Errorf("%s: expected an error", name)
continue
}
if code := exitCodeFor(err); code != 2 {
t.Errorf("%s: exit code = %d, want 2 (err: %v)", name, code, err)
}
}
if err := cmdScaffold([]string{"differential", "/nonexistent/file.s"}); err == nil {
t.Error("scaffold on a missing file should fail")
} else if code := exitCodeFor(err); code != 1 {
t.Errorf("scaffold on a missing file: exit code = %d, want 1", code)
}
}
// TestCmdAsmFormatValidation checks that an unknown --format exits 2 with
// and without -o, instead of assembling and silently dumping a raw image.
func TestCmdAsmFormatValidation(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out := filepath.Join(t.TempDir(), "f.bin")
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", path}) }); code != 2 {
t.Errorf("asm --format bogus without -o: code = %d, want 2", code)
}
if _, _, code := capture(func() int { return cmdAsm([]string{"--format", "bogus", "-o", out, path}) }); code != 2 {
t.Errorf("asm --format bogus with -o: code = %d, want 2", code)
}
}
// TestCmdAsmOutputFile pins the documented -o behaviour: the output goes to
// the file and stdout carries no hex dump; without -o the dump is the output.
func TestCmdAsmOutputFile(t *testing.T) {
path := writeTemp(t, "f_amd64.s", clean)
out := filepath.Join(t.TempDir(), "f.bin")
stdout, _, code := capture(func() int { return cmdAsm([]string{"-o", out, path}) })
if code != 0 {
t.Fatalf("code = %d", code)
}
if strings.Contains(stdout, "0000:") {
t.Errorf("stdout carries a hex dump despite -o:\n%s", stdout)
}
if !strings.Contains(stdout, "wrote ") {
t.Errorf("stdout misses the wrote line:\n%s", stdout)
}
b, err := os.ReadFile(out)
if err != nil {
t.Fatal(err)
}
if len(b) == 0 {
t.Error("the output file is empty")
}
stdout, _, code = capture(func() int { return cmdAsm([]string{path}) })
if code != 0 {
t.Fatalf("without -o: code = %d", code)
}
if !strings.Contains(stdout, "0000:") {
t.Errorf("without -o the hex dump is missing:\n%s", stdout)
}
}
// TestVerifyNonJITAMD64GroundTruth drives the cross-architecture
// ground-truth path for an amd64 kernel: the path a host of any other
// architecture takes, which must compare against the toolchain rather than
// refuse to run.
func TestVerifyNonJITAMD64GroundTruth(t *testing.T) {
if testing.Short() {
t.Skip("runs go tool asm")
}
path := writeTemp(t, "f_amd64.s", clean)
out, _, code := capture(func() int { return cmdVerifyNonJIT(path, arch.AMD64, true, false) })
if code != 0 {
t.Fatalf("code = %d (%s)", code, out)
}
if !strings.Contains(out, "1/1 matched") {
t.Errorf("output misses the matched report:\n%s", out)
}
}
// TestVerifySmokeCrashIsolation checks that a function faulting on its
// zeroed smoke arguments is reported as CRASH by a child process instead of
// killing `gasm verify` itself.
@@ -274,7 +368,7 @@ func TestVerifySmokeCrashIsolation(t *testing.T) {
}
if exitErr, ok := err.(*exec.ExitError); ok {
if ws, ok := exitErr.Sys().(syscall.WaitStatus); ok && ws.Signaled() {
t.Fatalf("verify died from %v — the crash was not isolated:\n%s", ws.Signal(), out)
t.Fatalf("verify died from %v; the crash was not isolated:\n%s", ws.Signal(), out)
}
}
if !strings.Contains(string(out), "CRASH") {
@@ -292,3 +386,77 @@ func TestSweepCheckLines(t *testing.T) {
t.Errorf("sweepCheckLines = %q, want %q", got, want)
}
}
// TestRunCorpusAudit drives the corpus audit over a small fixture tree: one
// suffixed amd64 file, one suffixed arm64 file whose body is not arm64, one
// generic file, and one file that does not parse.
func TestRunCorpusAudit(t *testing.T) {
dir := t.TempDir()
write := func(name, src string) {
t.Helper()
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
t.Fatal(err)
}
}
write("good_amd64.s", "#include \"textflag.h\"\nTEXT ·add(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
write("bad_arm64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0-0\n\tMOVQ AX, BX\n\tRET\n")
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
if stats.files != 4 {
t.Errorf("files = %d, want 4", stats.files)
}
if stats.generic != 2 {
t.Errorf("generic = %d, want 2 (generic.s and broken.s)", stats.generic)
}
// good_amd64 and generic.s assemble everywhere they are attempted.
if stats.full != 2 {
t.Errorf("full = %d, want 2", stats.full)
}
get := func(name string) *corpusTally {
for i, tg := range stats.targets {
if tg.name == name {
return stats.tallies[i]
}
}
t.Fatalf("no tally for %s", name)
return nil
}
// amd64: good_amd64 + generic.s + broken.s; the broken file fails to parse.
if a := get("amd64"); a.attempted != 3 || a.assembled != 2 {
t.Errorf("amd64 = %d/%d, want 2/3", a.assembled, a.attempted)
}
// arm64: bad_arm64 (MOVQ is not arm64) + generic.s + broken.s.
if a := get("arm64"); a.attempted != 3 || a.assembled != 1 {
t.Errorf("arm64 = %d/%d, want 1/3", a.assembled, a.attempted)
}
if r := get("amd64").reasons["instruction not encodable"]; r != 0 {
t.Errorf("amd64 unexpected unencodable reason: %d", r)
}
if r := get("arm64").reasons["instruction not encodable"]; r != 1 {
t.Errorf("arm64 unencodable reasons = %d, want 1", r)
}
}
// TestCompareGroundTruthPadding pins the padding-aware ground-truth
// comparison: the toolchain pads text symbols to 16-byte boundaries, so
// trailing zeros in the reference must not read as a mismatch, while any
// non-zero tail still must.
func TestCompareGroundTruthPadding(t *testing.T) {
code := []byte{0x48, 0x8b, 0x07, 0xc3} // 4 bytes, not a multiple of 16
img := &asm.Image{Code: code, Funcs: []asm.FuncLayout{{Name: "f", Offset: 0, Size: len(code)}}}
padded := append(append([]byte(nil), code...), 0, 0, 0)
matched, total, diffs := compareGroundTruth(img, map[string][]byte{"f": padded})
if matched != 1 || total != 1 || diffs != 0 {
t.Fatalf("zero padding should match: matched=%d total=%d diffs=%d", matched, total, diffs)
}
dirty := append(append([]byte(nil), code...), 0, 0x90, 0)
matched, _, diffs = compareGroundTruth(img, map[string][]byte{"f": dirty})
if matched != 0 || diffs != 1 {
t.Fatalf("non-zero padding must mismatch: matched=%d diffs=%d", matched, diffs)
}
}
+241
View File
@@ -0,0 +1,241 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"os"
"os/exec"
"path/filepath"
"regexp"
"strings"
"testing"
)
// TestManPagesTrackTheCLI builds the binary once, then compares every
// command's live `-h` output with its docs/man/gasm-<command>.1 page: the
// flag sets must agree both ways, and the page's SYNOPSIS line must carry
// the command's usage line. A flag or a usage change that skips the man
// page fails here, so the pages cannot drift from the binary.
func TestManPagesTrackTheCLI(t *testing.T) {
if testing.Short() {
t.Skip("builds the gasm binary")
}
bin := filepath.Join(t.TempDir(), "gasm")
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
t.Fatalf("build gasm: %v\n%s", err, out)
}
for _, cmd := range []string{
"tokens", "parse", "fmt", "lint", "asm", "dis", "verify",
"debug", "diff", "profile", "audit-instructions", "scaffold", "lsp",
} {
t.Run(cmd, func(t *testing.T) {
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm-"+cmd+".1"))
if err != nil {
t.Fatalf("read man page: %v", err)
}
page := string(raw)
out, _ := exec.Command(bin, cmd, "-h").CombinedOutput()
help := string(out)
binFlags := helpFlags(help)
pageFlags := roffFlags(page)
for f := range binFlags {
if !pageFlags[f] {
t.Errorf("flag -%s is in the binary's help but missing from the man page", f)
}
}
for f := range pageFlags {
if !binFlags[f] {
t.Errorf("flag -%s is in the man page but the binary does not accept it", f)
}
}
want := helpUsage(help)
got := roffSynopsis(page)
if want != "" && got != want {
t.Errorf("SYNOPSIS drift:\n page: %s\nbinary: %s", got, want)
}
})
}
}
// TestManCommandsTrackHelp compares the gasm(1) COMMANDS list with the
// top-level help output, so a subcommand added to the binary cannot miss
// its man entry and a stale entry cannot outlive its command.
func TestManCommandsTrackHelp(t *testing.T) {
if testing.Short() {
t.Skip("builds the gasm binary")
}
bin := filepath.Join(t.TempDir(), "gasm")
if out, err := exec.Command("go", "build", "-o", bin, ".").CombinedOutput(); err != nil {
t.Fatalf("build gasm: %v\n%s", err, out)
}
raw, err := os.ReadFile(filepath.Join("..", "..", "docs", "man", "gasm.1"))
if err != nil {
t.Fatalf("read man page: %v", err)
}
helpOut, err := exec.Command(bin, "--help").Output()
if err != nil {
t.Fatalf("gasm --help: %v", err)
}
binCmds := helpCommands(string(helpOut))
pageCmds := roffCommands(string(raw))
for c := range binCmds {
if !pageCmds[c] {
t.Errorf("command %q is in the binary's help but missing from gasm(1) COMMANDS", c)
}
}
for c := range pageCmds {
if !binCmds[c] {
t.Errorf("command %q is in gasm(1) COMMANDS but the binary does not list it", c)
}
}
}
// helpFlags extracts the flag names from a `gasm <cmd> -h` output.
func helpFlags(help string) map[string]bool {
m := map[string]bool{}
inFlags := false
for line := range strings.SplitSeq(help, "\n") {
if strings.TrimRight(line, " \t") == "Flags:" {
inFlags = true
continue
}
if !inFlags {
continue
}
if !strings.HasPrefix(line, " -") {
continue
}
token := strings.FieldsFunc(strings.TrimLeft(line, " "), func(r rune) bool {
return r == ' ' || r == '\t'
})
if len(token) == 0 {
continue
}
m[strings.TrimLeft(token[0], "-")] = true
}
return m
}
var roffEscape = regexp.MustCompile(`\\f[BIRP]`)
// roffFlags extracts the flag names from a man page's OPTIONS section.
func roffFlags(page string) map[string]bool {
m := map[string]bool{}
inOptions := false
for line := range strings.SplitSeq(page, "\n") {
if strings.HasPrefix(line, ".SH ") {
inOptions = strings.HasPrefix(line, ".SH OPTIONS")
continue
}
if !inOptions {
continue
}
// Flag entries are written as either `.B \-flag` or `\fB\-flag`.
var body string
switch {
case strings.HasPrefix(line, `.B \-`):
body = line[3:]
case strings.HasPrefix(line, `\fB\-`):
body = line[1:]
default:
continue
}
name := roffEscape.ReplaceAllString(body, "")
name = strings.ReplaceAll(name, `\-`, "-")
name = strings.TrimSpace(name)
if i := strings.IndexAny(name, " \t"); i >= 0 {
name = name[:i]
}
m[strings.TrimLeft(name, "-")] = true
}
return m
}
// helpCommands extracts the command names from the top-level help output's
// Commands section.
func helpCommands(help string) map[string]bool {
m := map[string]bool{}
inCmds := false
for line := range strings.SplitSeq(help, "\n") {
if strings.TrimSpace(line) == "Commands:" {
inCmds = true
continue
}
if !inCmds {
continue
}
t := strings.TrimSpace(line)
if t == "" {
break
}
name, _, _ := strings.Cut(t, " ")
m[name] = true
}
return m
}
// roffCommands extracts the command names from gasm(1)'s COMMANDS section,
// where each entry is written as `.B gasm\-<name>(1)` or `.B gasm <name>`.
func roffCommands(page string) map[string]bool {
m := map[string]bool{}
inCmds := false
for line := range strings.SplitSeq(page, "\n") {
if strings.HasPrefix(line, ".SH ") {
inCmds = strings.HasPrefix(line, ".SH COMMANDS")
continue
}
if !inCmds || !strings.HasPrefix(line, ".B gasm") {
continue
}
entry := strings.ReplaceAll(strings.TrimPrefix(line, ".B "), `\-`, "-")
entry = strings.TrimSuffix(entry, "(1)")
switch {
case strings.HasPrefix(entry, "gasm-"):
m[strings.TrimPrefix(entry, "gasm-")] = true
case strings.HasPrefix(entry, "gasm "):
m[strings.TrimPrefix(entry, "gasm ")] = true
}
}
return m
}
// helpUsage returns the command's usage line without the "Usage: " prefix.
func helpUsage(help string) string {
for line := range strings.SplitSeq(help, "\n") {
if strings.HasPrefix(line, "Usage: ") {
return normaliseUsage(line[len("Usage: "):])
}
}
return ""
}
// roffSynopsis returns the page's SYNOPSIS usage line, unescaped.
func roffSynopsis(page string) string {
inSyn := false
for line := range strings.SplitSeq(page, "\n") {
if strings.HasPrefix(line, ".SH ") {
inSyn = strings.HasPrefix(line, ".SH SYNOPSIS")
continue
}
if !inSyn || !strings.HasPrefix(line, ".B ") {
continue
}
return normaliseUsage(strings.ReplaceAll(line[3:], `\-`, "-"))
}
return ""
}
// normaliseUsage flattens whitespace and drops the roff font escapes so that
// the binary's usage line and the page's SYNOPSIS line compare equal.
func normaliseUsage(s string) string {
s = roffEscape.ReplaceAllString(s, "")
return strings.Join(strings.Fields(s), " ")
}
+109
View File
@@ -0,0 +1,109 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"os"
"path/filepath"
"strings"
"testing"
)
// writeTree writes a directory of files and returns its root.
func writeTree(t *testing.T, files map[string]string) string {
t.Helper()
dir := t.TempDir()
for name, content := range files {
path := filepath.Join(dir, name)
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
}
return dir
}
// TestAsmMacroAndIncludeEndToEnd drives `gasm asm` over a source with an
// in-file parameterised macro and an include resolved through -I, and checks
// the assembled bytes came from the expansion (the loop body counts six
// increments, two per expanded iteration).
func TestAsmMacroAndIncludeEndToEnd(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
dir := writeTree(t, map[string]string{
"inc/consts.h": "#define NITER 3\n",
"main_amd64.s": "#include \"textflag.h\"\n" +
"#include \"consts.h\"\n" +
"#define STEP(r) ADDQ $1, r; ADDQ $1, r\n" +
"TEXT ·f(SB), NOSPLIT, $0-8\n" +
"\tXORQ AX, AX\n" +
"\tMOVQ $NITER, CX\n" +
"loop:\n" +
"\tSTEP(AX)\n" +
"\tDECQ CX\n" +
"\tJNZ loop\n" +
"\tMOVQ AX, ret+0(FP)\n" +
"\tRET\n",
})
stdout, stderr, code := capture(func() int {
return cmdAsm([]string{"-I", filepath.Join(dir, "inc"), "-GOARCH", "amd64", filepath.Join(dir, "main_amd64.s")})
})
if code != 0 {
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
}
// The macro expanded to two ADDQ $1 encodings in the static body; the
// iteration count lives in the runtime loop.
if n := strings.Count(stdout, "83 c0 01"); n != 2 {
t.Errorf("found %d ADDQ $1 encodings in the image, want 2:\n%s", n, stdout)
}
}
// TestAsmIncludeResolutionOrder pins the -I search order end to end: the
// including file's directory wins over the -I directories.
func TestAsmIncludeResolutionOrder(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
dir := writeTree(t, map[string]string{
"src/main_amd64.s": "#include \"textflag.h\"\n" +
"#include \"vals.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tMOVQ $VAL, AX\n" +
"\tRET\n",
"src/vals.h": "#define VAL 1\n",
"late/vals.h": "#define VAL 2\n",
"early/vals.h": "#define VAL 3\n",
})
stdout, stderr, code := capture(func() int {
return cmdAsm([]string{"-I", filepath.Join(dir, "early"), "-I", filepath.Join(dir, "late"),
"-GOARCH", "amd64", filepath.Join(dir, "src", "main_amd64.s")})
})
if code != 0 {
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
}
// VAL came from src/vals.h, not from either -I directory: the image
// loads the immediate 1.
if !strings.Contains(stdout, "b8 01 00 00 00") {
t.Errorf("expected the source-directory VAL (immediate 1) in:\n%s", stdout)
}
}
// TestAsmMissingIncludeIsAnError pins the diagnostic for an include that
// resolves nowhere on the assembly path.
func TestAsmMissingIncludeIsAnError(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
path := writeTemp(t, "main_amd64.s", "#include \"textflag.h\"\n#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
_, stderr, code := capture(func() int { return cmdAsm([]string{"-GOARCH", "amd64", path}) })
if code == 0 {
t.Fatal("gasm asm accepted a file whose include resolves nowhere")
}
if !strings.Contains(stderr, `#include "nothere.h"`) {
t.Errorf("stderr does not name the failing include: %s", stderr)
}
}
+2 -2
View File
@@ -19,7 +19,7 @@ import (
// file: a Go test that seeds random states, drives both the assembly kernel
// and a caller-provided portable reference, and compares the outputs
// byte-for-byte. The lesson this encodes: a pipeline-level fuzz cannot see
// an unwired kernel — only a direct-call differential against the portable
// an unwired kernel, only a direct-call differential against the portable
// specification can, so every kernel ships with one.
//
// The generated file follows two conventions the caller fills in:
@@ -44,7 +44,7 @@ bodies, place the file in the kernel's package, and run it in CI.
rest = rest[1:]
}
if len(rest) != 1 {
return fmt.Errorf("usage: gasm scaffold differential <file.s>")
return &usageError{fmt.Errorf("usage: gasm scaffold differential <file.s>")}
}
path := rest[0]
src, err := os.ReadFile(path)
+140
View File
@@ -0,0 +1,140 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"fmt"
"slices"
"strings"
)
// unifiedDiff renders a unified diff with three lines of context between the
// two line slices, in the form `gofmt -d` prints. An empty result means the
// inputs are identical.
func unifiedDiff(name string, a, b []string) string {
if slices.Equal(a, b) {
return ""
}
var out strings.Builder
fmt.Fprintf(&out, "--- %s\n+++ %s\n", name, name)
// Longest common subsequence over the lines (assembly files are small
// enough for the quadratic table).
n, m := len(a), len(b)
lcs := make([][]int, n+1)
for i := range lcs {
lcs[i] = make([]int, m+1)
}
for i := n - 1; i >= 0; i-- {
for j := m - 1; j >= 0; j-- {
if a[i] == b[j] {
lcs[i][j] = lcs[i+1][j+1] + 1
} else if lcs[i+1][j] >= lcs[i][j+1] {
lcs[i][j] = lcs[i+1][j]
} else {
lcs[i][j] = lcs[i][j+1]
}
}
}
// Walk the LCS once, assigning every op its absolute position in both
// files (1-based, the position an insertion sits before).
type op struct {
kind byte // ' ', '-' or '+'
aLine, bLine int
text string
}
var ops []op
aPos, bPos := 0, 0
emit := func(kind byte, text string) {
ops = append(ops, op{kind: kind, aLine: aPos + 1, bLine: bPos + 1, text: text})
switch kind {
case ' ':
aPos++
bPos++
case '-':
aPos++
case '+':
bPos++
}
}
i, j := 0, 0
for i < n && j < m {
switch {
case a[i] == b[j]:
emit(' ', a[i])
i++
j++
case lcs[i+1][j] >= lcs[i][j+1]:
emit('-', a[i])
i++
default:
emit('+', b[j])
j++
}
}
for ; i < n; i++ {
emit('-', a[i])
}
for ; j < m; j++ {
emit('+', b[j])
}
// Group the edits into hunks: consecutive changes separated by more than
// twice the context lines start a new hunk.
const context = 3
var changes []int
for k, o := range ops {
if o.kind != ' ' {
changes = append(changes, k)
}
}
for g := 0; g < len(changes); {
last := g
for last+1 < len(changes) && changes[last+1]-changes[last]-1 <= 2*context {
last++
}
lo := max(0, changes[g]-context)
hi := min(len(ops), changes[last]+1+context)
// The header numbers are the first line of each side actually shown:
// the first context, deletion or insertion line. A hunk that shows
// no old lines is a pure insertion and reports the position it sits
// before (0 at the top of the file); the mirror rule holds for a
// pure deletion.
aStart := ops[lo].aLine - 1
bStart := ops[lo].bLine - 1
countA, countB := 0, 0
for _, o := range ops[lo:hi] {
switch o.kind {
case ' ':
countA++
countB++
case '-':
countA++
case '+':
countB++
}
}
for _, o := range ops[lo:hi] {
if o.kind != '+' {
aStart = o.aLine
break
}
}
for _, o := range ops[lo:hi] {
if o.kind != '-' {
bStart = o.bLine
break
}
}
fmt.Fprintf(&out, "@@ -%d,%d +%d,%d @@\n", aStart, countA, bStart, countB)
for _, o := range ops[lo:hi] {
out.WriteByte(o.kind)
out.WriteString(o.text)
out.WriteByte('\n')
}
g = last + 1
}
return out.String()
}
+94
View File
@@ -0,0 +1,94 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"slices"
"strings"
"testing"
)
func lines(ss ...string) []string { return ss }
func TestUnifiedDiffIdentical(t *testing.T) {
if got := unifiedDiff("f", lines("a", "b"), lines("a", "b")); got != "" {
t.Errorf("identical inputs produced %q, want empty", got)
}
}
func TestUnifiedDiffSingleChange(t *testing.T) {
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
b := lines("1", "2", "3!", "4", "5", "6", "7", "8")
want := "--- f\n+++ f\n" +
"@@ -1,6 +1,6 @@\n" +
" 1\n 2\n-3\n+3!\n 4\n 5\n 6\n"
if got := unifiedDiff("f", a, b); got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffInsertAtStart(t *testing.T) {
got := unifiedDiff("f", lines("x"), lines("new", "x"))
// The single existing line is shown as trailing context, so the hunk
// covers it.
want := "--- f\n+++ f\n@@ -1,1 +1,2 @@\n+new\n x\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffDeleteAtEnd(t *testing.T) {
got := unifiedDiff("f", lines("x", "y"), lines("x"))
want := "--- f\n+++ f\n@@ -1,2 +1,1 @@\n x\n-y\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func TestUnifiedDiffTwoHunks(t *testing.T) {
var a, b []string
for i := 1; i <= 20; i++ {
a = append(a, itoa(i))
b = append(b, itoa(i))
}
b[1] = "2!"
b[17] = "18!"
got := unifiedDiff("f", a, b)
if !strings.Contains(got, "@@ -1,5 +1,5 @@\n 1\n-2\n+2!\n 3\n 4\n 5\n") {
t.Errorf("first hunk wrong:\n%s", got)
}
if !strings.Contains(got, "@@ -15,6 +15,6 @@\n 15\n 16\n 17\n-18\n+18!\n 19\n 20\n") {
t.Errorf("second hunk wrong:\n%s", got)
}
}
// TestUnifiedDiffAdjacentHunks merges changes separated by exactly twice the
// context into one hunk.
func TestUnifiedDiffAdjacentHunks(t *testing.T) {
a := lines("1", "2", "3", "4", "5", "6", "7", "8")
b := slices.Clone(a)
b[0] = "1!"
b[7] = "8!"
got := unifiedDiff("f", a, b)
want := "--- f\n+++ f\n" +
"@@ -1,8 +1,8 @@\n" +
"-1\n+1!\n 2\n 3\n 4\n 5\n 6\n 7\n-8\n+8!\n"
if got != want {
t.Errorf("diff = %q, want %q", got, want)
}
}
func itoa(n int) string {
if n == 0 {
return "0"
}
var buf [4]byte
i := len(buf)
for n > 0 {
i--
buf[i] = byte('0' + n%10)
n /= 10
}
return string(buf[i:])
}
+101 -47
View File
@@ -9,11 +9,11 @@ import "strings"
import "fmt"
// Breakpoint is one INT3 breakpoint in the debuggee.
// Breakpoint is one software breakpoint in the debuggee.
type Breakpoint struct {
Addr uint64 // absolute address in the debuggee
Label string // source label ("" for raw addresses)
Orig byte // original byte at Addr (restored on removal)
Orig []byte // original bytes at Addr (restored on removal)
Enabled bool
Cond *Condition // optional condition (nil = unconditional)
hits int
@@ -32,11 +32,15 @@ type Condition struct {
MemAddr uint64 // memory address (for register-memory comparison, prefixed with *)
}
// Eval checks the condition against the current registers.
func (c *Condition) Eval(regs *Regs) bool {
// Eval checks the condition against the current registers. For the
// register-memory form, mem reads an 8-byte little-endian word from the
// debuggee; it may be nil when no reader is available. Anything that cannot
// be decided (unknown register or operator, unreadable memory) does not
// block the breakpoint.
func (c *Condition) Eval(regs *Regs, mem func(addr uint64) (uint64, bool)) bool {
actual, ok := regs.RegValue(c.Reg)
if !ok {
return true // unknown register — don't block
return true // unknown register, don't block
}
var expected uint64
switch {
@@ -48,9 +52,16 @@ func (c *Condition) Eval(regs *Regs) bool {
}
expected = v
case c.MemAddr != 0:
// Register-memory comparison — requires a Session, not available here.
// Fall back to treating as constant (the caller should resolve).
expected = c.Value
// Register-memory comparison, resolved in the debuggee at
// evaluation time.
if mem == nil {
return true
}
v, ok := mem(c.MemAddr)
if !ok {
return true
}
expected = v
default:
expected = c.Value
}
@@ -72,8 +83,19 @@ func (c *Condition) Eval(regs *Regs) bool {
}
}
// Breakpoints manages the set of breakpoints for a Session.
// Breakpoints manages software breakpoints for a debuggee.
// String renders the condition for display.
func (c *Condition) String() string {
switch {
case c.Reg2 != "":
return fmt.Sprintf("%s %s %s", c.Reg, c.Op, c.Reg2)
case c.MemAddr != 0:
return fmt.Sprintf("%s %s *%#x", c.Reg, c.Op, c.MemAddr)
default:
return fmt.Sprintf("%s %s %#x", c.Reg, c.Op, c.Value)
}
}
// Breakpoints manages the software breakpoints of one Session.
type Breakpoints struct {
t tracer
bps map[uint64]*Breakpoint
@@ -84,6 +106,18 @@ func NewBreakpoints(t tracer) *Breakpoints {
return &Breakpoints{t: t, bps: make(map[uint64]*Breakpoint)}
}
// breakpointMask is the byte mask of the breakpoint instruction inside a
// peeked word: the low len(breakpointInsn) bytes, because every supported
// architecture is little-endian and patches the instruction at the lowest
// address of the word.
func breakpointMask() uint64 {
var mask uint64
for range breakpointInsn {
mask = (mask << 8) | 0xFF
}
return mask
}
// Set installs a breakpoint at addr (replaces any existing one).
func (bm *Breakpoints) Set(addr uint64, label string) (*Breakpoint, error) {
return bm.SetWithCond(addr, label, nil)
@@ -101,13 +135,12 @@ func (bm *Breakpoints) SetWithCond(addr uint64, label string, cond *Condition) (
if err != nil {
return nil, err
}
orig := byte(word)
// Patch with the breakpoint instruction, preserving the rest of the word.
mask := uint64(0)
for range breakpointInsn {
mask = (mask << 8) | 0xFF
orig := make([]byte, len(breakpointInsn))
for i := range orig {
orig[i] = byte(word >> (8 * i))
}
patched := (word &^ mask) | breakpointWord(breakpointInsn)
// Patch with the breakpoint instruction, preserving the rest of the word.
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
if err := bm.t.Poke(addr, patched); err != nil {
return nil, err
}
@@ -135,26 +168,40 @@ func (bm *Breakpoints) Info() string {
}
cond := ""
if bp.Cond != nil {
cond = fmt.Sprintf(" if %s %s %#x", bp.Cond.Reg, bp.Cond.Op, bp.Cond.Value)
cond = " if " + bp.Cond.String()
}
result.WriteString(fmt.Sprintf(" %d: %s at %#x [%s, %d hits]%s\n", i, label, bp.Addr, status, bp.hits, cond))
}
return result.String()
}
// Clear removes the breakpoint at addr, restoring the original byte.
// restore writes the saved original bytes back over the breakpoint
// instruction, preserving the rest of the peeked word. It reports whether
// both the peek and the poke succeeded.
func (bm *Breakpoints) restore(addr uint64, bp *Breakpoint) bool {
word, err := bm.t.Peek(addr)
if err != nil {
return false
}
orig := uint64(0)
for i, b := range bp.Orig {
orig |= uint64(b) << (8 * i)
}
return bm.t.Poke(addr, (word&^breakpointMask())|orig) == nil
}
// Clear removes the breakpoint at addr, restoring the original bytes.
func (bm *Breakpoints) Clear(addr uint64) error {
bp, ok := bm.bps[addr]
if !ok {
return fmt.Errorf("debug: no breakpoint at %#x", addr)
}
word, err := bm.t.Peek(addr)
if err != nil {
return err
}
restored := (word &^ 0xFF) | uint64(bp.Orig)
if err := bm.t.Poke(addr, restored); err != nil {
return err
if !bm.restore(addr, bp) {
word, err := bm.t.Peek(addr)
if err != nil {
return err
}
return fmt.Errorf("debug: restore breakpoint at %#x failed, word is %#x", addr, word)
}
delete(bm.bps, addr)
return nil
@@ -186,43 +233,54 @@ func (bm *Breakpoints) All() []*Breakpoint {
// HandleTrap is called after the debuggee stops on SIGTRAP. It checks
// whether the trap was caused by one of our breakpoints (PC-adjust matches
// a breakpoint address), restores the original byte, rewinds PC, and
// a breakpoint address), restores the original bytes, rewinds PC, and
// returns the breakpoint that was hit (or nil if it was a single-step).
// Hits returns how many times the breakpoint has been hit.
func (bp *Breakpoint) Hits() int { return bp.hits }
func (bm *Breakpoints) HandleTrap(regs *Regs) *Breakpoint {
// After a breakpoint trap, PC points past the breakpoint instruction.
// On amd64 the kernel reports the trap with RIP past the INT3; on the
// other supported architectures the PC still stands on the trap
// instruction, which breakpointPCAdjust encodes per architecture.
trapAddr := regs.GetPC() - uint64(breakpointPCAdjust)
bp, ok := bm.bps[trapAddr]
if !ok || !bp.Enabled {
return nil // single-step trap or unknown
}
// Check the condition (if any).
if bp.Cond != nil && !bp.Cond.Eval(regs) {
// Condition not met — restore the byte but do NOT rewind RIP.
// The process continues from the next instruction (past the INT3).
word, err := bm.t.Peek(trapAddr)
if err == nil {
restored := (word &^ 0xFF) | uint64(bp.Orig)
bm.t.Poke(trapAddr, restored)
if bp.Cond != nil && !bp.Cond.Eval(regs, bm.peekValue) {
// Condition not met: step the original instruction and re-arm the
// breakpoint, leaving the debuggee stopped just past it, ready to
// resume silently. The PC must be rewound first: on architectures
// that report the trap past the instruction (amd64) it would
// otherwise sit on the second byte of the replaced instruction.
if !bm.restore(trapAddr, bp) {
return nil
}
// RIP is already past the INT3 (trapAddr + 1). Don't rewind.
regs.SetPC(trapAddr)
if err := bm.t.SetRegs(regs); err != nil {
return nil
}
if err := bm.t.Step(); err != nil {
return nil
}
bm.Reinsert(trapAddr)
return nil
}
bp.hits++
// Restore the original byte.
word, err := bm.t.Peek(trapAddr)
if err == nil {
restored := (word &^ 0xFF) | uint64(bp.Orig)
bm.t.Poke(trapAddr, restored)
}
// Rewind PC to re-execute the original instruction.
// Restore the original bytes and rewind PC to re-execute them.
bm.restore(trapAddr, bp)
regs.SetPC(trapAddr)
bm.t.SetRegs(regs)
return bp
}
// peekValue adapts tracer.Peek to the Condition value reader.
func (bm *Breakpoints) peekValue(addr uint64) (uint64, bool) {
v, err := bm.t.Peek(addr)
return v, err == nil
}
// Reinsert re-inserts the breakpoint at addr after a single-step past it.
// Called after Step() when we want the breakpoint to fire again on the
// next Continue().
@@ -235,11 +293,7 @@ func (bm *Breakpoints) Reinsert(addr uint64) error {
if err != nil {
return err
}
mask := uint64(0)
for range breakpointInsn {
mask = (mask << 8) | 0xFF
}
patched := (word &^ mask) | breakpointWord(breakpointInsn)
patched := (word &^ breakpointMask()) | breakpointWord(breakpointInsn)
return bm.t.Poke(addr, patched)
}
+265
View File
@@ -0,0 +1,265 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux
package debug
// Architecture-neutral tests: label and line tables, and the breakpoint
// manager against the mock tracer. These do not launch a debuggee, so they
// build on every supported linux architecture.
import (
"strings"
"testing"
)
func TestLineAt(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
{Offset: 15, Line: 8},
}
tests := []struct {
offset int
want int
}{
{0, 5},
{1, 5},
{4, 5},
{5, 6},
{7, 6},
{10, 7},
{12, 7},
{15, 8},
{20, 8},
}
for _, tt := range tests {
got := lineAt(lines, tt.offset)
if got != tt.want {
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
}
}
// Empty table.
if lineAt(nil, 5) != 0 {
t.Error("lineAt(nil, 5) should return 0")
}
}
func TestOffsetForLine(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
}
tests := []struct {
line int
want int
}{
{5, 0},
{6, 5},
{7, 10},
{99, -1}, // not found
{0, -1}, // not found
}
for _, tt := range tests {
got := offsetForLine(lines, tt.line)
if got != tt.want {
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
}
}
}
func TestNearestLabel(t *testing.T) {
labels := []Label{
{Name: "start", Offset: 0},
{Name: "loop", Offset: 10},
{Name: "done", Offset: 20},
}
tests := []struct {
offset int
want string
}{
{0, "start"},
{5, "start"},
{10, "loop"},
{15, "loop"},
{20, "done"},
{25, "done"},
}
for _, tt := range tests {
got := nearestLabel(labels, tt.offset)
if got != tt.want {
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
}
}
}
func TestBreakpointsSetAndClear(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
// Set a breakpoint at address 0x1000.
bp, err := bm.Set(0x1000, "test")
if err != nil {
t.Fatalf("Set: %v", err)
}
if !bp.Enabled {
t.Error("breakpoint not enabled")
}
if bp.Label != "test" {
t.Errorf("label = %q, want test", bp.Label)
}
// Verify Peek was called.
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
}
// Verify Poke wrote the breakpoint instruction's bytes.
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
t.Errorf("pokes = %v", tr.pokes)
}
if got := tr.pokes[0].val & breakpointMask(); got != breakpointWord(breakpointInsn) {
t.Errorf("patched bytes %#x, want %#x", got, breakpointWord(breakpointInsn))
}
// At should find it.
if bm.At(0x1000) == nil {
t.Error("At(0x1000) returned nil")
}
// All should return it.
all := bm.All()
if len(all) != 1 {
t.Errorf("All() = %d breakpoints, want 1", len(all))
}
// Clear it.
if err := bm.Clear(0x1000); err != nil {
t.Fatalf("Clear: %v", err)
}
if bm.At(0x1000) != nil {
t.Error("At(0x1000) after Clear should be nil")
}
}
// TestBreakpointRestoreWidth proves the restore path writes back every
// byte of the breakpoint instruction's width, not just the first byte: on
// arm64, riscv64 and loong64 the instruction is four bytes, and restoring
// one byte would leave three bytes of the trap instruction in place.
func TestBreakpointRestoreWidth(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
tr.mem[0x3000] = 0x11
tr.mem[0x3001] = 0x22
tr.mem[0x3002] = 0x33
tr.mem[0x3003] = 0x44
if _, err := bm.Set(0x3000, "width"); err != nil {
t.Fatalf("Set: %v", err)
}
for i, b := range breakpointInsn {
if tr.mem[0x3000+uint64(i)] != b {
t.Fatalf("byte %d after Set = %#x, want the breakpoint byte %#x", i, tr.mem[0x3000+uint64(i)], b)
}
}
if len(bm.At(0x3000).Orig) != len(breakpointInsn) {
t.Fatalf("Orig holds %d bytes, want %d", len(bm.At(0x3000).Orig), len(breakpointInsn))
}
if err := bm.Clear(0x3000); err != nil {
t.Fatalf("Clear: %v", err)
}
want := []byte{0x11, 0x22, 0x33, 0x44}
for i, b := range want {
if tr.mem[0x3000+uint64(i)] != b {
t.Errorf("byte %d after Clear = %#x, want %#x (restore must cover the full instruction width)", i, tr.mem[0x3000+uint64(i)], b)
}
}
}
func TestBreakpointsSetWithCond(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
if err != nil {
t.Fatalf("SetWithCond: %v", err)
}
if bp.Cond == nil || bp.Cond.Value != 42 {
t.Error("condition not set")
}
// Re-setting the same address should update the condition.
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
if err != nil {
t.Fatalf("SetWithCond (update): %v", err)
}
if bp2.Cond.Value != 100 {
t.Error("condition not updated")
}
// Should have only 1 Peek (first Set), second is update (no Peek needed).
if len(tr.peeks) != 1 {
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
}
}
func TestBreakpointsClearAll(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x1000, "a")
bm.Set(0x2000, "b")
bm.Set(0x3000, "c")
if len(bm.All()) != 3 {
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
}
bm.ClearAll()
if len(bm.All()) != 0 {
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
}
}
func TestBreakpointInfo(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x4000, "info_test")
info := bm.Info()
if info == "" {
t.Error("Info returned empty string")
}
if !strings.Contains(info, "info_test") {
t.Errorf("Info %q does not contain label", info)
}
}
// TestConditionString covers the display of all three condition forms.
func TestConditionString(t *testing.T) {
tests := []struct {
cond Condition
want string
}{
{Condition{Reg: "rax", Op: "==", Value: 42}, "rax == 0x2a"},
{Condition{Reg: "rax", Op: "!=", Reg2: "rbx"}, "rax != rbx"},
{Condition{Reg: "rax", Op: "<", MemAddr: 0x5000}, "rax < *0x5000"},
}
for _, tt := range tests {
if got := tt.cond.String(); got != tt.want {
t.Errorf("Condition.String() = %q, want %q", got, tt.want)
}
}
}
+42 -193
View File
@@ -6,7 +6,6 @@
package debug
import (
"strings"
"testing"
)
@@ -48,7 +47,7 @@ func TestConditionEval(t *testing.T) {
}
for _, tt := range tests {
got := tt.cond.Eval(regs)
got := tt.cond.Eval(regs, nil)
if got != tt.want {
t.Errorf("Condition{%q %q %d}.Eval() = %v, want %v",
tt.cond.Reg, tt.cond.Op, tt.cond.Value, got, tt.want)
@@ -56,65 +55,33 @@ func TestConditionEval(t *testing.T) {
}
}
func TestLineAt(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
{Offset: 15, Line: 8},
}
tests := []struct {
offset int
want int
}{
{0, 5},
{1, 5},
{4, 5},
{5, 6},
{7, 6},
{10, 7},
{12, 7},
{15, 8},
{20, 8},
}
for _, tt := range tests {
got := lineAt(lines, tt.offset)
if got != tt.want {
t.Errorf("lineAt(lines, %d) = %d, want %d", tt.offset, got, tt.want)
// TestConditionEvalMem covers the register-memory form: the value is read
// through the supplied reader, and a missing or failing reader must not
// block the breakpoint.
func TestConditionEvalMem(t *testing.T) {
regs := &Regs{RAX: 7}
mem := func(addr uint64) (uint64, bool) {
if addr == 0x5000 {
return 7, true
}
return 0, false
}
// Empty table.
if lineAt(nil, 5) != 0 {
t.Error("lineAt(nil, 5) should return 0")
eq := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
if !eq.Eval(regs, mem) {
t.Error("register-memory comparison with matching word should hold")
}
}
func TestOffsetForLine(t *testing.T) {
lines := []SourceLine{
{Offset: 0, Line: 5},
{Offset: 5, Line: 6},
{Offset: 10, Line: 7},
ne := Condition{Reg: "rax", Op: "!=", MemAddr: 0x5000}
if ne.Eval(regs, mem) {
t.Error("register-memory comparison with mismatching word should not hold")
}
tests := []struct {
line int
want int
}{
{5, 0},
{6, 5},
{7, 10},
{99, -1}, // not found
{0, -1}, // not found
bad := Condition{Reg: "rax", Op: "==", MemAddr: 0x6000}
if !bad.Eval(regs, mem) {
t.Error("unreadable memory must not block the breakpoint")
}
for _, tt := range tests {
got := offsetForLine(lines, tt.line)
if got != tt.want {
t.Errorf("offsetForLine(lines, %d) = %d, want %d", tt.line, got, tt.want)
}
noReader := Condition{Reg: "rax", Op: "==", MemAddr: 0x5000}
if !noReader.Eval(regs, nil) {
t.Error("missing memory reader must not block the breakpoint")
}
}
@@ -141,142 +108,8 @@ func TestDecodeRflags(t *testing.T) {
}
}
func TestNearestLabel(t *testing.T) {
labels := []Label{
{Name: "start", Offset: 0},
{Name: "loop", Offset: 10},
{Name: "done", Offset: 20},
}
tests := []struct {
offset int
want string
}{
{0, "start"},
{5, "start"},
{10, "loop"},
{15, "loop"},
{20, "done"},
{25, "done"},
}
for _, tt := range tests {
got := nearestLabel(labels, tt.offset)
if got != tt.want {
t.Errorf("nearestLabel(labels, %d) = %q, want %q", tt.offset, got, tt.want)
}
}
}
func TestBreakpointsSetAndClear(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
// Set a breakpoint at address 0x1000.
bp, err := bm.Set(0x1000, "test")
if err != nil {
t.Fatalf("Set: %v", err)
}
if !bp.Enabled {
t.Error("breakpoint not enabled")
}
if bp.Label != "test" {
t.Errorf("label = %q, want test", bp.Label)
}
// Verify Peek was called.
if len(tr.peeks) != 1 || tr.peeks[0] != 0x1000 {
t.Errorf("peeks = %v, want [0x1000]", tr.peeks)
}
// Verify Poke wrote INT3.
if len(tr.pokes) != 1 || tr.pokes[0].addr != 0x1000 {
t.Errorf("pokes = %v", tr.pokes)
}
// At should find it.
if bm.At(0x1000) == nil {
t.Error("At(0x1000) returned nil")
}
// All should return it.
all := bm.All()
if len(all) != 1 {
t.Errorf("All() = %d breakpoints, want 1", len(all))
}
// Clear it.
if err := bm.Clear(0x1000); err != nil {
t.Fatalf("Clear: %v", err)
}
if bm.At(0x1000) != nil {
t.Error("At(0x1000) after Clear should be nil")
}
}
func TestBreakpointsSetWithCond(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
cond := &Condition{Reg: "rax", Op: "==", Value: 42}
bp, err := bm.SetWithCond(0x2000, "cond_test", cond)
if err != nil {
t.Fatalf("SetWithCond: %v", err)
}
if bp.Cond == nil || bp.Cond.Value != 42 {
t.Error("condition not set")
}
// Re-setting the same address should update the condition.
cond2 := &Condition{Reg: "rbx", Op: "<", Value: 100}
bp2, err := bm.SetWithCond(0x2000, "cond_test2", cond2)
if err != nil {
t.Fatalf("SetWithCond (update): %v", err)
}
if bp2.Cond.Value != 100 {
t.Error("condition not updated")
}
// Should have only 1 Peek (first Set), second is update (no Peek needed).
if len(tr.peeks) != 1 {
t.Errorf("expected 1 Peek, got %d", len(tr.peeks))
}
}
func TestBreakpointsClearAll(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x1000, "a")
bm.Set(0x2000, "b")
bm.Set(0x3000, "c")
if len(bm.All()) != 3 {
t.Fatalf("expected 3 breakpoints, got %d", len(bm.All()))
}
bm.ClearAll()
if len(bm.All()) != 0 {
t.Errorf("ClearAll: expected 0 breakpoints, got %d", len(bm.All()))
}
}
func TestBreakpointInfo(t *testing.T) {
tr := newMockTracer()
bm := NewBreakpoints(tr)
bm.Set(0x4000, "info_test")
info := bm.Info()
if info == "" {
t.Error("Info returned empty string")
}
if !strings.Contains(info, "info_test") {
t.Errorf("Info %q does not contain label", info)
}
}
func TestWatchpointSlotTracking(t *testing.T) {
wpSlots = [4]bool{} // reset
s := &Session{}
s := &Session{} // per-session slots start free
// All four slots are free initially.
for i := range 4 {
@@ -289,8 +122,8 @@ func TestWatchpointSlotTracking(t *testing.T) {
}
// Manually mark slots 0 and 2 as used (simulating successful SetWatchpoint).
wpSlots[0] = true
wpSlots[2] = true
s.wpSlots[0] = true
s.wpSlots[2] = true
if !s.IsWatchpointSlotUsed(0) {
t.Error("slot 0 should be in use")
@@ -318,9 +151,25 @@ func TestWatchpointSlotTracking(t *testing.T) {
// Mark all slots used: FindFreeWatchpointSlot returns -1.
for i := range 4 {
wpSlots[i] = true
s.wpSlots[i] = true
}
if got := s.FindFreeWatchpointSlot(); got != -1 {
t.Errorf("FindFreeWatchpointSlot() with all slots used = %d, want -1", got)
}
}
// TestUnwatchSlotBound checks the bound the REPL parses against: it must
// cover the architecture's whole slot range, not a hardcoded 0-3.
func TestUnwatchSlotBound(t *testing.T) {
max := maxWatchpoints()
if max < 4 {
t.Fatalf("maxWatchpoints() = %d, want at least 4", max)
}
s := &Session{}
if s.IsWatchpointSlotUsed(max - 1) {
t.Errorf("slot %d should be free initially", max-1)
}
if s.IsWatchpointSlotUsed(max) {
t.Errorf("slot %d must be out of range", max)
}
}
+14 -11
View File
@@ -9,27 +9,22 @@ import (
"fmt"
"strings"
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
// memory and returns its text representation and length in bytes.
func (s *Session) Disassemble(addr uint64) (string, int, error) {
// Read up to 15 bytes (max x86 instruction length).
mem, err := s.ReadMemory(addr, 15)
if err != nil {
// Try a shorter read if we're near a page boundary.
mem, err = s.ReadMemory(addr, 1)
if err != nil {
return "", 0, err
}
return "", 0, err
}
inst, err := x86asm.Decode(mem, 64)
ins, err := disasm.Decode(arch.AMD64, mem, addr)
if err != nil {
return "???", 1, nil
return "", 0, err
}
text := x86asm.IntelSyntax(inst, addr, nil)
return text, inst.Len, nil
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr and returns
@@ -51,3 +46,11 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
}
return result.String()
}
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
// call. The first token must match exactly: a prefix test would also catch
// unrelated mnemonics.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
return strings.ToLower(m) == "call"
}
+18 -5
View File
@@ -7,8 +7,10 @@ package debug
import (
"fmt"
"strings"
"golang.org/x/arch/arm64/arm64asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
@@ -18,12 +20,11 @@ func (s *Session) Disassemble(addr uint64) (string, int, error) {
if err != nil {
return "", 0, err
}
inst, err := arm64asm.Decode(mem)
ins, err := disasm.Decode(arch.ARM64, mem, addr)
if err != nil {
return "???", 4, nil
return "", 0, err
}
text := arm64asm.GoSyntax(inst, addr, nil, nil)
return text, 4, nil
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
@@ -44,3 +45,15 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
}
return result
}
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
// The first token must match exactly so branches never match.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl":
return true
}
return false
}
+19 -5
View File
@@ -7,8 +7,10 @@ package debug
import (
"fmt"
"strings"
"golang.org/x/arch/loong64/loong64asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
@@ -18,12 +20,11 @@ func (s *Session) Disassemble(addr uint64) (string, int, error) {
if err != nil {
return "", 0, err
}
inst, err := loong64asm.Decode(mem)
ins, err := disasm.Decode(arch.LOONG64, mem, addr)
if err != nil {
return "???", 4, nil
return "", 0, err
}
text := loong64asm.GoSyntax(inst, addr, nil)
return text, 4, nil
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
@@ -44,3 +45,16 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
}
return result
}
// isCallInsn reports whether disassembled text (loong64asm.GoSyntax) is a
// call. GoSyntax renders bl and jirl calls as CALL (jirl returns print
// RET); the native mnemonics are accepted too. The first token must match
// exactly: a "bl" prefix would catch bltz and other branches.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "bl", "jirl":
return true
}
return false
}
+20 -5
View File
@@ -7,8 +7,10 @@ package debug
import (
"fmt"
"strings"
"golang.org/x/arch/riscv64/riscv64asm"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
)
// Disassemble decodes the instruction at the given address in the debuggee's
@@ -18,12 +20,11 @@ func (s *Session) Disassemble(addr uint64) (string, int, error) {
if err != nil {
return "", 0, err
}
inst, err := riscv64asm.Decode(mem)
ins, err := disasm.Decode(arch.RISCV, mem, addr)
if err != nil {
return "???", 4, nil
return "", 0, err
}
text := riscv64asm.GoSyntax(inst, addr, nil, nil)
return text, inst.Len, nil
return ins.Text, ins.Len, nil
}
// DisassembleN decodes up to n instructions starting at addr.
@@ -44,3 +45,17 @@ func (s *Session) DisassembleN(addr uint64, n int) string {
}
return result
}
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
// are accepted too. The first token must match exactly: a prefix test on
// "bl" would catch branches on other architectures, and jalr as ret prints
// RET, which must not be stepped over.
func isCallInsn(text string) bool {
m, _, _ := strings.Cut(text, " ")
switch strings.ToLower(m) {
case "call", "jal", "jalr":
return true
}
return false
}
+22 -2
View File
@@ -73,9 +73,29 @@ func decodeRflags(f uint64) string {
return flags[:len(flags)-1]
}
// archReturnAddr reads the return address from the stack (amd64 ABI0 convention).
// archReturnAddr reads the return address of the current frame (amd64
// ABI0 convention). A function that contains a CALL (or has a frame) is
// assembled with the prologue PUSHQ BP; MOVQ SP, BP, so mid-function the
// word at SP is the saved caller BP, a stack address, and the return
// address sits further up. Walk the stack from SP and take the first word
// that lies in an executable mapping: stack and data words never do, a
// return address always does.
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
return s.Peek(regs.GetSP())
ranges := execRanges(s.pid)
for off := uint64(0); off < 512; off += 8 {
word, err := s.Peek(regs.RSP + off)
if err != nil {
break
}
for _, r := range ranges {
if word >= r.lo && word < r.hi {
return word, nil
}
}
}
// No mapping available or nothing code-like on the stack: fall back to
// the raw entry convention, [SP] before any push.
return s.Peek(regs.RSP)
}
// archSPLabel returns the SP register name for display.
+6 -3
View File
@@ -5,7 +5,10 @@
package debug
import "fmt"
import (
"encoding/binary"
"fmt"
)
func printRegs(regs *Regs, codeBase, funcOff uint64) {
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
@@ -31,8 +34,8 @@ func printRegs(regs *Regs, codeBase, funcOff uint64) {
func printVectorRegs(v *VectorRegs) {
fmt.Println("\n Vector registers (V0-V31):")
for i := 0; i < 32; i += 2 {
fmt.Printf(" V%-2d = %016x%016x\n", i, v.V[i][8], v.V[i][0])
fmt.Printf(" V%-2d = %016x%016x\n", i+1, v.V[i+1][8], v.V[i+1][0])
fmt.Printf(" V%-2d = %016x%016x\n", i, binary.LittleEndian.Uint64(v.V[i][8:16]), binary.LittleEndian.Uint64(v.V[i][0:8]))
fmt.Printf(" V%-2d = %016x%016x\n", i+1, binary.LittleEndian.Uint64(v.V[i+1][8:16]), binary.LittleEndian.Uint64(v.V[i+1][0:8]))
}
}
+427
View File
@@ -0,0 +1,427 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
//go:build linux && amd64
package debug
import (
"bytes"
"fmt"
"io"
"os"
"path/filepath"
"runtime"
"strings"
"testing"
"time"
"unsafe"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
)
// Integration tests beyond the basic entry breakpoint: hardware watchpoints,
// conditional breakpoints, next/finish over a CALL, faulting kernels and the
// xstate vector-register readout. All drive a real ptrace session, so they
// run on amd64 hosts only.
// writeKernel writes an assembly source to a temporary file with the
// architecture suffix the assembler dispatcher expects.
func writeKernel(t *testing.T, src string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "kernel_amd64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatalf("write kernel: %v", err)
}
return path
}
// launchKernel launches a session for the kernel source and returns the
// session, its breakpoint manager and the function layout.
func launchKernel(t *testing.T, bin, path, funcName string, args []byte) (*Session, *Breakpoints, asm.FuncLayout) {
t.Helper()
k, err := verify.Load(path)
if err != nil {
t.Fatalf("Load: %v", err)
}
t.Cleanup(k.Close)
fl, err := k.Func(funcName)
if err != nil {
t.Fatalf("Func: %v", err)
}
if len(args) < fl.Args {
padded := make([]byte, fl.Args)
copy(padded, args)
args = padded
}
sess, err := Launch(bin, path, funcName, args)
if err != nil {
t.Fatalf("Launch: %v", err)
}
t.Cleanup(sess.Kill)
bm := NewBreakpoints(sess)
return sess, bm, fl
}
// runToEntry resumes the freshly launched debuggee until the breakpoint at
// the function entry traps, mirroring the REPL continue loop: the debuggee
// SIGSTOPs twice (launch barrier and entry barrier) before entering the JIT
// call.
func runToEntry(t *testing.T, sess *Session, bm *Breakpoints, entry uint64) {
t.Helper()
for range 50 {
for _, bp := range bm.All() {
bm.Reinsert(bp.Addr)
}
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
if sess.Exited() {
t.Fatal("debuggee exited before the entry breakpoint trapped")
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if bm.HandleTrap(&regs) != nil {
return
}
}
t.Fatal("no entry breakpoint trap after 50 resumes")
}
// captureStdout runs fn with os.Stdout redirected to a pipe and returns
// what it printed (the REPL writes its reports to stdout).
func captureStdout(t *testing.T, fn func()) string {
t.Helper()
r, w, err := os.Pipe()
if err != nil {
t.Fatalf("pipe: %v", err)
}
old := os.Stdout
os.Stdout = w
done := make(chan string, 1)
go func() {
b, _ := io.ReadAll(r)
done <- string(b)
}()
defer func() { os.Stdout = old }()
fn()
w.Close()
return <-done
}
// TestWatchpointArmRunHit proves the debug-register offsets: the watchpoint
// must fire on the store, with si_addr naming the watched address. The
// kernel writes its return value to ret+0(FP), which is the 8-byte word
// right above the stack pointer at entry.
func TestWatchpointArmRunHit(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func wpret() int64
TEXT ·wpret(SB), NOSPLIT, $0-8
MOVQ $0x5a5a5a5a5a5a5a5a, AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "wpret", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
watched := regs.RSP + 8 // ret+0(FP): the store target
slot := sess.FindFreeWatchpointSlot()
if slot < 0 {
t.Fatal("no free watchpoint slot")
}
if err := sess.SetWatchpoint(slot, watched, WatchWrite, 8); err != nil {
t.Fatalf("SetWatchpoint: %v (wrong debug-register offsets?)", err)
}
if err := sess.Continue(); err != nil {
t.Fatalf("Continue: %v", err)
}
reason, addr := sess.StopInfo()
if reason != StopWatchpoint {
t.Fatalf("stop reason = %v, want StopWatchpoint (DR0-DR3/DR7 offsets are wrong)", reason)
}
if addr != watched {
t.Fatalf("watchpoint address = %#x, want %#x", addr, watched)
}
// The watched word holds the stored value: x86 data breakpoints are
// reported with the access complete.
if word, err := sess.Peek(watched); err != nil || word != 0x5a5a5a5a5a5a5a5a {
t.Errorf("watched word = %#x (err %v), want 0x5a5a5a5a5a5a5a5a", word, err)
}
if err := sess.ClearWatchpoint(slot); err != nil {
t.Fatalf("ClearWatchpoint: %v", err)
}
}
// TestConditionalBreakpointFalseThenTrue proves the false-condition path:
// the breakpoint steps over the original instruction, re-arms itself and
// keeps running silently, and the true condition stops exactly once with the
// register in the expected state.
func TestConditionalBreakpointFalseThenTrue(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func countdown(n int64) int64
TEXT ·countdown(SB), NOSPLIT, $0-16
MOVQ n+0(FP), CX
loop:
DECQ CX
CMPQ CX, $0
JNE loop
MOVQ CX, ret+8(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "countdown", []byte{8})
loopAddr := sess.CodeBase() + uint64(fl.Offset) + uint64(fl.Labels["loop"])
// The length of the breakpointed instruction, from a disassembly taken
// before the INT3 is patched in.
_, insnLen, err := sess.Disassemble(loopAddr)
if err != nil || insnLen <= 0 {
t.Fatalf("Disassemble at %#x: len=%d err=%v", loopAddr, insnLen, err)
}
cond := &Condition{Reg: "rcx", Op: "==", Value: 1}
bp, err := bm.SetWithCond(loopAddr, "loop", cond)
if err != nil {
t.Fatalf("SetWithCond: %v", err)
}
hits := 0
exited := false
for range 200 {
for _, b := range bm.All() {
bm.Reinsert(b.Addr)
}
if err := sess.Continue(); err != nil {
exited = true
break // the debuggee finished
}
if sess.Exited() {
exited = true
break
}
if sig := sess.LastSignal(); sig != 0 {
t.Fatalf("unexpected signal stop %v", sig)
}
regs, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if hit := bm.HandleTrap(&regs); hit != nil {
hits++
if regs.RCX != 1 {
t.Fatalf("hit with RCX=%d, want 1", regs.RCX)
}
// Park after the instruction, as the REPL does.
if err := sess.Step(); err != nil {
t.Fatalf("Step: %v", err)
}
} else {
// A false evaluation must leave the debuggee past the whole
// original instruction: a PC inside it (trapAddr+1 on amd64)
// means the resume happens mid-instruction.
fresh, err := sess.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
if fresh.RIP > loopAddr && fresh.RIP < loopAddr+uint64(insnLen) {
t.Fatalf("false evaluation left the PC at %#x, inside the %d-byte instruction at %#x",
fresh.RIP, insnLen, loopAddr)
}
}
}
if hits != 1 {
t.Fatalf("conditional breakpoint hit %d times, want exactly 1 (false evaluations must run through silently)", hits)
}
if bp.Hits() != 1 {
t.Errorf("bp.Hits() = %d, want 1", bp.Hits())
}
if !exited || !sess.Exited() {
t.Fatal("debuggee did not run to completion after the conditional hit")
}
}
// TestNextAndFinishOverCall proves next and finish evaluate the trap with
// registers fetched after the stop: next lands exactly on the instruction
// after the CALL, and finish stops exactly on the return address.
func TestNextAndFinishOverCall(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func caller(x int64) int64
// The argument travels in AX: FP argument slots of CALL-bearing functions
// are an assembler concern outside this test's scope.
TEXT ·caller(SB), NOSPLIT, $0-16
MOVQ $5, AX
CALL ·bump(SB)
aftercall:
MOVQ AX, ret+8(FP)
RET
// func bump(x int64) int64
TEXT ·bump(SB), NOSPLIT, $0-0
ADDQ $3, AX
RET
`
path := writeKernel(t, kernel)
// next: step the prologue and the constant load (3 instructions), then
// step over the CALL and check the landing address and RAX.
sess, bm, fl := launchKernel(t, bin, path, "caller", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
afterOff := uint64(fl.Labels["aftercall"])
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), fl.Offset, fl.Size, fl.Args, nil, nil,
strings.NewReader("step 3\nnext\nregs\nquit\n"))
})
if !strings.Contains(out, fmt.Sprintf("func+%#x", afterOff)) {
t.Errorf("next did not land on the instruction after the CALL (func+%#x); output:\n%s", afterOff, out)
}
if !strings.Contains(out, "RAX = 0x0000000000000008") {
t.Errorf("callee did not run exactly once under next (want RAX=8); output:\n%s", out)
}
// finish: run to the return address read off the stack at entry.
sess2, bm2, fl2 := launchKernel(t, bin, path, "caller", nil)
entry2 := sess2.CodeBase() + uint64(fl2.Offset)
if _, err := bm2.Set(entry2, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess2, bm2, entry2)
regs, err := sess2.GetRegs()
if err != nil {
t.Fatalf("GetRegs: %v", err)
}
retAddr, err := sess2.Peek(regs.RSP)
if err != nil {
t.Fatalf("Peek return address: %v", err)
}
out2 := captureStdout(t, func() {
REPL(sess2, bm2, sess2.CodeBase(), fl2.Offset, fl2.Size, fl2.Args, nil, nil,
strings.NewReader("step 1\nfinish\nquit\n"))
})
want := fmt.Sprintf("finished, now at %#x\n", retAddr)
if !strings.Contains(out2, want) {
t.Errorf("finish stopped at the wrong PC; want %q in output:\n%s", want, out2)
}
}
// TestSignalStopSurfaced proves a faulting kernel surfaces as a reported
// stop instead of an infinite fault loop. A regression here hangs, so a
// watchdog fails the run rather than letting CI stall.
func TestSignalStopSurfaced(t *testing.T) {
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func crash() int64
TEXT ·crash(SB), NOSPLIT, $0-8
XORQ AX, AX
MOVQ (AX), AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, _ := launchKernel(t, bin, path, "crash", nil)
timer := time.AfterFunc(time.Minute, func() {
panic("watchdog: the debugger hung on the faulting kernel instead of reporting the signal stop")
})
defer timer.Stop()
out := captureStdout(t, func() {
REPL(sess, bm, sess.CodeBase(), 0, 0, 0, nil, nil,
strings.NewReader("continue\nquit\n"))
})
if !strings.Contains(out, "stopped on signal") {
t.Errorf("SIGSEGV did not surface as a reported stop; output:\n%s", out)
}
if !sess.Exited() {
t.Error("debuggee should be killed by quit after the signal stop")
}
}
// TestGetVectorRegsXState proves the NT_X86_XSTATE readout: the request
// succeeds on a normal process and the XMM halves agree with
// PTRACE_GETFPREGS.
func TestGetVectorRegsXState(t *testing.T) {
// The FPRegs layout must mirror the kernel's user_fpregs_struct
// exactly: PTRACE_GETFPREGS fills all 512 bytes, so a short struct
// overflows the caller's memory.
if got := unsafe.Sizeof(FPRegs{}); got != 512 {
t.Fatalf("sizeof(FPRegs) = %d, want 512", got)
}
if got := unsafe.Offsetof(FPRegs{}.XMM); got != 160 {
t.Fatalf("offsetof(FPRegs.XMM) = %d, want 160", got)
}
runtime.LockOSThread()
defer runtime.UnlockOSThread()
bin := buildGasm(t)
const kernel = `#include "textflag.h"
// func vprobe() int64
TEXT ·vprobe(SB), NOSPLIT, $0-8
MOVQ $1, AX
MOVQ AX, ret+0(FP)
RET
`
path := writeKernel(t, kernel)
sess, bm, fl := launchKernel(t, bin, path, "vprobe", nil)
entry := sess.CodeBase() + uint64(fl.Offset)
if _, err := bm.Set(entry, "entry"); err != nil {
t.Fatalf("Set: %v", err)
}
runToEntry(t, sess, bm, entry)
v, err := sess.GetVectorRegs()
if err != nil {
t.Fatalf("GetVectorRegs: %v", err)
}
fp, err := sess.GetFPRegs()
if err != nil {
t.Fatalf("GetFPRegs: %v", err)
}
for i := range 16 {
if !bytes.Equal(v.YMM[i][:16], fp.XMM[i][:]) {
t.Errorf("YMM%d low half %x, want the FPRegs XMM half %x", i, v.YMM[i][:16], fp.XMM[i][:])
}
}
}
+67 -28
View File
@@ -22,7 +22,14 @@ type Session struct {
cmd *exec.Cmd
stopped bool
exited bool
codeBase uint64 // base address of the JIT code in the debuggee
codeBase uint64 // base address of the JIT code in the debuggee
tmpDir string // scratch directory of the session, removed on Kill
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 DBGWVR0-15)
// lastSignal holds the signal of the most recent stop when that stop
// was a genuine signal-delivery-stop the caller must see (a fault such
// as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps,
// single-steps, SIGSTOP and suppressed runtime signals.
lastSignal syscall.Signal
}
// Launch starts the debuggee subprocess (gasm debug --target ...) and
@@ -77,7 +84,7 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
}
s := &Session{pid: cmd.Process.Pid, cmd: cmd}
s := &Session{pid: cmd.Process.Pid, cmd: cmd, tmpDir: tmpDir}
readyFile := filepath.Join(tmpDir, "ready")
for range 500 {
@@ -124,29 +131,17 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s
return s, bufAddrs, nil
}
// wait waits for the debuggee to stop and returns the wait status.
func (s *Session) wait() error {
var ws syscall.WaitStatus
_, err := syscall.Wait4(s.pid, &ws, 0, nil)
if err != nil {
return err
}
if ws.Exited() {
s.exited = true
return fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
}
s.stopped = true
return nil
}
// waitStopped consumes ptrace-stop events until one the debugger cares
// about arrives: SIGTRAP (a breakpoint or a completed single-step) or the
// debuggee's own SIGSTOP. A Go tracee's runtime raises SIGURG for
// asynchronous preemption, and every signal on a traced thread surfaces as
// a signal-delivery-stop, so those are suppressed and the tracee resumed
// without them. Runtime noise is why a single wait can return in the
// middle of runtime code and a resume can then fail: the event stream must
// be drained by the tracer.
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
// runtime raises SIGURG for asynchronous preemption, and every signal on a
// traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed
// and the tracee resumed without it. Every other signal (SIGSEGV, SIGBUS,
// SIGFPE, SIGILL, ...) is returned to the caller: resuming with signal 0
// would restart the faulting instruction and fault forever, so a faulting
// kernel must surface as a stop the caller reports. Runtime noise is also
// why a single wait can return in the middle of runtime code and a resume
// can then fail: the event stream must be drained by the tracer.
func (s *Session) waitStopped() (syscall.Signal, error) {
for {
var ws syscall.WaitStatus
@@ -164,10 +159,12 @@ func (s *Session) waitStopped() (syscall.Signal, error) {
switch sig := ws.StopSignal(); sig {
case syscall.SIGTRAP, syscall.SIGSTOP:
s.stopped = true
s.lastSignal = 0
return sig, nil
default:
// Runtime noise (SIGURG preemption and friends): resume the
// tracee without delivering the signal.
case syscall.SIGURG:
// Go runtime asynchronous preemption: resume the tracee
// without delivering the signal.
s.lastSignal = 0
if _, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_CONT),
@@ -176,10 +173,22 @@ func (s *Session) waitStopped() (syscall.Signal, error) {
); errno != 0 {
return 0, fmt.Errorf("debug: PTRACE_CONT: %w", errno)
}
default:
// A genuine signal-delivery-stop. Report it; the caller
// decides how to proceed.
s.stopped = true
s.lastSignal = sig
return sig, nil
}
}
}
// LastSignal returns the signal of the most recent stop when that stop was
// a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL
// or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and
// suppressed runtime signals.
func (s *Session) LastSignal() syscall.Signal { return s.lastSignal }
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
func (s *Session) Peek(addr uint64) (uint64, error) {
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
@@ -293,7 +302,8 @@ func (s *Session) Pid() int { return s.pid }
// CodeBase returns the base address of the JIT code in the debuggee.
func (s *Session) CodeBase() uint64 { return s.codeBase }
// Kill terminates the debuggee.
// Kill terminates the debuggee and removes the session's scratch
// directory, so a successful session leaves no gasm-debug-* debris behind.
func (s *Session) Kill() {
if !s.exited {
syscall.Kill(s.pid, syscall.SIGKILL)
@@ -303,6 +313,35 @@ func (s *Session) Kill() {
if s.cmd != nil && s.cmd.Process != nil {
s.cmd.Wait()
}
if s.tmpDir != "" {
os.RemoveAll(s.tmpDir)
s.tmpDir = ""
}
}
// execRange is one executable mapping of the debuggee.
type execRange struct {
lo, hi uint64
}
// execRanges parses the debuggee's executable mappings from /proc/pid/maps.
func execRanges(pid int) []execRange {
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
if err != nil {
return nil
}
var out []execRange
for line := range strings.SplitSeq(string(data), "\n") {
fields := strings.Fields(line)
if len(fields) < 2 || !strings.Contains(fields[1], "x") {
continue
}
var lo, hi uint64
if _, err := fmt.Sscanf(fields[0], "%x-%x", &lo, &hi); err == nil {
out = append(out, execRange{lo, hi})
}
}
return out
}
// findRWXMapping reads /proc/pid/maps and returns the base address of the
+68 -11
View File
@@ -6,6 +6,7 @@
package debug
import (
"encoding/binary"
"fmt"
"syscall"
"unsafe"
@@ -44,20 +45,24 @@ func (s *Session) SetRegs(regs *Regs) error {
return nil
}
// FPRegs holds the x87 FPU and SSE (XMM) register state from PTRACE_GETFPREGS.
// FPRegs holds the x87 FPU and SSE (XMM) register state from
// PTRACE_GETFPREGS. The layout is the kernel's struct user_fpregs_struct
// (sys/user.h), the FXSAVE image: 512 bytes with XMM0-15 at offset 160.
// The i387 fcs/ds segment fields do not exist in the 64-bit layout. The
// size matters: the copy fills all 512 bytes, so a short or misaligned
// struct makes PTRACE_GETFPREGS overflow the caller's memory.
type FPRegs struct {
FCW uint16
FSW uint16
FTW byte
FTW uint16
FOP uint16
FIP uint64
FCS uint16
FDP uint64
FDS uint16
MXCSR uint32
MXCSRMask uint32
ST [8][16]byte // x87 stack (10 bytes per reg, padded to 16)
XMM [16][16]byte // XMM0-15
XMM [16][16]byte // XMM0-15, struct offset 160
Reserved [96]byte // FXSAVE padding, to the full 512 bytes
}
// GetFPRegs retrieves the FPU/SSE register state of the stopped debuggee.
@@ -82,16 +87,68 @@ type VectorRegs struct {
YMM [16][32]byte // YMM0-15 (full 256-bit values)
}
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET + XSAVE.
// NT_X86_XSTATE (0x202), the xsave extended-state regset
// (include/uapi/linux/elf.h).
const ntX86XState = 0x202
// Layout of the buffer PTRACE_GETREGSET returns for NT_X86_XSTATE: the
// 512-byte legacy fxsave image (x87 state in 0-159, XMM0-15 in 160-511),
// then the 64-byte xsave header whose first 8 bytes are xstate_bv, then one
// component per set feature bit, each 64-byte aligned. The YMM high halves
// are the first extended component, at offset 576; that offset is fixed by
// the ISA on AVX-capable x86-64. XFEATURE_MASK_YMM is bit 2 of xstate_bv
// (arch/x86/include/asm/fpu/types.h); the high halves are zero when the bit
// is clear.
const (
xsaveXMMOffset = 160
xsaveXMMSize = 256
xsaveHeaderOffset = 512
xsaveBVOffset = xsaveHeaderOffset
ymmOffset = xsaveHeaderOffset + 64 // 576
ymmSize = 256 // 16 registers, 16 bytes each
xfeatureMaskYMM = 1 << 2
xstateMaxBuffer = 4096 // CPUID(0xD).xsave_size is far below this
)
// GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET on
// NT_X86_XSTATE. The low (XMM) halves always come from the legacy image;
// the high halves are copied only when xstate_bv reports the YMM feature,
// and read as zero otherwise. When the regset request fails the FP image
// still provides correct XMM halves, so that is the fallback.
func (s *Session) GetVectorRegs() (VectorRegs, error) {
var v VectorRegs
fp, err := s.GetFPRegs()
if err != nil {
return v, err
buf := make([]byte, xstateMaxBuffer)
iovec := syscall.Iovec{
Base: &buf[0],
Len: uint64(len(buf)),
}
_, _, errno := syscall.Syscall6(
syscall.SYS_PTRACE,
uintptr(syscall.PTRACE_GETREGSET),
uintptr(s.pid),
uintptr(ntX86XState),
uintptr(unsafe.Pointer(&iovec)),
0, 0,
)
if errno != 0 {
fp, err := s.GetFPRegs()
if err != nil {
return v, err
}
for i := range 16 {
copy(v.YMM[i][:16], fp.XMM[i][:])
}
return v, nil
}
n := int(iovec.Len)
for i := range 16 {
for j := range 16 {
v.YMM[i][j] = fp.XMM[i][j]
copy(v.YMM[i][:16], buf[xsaveXMMOffset+16*i:xsaveXMMOffset+16*i+16])
}
if n >= ymmOffset+ymmSize {
if binary.LittleEndian.Uint64(buf[xsaveBVOffset:xsaveBVOffset+8])&xfeatureMaskYMM != 0 {
for i := range 16 {
copy(v.YMM[i][16:], buf[ymmOffset+16*i:ymmOffset+16*i+16])
}
}
}
return v, nil
+5 -1
View File
@@ -91,5 +91,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
// breakpointInsn is the software breakpoint instruction.
var breakpointInsn = []byte{0xCC} // INT3
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap. x86-64 reports the #DB for INT3 with RIP on the byte after the
// INT3 (Intel SDM vol 3, "Debug Exceptions"), so the trap address is
// PC-1. The other supported architectures leave the PC on the trap
// instruction and use 0 there.
const breakpointPCAdjust = 1
+8 -2
View File
@@ -130,5 +130,11 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
// breakpointInsn is the software breakpoint instruction (BRK #0).
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0, because the arm64 kernel delivers the BRK SIGTRAP with the PC
// still on the BRK. do_el0_brk64 calls send_user_sigtrap, which uses
// instruction_pointer(regs) unmodified (arch/arm64/kernel/debug-monitors.c);
// only the kernel-internal skip paths advance the PC. GDB history agrees:
// decr_pc_after_break on aarch64 Linux is 0 (the +4 variant was a QEMU bug,
// sourceware PR 17280).
const breakpointPCAdjust = 0
+6 -2
View File
@@ -126,5 +126,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
// breakpointInsn is the software breakpoint instruction (BRK $0).
var breakpointInsn = []byte{0x05, 0x00, 0x2a, 0x00} // break 0
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0, because the kernel delivers the break SIGTRAP with csr_era
// still on the break instruction. do_bp passes regs->csr_era straight to
// force_sig_fault(SIGTRAP, TRAP_BRKPT, ...) and never adjusts era on the
// signal path (arch/loongarch/kernel/traps.c).
const breakpointPCAdjust = 0
+6 -2
View File
@@ -126,5 +126,9 @@ func (r *Regs) RegValue(name string) (uint64, bool) {
// breakpointInsn is the software breakpoint instruction (EBREAK).
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
// breakpointPCAdjust is how far PC is past the breakpoint instruction after a trap.
const breakpointPCAdjust = 4
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
// a trap: 0, because the kernel delivers the EBREAK SIGTRAP with sepc still
// on the ebreak. handle_break passes regs->epc straight to
// force_sig_fault(SIGTRAP, TRAP_BRKPT, ...) and only the kernel-internal
// WARN/CFI paths advance epc (arch/riscv/kernel/traps.c).
const breakpointPCAdjust = 0
+76 -20
View File
@@ -7,9 +7,10 @@ package debug
import (
"bufio"
"cmp"
"fmt"
"io"
"sort"
"slices"
"strconv"
"strings"
)
@@ -32,7 +33,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
entryAddr := codeBase + uint64(funcOffset)
fmt.Printf("stopped at function entry: %#x (%d bytes)\n", entryAddr, funcSize)
fmt.Println("commands: break <label|addr> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
fmt.Println("commands: break <label|addr|line> | step [n] | continue | disas [n] | regs | where | x <addr> [len] | w <addr> <val...> | labels | quit")
scanner := bufio.NewScanner(in)
@@ -93,7 +94,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
regs, _ := s.GetRegs()
pc := regs.GetPC()
text, instLen, _ := s.Disassemble(pc)
if strings.HasPrefix(strings.ToLower(text), "call") || strings.HasPrefix(strings.ToLower(text), "bl") {
if isCallInsn(text) {
afterAddr := pc + uint64(instLen)
_, err := bm.Set(afterAddr, "(next)")
if err != nil {
@@ -108,6 +109,21 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
bm.Clear(afterAddr)
continue
}
if s.Exited() {
bm.Clear(afterAddr)
fmt.Println("debuggee exited")
continue
}
if sig := s.LastSignal(); sig != 0 {
bm.Clear(afterAddr)
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue
}
// Fetch the registers after the stop: the trap must be
// evaluated against the real PC, not the pre-Continue
// snapshot, and a stale SetRegs would clobber live state.
regs, _ = s.GetRegs()
bm.HandleTrap(&regs)
bm.Clear(afterAddr)
} else {
@@ -143,9 +159,21 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
bm.Clear(retAddr)
continue
}
if !s.Exited() {
bm.HandleTrap(&regs)
if s.Exited() {
bm.Clear(retAddr)
fmt.Println("debuggee exited")
continue
}
if sig := s.LastSignal(); sig != 0 {
bm.Clear(retAddr)
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x\n", sig, regs.GetPC())
continue
}
// Fetch the registers after the stop, as the continue case
// does: HandleTrap must see the PC the trap left behind.
regs, _ = s.GetRegs()
bm.HandleTrap(&regs)
bm.Clear(retAddr)
if s.Exited() {
fmt.Println("debuggee exited")
@@ -171,6 +199,15 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println("debuggee exited")
break
}
if sig := s.LastSignal(); sig != 0 {
// A genuine signal-delivery-stop (a fault): report it
// and return to the prompt. Continuing would restart
// the faulting instruction and fault forever.
regs, _ := s.GetRegs()
fmt.Printf("stopped on signal %v at %#x (func+%#x)\n",
sig, regs.GetPC(), regs.GetPC()-codeBase-uint64(funcOffset))
break
}
reason, wpAddr := s.StopInfo()
if reason == StopWatchpoint {
fmt.Printf("watchpoint hit at %#x\n", wpAddr)
@@ -196,7 +233,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "break", "b":
if len(parts) < 2 {
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val>]")
fmt.Println("usage: break <label|addr|line> [if <reg> <op> <val|reg|*addr>]")
continue
}
var addr uint64
@@ -221,13 +258,27 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
reg := strings.ToLower(parts[3])
op := parts[4]
operand := parts[5]
if val, err := strconv.ParseUint(operand, 0, 64); err == nil {
cond = &Condition{Reg: reg, Op: op, Value: val}
} else {
cond = &Condition{Reg: reg, Op: op, Reg2: strings.ToLower(operand)}
switch {
case strings.HasPrefix(operand, "*"):
// Memory operand: compare against the 8-byte word at
// the address, resolved in the debuggee when the
// breakpoint is evaluated.
addr, err := strconv.ParseUint(strings.TrimPrefix(operand, "*"), 0, 64)
if err != nil {
fmt.Printf("invalid memory operand: %s\n", operand)
continue
}
cond = &Condition{Reg: reg, Op: op, MemAddr: addr}
default:
val, err := strconv.ParseUint(operand, 0, 64)
if err == nil {
cond = &Condition{Reg: reg, Op: op, Value: val}
} else {
cond = &Condition{Reg: reg, Op: op, Reg2: strings.ToLower(operand)}
}
}
} else if len(parts) >= 4 && parts[2] == "if" {
fmt.Println("usage: break <label|addr> if <reg> <op> <value|reg>")
fmt.Println("usage: break <label|addr|line> if <reg> <op> <value|reg|*addr>")
continue
}
bp, err := bm.SetWithCond(addr, label, cond)
@@ -237,7 +288,7 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
}
condStr := ""
if cond != nil {
condStr = fmt.Sprintf(" if %s %s %#x", cond.Reg, cond.Op, cond.Value)
condStr = " if " + cond.String()
}
fmt.Printf("breakpoint set: %s at %#x (func+%#x)%s\n", bp.Label, bp.Addr, bp.Addr-codeBase-uint64(funcOffset), condStr)
@@ -277,7 +328,11 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
addr, _ = resolveAddr(parts[1], codeBase, uint64(funcOffset), labels)
}
if len(parts) > 2 {
length, _ = strconv.Atoi(parts[2])
// A malformed or non-positive length would panic
// ReadMemory's make; fall back to the default instead.
if n, err := strconv.Atoi(parts[2]); err == nil && n > 0 {
length = n
}
}
mem, err := s.ReadMemory(addr, length)
if err != nil {
@@ -336,10 +391,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
}
case "labels", "l":
sorted := make([]Label, len(labels))
copy(sorted, labels)
sort.Slice(sorted, func(i, j int) bool { return sorted[i].Offset < sorted[j].Offset })
for _, l := range sorted {
slices.SortFunc(labels, func(a, b Label) int { return cmp.Compare(a.Offset, b.Offset) })
for _, l := range labels {
fmt.Printf(" func+%#04x %s\n", l.Offset, l.Name)
}
@@ -369,7 +422,10 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
fmt.Println()
case "help", "h", "?":
fmt.Printf(` break <label|addr> [if <reg> <op> <val>] set a breakpoint
fmt.Printf(` break <label|addr|line> [if <reg> <op> <val|reg|*addr>]
set a breakpoint, optionally conditional on a
register compared to a constant, a register, or the
8-byte word at *addr
delete <label|addr> remove a breakpoint
info break list all breakpoints
watch <addr> [r|w] [size] set a hardware watchpoint (write by default)
@@ -463,8 +519,8 @@ func REPL(s *Session, bm *Breakpoints, codeBase uint64, funcOffset, funcSize, ar
case "unwatch":
if len(parts) >= 2 {
slot, err := strconv.Atoi(parts[1])
if err != nil || slot < 0 || slot > 3 {
fmt.Println("usage: unwatch [<slot>]")
if err != nil || slot < 0 || slot >= maxWatchpoints() {
fmt.Printf("usage: unwatch [<slot 0-%d>]\n", maxWatchpoints()-1)
continue
}
if err := s.ClearWatchpoint(slot); err != nil {

Some files were not shown because too many files have changed in this diff Show More