Compare commits

...
17 Commits
Author SHA1 Message Date
petrbalvin 97dfaa7526 docs: changelog for macro expansion and the corrected corpus audit
Test / test (push) Successful in 2m14s
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 66aa4dbc8b test(verify): register the campaign kernels in the ground-truth suites
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin dce5d31462 feat(amd64): LOCK and REP prefixes, literal data pseudo-ops and ADJSP
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 9dc3987e02 feat(riscv64,loong64): PCALIGN, branch relaxation and operand shapes
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 9b238a525a feat(arm64): wide immediates, SIMD compare and system operand forms
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin ad82aac663 feat(parser): macro expansion, conditionals and include splicing with -I
Assisted-by: GLM 5.3 Flash
2026-09-20 14:25:47 +02:00
petrbalvin 0629f5e2df feat(arm64): assemble PCALIGN padding and BYTE literal bytes
Test / test (push) Successful in 2m16s
Assisted-by: GLM 5.3 Flash
2026-09-20 11:49:05 +02:00
petrbalvin ecb203dcf5 fix(lexer): treat trailing CR as line end so comment text is idempotent
Test / test (push) Successful in 2m13s
Assisted-by: GLM 5.3 Flash
2026-09-20 11:40:39 +02:00
petrbalvin 6c672567f3 feat(amd64): assemble the double-shift and static-SB operand shapes
Assisted-by: GLM 5.3 Flash
2026-09-20 11:40:39 +02:00
petrbalvin cc6e416c59 fix(lint): exempt shift counts, SETcc and ABIInternal from false positives
Assisted-by: GLM 5.3 Flash
2026-09-20 11:40:39 +02:00
petrbalvin c66a47973a fix(format): preserve square brackets in SIMD operands
Test / test (push) Successful in 2m15s
Assisted-by: GLM 5.3 Flash
2026-09-20 09:58:25 +02:00
petrbalvin 9629897202 docs: changelog and readme for the instruction wave and the honest corpus rate
Test / test (push) Successful in 2m17s
Assisted-by: GLM 5.3 Flash
2026-09-20 06:45:03 +02:00
petrbalvin 5399a8a724 feat(audit): probe the new operand shapes and measure attemptable files
Assisted-by: GLM 5.3 Flash
2026-09-20 06:45:03 +02:00
petrbalvin de5d9f358e feat(riscv64,loong64): encode AMO atomics, vector slices and bit ops
Assisted-by: GLM 5.3 Flash
2026-09-20 06:44:51 +02:00
petrbalvin ca3fdce0e0 feat(arm64): encode pairs, atomics, crypto, system and NEON slices
Assisted-by: GLM 5.3 Flash
2026-09-20 06:44:51 +02:00
petrbalvin fc2d92eabd feat(amd64): encode the GOROOT instruction families
Assisted-by: GLM 5.3 Flash
2026-09-20 06:44:51 +02:00
petrbalvin 39d2e80145 ci(release): refuse empty assets and verify what the release serves
Test / test (push) Successful in 2m10s
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 02:01:36 +02:00
80 changed files with 14469 additions and 379 deletions
+37
View File
@@ -325,6 +325,17 @@ jobs:
chomp $id;
my @files = grep { -f $_ } glob(q{dist/*/*});
@files or die qq{ERROR: no assets under dist/\n};
# A file that arrived empty from the artifact step would be uploaded as an
# empty attachment, every status would still be 201, and the run would go
# green over a release nobody can install. Refuse it here, before the
# upload, and verify what was stored afterwards.
my %size;
for my $path (@files) {
my $n = -s $path // 0;
(my $name = $path) =~ s{.*/}{};
$n > 0 or die qq{ERROR: $path is empty, so there is nothing to upload\n};
$size{$name} = $n;
}
my $bad = 0;
for my $path (@files) {
(my $name = $path) =~ s{.*/}{};
@@ -346,5 +357,31 @@ jobs:
printf qq{%s: HTTP %s\n}, $name, $code;
$bad = 1 if $code ne q{201};
}
# Read every asset back through the release download route and require the
# served length to be the file that was sent: stored but empty is a broken
# release however green the run looks.
open(my $v, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
my $v = <$v>;
close($v);
chomp $v;
for my $name (sort keys %size) {
my $url = qq{$ENV{GITEA_SERVER_URL}/$ENV{GITEA_REPOSITORY}/releases/download/v$v/$name};
my @head = (q{curl}, q{-sS}, q{-I}, q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}}, $url);
open(my $h, q{-|}, @head) or die qq{curl: $!};
my $len;
my $status;
while (my $l = <$h>) {
$status = $1 if $l =~ m{^HTTP/\S+\s+(\d+)};
$len = $1 if $l =~ m{^content-length:\s*(\d+)}i;
}
my $ok = close($h);
$len = defined $len ? $len : 0;
if (!$ok || $status != 200 || $len != $size{$name}) {
printf qq{ERROR: %s serves %s bytes, expected %d\n}, $name, $len, $size{$name};
$bad = 1;
next;
}
printf qq{%s: serves %d bytes\n}, $name, $len;
}
exit($bad ? 1 : 0);
'
+38
View File
@@ -9,6 +9,39 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
`gasm audit-instructions` now preprocess assembly the way the
toolchain does: object and parameterised `#define` macros expand at
the point of use, `#undef` and the `#ifdef`/`#ifndef`/`#else`/
`#endif` family select branches, `#include` splices headers resolved
through the source directory and the new repeatable `-I` flag, `;`
separates statements, and constant expressions left in operands
(`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion
happens only on the assembly path: `gasm lint`, `gasm fmt` and the
language server keep reading the raw file.
- **The GOROOT instruction wave, part 1.** The encoder now covers the
instruction families GOROOT's real code uses that gasm lacked,
byte-verified against `go tool asm`: on amd64 the carry ALU, the
atomics (CMPXCHG, XADD, XCHG), AES-NI, SHA-1/256, PCLMULQDQ, CRC32,
GFNI, ADX, BMI, the string primitives, the system set (CPUID, RDTSC,
SYSCALL, fences, MXCSR) and the SSE/AVX/EVEX gaps; on arm64 the pair
loads and stores (LDP/STP), acquire/release and LSE atomics, AES and
SHA, the system operations, the bit ops and the NEON slice including
structure loads and the literal-pool moves; on riscv64 the RV64A AMO
family with aq/rl ordering, the Zbb pseudos with their RVC
compressions, the FMA forms and the RVV slice with `vsetvli`/
`vsetivli`; on loong64 the AM atomics with acquire/release forms, the
LSX/LASX slice, the `VMOVQ`/`XVMOVQ` transfer family and FSEL.
Also fixed on the way: arm64 `CASD`/`CASW` lacked an opcode bit, and
riscv64 `VSETVLI` with an immediate length now canonicalises to
`vsetivli` as the toolchain does.
- **The corpus audit measures honestly.** Files named for Go ports gasm
does not target (arm, 386, s390x, ...) are no longer attempted for the
four supported architectures (no supported build compiles them), and
the headline rate is reported over attemptable files: 136 of 433 on
the full corpus (31.4 %), 135 of 383 on real code (35.2 %), from the
127 that the previous release measured. The probe battery that
decides encodability gained the operand shapes the new families use.
-
## [0.34.0] - 2026-09-20
@@ -106,6 +139,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Fixed
- **The corpus audit attempts fewer files that no build would compile.**
Files named for Go ports gasm does not target (arm, 386, s390x, ...)
are reported as other-port and never attempted, the headline rate is
computed over attemptable files, and the audit searches the
toolchain's shipped headers (funcdata.h and friends) automatically.
- **riscv64 JALR silently jumped to the wrong register.** The trampoline
form `JALR X0, 0(X5)` read the memory operand's base as the destination,
encoding a jump to X0 with no diagnostic; the destination is the first
+3 -2
View File
@@ -124,8 +124,9 @@ can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 127 of 627 files (20.3 %)
assembling for every target architecture today, with the top failure
`gasm audit-instructions --corpus` reports 136 of 433 attemptable files
(31.4 %) assembling for every target architecture today (files named for
other Go ports are counted but never attempted), with the top failure
reasons per architecture; the number moves with every release.
### Validation status
+4
View File
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
}
// x87 stack registers (FMOVD and the other x87 moves).
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
}
return regs
}
+27
View File
@@ -150,6 +150,33 @@ func arm64Curated() []Instr {
t = append(t, i(op, "Atomic memory operation"))
}
// Register-pair loads and stores.
for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} {
t = append(t, ic(op, "Register-pair load or store", 2, 2))
}
// Cache maintenance and prefetch.
t = append(t, i("DC", "Data cache maintenance"))
t = append(t, i("PRFM", "Memory prefetch"))
for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} {
t = append(t, i(op, "Atomic memory operation with acquire and release semantics"))
}
// Cryptographic extensions.
for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} {
t = append(t, i(op, "AES round"))
}
for _, op := range []string{
"SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1",
"SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1",
"SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1",
} {
t = append(t, i(op, "SHA round"))
}
for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} {
t = append(t, i(op, "Three-way XOR / rotate crypto vector operation"))
}
// Floating-point scalar.
for _, op := range []string{
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
+2654 -94
View File
File diff suppressed because it is too large Load Diff
+753 -27
View File
@@ -27,7 +27,14 @@ package asm
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
import "maps"
import (
"maps"
"math/bits"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
@@ -153,6 +160,67 @@ func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
}
// ---- logical immediate ----
// a64LogicalImm encodes v as the AArch64 logical (bitmask) immediate for the
// given lane width (32 or 64): it returns the N, immr and imms fields of the
// imm13 encoding. The algorithm mirrors cmd/internal/obj/arm64's
// encodeLogicalImmArrEncoding: replicate the value, shrink it to the smallest
// repeating element, find the run of ones and its rotation. ok is false when
// v is not expressible (all zeros, all ones, or not a single cyclic run).
func a64LogicalImm(v int64, width int) (n, immr, imms uint32, ok bool) {
u := uint64(v)
if width == 32 {
u &= 0xFFFFFFFF
}
size := uint64(width)
mask := ^uint64(0)
if size < 64 {
mask = uint64(1)<<size - 1
}
u &= mask
// All zeros and all ones are MOV territory, not bitmask immediates.
if u == 0 || u == mask {
return 0, 0, 0, false
}
// Shrink to the smallest repeating element.
for size > 2 {
half := size / 2
hm := uint64(1)<<half - 1
if u&hm == u>>half&hm {
size = half
u &= hm
} else {
break
}
}
ones := bits.OnesCount64(u)
// Find the right-rotation that lays the ones out contiguously at the
// bottom of the element; the hardware applies the inverse rotation.
em := uint64(1)<<size - 1
expected := uint64(1)<<ones - 1
rot := -1
for r := 0; r < int(size); r++ {
rotated := u>>r | u<<(int(size)-r)
if size < 64 {
rotated &= em
}
if rotated == expected {
rot = r
break
}
}
if rot < 0 {
return 0, 0, 0, false
}
if size == 64 {
n = 1
}
immr = uint32((int(size) - rot) % int(size))
imms = ^uint32(uint32(size*2-1))&0x3F | uint32(ones-1)
return n, immr, imms, true
}
// ---- load/store (unsigned immediate, scaled) ----
// a64LSU encodes a load/store register (unsigned immediate, scaled):
@@ -240,6 +308,8 @@ const (
a64CondLT = 0xb
a64CondGT = 0xc
a64CondLE = 0xd
a64CondAL = 0xe
a64CondNV = 0xf
)
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
@@ -260,6 +330,8 @@ var arm64CondMap = map[string]uint32{
"LT": a64CondLT,
"GT": a64CondGT,
"LE": a64CondLE,
"AL": a64CondAL,
"NV": a64CondNV,
}
// ---- instruction format tags ----
@@ -267,28 +339,47 @@ var arm64CondMap = map[string]uint32{
type a64Format uint8
const (
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
a64FCondCmp // conditional compare: CCMP, CCMN
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
a64FVTBL // SIMD table lookup: VTBL
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
@@ -401,6 +492,16 @@ func init() {
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
// The widening multiplies: a 64-bit result riding the same layout, the
// three-operand forms reading the accumulate register as ZR.
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
// ---- move wide ----
// MOVZ/MOVN/MOVK
@@ -450,6 +551,15 @@ func init() {
// ---- bitfield ----
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
@@ -622,10 +732,626 @@ func init() {
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
// ---- SIMD basics ----
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
// ---- SIMD: the arrangement-aware tables in this file carry VADD,
// VSUB, VMUL and every other three-register vector op. ----
// ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ----
dp1 := map[string]uint32{
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
// the source width.
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
"REV16W": 0x5ac00400,
}
for m, op := range dp1 {
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
}
// ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ----
a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000}
a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000}
a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000}
a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000}
// ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ----
a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000}
a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000}
a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000}
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
// ---- compare/test and branch ----
a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000}
a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000}
a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000}
a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000}
a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000}
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00}
a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00}
a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00}
a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00}
a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00}
a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00}
a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00}
// ---- LSE atomics with acquire and release semantics ----
// CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/
// LDOR/SWP families leave Rs free for the returned value.
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
"SWPALB": 0x38e08000,
"SWPALW": 0xb8e08000,
"SWPALD": 0xf8e08000,
}
for m, op := range lse {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// The remaining width and ordering spellings of the same shapes, and the
// CAS compare-and-swap family, word-verified against go tool asm.
lseMore := map[string]uint32{
"LDADDAB": 0x38a00000,
"LDADDAH": 0x78a00000,
"LDADDALB": 0x38e00000,
"LDADDALH": 0x78e00000,
"LDADDLB": 0x38600000,
"LDADDLD": 0xf8600000,
"LDADDLH": 0x78600000,
"LDADDLW": 0xb8600000,
"LDCLRAB": 0x38a01000,
"LDCLRAH": 0x78a01000,
"LDCLRALH": 0x78e01000,
"LDCLRB": 0x38201000,
"LDCLRD": 0xf8201000,
"LDCLRH": 0x78201000,
"LDCLRLB": 0x38601000,
"LDCLRLD": 0xf8601000,
"LDCLRLH": 0x78601000,
"LDCLRLW": 0xb8601000,
"LDCLRW": 0xb8201000,
"LDEORAB": 0x38a02000,
"LDEORAD": 0xf8a02000,
"LDEORAH": 0x78a02000,
"LDEORALB": 0x38e02000,
"LDEORALH": 0x78e02000,
"LDEORAW": 0xb8a02000,
"LDEORB": 0x38202000,
"LDEORD": 0xf8202000,
"LDEORH": 0x78202000,
"LDEORLB": 0x38602000,
"LDEORLD": 0xf8602000,
"LDEORLH": 0x78602000,
"LDEORLW": 0xb8602000,
"LDEORW": 0xb8202000,
"LDORAB": 0x38a03000,
"LDORAD": 0xf8a03000,
"LDORAH": 0x78a03000,
"LDORALH": 0x78e03000,
"LDORAW": 0xb8a03000,
"LDORB": 0x38203000,
"LDORD": 0xf8203000,
"LDORH": 0x78203000,
"LDORLB": 0x38603000,
"LDORLD": 0xf8603000,
"LDORLH": 0x78603000,
"LDORLW": 0xb8603000,
"LDORW": 0xb8203000,
"SWPAB": 0x38a08000,
"SWPAD": 0xf8a08000,
"SWPAH": 0x78a08000,
"SWPALH": 0x78e08000,
"SWPAW": 0xb8a08000,
"SWPB": 0x38208000,
"SWPH": 0x78208000,
"SWPLB": 0x38608000,
"SWPLD": 0xf8608000,
"SWPLH": 0x78608000,
"SWPLW": 0xb8608000,
"CASAD": 0xc8e07c00,
"CASALB": 0x08e0fc00,
"CASLW": 0x88a0fc00,
}
for m, op := range lseMore {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
dpsrExtra := map[string]uint32{
"ADC": 0x9a000000, "ADCW": 0x1a000000,
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
"SBC": 0xda000000, "SBCW": 0x5a000000,
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
"NGC": 0xda000000, "NGCW": 0x5a000000,
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
"NEGSW": 0x6b000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
}
for m, op := range dpsrExtra {
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
}
// ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ----
crypto2 := map[string]uint32{
"AESD": 0x4e285800, "AESE": 0x4e284800,
"AESIMC": 0x4e287800, "AESMC": 0x4e286800,
"SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800,
"SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000,
}
for m, op := range crypto2 {
a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op}
}
crypto3 := map[string]uint32{
"SHA1C": 0x5e000000, "SHA1P": 0x5e001000,
"SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000,
"SHA256H": 0x5e004000, "SHA256H2": 0x5e005000,
"SHA256SU1": 0x5e006000, "SHA512H": 0xce608000,
"SHA512H2": 0xce608400, "SHA512SU1": 0xce608800,
}
for m, op := range crypto3 {
a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op}
}
// ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ----
a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000}
a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000}
a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000}
a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000}
a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL}
a64InstrTable["VDUP"] = a64Enc{format: a64FDUP}
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
// for instructions that exist at a single arrangement and carry that
// arrangement's bits inside the base already, the fixed flag.
type a64SimdVSpec struct {
base uint32
arrs uint16
fixed bool
}
// a64Arr names the vector arrangements the encoders deal with, indexed by
// a64Arr. The source spellings put the element letter first: B8, H4, S2,
// D1 and the 128-bit halves B16, H8, S4, D2.
const (
a64Arr8B = iota
a64Arr16B
a64Arr4H
a64Arr8H
a64Arr2S
a64Arr4S
a64Arr2D
a64ArrD1
a64ArrQ1
a64ArrCount
)
// a64ArrNames maps an arrangement to its source spelling (element letter
// first, as the toolchain writes it).
var a64ArrNames = [a64ArrCount]string{
a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8",
a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1",
}
// a64ArrIndex resolves a source spelling to its a64Arr index, -1 when
// unknown.
func a64ArrIndex(s string) int {
for i, n := range a64ArrNames {
if n == s {
return i
}
}
return -1
}
// a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q)
// as it appears in element operands such as V13.S[0].
func a64ElemLetter(s string) bool {
switch s {
case "B", "H", "S", "D", "Q":
return true
}
return false
}
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
// accept: H, S and D widths for the pairwise data-processing, H and S for
// the across-vector reductions.
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
// flag alone, without the size bits: the FP converts, the FP round-to-integral
// and pairwise compares among them. Word-verified against go tool asm.
var a64SimdQOnly = map[string]bool{
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
"VFABS": true, "VFNEG": true, "VFSQRT": true,
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
"VFMAXV": true, "VFMAXNMV": true,
}
// a64ArrBits carries the fixed bits an arrangement contributes to the
// three-same word shape: the element size at bits 23:22 and the 128-bit
// flag at bit 30. Bit 29 belongs to the instruction's own base.
var a64ArrBits = [a64ArrCount]uint32{
a64Arr8B: 0,
a64Arr16B: 1 << 30,
a64Arr4H: 1 << 22,
a64Arr8H: 1<<30 | 1<<22,
a64Arr2S: 1 << 23,
a64Arr4S: 1<<30 | 1<<23,
a64Arr2D: 1<<30 | 1<<23 | 1<<22,
a64ArrD1: 1<<23 | 1<<22,
a64ArrQ1: 0,
}
// a64SimdVTable holds the arrangement-aware three-register SIMD
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VCMGE": {0x0e203c00, 0x7f, false},
"VCMGT": {0x0e203400, 0x7f, false},
"VCMHI": {0x2e203400, 0x7f, false},
"VCMHS": {0x2e203c00, 0x7f, false},
// FP compares take H, S and D arrangements only (the toolchain rejects
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
// FP arithmetic shares the same arrangement restriction.
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
// Saturating, halving, polynomial and pairwise arithmetic, the logical
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
// tool asm.
"VBIC": {0x0e601c00, 0x7f, false},
"VBIF": {0x2ee01c00, 0x7f, false},
"VBIT": {0x6ea01c00, 0x7f, false},
"VBSL": {0x6e601c00, 0x7f, false},
"VCMTST": {0x0e208c00, 0x7f, false},
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VMLA": {0x4ea09400, 0x7f, false},
"VMLS": {0x6ea09400, 0x7f, false},
"VORN": {0x4ee01c00, 0x7f, false},
"VSHADD": {0x4ea00400, 0x7f, false},
"VSRHADD": {0x4ea01400, 0x7f, false},
"VUHADD": {0x6ea00400, 0x7f, false},
"VURHADD": {0x6ea01400, 0x7f, false},
"VSMAX": {0x4ea06400, 0x7f, false},
"VSMIN": {0x4ea06c00, 0x7f, false},
"VSMAXP": {0x4ea0a400, 0x7f, false},
"VSMINP": {0x4ea0ac00, 0x7f, false},
"VUMAX": {0x2e206400, 0x7f, false},
"VUMIN": {0x2e206c00, 0x7f, false},
"VUMAXP": {0x6ea0a400, 0x7f, false},
"VUMINP": {0x6ea0ac00, 0x7f, false},
"VSQADD": {0x4ea00c00, 0x7f, false},
"VUQADD": {0x6ea00c00, 0x7f, false},
"VSQSUB": {0x4ea02c00, 0x7f, false},
"VUQSUB": {0x6ea02c00, 0x7f, false},
"VSSHL": {0x4ee04400, 0x7f, false},
"VUSHL": {0x6ee04400, 0x7f, false},
"VUZP1": {0x0e001800, 0x7f, false},
"VUZP2": {0x4ec05800, 0x7f, false},
"VTRN1": {0x4ec02800, 0x7f, false},
"VTRN2": {0x4ec06800, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
}
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
// combination for them, so they stay out and the encoder rejects the shape.
var a64SimdVZero = map[string]uint32{
"VCMEQ": 0x0e209800,
"VCMGT": 0x0e208800,
"VCMGE": 0x2e208800,
"VCMLT": 0x0e20a800,
"VCMLE": 0x2e209800,
// FP compares against (0.0): the register forms above carry the U and op
// bits; the zero forms reshape them.
"VFCMEQ": 0x0ea0d800,
"VFCMGE": 0x2ea0c800,
"VFCMGT": 0x0ea0c800,
"VFCMLE": 0x2ea0d800,
"VFCMLT": 0x0ea0e800,
}
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
// Two-register data-processing across one arrangement.
"VABS": {0x0e20b800, 0x7f, false},
"VNEG": {0x2e20b800, 0x7f, false},
"VCLS": {0x0e204800, 0x7f, false},
"VCLZ": {0x2e204800, 0x7f, false},
"VCNT": {0x0e205800, 0x7f, false},
"VNOT": {0x2e205800, 0x7f, false},
"VSQABS": {0x0e207800, 0x7f, false},
"VSQNEG": {0x2e207800, 0x7f, false},
"VRBIT": {0x6e605800, 0x7f, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
"VFABS": {0x0ea0f800, fpSimdArrs, false},
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
"VFRINTN": {0x0e218800, fpSimdArrs, false},
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
"VFRINTM": {0x0e219800, fpSimdArrs, false},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
// Across-vector reductions: the operand arrangement rides as usual and
// the destination stays a bare V register.
"VADDV": {0x0e31b800, 0x3f, false},
"VSMAXV": {0x0e30a800, 0x3f, false},
"VSMINV": {0x0e31a800, 0x3f, false},
"VUMAXV": {0x2e30a800, 0x3f, false},
"VUMINV": {0x2e31a800, 0x3f, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
// carry when they spell one at all; a bare V/F spelling is accepted as is.
var a64CryptoArr = map[string]int{
"AESD": a64Arr16B, "AESE": a64Arr16B, "AESIMC": a64Arr16B, "AESMC": a64Arr16B,
"SHA1H": a64Arr4S, "SHA1SU1": a64Arr4S, "SHA256SU0": a64Arr4S, "SHA512SU0": a64Arr2D,
"SHA1C": a64Arr4S, "SHA1P": a64Arr4S, "SHA1M": a64Arr4S, "SHA1SU0": a64Arr4S,
"SHA256H": a64Arr4S, "SHA256H2": a64Arr4S, "SHA256SU1": a64Arr4S,
"SHA512H": a64Arr2D, "SHA512H2": a64Arr2D, "SHA512SU1": a64Arr2D,
}
// a64DCOps maps the data-cache maintenance operation names to their fixed
// word (the register rides bits 4:0).
var a64DCOps = map[string]uint32{
"IVAC": 0xd5087620, "ZVA": 0xd50b7420,
"CVAC": 0xd50b7a20, "CVAU": 0xd50b7b20, "CIVAC": 0xd50b7e20,
}
// a64MRSOps maps the system register names GOROOT reads to their fixed word
// (the destination register rides bits 4:0).
var a64MRSOps = map[string]uint32{
"ELR_EL1": 0xd5384020, "MIDR_EL1": 0xd5380000,
"ID_AA64PFR0_EL1": 0xd5380400, "ID_AA64ISAR0_EL1": 0xd5380600,
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
"NZCV": 0xd53b4200, "FPCR": 0xd53b4400, "FPSR": 0xd53b4420,
}
// a64MSRRegOps maps the system register names GOROOT writes through the
// MSR (register) form, spelled in Go assembly as MOVD Rn, <sysreg> or
// MSR Rn, <sysreg>; the source register rides bits 4:0.
var a64MSRRegOps = map[string]uint32{
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
"ELR_EL1": 0xd5184020,
}
// a64MSROps maps the system register names GOROOT writes to their fixed
// word; the immediate rides CRm at bits 11:8 and Rt is the fixed 11111.
var a64MSROps = map[string]uint32{
"SPSel": 0xd50040a0, "DAIFSet": 0xd50340c0, "DAIFClr": 0xd50340e0, "DIT": 0xd5034040,
}
// a64PRFOps maps the prefetch operation names to their prfop immediate
// (word = 0xf9800000 | Rn<<5 | prfop).
var a64PRFOps = map[string]int{
"PLDL1KEEP": 0x00, "PLDL1STRM": 0x01, "PLDL2KEEP": 0x02, "PLDL2STRM": 0x03,
"PLDL3KEEP": 0x04, "PLDL3STRM": 0x05,
"PLIL1KEEP": 0x08, "PLIL1STRM": 0x09, "PLIL2KEEP": 0x0a, "PLIL2STRM": 0x0b,
"PLIL3KEEP": 0x0c, "PLIL3STRM": 0x0d,
"PSTL1KEEP": 0x10, "PSTL1STRM": 0x11, "PSTL2KEEP": 0x12, "PSTL2STRM": 0x13,
"PSTL3KEEP": 0x14, "PSTL3STRM": 0x15,
}
// a64VLD1Base holds the fixed words of the multi-register structure
// accesses, indexed by register count 1..4, before the Q and size bits.
// Post-index spellings add 0x9f0000 (post bit and Rm = 11111).
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
reg int
arr string
idx int
hasIdx bool
}
// a64VecReg parses a vector register operand: V0..V31 (F0..F31 as an alias,
// the same architectural registers the scalar floating-point spellings use),
// optionally with an arrangement suffix such as V0.B16 and, for element
// forms, a lane index such as V13.S[0]. It reports ok=false for anything
// else, including X/W and R spellings, which the toolchain's vector
// operands reject as well.
func a64VecReg(name string) (v a64Vec, ok bool) {
s := strings.TrimSpace(name)
if i := strings.IndexByte(s, '.'); i >= 0 {
v.arr = strings.TrimSpace(s[i+1:])
s = s[:i]
}
if v.arr != "" {
// Element form: B[3], S[2] and friends.
if j := strings.IndexByte(v.arr, '['); j >= 0 {
k := strings.LastIndexByte(v.arr, ']')
if k < j {
return v, false
}
n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k]))
if err != nil || n < 0 {
return v, false
}
v.idx, v.hasIdx = n, true
v.arr = strings.TrimSpace(v.arr[:j])
}
if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) {
return v, false
}
}
if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') {
return v, false
}
n := 0
for i := 1; i < len(s); i++ {
if s[i] < '0' || s[i] > '9' {
return v, false
}
n = n*10 + int(s[i]-'0')
}
if n > 31 {
return v, false
}
v.reg = n
return v, true
}
// a64ElemField encodes a lane index for the copy/insert group: imm5 = the
// index shifted by the element scale, with the scale's own bit set. B gets
// shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4.
func a64ElemField(arr string, idx int) (uint32, bool) {
var shift, low uint32
switch arr {
case "B8", "B16", "B":
shift, low = 1, 1
case "H4", "H8", "H":
shift, low = 2, 2
case "S2", "S4", "S":
shift, low = 3, 4
case "D1", "D2", "D":
shift, low = 4, 8
default:
return 0, false
}
if idx < 0 || idx >= 1<<(5-shift) {
return 0, false
}
return uint32(idx)<<shift | low, true
}
// a64VecListOf recovers the register list of a VLD1/VST1/VTBL operand run.
// The parser keeps parenthesised groups whole but splits bracketed lists on
// the commas, so a list arrives as one operand run whose first Raw starts
// with "[" and whose last Raw ends with "]". It returns the parsed
// registers with the brackets and spaces removed.
func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool) {
if start >= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") {
return nil, 0, false
}
end = start
for end < len(ops) {
if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") {
break
}
end++
}
if end >= len(ops) {
return nil, 0, false
}
for i := start; i <= end; i++ {
s := strings.TrimSpace(ops[i].Raw)
s = strings.TrimPrefix(s, "[")
s = strings.TrimSuffix(s, "]")
if s == "" && len(ops) > start+1 {
return nil, 0, false
}
for part := range strings.SplitSeq(s, ",") {
v, ok := a64VecReg(part)
if !ok {
return nil, 0, false
}
vs = append(vs, v)
}
}
return vs, end, true
}
// ---- load/store helper tables ----
+613 -18
View File
@@ -4,6 +4,7 @@
package asm
import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -472,12 +473,456 @@ TEXT ·f(SB), NOSPLIT, $0-0
}
}
// TestArm64SIMD tests SIMD encoding (via the instruction table).
// TestArm64SIMD tests SIMD encoding (via the arrangement-aware table).
func TestArm64SIMD(t *testing.T) {
// Verify SIMD instructions are in the table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
if _, ok := a64InstrTable[mnem]; !ok {
t.Errorf("%s not in instruction table", mnem)
// Verify SIMD instructions are in the arrangement table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL", "VAND", "VEOR", "VORR", "VCMEQ", "VZIP1", "VZIP2"} {
if _, ok := a64SimdVTable[mnem]; !ok {
t.Errorf("%s not in the SIMD arrangement table", mnem)
}
}
}
// TestArm64CarryAndBitOps pins the carry-setting arithmetic, the widening
// multiplies and the data-processing (1 source) group against go tool asm.
func TestArm64CarryAndBitOps(t *testing.T) {
got := arm64Words(t, "\tADC R0, R2, R12\n\tADCS $0, R1\n\tSBCS R5, R9, R5\n\tSBC R25, R10, R26\n"+
"\tMUL R4, R3, R0\n\tUMULH R24, R20, R24\n\tSMULH R1, R2, R3\n\tMSUB R19, R16, R26, R2\n"+
"\tRBIT R11, R4\n\tREV R1, R2\n\tCLZ R21, R9\n\tREVW R1, R2\n\tCLSW R1, R2\n")
want := []uint32{
0x9a00004c, // ADC R12, R2, R0
0xba1f0021, // ADCS R1, R1, ZR
0xfa050125, // SBCS R5, R9, R5
0xda19015a, // SBC R26, R10, R25
0x9b047c60, // MUL R0, R3, R4
0x9bd87e98, // UMULH R24, R20, R24
0x9b417c43, // SMULH R3, R2, R1
0x9b13c342, // MSUB R2, R26, R19, R16
0xdac00164, // RBIT R4, R11
0xdac00c22, // REV R2, R1
0xdac012a9, // CLZ R9, R21
0x5ac00822, // REVW R2, R1
0x5ac01422, // CLSW R2, R1
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64BitfieldExtract pins UBFX/SBFX: immr wraps to the register
// width, an out-of-range imms is an error.
func TestArm64BitfieldExtract(t *testing.T) {
got := arm64Words(t, "\tUBFX $33, R17, $25, R5\n\tUBFXW $4, R1, $9, R2\n")
want := []uint32{
0xd361e625, // UBFX immr=1 (33 wrapped), imms=25
0x53043022, // UBFXW immr=4, imms=9
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
for _, body := range []string{"\tUBFX $33, R17, $70, R5\n", "\tUBFX $-1, R17, $3, R5\n"} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64CondCompare pins CCMP/CCMN.
func TestArm64CondCompare(t *testing.T) {
got := arm64Words(t, "\tCCMP LE, R7, $19, $3\n\tCCMP LT, R30, R6, $7\n\tCCMN EQ, R1, R2, $3\n\tCCMPW LE, R7, $19, $3\n")
want := []uint32{
0xfa53d8e3, // CCMP imm form
0xfa46b3c7, // CCMP register form
0xba420023, // CCMN register form
0x7a53d8e3, // CCMPW
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64CompareBranch pins CBZ/CBNZ/TBZ/TBNZ against a label five and
// six words ahead, matching go tool asm's own offsets.
func TestArm64CompareBranch(t *testing.T) {
// Layout: CBZ(0) TBZ(4) TBNZ(8) CBNZ(12) NOP(16) NOP(17th word...) done.
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tCBZ R1, done\n\tTBZ $4, R7, done\n\tTBNZ $33, R7, done\n\tCBNZW R2, done\n" +
"\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
// done sits at word 6 from each branch's own pc: CBZ rel 6, TBZ rel 5,
// TBNZ rel 4, CBNZW rel 3.
want := []uint32{
0xb40000c1, // CBZ R1, +6
0x362000a7, // TBZ $4, R7, +5
0xb7080087, // TBNZ $33, R7, +4
0x35000062, // CBNZW R2, +3
0xd503201f, 0xd503201f, 0xd503201f,
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64ADR pins ADR against a forward label.
func TestArm64ADR(t *testing.T) {
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tADR done, R10\n\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
// rel = 12 bytes: immlo 0, immhi 3.
want := []uint32{0x1000006a, 0xd503201f, 0xd503201f, 0xd503201f, 0xd65f03c0}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64PairLoadStore pins LDP/STP/LDPW/FLDPD/FSTPD.
func TestArm64PairLoadStore(t *testing.T) {
got := arm64Words(t, "\tSTP (R2, R3), 8(R5)\n\tLDP -8(R5), (R2, R3)\n\tLDPW 4(R0), (R1, R2)\n\tSTPW (R1, R2), 4(R0)\n"+
"\tFLDPD 8(R0), (F1, F2)\n\tFSTPD (F3, F4), -8(R5)\n")
want := []uint32{
0xa9008ca2, // STP (R2, R3), 8(R5)
0xa97f8ca2, // LDP -8(R5), (R2, R3)
0x29408801, // LDPW 4(R0), (R1, R2)
0x29008801, // STPW (R1, R2), 4(R0)
0x6d408801, // FLDPD 8(R0), (F1, F2)
0x6d3f90a3, // FSTPD (F3, F4), -8(R5)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AcquireRelease pins LDAR/STLR and the acquire/release LSE
// families.
func TestArm64AcquireRelease(t *testing.T) {
got := arm64Words(t, "\tLDAR (R27), R22\n\tLDARB (R25), R2\n\tLDARW (R12), R29\n\tSTLR R3, (R24)\n\tSTLRB R11, (R22)\n"+
"\tCASALD R5, (R6), R7\n\tLDADDALD R5, (R6), R7\n\tLDCLRALB R5, (R6), R7\n\tLDORALD R5, (RSP), R7\n\tSWPALW R5, (R6), R7\n")
want := []uint32{
0xc8dfff76, // LDAR R22, (R27)
0x08dfff22, // LDARB R2, (R25)
0x88dffd9d, // LDARW R29, (R12)
0xc89fff03, // STLR R3, (R24)
0x089ffecb, // STLRB R11, (R22)
0xc8e5fcc7, // CASALD R7, (R6), R5
0xf8e500c7, // LDADDALD R7, (R6), R5
0x38e510c7, // LDCLRALB R7, (R6), R5
0xf8e533e7, // LDORALD R7, (RSP), R5
0xb8e580c7, // SWPALW R7, (R6), R5
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64System pins BRK, SVC, the barriers, cache maintenance and the
// system register accesses.
func TestArm64System(t *testing.T) {
got := arm64Words(t, "\tBRK $35943\n\tBRK\n\tSVC $7165\n\tDMB $1\n\tDSB $1\n\tISB $15\n"+
"\tDC ZVA, R4\n\tDC IVAC, R1\n\tMRS DCZID_EL0, R3\n\tMRS CNTVCT_EL0, R0\n\tMSR $9, DAIFSet\n\tMSR $3, SPSel\n"+
"\tPRFM (R0), PLDL1KEEP\n\tPRFM (R3), PLDL3KEEP\n\tPRFM (R2), $25\n")
want := []uint32{
0xd4318ce0, // BRK $35943
0xd4200000, // BRK
0xd4037fa1, // SVC $7165
0xd50331bf, // DMB $1
0xd503319f, // DSB $1
0xd5033fdf, // ISB $15
0xd50b7424, // DC ZVA, R4
0xd5087621, // DC IVAC, R1
0xd53b00e3, // MRS DCZID_EL0, R3
0xd53be040, // MRS CNTVCT_EL0, R0
0xd50349df, // MSR $9, DAIFSet
0xd50043bf, // MSR $3, SPSel
0xf9800000, // PRFM (R0), PLDL1KEEP
0xf9800064, // PRFM (R3), PLDL3KEEP
0xf9800059, // PRFM (R2), $25
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64Crypto pins the AES and SHA families.
func TestArm64Crypto(t *testing.T) {
got := arm64Words(t, "\tAESE V31.B16, V29.B16\n\tAESD V22.B16, V19.B16\n\tAESIMC V12.B16, V27.B16\n\tAESMC V14.B16, V28.B16\n"+
"\tSHA1C V8.S4, V8, V2\n\tSHA1H V17, V25\n\tSHA1P V3.S4, V20, V27\n\tSHA1SU0 V17.S4, V13.S4, V16.S4\n\tSHA1SU1 V24.S4, V23.S4\n"+
"\tSHA256H V4.S4, V2, V11\n\tSHA256H2 V6.S4, V16, V11\n\tSHA256SU0 V0.S4, V16.S4\n\tSHA256SU1 V31.S4, V3.S4, V15.S4\n"+
"\tSHA512H V2.D2, V1, V0\n\tSHA512H2 V4.D2, V3, V2\n\tSHA512SU0 V9.D2, V8.D2\n\tSHA512SU1 V7.D2, V6.D2, V5.D2\n")
want := []uint32{
0x4e284bfd, // AESE
0x4e285ad3, // AESD
0x4e28799b, // AESIMC
0x4e2869dc, // AESMC
0x5e080102, // SHA1C
0x5e280a39, // SHA1H
0x5e03129b, // SHA1P
0x5e1131b0, // SHA1SU0
0x5e281b17, // SHA1SU1
0x5e04404b, // SHA256H
0x5e06520b, // SHA256H2
0x5e282810, // SHA256SU0
0x5e1f606f, // SHA256SU1
0xce628020, // SHA512H
0xce648462, // SHA512H2
0xcec08128, // SHA512SU0
0xce6788c5, // SHA512SU1
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLogical pins the arrangement-aware three- and two-register
// SIMD paths.
func TestArm64SIMDLogical(t *testing.T) {
got := arm64Words(t, "\tVADD V1.B16, V2.B16, V3.B16\n\tVAND V4.B16, V4.B16, V9.B16\n\tVEOR V0.B16, V1.B16, V0.B16\n"+
"\tVORR V5.B16, V4.B16, V3.B16\n\tVADDP V1.H8, V2.H8, V3.H8\n\tVZIP1 V16.H8, V3.H8, V19.H8\n\tVZIP2 V22.D2, V25.D2, V21.D2\n"+
"\tVCMEQ V24.S4, V13.S4, V12.S4\n\tVCMEQ $0, V2.H4, V3.H4\n\tVREV32 V2.H8, V1.H8\n\tVREV64 V2.S4, V3.S4\n\tVUADDLV V31.S4, V11\n"+
"\tVPMULL V2.D1, V1.D1, V3.Q1\n\tVPMULL2 V2.B16, V1.B16, V4.H8\n\tVRAX1 V26.D2, V29.D2, V30.D2\n\tVMOV V2.B16, V4.B16\n")
want := []uint32{
0x4e218443, // VADD 16B
0x4e241c89, // VAND
0x6e201c20, // VEOR
0x4ea51c83, // VORR
0x4e61bc43, // VADDP 8H
0x4e503873, // VZIP1 8H
0x4ed67b35, // VZIP2 2D
0x6eb88dac, // VCMEQ 4S
0x0e609843, // VCMEQ $0, 4H
0x6e600841, // VREV32 8H
0x4ea00843, // VREV64 4S
0x6eb03beb, // VUADDLV 4S
0x0ee2e023, // VPMULL D1
0x4e22e024, // VPMULL2 16B
0xce7a8fbe, // VRAX1 2D
0x4ea21c44, // VMOV 16B pair
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDWide pins the four-register crypto group, VXAR, VEXT and the
// shift-by-immediate encodings.
func TestArm64SIMDWide(t *testing.T) {
got := arm64Words(t, "\tVEOR3 V2.B16, V7.B16, V12.B16, V25.B16\n\tVBCAX V1.B16, V2.B16, V26.B16, V31.B16\n"+
"\tVXAR $63, V27.D2, V21.D2, V26.D2\n\tVEXT $4, V2.B8, V1.B8, V3.B8\n\tVEXT $8, V2.B16, V1.B16, V3.B16\n"+
"\tVSHL $7, V22.D2, V25.D2\n\tVUSHR $6, V22.H8, V23.H8\n\tVSRI $24, V1.S4, V2.S4\n")
want := []uint32{
0xce070999, // VEOR3
0xce22075f, // VBCAX
0xce9bfeba, // VXAR
0x2e022023, // VEXT B8
0x6e024023, // VEXT B16
0x4f4756d9, // VSHL D2 $7
0x6f1a06d7, // VUSHR H8 $6
0x6f284422, // VSRI S4 $24
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDElement pins VDUP and the VMOV element forms.
func TestArm64SIMDElement(t *testing.T) {
got := arm64Words(t, "\tVDUP V31.B[15], V18\n\tVDUP V19.S[3], V18.S4\n\tVDUP V1.D[1], V2.D2\n"+
"\tVMOV V13.S[0], R20\n\tVMOV V11.B[11], V16.B[12]\n\tVMOV R20, V21.B[2]\n")
want := []uint32{
0x5e1f07f2, // VDUP element to register
0x4e1c0672, // VDUP element across S4
0x4e180422, // VDUP element across D2
0x0e043db4, // VMOV element to register
0x6e195d70, // VMOV element to element
0x4e051e95, // VMOV register into element
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLoadStore pins the structure loads and stores.
func TestArm64SIMDLoadStore(t *testing.T) {
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
"\tVLD1.P 32(R1), [V2.B16, V3.B16]\n\tVST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)\n\tVST1.P [V2.B16], (R1)\n"+
"\tVLD1R (R1), [V9.B8]\n\tVLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]\n")
want := []uint32{
0x4c407055, // VLD1 one register
0x4c40a022, // VLD1 two registers
0x0c402fae, // VLD1 four registers D1
0x4cdfa022, // VLD1.P two registers
0x4c0029c2, // VST1 four registers S4
0x4c9f7022, // VST1.P one register
0x0d40c029, // VLD1R
0x0d60e000, // VLD4R
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64MoviLiteral pins the VMOVS/VMOVD/VMOVQ constant loads: three
// words each (ADRP, ADD, wide load) plus the pooled literal in the data
// section.
func TestArm64MoviLiteral(t *testing.T) {
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tVMOVS $0x80402010, V11\n\tVMOVD $0x8040201008040201, V20\n" +
"\tVMOVQ $0x7040201008040201, $0x8040201008040201, V10\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12*3+4 {
t.Errorf("func size = %d, want %d", img.Funcs[0].Size, 12*3+4)
}
want := []uint32{
0x9000001b, 0x9100037b, 0xbd40036b, // VMOVS: ADRP, ADD, LDR S
0x9000001b, 0x9100037b, 0xfd400374, // VMOVD: ADRP, ADD, LDR D
0x9000001b, 0x9100037b, 0x3dc0036a, // VMOVQ: ADRP, ADD, LDR Q
0xd65f03c0,
}
got := leWords(img.Code)
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
// The literals sit in the data section.
var found32, found64, found128 bool
for _, d := range img.DataSyms {
switch d.Name {
case "$i32.80402010":
found32 = d.Size == 4
case "$i64.8040201008040201":
found64 = d.Size == 8
case "$i128.80402010080402017040201008040201":
found128 = d.Size == 16
}
}
if !found32 || !found64 || !found128 {
t.Errorf("literals missing: i32=%v i64=%v i128=%v", found32, found64, found128)
}
}
// TestArm64MOVK pins standalone MOVK with the hw field derived from the
// chunk position.
func TestArm64MOVK(t *testing.T) {
got := arm64Words(t, "\tMOVK $1234, R5\n\tMOVK $305397760, R5\n\tMOVKW $1234, R5\n")
want := []uint32{
0xf2809a45, // MOVK hw=0
0xf2a24685, // MOVK hw=1
0x72809a45, // MOVKW hw=0
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
@@ -851,20 +1296,170 @@ func TestArm64ExclNoOffset(t *testing.T) {
}
}
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
// rejected instead of wrapping through int32.
func TestArm64AddSubImmRange(t *testing.T) {
for _, body := range []string{
"\tADD $0x100000000, R0, R1\n",
"\tSUB $-0x100000000, R0, R1\n",
"\tCMP $0x100000000, R0\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
// TestArm64AddSubImmWide pins the wide-immediate classification the toolchain
// applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split
// into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP,
// and the MOVZ/MOVN/MOVK materialisations followed by the register form.
// Comparisons never split, and the W forms classify the 32-bit value. Every
// word is go tool asm's own for the same source.
func TestArm64AddSubImmWide(t *testing.T) {
got := arm64Words(t, strings.Join([]string{
"\tADD $0xaaaaaa, R2, R3",
"\tSUB $0xaaaaaa, R2",
"\tADD $0x186a0, R2, R5",
"\tADD $0x1ffe00, R2, R3",
"\tADD $0x3fffffffc000, R5",
"\tADD $-100000, R2, R3",
"\tADD $-2048, R2, R3",
"\tCMP $0xaaaaaa, R2",
"\tCMP $0xffffffffffa0, R3",
"\tCMPW $27745, R2",
"\tCMPW $0x60060, R2",
"\tADDS $0xaaaaaa, R2, R3",
"\tADD $0x12345678, R2, R3",
"\tADDW $0x60060, R2",
"\tSUB $0xe7791f700, R3, R1",
"\tADDW $0x12345678, R2, R3",
"\tCMN $0x1000000, R2",
}, "\n")+"\n")
want := []uint32{
0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split
0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn
0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split
0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split
0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP
0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK
0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD
0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK
0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup
0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW
0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split
0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS
0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK
0x11018042, 0x11418042, // ADDW $0x60060: W split
0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700
0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW
0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i])
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
// TestArm64CarryImmWide pins the carry family's $0 spellings in two and
// three operands, the ROR shift on the logical group (and its rejection for
// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector
// alias with an element selector. Words are go tool asm's own.
func TestArm64CarryShiftAlias(t *testing.T) {
got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+
"\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+
"\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n")
want := []uint32{
0x9a1f0294, // ADC ZR, R20, R20
0x9a1f0284, // ADC ZR, R20, R4
0xfa1f008c, // SBCS ZR, R4, R12
0xfa0f008c, // SBCS R15, R4, R12
0x0ac91e7a, // ANDW R9 ROR 7, R19, R26
0x8ac18443, // AND R1 ROR 33, R2, R3
0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1
0xda0203e7, // SBC ZR, R7, R2
0x9b0eff77, // MSUB ZR, R27, R14, R23
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i])
}
}
// ROR on an arithmetic form is unallocated: the toolchain reports an
// unsupported shift operator.
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Error("ADD R1@>33: expected an error, got none")
}
}
// TestArm64VecAliasElement pins the register-alias rewrite inside a vector
// operand with an element selector and inside a split register list: the
// aliases resolve textually where the parser carries the selector apart from
// the name. Words are go tool asm's own.
func TestArm64VecAliasElement(t *testing.T) {
src := `#include "textflag.h"
#define POLY V15
#define ACC0 V8
#define ACC1 V9
TEXT ·f(SB), NOSPLIT, $0-0
VMOV R1, POLY.D[0]
VEOR POLY.B16, POLY.B16, POLY.B16
VLD1 (R0), [ACC0.B16]
VLD1.P (R0), [ACC0.B16, ACC1.B16]
VST1.P [ACC0.B16, ACC1.B16], 32(R1)
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
want := []uint32{
0x4e081c2f, // INS V15.D[0], R1
0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16
0x4c407008, // VLD1 (R0), [V8.B16]
0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16]
0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1)
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies
// once the value leaves every imm12 form: a constant sequence into REGTMP
// (R27) followed by the register form. SUB $-0x100000000 is a bitmask
// immediate, so it rides the ORR form; the others take MOVZ. Words are go
// tool asm's own.
func TestArm64AddSubImmBeyond32(t *testing.T) {
got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n")
want := []uint32{
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
0x8b1b0001, // ADD R27, R0, R1
0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask)
0xcb1b0001, // SUB R27, R0, R1
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
0xeb1b001f, // CMP R27, R0
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
+55
View File
@@ -67,6 +67,9 @@ type spadjStep struct {
// patch sites (for the file-level layout to resolve), the label table and the
// stack-adjustment boundaries.
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
if err := checkAdjspBalance(t); err != nil {
return nil, nil, nil, nil, nil, err
}
fi := computeFrame(t)
chain := jumpChain(t)
resolve := func(name string) string {
@@ -203,6 +206,14 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
)
}
// frameBase is the SP delta the prologue leaves: 8 for the saved base
// pointer plus the frame, 0 frameless. bodyDelta tracks the ADJSP
// statements' straight-line sum, so a mid-body step's value is the
// frame base plus what the body has opened so far.
frameBase, bodyDelta := 0, 0
if fi.useFP {
frameBase = 8 + fi.size
}
pos := guardLen + len(fi.prologue)
for i, stmt := range t.Body {
s, ok := stmt.(*ast.Instr)
@@ -230,6 +241,16 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
ps[k].kind = RelCall
}
}
if strings.ToUpper(s.Mnemonic.Text) == "ADJSP" && len(s.Operands) == 1 && s.Operands[0].Imm.HasVal {
// The statement shifted SP mid-body: record the new running
// delta as the value in effect from just past the instruction.
v := s.Operands[0].Imm.Val
if s.Operands[0].Imm.Neg {
v = -v
}
bodyDelta += int(v)
steps = append(steps, spadjStep{pos + len(code), frameBase + bodyDelta})
}
patches = append(patches, ps...)
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
out = append(out, code...)
@@ -412,6 +433,40 @@ func hasCall(t *ast.Text) bool {
return false
}
// checkAdjspBalance mirrors the toolchain's push/pop walk: every ADJSP
// shifts SP away from the entry state and every RET must see the shifts
// closed. The assembler's own prologue and epilogue contribute matching
// deltas on both sides, so the statements' straight-line sum must be zero
// at each RET; branches do not reset the walk, which runs over the program
// list in source order. go tool asm reports an offender as "unbalanced
// PUSH/POP" (verified against ADJSP $16 before a RET, accepted as a
// $16/$-16 pair, per-RET rather than per-function).
func checkAdjspBalance(t *ast.Text) error {
delta := 0
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
switch strings.ToUpper(in.Mnemonic.Text) {
case "ADJSP":
if len(in.Operands) != 1 || !in.Operands[0].Imm.HasVal {
continue // reported during emission
}
v := in.Operands[0].Imm.Val
if in.Operands[0].Imm.Neg {
v = -v
}
delta += int(v)
case "RET":
if delta != 0 {
return fmt.Errorf("unbalanced PUSH/POP")
}
}
}
return nil
}
// guardLen returns the byte length of the stack-split guard prefix. The
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
// short form and 6 in the long form.
+129
View File
@@ -439,3 +439,132 @@ func TestSubSPEncodings(t *testing.T) {
}
}
}
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
// asserts the three behaviours the toolchain shows: each prefix statement is
// a standalone byte with a PC of its own (so a label placed on the LOCK
// points at the F0), the data pseudo-ops write their literal bytes inline,
// and END terminates nothing (the statements after it still belong to the
// function and carry no trace of it).
func TestAssemblePseudoStatements(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·pseudo(SB), NOSPLIT, $0-0
pfx:
LOCK
CMPXCHGQ AX, (BX)
REP
MOVSQ
BYTE $0x0f
BYTE $0x1f
WORD $0x1234
END
BYTE $0x02
RET
`)
code, labels, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
want := []byte{
0xf0,
0x48, 0x0f, 0xb1, 0x03,
0xf3, 0x48, 0xa5,
0x0f, 0x1f, 0x34, 0x12,
0x02, 0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
// The label sits on the LOCK byte, exactly where the toolchain's PC
// listing puts it.
if off := labels["pfx"]; off != 0 {
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
}
// The trailing BYTE lands where the layout says: after the 8 bytes of
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
// none.
if code[12] != 0x02 {
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
}
}
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
// ADJSP: the straight-line sum of the adjustments must be zero at each
// RET, branches in between counting for nothing (verified against go tool
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
// $16/$-16 pair with a JMP in between assembles).
func TestAssembleAdjspBalance(t *testing.T) {
// Balanced pair with a branch in between, bytes pinned from go tool asm.
fn := firstText(t, `
#include "textflag.h"
TEXT ·adjsp(SB), NOSPLIT, $0-0
ADJSP $16
JMP body
body:
ADJSP $-16
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
if hexBytes(code) != hexBytes(want) {
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
// Unbalanced at the RET: the toolchain diagnoses, so must we.
_, _, err = Assemble(firstText(t, `
#include "textflag.h"
TEXT ·unbalanced(SB), NOSPLIT, $0-0
ADJSP $16
RET
`))
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
}
// The check runs per RET: a closed pair before the first RET does not
// excuse an open adjustment before the second.
_, _, err = Assemble(firstText(t, `
#include "textflag.h"
TEXT ·tworet(SB), NOSPLIT, $0-0
ADJSP $8
ADJSP $-8
RET
mid:
ADJSP $8
RET
`))
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
}
// A framed function: the assembler's own prologue and epilogue
// contribute matching deltas, so the pair in the body still balances,
// and the bytes match go tool asm end to end.
fn = firstText(t, `
#include "textflag.h"
TEXT ·framed(SB), $16-8
ADJSP $8
ADJSP $-8
RET
`)
code, _, err = Assemble(fn)
if err != nil {
t.Fatalf("Assemble framed: %v", err)
}
want = []byte{
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
0xC3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
+35 -8
View File
@@ -18,7 +18,14 @@ func Encodable(mnemonic string) bool {
// Fixed-name instructions (no size suffix).
switch upper {
case "RET", "NOP", "CALL", "JMP":
case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The literal-data pseudo-ops, the accepted-and-ignored END and the
// SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP":
return true
}
if _, ok := noOperandTable[upper]; ok {
return true
}
if _, ok := condCode(upper); ok {
@@ -31,7 +38,7 @@ func Encodable(mnemonic string) bool {
return false
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" {
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return true
}
@@ -53,13 +60,28 @@ func Encodable(mnemonic string) bool {
}
}
// Legacy SSE shuffles and packed binaries dispatch on the full name.
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
// do the imm8-controlled instructions, the lane extracts and inserts and
// the packed integer shifts (their trailing width letters belong to the
// mnemonic).
if _, ok := sseShufTable[upper]; ok {
return true
}
if _, ok := sseBinTable[upper]; ok {
return true
}
if _, ok := sseImm3Table[upper]; ok {
return true
}
if _, ok := sseExtractTable[upper]; ok {
return true
}
if _, ok := sseInsertTable[upper]; ok {
return true
}
if _, ok := sseShiftImm[upper]; ok {
return true
}
// The size-suffix split: retry the tables and the scalar switch on the
// base.
@@ -74,12 +96,15 @@ func Encodable(mnemonic string) bool {
}
}
switch base2 {
case "MOV",
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
case "MOV", "MOVD",
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
"TEST",
"LEA",
"INC", "DEC", "NEG", "NOT",
"SHL", "SHR", "SAR",
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
"BT", "BTS", "BTR", "BTC",
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
"MOVS", "STOS",
"IMUL", "IMUL3",
"PUSH", "POP",
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
@@ -88,7 +113,9 @@ func Encodable(mnemonic string) bool {
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
"CVTSL2SD", "CVTSQ2SD",
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
"FMOVD",
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return true
}
// Full-name dispatches the size split would eat (a trailing width
+159 -7
View File
@@ -58,6 +58,51 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if cc, ok := condCode(upper); ok {
return e.encodeJcc(cc, ops)
}
// No-operand system and string-control instructions (CPUID, RDTSC,
// SYSCALL, the fences, UNDEF, …).
if op, ok := noOperandTable[upper]; ok {
if len(ops) != 0 {
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
}
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper {
case "POPFQ":
if len(ops) != 0 {
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
}
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
case "PUSHFQ":
if len(ops) != 0 {
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
}
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT":
return e.encodeInt(ops)
case "LDMXCSR":
return e.encodeMxcsr(2, ops)
case "STMXCSR":
return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm).
case "CMPSD":
return e.encodeCmpsd(ops)
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2":
return e.encodeSha256rnds2(ops)
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
// the SUBQ and ADDQ forms.
case "BYTE", "WORD", "LONG", "QUAD":
return e.encodeData(upper, ops)
case "END":
return e.encodeEnd(ops)
case "ADJSP":
return e.encodeAdjsp(ops)
}
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
@@ -67,7 +112,8 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if err != nil {
return err
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return e.encodeVec(base, ops, sfx)
}
if sfx.any() {
@@ -101,6 +147,21 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if m, ok := sseBinTable[base]; ok {
return e.encodeSSEBin(m, ops)
}
// The imm8-controlled legacy instructions, the lane extracts and inserts
// and the packed integer shifts all dispatch on the full name: a trailing
// width letter here belongs to the mnemonic, not to the size split.
if m, ok := sseImm3Table[upper]; ok {
return e.encodeSSEImm3(m, ops)
}
if m, ok := sseExtractTable[upper]; ok {
return e.encodeSSEExtract(m, ops)
}
if m, ok := sseInsertTable[upper]; ok {
return e.encodeSSEInsert(m, ops)
}
if _, ok := sseShiftImm[upper]; ok {
return e.encodeSSEShift(upper, ops)
}
// PMOVMSKB ends in a width letter the size split would eat, so it
// dispatches on the full name like the packed binaries above.
if upper == "PMOVMSKB" {
@@ -109,16 +170,36 @@ func (e *enc) encode(mnem string, ops []Operand) error {
switch base {
case "MOV":
return e.encodeMov(ops, size)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
// REX.W and all.
case "MOVD":
return e.encodeMov(ops, 8)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
return e.encodeALU(aluOp[base], ops, size)
case "TEST":
return e.encodeTest(ops, size)
case "LEA":
return e.encodeLea(ops, size)
case "INC", "DEC", "NEG", "NOT":
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
return e.encodeUnary(unaryOp[base], ops, size)
case "SHL", "SHR", "SAR":
return e.encodeShift(shiftOp[base], ops, size)
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
return e.encodeShift(base, ops, size)
case "BT", "BTS", "BTR", "BTC":
return e.encodeBitTest(base, ops, size)
case "XCHG":
return e.encodeExchange(ops, size)
case "CMPXCHG":
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
case "XADD":
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
case "CRC32":
return e.encodeCrc32(ops, size)
case "ADCX":
return e.encodeCarryExt(0x66, ops, size)
case "ADOX":
return e.encodeCarryExt(0xF3, ops, size)
case "MOVS", "STOS":
return e.encodeStringOp(base, ops, size)
case "IMUL", "IMUL3":
return e.encodeImul(ops, size)
case "PUSH":
@@ -136,7 +217,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD":
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
return e.encodeCvtInt(base, ops, size)
case "FMOVD":
return e.encodeFmov(ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return e.encodeSSEMove(sseMoveTable[base], ops)
}
return fmt.Errorf("unsupported instruction %q", mnem)
@@ -166,6 +251,73 @@ var prefetchVariant = map[string]int{
"PREFETCHT2": 3,
}
// dataWidth is the literal byte count of each data-emission pseudo-op.
var dataWidth = map[string]int{
"BYTE": 1,
"WORD": 2,
"LONG": 4,
"QUAD": 8,
}
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
// little-endian, with no opcode lookup. The value is truncated to the
// width rather than range-checked, exactly as go tool asm behaves (BYTE
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
// exactly one immediate is accepted: the toolchain rejects a list such as
// BYTE $1, $2, $3.
func (e *enc) encodeData(mnem string, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("%s requires an integer immediate", mnem)
}
width := dataWidth[mnem]
out := make([]byte, width)
u := uint64(imm)
for i := range width {
out[i] = byte(u >> (8 * i))
}
e.out = append(e.out, out...)
return nil
}
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
// entirely: the AEND Prog is skipped when the program list is flushed, so
// the statements after an END still belong to the same function and the
// encoded body carries no trace of it, whatever operands follow the name
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
func (e *enc) encodeEnd(ops []Operand) error {
return nil
}
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
// picks (the same selection subSP and addSP make for the frame). go tool
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
// statement's effect on the SP balance is checked by the function-level
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
func (e *enc) encodeAdjsp(ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("ADJSP requires an integer immediate")
}
switch v := int(imm); {
case v > 0:
e.out = append(e.out, subSP(v)...)
case v < 0:
e.out = append(e.out, addSP(-v)...)
default:
return fmt.Errorf("ADJSP $0 has no encoding")
}
return nil
}
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
func splitSize(upper string) (base string, size int) {
if upper == "" {
@@ -195,7 +347,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
if ss, ok := scatterTable[upper]; ok {
return e.encodeScatter(upper, ss, ops, sfx)
}
if upper == "KMOVW" || upper == "KMOVQ" {
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
+437
View File
@@ -200,10 +200,60 @@ func TestUnary(t *testing.T) {
func TestShift(t *testing.T) {
checkSyntax(t, "shl rdx, 0x2", "SHLQ", Imm(2), DX)
checkSyntax(t, "shl rdx, cl", "SHLQ", CL, DX)
checkSyntax(t, "shl rdx, cl", "SHLQ", CX, DX)
checkSyntax(t, "shl rdx, 0x1", "SHLQ", Imm(1), DX)
checkSyntax(t, "sar rcx, 0x1f", "SARQ", Imm(31), CX)
}
// TestDoubleShift pins the three-operand SHL/SHR form, which encodes as
// SHLD/SHRD: go tool asm accepts it for SHL/SHR at W/L/Q widths and rejects
// it for SAR, SAL, the rotates and the B width. The byte pins mirror the
// oracle's objdump output (48 0f a4 fe 0d for the first case, and so on).
func TestDoubleShift(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string // hex encoding
}{
{"SHLQ imm", "SHLQ", []Operand{Imm(0x0d), DI, SI}, "480fa4fe0d"},
{"SHLQ CX high regs", "SHLQ", []Operand{CX, Reg{idx: 8, size: 8}, Reg{idx: 9, size: 8}}, "4d0fa5c1"},
{"SHRQ imm", "SHRQ", []Operand{Imm(1), AX, CX}, "480facc101"},
{"SHLW imm", "SHLW", []Operand{Imm(1), AX, CX}, "660fa4c101"},
{"SHRD CL", "SHRQ", []Operand{CL, AX, CX}, "480fadc1"},
{"SHLD imm high regs", "SHLQ", []Operand{Imm(2), Reg{idx: 10, size: 8}, Reg{idx: 11, size: 8}}, "4d0fa4d302"},
{"SHRD imm max", "SHRQ", []Operand{Imm(63), Reg{idx: 9, size: 8}, Reg{idx: 15, size: 8}}, "4d0faccf3f"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
// Rejected forms: the oracle rejects every one of these.
rejected := []struct {
name string
mnem string
ops []Operand
}{
{"SARQ three operands", "SARQ", []Operand{Imm(1), AX, CX}},
{"SALQ three operands", "SALQ", []Operand{Imm(1), AX, CX}},
{"ROLQ three operands", "ROLQ", []Operand{Imm(1), AX, CX}},
{"SHLB three operands", "SHLB", []Operand{Imm(1), AL, CL}},
{"SHRQ memory source", "SHRQ", []Operand{Imm(1), Ptr(AX, 0, 8), CX}},
{"SHRQ ECX count", "SHRQ", []Operand{Reg{idx: 1, size: 4}, AX, CX}},
}
for _, c := range rejected {
if _, err := Encode(c.mnem, c.ops...); err == nil {
t.Errorf("%s: Encode succeeded, want rejection", c.name)
}
}
}
func TestImul(t *testing.T) {
checkSyntax(t, "imul rdx, rcx", "IMULQ", CX, DX)
checkSyntax(t, "imul edx, edx, 0x3", "IMULL", Imm(3), DX, DX)
@@ -277,6 +327,12 @@ func TestSSEMoveGroundTruth(t *testing.T) {
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
// Static-symbol (SB) references: the GOROOT crypto kernels load and
// store octa constants by name (MOVOU bswapMask<>+0(SB), X0).
{"MOVOU sym,X0", "MOVOU", []Operand{sbMem{size: 16, name: "bswapMask"}, vreg(t, "X0")}, "f30f6f0500000000", "MOVDQU"},
{"MOVOU X0,sym+8", "MOVOU", []Operand{vreg(t, "X0"), sbMem{size: 16, name: "bswapMask", addend: 8}}, "f30f7f0500000000", "MOVDQU"},
{"MOVO sym,X1", "MOVO", []Operand{sbMem{size: 16, name: "gcmPoly"}, vreg(t, "X1")}, "660f6f0d00000000", "MOVDQA"},
{"MOVO X2,sym", "MOVO", []Operand{vreg(t, "X2"), sbMem{size: 16, name: "gcmPoly"}}, "660f7f1500000000", "MOVDQA"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
@@ -546,6 +602,248 @@ func TestEncodableCmovSize(t *testing.T) {
}
}
// TestCarryShiftMulGroundTruth pins the carry-flag ALU family (ADC/SBB with
// their accumulator immediate forms), the rotate family, MUL/DIV/IDIV and the
// bit-test family byte for byte against go tool asm (see
// testdata/verify/scalar_amd64.s).
func TestCarryShiftMulGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ADCQ AX,BX", "ADCQ", []Operand{AX, BX}, "4811c3"},
{"ADCL AX,BX", "ADCL", []Operand{AX, BX}, "11c3"},
{"ADCB AL,BL", "ADCB", []Operand{AL, BL}, "10c3"},
{"ADCW AX,BX", "ADCW", []Operand{AX, BX}, "6611c3"},
{"SBBQ AX,BX", "SBBQ", []Operand{AX, BX}, "4819c3"},
{"ADCQ $5,BX", "ADCQ", []Operand{Imm(5), BX}, "4883d305"},
{"ADCQ $300,BX", "ADCQ", []Operand{Imm(300), BX}, "4881d32c010000"},
{"ADCQ $300,AX", "ADCQ", []Operand{Imm(300), AX}, "48152c010000"},
{"ADCB $5,AL", "ADCB", []Operand{Imm(5), AL}, "1405"},
{"SBBQ $300,AX", "SBBQ", []Operand{Imm(300), AX}, "481d2c010000"},
{"ADCQ AX,(BX)", "ADCQ", []Operand{AX, Ptr(BX, 0, 8)}, "481103"},
{"ROLQ $3,AX", "ROLQ", []Operand{Imm(3), AX}, "48c1c003"},
{"ROLL CX,BX", "ROLL", []Operand{CL, BX}, "d3c3"},
{"RORQ CL,AX", "RORQ", []Operand{CL, AX}, "48d3c8"},
{"RCRQ $1,BX", "RCRQ", []Operand{Imm(1), BX}, "48d1db"},
{"RCLQ $3,AX", "RCLQ", []Operand{Imm(3), AX}, "48c1d003"},
{"RORB CL,BL", "RORB", []Operand{CL, BL}, "d2cb"},
{"SALQ $2,AX", "SALQ", []Operand{Imm(2), AX}, "48c1e002"},
{"ROLW $1,AX", "ROLW", []Operand{Imm(1), AX}, "66d1c0"},
{"MULQ CX", "MULQ", []Operand{CX}, "48f7e1"},
{"MULL CX", "MULL", []Operand{CX}, "f7e1"},
{"MULB CL", "MULB", []Operand{CL}, "f6e1"},
{"DIVL CX", "DIVL", []Operand{CX}, "f7f1"},
{"IDIVQ CX", "IDIVQ", []Operand{CX}, "48f7f9"},
{"MULW CX", "MULW", []Operand{CX}, "66f7e1"},
{"BTQ AX,DX", "BTQ", []Operand{AX, DX}, "480fa3c2"},
{"BTL AX,DX", "BTL", []Operand{AX, DX}, "0fa3c2"},
{"BTW AX,DX", "BTW", []Operand{AX, DX}, "660fa3c2"},
{"BTQ $3,BX", "BTQ", []Operand{Imm(3), BX}, "480fbae303"},
{"BTQ $3,(AX)", "BTQ", []Operand{Imm(3), Ptr(AX, 0, 8)}, "480fba2003"},
{"BTSQ $5,BX", "BTSQ", []Operand{Imm(5), BX}, "480fbaeb05"},
{"BTCQ AX,BX", "BTCQ", []Operand{AX, BX}, "480fbbc3"},
{"BTRQ $7,BX", "BTRQ", []Operand{Imm(7), BX}, "480fbaf307"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// The bit-test immediate is an unsigned bit index with the negative
// spelling accepted, the shuffle convention: BTQ $300 must be rejected.
if _, err := Encode("BTQ", Imm(300), AX); err == nil {
t.Errorf("BTQ $300: expected an error, got none")
}
}
// TestAtomicSystemGroundTruth pins the exchange/compare-exchange/accumulate
// family, the string primitives, the flag and system instructions, the MXCSR
// pair, the scalar float-to-int conversions and the x87 FMOVD byte for byte
// against go tool asm (see testdata/verify/atomics_amd64.s and
// testdata/verify/system_amd64.s).
func TestAtomicSystemGroundTruth(t *testing.T) {
r8 := Reg{idx: 8, size: 8}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"XCHGQ AX,BX", "XCHGQ", []Operand{AX, BX}, "4893"},
{"XCHGQ BX,AX", "XCHGQ", []Operand{BX, AX}, "4893"},
{"XCHGL AX,BX", "XCHGL", []Operand{AX, BX}, "93"},
{"XCHGB AL,BL", "XCHGB", []Operand{AL, BL}, "86c3"},
{"XCHGW AX,BX", "XCHGW", []Operand{AX, BX}, "6693"},
{"XCHGQ R8,R9", "XCHGQ", []Operand{r8, Reg{idx: 9, size: 8}}, "4d87c1"},
{"XCHGQ BX,(AX)", "XCHGQ", []Operand{BX, Ptr(AX, 0, 8)}, "488718"},
{"XCHGQ (AX),BX", "XCHGQ", []Operand{Ptr(AX, 0, 8), BX}, "488718"},
{"XCHGQ AX,(BX)", "XCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "488703"},
{"CMPXCHGL AX,BX", "CMPXCHGL", []Operand{AX, BX}, "0fb1c3"},
{"CMPXCHGQ AX,(BX)", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
{"CMPXCHGB AL,(BX)", "CMPXCHGB", []Operand{AL, Ptr(BX, 0, 1)}, "0fb003"},
{"CMPXCHGW AX,BX", "CMPXCHGW", []Operand{AX, BX}, "660fb1c3"},
{"XADDL AX,BX", "XADDL", []Operand{AX, BX}, "0fc1c3"},
{"XADDQ AX,(BX)", "XADDQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fc103"},
{"XADDB AL,(BX)", "XADDB", []Operand{AL, Ptr(BX, 0, 1)}, "0fc003"},
{"XADDW AX,BX", "XADDW", []Operand{AX, BX}, "660fc1c3"},
{"ADCXL AX,CX", "ADCXL", []Operand{AX, CX}, "660f38f6c8"},
{"ADCXQ AX,CX", "ADCXQ", []Operand{AX, CX}, "66480f38f6c8"},
{"ADOXL AX,CX", "ADOXL", []Operand{AX, CX}, "f30f38f6c8"},
{"ADOXQ AX,CX", "ADOXQ", []Operand{AX, CX}, "f3480f38f6c8"},
{"CRC32B AX,CX", "CRC32B", []Operand{AX, CX}, "f20f38f0c8"},
{"CRC32W AX,CX", "CRC32W", []Operand{AX, CX}, "66f20f38f1c8"},
{"CRC32L AX,CX", "CRC32L", []Operand{AX, CX}, "f20f38f1c8"},
{"CRC32Q AX,CX", "CRC32Q", []Operand{AX, CX}, "f2480f38f1c8"},
{"CRC32L (AX),CX", "CRC32L", []Operand{Ptr(AX, 0, 4), CX}, "f20f38f108"},
{"MOVSQ", "MOVSQ", []Operand{}, "48a5"},
{"MOVSL", "MOVSL", []Operand{}, "a5"},
{"MOVSB", "MOVSB", []Operand{}, "a4"},
{"MOVSW", "MOVSW", []Operand{}, "66a5"},
{"STOSB", "STOSB", []Operand{}, "aa"},
{"STOSQ", "STOSQ", []Operand{}, "48ab"},
{"STOSL", "STOSL", []Operand{}, "ab"},
{"STOSW", "STOSW", []Operand{}, "66ab"},
{"CLD", "CLD", []Operand{}, "fc"},
{"STD", "STD", []Operand{}, "fd"},
{"POPFQ", "POPFQ", []Operand{}, "9d"},
{"PUSHFQ", "PUSHFQ", []Operand{}, "9c"},
{"CPUID", "CPUID", []Operand{}, "0fa2"},
{"RDTSC", "RDTSC", []Operand{}, "0f31"},
{"RDTSCP", "RDTSCP", []Operand{}, "0f01f9"},
{"SYSCALL", "SYSCALL", []Operand{}, "0f05"},
{"XGETBV", "XGETBV", []Operand{}, "0f01d0"},
{"PAUSE", "PAUSE", []Operand{}, "f390"},
{"LFENCE", "LFENCE", []Operand{}, "0faee8"},
{"MFENCE", "MFENCE", []Operand{}, "0faef0"},
{"SFENCE", "SFENCE", []Operand{}, "0faef8"},
{"UNDEF", "UNDEF", []Operand{}, "0f0b"},
{"INT $3", "INT", []Operand{Imm(3)}, "cd03"},
{"LDMXCSR (AX)", "LDMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae10"},
{"STMXCSR (AX)", "STMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae18"},
{"CVTSD2SL X0,AX", "CVTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2dc0"},
{"CVTTSD2SQ X0,AX", "CVTTSD2SQ", []Operand{vreg(t, "X0"), AX}, "f2480f2cc0"},
{"CVTTSD2SL X0,AX", "CVTTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2cc0"},
{"CVTSS2SQ X0,AX", "CVTSS2SQ", []Operand{vreg(t, "X0"), AX}, "f3480f2dc0"},
{"FMOVD (AX),F0", "FMOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "F0")}, "dd00"},
{"FMOVD F0,(AX)", "FMOVD", []Operand{vreg(t, "F0"), Ptr(AX, 0, 8)}, "dd10"},
{"FMOVD F0,F1", "FMOVD", []Operand{vreg(t, "F0"), vreg(t, "F1")}, "ddd1"},
{"MOVD AX,X0", "MOVD", []Operand{AX, vreg(t, "X0")}, "66480f6ec0"},
{"MOVD X0,AX", "MOVD", []Operand{vreg(t, "X0"), AX}, "66480f7ec0"},
{"MOVD X0,X1", "MOVD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f30f7ec8"},
{"MOVD (AX),X0", "MOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "X0")}, "f30f7e00"},
{"MOVD X0,(AX)", "MOVD", []Operand{vreg(t, "X0"), Ptr(AX, 0, 8)}, "660fd600"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// LDMXCSR/STMXCSR take a memory operand only.
if _, err := Encode("LDMXCSR", AX); err == nil {
t.Errorf("LDMXCSR AX: expected an error, got none")
}
}
// TestSSEGapsGroundTruth pins the legacy SSE gap families: the scalar
// compare and square root, the Plan 9 packed spellings, the imm8-controlled
// shuffles, the lane extracts and inserts, the packed integer shifts and the
// AES/SHA round instructions, byte for byte against go tool asm (see
// testdata/verify/crypto_amd64.s and testdata/verify/sse_amd64.s).
func TestSSEGapsGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ANDNPD X0,X1", "ANDNPD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f55c8"},
{"ANDNPS X0,X1", "ANDNPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f55c8"},
{"COMISD X0,X1", "COMISD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f2fc8"},
{"SQRTSD X0,X1", "SQRTSD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f20f51c8"},
{"PSHUFL $3,X0,X1", "PSHUFL", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f70c803"},
{"PALIGNR $2,X0,X1", "PALIGNR", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0fc802"},
{"PBLENDW $3,X0,X1", "PBLENDW", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0ec803"},
{"PCMPESTRI $1,X0,X1", "PCMPESTRI", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}, "660f3a61c801"},
{"PCLMULQDQ $0,X0,X1", "PCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3a44c800"},
{"PCLMULQDQ $0,(AX),X1", "PCLMULQDQ", []Operand{Imm(0), Ptr(AX, 0, 16), vreg(t, "X1")}, "660f3a440800"},
{"PEXTRB $1,X0,AX", "PEXTRB", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a14c001"},
{"PEXTRD $1,X0,AX", "PEXTRD", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a16c001"},
{"PEXTRQ $1,X0,AX", "PEXTRQ", []Operand{Imm(1), vreg(t, "X0"), AX}, "66480f3a16c001"},
{"PEXTRW $1,X0,AX", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), AX}, "660fc5c001"},
{"PEXTRW $1,X0,(AX)", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), Ptr(AX, 0, 2)}, "660f3a150001"},
{"PINSRB $1,AX,X0", "PINSRB", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a20c001"},
{"PINSRD $1,AX,X0", "PINSRD", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a22c001"},
{"PINSRQ $1,AX,X0", "PINSRQ", []Operand{Imm(1), AX, vreg(t, "X0")}, "66480f3a22c001"},
{"PINSRW $1,AX,X0", "PINSRW", []Operand{Imm(1), AX, vreg(t, "X0")}, "660fc4c001"},
{"PINSRW $1,(AX),X0", "PINSRW", []Operand{Imm(1), Ptr(AX, 0, 2), vreg(t, "X0")}, "660fc40001"},
{"PSLLL $2,X0", "PSLLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72f002"},
{"PSRAL $2,X0", "PSRAL", []Operand{Imm(2), vreg(t, "X0")}, "660f72e002"},
{"PSRLL $2,X0", "PSRLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72d002"},
{"PSRLQ $2,X0", "PSRLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d002"},
{"PSLLQ $2,X0", "PSLLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f002"},
{"PSLLW $2,X0", "PSLLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71f002"},
{"PSRLW $2,X0", "PSRLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71d002"},
{"PSRAW $2,X0", "PSRAW", []Operand{Imm(2), vreg(t, "X0")}, "660f71e002"},
{"PSLLDQ $2,X0", "PSLLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f802"},
{"PSRLDQ $2,X0", "PSRLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d802"},
{"PSLLL X0,X1", "PSLLL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ff2c8"},
{"PSRLQ X0,X1", "PSRLQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660fd3c8"},
{"PSLLL (AX),X1", "PSLLL", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660ff208"},
{"PSUBL X0,X1", "PSUBL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffac8"},
{"PADDL X0,X1", "PADDL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffec8"},
{"PCMPEQL X0,X1", "PCMPEQL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f76c8"},
{"PUNPCKLBW X0,X1", "PUNPCKLBW", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f60c8"},
{"MOVOA X0,X1", "MOVOA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f6fc8"},
{"MOVOA (AX),X1", "MOVOA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f6f08"},
{"MOVOA X0,(AX)", "MOVOA", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "660f7f00"},
{"AESIMC X0,X1", "AESIMC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dbc8"},
{"AESIMC (AX),X1", "AESIMC", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f38db08"},
{"AESENC X0,X1", "AESENC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dcc8"},
{"AESENCLAST X0,X1", "AESENCLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38ddc8"},
{"AESDEC X0,X1", "AESDEC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dec8"},
{"AESDECLAST X0,X1", "AESDECLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dfc8"},
{"AESKEYGENASSIST $0,X0,X1", "AESKEYGENASSIST", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3adfc800"},
{"SHA1MSG1 X0,X1", "SHA1MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c9c8"},
{"SHA1MSG2 X0,X1", "SHA1MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cac8"},
{"SHA1NEXTE X0,X1", "SHA1NEXTE", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c8c8"},
{"SHA1RNDS4 $0,X0,X1", "SHA1RNDS4", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "0f3accc800"},
{"SHA256MSG1 X0,X1", "SHA256MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38ccc8"},
{"SHA256MSG2 X0,X1", "SHA256MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cdc8"},
{"SHA256RNDS2 X0,X1,X2", "SHA256RNDS2", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "0f38cbd1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// SHA256RNDS2's first operand must be the literal X0.
if _, err := Encode("SHA256RNDS2", vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")); err == nil {
t.Errorf("SHA256RNDS2 X1,...: expected an error, got none")
}
// PSLLDQ has no variable-count form.
if _, err := Encode("PSLLDQ", vreg(t, "X0"), vreg(t, "X1")); err == nil {
t.Errorf("PSLLDQ X0,X1: expected an error, got none")
}
}
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
// byte for byte (no prefix / 66 / F2 / F3 variants).
func TestSSEBinGroundTruth(t *testing.T) {
@@ -627,3 +925,142 @@ func TestMOVQXMMGroundTruth(t *testing.T) {
}
}
}
// TestPrefixStatements pins LOCK, REP and REPN. go tool asm encodes each as
// a standalone one-byte instruction with a PC of its own (F0, F3, F2), not a
// prefix field merged into the following instruction, and it validates
// nothing about the pairing (LOCK before NOP assembles). The prefixed
// atomic and string shapes are the bytes the runtime's own kernels need.
func TestPrefixStatements(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"LOCK", "LOCK", nil, "f0"},
{"REP", "REP", nil, "f3"},
{"REPN", "REPN", nil, "f2"},
// LOCK; CMPXCHGQ AX, (BX)
{"LOCK CMPXCHGQ", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
// REP; MOVSQ
{"REP MOVSQ", "MOVSQ", nil, "48a5"},
// REPN; MOVSB
{"REPN MOVSB", "MOVSB", nil, "a4"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// The prefix statements take no operands, as the toolchain reports for
// LOCK AX.
if _, err := Encode("LOCK", AX); err == nil {
t.Error("LOCK AX assembled, want an error")
}
if _, err := Encode("REP", Imm(1)); err == nil {
t.Error("REP $1 assembled, want an error")
}
}
// TestDataEmission pins BYTE, WORD, LONG and QUAD: the immediate lands in
// the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
// lookup, truncated to the width rather than range-checked (go tool asm
// emits FF for BYTE $0x1FF and 45 23 for WORD $0x12345, both silently).
func TestDataEmission(t *testing.T) {
cases := []struct {
name string
mnem string
imm Imm
want string
}{
{"BYTE", "BYTE", 0x0f, "0f"},
{"BYTE negative", "BYTE", -1, "ff"},
{"BYTE truncated", "BYTE", 0x1ff, "ff"},
{"WORD", "WORD", 0x1234, "3412"},
{"WORD negative", "WORD", -1, "ffff"},
{"WORD truncated", "WORD", 0x12345, "4523"},
{"LONG", "LONG", 0x11223344, "44332211"},
{"LONG negative", "LONG", -1, "ffffffff"},
{"QUAD", "QUAD", 0x1122334455667788, "8877665544332211"},
{"QUAD negative", "QUAD", -2, "feffffffffffffff"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.imm)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// Exactly one immediate: the toolchain rejects BYTE $1, $2, $3, and a
// register or a missing operand is no immediate at all.
if _, err := Encode("BYTE"); err == nil {
t.Error("BYTE with no operand assembled, want an error")
}
if _, err := Encode("BYTE", Imm(1), Imm(2)); err == nil {
t.Error("BYTE $1, $2 assembled, want an error")
}
if _, err := Encode("WORD", AX); err == nil {
t.Error("WORD AX assembled, want an error")
}
}
// TestEndIgnored pins END: go tool asm drops the statement entirely, so it
// encodes to zero bytes and takes any operands without complaint (the
// toolchain accepts END $0 and END AX alike).
func TestEndIgnored(t *testing.T) {
for _, ops := range [][]Operand{nil, {Imm(0)}, {AX}} {
code, err := Encode("END", ops...)
if err != nil {
t.Errorf("END: %v", err)
continue
}
if len(code) != 0 {
t.Errorf("END = %x, want no bytes", code)
}
}
}
// TestAdjsp pins ADJSP: a positive immediate is SUBQ $imm, SP, a negative
// one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude picks; $0
// has no encoding (go tool asm refuses ADJSP $0 outright).
func TestAdjsp(t *testing.T) {
cases := []struct {
name string
imm Imm
want string
}{
{"imm8", 112, "4883ec70"},
{"imm8 negative", -112, "4883c470"},
{"imm32", 200, "4881ecc8000000"},
{"imm32 negative", -200, "4881c4c8000000"},
{"small", 8, "4883ec08"},
}
for _, c := range cases {
code, err := Encode("ADJSP", c.imm)
if err != nil {
t.Errorf("%s: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("ADJSP %d = %s, want %s", int64(c.imm), got, c.want)
}
}
if _, err := Encode("ADJSP", Imm(0)); err == nil {
t.Error("ADJSP $0 assembled, want an error")
}
if _, err := Encode("ADJSP"); err == nil {
t.Error("ADJSP with no operand assembled, want an error")
}
if _, err := Encode("ADJSP", AX); err == nil {
t.Error("ADJSP AX assembled, want an error")
}
}
+35 -18
View File
@@ -180,10 +180,19 @@ var evexTable = map[string]evexSpec{
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38, permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2B": {2, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38, population count (reg=dst, rm=src; W selects byte/word
// against dword/qword).
"VPOPCNTB": {2, 0x54, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPOPCNTD": {2, 0x55, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPOPCNTQ": {2, 0x55, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F.W1, the qword spelling of the packed OR (VPORQ has no VEX
// form in the Go assembler: it always encodes through EVEX).
"VPORQ": {1, 0xEB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -1435,18 +1444,20 @@ var evexKOperand = map[string]bool{
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
// prefix and W for the wider widths.
// direction, kk (k → k), kmem (k → mem), gprk (GPR/mem → k) and kgpr
// (k → GPR). Each direction group carries its own mandatory prefix and W:
// the k-destination/source forms share one pair, the GPR forms another.
type kmovSpec struct {
kk, kmem, gprk, kgpr byte
gprPP int
w int
kPP, kW int // prefix and VEX.W for the k forms
gprPP, gprW int // prefix and VEX.W for the GPR forms
}
var kmovTable = map[string]kmovSpec{
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0, 0, 0},
"KMOVB": {0x90, 0x91, 0x92, 0x93, 1, 0, 1, 0},
"KMOVD": {0x90, 0x91, 0x92, 0x93, 1, 1, 3, 0},
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 0, 1, 3, 1},
}
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
@@ -1460,14 +1471,14 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
dstReg, dstIsReg := dst.(Reg)
srcK := srcIsReg && srcReg.mask
dstK := dstIsReg && dstReg.mask
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
switch {
case srcK && dstK:
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
// k ← k: reg = dst, rm = src.
spec := vexSpec{mapSel: 1, opcode: ks.kk, w: ks.kW, pp: ks.kPP, opdigit: -1}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
case srcK && dstIsReg:
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
spec.pp = ks.gprPP
// GPR ← k: reg = dst, rm = src.
spec := vexSpec{mapSel: 1, opcode: ks.kgpr, w: ks.gprW, pp: ks.gprPP, opdigit: -1}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
@@ -1477,11 +1488,17 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
if _, ok := dst.(Mem); !ok {
return fmt.Errorf("%s: invalid destination operand", upper)
}
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
// mem ← k: reg = src, rm = dst.
spec := vexSpec{mapSel: 1, opcode: ks.kmem, w: ks.kW, pp: ks.kPP, opdigit: -1}
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
case dstK:
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
spec.pp = ks.gprPP
// k ← GPR: reg = dst, rm = src. A memory source shares the k ← k
// opcode and prefix group (the ykmovb layout the Go assembler uses).
opcode, w, pp := ks.gprk, ks.gprW, ks.gprPP
if memOperand(src) {
opcode, w, pp = ks.kk, ks.kW, ks.kPP
}
spec := vexSpec{mapSel: 1, opcode: opcode, w: w, pp: pp, opdigit: -1}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
}
return fmt.Errorf("%s requires a K register operand", upper)
+19
View File
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
// The qword OR spelling always encodes through EVEX.
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
// Byte permute and population count.
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
// Align (NDS + imm8).
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
@@ -52,6 +61,16 @@ func TestEvexGroundTruth(t *testing.T) {
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
// Moves, incl. disp8×N (64 for a 512-bit operand).
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
+706 -14
View File
@@ -14,30 +14,82 @@ var aluOp = map[string]struct {
}{
"ADD": {0x01, 0},
"OR": {0x09, 1},
"ADC": {0x11, 2},
"SBB": {0x19, 3},
"AND": {0x21, 4},
"SUB": {0x29, 5},
"XOR": {0x31, 6},
"CMP": {0x39, 7},
}
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
// mode); NEG/NOT use the 0xF6/0xF7 group.
// unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode.
// INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes
// in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4,
// DIV /6, IDIV /7; the accumulator is the implicit other operand).
var unaryOp = map[string]struct {
digit int
op byte
}{
"INC": {0, 0xFF},
"DEC": {1, 0xFF},
"NOT": {2, 0xF7},
"NEG": {3, 0xF7},
"INC": {0, 0xFF},
"DEC": {1, 0xFF},
"NOT": {2, 0xF7},
"NEG": {3, 0xF7},
"MUL": {4, 0xF7},
"DIV": {6, 0xF7},
"IDIV": {7, 0xF7},
}
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
// shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the
// 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4).
var shiftOp = map[string]int{
"SHL": 4,
"SAL": 4,
"SHR": 5,
"SAR": 7,
"ROL": 0,
"ROR": 1,
"RCL": 2,
"RCR": 3,
}
// bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form;
// the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's
// opcode row.
var bitTestOp = map[string]int{
"BT": 4,
"BTS": 5,
"BTR": 6,
"BTC": 7,
}
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
// fence names carry their opcode inside the 0F AE /digit group spelled out in
// full (E8/F0/F8), and PAUSE is F3 90.
//
// LOCK, REP and REPN are the prefix statements. go tool asm encodes each as
// a standalone one-byte instruction with a PC of its own (F0, F3 and F2
// respectively), not as a prefix field merged into the next instruction: the
// statement that follows is encoded unaware of it, and nothing validates
// that the pairing is a legal one (LOCK before NOP assembles without
// complaint, each byte pinned against the toolchain). Because the bytes
// land in the stream before the following statement anyway, a LOCKed
// CMPXCHGQ encodes identically to a prefixed form.
var noOperandTable = map[string][]byte{
"CPUID": {0x0F, 0xA2},
"RDTSC": {0x0F, 0x31},
"RDTSCP": {0x0F, 0x01, 0xF9},
"SYSCALL": {0x0F, 0x05},
"XGETBV": {0x0F, 0x01, 0xD0},
"CLD": {0xFC},
"STD": {0xFD},
"PAUSE": {0xF3, 0x90},
"LFENCE": {0x0F, 0xAE, 0xE8},
"MFENCE": {0x0F, 0xAE, 0xF0},
"SFENCE": {0x0F, 0xAE, 0xF8},
"UNDEF": {0x0F, 0x0B},
"LOCK": {0xF0},
"REP": {0xF3},
"REPN": {0xF2},
}
// --- MOV --------------------------------------------------------------------
@@ -309,6 +361,13 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if err != nil {
return err
}
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
// the destination is AL, the form the Go assembler prefers here.
if r, ok := dst.(Reg); ok && r.idx == 0 {
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
i.imm = immBytes
return e.emit(i)
}
i := newInstr(1, []byte{0x80})
if err := setRMDigit(i, digit, dst, 1); err != nil {
return err
@@ -451,13 +510,34 @@ func (e *enc) encodeUnary(op struct {
// --- SHL/SHR/SAR ------------------------------------------------------------
func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
// doubleShiftOp maps the two mnemonics whose three-operand form go tool asm
// accepts to the SHLD/SHRD opcode pair (imm8 form, CL form). SAR, SAL and
// the rotates have no such form: the oracle rejects SARQ/ROLQ with three
// operands, and so do we.
var doubleShiftOp = map[string][2]byte{
"SHL": {0xA4, 0xA5}, // SHLD
"SHR": {0xAC, 0xAD}, // SHRD
}
// isShiftCountCL reports whether a count operand is the CL register or its
// CX spelling: go tool asm accepts both (CX names the same low byte) and
// rejects ECX/RCX.
func isShiftCountCL(o Operand) bool {
reg, ok := o.(Reg)
return ok && reg.idx == 1 && (reg.size == 1 || reg.size == 2)
}
func (e *enc) encodeShift(base string, ops []Operand, size int) error {
digit := shiftOp[base]
if len(ops) == 3 {
return e.encodeDoubleShift(base, ops, size)
}
if len(ops) != 2 {
return fmt.Errorf("shift expects 2 operands, got %d", len(ops))
}
count, dst := ops[0], ops[1]
// Count is $1, %CL, or an imm8.
if reg, ok := count.(Reg); ok && reg.idx == 1 && reg.size <= 1 {
// Count is $1, CL (or its CX spelling), or an imm8.
if isShiftCountCL(count) {
// CL: 0xD2 (8-bit) / 0xD3.
op := byte(0xD3)
if size == 1 {
@@ -504,6 +584,44 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
return e.emit(i)
}
// encodeDoubleShift emits the three-operand SHL/SHR form, which the Go
// assembler spells as a shift but encodes as SHLD/SHRD (0F A4/A5, 0F AC/AD):
// the first operand is the count ($imm or CL), the second feeds the vacated
// bits (the reg field) and the third is the shifted value (the r/m field),
// matching go tool asm byte for byte. The W/L/Q widths exist; the oracle
// rejects the three-operand B form and every SAR/rotate one.
func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
opc, ok := doubleShiftOp[base]
if !ok || size == 1 {
return fmt.Errorf("%s: shift expects 2 operands, got %d", base, len(ops))
}
count, src, dst := ops[0], ops[1], ops[2]
srcReg, ok := src.(Reg)
if !ok {
return fmt.Errorf("%s: middle operand must be a register, like go tool asm", base)
}
i := newInstr(size, []byte{0x0F, opc[0]})
if isShiftCountCL(count) {
// CL (or CX) form: 0F A5/AD.
i.opcode[1] = opc[1]
} else {
imm, ok := count.(Imm)
if !ok {
return fmt.Errorf("shift count must be $1, CL or an immediate")
}
// The count is an unsigned imm8: the same range convention as the
// two-operand shift above.
if imm < 0 || imm > 255 {
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
}
i.imm = []byte{byte(imm)}
}
if err := setRMReg(i, srcReg.idx, srcReg.idx >= 8, false, dst, size); err != nil {
return err
}
return e.emit(i)
}
// --- IMUL -------------------------------------------------------------------
func (e *enc) encodeImul(ops []Operand, size int) error {
@@ -913,6 +1031,7 @@ type sseMove struct {
var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
"MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
@@ -938,12 +1057,12 @@ func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
op = m.load
reg, rm = dstReg, src
case srcVec:
if _, ok := dst.(Mem); !ok {
if !isX86Mem(dst) {
return fmt.Errorf("SSE move: invalid destination operand")
}
reg, rm = srcReg, dst
case dstVec:
if _, ok := src.(Mem); !ok {
if !isX86Mem(src) {
return fmt.Errorf("SSE move: invalid source operand")
}
op = m.load
@@ -1000,10 +1119,120 @@ var sseBinTable = map[string]sseBin{
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
"PCMPEQD": {0x66, 0x76, false},
"PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false},
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
"PCMPGTD": {0x66, 0x66, false},
"PSHUFB": {0x66, 0x00, true},
// Scalar compares and square root, packed adds/subtracts and the byte
// unpack, the spellings the Plan 9 table uses (COMISD orders the
// operands like every other two-operand form).
"ANDNPD": {0x66, 0x55, false},
"ANDNPS": {0x00, 0x55, false},
"COMISD": {0x66, 0x2F, false},
"SQRTSD": {0xF2, 0x51, false},
"PADDL": {0x66, 0xFE, false},
"PSUBL": {0x66, 0xFA, false},
"PUNPCKLBW": {0x66, 0x60, false},
// AES round functions (66 0F38) and the SHA message schedule helpers
// (no prefix, 0F38).
"AESENC": {0x66, 0xDC, true},
"AESENCLAST": {0x66, 0xDD, true},
"AESDEC": {0x66, 0xDE, true},
"AESDECLAST": {0x66, 0xDF, true},
"AESIMC": {0x66, 0xDB, true},
"SHA1MSG1": {0x00, 0xC9, true},
"SHA1MSG2": {0x00, 0xCA, true},
"SHA1NEXTE": {0x00, 0xC8, true},
"SHA256MSG1": {0x00, 0xCC, true},
"SHA256MSG2": {0x00, 0xCD, true},
}
// sseImm3 describes a legacy SSE instruction taking a leading imm8 and two
// further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and
// map3A select the opcode map the same way as sseBin's.
type sseImm3 struct {
prefix byte
op byte
map3A bool // opcode lives under 0F3A instead of 0F38
}
// sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3
// align/blend shuffles, the string compare, carry-less multiply and the AES
// key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings.
var sseImm3Table = map[string]sseImm3{
"PALIGNR": {0x66, 0x0F, true},
"PBLENDW": {0x66, 0x0E, true},
"PCMPESTRI": {0x66, 0x61, true},
"PCLMULQDQ": {0x66, 0x44, true},
"AESKEYGENASSIST": {0x66, 0xDF, true},
"SHA1RNDS4": {0x00, 0xCC, true},
}
// sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM
// source and rm = the destination (GPR or memory). PEXTRW's GPR destination
// uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one,
// so it carries both opcodes.
type sseExtract struct {
op []byte
opMem []byte // used when the destination is memory; nil shares op
rexW bool // PEXTRQ's REX.W
}
var sseExtractTable = map[string]sseExtract{
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
}
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
// destination and rm = the source (GPR or memory).
type sseInsert struct {
op []byte
rexW bool // PINSRQ's REX.W
}
var sseInsertTable = map[string]sseInsert{
"PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false},
"PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false},
"PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true},
"PINSRW": {[]byte{0x0F, 0xC4}, false},
}
// sseShiftImm maps the legacy packed integer shifts' immediate form:
// OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L
// (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ.
var sseShiftImm = map[string]sseShift{
"PSLLW": {0x71, 6},
"PSRLW": {0x71, 2},
"PSRAW": {0x71, 4},
"PSLLL": {0x72, 6},
"PSRLL": {0x72, 2},
"PSRAL": {0x72, 4},
"PSLLQ": {0x73, 6},
"PSRLQ": {0x73, 2},
"PSLLDQ": {0x73, 7},
"PSRLDQ": {0x73, 3},
}
// sseShiftVar maps the variable-count forms (the count comes from an XMM
// register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no
// variable form.
var sseShiftVar = map[string]byte{
"PSLLW": 0xF1,
"PSRLW": 0xD1,
"PSRAW": 0xE1,
"PSLLL": 0xF2,
"PSRLL": 0xD2,
"PSRAL": 0xE2,
"PSLLQ": 0xF3,
"PSRLQ": 0xD3,
}
// sseShift is one /digit selector in the 0F 71/72/73 immediate group.
type sseShift struct {
op byte
digit int
}
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
@@ -1016,6 +1245,7 @@ type sseShuf struct {
var sseShufTable = map[string]sseShuf{
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
"PSHUFL": {0x66, 0x70},
}
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
@@ -1090,3 +1320,465 @@ func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
}
return e.emit(i)
}
// --- carry, bit test, exchange and accumulate -------------------------------
// encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9
// order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with
// reg = index, rm = target); an immediate index uses 0F BA /digit with imm8.
func (e *enc) encodeBitTest(name string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
digit := bitTestOp[name]
index, target := ops[0], ops[1]
if reg, ok := index.(Reg); ok {
// Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC),
// the /digit base plus eight per step.
i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3})
if err := setRM(i, reg, target, size); err != nil {
return err
}
return e.emit(i)
}
imm, ok := index.(Imm)
if !ok {
return fmt.Errorf("%s index must be a register or an immediate", name)
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
i := newInstr(size, []byte{0x0F, 0xBA})
if err := setRMDigit(i, digit, target, size); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeExchange encodes XCHG. A register-to-register exchange where either
// operand is AX uses the 0x90+r accumulator form (with REX.W for the quad
// form, as the Go assembler emits it); everything else uses 0x86/0x87 with
// the register operand in ModRM.reg, the memory (or second register) in r/m.
func (e *enc) encodeExchange(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsReg := src.(Reg)
dstReg, dstIsReg := dst.(Reg)
if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) {
// 0x90+r: r is the non-AX register, whichever side it sits on.
r := dstReg
if srcReg.idx == 0 {
r = dstReg
} else {
r = srcReg
}
i := newInstr(size, []byte{0x90 + byte(r.idx&7)})
i.rexB = r.idx >= 8
return e.emit(i)
}
op := byte(0x87)
if size == 1 {
op = 0x86
}
switch {
case srcIsReg:
i := newInstr(size, []byte{op})
if err := setRM(i, srcReg, dst, size); err != nil {
return err
}
return e.emit(i)
case dstIsReg:
i := newInstr(size, []byte{op})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("XCHG: at least one operand must be a register")
}
// encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG
// (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the
// destination writable (register or memory).
func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok {
return fmt.Errorf("%s source must be a register", name)
}
opc := op
if size == 1 {
opc = op8
}
i := newInstr(size, []byte{0x0F, opc})
if err := setRM(i, srcReg, ops[1], size); err != nil {
return err
}
return e.emit(i)
}
// encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for
// the rest; the word form carries a 0x66 operand-size prefix (66 F2, the
// prefix order the Go assembler emits) and the quad form REX.W. reg = GPR
// accumulator, rm = the data source.
func (e *enc) encodeCrc32(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("CRC32 destination must be a general register")
}
i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1}
if size > 1 {
i.opcode[2] = 0xF1
}
i.rexW = size == 8
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg =
// destination, rm = source, the carry/overflow flag as the carry-in.
func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("ADCX/ADOX destination must be a general register")
}
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8}
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// --- string primitives, flags and INT ----------------------------------------
// encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and
// STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or
// REX.W prefix.
func (e *enc) encodeStringOp(base string, ops []Operand, size int) error {
if len(ops) != 0 {
return fmt.Errorf("%s takes no operands, got %d", base, len(ops))
}
var op byte
switch base {
case "MOVS":
op = 0xA5
if size == 1 {
op = 0xA4
}
case "STOS":
op = 0xAB
if size == 1 {
op = 0xAA
}
default:
return fmt.Errorf("unsupported string instruction %q", base)
}
return e.emit(newInstr(size, []byte{op}))
}
// encodeInt encodes INT with its single imm8 operand. The field takes the
// low byte silently inside the 32-bit span, matching the scalar convention
// (go tool asm encodes INT $256 as CD 00).
func (e *enc) encodeInt(ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("INT expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("INT operand must be an immediate")
}
if imm < -(1<<31) || imm > (1<<32)-1 {
return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm))
}
return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}})
}
// encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a
// single 32-bit memory operand.
func (e *enc) encodeMxcsr(digit int, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops))
}
m, ok := ops[0].(Mem)
if !ok {
return fmt.Errorf("MXCSR instruction requires a memory operand")
}
i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1}
if err := setMem(i, digit, m); err != nil {
return err
}
return e.emit(i)
}
// cvtIntOp maps the scalar float-to-integer conversions to their mandatory
// prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating
// CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width.
var cvtIntOp = map[string]struct {
prefix byte
op byte
}{
"CVTSD2S": {0xF2, 0x2D},
"CVTTSD2S": {0xF2, 0x2C},
"CVTSS2S": {0xF3, 0x2D},
"CVTTSS2S": {0xF3, 0x2C},
}
// encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C
// with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the
// quad spellings.
func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
}
spec := cvtIntOp[base]
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("%s destination must be a general register", base)
}
i := newInstr(size, []byte{0x0F, spec.op})
i.prefix = spec.prefix
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
// encodeFmov encodes the x87 double move. The memory forms are DD /0
// (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register
// move is DD C0+dst (FLD st(dst)), the form the Go assembler emits.
func (e *enc) encodeFmov(ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsF := src.(Reg)
dstReg, dstIsF := dst.(Reg)
srcF := srcIsF && srcReg.fp
dstF := dstIsF && dstReg.fp
switch {
case srcF && dstF:
// The register form is DD /2 with rm = the destination (FST st(dst)).
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setRMDigit(i, 2, dstReg, 8); err != nil {
return err
}
return e.emit(i)
case dstF:
m, ok := src.(Mem)
if !ok {
return fmt.Errorf("FMOVD: invalid source operand")
}
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setMem(i, 0, m); err != nil {
return err
}
return e.emit(i)
case srcF:
m, ok := dst.(Mem)
if !ok {
return fmt.Errorf("FMOVD: invalid destination operand")
}
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setMem(i, 2, m); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("FMOVD needs an x87 register operand")
}
// --- legacy SSE imm8, extract, insert and packed shift families --------------
// encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src,
// dst with reg = dst, rm = src and the immediate appended last (PALIGNR,
// PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4).
func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("SSE imm8 instruction needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
src, dst := ops[1], ops[2]
dstReg, ok2 := dst.(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("SSE imm8 instruction destination must be a vector register")
}
opcode := []byte{0x0F, 0x38, m.op}
if m.map3A {
opcode = []byte{0x0F, 0x3A, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the
// XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and
// PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the
// SSE4.1 0F3A 15 one).
func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("extract needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
srcReg, srcVec := vecReg(ops[1])
if !srcVec {
return fmt.Errorf("extract source must be an XMM register")
}
opcode := m.op
if m.opMem != nil && memOperand(ops[2]) {
opcode = m.opMem
}
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
if err := setRM(i, srcReg, ops[2], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the
// XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and
// PINSRW).
func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("insert needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
dstReg, dstVec := vecReg(ops[2])
if !dstVec {
return fmt.Errorf("insert destination must be an XMM register")
}
i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW}
if err := setRM(i, dstReg, ops[1], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEShift encodes the legacy packed integer shifts. The immediate
// form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form
// OP count, dst carries the count in an XMM register (or memory) on the
// 66 0F D1-F3 opcodes. The destination is always the register written.
func (e *enc) encodeSSEShift(name string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s destination must be the second, vector operand", name)
}
if imm, isImm := ops[0].(Imm); isImm {
spec := sseShiftImm[name]
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
if !vecOrMem(ops[0]) {
return fmt.Errorf("%s count must be an immediate, a vector register or memory", name)
}
op, ok := sseShiftVar[name]
if !ok {
return fmt.Errorf("%s has no variable-count form", name)
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
return e.emit(i)
}
// encodeCmpsd encodes CMPSD, the scalar double compare with its predicate
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
// F2 0F C2 with reg = dst, rm = src.
func (e *enc) encodeCmpsd(ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
}
imm, ok := ops[2].(Imm)
if !ok {
return fmt.Errorf("CMPSD predicate must be an immediate")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("CMPSD destination must be a vector register")
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the
// literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no
// prefix, reg = dst, rm = src; X0 is implicit on the wire).
func (e *enc) encodeSha256rnds2(ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops))
}
x0, ok := ops[0].(Reg)
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
return fmt.Errorf("SHA256RNDS2 first operand must be X0")
}
dstReg, ok2 := ops[2].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("SHA256RNDS2 destination must be a vector register")
}
i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[1], 8); err != nil {
return err
}
return e.emit(i)
}
+814 -45
View File
@@ -46,30 +46,70 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
}
// Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{}
pos := guardLen + len(prologue)
// The toolchain's parser counts N(PC) displacements over the source
// instructions at a uniform 4 bytes each, so a PC-relative branch
// resolves to the instruction N slots away in body order; the resolved
// target then participates in layout and loop-head padding like any
// branch target.
instrs := make([]*ast.Instr, 0, len(t.Body))
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
pos += loong64InstrSize(s, fi)
if in, ok := stmt.(*ast.Instr); ok && strings.ToUpper(in.Mnemonic.Text) != "PCALIGN" {
instrs = append(instrs, in)
}
}
// Pass 2: encode. The guard prefix precedes the prologue; its branches
// target the morestack block at the end of the function, which the first
// pass has sized.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += loong64InstrSize(in, fi)
}
parseIndex := make(map[*ast.Instr]int, len(instrs))
for i, in := range instrs {
parseIndex[in] = i
}
pcRelTarget := make(map[*ast.Instr]*ast.Instr)
for _, in := range instrs {
off, ok := loong64PCRelOffset(in)
if !ok {
continue
}
bodyLen = p - (guardLen + len(prologue))
tgt := parseIndex[in] + off
if tgt < 0 || tgt >= len(instrs) {
continue
}
pcRelTarget[in] = instrs[tgt]
}
// Pass 1: label offsets from the instruction sizes. PCALIGN contributes
// only its padding. On top of the explicit PCALIGNs, the toolchain pads
// every backward-branch target (loop head) to a 16-byte boundary, so the
// layout runs to a fixpoint over the alignment set.
loopAligns := map[string]bool{}
alignInstrs := map[*ast.Instr]bool{}
for {
offsets, _, pcs, _ := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
changed := false
for _, in := range instrs {
// A backward PC-relative target is the resolved instruction.
if tgt, ok := pcRelTarget[in]; ok && pcs[tgt] < pcs[in] && !alignInstrs[tgt] {
alignInstrs[tgt] = true
changed = true
}
target, ok := loong64BranchTarget(in)
if !ok {
continue
}
tOff, ok := offsets[target]
if !ok || tOff >= pcs[in] || loopAligns[target] {
continue
}
loopAligns[target] = true
changed = true
}
if !changed {
break
}
}
// Final layout with the complete alignment set.
offsets, alignPad, pcs, bodyEnd := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
bodyLen := bodyEnd - (guardLen + len(prologue))
pcRelPcs := make(map[*ast.Instr]int, len(pcRelTarget))
for in, tgt := range pcRelTarget {
pcRelPcs[in] = pcs[tgt]
}
var out []byte
if fi.needSplit {
@@ -84,7 +124,20 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
if !ok {
continue
}
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve)
// PCALIGN pads to the requested boundary with andi $0, $0, 0, the
// architecture's NOP, and encodes to nothing itself.
if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" {
pad := loong64PCAlignPad(pc, in)
out = append(out, loong64PadBytes(pad)...)
pc += pad
continue
}
// Loop-head alignment padding precedes the instruction.
if pad := alignPad[in]; pad > 0 {
out = append(out, loong64PadBytes(pad)...)
pc += pad
}
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve, pcRelPcs)
if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
@@ -115,6 +168,115 @@ func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry,
return out, offsets, relocs, lines, spadj, nil
}
// loong64PCRelOffset reports the N of a branch operand spelled N(PC): the
// displacement counted in source instructions from the branch itself.
func loong64PCRelOffset(instr *ast.Instr) (int, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
branch := false
switch mnem {
case "JMP":
branch = len(instr.Operands) == 1
case "JAL", "CALL", "BL":
branch = len(instr.Operands) == 1 || len(instr.Operands) == 2
case "BFPT", "BFPF":
branch = len(instr.Operands) == 1
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
branch = len(instr.Operands) >= 2
}
if !branch {
return 0, false
}
op := instr.Operands[len(instr.Operands)-1]
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
return int(op.Addr.Offset), true
}
return 0, false
}
// loong64Layout walks the function body once and returns the label offsets,
// the loop-alignment padding due before each instruction (a pad of 0 needs
// nothing), the pc each instruction starts at (its padding included) and the
// first pc past the body. Explicit PCALIGN pads, the alignment pads for the
// labels in aligns and those for the instructions in alignInstrs (backward
// PC-relative targets) all contribute, mirroring the toolchain's layout
// pass.
func loong64Layout(t *ast.Text, start int, fi loong64FrameInfo, aligns map[string]bool, alignInstrs map[*ast.Instr]bool) (map[string]int, map[*ast.Instr]int, map[*ast.Instr]int, int) {
offsets := map[string]int{}
alignPad := map[*ast.Instr]int{}
pcs := map[*ast.Instr]int{}
pos := start
pendingAlign := false
var pendingNames []string
explicit := false
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
if aligns[s.Name.Text] {
pendingAlign = true
}
pendingNames = append(pendingNames, s.Name.Text)
// Provisional: a branch to the label lands here unless a loop
// alignment pad follows, in which case the label resolves to the
// padded instruction (the toolchain's labels bind to the branch
// target instruction, which the padding pass precedes).
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pos += loong64PCAlignPad(pos, s)
explicit = true
continue
}
if pendingAlign {
pendingAlign = false
if pos&15 != 0 {
alignPad[s] = 16 - pos&15
}
}
if alignInstrs[s] && pos&15 != 0 {
alignPad[s] = 16 - pos&15
}
if !explicit {
for _, n := range pendingNames {
offsets[n] = pos + alignPad[s]
}
}
pendingNames = nil
explicit = false
pcs[s] = pos + alignPad[s]
pos += alignPad[s] + loong64InstrSize(s, fi)
}
}
return offsets, alignPad, pcs, pos
}
// loong64BranchTarget reports the local label a branch-like instruction
// transfers to, the loop-head signal the toolchain derives from backward
// branch targets.
func loong64BranchTarget(instr *ast.Instr) (string, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
var op *ast.Operand
switch {
case mnem == "JMP" || mnem == "JAL" || mnem == "BFPT" || mnem == "BFPF":
if len(ops) != 1 {
return "", false
}
op = ops[0]
case mnem == "TEQ" || mnem == "TNE":
return "", false
case len(ops) >= 2:
op = ops[len(ops)-1]
default:
return "", false
}
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's
// branch-chasing pass: a label whose first instruction is an unconditional
// local jump redirects its own jumpers to the ultimate target. The Go
@@ -177,21 +339,51 @@ func l64LabelOK(op *ast.Operand) (string, bool) {
return "", false
}
// l64SubToAdd rewrites the SUB family with an immediate first operand onto
// its ADD counterpart with the negated immediate: LoongArch has no
// subtract-immediate instructions, and the toolchain folds SUB $v into the
// ADD immediate form through the same optab matching (the $0 fold into 3R
// and the large-constant materialisations included). The negation is the
// second result; the operand is left untouched because the size pass
// normalises the same instruction.
func l64SubToAdd(mnem string, ops []*ast.Operand) (string, bool) {
if len(ops) >= 2 && isImmOperand(ops[0]) {
switch mnem {
case "SUB":
return "ADD", true
case "SUBW":
return "ADDW", true
case "SUBV", "SUBVU":
return "ADDV", true
}
}
return mnem, false
}
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for
// most, more for the multi-instruction expansions.
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
var neg bool
mnem, neg = l64SubToAdd(mnem, ops)
if mnem == "RET" {
return len(loong64Return(fi))
}
switch mnem {
case "TEQ", "TNE":
return 8 // bne/beq over the BREAK, then BREAK
case "PRELDX":
return 20 // the four-instruction constant materialisation + preldx
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return loong64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := l64Imm64(ops[0])
if neg {
v = -v
}
if v == 0 {
return 4 // folds into the 3R form (rk = R0)
}
@@ -229,11 +421,50 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
return 4
}
// loong64PCAlignPad returns the padding PCALIGN inserts before the next
// instruction so that it starts at the requested boundary relative to the
// function start. The boundary must be a power of two between 8 and 2048, as
// the toolchain requires; anything else pads nothing.
func loong64PCAlignPad(pos int, instr *ast.Instr) int {
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
return 0
}
align := int(immFromOperand(instr.Operands[0]))
if align < 8 || align > 2048 || align&(align-1) != 0 {
return 0
}
return (align - pos%align) % align
}
// loong64PadBytes renders PCALIGN padding: the toolchain emits andi $0, $0, 0
// (the architecture's NOP) for every full 4 bytes of pad.
func loong64PadBytes(pad int) []byte {
nop := l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0))
out := make([]byte, 0, pad/4*len(nop))
for i := 0; i < pad/4; i++ {
out = append(out, nop...)
}
return out
}
// encodeLOONG64Instr encodes a single LoongArch instruction.
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
// The SUB family with an immediate first operand folds onto the ADD
// immediate form with the negated immediate; the negation happens on a
// copy of the operand, never on the shared syntax tree.
mnem, neg := l64SubToAdd(mnem, ops)
if neg {
c := *ops[0]
c.Imm.Val = -c.Imm.Val
ops2 := make([]*ast.Operand, len(ops))
ops2[0] = &c
copy(ops2[1:], ops[1:])
ops = ops2
}
// Pseudo-instructions and the branches first.
switch mnem {
case "RET":
@@ -249,10 +480,80 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "NEGW", "NEGV":
// The integer negation pseudo is a subtract from zero:
// NEGW src, dst → sub.w r0, src, dst.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
src, dst := l64Reg(ops[0]), l64Reg(ops[1])
if src < 0 || dst < 0 {
return nil, fmt.Errorf("invalid register operand")
}
sub := l64InstrTable["SUBW"].op
if mnem == "NEGV" {
sub = l64InstrTable["SUBV"].op
}
return l64wordLE(l64rrr(sub, src, 0, dst)), nil
case "TEQ", "TNE":
// The trap pseudo expands to two instructions: bne/beq rj, rd over
// the BREAK (offset 2 instruction units), then BREAK $code.
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
code := int(immFromOperand(ops[0]))
rj, rd := 0, l64Reg(ops[len(ops)-1])
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
bop := l64branchTable["BNE"]
if mnem == "TNE" {
bop = l64branchTable["BEQ"]
}
return l64WordsLE(
l64irr16(bop, 2, rj, rd),
l64i15(l64InstrTable["BREAK"].op, code),
), nil
case "PRELDX":
// preldx offset(Rbase), $n, $hint: the 64-bit descriptor n packs
// (addrSeq, blockSize, blockNums, stride); the constant v built from
// it materialises in R30 across four instructions, then the preldx.
if len(ops) != 3 || !isMemOperand(ops[0]) || !isImmOperand(ops[1]) || !isImmOperand(ops[2]) {
return nil, fmt.Errorf("PRELDX expects offset(reg), $n, $hint")
}
rj := loong64RegNum(ops[0].Addr.Base)
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
n := uint64(l64Imm64(ops[1]))
hint := int(l64Imm64(ops[2]))
addrSeq := (n >> 0) & 0x1
blkSize := (n >> 1) & 0x7ff
blkNums := (n >> 12) & 0x1ff
stride := (n >> 21) & 0xffff
v := uint64(ops[0].Addr.Offset)&0xffff + addrSeq<<16 +
((blkSize/16)-1)<<20 + (blkNums-1)<<32 + stride<<44
const (
lu12iw = 0x0a << 25
lu32id = 0x0b << 25
lu52id = 0x00c << 22
ori = 0x00e << 22
preldx = 0x7058 << 15
)
return l64WordsLE(
l64ir(lu12iw, int(uint32(v>>12)), 30),
l64irr(ori, int(uint32(v)), 30, 30),
l64ir(lu32id, int(uint32(v>>32)), 30),
l64irr(lu52id, int(uint32(v>>52)), 30, 30),
l64rrr(preldx, 30, rj, hint),
), nil
case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs)
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs, pcRelPcs)
case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs)
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs, pcRelPcs)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs)
}
@@ -262,12 +563,12 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
if mnem == "JIRL" {
return encodeLOONG64Jirl(op, ops)
}
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve)
return encodeLOONG64Branch16(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
}
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
// BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0).
if op, ok := l64branch21Table[mnem]; ok {
return encodeLOONG64Branch21(mnem, op, ops, pc, offsets, resolve)
return encodeLOONG64Branch21(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
}
// B/BL aliases reached only via JMP/JAL above.
@@ -321,6 +622,16 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before
// the integer/FP table (their mnemonics overlap the table's 2R format
// but resolve vector-bank registers).
if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled {
if err != nil {
return nil, err
}
return code, nil
}
enc, ok := l64InstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
@@ -523,11 +834,23 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
//
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc) ([]byte, error) {
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
}
op := instr.Operands[0]
// PC-relative displacement: N(PC) resolves to the instruction N slots
// away in source order (the toolchain's parse-time count), and the field
// carries the final pc distance in instruction units.
if op.Addr.Sym == nil && op.Addr.Base == "PC" {
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, op.Addr.Offset)
}
v := (targetPc - pc) >> 2
opc := l64jumpTable[mnem]
return l64wordLE(l64bbl(opc, v)), nil
}
if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil {
// Indirect: (rj) → jirl.
rj := loong64RegNum(op.Addr.Base)
@@ -610,16 +933,28 @@ func l64offsetOperand(op *ast.Operand) (int32, bool) {
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
func encodeLOONG64Branch16(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
target := resolve(l64Label(ops[len(ops)-1]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
var target string
var v int
lastOp := ops[len(ops)-1]
if lastOp.Kind == ast.OpAddr && lastOp.Addr.Sym == nil && lastOp.Addr.Base == "PC" {
// N(PC) resolves to the instruction N slots away in source order.
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, lastOp.Addr.Offset)
}
v = (targetPc - pc) >> 2
} else {
target = resolve(l64Label(lastOp))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v = (targetOff - pc) >> 2
}
v := (targetOff - pc) >> 2
if len(ops) == 2 {
// Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/
// BGEZ-family aliases encoded with rj in the rj field.
@@ -680,33 +1015,55 @@ func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, o
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
// and a 16-bit offset, are handled separately.
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 {
func encodeLOONG64Branch21(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
isBF := mnem == "BFPT" || mnem == "BFPF"
if len(ops) != 2 && !(isBF && (len(ops) == 1 || len(ops) == 2)) {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
target := resolve(l64Label(ops[1]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
rj := 0 // BFPT/BFPF default to FCC0
if mnem != "BFPT" && mnem != "BFPF" {
var rj int
tgtOp := ops[len(ops)-1]
if isBF {
// BFPT/BFPF test an FCC condition register, defaulting to FCC0 when
// spelled without one.
rj = 0
if len(ops) == 2 {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
} else {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
var v int
if tgtOp.Kind == ast.OpAddr && tgtOp.Addr.Sym == nil && tgtOp.Addr.Base == "PC" {
// N(PC) resolves to the instruction N slots away in source order.
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, tgtOp.Addr.Offset)
}
v = (targetPc - pc) >> 2
} else {
target := resolve(l64Label(tgtOp))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v = (targetOff - pc) >> 2
}
if mnem == "BGTZ" || mnem == "BLEZ" {
// The toolchain swaps the register into the rd field and keeps the
// 16-bit offset form.
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
return nil, fmt.Errorf("branch %d too far (16-bit range)", v)
}
return l64wordLE(l64irr16(op, v, 0, rj)), nil
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
return nil, fmt.Errorf("branch %d too far (21-bit range)", v)
}
return l64wordLE(l64ir21(op, v, rj)), nil
}
@@ -1490,3 +1847,415 @@ func l64Label(op *ast.Operand) string {
}
return op.Raw
}
// ---- LSX/LASX (V*/XV*) vector dispatch ----
// l64VecOperand describes a vector register operand: the 5-bit register
// number, its bank and an optional width or element suffix (V0.B16,
// V1.V[0], X3.WU[2]). The parser hands suffixed operands over verbatim
// (the element index survives only in the raw text), so the suffix is
// scanned from op.Raw.
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
unsig bool // the suffix carries the U marker (.BU[0])
hasSuf bool // any suffix present
}
// l64ParseVecOperand parses a vector register operand with an optional
// width or element suffix. ok reports whether the operand names a vector
// register at all (V or X bank, with or without a suffix).
func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
if op.Kind == ast.OpImmediate {
return v, false
}
name := strings.ReplaceAll(op.Raw, " ", "")
if name == "" || (name[0] != 'V' && name[0] != 'X') {
return v, false
}
i := 1
num := 0
for i < len(name) && name[i] >= '0' && name[i] <= '9' {
num = num*10 + int(name[i]-'0')
if num > 31 {
return v, false
}
i++
}
if i == 1 {
return v, false // no register digits
}
v.num, v.lasx = num, name[0] == 'X'
if i == len(name) {
return v, true
}
if name[i] != '.' || i+2 > len(name) {
return v, false
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
return v, false
}
v.width, v.hasSuf = w, true
i++
if i < len(name) && name[i] == 'U' {
v.unsig = true
i++
}
if i < len(name) && name[i] == '[' {
// Element form .T[i]: the closing bracket ends the operand.
if name[len(name)-1] != ']' || i+2 > len(name)-1 {
return v, false
}
idx := 0
for _, c := range name[i+1 : len(name)-1] {
if c < '0' || c > '9' {
return v, false
}
idx = idx*10 + int(c-'0')
if idx > 31 {
return v, false
}
}
v.elem, v.hasEl = idx, true
return v, true
}
// Width form .T<lanes>: the trailing digits give the lane count.
lanes := 0
if i >= len(name) {
return v, false
}
for ; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return v, false
}
lanes = lanes*10 + int(name[i]-'0')
if lanes > 64 {
return v, false
}
}
v.lanes = lanes
return v, true
}
// l64VecSuffixWidth validates a width suffix against the bank (LSX:
// B16/H8/W4/V2, LASX: B32/H16/W8/V4) and returns the encoded 2-bit width
// selector of vreplgr2vr and vldrepl.
func l64VecSuffixWidth(lasx bool, v l64VecOperand) (int, bool) {
want := map[byte]int{'B': 16, 'H': 8, 'W': 4, 'V': 2}
if lasx {
want = map[byte]int{'B': 32, 'H': 16, 'W': 8, 'V': 4}
}
lanes, ok := want[v.width]
if !ok || lanes != v.lanes {
return 0, false
}
switch v.width {
case 'B':
return 0, true
case 'H':
return 1, true
case 'W':
return 2, true
default:
return 3, true
}
}
// l64VecElementBase validates an element suffix against the bank and
// returns the encoded index field: the index rides in the rk field above a
// per-width base (vpickve2gr/vinsgr2vr give ui4 to .b, ui3 to .h, ui2 to .w
// and ui1 to .d). The LASX bank has no .b/.h element forms: the toolchain
// rejects `XVMOVQ R4, X2.B[0]` and `XVMOVQ X3.B[31], R5`.
func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) {
limit, base := 0, 0
switch v.width {
case 'B':
if lasx {
return 0, false
}
limit, base = 15, 0
case 'H':
if lasx {
return 0, false
}
limit, base = 7, 16
case 'W':
limit, base = 3, 24
if lasx {
limit, base = 7, 16
}
case 'V':
limit, base = 1, 28
if lasx {
limit, base = 3, 24
}
default:
return 0, false
}
if v.elem > limit {
return 0, false
}
return base + v.elem, true
}
// encodeLOONG64Vector encodes the LSX/LASX mnemonics the table marks as
// vector plus the VMOVQ/XVMOVQ move family. handled reports whether the
// mnemonic belongs to the vector slice; the operand shapes and opcode
// constants reproduce GOARCH=loong64 `go tool asm` exactly.
func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) {
if mnem == "VMOVQ" || mnem == "XVMOVQ" {
code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi)
return code, true, err
}
lasx, ok := l64VecBank[mnem]
if !ok {
return nil, false, nil
}
ops := instr.Operands
bank := "V"
if lasx {
bank = "X"
}
vec := func(op *ast.Operand) (int, error) {
v, isVec := l64ParseVecOperand(op)
if !isVec || v.lasx != lasx || v.hasSuf {
return -1, fmt.Errorf("%s: expected a bare %s0-%s31 vector register, got %q", mnem, bank, bank, op.Raw)
}
return v.num, nil
}
// Two-operand forms (vpcnt.v): INSTR vj, vd.
if l64Vec2R[mnem] {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[1])
if err != nil {
return nil, true, err
}
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, vd)), true, nil
}
// Immediate forms: INSTR $imm, vd or INSTR $imm, vj, vd.
if e, imm := l64VecImmInfo[mnem]; imm && len(ops) >= 2 && isImmOperand(ops[0]) {
if len(ops) > 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
imm := int(immFromOperand(ops[0]))
if imm < e.min || imm > e.max {
return nil, true, fmt.Errorf("%s: immediate out of range [%d, %d]", mnem, e.min, e.max)
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64irr(e.op, (imm+e.bias)&e.mask, vj, vd)), true, nil
}
// Vector-to-condition forms: INSTR vj, FCCn.
if l64InstrTable[mnem].format == l64Fvcf {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
if loong64RegClass(operandRegName(ops[1])) != l64ClsFCC {
return nil, true, fmt.Errorf("%s: expected an FCC condition flag, got %q", mnem, ops[1].Raw)
}
fcc := loong64RegNum(operandRegName(ops[1]))
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, fcc)), true, nil
}
// Three-register forms: INSTR vk, vj, vd or INSTR vk, vd (vj = vd).
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
vk, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64rrr(l64InstrTable[mnem].op, vk, vj, vd)), true, nil
}
// encodeLOONG64Vmovq encodes the VMOVQ/XVMOVQ move family. One mnemonic
// covers the whole LSX/LASX transfer surface, dispatched by operand shape
// exactly as the toolchain's table does:
//
// VMOVQ vd, off(rj) vst VMOVQ off(rj), vd vld
// VMOVQ vd, (rj)(rk) vstx VMOVQ (rj)(rk), vd vldx
// VMOVQ off(rj), vd.T vldrepl (load and replicate one element)
// VMOVQ vj, vd vori.b $0 (a register move)
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
if lasx {
bank = "X"
}
if len(ops) != 2 {
return nil, fmt.Errorf("VMOVQ expects 2 operands, got %d", len(ops))
}
src, srcVec := l64ParseVecOperand(ops[0])
dst, dstVec := l64ParseVecOperand(ops[1])
srcMem := isMemOperand(ops[0])
dstMem := isMemOperand(ops[1])
srcIdx := srcMem && ops[0].Addr.Index != ""
dstIdx := dstMem && ops[1].Addr.Index != ""
intReg := func(op *ast.Operand) (int, error) {
if isMemOperand(op) {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
name := operandRegName(op)
if loong64RegClass(name) != l64ClsGR {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
return loong64RegNum(name), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
if src.hasSuf || dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: a register move takes bare %s registers", bank)
}
if src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
return l64wordLE(l64rr(enc.move, src.num, dst.num)), nil
}
// Store: VMOVQ vd, off(rj) or VMOVQ vd, (rj)(rk).
if srcVec && dstMem {
if src.hasSuf || src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected a bare %s0-%s31 register as the stored value", bank, bank)
}
if dstIdx {
rj, rk := loong64RegNum(ops[1].Addr.Base), loong64RegNum(ops[1].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(enc.st, int(off), rj, src.num)), nil
}
// Load: VMOVQ off(rj), vd, the indexed VMOVQ (rj)(rk), vd, and the
// load-and-replicate form VMOVQ off(rj), vd.T.
if srcMem && dstVec {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
if srcIdx {
if dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: an indexed load takes a bare %s register", bank)
}
rj, rk := loong64RegNum(ops[0].Addr.Base), loong64RegNum(ops[0].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
op := enc.ld
if dst.hasSuf {
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
case 2:
op = enc.replW
default:
op = enc.replD
}
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}
// Element extract: VMOVQ vj.T[i], rd (vpickve2gr, signed or unsigned).
if srcVec && src.hasEl && !dstVec && !dstMem {
if src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
idx, ok := l64VecElementBase(lasx, src)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[0].Raw)
}
rd, err := intReg(ops[1])
if err != nil {
return nil, err
}
op := enc.pickS
if src.unsig {
op = enc.pickU
}
return l64wordLE(l64irr(op, idx, src.num, rd)), nil
}
// Insert and duplicate: VMOVQ rj, vd.T[i] (vinsgr2vr) and
// VMOVQ rj, vd.T (vreplgr2vr).
if !srcVec && !srcMem && dstVec && dst.hasSuf {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
rs, err := intReg(ops[0])
if err != nil {
return nil, err
}
if dst.hasEl {
idx, ok := l64VecElementBase(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.ins, idx, rs, dst.num)), nil
}
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid width suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.dup, w, rs, dst.num)), nil
}
return nil, fmt.Errorf("VMOVQ: unsupported operand combination %q, %q", ops[0].Raw, ops[1].Raw)
}
+171 -6
View File
@@ -30,7 +30,10 @@ package asm
// of the immediate and register fields), mirroring the toolchain's OP_*
// helpers, so each l64* function only ORs its fields in.
import "maps"
import (
"maps"
"strings"
)
// loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
@@ -103,7 +106,12 @@ func loong64RegNum(name string) int {
case "R31", "S8":
return 31
}
// F0-F31, FCC0-FCC7, FCSR0-FCSR31.
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
// X0-X31) are deliberately NOT accepted here: they are a separate
// register class, and the toolchain rejects V/X names wherever an
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
// resolved only through loong64VecRegNum.
if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], 31)
}
@@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int {
return -1
}
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
// of their own: the toolchain accepts them only in the vector operands of the
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
// spellings never reach the integer/FP resolver.
func loong64VecRegNum(name string) int {
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
return -1
}
return loong64RegSpecial(name[1:], 31)
}
// ---- format helpers ----
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
@@ -247,7 +268,7 @@ const (
l64Firr14 // 2RI14 (ldptr/stptr)
l64Firr16 // 2RI16 (addu16i.d)
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
l64Firir // bstrins/bstrpick
l64Firrr // alsl
l64Fi15 // syscall/break/dbar
@@ -255,6 +276,8 @@ const (
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
l64Fshift // 2RI12 with a 5/6-bit shift immediate
l64Fpreld // preld (2RI12 + 5-bit hint)
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -277,10 +300,68 @@ type l64DualEnc struct {
var l64DualTable = map[string]l64DualEnc{}
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
// yet; the base integer, memory and floating-point ISA is complete.
// to their encoding.
var l64InstrTable = map[string]l64Enc{}
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
type l64Vec3Enc struct {
op uint32
lasx bool
}
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
// the opcode, the bank, the accepted immediate range, the bias the toolchain
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
type l64VecImmEnc struct {
op uint32
lasx bool
min, max int
bias int
mask int
}
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
// each accepts; presence in the map routes the mnemonic through the vector
// dispatcher rather than the integer/FP formats.
var l64VecBank = map[string]bool{}
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
var l64VecImmInfo = map[string]l64VecImmEnc{}
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
// vpcnt.v).
var l64Vec2R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
pickS, pickU uint32 // vpickve2gr.{,u} element extract
ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move
}
var l64VmovqTable = map[bool]l64VmovqEnc{
false: { // VMOVQ, the LSX (V) bank
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
},
true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
},
}
func init() {
// 3R, integer.
rrr := map[string]uint32{
@@ -360,6 +441,10 @@ func init() {
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
// LSX: convert a 64-bit integer lane to a double float. The operand
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
// so the entry stays on the 2R integer/FP format.
"FFINTDV": 0x474a << 10,
}
for m, op := range rr {
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
@@ -416,12 +501,14 @@ func init() {
// LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R, fused multiply-add.
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
// condition flag, the layout matches the 4R shape).
rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
"FSEL": 0x340 << 18,
}
for m, op := range rrrr {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
@@ -455,6 +542,10 @@ func init() {
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
// is rejected by the oracle.
am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
@@ -472,10 +563,84 @@ func init() {
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
// The _dbar (acquire/release) add, and, or variants: opcodes read off
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
}
for m, op := range am {
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
}
// ---- LSX/LASX (V*/XV*) ----
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
// most of them), not assumed from the LoongArch manual.
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
vec3 := map[string]l64Vec3Enc{
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
"XVADDV": {0xE817 << 15, true},
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
}
for m, e := range vec3 {
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
l64VecBank[m] = e.lasx
}
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
// range, bias and field mask are the ones the toolchain encodes: vandi.b
// stores the raw 8-bit constant, vsrai.b stores imm+8 (byte-lane bias),
// vseqi.b and vseqi.d store 5-bit and 7-bit two's-complement values.
// The mnemonics that also have a register form (VSEQB, VSEQV, VSRAB,
// VROTRW) keep their three-register entry in l64InstrTable; the
// dispatcher picks the immediate opcode from l64VecImmInfo by operand
// kind, so the immediate entries must not overwrite the table.
vecImm := map[string]l64VecImmEnc{
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -64, 63, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -64, 63, 0, 0x7F},
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
}
for m, e := range vecImm {
l64VecImmInfo[m] = e
l64VecBank[m] = e.lasx
}
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
// vsetallnez.*): the sub-op rides in the rk field.
vecCf := map[string]uint32{
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
}
for m, op := range vecCf {
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
l64VecBank[m] = strings.HasPrefix(m, "XV")
}
// Lane popcount: INSTR vj, vd (the 2R layout with the opcode extending
// over the unused vk field).
vec2r := map[string]l64Vec3Enc{
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
}
for m, e := range vec2r {
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
l64VecBank[m] = e.lasx
l64Vec2R[m] = true
}
}
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
+261
View File
@@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) {
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
}
}
// The X/V spellings name the LSX/LASX vector banks, a register class of
// their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0`
// with "unrecognized instruction" while assembling `VADDV V0, V1, V2`
// and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector
// operands resolve through loong64VecRegNum only.
vecCases := map[string]int{
"V0": 0, "V31": 31, "X0": 0, "X31": 31,
"R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1,
}
for name, want := range vecCases {
if got := loong64VecRegNum(name); got != want {
t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want)
}
}
}
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
@@ -328,3 +342,250 @@ TEXT ·f(SB), NOSPLIT, $0-0
0x4C000020, // jirl r0, r1, 0 (RET)
)
}
// TestLOONG64_vector pins the LSX/LASX slice against words read off
// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own
// loong64enc1.s): the three-register forms, the immediate forms with their
// biases, the vector-to-condition forms, lane popcount, the FP conversion,
// FSEL and the VMOVQ move family.
func TestLOONG64_vector(t *testing.T) {
t.Run("three-register and immediate forms", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VADDV V1, V2, V3
VADDW V1, V2, V3
VADDV V2, V1
VANDV V1, V2
VXORV V1, V2, V3
VSEQB V1, V2, V3
VSEQV V1, V2, V3
VSRAB V1, V2, V3
VROTRW V1, V2, V3
VANDB $0, V2, V3
VANDB $255, V2
VSEQB $3, V2, V3
VSEQV $15, V2, V3
VSEQV $-15, V2, V3
VSRAB $7, V1, V2
VROTRW $16, V1, V2
VPCNTV V1, V2
XVADDV X1, X2, X3
XVXORV X1, X2, X3
XVSEQB X1, X2, X3
XVPCNTV X1, X2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x700B8443, // vadd.v v3, v2, v1
0x700B0443, // vadd.w
0x700B8821, // vadd.v v1, v1, v2 (two-operand form)
0x71260442, // vand.v v2, v2, v1
0x71270443, // vxor.v
0x70000443, // vseq.b
0x70018443, // vseq.d
0x70EC0443, // vsra.b
0x70EF0443, // vrotr.w
0x73D00043, // vandi.b v3, v2, 0
0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form)
0x72800C43, // vseqi.b v3, v2, 3
0x7281BC43, // vseqi.d v3, v2, 15
0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement)
0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8)
0x72A0C022, // vrotri.w v2, v1, 16
0x729C2C22, // vpcnt.d v2, v1
0x740B8443, // xvadd.d x3, x2, x1
0x75270443, // xvxor.d
0x74000443, // xvseq.b
0x769C2C22, // xvpcnt.d x2, x1
0x4C000020,
)
})
t.Run("vector-to-condition", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETNEV V1, FCC0
VSETANYEQB V1, FCC0
VSETANYEQV V2, FCC0
VSETALLNEV V0, FCC0
XVSETNEV X1, FCC0
XVSETALLNEV X1, FCC0
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x729C9C20, // vsetnez.d fcc0, v1
0x729CA020, // vsetanyeqz.b
0x729CAC40, // vsetanyeqz.d
0x729CBC00, // vsetallnez.d
0x769C9C20, // xvsetnez.d
0x769CBC20, // xvsetallnez.d
0x4C000020,
)
})
t.Run("FP convert and FSEL", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
FFINTDV F0, F1
FSEL FCC0, F3, F4, F3
FSEL FCC1, F1, F2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x011D2801, // ffint.d.v f1, f0
0x0D000C83, // fsel f3, f4, f3, fcc0
0x0D008442, // fsel f2, f2, f1, fcc1
0x4C000020,
)
})
t.Run("VMOVQ move family", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMOVQ V1, V9
VMOVQ (R4), V2
VMOVQ 16(R4), V2
VMOVQ V0, (R4)
VMOVQ V0, 32(R4)
VMOVQ (R4)(R7), V3
VMOVQ V3, (R4)(R7)
VMOVQ R6, V0.B16
VMOVQ R6, V12.W4
VMOVQ (R4), V4.W4
XVMOVQ X3, X7
XVMOVQ (R4), X2
XVMOVQ X0, (R4)
XVMOVQ (R4)(R7), X4
XVMOVQ X0, (R4)(R7)
XVMOVQ R6, X0.B32
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x732D0029, // vori.b v9, v1, 0 (register move)
0x2C000082, // vld v2, r4, 0
0x2C004082, // vld v2, r4, 16
0x2C400080, // vst v0, r4, 0
0x2C408080, // vst v0, r4, 32
0x38401C83, // vldx v3, r4, r7
0x38441C83, // vstx v3, r4, r7
0x729F00C0, // vreplgr2vr.b v0, r6
0x729F08CC, // vreplgr2vr.w v12, r6
0x30200084, // vldrepl.w v4, r4, 0
0x772D0067, // xvori.b x7, x3, 0
0x2C800082, // xvld x2, r4, 0
0x2CC00080, // xvst x0, r4, 0
0x38481C84, // xvldx x4, r4, r7
0x384C1C80, // xvstx x0, r4, r7
0x769F00C0, // xvreplgr2vr.b x0, r6
0x4C000020,
)
})
t.Run("element extract and insert", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMOVQ V0.V[0], R10
VMOVQ V6.V[1], R8
VMOVQ R9, V1.V[0]
XVMOVQ X0.V[0], R10
XVMOVQ X5.W[7], R7
XVMOVQ R4, X7.V[3]
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72EFF00A, // vpickve2gr.d r10, v0, 0
0x72EFF4C8, // vpickve2gr.d r8, v6, 1
0x72EBF121, // vinsgr2vr.d v1, r9, 0
0x76EFE00A, // xvpickve2gr.d r10, x0, 0
0x76EFDCA7, // xvpickve2gr.w r7, x5, 7
0x76EBEC87, // xvinsgr2vr.d x7, r4, 3
0x4C000020,
)
})
}
// TestLOONG64_vectorErrors pins the register-class and range diagnostics of
// the vector slice; each shape is rejected by the oracle as well
// (GOARCH=loong64 go tool asm).
func TestLOONG64_vectorErrors(t *testing.T) {
cases := []string{
// Integer registers in vector positions.
`TEXT ·e(SB), NOSPLIT, $0
VADDV R4, R5, R6
RET
`,
// Crossed banks: LSX spellings take V, LASX spellings X.
`TEXT ·e(SB), NOSPLIT, $0
VADDV X1, X2, X3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVADDV V1, V2, V3
RET
`,
// The LASX bank has no .b/.h element forms.
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ R4, X2.B[0]
RET
`,
// Immediate ranges.
`TEXT ·e(SB), NOSPLIT, $0
VANDB $256, V2
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VSEQB $16, V2, V3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VROTRW $32, V1, V2
RET
`,
// VSET* wants an FCC flag, not a vector register.
`TEXT ·e(SB), NOSPLIT, $0
VSETNEV V1, V2
RET
`,
}
for i, src := range cases {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants.
// The oracle words come from GOARCH=loong64 go tool objdump of kernels
// assembled with go tool asm, and match the toolchain's loong64enc1.s.
func TestLOONG64_dbarAtomics(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·atoms(SB), NOSPLIT, $0
AMADDDBW R14, (R13), R12
AMADDDBV R14, (R13), R12
AMANDDBW R5, (R4), R6
AMANDDBV R5, (R4), R6
AMORDBW R5, (R4), R0
AMORDBV R5, (R4), R6
AMSWAPDBW R5, (R4), R6
AMCASDBV R6, (R4), R5
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x386A39AC, // amadd_db.w r12, r13, r14
0x386AB9AC, // amadd_db.d
0x386B1486, // amand_db.w r6, r4, r5
0x386B9486, // amand_db.d
0x386C1480, // amor_db.w r0, r4, r5
0x386C9486, // amor_db.d
0x38691486, // amswap_db.w
0x385B9885, // amcas_db.w
0x4C000020,
)
}
+4 -1
View File
@@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42
}
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
// invalid register operands and operand-count mismatches.
// invalid register operands and operand-count mismatches. The X0 and
// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects
// `BEQZ X0` (the X bank is not an integer register) and the two-register
// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`).
func TestLOONG64_errors(t *testing.T) {
cases := []string{
`TEXT ·e(SB), NOSPLIT, $0
+13
View File
@@ -48,3 +48,16 @@ type sbMem struct {
}
func (sbMem) isOperand() {}
// isX86Mem reports whether the operand is an amd64 memory reference: a base
// or indexed Mem, or an SB-relative sbMem. Encoders that gate on "memory in
// this position" must accept both; the r/m emitters distinguish the two
// themselves.
func isX86Mem(o Operand) bool {
switch o.(type) {
case Mem, sbMem:
return true
default:
return false
}
}
+6 -1
View File
@@ -17,12 +17,13 @@ import "strings"
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0-K7.
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register
}
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
@@ -144,6 +145,10 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ {
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
}
// x87 stack: F0..F7.
for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
}
return m
}
+1130 -47
View File
File diff suppressed because it is too large Load Diff
+150 -30
View File
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
case "F31", "FT11":
return 31
default:
// Vector registers V0-V31 (the "V" extension). They share the
// register numbering with the integer file: a bare number 0-31.
if len(name) >= 2 && name[0] == 'V' {
if n, ok := parseRegDigits(name[1:], 31); ok {
return n
}
}
return -1
}
}
// parseRegDigits parses a decimal register suffix and reports whether it is
// within [0, max].
func parseRegDigits(digits string, max int) (int, bool) {
if digits == "" {
return 0, false
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return 0, false
}
n = n*10 + int(digits[i]-'0')
if n > max {
return 0, false
}
}
return n, true
}
// RISC-V instruction encoding parameters.
type riscvEnc struct {
opcode uint32 // bits [6:0]
@@ -193,6 +219,9 @@ var riscvInstrTable = map[string]riscvEnc{
"DIVUW": {0x3B, 0x5, 0x01},
"REMW": {0x3B, 0x6, 0x01},
"REMUW": {0x3B, 0x7, 0x01},
// Zicond conditional zeroing.
"CZEROEQZ": {0x33, 0x5, 0x07},
"CZERONEZ": {0x33, 0x7, 0x07},
// RV64I, I-type arithmetic.
"ADDI": {0x13, 0x0, 0x00},
"ADDIW": {0x1B, 0x0, 0x00},
@@ -221,36 +250,47 @@ var riscvInstrTable = map[string]riscvEnc{
"BGE": {0x63, 0x5, 0x00},
"BLTU": {0x63, 0x6, 0x00},
"BGEU": {0x63, 0x7, 0x00},
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
// with the register operands swapped.
"BGT": {0x63, 0x4, 0x00},
"BLE": {0x63, 0x5, 0x00},
"BGTU": {0x63, 0x6, 0x00},
"BLEU": {0x63, 0x7, 0x00},
// U-type.
"LUI": {0x37, 0x0, 0x00},
"AUIPC": {0x17, 0x0, 0x00},
// System.
"ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00},
"ECALL": {0x73, 0x0, 0x00},
"EBREAK": {0x73, 0x0, 0x00},
"FENCE": {0x0F, 0x0, 0x00},
"FENCE.TSO": {0x0F, 0x0, 0x00},
"PAUSE": {0x0F, 0x0, 0x00},
// JALR, indirect jump/call (I-type).
"JALR": {0x67, 0x0, 0x00},
// RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
"AMOORW": {0x2F, 0x2, 0x06 << 2},
"AMOORD": {0x2F, 0x3, 0x06 << 2},
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
// the lower two, exactly as the toolchain writes them: every AMO sets both
// aq and rl (funct7 |= 3).
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
// RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00},
@@ -273,12 +313,23 @@ var riscvInstrTable = map[string]riscvEnc{
"FMAXS": {0x53, 0x1, 0x14},
"FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15},
// FP sign injection (double): rs2 carries the sign source.
"FSGNJD": {0x53, 0x0, 0x11},
"FSGNJS": {0x53, 0x0, 0x10},
"FSGNJX": {0x53, 0x0, 0x14},
"FSGNJXD": {0x53, 0x0, 0x15},
"FSGNJXS": {0x53, 0x0, 0x14},
"FSGNJND": {0x53, 0x1, 0x11},
"FSGNJNS": {0x53, 0x1, 0x10},
"FSGNJNX": {0x53, 0x1, 0x14},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2},
"LRD": {0x2F, 0x3, 0x02 << 2},
"SCW": {0x2F, 0x2, 0x03 << 2},
"SCD": {0x2F, 0x3, 0x03 << 2},
// The toolchain gives LR acquire ordering (aq = 1) and SC release
// ordering (rl = 1).
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
// FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50},
@@ -296,11 +347,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
}
// riscvAMOType encodes an atomic (AMO) instruction.
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
// (the table stores the full field, so the word needs no reassembly).
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
}
@@ -342,6 +393,10 @@ var riscvCvtTable = map[string]riscvCvtEnc{
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
// The toolchain's W/D suffix spellings of the same moves.
"FMVXS": {0x70, 0x0, 0x53},
"FMVFS": {0x78, 0x0, 0x53},
"FMVSX": {0x79, 0x0, 0x53},
}
// riscvCvtType encodes an FP conversion instruction.
@@ -441,6 +496,71 @@ func riscvJType(rd int, offset int32) uint32 {
0x6F // JAL opcode
}
// ---- RVV ("V" extension) encoding helpers ----
// The OP-V major opcode and its funct3 subclasses.
const (
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
riscvVf3VV = 0x0 // vector-vector
riscvVf3MV = 0x2 // vector mask
riscvVf3VI = 0x3 // vector-immediate
riscvVf3VX = 0x4 // vector-scalar
riscvVf3Cfg = 0x7 // vsetvli
)
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
// group multiplier in [2:0], the selected element width in [5:3] and the
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
func riscvVType(vsew, vlmul, vta, vma int) int {
return vlmul | vsew<<3 | vta<<6 | vma<<7
}
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
// the toolchain writes above its 10-bit vtype).
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
imm := vtype & 0x3FF
if vsetivli {
imm |= 0xC00
}
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
uint32(rd)<<7 | riscvOpV
}
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
// segment count in bits [31:29], the addressing mode in bits [28:26], the
// unmasked bit at 25 and the width in funct3. width follows the load
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
uint32(rd)<<7 | op
}
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
// the standard positions. vs1 may name an integer register for the *VX forms
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
}
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
// fields live where the second source register would be: rs1Field and vs2 are
// written verbatim (the oracle writes fixed non-zero constants there for some
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
}
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
func riscvSegNF(n int) int32 { return int32(n - 1) }
// ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
+242 -6
View File
@@ -5,6 +5,8 @@ package asm
import (
"bytes"
"encoding/binary"
"encoding/hex"
"strings"
"testing"
@@ -866,7 +868,7 @@ func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]in
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
instr := fn.Body[0].(*ast.Instr)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil)
}
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
@@ -903,9 +905,10 @@ func TestRISCVBranchJumpRange(t *testing.T) {
}
}
// TestRISCVBranchFarBody drives the range check through the full two-pass
// assembler: a forward branch over a body larger than the B-type span must
// error rather than wrap.
// TestRISCVBranchFarBody drives the relaxation pass through the full
// assembler: a forward branch over a body larger than the B-type span is
// rewritten as an inverted branch over an inserted JMP, the same layout the
// toolchain produces, instead of wrapping to a wrong target.
func TestRISCVBranchFarBody(t *testing.T) {
var sb strings.Builder
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
@@ -914,8 +917,20 @@ func TestRISCVBranchFarBody(t *testing.T) {
}
sb.WriteString("done:\n\tRET\n")
fn := firstTextRISCV(t, sb.String())
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
t.Error("expected a branch-out-of-range error, got none")
out, _, _, _, _, err := assembleRISCV(fn)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
// The relaxed branch at offset 0 targets the inserted JMP at 4 (bne
// x10, x11, +4); the JMP at 4 carries the far forward displacement.
wantBranch := wordLE(riscvBType(riscvEnc{0x63, 0x1, 0x00}, 10, 11, 4))
if !bytes.Equal(out[0:4], wantBranch) {
t.Errorf("relaxed branch = %x, want %x", out[0:4], wantBranch)
}
// done sits after 1100 ADDs: 4 + 4400, i.e. offset 4404 from the JMP at 4.
wantJmp := wordLE(riscvJType(0, 4404))
if !bytes.Equal(out[4:8], wantJmp) {
t.Errorf("inserted JMP = %x, want %x", out[4:8], wantJmp)
}
}
@@ -971,3 +986,224 @@ TEXT ·edge(SB), NOSPLIT, $0
t.Errorf("int32-span immediates must assemble: %v", err)
}
}
// riscvWants decodes code as little-endian words and pins each one; the
// expected values below were read off GOARCH=riscv64 go tool objdump of
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWants(t *testing.T, code []byte, want ...uint32) {
t.Helper()
got := make([]uint32, 0, len(code)/4)
for i := 0; i+4 <= len(code); i += 4 {
got = append(got, binary.LittleEndian.Uint32(code[i:]))
}
if len(got) < len(want) {
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
}
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// riscvWantsHex pins the exact hex encoding of a function's instruction
// bytes, including any 2-byte compressed instructions in the stream; the
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
// assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
t.Helper()
got := hex.EncodeToString(code)
if got != wantHex {
t.Errorf("code = %s, want %s", got, wantHex)
}
}
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
// the time CSR).
func TestRISCV_extendedPseudos(t *testing.T) {
t.Run("logic and minmax", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·l(SB), NOSPLIT, $0
ANDN X19, X20, X21
ANDN X19, X20
ORN X20, X19
MAX X26, X28, X29
MIN X29, X30, X5
MAX X5, X5
MAX X5, X5, X6
SEQZ X5, X6
NEG X5, X6
NOT X5
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
// then SEQZ, NEG, NOT and RDTIME.
riscvWantsHex(t, code,
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
"9682"+"1683"+
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
})
t.Run("rotate", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·r(SB), NOSPLIT, $0
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORIW $31, X13, X14
RORIW $1, X14, X15
RORIW $3, X14
RORW X15, X16, X17
RORW $31, X13
RET
`)
code := assembleRISCVHelper(t, fn)
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
riscvWantsHex(t, code,
"b30fa040b39ff50133d6a50033e6cf00"+
"b30fa040b39ff501b3d5a500b3e5bf00"+
"93dff5038605b3e5bf00"+
"9bdff6011b97160033e7ef00"+
"9b5f17009b17f701b3e7ff00"+
"9b5f37001b17d70133e7ef00"+
"b30ff040bb1ff801bb58f800b3e81f01"+
"9bdff6019b961600b3e6df00"+"67800000")
})
t.Run("fp and branches", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
FABSD F1, F2
FSGNJD F1, F0, F2
FMADDD F1, F2, F3, F4
FMSUBD F1, F2, F3, F4
FNMSUBD F1, F2, F3, F4
BGT X5, X6, tgt
BLE X5, X6, tgt
BGTU X5, X6, tgt
BLEU X5, X6, tgt
tgt:
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWantsHex(t, code,
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
"f32210c0"+"67800000")
})
}
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
// exactly as GOARCH=riscv64 go tool asm encodes them.
func TestRISCV_amoWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0
AMOSWAPW X5, (X6), X7
AMOSWAPD X5, (X6), X7
AMOADDW X5, (X6), X7
AMOADDD X5, (X6), X7
AMOANDW X5, (X6), X7
AMOANDD X5, (X6), X7
AMOORW X5, (X6), X7
AMOORD X5, (X6), X7
AMOXORW X5, (X6), X7
AMOXORD X5, (X6), X7
AMOMAXW X5, (X6), X7
AMOMAXD X5, (X6), X7
AMOMAXUW X5, (X6), X7
AMOMAXUD X5, (X6), X7
AMOMINUW X5, (X6), X7
AMOMINUD X5, (X6), X7
LRW (X5), X6
LRD (X5), X6
SCW X5, (X6), X7
SCD X5, (X6), X7
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0E5323AF, // amoswap.w
0x0E5333AF, // amoswap.d
0x065323AF, // amoaddd.w
0x065333AF, // amoadd.d
0x665323AF, // amoand.w
0x665333AF, // amoand.d
0x465323AF, // amoor.w
0x465333AF, // amoor.d
0x265323AF, // amoxor.w
0x265333AF, // amoxor.d
0xA65323AF, // amomax.w
0xA65333AF, // amomax.d
0xE65323AF, // amomaxu.w
0xE65333AF, // amomaxu.d
0xC65323AF, // amominu.w
0xC65333AF, // amominu.d
0x1402A32F, // lr.w (aq)
0x1402B32F, // lr.d
0x1A5323AF, // sc.w (rl)
0x1A5333AF, // sc.d
)
}
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
// toolchain canonicalises an immediate avl to vsetivli even under the
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
// identical), which is what the 0xC00 bit of the first word carries.
func TestRISCV_vectorWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETVLI X5, E8, M8, TA, MA, X6
VSETIVLI $4, E32, M1, TA, MA, X0
VSETVLI $15, E32, M1, TA, MA, X12
VADDVV V1, V2, V3
VADDVX X12, V12, V12
VXORVV V8, V16, V24
VMSEQVX X12, V8, V0
VMSNEVV V8, V16, V0
VSLLVI $8, V28, V30
VSRLVI $25, V29, V29
VFIRSTM V0, X6
VIDV V12
VMV4RV V8, V24
VLE8V (X10), V8
VSE8V V24, (X10)
VSE32V V9, (X11)
VLSSEG4E32V (X14), X0, V0
VLSSEG8E32V (X10), X0, V4
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
0xCD027057, // vsetivli x0, 4
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
0x022081D7, // vadd.vv v3, v2, v1
0x02C64657, // vadd.vx v12, v12, x12
0x2F040C57, // vxor.vv v24, v16, v8
0x62864057, // vmseq.vx v0, v8, x12
0x67040057, // vmsne.vv v0, v16, v8
0x97C43F57, // vsll.vi v30, v28, 8
0xA3DCBED7, // vsrl.vi v29, v29, 25
0x4208A357, // vmfirst.m x6, v0
0x5208A657, // vid.v v12
0x9E81BC57, // vmv4r.v v24, v8
0x02050407, // vle8.v v8, (x10)
0x02050C27, // vse8.v v24, (x10)
0x0205E4A7, // vse32.v v9, (x11)
0x6A076007, // vlsseg4e32.v v0, (x14), x0
0xEA056207, // vlsseg8e32.v v4, (x10), x0
)
}
+144 -1
View File
@@ -41,7 +41,7 @@ const (
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
@@ -52,6 +52,15 @@ const (
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
// vexZeroAll is the no-operand form that zeroes the full upper state
// (VZEROALL, the L = 1 twin of VZEROUPPER).
vexZeroAll
// vexNDS3GPR is the three-operand NDS form over general-purpose
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
vexNDS3GPR
// vexImmRMGPR is the immediate form over general-purpose registers
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
vexImmRMGPR
)
// vexSpec describes one VEX instruction's encoding parameters.
@@ -125,6 +134,12 @@ var vexTable = map[string]vexSpec{
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// Scalar fused multiply-add (NDS form). The Go assembler carries the
// same 66 prefix as the packed forms on every FMA row, and W1 on the
// double-precision spellings, so SD shares PD's prefix/W pair and the
// scalar width rides on the W bit.
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
@@ -192,6 +207,31 @@ var vexTable = map[string]vexSpec{
// VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
// falls through to the EVEX table.
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
@@ -200,6 +240,14 @@ var vexTable = map[string]vexSpec{
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
// YMM (the encoder rejects an XMM destination, as go tool asm does).
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
// memory destination in rm).
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
@@ -290,6 +338,8 @@ type vexMoveSpec struct {
var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, aligned integer move.
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
@@ -324,6 +374,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
}
}
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
// only; an XMM destination is rejected exactly as go tool asm does.
if mnemUpper == "VBROADCASTI128" {
dstReg, ok := ops[len(ops)-1].(Reg)
if len(ops) != 2 || !ok || dstReg.size != 32 {
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
}
}
if ms, ok := vexMoveTable[mnemUpper]; ok {
return e.encodeVexMove(mnemUpper, ms, ops)
}
@@ -356,6 +414,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
case vexZeroAll:
return e.encodeVexZeroAll(mnemUpper, spec, ops)
case vexNDS3GPR:
return e.encodeVexNDS3GPR(spec, ops)
case vexImmRMGPR:
return e.encodeVexImmRMGPR(spec, ops)
case vexRMRev:
return e.encodeVexRMRev(spec, ops)
}
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
}
@@ -607,6 +673,83 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
return nil
}
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
// twin of VZEROUPPER.
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 0 {
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
}
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
return nil
}
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
// rm = src2 and L = 0.
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
vvvvReg, ok := src1.(Reg)
if !ok || vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
}
// encodeVexImmRMGPR encodes the immediate form over general-purpose
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
// a store with no register-destination form).
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("store source must be a vector register")
}
if !memOperand(ops[1]) {
return fmt.Errorf("store destination must be memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
+59
View File
@@ -19,6 +19,20 @@ func vreg(t *testing.T, name string) Reg {
return r
}
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
// go tool asm in the ground-truth tests instead.
var x86asmUnrecognised = map[string]bool{
"ANDNL": true,
"ANDNQ": true,
"MULXL": true,
"MULXQ": true,
"RORXL": true,
"RORXQ": true,
"VFMADD213SD": true,
"VFNMADD231SD": true,
}
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
// instruction and verifies it round-trips through the x86 decoder to the same
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
@@ -37,8 +51,15 @@ func TestVexNDS3(t *testing.T) {
t.Errorf("%s: Encode: %v", mnem, err)
continue
}
// The x86 decoder's table lacks a handful of rows the Go assembler
// emits (the scalar 213/231 FMA spellings among them); those are
// pinned byte for byte against go tool asm in TestVexGroundTruth
// instead of round-tripped here.
inst, err := x86asm.Decode(code, 64)
if err != nil {
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
continue
}
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
continue
}
@@ -184,6 +205,38 @@ func TestVexGroundTruth(t *testing.T) {
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
// Packed single XOR and byte compare (NDS form).
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
// Octa byte shifts (vvvv carries the destination).
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
// Two-operand test and the non-temporal and broadcast stores.
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
// Aligned integer move and the full zeroing form.
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
// BMI1/BMI2 general-register VEX forms.
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
// Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
@@ -287,6 +340,12 @@ func TestVexGroundTruth(t *testing.T) {
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
// The decoder's AVX/BMI table lacks a few rows the Go
// assembler emits (the GPR VEX forms and the scalar FMA
// spellings); their bytes are the ground truth here.
if x86asmUnrecognised[c.mnem] {
continue
}
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
continue
}
+127 -22
View File
@@ -37,7 +37,7 @@ import (
// construction and are excluded from the diff; the other architectures list
// their conditional branches outright.
func cmdAuditInstructions(args []string) error {
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]", `
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]", `
Compare the gasm encoder for the given architecture (default amd64) against
go tool asm and print the diff: superset encodings (gasm-only, shippable via
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
@@ -57,11 +57,13 @@ per-architecture pass rates and the most common failure reasons, which drive
the encodability backlog by frequency rather than by table order.
`)
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
if err := fs.Parse(args); err != nil {
return err
}
if *corpus {
return cmdAuditCorpus(fs.Args())
return cmdAuditCorpus(fs.Args(), dirs)
}
archName := "amd64"
switch n := len(fs.Args()); {
@@ -266,18 +268,62 @@ func probeShapes(a arch.Arch) []string {
// and takes R register spellings.
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
"GE, F0, F1, F2", "NE, F0, F1, $0",
// Pairs, acquire/release and exclusive atomics, LSE-AL forms.
"(R0), R1", "R0, (R1)", "R1, (R2), R3", "(R2, R3), 8(R1)",
"8(R1), (R2, R3)", "R1, R2, (R3)", "(R0)",
// System operations and their register/operand names.
"$4, R1, p2", "$35943", "$1", "$1, SPSel", "SPSel, R0",
"IVAC, R0", "(R0), PLDL1KEEP", "R1, R2, R3, R4",
// SIMD element, structure and literal-pool forms.
"(R0), [V1.B16]", "[V1.B16], (R0)", "V13.S[0], R1",
"R1, V2.B[3]", "$4, V1.B16, V2.B16", "V1.B16, (R0)",
"(R0), V1.B16", "",
// The spellings GOROOT's own kernels use, from the
// differential kernels this table was proven against.
"R0, p2", "R0, R1", "F0, F1, F2, F3", "$4, V1.B16, V2.B16, V3.B16, V4.B16",
"(R0), [V0.B8, V1.B8, V2.B8, V3.B8]", "$1, $2, V1",
"R0, R1, p2", "p2, R1", "$1234, R1", "DCZID_EL0, R1",
"$0", "R1, $4, EQ", "$33, R1, $25, R2", "$4, R1, p2",
"$4, V1.B8, V2.B8, V3.B8", "$63, V1.D2, V2.D2, V3.D2",
"V1.B16, [V2.B16], V3.B16", "V1.B8, [V2.B16, V3.B16], V4.B8",
"$4, V1.B16, V2.B16, V3.B16", "$15, V1", "V1, V2, p2",
"R0, R1, $1, $4, p2",
}
case arch.RISCV:
return []string{
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
"X5, X6, p2", "p2(SB)",
// AMO atomics: destination, base, source.
"R5, (R4), R6", "X5, (X4), X6",
// Segment stores take the first vector register aligned
// to the segment count, as the toolchain requires.
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
// The FP multiply-add family takes four registers.
"F0, F1, F2, F3",
// The RVV slice: register, vector-register and vtype forms.
"V1, V2, V3", "V1, X5, V2", "V1", "V1, (X5)", "(X5), V1",
"$15, V1", "$15", "V1, V2", "V1, X5",
"X5, X6, p2", "R5, R6, p2",
"X5, E8, M8, TA, MA, X6", "$4, E32, M1, TA, MA, X1",
"(X5), X6, V1, V2",
"",
}
case arch.LOONG64:
return []string{
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
// AMO atomics: destination, base, source.
"R5, (R4), R6", "X5, (X4), X6",
// Segment stores take the first vector register aligned
// to the segment count, as the toolchain requires.
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
// The LSX and LASX banks share the 5-bit numbering with F.
"V1, V2, V3", "X1, X2, X3", "V1, V2", "X1, X2", "V1", "X1",
// The vector compare-to-flag forms land in an FCC register.
"V1, FCC0", "X1, FCC0",
"",
}
}
return nil
@@ -351,8 +397,11 @@ func (t *corpusTally) fail(path, reason string) {
}
}
// cmdAuditCorpus implements audit-instructions --corpus.
func cmdAuditCorpus(args []string) error {
// cmdAuditCorpus implements audit-instructions --corpus. The include
// directories carry #include resolution over a corpus whose files refer to
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
// needs.
func cmdAuditCorpus(args []string, dirs includeDirs) error {
if len(args) > 1 {
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
}
@@ -366,7 +415,25 @@ func cmdAuditCorpus(args []string) error {
}
root = filepath.Join(strings.TrimSpace(string(out)), "src")
}
stats, err := runCorpusAudit(root)
// The toolchain's shipped headers (funcdata.h and friends) define the
// macros GOROOT files include; a corpus audit measures those files, so
// the header directory joins the search path automatically. go_asm.h
// is compiler-generated per package and stays unresolvable on purpose.
if out, err := exec.Command("go", "env", "GOROOT").Output(); err == nil {
pkgInclude := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
if fi, err := os.Stat(pkgInclude); err == nil && fi.IsDir() {
seen := false
for _, d := range dirs {
if d == pkgInclude {
seen = true
}
}
if !seen {
dirs = append(dirs, pkgInclude)
}
}
}
stats, err := runCorpusAudit(root, dirs)
if err != nil {
return err
}
@@ -376,16 +443,41 @@ func cmdAuditCorpus(args []string) error {
// corpusStats is the outcome of one corpus audit run.
type corpusStats struct {
root string
files int
generic int // files attempted for all four architectures
full int // files that assembled for every target architecture
targets []corpusTarget
tallies []*corpusTally
root string
files int
generic int // files attempted for all four architectures
otherPort int // files named for another Go port: never attempted
full int // files that assembled for every target architecture
targets []corpusTarget
tallies []*corpusTally
}
// runCorpusAudit assembles every .s file under root and returns the stats.
func runCorpusAudit(root string) (*corpusStats, error) {
// goPortSuffixes lists every architecture the Go project ports to. A file
// named for one of them belongs to that port's build, not to the generic
// set, even when gasm does not support the architecture.
var goPortSuffixes = []string{
"386", "amd64", "arm", "arm64", "loong64", "mips", "mips64",
"mips64le", "mipsle", "ppc64", "ppc64le", "riscv", "riscv64",
"s390x", "wasm",
}
// otherPortFile reports whether the file's name carries a Go-architecture
// suffix gasm does not support.
func otherPortFile(path string) bool {
base := path
if i := strings.LastIndexByte(base, '/'); i >= 0 {
base = base[i+1:]
}
for _, sfx := range goPortSuffixes {
if strings.HasSuffix(base, "_"+sfx+".s") {
return true
}
}
return false
}
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
files, err := asmFiles(root)
if err != nil {
return nil, err
@@ -403,14 +495,14 @@ func runCorpusAudit(root string) (*corpusStats, error) {
}
// full is the north-star number: a file counts when every architecture
// its name allows assembles it.
full, generic := 0, 0
full, generic, otherPort := 0, 0, 0
for _, path := range files {
src, err := readSource(path)
if err != nil {
return nil, err
}
f, errs := parser.Parse(path, src)
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
var wanted []int // indexes into targets
if a := arch.FromFilename(path); a != arch.Unknown {
@@ -419,6 +511,13 @@ func runCorpusAudit(root string) (*corpusStats, error) {
wanted = append(wanted, i)
}
}
} else if otherPortFile(path) {
// A file named for a Go port gasm does not support (arm,
// 386, s390x, ...) is compiled by no supported-arch build,
// so it is neither generic nor a per-arch attempt: counting
// it as generic would make the headline unreachably low
// for reasons no supported target can fix.
otherPort++
} else {
generic++
for i := range targets {
@@ -449,19 +548,25 @@ func runCorpusAudit(root string) (*corpusStats, error) {
}
return &corpusStats{
root: root,
files: len(files),
generic: generic,
full: full,
targets: targets,
tallies: tallies,
root: root,
files: len(files),
generic: generic,
otherPort: otherPort,
full: full,
targets: targets,
tallies: tallies,
}, nil
}
// printCorpusStats renders the corpus audit report.
func printCorpusStats(s *corpusStats) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures)\n", s.root, s.files, s.generic)
fmt.Printf(" assemble for every target architecture: %d (%.1f%%)\n", s.full, 100*float64(s.full)/float64(max(s.files, 1)))
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort)
// The rate is over the files a supported build would attempt: the
// other ports' files sit in the count for completeness but can never
// assemble, so counting them in the denominator would report the gap
// of architectures gasm deliberately does not target.
attemptable := max(s.files-s.otherPort, 1)
fmt.Printf(" assemble for every target architecture: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
for i, tg := range s.targets {
t := s.tallies[i]
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
+26 -11
View File
@@ -240,6 +240,16 @@ func readSource(path string) (string, error) {
return string(b), err
}
// includeDirs collects repeatable -I flags: the directories searched for
// #include files during macro expansion and include splicing.
type includeDirs []string
func (d *includeDirs) String() string { return strings.Join(*d, ",") }
func (d *includeDirs) Set(v string) error {
*d = append(*d, v)
return nil
}
func cmdTokens(args []string) int {
fs := newCommand("tokens", "gasm tokens <file>", `
Print the lexical token stream of FILE: position, token kind and text, one
@@ -476,7 +486,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
}
func cmdAsm(args []string) int {
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>", `
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>", `
Assemble FILE without the Go toolchain: every TEXT function is encoded to
machine code and printed as a hex dump. Supported architectures: amd64
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
@@ -498,9 +508,11 @@ and the format version from go version).
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
var dirs includeDirs
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
fs.Parse(args)
if fs.NArg() != 1 {
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>")
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>")
return 2
}
// The format is validated before anything else, so a bogus value exits 2
@@ -526,7 +538,7 @@ and the format version from go version).
fmt.Fprintln(os.Stderr, "gasm:", err)
return 1
}
f, errs := parser.Parse(path, src)
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
@@ -634,7 +646,7 @@ and the format version from go version).
// cmdDiff compares the machine code of two assembly files.
func cmdDiff(args []string) int {
set := newCommand("diff", "gasm diff [-GOARCH arch] <file1.s> <file2.s>", `
set := newCommand("diff", "gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>", `
Compare the machine code produced by assembling two files.
Shows which functions differ and the byte-level differences.
Useful for verifying that two implementations produce identical code,
@@ -645,9 +657,11 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
`)
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64")
var dirs includeDirs
set.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
set.Parse(args)
if set.NArg() != 2 {
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>")
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>")
return 2
}
path1, path2 := set.Arg(0), set.Arg(1)
@@ -675,12 +689,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
}
// Assemble both files.
img1, err := assemblePath(path1, forced)
img1, err := assemblePath(path1, forced, dirs)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
return 1
}
img2, err := assemblePath(path2, forced)
img2, err := assemblePath(path2, forced, dirs)
if err != nil {
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
return 1
@@ -755,14 +769,15 @@ func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
}
}
// assemblePath reads, parses and assembles a file (used by cmdDiff). A
// non-Unknown forced architecture overrides the file-name suffix.
func assemblePath(path string, forced arch.Arch) (*asm.Image, error) {
// assemblePath reads, preprocesses, parses and assembles a file (used by
// cmdDiff). A non-Unknown forced architecture overrides the file-name
// suffix.
func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image, error) {
src, err := readSource(path)
if err != nil {
return nil, err
}
f, errs := parser.Parse(path, src)
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs})
for _, e := range errs {
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
}
+1 -1
View File
@@ -403,7 +403,7 @@ func TestRunCorpusAudit(t *testing.T) {
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
stats, err := runCorpusAudit(dir)
stats, err := runCorpusAudit(dir, nil)
if err != nil {
t.Fatalf("runCorpusAudit: %v", err)
}
+109
View File
@@ -0,0 +1,109 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package main
import (
"os"
"path/filepath"
"strings"
"testing"
)
// writeTree writes a directory of files and returns its root.
func writeTree(t *testing.T, files map[string]string) string {
t.Helper()
dir := t.TempDir()
for name, content := range files {
path := filepath.Join(dir, name)
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
}
return dir
}
// TestAsmMacroAndIncludeEndToEnd drives `gasm asm` over a source with an
// in-file parameterised macro and an include resolved through -I, and checks
// the assembled bytes came from the expansion (the loop body counts six
// increments, two per expanded iteration).
func TestAsmMacroAndIncludeEndToEnd(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
dir := writeTree(t, map[string]string{
"inc/consts.h": "#define NITER 3\n",
"main_amd64.s": "#include \"textflag.h\"\n" +
"#include \"consts.h\"\n" +
"#define STEP(r) ADDQ $1, r; ADDQ $1, r\n" +
"TEXT ·f(SB), NOSPLIT, $0-8\n" +
"\tXORQ AX, AX\n" +
"\tMOVQ $NITER, CX\n" +
"loop:\n" +
"\tSTEP(AX)\n" +
"\tDECQ CX\n" +
"\tJNZ loop\n" +
"\tMOVQ AX, ret+0(FP)\n" +
"\tRET\n",
})
stdout, stderr, code := capture(func() int {
return cmdAsm([]string{"-I", filepath.Join(dir, "inc"), "-GOARCH", "amd64", filepath.Join(dir, "main_amd64.s")})
})
if code != 0 {
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
}
// The macro expanded to two ADDQ $1 encodings in the static body; the
// iteration count lives in the runtime loop.
if n := strings.Count(stdout, "83 c0 01"); n != 2 {
t.Errorf("found %d ADDQ $1 encodings in the image, want 2:\n%s", n, stdout)
}
}
// TestAsmIncludeResolutionOrder pins the -I search order end to end: the
// including file's directory wins over the -I directories.
func TestAsmIncludeResolutionOrder(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
dir := writeTree(t, map[string]string{
"src/main_amd64.s": "#include \"textflag.h\"\n" +
"#include \"vals.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tMOVQ $VAL, AX\n" +
"\tRET\n",
"src/vals.h": "#define VAL 1\n",
"late/vals.h": "#define VAL 2\n",
"early/vals.h": "#define VAL 3\n",
})
stdout, stderr, code := capture(func() int {
return cmdAsm([]string{"-I", filepath.Join(dir, "early"), "-I", filepath.Join(dir, "late"),
"-GOARCH", "amd64", filepath.Join(dir, "src", "main_amd64.s")})
})
if code != 0 {
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
}
// VAL came from src/vals.h, not from either -I directory: the image
// loads the immediate 1.
if !strings.Contains(stdout, "b8 01 00 00 00") {
t.Errorf("expected the source-directory VAL (immediate 1) in:\n%s", stdout)
}
}
// TestAsmMissingIncludeIsAnError pins the diagnostic for an include that
// resolves nowhere on the assembly path.
func TestAsmMissingIncludeIsAnError(t *testing.T) {
if testing.Short() {
t.Skip("runs the assembler end to end")
}
path := writeTemp(t, "main_amd64.s", "#include \"textflag.h\"\n#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
_, stderr, code := capture(func() int { return cmdAsm([]string{"-GOARCH", "amd64", path}) })
if code == 0 {
t.Fatal("gasm asm accepted a file whose include resolves nowhere")
}
if !strings.Contains(stderr, `#include "nothere.h"`) {
t.Errorf("stderr does not name the failing include: %s", stderr)
}
}
+12 -3
View File
@@ -141,12 +141,13 @@ gasm lint kernel_amd64.s
## asm
```text
Usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>
Usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-o out] <file>
```
| Flag | Default | Effect |
|---|---|---|
| `-format` | `raw` | output format: `raw` (concatenated image), `elf` or `goobj` (Go object) |
| `-I` | empty | directory to search for `#include` files; may be repeated, searched in order after the source directory |
| `-p` | empty | package path for `--format goobj`, qualifying the exported symbols |
| `-GOARCH` | empty | target architecture: `amd64`, `arm64`, `riscv64` or `loong64`; overrides the file-name suffix |
| `-o` | empty | write the output to this file instead of a hex dump on stdout |
@@ -162,6 +163,13 @@ system toolchain; `goobj` emits the Go toolchain's own object format, which
installed: the object preamble is captured from `go tool asm` and the format
version from `go version`. `raw` and `elf` need no toolchain at all.
Assembly preprocessing matches the toolchain's: `#define` macros (object and
parameterised) expand at the point of use, `#undef`, `#ifdef`, `#ifndef`,
`#else` and `#endif` behave as in `go tool asm`, `;` separates statements,
and `#include "file"` splices the named file in, resolved against the source
directory and then each `-I` directory in order. `textflag.h` is the one
header that is not spliced: gasm consumes its flag names natively.
```sh
gasm asm hello_amd64.s
```
@@ -305,12 +313,13 @@ gasm debug --func add --cover hello_amd64.s
## diff
```text
Usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>
Usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>
```
| Flag | Default | Effect |
|---|---|---|
| `-GOARCH` | empty | target architecture for both files, overriding the file-name suffixes |
| `-I` | empty | directory to search for `#include` files; may be repeated, searched in order after the source directory |
| `-map` | empty | comma-separated `old=new` pairs to match functions with different names |
Functions are paired by exact name unless `--map` says otherwise, so
@@ -348,7 +357,7 @@ add: 16 bytes, args=24, frame=0 NOSPLIT
## audit-instructions
```text
Usage: gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]
Usage: gasm audit-instructions [--corpus [dir]] [-I dir] [amd64|arm64|riscv64|loong64]
```
Compare the gasm encoder for the given architecture (default amd64) against the
+5 -1
View File
@@ -2,7 +2,7 @@
.SH NAME
gasm-asm \- assemble Plan 9 assembly without the Go toolchain
.SH SYNOPSIS
.B gasm asm [\-\-format raw|elf|goobj] [\-p pkg] [\-GOARCH arch] [\-o out] <file>
.B gasm asm [\-\-format raw|elf|goobj] [\-I dir] [\-p pkg] [\-GOARCH arch] [\-o out] <file>
.SH DESCRIPTION
Assemble FILE without the Go toolchain: every TEXT function is encoded
to machine code and printed as a hex dump. Supported architectures:
@@ -47,6 +47,10 @@ functions link too.
.B \-\-format \fIraw|elf|goobj\fR
Output format; the default is raw.
.TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory.
.TP
.B \-p \fIpkg\fR
Package path for --format goobj, qualifying the exported symbols.
.TP
+7 -1
View File
@@ -2,7 +2,7 @@
.SH NAME
gasm-audit-instructions \- diff the encoder against the Go toolchain, or measure a corpus
.SH SYNOPSIS
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [amd64|arm64|riscv64|loong64]
.B gasm audit\-instructions [\-\-corpus [\fIdir\fR]] [\-I dir] [amd64|arm64|riscv64|loong64]
.SH DESCRIPTION
Compare the gasm encoder for the given architecture (default amd64)
against
@@ -38,6 +38,12 @@ second.
.B \-\-corpus [\fIdir\fR]
Assemble a corpus of .s files and report pass rates and failure
reasons.
.TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory. A corpus run whose files include
toolchain headers (such as GOROOT/pkg/include) needs it, the same -I a
toolchain comparison takes.
.SH EXIT STATUS
The mnemonic-diff mode reports through its output and exits 0; a failed
probe or an unknown architecture exits non-zero.
+5 -1
View File
@@ -2,7 +2,7 @@
.SH NAME
gasm-diff \- compare the machine code of two assembly files
.SH SYNOPSIS
.B gasm diff [\-GOARCH arch] <file1.s> <file2.s>
.B gasm diff [\-GOARCH arch] [\-I dir] <file1.s> <file2.s>
.SH DESCRIPTION
Compare the machine code produced by assembling two files. Shows which
functions differ and the byte-level differences. Useful for verifying
@@ -20,6 +20,10 @@ pairs two variants regardless of suffix.
Target architecture for both files: amd64, arm64, riscv64 or loong64;
overrides the file-name suffixes.
.TP
.B \-I \fIdir\fR
Directory to search for #include files; may be repeated, searched in
order after the source directory.
.TP
.B \-\-map \fIspec\fR
Comma-separated old=new pairs to match functions with different names.
.SH EXIT STATUS
+49 -2
View File
@@ -300,6 +300,26 @@ func wouldMerge(prev, cur token.Token) bool {
return len(kinds) != 2 || kinds[0] != prev.Kind || kinds[1] != cur.Kind
}
// isOperandBracket reports whether t is one of the square-bracket tokens the
// lexer emits, as Illegal tokens carrying their spelling, for the arm64 and
// loong64 register lists and element selectors that valid GAsm source
// contains.
func isOperandBracket(t token.Token) bool {
return t.Kind == token.Illegal && (t.Text == "[" || t.Text == "]")
}
// isOpenBracket reports whether t is the '[' of a register list or element
// selector.
func isOpenBracket(t token.Token) bool {
return t.Kind == token.Illegal && t.Text == "["
}
// isCloseBracket reports whether t is the ']' that closes a register list or
// element selector.
func isCloseBracket(t token.Token) bool {
return t.Kind == token.Illegal && t.Text == "]"
}
// spaceBetween decides whether a single space separates prev and cur.
func spaceBetween(prev, cur token.Token) bool {
// '/' beside '/' or '*' would form a comment opener in the output and
@@ -308,6 +328,22 @@ func spaceBetween(prev, cur token.Token) bool {
return true
}
switch cur.Kind {
case token.Illegal:
// A closing bracket always glues to the text it closes. An opening
// bracket glues to the operand it extends (V31.B[15]) but takes its
// own space after a comma, a mnemonic or an operator, exactly like
// the parenthesis rule below. Any other Illegal spelling is stray.
if isCloseBracket(cur) {
return false
}
if isOpenBracket(cur) {
switch prev.Kind {
case token.Ident, token.Number, token.RParen, token.RAngle:
return false
}
return true
}
return true
case token.RParen:
return false
case token.Comma:
@@ -328,6 +364,10 @@ func spaceBetween(prev, cur token.Token) bool {
}
}
switch prev.Kind {
case token.Illegal:
// '[' opens a bracket group and glues to what follows; ']' closes
// one, and what comes next takes its own space.
return !isOpenBracket(prev)
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash, token.Pipe:
return false
case token.Dollar:
@@ -358,8 +398,15 @@ func splitLines(toks []token.Token) [][]token.Token {
// Illegal tokens carry no canonical spelling: the parser
// reports them as errors where they matter, and the formatter
// drops them so that a stray character cannot survive into the
// output and make the next pass render a different file.
continue
// output and make the next pass render a different file. The
// square brackets of the arm64 and loong64 vector syntaxes are
// the one exception: the lexer gives them no dedicated kind,
// but a register list [V0.B16, V1.B16] and an element selector
// V0.B[3] are valid, load-bearing source, so their tokens stay
// in the stream and renderOps glues them back where they were.
if !isOperandBracket(t) {
continue
}
}
if t.Kind == token.Newline {
lines = append(lines, cur)
+115
View File
@@ -153,6 +153,121 @@ func TestOperandSpacing(t *testing.T) {
}
}
// TestVectorBracketSpacing pins the square-bracket operand forms of the
// arm64 and loong64 vector syntaxes. The lexer emits '[' and ']' as Illegal
// tokens carrying their spelling, and renderOps must glue them back exactly
// where they were: a register list and an element selector are load-bearing
// operands the assembler reads out of the operand text, so no bracket may be
// dropped, and the canonical spelling inside the brackets is tight.
func TestVectorBracketSpacing(t *testing.T) {
cases := map[string]string{
// Register lists of one to four registers.
"[V21.B16]": "[V21.B16]",
"[V17.B16, V18.B16]": "[V17.B16, V18.B16]",
"[V18.D1, V19.D1, V20.D1]": "[V18.D1, V19.D1, V20.D1]",
"[V14.B16, V15.B16, V16.B16, V17.B16]": "[V14.B16, V15.B16, V16.B16, V17.B16]",
// Element selectors.
"V31.B[15]": "V31.B[15]",
"V19.S[0]": "V19.S[0]",
"V1.D[1]": "V1.D[1]",
"V11.B[11], V16.B[12]": "V11.B[11], V16.B[12]",
// Lists beside address operands, on either side.
"32(R1), [V2.B16, V3.B16]": "32(R1), [V2.B16, V3.B16]",
"[V2.S4, V3.S4], (R14)": "[V2.S4, V3.S4], (R14)",
"(R24), [V18.D1, V19.D1]": "(R24), [V18.D1, V19.D1]",
// A spaced spelling canonicalises to the tight one.
"[ V21.B16 ]": "[V21.B16]",
"V31.B [15]": "V31.B[15]",
}
for in, want := range cases {
toks := lexOperands(in)
if got := renderOps(toks); got != want {
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
}
}
}
// TestSIMDBracketRoundTrip formats whole functions carrying the bracket
// shapes of the arm64 vector kernels and pins the output byte for byte. The
// brackets are load-bearing: formatting must not change what the file
// assembles to, so the formatted text keeps every bracket, re-formats to
// itself and still parses cleanly.
func TestSIMDBracketRoundTrip(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"VDUP V31.B[15], R3\n" +
"VTBL V22.B16, [V28.B16], V11.B16\n" +
"VLD1 (R2), [V21.B16]\n" +
"VMOVQ $0x70, $0x80, V10\n" +
"RET\n"
want := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tVDUP V31.B[15], R3\n" +
"\tVTBL V22.B16, [V28.B16], V11.B16\n" +
"\tVLD1 (R2), [V21.B16]\n" +
"\tVMOVQ $0x70, $0x80, V10\n" +
"\tRET\n"
got := Source(in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
if again := Source(got); again != got {
t.Fatalf("not idempotent:\n%q", again)
}
if n := strings.Count(got, "["); n != 3 {
t.Errorf("output carries %d '[', want 3:\n%s", n, got)
}
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
t.Errorf("formatted output no longer parses: %v", errs)
}
}
// TestBracketFormsRoundTrip runs every bracket shape of the vector kernels
// through a full format pass as its own single-instruction function, where
// the canonical form is the line itself indented: formatting must be a no-op
// on each, so no bracket moves, vanishes or gains a space.
func TestBracketFormsRoundTrip(t *testing.T) {
for _, instr := range []string{
"VDUP V31.B[15], V18",
"VDUP V19.S[3], V18.S4",
"VDUP V1.D[1], V2.D2",
"VMOV V13.S[0], R20",
"VMOV V11.B[11], V16.B[12]",
"VMOV R20, V21.B[2]",
"VTBL V22.B16, [V28.B16], V11.B16",
"VTBL V18.B8, [V17.B16, V18.B16], V22.B8",
"VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8",
"VLD1 (R2), [V21.B16]",
"VLD1 (R24), [V18.D1, V19.D1, V20.D1]",
"VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]",
"VLD1.P 32(R1), [V2.B16, V3.B16]",
"VLD1R (R1), [V9.B8]",
"VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]",
"VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)",
"VST1.P [V2.B16], (R1)",
"VST1.P [V2.B16, V3.B16], 32(R1)",
"VMOVQ $0x70, $0x80, V10",
} {
src := "TEXT ·f(SB), NOSPLIT, $0\n" + instr + "\nRET\n"
want := "TEXT ·f(SB), NOSPLIT, $0\n\t" + instr + "\n\tRET\n"
got := Source(src)
if got != want {
t.Errorf("formatting %q:\n got %q\n want %q", instr, got, want)
continue
}
if again := Source(got); again != got {
t.Errorf("not idempotent for %q:\n%q", instr, again)
}
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
t.Errorf("formatted output of %q no longer parses: %v", instr, errs)
}
}
}
// TestFlagListRoundTrip pins the '|' flag separator and the <ABIInternal>
// marker through a full format pass: the bars the Go toolchain requires and
// the ABI bracket must survive byte for byte, on TEXT and GLOBL alike.
+6
View File
@@ -31,6 +31,12 @@ func FuzzFormatIdempotency(f *testing.F) {
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n")
f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n")
f.Add("garbage ### ???\n")
// Line-ending whitespace at the edge of a comment: a CR followed by more
// trailing whitespace once survived the first pass and disappeared on
// re-lexing, so formatting was not idempotent.
f.Add("//\r ")
f.Add("// loop \r\t\nMOVQ AX, BX\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0 // tail\r\n\tMOVQ AX, BX\r\n\tRET\r\n")
f.Fuzz(func(t *testing.T, src string) {
once := Source(src)
@@ -0,0 +1,2 @@
go test fuzz v1
string("//\r ")
+51 -4
View File
@@ -119,7 +119,9 @@ func (l *Lexer) Next() token.Token {
// is a C-preprocessor line continuation (used by #define macros in the
// runtime .s files): splice the lines together by consuming both, so
// the whole macro becomes one logical line that the parser treats as an
// opaque preprocessor directive.
// opaque preprocessor directive. The backslash may also reach its
// newline across whitespace and a trailing comment ("…; \ // note\n"),
// which the toolchain's scanner skips the same way.
for {
c := l.cur()
if c == ' ' || c == '\t' || c == '\r' {
@@ -136,6 +138,16 @@ func (l *Lexer) Next() token.Token {
}
continue
}
if c == '\\' && l.continuationAhead() {
l.advance() // backslash, then the runes the scan saw
for !l.atEnd() && l.cur() != '\n' {
l.advance()
}
if !l.atEnd() {
l.advance() // the newline that closes the continuation
}
continue
}
break
}
@@ -185,16 +197,42 @@ func (l *Lexer) Next() token.Token {
}
}
// continuationAhead reports, without consuming anything, whether the
// backslash at the current position closes onto a newline through nothing
// but horizontal whitespace and one line comment. Positions after the
// backslash are inspected directly on the rune slice so a non-match leaves
// the scanner state untouched.
func (l *Lexer) continuationAhead() bool {
i := l.i + 1
for i < len(l.src) {
switch r := l.src[i]; {
case r == ' ' || r == '\t' || r == '\r':
i++
case r == '/' && i+1 < len(l.src) && l.src[i+1] == '/':
for i < len(l.src) && l.src[i] != '\n' {
i++
}
default:
return r == '\n'
}
}
return false
}
// lineComment consumes a // comment up to, but not including, the newline. A
// trailing \r is part of a CRLF line ending rather than comment content:
// dropping it keeps the formatter's output uniformly LF-terminated.
// trailing run of \r, spaces and tabs is line-ending whitespace rather than
// comment content, so it never enters the token text. Trimming only a \r
// directly before the token's end would make the text depend on what follows
// the comment (a newline or the end of the input): "//x\r " would carry the
// "\r " while "//x\r\n" would not, and a formatter that terminates the line
// with \n would then re-lex its own output to a shorter comment.
func (l *Lexer) lineComment(start token.Position) token.Token {
var b strings.Builder
for !l.atEnd() && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Comment, start, strings.TrimSuffix(b.String(), "\r"))
return l.make(token.Comment, start, strings.TrimRight(b.String(), " \t\r"))
}
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
@@ -399,6 +437,15 @@ func (l *Lexer) punct(start token.Position) token.Token {
case '|':
l.advance()
return l.make(token.Pipe, start, "|")
case ';':
l.advance()
return l.make(token.Semicolon, start, ";")
case '&':
l.advance()
return l.make(token.Ampersand, start, "&")
case '~':
l.advance()
return l.make(token.Tilde, start, "~")
default:
// Unknown rune: emit it as Illegal and move on.
l.advance()
+13
View File
@@ -85,6 +85,19 @@ func TestLabelAndComment(t *testing.T) {
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
}
func TestLineCommentTrailingWhitespace(t *testing.T) {
// A trailing run of CR, spaces and tabs is line-ending whitespace, not
// comment content. The token text must not depend on what follows the
// comment: before the trim covered only a CR directly before the token's
// end, "// loop\r " kept the CR while "// loop\r\n" dropped it, and the
// formatter re-lexed its own output to a shorter comment.
eq(t, texts("// loop\r"), []string{"// loop"})
eq(t, texts("// loop\r "), []string{"// loop"})
eq(t, texts("// loop \r\t\nMOVQ AX, BX"), []string{"// loop", "MOVQ", "AX", ",", "BX"})
// A CR inside the comment is content and stays.
eq(t, texts("// loops\rall"), []string{"// loops\rall"})
}
func TestAVX512Mnemonics(t *testing.T) {
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
+6
View File
@@ -21,6 +21,12 @@ import (
// The check requires a parseable signature; functions without one, and
// functions whose parameters are all covered by frame reads, stay silent.
func checkABI0Args(t *ast.Text) []Diagnostic {
// An explicit <ABIInternal> TEXT reads its arguments from the register
// file by declaration (runtime·memmove<ABIInternal> is the canonical
// example), so the ABI0 frame contract does not apply to it.
if t.Name != nil && t.Name.ABI != "" {
return nil
}
params, ok := abiParamNames(t.Doc)
if !ok || len(params) == 0 {
return nil
+17
View File
@@ -82,6 +82,23 @@ func TestABIArgSizeSkipsRegisterABI(t *testing.T) {
}
}
// TestABI0ArgsSkipsABIInternal verifies the frame-read check does not fire for
// a TEXT declared <ABIInternal>: runtime·memmove<ABIInternal> and friends read
// their arguments from the register file by declaration, which is the correct
// spelling there, not the register-args port bug the rule hunts.
func TestABI0ArgsSkipsABIInternal(t *testing.T) {
diags := lintSrc(t, "#include \"textflag.h\"\n"+
"// func memmove(to, from unsafe.Pointer, n uintptr)\n"+
"TEXT ·memmove<ABIInternal>(SB), NOSPLIT, $0-24\n"+
"\tMOVQ AX, DI\n"+
"\tMOVQ BX, SI\n"+
"\tMOVQ CX, BX\n"+
"\tRET\n")
if codes(diags)[CodeABI0RegisterArgs] != 0 {
t.Fatalf("ABIInternal TEXT must not be checked against the FP frame: %+v", diags)
}
}
// TestUnreachableCode exercises the dead-code detection and its guard rails.
func TestUnreachableCode(t *testing.T) {
// Code after a RET is unreachable.
+111 -8
View File
@@ -338,10 +338,8 @@ func lintText(t *ast.Text, tab *arch.Table, archKnown bool, cfg Config, macros m
}
if isJump(cfg.Arch, upper) {
for _, op := range st.Operands {
if name, pos, ok := localLabelRef(op); ok && !tab.IsRegister(name) && !arch.IsPseudoReg(name) {
referenced[name] = pos
}
if name, pos, ok := branchTargetRef(cfg.Arch, upper, st.Operands, tab); ok {
referenced[name] = pos
}
}
}
@@ -649,17 +647,65 @@ func localLabelRef(op *ast.Operand) (string, token.Position, bool) {
return sym.Name, op.Pos, true
}
// branchTargetRef returns the local label a branch transfers control to: the
// bare symbol in the destination position, the last operand, since that is
// where the Plan 9 branch target sits. A register-named target is a
// register-indirect branch (JMP AX, arm64 BR R5, riscv64 JALR X6, loong64
// JIRL R1) and yields no reference, unless the encoder reads the target
// positionally (positionalBranchTarget): there a label may legitimately
// collide with a register alias, riscv64 ZERO being the ABI name of X0, and
// a label named zero is ordinary code.
func branchTargetRef(a arch.Arch, upper string, ops []*ast.Operand, tab *arch.Table) (string, token.Position, bool) {
if len(ops) == 0 {
return "", token.Position{}, false
}
name, pos, ok := localLabelRef(ops[len(ops)-1])
if !ok {
return "", token.Position{}, false
}
if !positionalBranchTarget(a, upper) && (tab.IsRegister(name) || arch.IsPseudoReg(name)) {
return "", token.Position{}, false
}
return name, pos, true
}
// positionalBranchTarget reports whether the encoder reads a bare-symbol
// operand of the branch as its label target from a fixed position, without
// consulting the register file. The riscv64 branch, JMP and JAL encoders do
// (labelFromOperand in asm/riscv_assemble.go), as do the loong64 branch,
// BFPT/BFPF and jump encoders (l64Label in asm/loong64_assemble.go). amd64
// never does, because a bare register operand to JMP/CALL/Jcc is a
// register-indirect branch; nor do the register-indirect forms of the RISC
// families (arm64 BR/BLR, riscv64 JALR/JR, loong64 JIRL).
func positionalBranchTarget(a arch.Arch, upper string) bool {
switch a {
case arch.RISCV:
return riscvBranches[upper] || upper == "JMP" || upper == "JAL"
case arch.LOONG64:
return loong64Branches[upper] || upper == "JMP" || upper == "B" ||
upper == "JAL" || upper == "BL"
}
return false
}
// riscvBranches and loong64Branches are the conditional-branch mnemonics; they
// are listed explicitly rather than matched by a "B" prefix so that bit-manip
// instructions (BCLR, BSET, …) are never mistaken for branches.
// instructions (BCLR, BSET, …) are never mistaken for branches. The sets
// mirror the encoder's own branch cases: the B-type table entries
// (riscv_encode.go), the branch-zero pseudos and the reversed branches
// BGT/BGTU/BLE/BLEU (riscv_assemble.go), and for loong64 the 16-bit branch
// table plus the single-register forms of l64branch21Table (BEQZ/BNEZ and the
// floating-point branches BFPT/BFPF).
var riscvBranches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BEQZ": true, "BNEZ": true, "BLEZ": true, "BGEZ": true, "BLTZ": true, "BGTZ": true,
"BGT": true, "BGTU": true, "BLE": true, "BLEU": true,
}
var loong64Branches = map[string]bool{
"BEQ": true, "BNE": true, "BLT": true, "BGE": true, "BLTU": true, "BGEU": true,
"BLEZ": true, "BLTZ": true, "BGEZ": true, "BGTZ": true,
"BEQZ": true, "BNEZ": true, "BFPT": true, "BFPF": true,
}
// isJump reports whether the mnemonic is any branch.
@@ -676,7 +722,8 @@ func isJump(a arch.Arch, upper string) bool {
upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "CALL" || loong64Branches[upper] ||
upper == "JIRL" || upper == "JMP" || upper == "BR"
upper == "JIRL" || upper == "JMP" || upper == "BR" ||
upper == "B" || upper == "JAL" || upper == "BL"
default: // amd64
return upper == "CALL" || strings.HasPrefix(upper, "J")
}
@@ -692,7 +739,8 @@ func isUnconditionalJump(a arch.Arch, upper string) bool {
return upper == "JMP" || upper == "J" || upper == "JAL" ||
upper == "JALR" || upper == "JR" || upper == "BR"
case arch.LOONG64:
return upper == "JMP" || upper == "JIRL" || upper == "BR"
return upper == "JMP" || upper == "JIRL" || upper == "BR" || upper == "B" ||
upper == "JAL" || upper == "BL"
default:
return upper == "JMP"
}
@@ -792,6 +840,43 @@ func isSPReg(op *ast.Operand, a arch.Arch) bool {
return false
}
// shiftRotateBases are the shift and rotate mnemonics without their width
// suffix. These are the instructions whose encoder path (encodeShift) reads
// the count from the first operand.
var shiftRotateBases = map[string]bool{
"SHL": true, "SHR": true, "SAR": true, "SAL": true,
"ROL": true, "ROR": true, "RCL": true, "RCR": true,
}
// isShiftCountOperand reports whether operand i of mnem is the shift count.
// The ISA fixes the shift/rotate count register at CL: the D2/D3 group (and
// C0/C1 for immediates) encode the count outside the ModRM register field,
// so the count operand is 8-bit by definition no matter how wide the data is.
// The count arrives as the first of the two operands; the one-operand form
// does not exist.
func isShiftCountOperand(mnem string, i, nops int) bool {
if nops != 2 || i != 0 {
return false
}
if shiftRotateBases[mnem] {
return true
}
if len(mnem) > 1 {
switch mnem[len(mnem)-1] {
case 'Q', 'L', 'W', 'B':
return shiftRotateBases[mnem[:len(mnem)-1]]
}
}
return false
}
// isSetcc reports whether the mnemonic is a SETcc: SET plus a condition code.
// The membership test is the encoder's own SET dispatch, which asm.Encodable
// mirrors.
func isSetcc(mnem string) bool {
return strings.HasPrefix(mnem, "SET") && asm.Encodable(mnem)
}
// checkRegisterWidth detects amd64 register-width mismatches. The naming
// truth of the Go assembler governs: AX, BX, CX, DX, SI, DI, BP, SP and
// R8-R15 ARE the 64-bit register names (there are no separate EAX/RAX
@@ -802,6 +887,14 @@ func isSPReg(op *ast.Operand, a arch.Arch) bool {
// register (EAX under the gasm alias extension, or a byte form), and byte
// registers in L/W operations.
func checkRegisterWidth(mnem string, ops []*ast.Operand) string {
// A SETcc stores one byte: the destination is an 8-bit register or an
// 8-bit memory location by definition (0F 90+cc), whichever condition it
// tests. The trailing letter of spellings like SETPL or SETEQ is part of
// the condition code, not an operand width, so the whole family is
// exempt from the suffix logic.
if isSetcc(mnem) {
return ""
}
// Determine expected width from mnemonic suffix.
var expected int // 0=unknown, 8/4/2/1=bytes
switch {
@@ -816,10 +909,20 @@ func checkRegisterWidth(mnem string, ops []*ast.Operand) string {
default:
return "" // no suffix, can't determine width
}
for _, op := range ops {
for i, op := range ops {
if op.Kind != ast.OpAddr || op.Addr.Sym == nil {
continue
}
// Only a bare register carries a width to compare: frame and static
// symbol references (ch+8(FP), foo(SB)) and memory operands are not
// registers even when their name collides with one.
if op.Addr.Sym.Pseudo != "" || op.Addr.Base != "" || op.Addr.Index != "" {
continue
}
// The shift/rotate count is exempt: fixed at 8 bits by the ISA.
if isShiftCountOperand(mnem, i, len(ops)) {
continue
}
name := strings.ToLower(op.Addr.Sym.Name)
regWidth := amd64RegWidth(name)
if regWidth == 0 {
+214
View File
@@ -8,6 +8,7 @@ import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
@@ -21,6 +22,21 @@ func lintSrc(t *testing.T, src string) []Diagnostic {
return File(f, Config{Arch: arch.AMD64})
}
// lintArchFile parses and lints src under a, then hands the same file to
// assemble so the assertion is pinned against the encoder: a kernel the
// linter reasons about must also be one the encoder accepts.
func lintArchFile(t *testing.T, filename, src string, a arch.Arch, assemble func(*ast.File) (*asm.Image, error)) []Diagnostic {
t.Helper()
f, errs := parser.Parse(filename, src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := assemble(f); err != nil {
t.Fatalf("encoder rejects the kernel: %v", err)
}
return File(f, Config{Arch: a})
}
// lintSrcArch lints src under the architecture inferred from filename.
func lintSrcArch(t *testing.T, filename, src string) []Diagnostic {
t.Helper()
@@ -262,6 +278,138 @@ loop:
}
}
func TestRiscvBranchFamilyRegistersLabels(t *testing.T) {
// Every riscv64 pseudo-branch that references a label must register that
// reference: the reversed branches BGT/BGTU/BLE/BLEU (GOROOT's
// memmove_riscv64 branches with BGTU) and a label named like the ZERO
// register alias (GOROOT's memclr_riscv64 carries a label named zero;
// ZERO is the ABI name of X0) must not be reported unused.
diags := lintSrcArch(t, "f_riscv64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
BGTU X10, X11, backward
BGT X10, X11, zero
BLE X10, X11, one
BLEU X10, X11, two
BEQZ X10, zero
BNEZ X10, one
JMP two
backward:
RET
zero:
RET
one:
RET
two:
RET
`)
if codes(diags)[CodeUnusedLabel] != 0 {
t.Fatalf("branch-referenced labels must not be flagged unused: %+v", diags)
}
if codes(diags)[CodeUndefinedLabel] != 0 {
t.Fatalf("defined labels must resolve: %+v", diags)
}
// A branch to a truly undefined label still reports.
diags = lintSrcArch(t, "f_riscv64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
BGT X10, X11, nowhere
RET
`)
if codes(diags)[CodeUndefinedLabel] != 1 {
t.Fatalf("undefined branch target must be flagged: %+v", diags)
}
// A register-indirect JALR is not a label reference.
diags = lintSrcArch(t, "f_riscv64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
JALR X1
RET
`)
if codes(diags)[CodeUndefinedLabel] != 0 {
t.Fatalf("register operand of JALR is not a label: %+v", diags)
}
}
func TestLoong64BranchFamilyRegistersLabels(t *testing.T) {
// The loong64 jumps and single-register branches (JAL, B, BL, BEQZ/BNEZ,
// BFPT/BFPF) all reference their label from the last operand; GOROOT's
// own basic kernels tail-call with JAL, so the reference must register.
diags := lintSrcArch(t, "f_loong64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
BEQZ R4, fin
BNEZ R4, fin
BLTZ R4, fin
JAL fin
BL fin
B fin
RET
fin:
RET
`)
if codes(diags)[CodeUnusedLabel] != 0 {
t.Fatalf("branch-referenced labels must not be flagged unused: %+v", diags)
}
if codes(diags)[CodeUndefinedLabel] != 0 {
t.Fatalf("defined labels must resolve: %+v", diags)
}
}
// TestBranchFamiliesAssemble pins the lint branch sets to the encoder: every
// mnemonic the linter classifies as a riscv64 or loong64 label branch must be
// a branch the encoder actually assembles, with the label in the last
// operand. If the encoder gains or renames a branch, this test fails and the
// set follows it.
func TestBranchFamiliesAssemble(t *testing.T) {
riscvForms := map[string]string{}
for m := range riscvBranches {
riscvForms[m] = m + " X10, X11, tgt"
}
for _, m := range []string{"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ"} {
riscvForms[m] = m + " X10, tgt"
}
riscvForms["JMP"] = "JMP tgt"
riscvForms["JAL"] = "JAL tgt"
loongForms := map[string]string{}
for _, m := range []string{"BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU"} {
loongForms[m] = m + " R4, R5, tgt"
}
for _, m := range []string{"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ", "BFPT", "BFPF"} {
loongForms[m] = m + " R4, tgt"
}
loongForms["JMP"] = "JMP tgt"
loongForms["B"] = "B tgt"
loongForms["JAL"] = "JAL tgt"
loongForms["BL"] = "BL tgt"
for m, form := range riscvForms {
src := "#include \"textflag.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\t" + form + "\n" +
"tgt:\n" +
"\tRET\n"
diags := lintArchFile(t, "f_riscv64.s", src, arch.RISCV, asm.AssembleFileRISCV)
if codes(diags)[CodeUnusedLabel] != 0 || codes(diags)[CodeUndefinedLabel] != 0 {
t.Errorf("riscv64 %s: label reference not registered: %+v", m, diags)
}
}
for m, form := range loongForms {
src := "#include \"textflag.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\t" + form + "\n" +
"tgt:\n" +
"\tRET\n"
diags := lintArchFile(t, "f_loong64.s", src, arch.LOONG64, asm.AssembleFileLOONG64)
if codes(diags)[CodeUnusedLabel] != 0 || codes(diags)[CodeUndefinedLabel] != 0 {
t.Errorf("loong64 %s: label reference not registered: %+v", m, diags)
}
}
}
func TestInvalidTextflag(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
@@ -413,6 +561,72 @@ TEXT ·f(SB), NOSPLIT, $0
}
}
func TestRegisterWidthShiftCount(t *testing.T) {
// The shift and rotate count lives in CL by ISA definition (the D2/D3
// group encodes the count outside the ModRM register field), so the count
// operand is 8-bit no matter how wide the data is: SHLQ CL, AX is the
// normal spelling of a 64-bit shift. The data operand keeps its check.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
SHLQ CL, AX
SHRL CL, BX
SARQ CL, CX
ROLL CL, DX
RORQ CL, R8
RCLL CL, R9
RCRQ CL, R10
MOVQ CL, R10
RET
`)
if codes(diags)[CodeRegisterWidthMismatch] != 1 {
t.Fatalf("only the MOVQ CL data move must be flagged, got %+v", diags)
}
}
func TestRegisterWidthSetcc(t *testing.T) {
// A SETcc stores one byte whichever condition it tests (0F 90+cc), so
// SETNE AL is always right and the trailing letters of SETEQ, SETPL and
// SETLS are condition codes, not width suffixes.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
CMPQ AX, BX
SETNE AL
SETEQ AL
SETPL AL
SETLS AL
SETCC (BX)
SETGE (R8)
RET
`)
if codes(diags)[CodeRegisterWidthMismatch] != 0 {
t.Fatalf("SETcc destinations are 8-bit by definition: %+v", diags)
}
if codes(diags)[CodeUnknownInstr] != 0 {
t.Fatalf("every SETcc spelling must be known: %+v", diags)
}
}
func TestRegisterWidthFrameNames(t *testing.T) {
// GOROOT's BSD syscall stubs carry frame parameters whose names collide
// with byte register names (kevent's ch and nch): MOVQ ch+8(FP), SI is a
// frame reference, not the CH register.
diags := lintSrc(t, `
#include "textflag.h"
TEXT ·kevent(SB), NOSPLIT, $0-36
MOVL kq+0(FP), DI
MOVQ ch+8(FP), SI
MOVL nch+16(FP), DX
MOVQ ev+24(FP), R10
MOVQ AX, ret+32(FP)
RET
`)
if codes(diags)[CodeRegisterWidthMismatch] != 0 {
t.Fatalf("frame and static symbol names are not registers: %+v", diags)
}
}
func TestNonportableRegisterName(t *testing.T) {
diags := lintSrc(t, `
#include "textflag.h"
+146
View File
@@ -0,0 +1,146 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Constant-expression folding for operands. The toolchain's assembler
// evaluates arithmetic in every operand position, and macro-heavy GOROOT
// sources lean on it: parameterised bodies carry offsets like
// ((index*4)+0)(base), immediates like $(32-shift) and masks like
// $~63 or $(1<<0|1<<9). Substituting the parameters textually therefore
// leaves constant arithmetic behind, and the parser folds it here, keeping
// the operand AST identical to what the same literals written out would
// produce. Anything that is not a closed integer expression fails to fold
// and falls through to the ordinary operand paths.
package parser
import (
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// foldExpr evaluates the constant integer expression at the head of ts and
// returns its value together with the unconsumed tokens. ok is false when
// the tokens do not form an expression, which is the callers' signal to use
// the ordinary parsing paths.
func foldExpr(ts []token.Token) (val int64, rest []token.Token, ok bool) {
v, rest, ok := foldAdd(ts)
if !ok {
return 0, ts, false
}
return v, rest, true
}
// foldAdd parses addition-level expressions: +, - and | bind loosest, the
// Plan 9 convention that makes x<<1|3 read as (x<<1)|3.
func foldAdd(ts []token.Token) (int64, []token.Token, bool) {
v, rest, ok := foldMul(ts)
if !ok {
return 0, ts, false
}
for len(rest) > 0 {
kind := rest[0].Kind
if kind != token.Plus && kind != token.Minus && kind != token.Pipe {
return v, rest, true
}
w, r2, ok := foldMul(rest[1:])
if !ok {
return v, rest, true
}
switch kind {
case token.Plus:
v += w
case token.Minus:
v -= w
case token.Pipe:
v |= w
}
rest = r2
}
return v, rest, true
}
// foldMul parses multiplication-level expressions: *, / and the bit
// operators &, << and >>.
func foldMul(ts []token.Token) (int64, []token.Token, bool) {
v, rest, ok := foldFactor(ts)
if !ok {
return 0, ts, false
}
for len(rest) > 0 {
switch rest[0].Kind {
case token.Star:
w, r2, ok := foldFactor(rest[1:])
if !ok {
return v, rest, true
}
v *= w
rest = r2
case token.Slash:
w, r2, ok := foldFactor(rest[1:])
if !ok || w == 0 {
return v, rest, true
}
v /= w
rest = r2
case token.Ampersand:
w, r2, ok := foldFactor(rest[1:])
if !ok {
return v, rest, true
}
v &= w
rest = r2
case token.LShift:
w, r2, ok := foldFactor(rest[1:])
if !ok || w < 0 || w >= 64 {
return v, rest, true
}
v <<= uint(w)
rest = r2
case token.RShift:
w, r2, ok := foldFactor(rest[1:])
if !ok || w < 0 || w >= 64 {
return v, rest, true
}
v >>= uint(w)
rest = r2
default:
return v, rest, true
}
}
return v, rest, true
}
// foldFactor parses a number, a parenthesised expression, or a unary sign
// or complement.
func foldFactor(ts []token.Token) (int64, []token.Token, bool) {
if len(ts) == 0 {
return 0, ts, false
}
switch ts[0].Kind {
case token.Number:
v, ok := tryInt(ts[0].Text)
if !ok {
return 0, ts, false
}
return v, ts[1:], true
case token.LParen:
v, rest, ok := foldAdd(ts[1:])
if !ok || len(rest) == 0 || rest[0].Kind != token.RParen {
return 0, ts, false
}
return v, rest[1:], true
case token.Minus:
v, rest, ok := foldFactor(ts[1:])
if !ok {
return 0, ts, false
}
return -v, rest, true
case token.Plus:
return foldFactor(ts[1:])
case token.Tilde:
v, rest, ok := foldFactor(ts[1:])
if !ok {
return 0, ts, false
}
return ^v, rest, true
}
return 0, ts, false
}
+23
View File
@@ -408,6 +408,18 @@ func parseImmediate(g []token.Token) ast.Immediate {
return imm
}
}
// A constant expression introduced by '(' or '~'. Textual macro
// substitution leaves arithmetic such as $(32-shift) and $~63 behind,
// and the toolchain evaluates it in place; only shapes the ordinary
// paths below cannot read reach the folder, so every existing form
// keeps its exact parse.
if g[0].Kind == token.LParen || g[0].Kind == token.Tilde {
if v, rest, ok := foldExpr(g); ok && len(rest) == 0 {
imm.Val = v
imm.HasVal = true
return imm
}
}
i := 0
if g[i].Kind == token.Minus {
imm.Neg = true
@@ -459,6 +471,17 @@ func parseAddress(g []token.Token) ast.Address {
}
i := 0
// A parenthesised constant expression as the displacement: substituted
// macro bodies carry ((index*4)+0)(base) shapes. As with the signed
// number path below, the value is committed only when a base group
// follows.
if i < len(g) && g[i].Kind == token.LParen {
if v, rest, ok := foldExpr(g[i:]); ok && len(rest) > 0 && rest[0].Kind == token.LParen {
addr.Offset = v
addr.HasOff = true
i = len(g) - len(rest)
}
}
// Optional leading displacement before a '(' base group. A sign pushes
// the parenthesis one token further out: -4(DX) has it at i+2.
if isSignedNumber(g, i) {
+475
View File
@@ -0,0 +1,475 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// The preprocessor turns #define and #include directives into the token
// stream the parser really sees, the way the Go toolchain's assembler does:
// object and parameterised macros expand at the point of use, and an
// #include splices the named file's lines in place of the directive. The
// pass runs only on the assembly path (gasm asm, diff, the corpus audit),
// where the result is machine code; parsing for the linter, formatter and
// language server keeps the raw file so their view of #define lines, and
// therefore their macro-aware behaviour, is unchanged.
package parser
import (
"fmt"
"os"
"path/filepath"
"slices"
"strconv"
"strings"
"unicode/utf8"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Options controls the optional preprocessing applied before a file is
// parsed. The zero value reproduces Parse exactly.
type Options struct {
// IncludeDirs lists the -I directories searched for #include files,
// in order, after the including file's own directory.
IncludeDirs []string
// Expand enables macro expansion, include splicing and the
// statement-separator reading of ';' that the expanded bodies rely on.
Expand bool
}
// ParseWithOptions parses src like Parse, optionally preprocessing it first.
// The returned file is usable even when errors is non-empty.
func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) {
tokens := lexer.Tokenize(src)
var lines [][]token.Token
var errs []error
if opts.Expand {
pp := &preproc{opts: opts, macros: map[string]*macroDef{}}
lines = pp.fileLines(path, tokens, token.Position{})
errs = pp.errs
} else {
lines = splitLines(tokens)
}
p := &state{path: path}
p.parse(lines)
return p.file, append(errs, p.errs...)
}
// maxExpansionDepth bounds recursive macro expansion; the toolchain's
// assembler gives up after 100 nested invocations without producing a token.
const maxExpansionDepth = 100
// textflagHeader names the one header gasm does not splice: its flag macros
// (NOSPLIT, RODATA, …) are consumed by name throughout gasm's parser,
// encoders and linter, and expanding them to their numeric constants would
// leave every consumer blind to them.
const textflagHeader = "textflag.h"
// macroDef is one #define. A nil args slice is an object macro; a non-nil
// (possibly empty) one is parameterised, the C distinction between
// "#define A(x)" and "#define A (x)".
type macroDef struct {
name string
args []string
body []token.Token
}
// preproc carries the state of one expansion pass: the live macro table, the
// chain of files currently being read, for cycle detection, and the
// conditional-inclusion stack of #ifdef regions.
type preproc struct {
opts Options
macros map[string]*macroDef
errs []error
stack []string // absolute paths of files being read, innermost last
ifdefStack []bool // one entry per open #ifdef/#ifndef, its truth
}
// enabled reports whether the position being read is inside a live
// conditional branch. Directives inside a disabled branch contribute
// nothing, and its content lines are dropped, exactly as the toolchain's
// input stack does.
func (pp *preproc) enabled() bool {
return len(pp.ifdefStack) == 0 || pp.ifdefStack[len(pp.ifdefStack)-1]
}
func (pp *preproc) errorf(pos token.Position, format string, args ...any) {
pp.errs = append(pp.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)})
}
// fileLines tokenizes and preprocesses one file into logical lines.
// Directive lines are kept (the parser records them for the tooling);
// #include lines are replaced by the included file's lines. includePos is
// the position of the #include that pulled this file in, zero for the
// top-level file, and only serves cycle diagnostics.
func (pp *preproc) fileLines(path string, tokens []token.Token, includePos token.Position) [][]token.Token {
abs, err := filepath.Abs(path)
if err != nil {
abs = filepath.Clean(path)
}
if slices.Contains(pp.stack, abs) {
if includePos.IsValid() {
pp.errorf(includePos, "#include %q: include cycle (%s is already being read)", path, filepath.Base(path))
}
return nil
}
pp.stack = append(pp.stack, abs)
var out [][]token.Token
for _, line := range splitLines(tokens) {
if len(line) == 0 {
out = append(out, line)
continue
}
if line[0].Kind == token.Hash {
out = append(out, pp.directive(line, filepath.Dir(path))...)
continue
}
if !pp.enabled() {
continue
}
out = append(out, splitOnSemicolons(pp.expandTokens(line))...)
}
pp.stack = pp.stack[:len(pp.stack)-1]
if len(pp.stack) == 0 && len(pp.ifdefStack) > 0 {
// The stack is per-input, shared across includes, so only the
// top-level file's end can decide the input was left unclosed.
pp.errorf(token.Position{Line: 1, Column: 1}, "unclosed #ifdef or #ifndef")
}
return out
}
// directive processes one '#' line and returns the lines to keep in the
// stream: every directive line is kept as-is for the parser (which records
// it), except #include, which is replaced by the spliced content.
// Conditionals are tracked on every line; every other directive is inert
// inside a disabled branch.
func (pp *preproc) directive(line []token.Token, dir string) [][]token.Token {
if len(line) < 2 || line[1].Kind != token.Ident {
return [][]token.Token{line}
}
switch line[1].Text {
case "ifdef", "ifndef":
pp.ifdef(line, line[1].Text == "ifndef")
case "else":
pp.elseBranch(line)
case "endif":
pp.endif(line)
case "define":
if pp.enabled() {
pp.define(line)
}
case "undef":
if pp.enabled() {
pp.undef(line)
}
case "include":
if pp.enabled() {
return pp.include(line, dir)
}
default:
// #line and unknown directives are recorded but not interpreted:
// conservative support keeps the parser's view intact and files
// using them fail on their content, not silently.
}
return [][]token.Token{line}
}
// ifdef handles "#ifdef NAME" and "#ifndef NAME", pushing the branch's truth
// onto the conditional stack. A branch opened inside a disabled region is
// itself disabled, however the name resolves.
func (pp *preproc) ifdef(line []token.Token, inverted bool) {
truth := false
if len(line) >= 3 && line[2].Kind == token.Ident {
_, defined := pp.macros[line[2].Text]
truth = defined != inverted
} else {
pp.errorf(line[0].Pos, "expected identifier after #%s", line[1].Text)
}
if !pp.enabled() {
truth = false
}
pp.ifdefStack = append(pp.ifdefStack, truth)
}
// elseBranch flips the innermost conditional's truth, but only when the
// region enclosing it is itself live: the toolchain keeps outer overrides.
func (pp *preproc) elseBranch(line []token.Token) {
if len(pp.ifdefStack) == 0 {
pp.errorf(line[0].Pos, "unmatched #else")
return
}
if len(pp.ifdefStack) == 1 || pp.ifdefStack[len(pp.ifdefStack)-2] {
pp.ifdefStack[len(pp.ifdefStack)-1] = !pp.ifdefStack[len(pp.ifdefStack)-1]
}
}
// endif closes the innermost conditional.
func (pp *preproc) endif(line []token.Token) {
if len(pp.ifdefStack) == 0 {
pp.errorf(line[0].Pos, "unmatched #endif")
return
}
pp.ifdefStack = pp.ifdefStack[:len(pp.ifdefStack)-1]
}
// define parses "#define NAME[(formals)] body" into the macro table. The
// body runs to the end of the logical line (the lexer has already spliced
// backslash continuations) and stops at a comment, which never expands.
func (pp *preproc) define(line []token.Token) {
if len(line) < 3 || line[2].Kind != token.Ident {
return
}
name := line[2]
args := []string(nil)
body := line[3:]
// The definition is parameterised only when '(' follows the name
// directly; the toolchain separates "#define A(x)" from
// "#define A (x)" by adjacency, and so does the column check here.
if len(body) > 0 && body[0].Kind == token.LParen &&
body[0].Pos.Column == name.Pos.Column+utf8.RuneCountInString(name.Text) {
args = []string{}
i := 1
for i < len(body) && body[i].Kind != token.RParen {
if body[i].Kind == token.Ident {
args = append(args, body[i].Text)
}
i++
}
if i < len(body) {
body = body[i+1:]
} else {
body = nil
}
}
if i := slices.IndexFunc(body, func(t token.Token) bool { return t.Kind == token.Comment }); i >= 0 {
body = body[:i]
}
if _, exists := pp.macros[name.Text]; exists {
// The toolchain refuses redefinition, so a file the oracle accepts
// never redefines; failing here keeps that contract visible.
pp.errorf(name.Pos, "redefinition of macro %s", name.Text)
}
pp.macros[name.Text] = &macroDef{name: name.Text, args: args, body: pp.bodyWithBreaks(body)}
}
// bodyWithBreaks records the statement boundaries the continuations carry.
// The lexer splices backslash-continued lines into one logical line, but the
// toolchain keeps the newline as a token in the stored body, which is how a
// multi-instruction body without semicolons (the arm64 style) still splits
// into statements on expansion. A line change inside the logical line is
// exactly a continuation, so the boundary is restored from the positions.
func (pp *preproc) bodyWithBreaks(body []token.Token) []token.Token {
out := make([]token.Token, 0, len(body))
for i, t := range body {
if i > 0 && t.Pos.Line != body[i-1].Pos.Line {
out = append(out, token.Token{Kind: token.Newline, Text: "\n", Pos: t.Pos, End: t.Pos})
}
out = append(out, t)
}
return out
}
// undef handles "#undef NAME", which the toolchain honours and requires to
// name a defined macro.
func (pp *preproc) undef(line []token.Token) {
if len(line) < 3 || line[2].Kind != token.Ident {
return
}
if _, ok := pp.macros[line[2].Text]; !ok {
pp.errorf(line[2].Pos, "#undef for undefined macro %s", line[2].Text)
return
}
delete(pp.macros, line[2].Text)
}
// include resolves and splices "#include \"file\"". A header that cannot be
// read keeps the directive line in the stream, with a diagnostic.
func (pp *preproc) include(line []token.Token, dir string) [][]token.Token {
if len(line) < 3 || line[2].Kind != token.String {
return [][]token.Token{line}
}
header := line[2]
name, err := strconv.Unquote(header.Text)
if err != nil {
pp.errorf(header.Pos, "unquoting include file name: %v", err)
return [][]token.Token{line}
}
if filepath.Base(name) == textflagHeader {
// Flag macros are handled natively (see textflagHeader); the
// directive stays so tools still see the include.
return [][]token.Token{line}
}
resolved, ok := pp.resolve(name, dir)
if !ok {
searched := append([]string{dir}, pp.opts.IncludeDirs...)
pp.errorf(header.Pos, "#include %q: file not found (searched %s)", name, strings.Join(searched, ", "))
return [][]token.Token{line}
}
src, err := os.ReadFile(resolved)
if err != nil {
pp.errorf(header.Pos, "#include %q: %v", name, err)
return [][]token.Token{line}
}
return pp.fileLines(resolved, lexer.Tokenize(string(src)), header.Pos)
}
// resolve looks an include name up the way the toolchain does: as written
// (relative to the working directory), then relative to the including
// file's directory, then in each -I directory in order.
func (pp *preproc) resolve(name, dir string) (string, bool) {
candidates := []string{name}
if !filepath.IsAbs(name) {
candidates = append(candidates, filepath.Join(dir, name))
for _, d := range pp.opts.IncludeDirs {
candidates = append(candidates, filepath.Join(d, name))
}
}
for _, c := range candidates {
if st, err := os.Stat(c); err == nil && !st.IsDir() {
return c, true
}
}
return "", false
}
// expandTokens expands every macro invocation in a token sequence,
// recursively, with a depth guard. A body is spliced into the sequence in
// place and rescanned, the way the toolchain's input stack re-reads pushed
// tokens: an object macro may name a parameterised one, and the argument
// list of the expansion may then come from the tokens that follow.
func (pp *preproc) expandTokens(in []token.Token) []token.Token {
s := in
i := 0
consecutive := 0
for i < len(s) {
t := s[i]
if t.Kind != token.Ident {
i++
consecutive = 0
continue
}
def := pp.macros[t.Text]
if def == nil {
i++
consecutive = 0
continue
}
// The guard mirrors the toolchain's: 100 nested invocations in a
// row without a plain token between them means recursion.
consecutive++
if consecutive > maxExpansionDepth {
pp.errorf(t.Pos, "recursive macro invocation (deeper than %d levels)", maxExpansionDepth)
return nil
}
if def.args == nil {
s = append(s[:i], append(restamp(def.body, t.Pos), s[i+1:]...)...)
continue
}
// A parameterised macro invoked without its parentheses stands
// unexpanded, naming itself, as in the toolchain.
if i+1 >= len(s) || s[i+1].Kind != token.LParen {
i++
consecutive = 0
continue
}
args, next := pp.collectArgs(s, i+1, t)
if args == nil {
return nil
}
// A zero-argument macro may be invoked as NAME().
if len(def.args) == 0 && len(args) == 1 && len(args[0]) == 0 {
args = nil
}
if len(args) != len(def.args) {
pp.errorf(t.Pos, "wrong arg count for macro %s: got %d, want %d", t.Text, len(args), len(def.args))
i = next
consecutive = 0
continue
}
sub := make([]token.Token, 0, len(def.body))
for _, bt := range def.body {
if bt.Kind == token.Ident {
if k := slices.Index(def.args, bt.Text); k >= 0 {
sub = append(sub, restamp(args[k], t.Pos)...)
continue
}
}
sub = append(sub, bt)
}
s = append(s[:i], append(sub, s[next:]...)...)
}
return s
}
// collectArgs reads the actual argument tokens of an invocation; the opening
// parenthesis is at start. Commas separate arguments except inside nested
// parentheses. A nil result means the list was unterminated, which is a
// diagnostic.
func (pp *preproc) collectArgs(in []token.Token, start int, name token.Token) ([][]token.Token, int) {
var args [][]token.Token
var cur []token.Token
nesting := 0
for i := start + 1; i < len(in); i++ {
t := in[i]
switch t.Kind {
case token.LParen:
nesting++
cur = append(cur, t)
case token.RParen:
if nesting == 0 {
return append(args, cur), i + 1
}
nesting--
cur = append(cur, t)
case token.Comma:
if nesting == 0 {
args = append(args, cur)
cur = nil
continue
}
cur = append(cur, t)
case token.Comment:
pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text)
return nil, i
default:
cur = append(cur, t)
}
}
pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text)
return nil, len(in)
}
// restamp copies body tokens to the invocation's position, so diagnostics
// and the line table point where the macro was used, as the toolchain's
// input stack does.
func restamp(body []token.Token, pos token.Position) []token.Token {
out := make([]token.Token, len(body))
for i, t := range body {
t.Pos, t.End = pos, pos
out[i] = t
}
return out
}
// splitOnSemicolons breaks a token sequence at ';' statement separators and
// at the Newline markers that record continuation boundaries inside macro
// bodies, producing the logical lines the parser expects. The separators
// carry no meaning beyond the break, so the pieces are exactly what the same
// statements on separate lines would produce.
func splitOnSemicolons(ts []token.Token) [][]token.Token {
var out [][]token.Token
start := 0
for i, t := range ts {
if t.Kind == token.Semicolon || t.Kind == token.Newline {
if i > start {
out = append(out, ts[start:i])
}
start = i + 1
}
}
if start < len(ts) {
out = append(out, ts[start:])
}
return out
}
+552
View File
@@ -0,0 +1,552 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package parser
import (
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// expand parses src with preprocessing enabled and returns the first TEXT's
// body instructions as "MNEMONIC operand|operand" strings, the shape the
// expansion assertions below compare against. Runs of spaces are
// collapsed: Raw renders a token group as its tokens joined with single
// spaces, so "$(32-7)" arrives as "$ ( 32 - 7 )" and the comparison must
// not depend on that spelling.
func expand(t *testing.T, src string) (*ast.File, []string) {
t.Helper()
f, errs := ParseWithOptions("t_amd64.s", src, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
ts := texts(f)
if len(ts) == 0 {
t.Fatalf("no TEXT in:\n%s", src)
}
var got []string
for _, s := range ts[0].Body {
in, ok := s.(*ast.Instr)
if !ok {
continue
}
var ops []string
for _, op := range in.Operands {
ops = append(ops, op.Raw)
}
line := in.Mnemonic.Text + " " + strings.Join(ops, ", ")
got = append(got, strings.ReplaceAll(line, " ", ""))
}
return f, got
}
func wantLines(t *testing.T, got []string, want ...string) {
t.Helper()
strip := func(lines []string) string {
var out []string
for _, l := range lines {
out = append(out, strings.ReplaceAll(l, " ", ""))
}
return strings.Join(out, "\n")
}
if strip(got) != strip(want) {
t.Errorf("expanded body:\n %s\nwant:\n %s", strings.Join(got, "\n "), strings.Join(want, "\n "))
}
}
func TestObjectMacroExpandsAtUse(t *testing.T) {
_, got := expand(t, `
#define REGTMP CX
#define TWICE ADDQ CX, AX; ADDQ CX, AX
TEXT ·f(SB), NOSPLIT, $0
MOVQ 8(SP), REGTMP
TWICE
RET
`)
wantLines(t, got,
"MOVQ 8(SP), CX",
"ADDQ CX, AX",
"ADDQ CX, AX",
"RET",
)
}
func TestParameterisedMacroSubstitutesArguments(t *testing.T) {
f, errs := ParseWithOptions("t_amd64.s", `
#define ROUND1(a, index, const, shift) \
ADDQ $const, a; \
MOVW (index*4)(SP), a; \
RORQ $(32-shift), a
TEXT ·f(SB), NOSPLIT, $0
ROUND1(AX, 3, 0xd76aa478, 7)
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
body := texts(f)[0].Body
add := body[0].(*ast.Instr)
if add.Mnemonic.Text != "ADDQ" || !add.Operands[0].Imm.HasVal ||
add.Operands[0].Imm.Val != 0xd76aa478 || add.Operands[1].Addr.Sym == nil ||
add.Operands[1].Addr.Sym.Name != "AX" {
t.Errorf("ADDQ operands substituted wrong: %+v %+v", add.Operands[0].Imm, add.Operands[1].Addr)
}
mov := body[1].(*ast.Instr)
if addr := mov.Operands[0].Addr; !addr.HasOff || addr.Offset != 12 {
t.Errorf("MOVW offset = %+v, want 12 from 3*4", addr)
}
ror := body[2].(*ast.Instr)
if !ror.Operands[0].Imm.HasVal || ror.Operands[0].Imm.Val != 25 {
t.Errorf("RORQ immediate = %+v, want 25 from (32-7)", ror.Operands[0].Imm)
}
}
func TestMacroArgumentsKeepCommasInParens(t *testing.T) {
// An argument may itself be an unparenthesised expression: the tokens
// substitute verbatim and the parser folds the result, as the
// toolchain's parser does.
f, errs := ParseWithOptions("t_amd64.s", `
#define LOAD(dst, off) MOVQ off(SP), dst
TEXT ·f(SB), NOSPLIT, $0
LOAD(AX, 1*8)
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
addr := in.Operands[0].Addr
if !addr.HasOff || addr.Offset != 8 {
t.Errorf("offset = %+v, want 8", addr)
}
if sym := in.Operands[1].Addr.Sym; sym == nil || sym.Name != "AX" {
t.Errorf("destination = %+v, want AX", in.Operands[1].Addr)
}
}
func TestNestedMacroInvocations(t *testing.T) {
// An object macro naming a parameterised one, and a parameterised body
// invoking another parameterised macro: the toolchain's input stack
// rescans substituted tokens, and so does expansion here.
_, got := expand(t, `
#define DOUBLE(x) ADDQ x, x
#define TWICE2 DOUBLE
#define FOUR(a, b) DOUBLE(a); DOUBLE(b)
TEXT ·f(SB), NOSPLIT, $0
TWICE2(AX)
FOUR(AX, CX)
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"ADDQ AX, AX",
"ADDQ CX, CX",
"RET",
)
}
func TestMultiLineBodySplitsWithoutSemicolons(t *testing.T) {
// The arm64 style: backslash-continued lines with no semicolons. The
// continuation newline is a statement boundary, as in the toolchain.
_, got := expand(t, `
#define PAIR \
ADDQ AX, AX \
MOVQ AX, CX
TEXT ·f(SB), NOSPLIT, $0
PAIR
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"MOVQ AX, CX",
"RET",
)
}
func TestZeroArgumentMacro(t *testing.T) {
_, got := expand(t, `
#define BARRIER()
TEXT ·f(SB), NOSPLIT, $0
BARRIER()
RET
`)
wantLines(t, got, "RET")
}
func TestParameterisedWithoutParensStandsAsName(t *testing.T) {
// A parameterised macro invoked without its parentheses names itself,
// which the parser then reports as an unknown instruction rather than
// silently expanding nothing.
f, errs := ParseWithOptions("t_amd64.s", `
#define M(x) ADDQ x, x
TEXT ·f(SB), NOSPLIT, $0
M
RET
`, Options{Expand: true})
if len(errs) != 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
if len(fn.Body) == 0 {
t.Fatal("body empty")
}
in, ok := fn.Body[0].(*ast.Instr)
if !ok || in.Mnemonic.Text != "M" {
t.Fatalf("bare parameterised macro did not stand as its name: %+v", fn.Body[0])
}
}
func TestDefinitionScoping(t *testing.T) {
// A definition applies from its point onward: the use before the
// #define stays untouched.
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
SPECIAL
#define SPECIAL ADDQ AX, AX
SPECIAL
RET
`)
wantLines(t, got,
"SPECIAL",
"ADDQ AX, AX",
"RET",
)
}
func TestUndefRemovesMacro(t *testing.T) {
_, got := expand(t, `
#define TEMP AX
TEXT ·f(SB), NOSPLIT, $0
TEMP
#undef TEMP
TEMP
RET
`)
wantLines(t, got,
"AX",
"TEMP",
"RET",
)
}
func TestUndefUndefinedMacroIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#undef NOSUCH\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "undefined macro NOSUCH") {
t.Fatalf("#undef of an undefined macro: got %v, want an error naming it", errs)
}
}
func TestRedefinitionIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define A X\n#define A Y\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "redefinition of macro A") {
t.Fatalf("redefinition: got %v, want an error", errs)
}
}
func TestRecursiveMacroIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define A B\n#define B A\nTEXT ·f(SB), NOSPLIT, $0\n\tA\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "recursive macro invocation") {
t.Fatalf("recursion: got %v, want a recursive-macro error, not a hang", errs)
}
}
func TestWrongArgumentCountIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#define M(a, b) ADDQ a, b\nTEXT ·f(SB), NOSPLIT, $0\n\tM(AX)\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "wrong arg count for macro M") {
t.Fatalf("arg count: got %v, want an error", errs)
}
}
func TestConditionalsSelectOneBranch(t *testing.T) {
_, got := expand(t, `
#define MODE2
TEXT ·f(SB), NOSPLIT, $0
#ifdef MODE2
ADDQ AX, AX
#else
SUBQ AX, AX
#endif
#ifndef MODE2
SUBQ CX, CX
#else
ADDQ CX, CX
#endif
RET
`)
wantLines(t, got,
"ADDQ AX, AX",
"ADDQ CX, CX",
"RET",
)
}
func TestConditionalsHideDefinitionsAndIncludes(t *testing.T) {
// A definition inside a disabled branch must not exist, and an
// unresolvable include there must not be followed.
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
#ifdef NOTDEFINED
#define HIDEN ADDQ AX, AX
#include "nowhere.h"
#endif
HIDEN
RET
`)
wantLines(t, got, "HIDEN", "RET")
}
func TestUnclosedConditionalIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#ifdef X\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unclosed #ifdef") {
t.Fatalf("unclosed conditional: got %v, want an error", errs)
}
}
func TestUnmatchedConditionalDelimitersAreErrors(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#endif\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unmatched #endif") {
t.Fatalf("unmatched #endif: got %v, want an error", errs)
}
_, errs = ParseWithOptions("t_amd64.s", "#else\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n", Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "unmatched #else") {
t.Fatalf("unmatched #else: got %v, want an error", errs)
}
}
// includeTree writes a directory of include files and returns its path.
func includeTree(t *testing.T, files map[string]string) string {
t.Helper()
dir := t.TempDir()
for name, content := range files {
path := filepath.Join(dir, name)
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
}
return dir
}
func TestIncludeSplicesAndDefinesAreShared(t *testing.T) {
dir := includeTree(t, map[string]string{
"consts.h": "#define KONST $42\n",
})
f, errs := ParseWithOptions("t_amd64.s", `
#include "consts.h"
TEXT ·f(SB), NOSPLIT, $0
MOVQ KONST, AX
RET
`, Options{Expand: true, IncludeDirs: []string{dir}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
if in.Mnemonic.Text != "MOVQ" || strings.ReplaceAll(in.Operands[0].Raw, " ", "") != "$42" {
t.Fatalf("include splicing failed: %+v", in)
}
}
func TestIncludeResolutionOrder(t *testing.T) {
// The including file's directory wins over the -I list, and the -I list
// is searched in order.
src := includeTree(t, map[string]string{
"inc/main.s": "#include \"which.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"inc/which.h": "#define WHO ONE\n",
"first/which.h": "#define WHO TWO\n",
"second/which.h": "#define WHO THREE\n",
})
main := filepath.Join(src, "inc", "main.s")
body, err := os.ReadFile(main)
if err != nil {
t.Fatal(err)
}
// The header exists in the including file's directory and in two -I
// directories; the source-directory copy must win.
f, errs := ParseWithOptions(main, string(body), Options{Expand: true, IncludeDirs: []string{
filepath.Join(src, "first"), filepath.Join(src, "second"),
}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
found := false
for _, d := range f.Decls {
if pp, ok := d.(*ast.Preproc); ok && strings.Contains(pp.Raw, "define WHO ONE") {
found = true
}
}
if !found {
t.Error("the including file's directory did not win include resolution")
}
}
func TestIncludeSearchesIncludeDirsInOrder(t *testing.T) {
src := includeTree(t, map[string]string{
"inc/main.s": "#include \"which.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"first/which.h": "#define WHO TWO\n",
"second/which.h": "#define WHO THREE\n",
})
main := filepath.Join(src, "inc", "main.s")
body, err := os.ReadFile(main)
if err != nil {
t.Fatal(err)
}
f, errs := ParseWithOptions(main, string(body), Options{Expand: true, IncludeDirs: []string{
filepath.Join(src, "first"), filepath.Join(src, "second"),
}})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
for _, d := range f.Decls {
if pp, ok := d.(*ast.Preproc); ok && strings.Contains(pp.Raw, "define WHO THREE") {
t.Error("the second -I directory was searched before the first")
}
}
}
func TestIncludeCycleIsDetected(t *testing.T) {
src := includeTree(t, map[string]string{
"a.s": "#include \"b.s\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
"b.s": "#include \"a.s\"\n",
})
_, errs := ParseWithOptions(filepath.Join(src, "a.s"), "#include \"b.s\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), "include cycle") {
t.Fatalf("include cycle: got %v, want a cycle diagnostic, not a hang", errs)
}
}
func TestUnresolvableIncludeIsAnError(t *testing.T) {
_, errs := ParseWithOptions("t_amd64.s", "#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n",
Options{Expand: true, IncludeDirs: []string{t.TempDir()}})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), `#include "nothere.h"`) {
t.Fatalf("missing include: got %v, want a clear diagnostic", errs)
}
}
func TestTextflagHeaderIsNeverSpliced(t *testing.T) {
// textflag.h resolves nowhere here, yet the file must parse: the flag
// names are consumed natively and the include stays in the tree.
f, errs := ParseWithOptions("t_amd64.s", `
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
hasInclude := false
for _, d := range f.Decls {
if _, ok := d.(*ast.Include); ok {
hasInclude = true
}
}
if !hasInclude {
t.Error("textflag.h include was dropped from the tree")
}
}
func TestSemicolonSplitsRawLinesToo(t *testing.T) {
_, got := expand(t, `
TEXT ·f(SB), NOSPLIT, $0
BYTE $0x0f; BYTE $0x1f
RET
`)
wantLines(t, got, "BYTE $0x0f", "BYTE $0x1f", "RET")
}
func TestParseUnchangedWithoutExpand(t *testing.T) {
// Without Expand the preprocessor must not exist: a macro invocation
// stays an unexpanded instruction line and ';' keeps the old parse.
f, errs := Parse("t_amd64.s", `
#define TWICE ADDQ AX, AX
TEXT ·f(SB), NOSPLIT, $0
TWICE
BYTE $0x0f; BYTE $0x1f
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
var mnemonics []string
for _, s := range fn.Body {
if in, ok := s.(*ast.Instr); ok {
mnemonics = append(mnemonics, in.Mnemonic.Text)
}
}
if strings.Join(mnemonics, " ") != "TWICE BYTE RET" {
t.Errorf("non-expanding parse changed: %v", mnemonics)
}
}
func TestConstantExpressionFolding(t *testing.T) {
// The shapes substituted macro bodies leave behind: parenthesised
// arithmetic in immediates and displacements, tilde complements. The
// assertions read the semantic fields; Raw keeps the operand's tokens
// in the canonicalised rendering, not the folded values.
f, errs := ParseWithOptions("t_amd64.s", `
TEXT ·f(SB), NOSPLIT, $0
RORQ $(32-7), AX
ANDQ $~63, AX
MOVQ ((2*4)+0)(SP), AX
MOVQ $((1<<3)|(1<<1)), AX
RET
`, Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
body := texts(f)[0].Body
ror := body[0].(*ast.Instr)
if !ror.Operands[0].Imm.HasVal || ror.Operands[0].Imm.Val != 25 {
t.Errorf("RORQ immediate = %+v, want 25", ror.Operands[0].Imm)
}
and := body[1].(*ast.Instr)
if !and.Operands[0].Imm.HasVal || and.Operands[0].Imm.Val != -64 {
t.Errorf("ANDQ immediate = %+v, want -64", and.Operands[0].Imm)
}
mov := body[2].(*ast.Instr)
addr := mov.Operands[0].Addr
if !addr.HasOff || addr.Offset != 8 || addr.Base != "SP" {
t.Errorf("MOVQ address = %+v, want 8(SP)", addr)
}
mov2 := body[3].(*ast.Instr)
if !mov2.Operands[0].Imm.HasVal || mov2.Operands[0].Imm.Val != 10 {
t.Errorf("MOVQ immediate = %+v, want 10", mov2.Operands[0].Imm)
}
}
func TestConstantExpressionFoldsWithoutExpand(t *testing.T) {
// Folding is a parser capability, not a preprocessing one: a
// hand-written $(32-7) folds the same way with expansion off.
f, errs := ParseWithOptions("t_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n\tRORQ $(32-7), AX\n\tRET\n", Options{})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
in := texts(f)[0].Body[0].(*ast.Instr)
if !in.Operands[0].Imm.HasVal || in.Operands[0].Imm.Val != 25 {
t.Errorf("Imm = %+v, want 25", in.Operands[0].Imm)
}
}
func TestNotAnExpressionFallsBack(t *testing.T) {
// Symbol immediates and floats must keep their ordinary parse.
f, errs := ParseWithOptions("t_amd64.s", "TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ $1.5, AX\n\tMOVQ $·sym(SB), AX\n\tRET\n", Options{})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
fn := texts(f)[0]
mov1 := fn.Body[0].(*ast.Instr)
if mov1.Operands[0].Imm.HasVal || mov1.Operands[0].Imm.Float != "1.5" {
t.Errorf("float immediate parsed as %+v", mov1.Operands[0].Imm)
}
mov2 := fn.Body[1].(*ast.Instr)
if mov2.Operands[0].Imm.Sym == nil {
t.Errorf("symbol immediate parsed as %+v", mov2.Operands[0].Imm)
}
}
+69
View File
@@ -0,0 +1,69 @@
// Atomics and carry-extending multi-word arithmetic: exchange,
// compare-exchange, exchange-add, ADCX/ADOX and the CRC-32 accumulator
// family. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func xchg(p *uint64, v uint64) uint64
TEXT ·xchg(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
XCHGQ BX, (AX)
XCHGQ BX, CX
XCHGL BX, CX
XCHGW BX, CX
XCHGB BL, CL
MOVQ AX, ret+16(FP)
RET
// func cmpxchg(p *uint64, old, new uint64) uint8
TEXT ·cmpxchg(SB), NOSPLIT, $0-25
MOVQ p+0(FP), AX
MOVQ old+8(FP), BX
MOVQ new+16(FP), CX
CMPXCHGQ CX, (AX)
CMPXCHGL CX, BX
CMPXCHGW CX, BX
CMPXCHGB CL, BL
SETEQ AL
MOVB AL, ret+24(FP)
RET
// func xadd(p *uint64, v uint64) uint64
TEXT ·xadd(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
XADDQ BX, (AX)
XADDL BX, CX
XADDW BX, CX
XADDB BL, CL
MOVQ AX, ret+16(FP)
RET
// func adcx_adox(lo, hi, x, y uint64) uint64
TEXT ·adcx_adox(SB), NOSPLIT, $0-40
MOVQ lo+0(FP), AX
MOVQ hi+8(FP), DX
MOVQ x+16(FP), BX
MOVQ y+24(FP), CX
ADCXQ BX, AX
ADOXQ CX, DX
ADCXL BX, AX
ADOXL CX, DX
XORQ BX, BX
ADCXQ BX, AX
MOVQ AX, ret+32(FP)
RET
// func crc32(crc uint32, p *byte, n int) uint32
TEXT ·crc32(SB), NOSPLIT, $0-28
MOVL crc+0(FP), AX
MOVQ p+8(FP), SI
MOVQ n+16(FP), CX
CRC32B (SI), AX
CRC32Q (SI), CX
CRC32L (SI), AX
MOVW (SI), DX
CRC32W DX, AX
MOVL AX, ret+24(FP)
RET
+72
View File
@@ -0,0 +1,72 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 synchronisation instructions: the
// acquire/release loads and stores, the exclusive family and the LSE
// atomics with acquire and release semantics, plus the register-pair
// loads and stores. Every function is byte-compared against go tool asm.
#include "textflag.h"
// func acquireRelease()
TEXT ·acquireRelease(SB), NOSPLIT, $0-0
LDAR (R1), R2
LDARB (R3), R4
LDARH (R5), R6
LDARW (R7), R8
STLR R2, (R1)
STLRB R4, (R3)
STLRH R6, (R5)
STLRW R8, (R7)
RET
// func exclusive()
TEXT ·exclusive(SB), NOSPLIT, $0-0
LDAXR (R1), R2
LDAXRB (R3), R4
LDAXRW (R5), R6
STLXR R2, (R1), R8
STLXRB R4, (R3), R8
STLXRW R6, (R5), R8
RET
// func lseAcquireRelease()
TEXT ·lseAcquireRelease(SB), NOSPLIT, $0-0
CASALD R1, (R3), R2
CASALW R4, (R6), R5
LDADDALD R1, (R3), R2
LDADDALW R4, (R6), R5
LDCLRALB R1, (R3), R2
LDCLRALW R4, (R6), R5
LDCLRALD R1, (R3), R2
LDORALB R1, (R3), R2
LDORALW R4, (R6), R5
LDORALD R1, (R3), R2
SWPALB R1, (R3), R2
SWPALW R4, (R6), R5
SWPALD R1, (R3), R2
RET
// func lseBase()
TEXT ·lseBase(SB), NOSPLIT, $0-0
LDADDD R1, (R3), R2
LDADDW R4, (R6), R5
CASD R1, (R3), R2
CASW R4, (R6), R5
SWPD R1, (R3), R2
SWPW R4, (R6), R5
RET
// func pairs()
TEXT ·pairs(SB), NOSPLIT, $0-0
LDP (R1), (R2, R3)
LDP 8(R4), (R5, R6)
LDP -16(R1), (R2, R3)
LDPW 4(R4), (R5, R6)
STP (R2, R3), 24(R7)
STP (R2, R3),-8(R7)
STPW (R1, R2), 4(R0)
FLDPD (R8), (F1, F2)
FLDPD 8(R8), (F3, F4)
FSTPD (F3, F4),-8(R9)
RET
+49
View File
@@ -0,0 +1,49 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the loong64 atomics: the AM* family in its plain
// and _dbar (acquire/release) forms, spelled as the runtime's
// atomic_loong64.s spells them. Every AM* takes three operands:
// value, (address), result.
#include "textflag.h"
TEXT ·plain(SB), NOSPLIT, $0-0
AMSWAPB R14, (R13), R12
AMSWAPH R14, (R13), R12
AMSWAPW R5, (R4), R6
AMSWAPV R5, (R4), R0
AMCASB R14, (R13), R12
AMCASH R6, (R4), R5
AMCASW R6, (R4), R5
AMCASV R6, (R4), R5
AMADDW R5, (R4), R0
AMADDV R14, (R13), R12
AMANDW R5, (R4), R6
AMANDV R5, (R4), R6
AMORW R5, (R4), R0
AMORV R5, (R4), R6
AMXORW R5, (R4), R6
AMXORV R5, (R4), R6
AMMAXW R5, (R4), R6
AMMAXV R5, (R4), R6
AMMINW R5, (R4), R6
AMMINV R5, (R4), R6
AMMAXWU R5, (R4), R6
AMMAXVU R5, (R4), R6
AMMINWU R5, (R4), R6
AMMINVU R5, (R4), R6
RET
TEXT ·dbar(SB), NOSPLIT, $0-0
AMADDDBW R5, (R4), R6
AMADDDBV R5, (R4), R6
AMANDDBW R5, (R6), R0
AMANDDBV R5, (R4), R6
AMORDBW R5, (R6), R0
AMORDBV R5, (R4), R6
AMSWAPDBW R5, (R4), R6
AMSWAPDBV R5, (R4), R0
AMCASDBW R6, (R4), R5
AMCASDBV R6, (R4), R5
RET
+35
View File
@@ -0,0 +1,35 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 atomics: the RV64A AMO family and the
// load-reserved / store-conditional pair, in the toolchain's spelling
// (value, (address), result). Both orderings sit in the encodings: the
// table gives every AMO aq and rl, LR acquire and SC release.
#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0-0
AMOSWAPW X5, (X6), X7
AMOSWAPD X5, (X6), X7
AMOADDW X5, (X6), X7
AMOADDD X5, (X6), X7
AMOANDW X5, (X6), X7
AMOANDD X5, (X6), X7
AMOORW X5, (X6), X7
AMOORD X5, (X6), X7
AMOXORW X5, (X6), X7
AMOXORD X5, (X6), X7
AMOMAXW X5, (X6), X7
AMOMAXD X5, (X6), X7
AMOMAXUW X5, (X6), X7
AMOMAXUD X5, (X6), X7
AMOMINUW X5, (X6), X7
AMOMINUD X5, (X6), X7
RET
TEXT ·lrsc(SB), NOSPLIT, $0-0
LRW (X5), X6
LRD (X5), X6
SCW X5, (X6), X7
SCD X5, (X6), X7
RET
+102
View File
@@ -0,0 +1,102 @@
// The AVX/AVX-512 gap families: fused scalar multiply-add, carries through
// GF(2^8) affine transforms, population counts, non-temporal stores, mask
// moves and the KMOV widths. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func avxblend(a, b []float64) float64
TEXT ·avxblend(SB), NOSPLIT, $0-56
MOVQ a_base+0(FP), SI
MOVQ b_base+24(FP), DI
VMOVUPD (SI), Y0
VMOVUPD (DI), Y1
VXORPS Y2, Y2, Y2
VSHUFPD $5, Y0, Y1, Y3
VMOVUPD Y3, (SI)
VPBLENDD $3, Y0, Y1, Y4
VPERM2F128 $1, Y4, Y0, Y0
VEXTRACTF128 $1, Y0, X1
VZEROALL
VMOVSD X1, ret+48(FP)
RET
// func avxint(p *byte, n int) uint64
TEXT ·avxint(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
VMOVDQU (SI), Y0
VPCMPEQB Y0, Y0, Y1
VPSLLDQ $2, X0, X0
VPSRLDQ $4, Y0, Y0
VPALIGNR $3, X0, X1, X1
VPCLMULQDQ $0, X0, X1, X2
VGF2P8AFFINEQB $7, X2, X0, X3
VPOPCNTB X3, X4
VPOPCNTD Y0, Y5
VPERMI2B X0, X1, X2
VPTEST X0, X0
VPMOVMSKB X1, AX
VZEROUPPER
MOVQ AX, ret+16(FP)
RET
// func avxnt(p *float64)
TEXT ·avxnt(SB), NOSPLIT, $0-8
MOVQ p+0(FP), DI
VMOVUPD (DI), Y0
VADDPD Y0, Y0, Y0
VMOVNTDQ Y0, (DI)
VMOVNTDQ X0, 16(DI)
VZEROALL
RET
// func avxmas(a, b []float64) float64
TEXT ·avxmas(SB), NOSPLIT, $0-56
MOVQ a_base+0(FP), SI
MOVQ b_base+24(FP), DI
VMOVSD (SI), X0
VMOVSD (DI), X1
VFMADD213SD X1, X0, X0
VFNMADD231SD X1, X0, X0
VADDSD X1, X0, X0
VMOVSD X0, ret+48(FP)
RET
// func avxgpr(x, y uint64) uint64
TEXT ·avxgpr(SB), NOSPLIT, $0-24
MOVQ x+0(FP), AX
MOVQ y+8(FP), BX
ANDNL BX, AX, CX
MULXQ BX, DX, SI
RORXL $3, AX, CX
RORXQ $7, BX, SI
MOVQ CX, ret+16(FP)
RET
// func avxmask(kin uint8, p *byte) uint8
TEXT ·avxmask(SB), NOSPLIT, $0-17
MOVQ p+8(FP), SI
KMOVB kin+0(FP), K1
KMOVB K1, K2
KMOVW K2, K1
KMOVD K1, K3
KMOVQ K3, K4
KMOVB K4, K1
KMOVB K1, AX
KMOVD K1, (SI)
MOVB AL, ret+8(FP)
RET
// func avx512(p *uint64, n int) uint64
TEXT ·avx512(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
VMOVDQU64 (SI), Z0
VPORQ Z0, Z0, Z1
VPOPCNTQ Z1, Z2
VPERMB Z1, Z0, Z2
VPXORD Z2, Z1, Z0
VMOVDQA64 Z0, (SI)
VZEROUPPER
XORQ AX, AX
MOVQ AX, ret+16(FP)
RET
+64
View File
@@ -0,0 +1,64 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 toolchain-synthesised instructions:
// the Zbb-style pseudos the assembler expands instruction-for-instruction
// (ANDN/ORN, MIN/MAX, ROR and friends, the reversed branches, FABSD), the
// CSR read RDTIME and the FP sign-injection and fused-multiply-add forms.
#include "textflag.h"
TEXT ·logic(SB), NOSPLIT, $0-0
ANDN X19, X20, X21
ANDN X19, X20
ANDN X21, X19, X21
ORN X20, X19
ORN X20, X19, X21
MAX X26, X28, X29
MAX X26, X28
MAXU X28, X29, X30
MAXU X28, X29
MIN X29, X30, X5
MIN X29, X30
MINU X30, X5, X6
MINU X30, X5
MAX X5, X5
MAX X5, X5, X6
SEQZ X5, X6
NEG X5, X6
NEG X5
NOT X5
NOT X5, X6
NOP
RET
TEXT ·rotate(SB), NOSPLIT, $0-0
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORIW $31, X13, X14
RORIW $1, X14, X15
RORIW $3, X14
RORW X15, X16, X17
RORW $31, X13
RET
TEXT ·fp(SB), NOSPLIT, $0-0
FABSD F1, F2
FSGNJD F1, F0, F2
FMADDD F1, F2, F3, F4
FMSUBD F1, F2, F3, F4
FNMSUBD F1, F2, F3, F4
FMADDS F1, F2, F3, F4
FNMADDS F1, F2, F3, F4
RET
TEXT ·branches(SB), NOSPLIT, $0-0
BGT X5, X6, tgt
BLE X5, X6, tgt
BGTU X5, X6, tgt
BLEU X5, X6, tgt
tgt:
RDTIME X5
RET
File diff suppressed because it is too large Load Diff
+48
View File
@@ -0,0 +1,48 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Carry arithmetic, logical shifts, register aliases with element selectors
// and the ADC/SBC immediate spellings: the shapes nat_arm64.s, p256 and
// gcm_arm64.s exercise. Byte-for-byte against go tool asm.
#include "textflag.h"
#define acc0 V8
#define acc1 V9
#define const0 R15
#define POLY V15
// carry pins the ADC/SBC family: the $0 spellings in two and three
// operands, and the register-carry forms.
TEXT ·carry(SB), NOSPLIT, $0-0
ADC $0, R20
ADC $0, R20, R4
SBCS $0, R4
SBCS $0, R4, R12
SBCS R15, R4, R12
SBC $0, R1
ADCSW $0, R2, R3
RET
// shift pins the shifted-register forms including ROR, which only the
// logical family accepts.
TEXT ·shift(SB), NOSPLIT, $0-0
ANDW R9@>7, R19, R26
AND R1@>33, R2, R3
ADD R1<<11, R2, R3
SUB R1->33, R2
ORR R5<<2, R6, R7
RET
// vecalias pins the vector aliases with element selectors and the
// structure loads with aliased members.
TEXT ·vecalias(SB), NOSPLIT, $0-0
MOVD $0xC2, R1
VMOV R1, POLY.D[0]
VMOV R0, POLY.D[1]
VEOR POLY.B16, POLY.B16, POLY.B16
VLD1 (R0), [acc0.B16]
VLD1.P (R0), [acc0.B16, acc1.B16]
VST1 [acc0.B16, acc1.B16], (R1)
VST1.P [acc0.B16, acc1.B16], 32(R1)
RET
+57
View File
@@ -0,0 +1,57 @@
// The AES-NI, SHA and carry-less multiply round instructions as GOROOT's
// crypto kernels spell them. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func aesround(blk, rk *byte)
TEXT ·aesround(SB), NOSPLIT, $0-16
MOVQ blk+0(FP), SI
MOVQ rk+8(FP), DI
MOVOU (SI), X0
MOVOU (DI), X1
AESENC X1, X0
AESENCLAST X1, X0
AESDEC X1, X0
AESDECLAST X1, X0
AESIMC X1, X2
AESKEYGENASSIST $1, X1, X3
MOVOU X0, (SI)
MOVOU X2, (DI)
RET
// func sha1block(p *byte, n int, h *[5]uint32)
TEXT ·sha1block(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ h+16(FP), DI
MOVOU (SI), X0
MOVOU 16(SI), X1
SHA1RNDS4 $0, X1, X0
SHA1NEXTE X1, X0
SHA1MSG1 X1, X2
SHA1MSG2 X1, X2
MOVOU X0, (DI)
RET
// func sha256block(p *byte, n int, h *[8]uint32)
TEXT ·sha256block(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ h+16(FP), DI
MOVOU (SI), X0
MOVOU 16(SI), X1
SHA256RNDS2 X0, X1, X0
SHA256MSG1 X1, X2
SHA256MSG2 X1, X2
MOVOU X0, (DI)
RET
// func pclmul(a, b *byte)
TEXT ·pclmul(SB), NOSPLIT, $0-16
MOVQ a+0(FP), SI
MOVQ b+8(FP), DI
MOVOU (SI), X0
MOVOU (DI), X1
PCLMULQDQ $0, X1, X0
PCLMULQDQ $17, (DI), X0
MOVOU X0, (SI)
RET
+42
View File
@@ -0,0 +1,42 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 cryptographic extension: the AES round
// instructions and the SHA1, SHA256 and SHA512 families. Every function is
// byte-compared against go tool asm.
#include "textflag.h"
// func aesRound()
TEXT ·aesRound(SB), NOSPLIT, $0-0
AESE V31.B16, V29.B16
AESD V22.B16, V19.B16
AESMC V14.B16, V28.B16
AESIMC V12.B16, V27.B16
RET
// func sha1Round()
TEXT ·sha1Round(SB), NOSPLIT, $0-0
SHA1C V8.S4, V8, V2
SHA1P V3.S4, V20, V27
SHA1M V0.S4, V27, V27
SHA1H V17, V25
SHA1SU0 V17.S4, V13.S4, V16.S4
SHA1SU1 V24.S4, V23.S4
RET
// func sha256Round()
TEXT ·sha256Round(SB), NOSPLIT, $0-0
SHA256H V4.S4, V2, V11
SHA256H2 V6.S4, V16, V11
SHA256SU0 V0.S4, V16.S4
SHA256SU1 V31.S4, V3.S4, V15.S4
RET
// func sha512Round()
TEXT ·sha512Round(SB), NOSPLIT, $0-0
SHA512H V2.D2, V1, V0
SHA512H2 V4.D2, V3, V2
SHA512SU0 V9.D2, V8.D2
SHA512SU1 V7.D2, V6.D2, V5.D2
RET
+33
View File
@@ -0,0 +1,33 @@
// The three-operand SHL/SHR forms, which go tool asm encodes as SHLD/SHRD:
// immediate and CL (or its CX spelling) counts at the Q and W widths, next
// to the two-operand CX-count spelling GOROOT's bignum kernels use. Every
// result is folded back so no instruction is dead.
#include "textflag.h"
// func dblshift(x, y uint64) uint64
TEXT ·dblshift(SB), NOSPLIT, $0-24
MOVQ x+0(FP), SI
MOVQ y+8(FP), DI
MOVQ $12, CX
SHLQ $13, SI, DI
SHRQ $7, DI, SI
SHLQ CX, SI, DI
SHRQ CX, DI, SI
SHLQ CX, SI
SHLQ $9, DI
SHLW $1, SI, DI
SHRW $3, DI, SI
XORQ DI, SI
MOVQ SI, ret+16(FP)
RET
// func dblshift32(a, b uint32) uint32
TEXT ·dblshift32(SB), NOSPLIT, $0-12
MOVL a+0(FP), SI
MOVL b+4(FP), DI
SHLL $5, SI, DI
SHRL $2, DI, SI
XORL SI, DI
MOVL DI, ret+8(FP)
RET
+66
View File
@@ -0,0 +1,66 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 integer slice: carry-setting arithmetic,
// widening multiplies, bit manipulation, conditional compares, the compare
// and test branches, ADR and the wide-constant moves. Every function is
// byte-compared against go tool asm.
#include "textflag.h"
// func carryArith()
TEXT ·carryArith(SB), NOSPLIT, $0-0
ADC R0, R2, R12
ADCS R23, R22, R22
ADC $0, R1
SBC R25, R10, R26
SBCS R5, R9, R5
SBCS $0, R1
RET
// func wideningMul()
TEXT ·wideningMul(SB), NOSPLIT, $0-0
MUL R4, R3, R0
MSUB R19, R16, R26, R2
SMULH R24, R20, R24
UMULH R24, R20, R24
RET
// func bitManip()
TEXT ·bitManip(SB), NOSPLIT, $0-0
RBIT R11, R4
REV R1, R2
CLZ R21, R9
REVW R1, R2
CLSW R1, R2
UBFX $33, R17, $25, R5
UBFXW $4, R1, $9, R2
RET
// func condCompare()
TEXT ·condCompare(SB), NOSPLIT, $0-0
CCMP LE, R7, $19, $3
CCMP LT, R30, R6, $7
CCMN EQ, R1, R2, $3
CCMPW LE, R7, $19, $3
RET
// func branchForms()
TEXT ·branchForms(SB), NOSPLIT, $0-0
CBZ R1, target
CBNZ R7, target
CBNZW R2, target
TBZ $4, R7, target
TBNZ $33, R7, target
ADR target, R10
target:
RET
// func wideMoves()
TEXT ·wideMoves(SB), NOSPLIT, $0-0
MOVK $1234, R5
MOVK $305397760, R5
MOVKW $1234, R5
MOVK $16771847290880, R21
RET
+51
View File
@@ -0,0 +1,51 @@
// The subtract-immediate fold, the TEQ/TNE trap pseudos, PRELDX, the FP
// condition branches and the N(PC) branch spellings, against the toolchain.
#include "textflag.h"
// func SubFold(x int64) int64
TEXT ·SubFold(SB), NOSPLIT, $0-16
MOVV x+0(FP), R8
SUBV $0, R8
SUBV $4, R9, R10
SUBV $4096, R11
SUBV $-4, R12
SUB $1, R13
SUBVU $4, R14
SUBV $1048576, R15
MOVV R8, ret+8(FP)
RET
// func Traps(x int64) int64
TEXT ·Traps(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
TEQ $4, R4, R5
TEQ $4, R4
TNE $6, R5, R6
MOVV R4, ret+8(FP)
RET
// func Prefetch(x int64) int64
TEXT ·Prefetch(SB), NOSPLIT, $0-16
MOVV x+0(FP), R7
PRELDX 0(R7), $0x80001021, $0
PRELDX -1(R7), $0x1021, $2
MOVV R7, ret+8(FP)
RET
// func BranchForms(x int64) int64
TEXT ·BranchForms(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
l1:
BFPT l1
BFPT FCC3, l1
BFPF l1
JMP -4(PC)
JAL 1(PC)
JAL (R4)
loop:
ADDV $1, R4
BEQ R4, R5, loop
BNE R4, l1
RET
+33
View File
@@ -0,0 +1,33 @@
// PCALIGN padding on loong64: andi $0, $0, 0 (the architecture's NOP), plus
// the automatic loop-head alignment to a 16-byte boundary.
#include "textflag.h"
// func Pad16(x int64) int64
TEXT ·Pad16(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
PCALIGN $16
ADDV $1, R4
MOVV R4, ret+8(FP)
RET
// func Pad32(x int64) int64
TEXT ·Pad32(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
PCALIGN $32
ADDV $1, R4
MOVV R4, ret+8(FP)
RET
// func LoopAlign(x int64) int64
TEXT ·LoopAlign(SB), NOSPLIT, $0-16
MOVV x+0(FP), R4
MOVV $10, R5
loop:
BEQ R4, R5, done
ADDV $1, R4
JMP loop
done:
MOVV R4, ret+8(FP)
RET
+35
View File
@@ -0,0 +1,35 @@
// PCALIGN padding on riscv64: 4-byte NOPs with a 2-byte compressed NOP when
// the pad is 2 mod 4, exactly as the toolchain lays the bytes down.
#include "textflag.h"
// func Pad8(x int64) int64
TEXT ·Pad8(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $8
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func Pad16(x int64) int64
TEXT ·Pad16(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $16
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func Pad32(x int64) int64
TEXT ·Pad32(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $32
ADD $1, X5
MOV X5, ret+8(FP)
RET
// func PadAfterOdd(x int64) int64
TEXT ·PadAfterOdd(SB), NOSPLIT, $0-16
MOV x+0(FP), X5
PCALIGN $8
ADD $1, X5
MOV X5, ret+8(FP)
RET
+103
View File
@@ -0,0 +1,103 @@
// Instruction prefixes: LOCK, REP and REPN. go tool asm encodes each
// statement as a standalone one-byte instruction with a PC of its own (F0,
// F3 and F2 respectively); the statement that follows is encoded unaware of
// it, and nothing validates the pairing. The shapes are the runtime's
// atomic read-modify-write family and the string moves, every result folded
// back.
#include "textflag.h"
// func cas64(ptr *uint64, old, new uint64) bool
TEXT ·cas64(SB), NOSPLIT, $0-25
MOVQ ptr+0(FP), BX
MOVQ old+8(FP), AX
MOVQ new+16(FP), CX
LOCK
CMPXCHGQ CX, 0(BX)
SETEQ ret+24(FP)
RET
// func casloop(addr *uint64, v uint64) uint64
// The runtime's Or64 shape: a LOCK inside a branch loop, the backward jump
// measuring over the prefix statement's own byte.
TEXT ·casloop(SB), NOSPLIT, $0-24
MOVQ addr+0(FP), BX
MOVQ v+8(FP), CX
loop:
MOVQ CX, DX
MOVQ (BX), AX
ORQ AX, DX
LOCK
CMPXCHGQ DX, (BX)
JNZ loop
MOVQ AX, ret+16(FP)
RET
// func xadd64(p *uint64, v uint64) uint64
TEXT ·xadd64(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
LOCK
XADDQ BX, (AX)
MOVQ AX, ret+16(FP)
RET
// func xaddw(p *uint16, v uint16) uint16
TEXT ·xaddw(SB), NOSPLIT, $0-12
MOVQ p+0(FP), AX
MOVW v+8(FP), BX
LOCK
XADDW BX, (AX)
MOVW AX, ret+8(FP)
RET
// func lockarith(p *uint64)
TEXT ·lockarith(SB), NOSPLIT, $0-8
MOVQ p+0(FP), AX
LOCK
ORQ CX, (AX)
LOCK
ANDL CX, (AX)
LOCK
INCQ (AX)
LOCK
DECQ (AX)
LOCK
ORB BX, (AX)
RET
// func repstring(dst, src *byte, n int)
// The memmove shapes: forward copy by quadwords, backward tails.
TEXT ·repstring(SB), NOSPLIT, $0-24
MOVQ dst+0(FP), DI
MOVQ src+8(FP), SI
REP
MOVSQ
REP
MOVSB
REPN
MOVSB
REP
STOSQ
REP
STOSB
RET
// func pfxlabel()
// Labels pinned on prefix statements' own bytes: pfx: sits on the LOCK,
// mid: on the REPN.
TEXT ·pfxlabel(SB), NOSPLIT, $0-0
pfx:
LOCK
XCHGL BX, (AX)
JMP done
mid:
REPN
MOVSB
done:
REP
STOSB
RET
+62
View File
@@ -0,0 +1,62 @@
// Literal data emission: BYTE, WORD, LONG and QUAD write the immediate
// into the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
// lookup, truncated to the width rather than range-checked; END is
// accepted and ignored, contributing no bytes and ending nothing. The
// shapes mirror the runtime's hand-laid markers
// (crypto/internal/boring/sig/sig_amd64.s) and its syscall stubs
// (runtime/sys_linux_amd64.s).
#include "textflag.h"
// func marker()
// A boring/crypto-style marker: a hand-laid forward branch whose skip
// distance is patched at runtime. One BYTE per statement, as the
// runtime's own file spells it: the semicolon-separated one-liner the
// sys_linux_amd64.s stub uses does not survive gasm fmt, which drops the
// statement separators.
TEXT ·marker(SB), NOSPLIT, $0-0
BYTE $0xEB
BYTE $0x1D
BYTE $0xF4
BYTE $0x48
BYTE $0xF4
BYTE $0x4B
BYTE $0xC3
RET
// func stub()
// The sys_linux_amd64.s stub bytes: the sign-extended
// "48 c7 c0 0f 00 00 00" form of MOVQ $rt_sigreturn, AX.
TEXT ·stub(SB), NOSPLIT, $0-0
BYTE $0x48
BYTE $0xc7
BYTE $0xc0
BYTE $0x0f
BYTE $0x00
BYTE $0x00
BYTE $0x00
RET
// func words()
// The wider literals, and an END that ends nothing: the WORD after it
// still lands in this function.
TEXT ·words(SB), NOSPLIT, $0-0
WORD $0x1234
WORD $-1
LONG $0x11223344
LONG $-1
QUAD $0x1122334455667788
QUAD $-2
END
WORD $0xBEEF
RET
// func trunc()
// Truncation, not a range check: each literal keeps its low bytes, exactly
// as go tool asm emits them.
TEXT ·trunc(SB), NOSPLIT, $0-0
BYTE $0x1FF
WORD $0x12345
LONG $0x123456789
QUAD $-2
RET
+76
View File
@@ -0,0 +1,76 @@
// Carry arithmetic, rotates, unsigned/signed division and bit tests: the
// scalar families GOROOT's big-number and crypto kernels use. Every result
// is folded back so no instruction is dead.
#include "textflag.h"
// func carry(a, b uint64) uint64
TEXT ·carry(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
ADCQ $0, AX
MOVQ BX, CX
SBBQ $1, CX
ADCL BX, AX
ADCB AL, BL
ADCW $7, CX
MOVQ AX, ret+16(FP)
RET
// func borrow(a, b uint64) uint64
TEXT ·borrow(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
SUBQ BX, AX
SBBQ $0, AX
SBBQ BX, CX
MOVQ AX, ret+16(FP)
RET
// func rot(x uint64, n uint32) uint64
TEXT ·rot(SB), NOSPLIT, $0-24
MOVQ x+0(FP), AX
MOVL n+8(FP), CX
ROLQ CL, AX
RORQ $7, AX
ROLL $1, AX
RORL CL, AX
RCLQ $1, AX
RCRQ CL, AX
ROLW $3, AX
SALQ $2, AX
SALB $1, AX
MOVQ AX, ret+8(FP)
RET
// func muldiv(a, b uint64) uint64
TEXT ·muldiv(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
MULQ BX
MULQ (BX)
MOVL (BX), CX
MULL CX
DIVQ BX
IDIVQ BX
MOVL a+0(FP), AX
DIVL CX
IDIVL CX
MOVQ AX, ret+16(FP)
RET
// func bitfield(w *uint64) uint64
TEXT ·bitfield(SB), NOSPLIT, $0-16
MOVQ (DI), AX
MOVQ (DI), CX
BTQ AX, CX
BTQ $3, (DI)
BTL AX, CX
BTW $1, CX
BTSQ $5, AX
BTRQ AX, CX
BTCQ $7, (DI)
SETCS AL
MOVQ AX, ret+8(FP)
RET
+98
View File
@@ -0,0 +1,98 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 NEON slice: the logical and arithmetic
// three-register operations, permutations, comparisons, shifts, the crypto
// four-register group, element moves, table lookups and the structure
// loads and stores. Every function is byte-compared against go tool asm.
#include "textflag.h"
// func simdLogic()
TEXT ·simdLogic(SB), NOSPLIT, $0-0
VADD V1.B16, V2.B16, V3.B16
VADD V1.B8, V2.B8, V3.B8
VSUB V1.S4, V2.S4, V3.S4
VMUL V1.H8, V2.H8, V3.H8
VAND V4.B16, V4.B16, V9.B16
VORR V5.B16, V4.B16, V3.B16
VEOR V0.B16, V1.B16, V0.B16
VADDP V1.H8, V2.H8, V3.H8
VCMEQ V24.S4, V13.S4, V12.S4
VCMEQ $0, V2.H4, V3.H4
RET
// func simdPerm()
TEXT ·simdPerm(SB), NOSPLIT, $0-0
VZIP1 V16.H8, V3.H8, V19.H8
VZIP1 V6.D2, V9.D2, V11.D2
VZIP2 V22.D2, V25.D2, V21.D2
VREV32 V2.H8, V1.H8
VREV64 V2.S4, V3.S4
VUADDLV V31.S4, V11
VEXT $4, V2.B8, V1.B8, V3.B8
VEXT $8, V2.B16, V1.B16, V3.B16
RET
// func simdShift()
TEXT ·simdShift(SB), NOSPLIT, $0-0
VSHL $7, V22.D2, V25.D2
VSHL $24, V1.S4, V2.S4
VUSHR $6, V22.H8, V23.H8
VUSHR $56, V1.D2, V2.D2
VSRI $24, V1.S4, V2.S4
VSRI $56, V1.D2, V2.D2
RET
// func simdCrypto4()
TEXT ·simdCrypto4(SB), NOSPLIT, $0-0
VEOR3 V2.B16, V7.B16, V12.B16, V25.B16
VBCAX V1.B16, V2.B16, V26.B16, V31.B16
VXAR $63, V27.D2, V21.D2, V26.D2
VRAX1 V26.D2, V29.D2, V30.D2
VPMULL V2.D1, V1.D1, V3.Q1
VPMULL V2.B8, V1.B8, V3.H8
VPMULL2 V2.D2, V1.D2, V4.Q1
VPMULL2 V2.B16, V1.B16, V4.H8
RET
// func simdElement()
TEXT ·simdElement(SB), NOSPLIT, $0-0
VDUP V31.B[15], V18
VDUP V19.S[3], V18.S4
VDUP V1.D[1], V2.D2
VMOV V13.S[0], R20
VMOV V11.B[11], V16.B[12]
VMOV R20, V21.B[2]
VMOV V2.B16, V4.B16
RET
// func simdTable()
TEXT ·simdTable(SB), NOSPLIT, $0-0
VTBL V22.B16, [V28.B16], V11.B16
VTBL V18.B8, [V17.B16, V18.B16], V22.B8
VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8
RET
// func simdLoadStore()
TEXT ·simdLoadStore(SB), NOSPLIT, $0-0
VLD1 (R2), [V21.B16]
VLD1 (R24), [V18.D1, V19.D1, V20.D1]
VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]
VLD1.P 32(R1), [V2.B16, V3.B16]
VLD1.P 64(R4), [V5.B16, V6.B16, V7.B16, V8.B16]
VLD1R (R1), [V9.B8]
VLD1R (R0), [V0.B16]
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)
VST1 [V14.H4, V15.H4, V16.H4], (R27)
VST1.P [V2.B16], (R1)
VST1.P [V2.B16, V3.B16], 32(R1)
RET
// func simdLiteral()
TEXT ·simdLiteral(SB), NOSPLIT, $0-0
VMOVS $0x80402010, V11
VMOVD $0x8040201008040201, V20
VMOVQ $0x7040201008040201, $0x8040201008040201, V10
RET
+77
View File
@@ -0,0 +1,77 @@
// The legacy SSE gap families: scalar compares and square roots, the Plan 9
// packed spellings, shuffles, lane extracts and inserts, packed integer
// shifts and the octa moves. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func cmporder(a, b *float64) int
TEXT ·cmporder(SB), NOSPLIT, $0-24
MOVQ a+0(FP), SI
MOVQ b+8(FP), DI
MOVSD (SI), X0
MOVSD (DI), X1
ANDNPD X0, X2
ANDNPS X0, X3
COMISD X0, X1
SQRTSD X0, X2
CMPSD X0, X1, $5
MOVL SI, CX
SETPL CL
MOVL CX, ret+16(FP)
RET
// func packed(w *uint64) uint64
TEXT ·packed(SB), NOSPLIT, $0-16
MOVQ w+0(FP), SI
MOVO (SI), X0
MOVOA (SI), X1
PADDL X0, X1
PSUBL X0, X1
PCMPEQL X0, X1
PUNPCKLBW X0, X1
PSHUFL $27, X0, X2
MOVOU X2, (SI)
MOVQ (SI), AX
MOVQ AX, ret+8(FP)
RET
// func lanes(p *byte, buf *byte)
TEXT ·lanes(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
MOVQ buf+8(FP), DI
MOVO (SI), X0
MOVQ SI, AX
PINSRB $1, AX, X0
PINSRW $2, AX, X0
PINSRD $3, AX, X0
PINSRQ $1, AX, X0
PEXTRB $1, X0, AX
PEXTRW $2, X0, AX
PEXTRD $3, X0, AX
PEXTRQ $1, X0, CX
PCMPESTRI $4, X0, X0
MOVB AL, (DI)
MOVOU X0, (SI)
RET
// func shifts(p *uint64)
TEXT ·shifts(SB), NOSPLIT, $0-8
MOVQ p+0(FP), SI
MOVO (SI), X0
MOVO X0, X1
PSLLW $3, X0
PSRLW $1, X1
PSRAW $2, X0
PSLLL $4, X0
PSRLL $5, X1
PSRAL $1, X0
PSLLQ $7, X0
PSRLQ $9, X1
PSLLL X1, X0
PSRLQ X0, X1
PSLLDQ $2, X0
PSRLDQ $4, X1
MOVOU X0, (SI)
MOVOU X1, 16(SI)
RET
+27
View File
@@ -0,0 +1,27 @@
// Legacy SSE octa moves against static (SB) symbols: the load and store
// shapes GOROOT's AES-CTR, AES-GCM and P-256 kernels spell (MOVOU
// bswapMask<>+0(SB), X0 and the reverse), including offsets into the symbol
// and the aligned MOVO pair. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func ssestatic() uint64
TEXT ·ssestatic(SB), NOSPLIT, $0-8
MOVOU bswapMask<>+0(SB), X0
MOVOU bswapMask<>+8(SB), X1
MOVO rodataMask<>+0(SB), X2
PXOR X1, X0
PXOR X2, X0
MOVOU X0, sink<>+0(SB)
MOVOU sink<>+0(SB), X3
PXOR X3, X0
MOVQ X0, AX
MOVQ AX, ret+0(FP)
RET
GLOBL bswapMask<>(SB), RODATA|NOPTR, $16
GLOBL rodataMask<>(SB), RODATA|NOPTR, $16
GLOBL sink<>(SB), NOPTR, $16
+76
View File
@@ -0,0 +1,76 @@
// System, string-primitive and x87 families: flag register moves, the
// serialising instructions, MOVS/STOS, the MXCSR pair, scalar float-to-int
// conversions and FMOVD. Every result is folded back so no instruction is
// dead.
#include "textflag.h"
// func system(x uint64) uint64
TEXT ·system(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
PUSHFQ
POPFQ
CPUID
RDTSC
RDTSCP
SYSCALL
XGETBV
PAUSE
LFENCE
MFENCE
SFENCE
UNDEF
XORQ AX, BX
MOVQ BX, ret+8(FP)
RET
// func stringprim(p *byte, n int) uint64
TEXT ·stringprim(SB), NOSPLIT, $0-24
MOVQ p+0(FP), DI
MOVQ n+8(FP), CX
LEAQ buf<>(SB), AX
MOVQ AX, SI
CLD
MOVSB
MOVSW
MOVSL
MOVSQ
STOSB
STOSQ
STOSL
STOSW
MOVQ DI, ret+16(FP)
RET
DATA buf<>+0x00(SB)/8, $0
GLOBL buf<>(SB), NOPTR, $8
// func intgate(x uint64) uint64
TEXT ·intgate(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
INT $3
MOVQ AX, ret+8(FP)
RET
// func fpmxcsr(x float64, csr *uint32) int64
TEXT ·fpmxcsr(SB), NOSPLIT, $0-24
MOVQ x+0(FP), X0
MOVQ csr+8(FP), AX
STMXCSR (AX)
LDMXCSR (AX)
CVTSD2SL X0, CX
CVTTSD2SQ X0, DX
MOVL (AX), SI
MOVQ SI, ret+8(FP)
RET
// func fmove(p *float64) float64
TEXT ·fmove(SB), NOSPLIT, $0-16
MOVQ p+0(FP), AX
FMOVD (AX), F0
FMOVD F0, F1
FMOVD F0, (AX)
MOVQ (AX), AX
MOVQ AX, ret+8(FP)
RET
+61
View File
@@ -0,0 +1,61 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 system instructions: barriers,
// cache maintenance, the system register accesses, supervisor calls,
// breakpoints and prefetches. Every function is byte-compared against
// go tool asm.
#include "textflag.h"
// func barriers()
TEXT ·barriers(SB), NOSPLIT, $0-0
DMB $15
DMB $1
DSB $15
DSB $4
ISB $15
ISB $1
RET
// func cacheOps()
TEXT ·cacheOps(SB), NOSPLIT, $0-0
DC ZVA, R4
DC IVAC, R1
DC CVAC, R2
DC CVAU, R3
DC CIVAC, R7
RET
// func sysRegs()
TEXT ·sysRegs(SB), NOSPLIT, $0-0
MRS DCZID_EL0, R3
MRS CNTVCT_EL0, R0
MRS CNTPCT_EL0, R1
MRS CNTFRQ_EL0, R2
MRS MIDR_EL1, R0
MRS ID_AA64PFR0_EL1, R0
MRS ID_AA64ISAR0_EL1, R0
MRS ID_AA64ISAR1_EL1, R0
MRS DIT, R0
MSR $3, SPSel
MSR $9, DAIFSet
MSR $6, DAIFClr
MSR $1, DIT
RET
// func exceptions()
TEXT ·exceptions(SB), NOSPLIT, $0-0
SVC $0
SVC $7165
BRK
BRK $35943
RET
// func prefetch()
TEXT ·prefetch(SB), NOSPLIT, $0-0
PRFM (R0), PLDL1KEEP
PRFM (R3), PLDL3KEEP
PRFM (R4), PSTL1KEEP
PRFM (R2), $25
RET
+85
View File
@@ -0,0 +1,85 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the loong64 LSX/LASX slice: every function pairs
// with the same instructions in the go tool asm ground truth.
#include "textflag.h"
TEXT ·threeReg(SB), NOSPLIT, $0-0
VADDV V1, V2, V3
VADDW V1, V2, V3
VADDV V2, V1
VANDV V1, V2, V3
VANDV V1, V2
VXORV V1, V2, V3
VXORV V1, V2
VSEQB V1, V2, V3
VSEQV V1, V2, V3
VSRAB V1, V2, V3
VROTRW V1, V2, V3
VPCNTV V1, V2
XVADDV X1, X2, X3
XVADDV X2, X1
XVANDV X1, X2, X3
XVXORV X1, X2, X3
XVSEQB X1, X2, X3
XVSEQV X1, X2, X3
XVPCNTV X1, X2
RET
TEXT ·immediates(SB), NOSPLIT, $0-0
VANDB $0, V2, V3
VANDB $255, V2
VSEQB $3, V2, V3
VSEQV $15, V2, V3
VSRAB $0, V1, V2
VSRAB $7, V1, V2
VSRAB $6, V1
VROTRW $0, V1, V2
VROTRW $16, V1, V2
VROTRW $16, V1
XVANDB $1, X2, X2
RET
TEXT ·conditions(SB), NOSPLIT, $0-0
VSETNEV V1, FCC0
VSETANYEQB V1, FCC0
VSETANYEQV V2, FCC0
VSETALLNEV V0, FCC0
XVSETNEV X1, FCC0
XVSETANYEQB X1, FCC0
XVSETANYEQV X1, FCC0
XVSETALLNEV X1, FCC0
RET
TEXT ·fpConvert(SB), NOSPLIT, $0-0
FFINTDV F0, F1
FSEL FCC0, F3, F4, F3
FSEL FCC1, F1, F2
RET
TEXT ·memMoves(SB), NOSPLIT, $0-0
VMOVQ V1, V9
VMOVQ (R4), V2
VMOVQ 16(R4), V2
VMOVQ V0, (R4)
VMOVQ V0, 32(R4)
VMOVQ V0,-16(R6)
VMOVQ (R4)(R7), V3
VMOVQ V3, (R4)(R7)
XVMOVQ X3, X7
XVMOVQ (R4), X2
XVMOVQ X0, (R4)
XVMOVQ (R4)(R7), X4
XVMOVQ X0, (R4)(R7)
RET
TEXT ·elements(SB), NOSPLIT, $0-0
VMOVQ R6, V0.B16
VMOVQ R6, V12.W4
XVMOVQ R6, X0.B32
VMOVQ (R4), V4.W4
VMOVQ (R10), V0.W4
XVMOVQ (R4), X0.B32
RET
+53
View File
@@ -0,0 +1,53 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 RVV slice: the instructions GOROOT's
// vector kernels use (crypto/internal/fips140/subtle/xor_riscv64.s,
// internal/bytealg and internal/chacha8rand), spelled as they spell them.
#include "textflag.h"
TEXT ·config(SB), NOSPLIT, $0-0
VSETVLI X5, E8, M8, TA, MA, X6
VSETVLI X11, E8, M8, TA, MA, X5
VSETVLI X12, E8, M8, TA, MA, X5
VSETVLI X13, E8, M8, TU, MU, X15
VSETIVLI $4, E32, M1, TA, MA, X0
VSETIVLI $15, E32, M1, TA, MA, X12
VSETVLI $15, E32, M1, TA, MA, X12
VSETVLI X10, E16, M1, TU, MU, X12
VSETVLI X10, E32, M2, TA, MA, X12
VSETVLI X10, E64, M8, TU, MU, X12
VSETIVLI $31, E32, M1, TA, MA, X12
RET
TEXT ·loadsStores(SB), NOSPLIT, $0-0
VLE8V (X10), V8
VLE8V (X11), V16
VLE8V (X12), V16
VIDV V12
VMV4RV V8, V24
VSE8V V24, (X10)
VSE32V V0, (X11)
VSE32V V8, (X11)
VSE32V V15, (X11)
RET
TEXT ·segmented(SB), NOSPLIT, $0-0
VLSSEG4E32V (X14), X0, V0
VLSSEG8E32V (X10), X0, V4
RET
TEXT ·crypto(SB), NOSPLIT, $0-0
VADDVV V20, V4, V4
VADDVV V27, V11, V11
VADDVX X12, V12, V12
VXORVV V8, V16, V24
VXORVV V13, V13, V13
VMSEQVX X12, V8, V0
VMSNEVV V8, V16, V0
VFIRSTM V0, X6
VFIRSTM V0, X7
VSLLVI $8, V28, V30
VSRLVI $25, V29, V29
RET
+70
View File
@@ -0,0 +1,70 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Wide-immediate arithmetic: every classification band of the ADD/SUB
// immediate family (single imm12, the ADDCON2 split, bitmask and MOVZ/MOVN/
// MOVK materialisations into REGTMP) plus the logical bitmask immediates and
// their materialised fallback. Byte-for-byte against go tool asm.
#include "textflag.h"
// imm12 covers the plain and shifted-by-12 imm12 forms.
TEXT ·imm12(SB), NOSPLIT, $0-0
ADD $1, R2, R3
ADD $0x000aaa, R2, R3
ADD $0xaaa000, R2
SUB $0x000aaa, R2, R3
SUB $0xaaa000, R2
ADDW $40960, R0
CMP $40960, R0
CMPW $40960, R0
RET
// split pins the ADDCON2 band: two imm12 instructions, low half first.
TEXT ·split(SB), NOSPLIT, $0-0
ADD $0xaaaaaa, R2, R3
SUB $0xaaaaaa, R2
ADD $0x186a0, R2, R5
SUB $0x186a0, R2, R3
ADDW $0x60060, R2
RET
// regtmp covers the single-word materialisations: MOVZ for a movcon value,
// MOVN for the complement form, the bitmask ORR otherwise.
TEXT ·regtmp(SB), NOSPLIT, $0-0
ADD $0x1ffe00, R2, R3
ADD $0x3fffffffc000, R5
ADD $-2048, R2, R3
ADD $-100000, R2, R3
CMP $0x1000000, R2
CMP $0x100000000, R0
SUB $-0x100000000, R0, R1
RET
// movseq covers the omovlconst sequences: MOVZ/MOVN ladders and the
// compare forms that never split.
TEXT ·movseq(SB), NOSPLIT, $0-0
ADD $0x12345678, R2, R3
SUB $0xe7791f700, R3, R1
CMP $0xaaaaaa, R2
CMP $0xffffffffffa0, R3
CMPW $27745, R2
CMPW $0x60060, R2
ADDS $0xaaaaaa, R2, R3
CMN $0x1000000, R2
ADDW $0x12345678, R2, R3
RET
// logical covers the bitmask immediates of the logical family and the
// materialised fallback for the values a bitmask cannot carry.
TEXT ·logical(SB), NOSPLIT, $0-0
AND $0x3ff00000, R2, R3
BIC $0x22220000, R3, R4
ORR $0x3ff00000, R2
EOR $0x3ff00000, R2, R3
ANDS $0x3ff00000, R2
ORNW $0x3ff00000, R2
EONW $0x3ff00000, R2
BICSW $0x6006000060060, R5
TST $0x4900000049, R0
RET
+11
View File
@@ -43,6 +43,13 @@ const (
At // @
Hash // #
Pipe // |
// Semicolon separates statements on one line (a Plan 9 statement
// terminator); Ampersand and Tilde are the expression operators & and ~
// of constant expressions. All three appear mostly inside macro bodies.
Semicolon // ;
Ampersand // &
Tilde // ~
)
var kindNames = map[Kind]string{
@@ -71,6 +78,10 @@ var kindNames = map[Kind]string{
At: "@",
Hash: "#",
Pipe: "|",
Semicolon: ";",
Ampersand: "&",
Tilde: "~",
}
// String returns a human-readable name for the kind.
+9
View File
@@ -26,6 +26,15 @@ func TestGroundTruthARM64(t *testing.T) {
"../testdata/verify/bigframe_arm64.s",
"../testdata/verify/guard_arm64.s",
"../testdata/verify/indirect_arm64.s",
"../testdata/verify/exclusive_arm64.s",
"../testdata/verify/shifts_arm64.s",
"../testdata/verify/atomics_arm64.s",
"../testdata/verify/crypto_arm64.s",
"../testdata/verify/integer_arm64.s",
"../testdata/verify/simd_arm64.s",
"../testdata/verify/widenimm_arm64.s",
"../testdata/verify/carryshift_arm64.s",
"../testdata/verify/system_arm64.s",
} {
t.Run(path, func(t *testing.T) {
src, err := os.ReadFile(path)
+13
View File
@@ -115,6 +115,19 @@ func TestGroundTruthAMD64(t *testing.T) {
"../testdata/verify/bigframe_amd64.s",
"../testdata/verify/guard_amd64.s",
"../testdata/verify/indirect_amd64.s",
"../testdata/verify/widen_amd64.s",
"../testdata/verify/scalar_amd64.s",
"../testdata/verify/atomics_amd64.s",
"../testdata/verify/system_amd64.s",
"../testdata/verify/crypto_amd64.s",
"../testdata/verify/sse_amd64.s",
"../testdata/verify/avx_amd64.s",
"../testdata/verify/pfx_amd64.s",
"../testdata/verify/rawdata_amd64.s",
"../testdata/verify/pfx_amd64.s",
"../testdata/verify/rawdata_amd64.s",
"../testdata/verify/doubleshift_amd64.s",
"../testdata/verify/ssestatic_amd64.s",
} {
t.Run(path, func(t *testing.T) {
f, errs := parser.Parse(path, mustRead(t, path))
+6
View File
@@ -25,6 +25,12 @@ func TestGroundTruthLOONG64(t *testing.T) {
"../testdata/verify/bigframe_loong64.s",
"../testdata/verify/guard_loong64.s",
"../testdata/verify/indirect_loong64.s",
"../testdata/verify/movwfp_loong64.s",
"../testdata/verify/branchu_loong64.s",
"../testdata/verify/atomics_loong64.s",
"../testdata/verify/vector_loong64.s",
"../testdata/verify/pcalign_loong64.s",
"../testdata/verify/l64forms_loong64.s",
"trampoline_loong64.s",
} {
t.Run(path, func(t *testing.T) {
+6
View File
@@ -29,6 +29,12 @@ func TestGroundTruthRISCV(t *testing.T) {
"../testdata/verify/guard_riscv64.s",
"../testdata/verify/indirect_riscv64.s",
"../testdata/verify/misc_riscv64.s",
"../testdata/verify/rvcstore_riscv64.s",
"../testdata/verify/atomics_riscv64.s",
"../testdata/verify/vector_riscv64.s",
"../testdata/verify/bitmanip_riscv64.s",
"../testdata/verify/pcalign_riscv64.s",
"../testdata/verify/branch_far_riscv64.s",
"trampoline_riscv64.s",
} {
t.Run(path, func(t *testing.T) {