Compare commits

...
5 Commits
Author SHA1 Message Date
petrbalvin 1529eba9ce docs: changelog and readme for the instruction wave and the honest corpus rate
Test / test (push) Successful in 2m12s
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 06:45:03 +02:00
petrbalvin 9cbd31e00b feat(audit): probe the new operand shapes and measure attemptable files
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 06:45:03 +02:00
petrbalvin 924e0013eb feat(riscv64,loong64): encode AMO atomics, vector slices and bit ops
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 06:44:51 +02:00
petrbalvin 56376a0e59 feat(arm64): encode pairs, atomics, crypto, system and NEON slices
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 06:44:51 +02:00
petrbalvin 70eb8fb8bd feat(amd64): encode the GOROOT instruction families
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 06:44:51 +02:00
44 changed files with 6290 additions and 127 deletions
+23
View File
@@ -9,6 +9,29 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
- **The GOROOT instruction wave, part 1.** The encoder now covers the
instruction families GOROOT's real code uses that gasm lacked,
byte-verified against `go tool asm`: on amd64 the carry ALU, the
atomics (CMPXCHG, XADD, XCHG), AES-NI, SHA-1/256, PCLMULQDQ, CRC32,
GFNI, ADX, BMI, the string primitives, the system set (CPUID, RDTSC,
SYSCALL, fences, MXCSR) and the SSE/AVX/EVEX gaps; on arm64 the pair
loads and stores (LDP/STP), acquire/release and LSE atomics, AES and
SHA, the system operations, the bit ops and the NEON slice including
structure loads and the literal-pool moves; on riscv64 the RV64A AMO
family with aq/rl ordering, the Zbb pseudos with their RVC
compressions, the FMA forms and the RVV slice with `vsetvli`/
`vsetivli`; on loong64 the AM atomics with acquire/release forms, the
LSX/LASX slice, the `VMOVQ`/`XVMOVQ` transfer family and FSEL.
Also fixed on the way: arm64 `CASD`/`CASW` lacked an opcode bit, and
riscv64 `VSETVLI` with an immediate length now canonicalises to
`vsetivli` as the toolchain does.
- **The corpus audit measures honestly.** Files named for Go ports gasm
does not target (arm, 386, s390x, ...) are no longer attempted for the
four supported architectures (no supported build compiles them), and
the headline rate is reported over attemptable files: 136 of 433 on
the full corpus (31.4 %), 135 of 383 on real code (35.2 %), from the
127 that the previous release measured. The probe battery that
decides encodability gained the operand shapes the new families use.
-
## [0.34.0] - 2026-09-20
+3 -2
View File
@@ -124,8 +124,9 @@ can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 127 of 627 files (20.3 %)
assembling for every target architecture today, with the top failure
`gasm audit-instructions --corpus` reports 136 of 433 attemptable files
(31.4 %) assembling for every target architecture today (files named for
other Go ports are counted but never attempted), with the top failure
reasons per architecture; the number moves with every release.
### Validation status
+4
View File
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
}
// x87 stack registers (FMOVD and the other x87 moves).
for i := 0; i <= 7; i++ {
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
}
return regs
}
+27
View File
@@ -150,6 +150,33 @@ func arm64Curated() []Instr {
t = append(t, i(op, "Atomic memory operation"))
}
// Register-pair loads and stores.
for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} {
t = append(t, ic(op, "Register-pair load or store", 2, 2))
}
// Cache maintenance and prefetch.
t = append(t, i("DC", "Data cache maintenance"))
t = append(t, i("PRFM", "Memory prefetch"))
for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} {
t = append(t, i(op, "Atomic memory operation with acquire and release semantics"))
}
// Cryptographic extensions.
for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} {
t = append(t, i(op, "AES round"))
}
for _, op := range []string{
"SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1",
"SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1",
"SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1",
} {
t = append(t, i(op, "SHA round"))
}
for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} {
t = append(t, i(op, "Three-way XOR / rotate crypto vector operation"))
}
// Floating-point scalar.
for _, op := range []string{
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
+1139 -19
View File
File diff suppressed because it is too large Load Diff
+423 -6
View File
@@ -27,7 +27,13 @@ package asm
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
import "maps"
import (
"maps"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
@@ -288,7 +294,25 @@ const (
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
a64FCondCmp // conditional compare: CCMP, CCMN
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
a64FVTBL // SIMD table lookup: VTBL
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
@@ -622,10 +646,403 @@ func init() {
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
// ---- SIMD basics ----
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
// ---- SIMD: the arrangement-aware tables in this file carry VADD,
// VSUB, VMUL and every other three-register vector op. ----
// ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ----
dp1 := map[string]uint32{
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
}
for m, op := range dp1 {
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
}
// ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ----
a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000}
a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000}
a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000}
a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000}
// ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ----
a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000}
a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000}
a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000}
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
// ---- compare/test and branch ----
a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000}
a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000}
a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000}
a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000}
a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000}
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00}
a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00}
a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00}
a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00}
a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00}
a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00}
a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00}
// ---- LSE atomics with acquire and release semantics ----
// CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/
// LDOR/SWP families leave Rs free for the returned value.
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
"SWPALB": 0x38e08000,
"SWPALW": 0xb8e08000,
"SWPALD": 0xf8e08000,
}
for m, op := range lse {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
dpsrExtra := map[string]uint32{
"ADC": 0x9a000000, "ADCW": 0x1a000000,
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
"SBC": 0xda000000, "SBCW": 0x5a000000,
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
}
for m, op := range dpsrExtra {
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
}
// ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ----
crypto2 := map[string]uint32{
"AESD": 0x4e285800, "AESE": 0x4e284800,
"AESIMC": 0x4e287800, "AESMC": 0x4e286800,
"SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800,
"SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000,
}
for m, op := range crypto2 {
a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op}
}
crypto3 := map[string]uint32{
"SHA1C": 0x5e000000, "SHA1P": 0x5e001000,
"SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000,
"SHA256H": 0x5e004000, "SHA256H2": 0x5e005000,
"SHA256SU1": 0x5e006000, "SHA512H": 0xce608000,
"SHA512H2": 0xce608400, "SHA512SU1": 0xce608800,
}
for m, op := range crypto3 {
a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op}
}
// ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ----
a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000}
a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000}
a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000}
a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000}
a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL}
a64InstrTable["VDUP"] = a64Enc{format: a64FDUP}
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
// for instructions that exist at a single arrangement and carry that
// arrangement's bits inside the base already, the fixed flag.
type a64SimdVSpec struct {
base uint32
arrs uint16
fixed bool
}
// a64Arr names the vector arrangements the encoders deal with, indexed by
// a64Arr. The source spellings put the element letter first: B8, H4, S2,
// D1 and the 128-bit halves B16, H8, S4, D2.
const (
a64Arr8B = iota
a64Arr16B
a64Arr4H
a64Arr8H
a64Arr2S
a64Arr4S
a64Arr2D
a64ArrD1
a64ArrQ1
a64ArrCount
)
// a64ArrNames maps an arrangement to its source spelling (element letter
// first, as the toolchain writes it).
var a64ArrNames = [a64ArrCount]string{
a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8",
a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1",
}
// a64ArrIndex resolves a source spelling to its a64Arr index, -1 when
// unknown.
func a64ArrIndex(s string) int {
for i, n := range a64ArrNames {
if n == s {
return i
}
}
return -1
}
// a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q)
// as it appears in element operands such as V13.S[0].
func a64ElemLetter(s string) bool {
switch s {
case "B", "H", "S", "D", "Q":
return true
}
return false
}
// a64ArrBits carries the fixed bits an arrangement contributes to the
// three-same word shape: the element size at bits 23:22 and the 128-bit
// flag at bit 30. Bit 29 belongs to the instruction's own base.
var a64ArrBits = [a64ArrCount]uint32{
a64Arr8B: 0,
a64Arr16B: 1 << 30,
a64Arr4H: 1 << 22,
a64Arr8H: 1<<30 | 1<<22,
a64Arr2S: 1 << 23,
a64Arr4S: 1<<30 | 1<<23,
a64Arr2D: 1<<30 | 1<<23 | 1<<22,
a64ArrD1: 1<<23 | 1<<22,
a64ArrQ1: 0,
}
// a64SimdVTable holds the arrangement-aware three-register SIMD
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
}
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
// carry when they spell one at all; a bare V/F spelling is accepted as is.
var a64CryptoArr = map[string]int{
"AESD": a64Arr16B, "AESE": a64Arr16B, "AESIMC": a64Arr16B, "AESMC": a64Arr16B,
"SHA1H": a64Arr4S, "SHA1SU1": a64Arr4S, "SHA256SU0": a64Arr4S, "SHA512SU0": a64Arr2D,
"SHA1C": a64Arr4S, "SHA1P": a64Arr4S, "SHA1M": a64Arr4S, "SHA1SU0": a64Arr4S,
"SHA256H": a64Arr4S, "SHA256H2": a64Arr4S, "SHA256SU1": a64Arr4S,
"SHA512H": a64Arr2D, "SHA512H2": a64Arr2D, "SHA512SU1": a64Arr2D,
}
// a64DCOps maps the data-cache maintenance operation names to their fixed
// word (the register rides bits 4:0).
var a64DCOps = map[string]uint32{
"IVAC": 0xd5087620, "ZVA": 0xd50b7420,
"CVAC": 0xd50b7a20, "CVAU": 0xd50b7b20, "CIVAC": 0xd50b7e20,
}
// a64MRSOps maps the system register names GOROOT reads to their fixed word
// (the destination register rides bits 4:0).
var a64MRSOps = map[string]uint32{
"ELR_EL1": 0xd5384020, "MIDR_EL1": 0xd5380000,
"ID_AA64PFR0_EL1": 0xd5380400, "ID_AA64ISAR0_EL1": 0xd5380600,
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
}
// a64MSROps maps the system register names GOROOT writes to their fixed
// word; the immediate rides CRm at bits 11:8 and Rt is the fixed 11111.
var a64MSROps = map[string]uint32{
"SPSel": 0xd50040a0, "DAIFSet": 0xd50340c0, "DAIFClr": 0xd50340e0, "DIT": 0xd5034040,
}
// a64PRFOps maps the prefetch operation names to their prfop immediate
// (word = 0xf9800000 | Rn<<5 | prfop).
var a64PRFOps = map[string]int{
"PLDL1KEEP": 0x00, "PLDL1STRM": 0x01, "PLDL2KEEP": 0x02, "PLDL2STRM": 0x03,
"PLDL3KEEP": 0x04, "PLDL3STRM": 0x05,
"PLIL1KEEP": 0x08, "PLIL1STRM": 0x09, "PLIL2KEEP": 0x0a, "PLIL2STRM": 0x0b,
"PLIL3KEEP": 0x0c, "PLIL3STRM": 0x0d,
"PSTL1KEEP": 0x10, "PSTL1STRM": 0x11, "PSTL2KEEP": 0x12, "PSTL2STRM": 0x13,
"PSTL3KEEP": 0x14, "PSTL3STRM": 0x15,
}
// a64VLD1Base holds the fixed words of the multi-register structure
// accesses, indexed by register count 1..4, before the Q and size bits.
// Post-index spellings add 0x9f0000 (post bit and Rm = 11111).
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
reg int
arr string
idx int
hasIdx bool
}
// a64VecReg parses a vector register operand: V0..V31 (F0..F31 as an alias,
// the same architectural registers the scalar floating-point spellings use),
// optionally with an arrangement suffix such as V0.B16 and, for element
// forms, a lane index such as V13.S[0]. It reports ok=false for anything
// else, including X/W and R spellings, which the toolchain's vector
// operands reject as well.
func a64VecReg(name string) (v a64Vec, ok bool) {
s := strings.TrimSpace(name)
if i := strings.IndexByte(s, '.'); i >= 0 {
v.arr = strings.TrimSpace(s[i+1:])
s = s[:i]
}
if v.arr != "" {
// Element form: B[3], S[2] and friends.
if j := strings.IndexByte(v.arr, '['); j >= 0 {
k := strings.LastIndexByte(v.arr, ']')
if k < j {
return v, false
}
n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k]))
if err != nil || n < 0 {
return v, false
}
v.idx, v.hasIdx = n, true
v.arr = strings.TrimSpace(v.arr[:j])
}
if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) {
return v, false
}
}
if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') {
return v, false
}
n := 0
for i := 1; i < len(s); i++ {
if s[i] < '0' || s[i] > '9' {
return v, false
}
n = n*10 + int(s[i]-'0')
}
if n > 31 {
return v, false
}
v.reg = n
return v, true
}
// a64ElemField encodes a lane index for the copy/insert group: imm5 = the
// index shifted by the element scale, with the scale's own bit set. B gets
// shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4.
func a64ElemField(arr string, idx int) (uint32, bool) {
var shift, low uint32
switch arr {
case "B8", "B16", "B":
shift, low = 1, 1
case "H4", "H8", "H":
shift, low = 2, 2
case "S2", "S4", "S":
shift, low = 3, 4
case "D1", "D2", "D":
shift, low = 4, 8
default:
return 0, false
}
if idx < 0 || idx >= 1<<(5-shift) {
return 0, false
}
return uint32(idx)<<shift | low, true
}
// a64VecListOf recovers the register list of a VLD1/VST1/VTBL operand run.
// The parser keeps parenthesised groups whole but splits bracketed lists on
// the commas, so a list arrives as one operand run whose first Raw starts
// with "[" and whose last Raw ends with "]". It returns the parsed
// registers with the brackets and spaces removed.
func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool) {
if start >= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") {
return nil, 0, false
}
end = start
for end < len(ops) {
if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") {
break
}
end++
}
if end >= len(ops) {
return nil, 0, false
}
for i := start; i <= end; i++ {
s := strings.TrimSpace(ops[i].Raw)
s = strings.TrimPrefix(s, "[")
s = strings.TrimSuffix(s, "]")
if s == "" && len(ops) > start+1 {
return nil, 0, false
}
for part := range strings.SplitSeq(s, ",") {
v, ok := a64VecReg(part)
if !ok {
return nil, 0, false
}
vs = append(vs, v)
}
}
return vs, end, true
}
// ---- load/store helper tables ----
+449 -5
View File
@@ -472,12 +472,456 @@ TEXT ·f(SB), NOSPLIT, $0-0
}
}
// TestArm64SIMD tests SIMD encoding (via the instruction table).
// TestArm64SIMD tests SIMD encoding (via the arrangement-aware table).
func TestArm64SIMD(t *testing.T) {
// Verify SIMD instructions are in the table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
if _, ok := a64InstrTable[mnem]; !ok {
t.Errorf("%s not in instruction table", mnem)
// Verify SIMD instructions are in the arrangement table.
for _, mnem := range []string{"VADD", "VSUB", "VMUL", "VAND", "VEOR", "VORR", "VCMEQ", "VZIP1", "VZIP2"} {
if _, ok := a64SimdVTable[mnem]; !ok {
t.Errorf("%s not in the SIMD arrangement table", mnem)
}
}
}
// TestArm64CarryAndBitOps pins the carry-setting arithmetic, the widening
// multiplies and the data-processing (1 source) group against go tool asm.
func TestArm64CarryAndBitOps(t *testing.T) {
got := arm64Words(t, "\tADC R0, R2, R12\n\tADCS $0, R1\n\tSBCS R5, R9, R5\n\tSBC R25, R10, R26\n"+
"\tMUL R4, R3, R0\n\tUMULH R24, R20, R24\n\tSMULH R1, R2, R3\n\tMSUB R19, R16, R26, R2\n"+
"\tRBIT R11, R4\n\tREV R1, R2\n\tCLZ R21, R9\n\tREVW R1, R2\n\tCLSW R1, R2\n")
want := []uint32{
0x9a00004c, // ADC R12, R2, R0
0xba1f0021, // ADCS R1, R1, ZR
0xfa050125, // SBCS R5, R9, R5
0xda19015a, // SBC R26, R10, R25
0x9b047c60, // MUL R0, R3, R4
0x9bd87e98, // UMULH R24, R20, R24
0x9b417c43, // SMULH R3, R2, R1
0x9b13c342, // MSUB R2, R26, R19, R16
0xdac00164, // RBIT R4, R11
0xdac00c22, // REV R2, R1
0xdac012a9, // CLZ R9, R21
0x5ac00822, // REVW R2, R1
0x5ac01422, // CLSW R2, R1
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64BitfieldExtract pins UBFX/SBFX: immr wraps to the register
// width, an out-of-range imms is an error.
func TestArm64BitfieldExtract(t *testing.T) {
got := arm64Words(t, "\tUBFX $33, R17, $25, R5\n\tUBFXW $4, R1, $9, R2\n")
want := []uint32{
0xd361e625, // UBFX immr=1 (33 wrapped), imms=25
0x53043022, // UBFXW immr=4, imms=9
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
for _, body := range []string{"\tUBFX $33, R17, $70, R5\n", "\tUBFX $-1, R17, $3, R5\n"} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
}
// TestArm64CondCompare pins CCMP/CCMN.
func TestArm64CondCompare(t *testing.T) {
got := arm64Words(t, "\tCCMP LE, R7, $19, $3\n\tCCMP LT, R30, R6, $7\n\tCCMN EQ, R1, R2, $3\n\tCCMPW LE, R7, $19, $3\n")
want := []uint32{
0xfa53d8e3, // CCMP imm form
0xfa46b3c7, // CCMP register form
0xba420023, // CCMN register form
0x7a53d8e3, // CCMPW
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64CompareBranch pins CBZ/CBNZ/TBZ/TBNZ against a label five and
// six words ahead, matching go tool asm's own offsets.
func TestArm64CompareBranch(t *testing.T) {
// Layout: CBZ(0) TBZ(4) TBNZ(8) CBNZ(12) NOP(16) NOP(17th word...) done.
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tCBZ R1, done\n\tTBZ $4, R7, done\n\tTBNZ $33, R7, done\n\tCBNZW R2, done\n" +
"\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
// done sits at word 6 from each branch's own pc: CBZ rel 6, TBZ rel 5,
// TBNZ rel 4, CBNZW rel 3.
want := []uint32{
0xb40000c1, // CBZ R1, +6
0x362000a7, // TBZ $4, R7, +5
0xb7080087, // TBNZ $33, R7, +4
0x35000062, // CBNZW R2, +3
0xd503201f, 0xd503201f, 0xd503201f,
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64ADR pins ADR against a forward label.
func TestArm64ADR(t *testing.T) {
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tADR done, R10\n\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
// rel = 12 bytes: immlo 0, immhi 3.
want := []uint32{0x1000006a, 0xd503201f, 0xd503201f, 0xd503201f, 0xd65f03c0}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64PairLoadStore pins LDP/STP/LDPW/FLDPD/FSTPD.
func TestArm64PairLoadStore(t *testing.T) {
got := arm64Words(t, "\tSTP (R2, R3), 8(R5)\n\tLDP -8(R5), (R2, R3)\n\tLDPW 4(R0), (R1, R2)\n\tSTPW (R1, R2), 4(R0)\n"+
"\tFLDPD 8(R0), (F1, F2)\n\tFSTPD (F3, F4), -8(R5)\n")
want := []uint32{
0xa9008ca2, // STP (R2, R3), 8(R5)
0xa97f8ca2, // LDP -8(R5), (R2, R3)
0x29408801, // LDPW 4(R0), (R1, R2)
0x29008801, // STPW (R1, R2), 4(R0)
0x6d408801, // FLDPD 8(R0), (F1, F2)
0x6d3f90a3, // FSTPD (F3, F4), -8(R5)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AcquireRelease pins LDAR/STLR and the acquire/release LSE
// families.
func TestArm64AcquireRelease(t *testing.T) {
got := arm64Words(t, "\tLDAR (R27), R22\n\tLDARB (R25), R2\n\tLDARW (R12), R29\n\tSTLR R3, (R24)\n\tSTLRB R11, (R22)\n"+
"\tCASALD R5, (R6), R7\n\tLDADDALD R5, (R6), R7\n\tLDCLRALB R5, (R6), R7\n\tLDORALD R5, (RSP), R7\n\tSWPALW R5, (R6), R7\n")
want := []uint32{
0xc8dfff76, // LDAR R22, (R27)
0x08dfff22, // LDARB R2, (R25)
0x88dffd9d, // LDARW R29, (R12)
0xc89fff03, // STLR R3, (R24)
0x089ffecb, // STLRB R11, (R22)
0xc8e5fcc7, // CASALD R7, (R6), R5
0xf8e500c7, // LDADDALD R7, (R6), R5
0x38e510c7, // LDCLRALB R7, (R6), R5
0xf8e533e7, // LDORALD R7, (RSP), R5
0xb8e580c7, // SWPALW R7, (R6), R5
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64System pins BRK, SVC, the barriers, cache maintenance and the
// system register accesses.
func TestArm64System(t *testing.T) {
got := arm64Words(t, "\tBRK $35943\n\tBRK\n\tSVC $7165\n\tDMB $1\n\tDSB $1\n\tISB $15\n"+
"\tDC ZVA, R4\n\tDC IVAC, R1\n\tMRS DCZID_EL0, R3\n\tMRS CNTVCT_EL0, R0\n\tMSR $9, DAIFSet\n\tMSR $3, SPSel\n"+
"\tPRFM (R0), PLDL1KEEP\n\tPRFM (R3), PLDL3KEEP\n\tPRFM (R2), $25\n")
want := []uint32{
0xd4318ce0, // BRK $35943
0xd4200000, // BRK
0xd4037fa1, // SVC $7165
0xd50331bf, // DMB $1
0xd503319f, // DSB $1
0xd5033fdf, // ISB $15
0xd50b7424, // DC ZVA, R4
0xd5087621, // DC IVAC, R1
0xd53b00e3, // MRS DCZID_EL0, R3
0xd53be040, // MRS CNTVCT_EL0, R0
0xd50349df, // MSR $9, DAIFSet
0xd50043bf, // MSR $3, SPSel
0xf9800000, // PRFM (R0), PLDL1KEEP
0xf9800064, // PRFM (R3), PLDL3KEEP
0xf9800059, // PRFM (R2), $25
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64Crypto pins the AES and SHA families.
func TestArm64Crypto(t *testing.T) {
got := arm64Words(t, "\tAESE V31.B16, V29.B16\n\tAESD V22.B16, V19.B16\n\tAESIMC V12.B16, V27.B16\n\tAESMC V14.B16, V28.B16\n"+
"\tSHA1C V8.S4, V8, V2\n\tSHA1H V17, V25\n\tSHA1P V3.S4, V20, V27\n\tSHA1SU0 V17.S4, V13.S4, V16.S4\n\tSHA1SU1 V24.S4, V23.S4\n"+
"\tSHA256H V4.S4, V2, V11\n\tSHA256H2 V6.S4, V16, V11\n\tSHA256SU0 V0.S4, V16.S4\n\tSHA256SU1 V31.S4, V3.S4, V15.S4\n"+
"\tSHA512H V2.D2, V1, V0\n\tSHA512H2 V4.D2, V3, V2\n\tSHA512SU0 V9.D2, V8.D2\n\tSHA512SU1 V7.D2, V6.D2, V5.D2\n")
want := []uint32{
0x4e284bfd, // AESE
0x4e285ad3, // AESD
0x4e28799b, // AESIMC
0x4e2869dc, // AESMC
0x5e080102, // SHA1C
0x5e280a39, // SHA1H
0x5e03129b, // SHA1P
0x5e1131b0, // SHA1SU0
0x5e281b17, // SHA1SU1
0x5e04404b, // SHA256H
0x5e06520b, // SHA256H2
0x5e282810, // SHA256SU0
0x5e1f606f, // SHA256SU1
0xce628020, // SHA512H
0xce648462, // SHA512H2
0xcec08128, // SHA512SU0
0xce6788c5, // SHA512SU1
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLogical pins the arrangement-aware three- and two-register
// SIMD paths.
func TestArm64SIMDLogical(t *testing.T) {
got := arm64Words(t, "\tVADD V1.B16, V2.B16, V3.B16\n\tVAND V4.B16, V4.B16, V9.B16\n\tVEOR V0.B16, V1.B16, V0.B16\n"+
"\tVORR V5.B16, V4.B16, V3.B16\n\tVADDP V1.H8, V2.H8, V3.H8\n\tVZIP1 V16.H8, V3.H8, V19.H8\n\tVZIP2 V22.D2, V25.D2, V21.D2\n"+
"\tVCMEQ V24.S4, V13.S4, V12.S4\n\tVCMEQ $0, V2.H4, V3.H4\n\tVREV32 V2.H8, V1.H8\n\tVREV64 V2.S4, V3.S4\n\tVUADDLV V31.S4, V11\n"+
"\tVPMULL V2.D1, V1.D1, V3.Q1\n\tVPMULL2 V2.B16, V1.B16, V4.H8\n\tVRAX1 V26.D2, V29.D2, V30.D2\n\tVMOV V2.B16, V4.B16\n")
want := []uint32{
0x4e218443, // VADD 16B
0x4e241c89, // VAND
0x6e201c20, // VEOR
0x4ea51c83, // VORR
0x4e61bc43, // VADDP 8H
0x4e503873, // VZIP1 8H
0x4ed67b35, // VZIP2 2D
0x6eb88dac, // VCMEQ 4S
0x0e609843, // VCMEQ $0, 4H
0x6e600841, // VREV32 8H
0x4ea00843, // VREV64 4S
0x6eb03beb, // VUADDLV 4S
0x0ee2e023, // VPMULL D1
0x4e22e024, // VPMULL2 16B
0xce7a8fbe, // VRAX1 2D
0x4ea21c44, // VMOV 16B pair
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDWide pins the four-register crypto group, VXAR, VEXT and the
// shift-by-immediate encodings.
func TestArm64SIMDWide(t *testing.T) {
got := arm64Words(t, "\tVEOR3 V2.B16, V7.B16, V12.B16, V25.B16\n\tVBCAX V1.B16, V2.B16, V26.B16, V31.B16\n"+
"\tVXAR $63, V27.D2, V21.D2, V26.D2\n\tVEXT $4, V2.B8, V1.B8, V3.B8\n\tVEXT $8, V2.B16, V1.B16, V3.B16\n"+
"\tVSHL $7, V22.D2, V25.D2\n\tVUSHR $6, V22.H8, V23.H8\n\tVSRI $24, V1.S4, V2.S4\n")
want := []uint32{
0xce070999, // VEOR3
0xce22075f, // VBCAX
0xce9bfeba, // VXAR
0x2e022023, // VEXT B8
0x6e024023, // VEXT B16
0x4f4756d9, // VSHL D2 $7
0x6f1a06d7, // VUSHR H8 $6
0x6f284422, // VSRI S4 $24
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDElement pins VDUP and the VMOV element forms.
func TestArm64SIMDElement(t *testing.T) {
got := arm64Words(t, "\tVDUP V31.B[15], V18\n\tVDUP V19.S[3], V18.S4\n\tVDUP V1.D[1], V2.D2\n"+
"\tVMOV V13.S[0], R20\n\tVMOV V11.B[11], V16.B[12]\n\tVMOV R20, V21.B[2]\n")
want := []uint32{
0x5e1f07f2, // VDUP element to register
0x4e1c0672, // VDUP element across S4
0x4e180422, // VDUP element across D2
0x0e043db4, // VMOV element to register
0x6e195d70, // VMOV element to element
0x4e051e95, // VMOV register into element
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64SIMDLoadStore pins the structure loads and stores.
func TestArm64SIMDLoadStore(t *testing.T) {
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
"\tVLD1.P 32(R1), [V2.B16, V3.B16]\n\tVST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)\n\tVST1.P [V2.B16], (R1)\n"+
"\tVLD1R (R1), [V9.B8]\n\tVLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]\n")
want := []uint32{
0x4c407055, // VLD1 one register
0x4c40a022, // VLD1 two registers
0x0c402fae, // VLD1 four registers D1
0x4cdfa022, // VLD1.P two registers
0x4c0029c2, // VST1 four registers S4
0x4c9f7022, // VST1.P one register
0x0d40c029, // VLD1R
0x0d60e000, // VLD4R
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64MoviLiteral pins the VMOVS/VMOVD/VMOVQ constant loads: three
// words each (ADRP, ADD, wide load) plus the pooled literal in the data
// section.
func TestArm64MoviLiteral(t *testing.T) {
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
"\tVMOVS $0x80402010, V11\n\tVMOVD $0x8040201008040201, V20\n" +
"\tVMOVQ $0x7040201008040201, $0x8040201008040201, V10\n\tRET\n"
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if img.Funcs[0].Size != 12*3+4 {
t.Errorf("func size = %d, want %d", img.Funcs[0].Size, 12*3+4)
}
want := []uint32{
0x9000001b, 0x9100037b, 0xbd40036b, // VMOVS: ADRP, ADD, LDR S
0x9000001b, 0x9100037b, 0xfd400374, // VMOVD: ADRP, ADD, LDR D
0x9000001b, 0x9100037b, 0x3dc0036a, // VMOVQ: ADRP, ADD, LDR Q
0xd65f03c0,
}
got := leWords(img.Code)
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
// The literals sit in the data section.
var found32, found64, found128 bool
for _, d := range img.DataSyms {
switch d.Name {
case "$i32.80402010":
found32 = d.Size == 4
case "$i64.8040201008040201":
found64 = d.Size == 8
case "$i128.80402010080402017040201008040201":
found128 = d.Size == 16
}
}
if !found32 || !found64 || !found128 {
t.Errorf("literals missing: i32=%v i64=%v i128=%v", found32, found64, found128)
}
}
// TestArm64MOVK pins standalone MOVK with the hw field derived from the
// chunk position.
func TestArm64MOVK(t *testing.T) {
got := arm64Words(t, "\tMOVK $1234, R5\n\tMOVK $305397760, R5\n\tMOVKW $1234, R5\n")
want := []uint32{
0xf2809a45, // MOVK hw=0
0xf2a24685, // MOVK hw=1
0x72809a45, // MOVKW hw=0
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
+32 -8
View File
@@ -18,7 +18,11 @@ func Encodable(mnemonic string) bool {
// Fixed-name instructions (no size suffix).
switch upper {
case "RET", "NOP", "CALL", "JMP":
case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2":
return true
}
if _, ok := noOperandTable[upper]; ok {
return true
}
if _, ok := condCode(upper); ok {
@@ -31,7 +35,7 @@ func Encodable(mnemonic string) bool {
return false
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" {
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return true
}
@@ -53,13 +57,28 @@ func Encodable(mnemonic string) bool {
}
}
// Legacy SSE shuffles and packed binaries dispatch on the full name.
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
// do the imm8-controlled instructions, the lane extracts and inserts and
// the packed integer shifts (their trailing width letters belong to the
// mnemonic).
if _, ok := sseShufTable[upper]; ok {
return true
}
if _, ok := sseBinTable[upper]; ok {
return true
}
if _, ok := sseImm3Table[upper]; ok {
return true
}
if _, ok := sseExtractTable[upper]; ok {
return true
}
if _, ok := sseInsertTable[upper]; ok {
return true
}
if _, ok := sseShiftImm[upper]; ok {
return true
}
// The size-suffix split: retry the tables and the scalar switch on the
// base.
@@ -74,12 +93,15 @@ func Encodable(mnemonic string) bool {
}
}
switch base2 {
case "MOV",
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
case "MOV", "MOVD",
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
"TEST",
"LEA",
"INC", "DEC", "NEG", "NOT",
"SHL", "SHR", "SAR",
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
"BT", "BTS", "BTR", "BTC",
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
"MOVS", "STOS",
"IMUL", "IMUL3",
"PUSH", "POP",
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
@@ -88,7 +110,9 @@ func Encodable(mnemonic string) bool {
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
"CVTSL2SD", "CVTSQ2SD",
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
"FMOVD",
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return true
}
// Full-name dispatches the size split would eat (a trailing width
+81 -6
View File
@@ -58,6 +58,41 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if cc, ok := condCode(upper); ok {
return e.encodeJcc(cc, ops)
}
// No-operand system and string-control instructions (CPUID, RDTSC,
// SYSCALL, the fences, UNDEF, …).
if op, ok := noOperandTable[upper]; ok {
if len(ops) != 0 {
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
}
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper {
case "POPFQ":
if len(ops) != 0 {
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
}
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
case "PUSHFQ":
if len(ops) != 0 {
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
}
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT":
return e.encodeInt(ops)
case "LDMXCSR":
return e.encodeMxcsr(2, ops)
case "STMXCSR":
return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm).
case "CMPSD":
return e.encodeCmpsd(ops)
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2":
return e.encodeSha256rnds2(ops)
}
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
@@ -67,7 +102,8 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if err != nil {
return err
}
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
return e.encodeVec(base, ops, sfx)
}
if sfx.any() {
@@ -101,6 +137,21 @@ func (e *enc) encode(mnem string, ops []Operand) error {
if m, ok := sseBinTable[base]; ok {
return e.encodeSSEBin(m, ops)
}
// The imm8-controlled legacy instructions, the lane extracts and inserts
// and the packed integer shifts all dispatch on the full name: a trailing
// width letter here belongs to the mnemonic, not to the size split.
if m, ok := sseImm3Table[upper]; ok {
return e.encodeSSEImm3(m, ops)
}
if m, ok := sseExtractTable[upper]; ok {
return e.encodeSSEExtract(m, ops)
}
if m, ok := sseInsertTable[upper]; ok {
return e.encodeSSEInsert(m, ops)
}
if _, ok := sseShiftImm[upper]; ok {
return e.encodeSSEShift(upper, ops)
}
// PMOVMSKB ends in a width letter the size split would eat, so it
// dispatches on the full name like the packed binaries above.
if upper == "PMOVMSKB" {
@@ -109,16 +160,36 @@ func (e *enc) encode(mnem string, ops []Operand) error {
switch base {
case "MOV":
return e.encodeMov(ops, size)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
// REX.W and all.
case "MOVD":
return e.encodeMov(ops, 8)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
return e.encodeALU(aluOp[base], ops, size)
case "TEST":
return e.encodeTest(ops, size)
case "LEA":
return e.encodeLea(ops, size)
case "INC", "DEC", "NEG", "NOT":
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
return e.encodeUnary(unaryOp[base], ops, size)
case "SHL", "SHR", "SAR":
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
return e.encodeShift(shiftOp[base], ops, size)
case "BT", "BTS", "BTR", "BTC":
return e.encodeBitTest(base, ops, size)
case "XCHG":
return e.encodeExchange(ops, size)
case "CMPXCHG":
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
case "XADD":
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
case "CRC32":
return e.encodeCrc32(ops, size)
case "ADCX":
return e.encodeCarryExt(0x66, ops, size)
case "ADOX":
return e.encodeCarryExt(0xF3, ops, size)
case "MOVS", "STOS":
return e.encodeStringOp(base, ops, size)
case "IMUL", "IMUL3":
return e.encodeImul(ops, size)
case "PUSH":
@@ -136,7 +207,11 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.encodeMovExtend(base, ops)
case "CVTSL2SD", "CVTSQ2SD":
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
return e.encodeCvtInt(base, ops, size)
case "FMOVD":
return e.encodeFmov(ops)
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
return e.encodeSSEMove(sseMoveTable[base], ops)
}
return fmt.Errorf("unsupported instruction %q", mnem)
@@ -195,7 +270,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
if ss, ok := scatterTable[upper]; ok {
return e.encodeScatter(upper, ss, ops, sfx)
}
if upper == "KMOVW" || upper == "KMOVQ" {
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
if sfx.any() {
return fmt.Errorf("%s takes no EVEX suffixes", upper)
}
+242
View File
@@ -546,6 +546,248 @@ func TestEncodableCmovSize(t *testing.T) {
}
}
// TestCarryShiftMulGroundTruth pins the carry-flag ALU family (ADC/SBB with
// their accumulator immediate forms), the rotate family, MUL/DIV/IDIV and the
// bit-test family byte for byte against go tool asm (see
// testdata/verify/scalar_amd64.s).
func TestCarryShiftMulGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ADCQ AX,BX", "ADCQ", []Operand{AX, BX}, "4811c3"},
{"ADCL AX,BX", "ADCL", []Operand{AX, BX}, "11c3"},
{"ADCB AL,BL", "ADCB", []Operand{AL, BL}, "10c3"},
{"ADCW AX,BX", "ADCW", []Operand{AX, BX}, "6611c3"},
{"SBBQ AX,BX", "SBBQ", []Operand{AX, BX}, "4819c3"},
{"ADCQ $5,BX", "ADCQ", []Operand{Imm(5), BX}, "4883d305"},
{"ADCQ $300,BX", "ADCQ", []Operand{Imm(300), BX}, "4881d32c010000"},
{"ADCQ $300,AX", "ADCQ", []Operand{Imm(300), AX}, "48152c010000"},
{"ADCB $5,AL", "ADCB", []Operand{Imm(5), AL}, "1405"},
{"SBBQ $300,AX", "SBBQ", []Operand{Imm(300), AX}, "481d2c010000"},
{"ADCQ AX,(BX)", "ADCQ", []Operand{AX, Ptr(BX, 0, 8)}, "481103"},
{"ROLQ $3,AX", "ROLQ", []Operand{Imm(3), AX}, "48c1c003"},
{"ROLL CX,BX", "ROLL", []Operand{CL, BX}, "d3c3"},
{"RORQ CL,AX", "RORQ", []Operand{CL, AX}, "48d3c8"},
{"RCRQ $1,BX", "RCRQ", []Operand{Imm(1), BX}, "48d1db"},
{"RCLQ $3,AX", "RCLQ", []Operand{Imm(3), AX}, "48c1d003"},
{"RORB CL,BL", "RORB", []Operand{CL, BL}, "d2cb"},
{"SALQ $2,AX", "SALQ", []Operand{Imm(2), AX}, "48c1e002"},
{"ROLW $1,AX", "ROLW", []Operand{Imm(1), AX}, "66d1c0"},
{"MULQ CX", "MULQ", []Operand{CX}, "48f7e1"},
{"MULL CX", "MULL", []Operand{CX}, "f7e1"},
{"MULB CL", "MULB", []Operand{CL}, "f6e1"},
{"DIVL CX", "DIVL", []Operand{CX}, "f7f1"},
{"IDIVQ CX", "IDIVQ", []Operand{CX}, "48f7f9"},
{"MULW CX", "MULW", []Operand{CX}, "66f7e1"},
{"BTQ AX,DX", "BTQ", []Operand{AX, DX}, "480fa3c2"},
{"BTL AX,DX", "BTL", []Operand{AX, DX}, "0fa3c2"},
{"BTW AX,DX", "BTW", []Operand{AX, DX}, "660fa3c2"},
{"BTQ $3,BX", "BTQ", []Operand{Imm(3), BX}, "480fbae303"},
{"BTQ $3,(AX)", "BTQ", []Operand{Imm(3), Ptr(AX, 0, 8)}, "480fba2003"},
{"BTSQ $5,BX", "BTSQ", []Operand{Imm(5), BX}, "480fbaeb05"},
{"BTCQ AX,BX", "BTCQ", []Operand{AX, BX}, "480fbbc3"},
{"BTRQ $7,BX", "BTRQ", []Operand{Imm(7), BX}, "480fbaf307"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// The bit-test immediate is an unsigned bit index with the negative
// spelling accepted, the shuffle convention: BTQ $300 must be rejected.
if _, err := Encode("BTQ", Imm(300), AX); err == nil {
t.Errorf("BTQ $300: expected an error, got none")
}
}
// TestAtomicSystemGroundTruth pins the exchange/compare-exchange/accumulate
// family, the string primitives, the flag and system instructions, the MXCSR
// pair, the scalar float-to-int conversions and the x87 FMOVD byte for byte
// against go tool asm (see testdata/verify/atomics_amd64.s and
// testdata/verify/system_amd64.s).
func TestAtomicSystemGroundTruth(t *testing.T) {
r8 := Reg{idx: 8, size: 8}
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"XCHGQ AX,BX", "XCHGQ", []Operand{AX, BX}, "4893"},
{"XCHGQ BX,AX", "XCHGQ", []Operand{BX, AX}, "4893"},
{"XCHGL AX,BX", "XCHGL", []Operand{AX, BX}, "93"},
{"XCHGB AL,BL", "XCHGB", []Operand{AL, BL}, "86c3"},
{"XCHGW AX,BX", "XCHGW", []Operand{AX, BX}, "6693"},
{"XCHGQ R8,R9", "XCHGQ", []Operand{r8, Reg{idx: 9, size: 8}}, "4d87c1"},
{"XCHGQ BX,(AX)", "XCHGQ", []Operand{BX, Ptr(AX, 0, 8)}, "488718"},
{"XCHGQ (AX),BX", "XCHGQ", []Operand{Ptr(AX, 0, 8), BX}, "488718"},
{"XCHGQ AX,(BX)", "XCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "488703"},
{"CMPXCHGL AX,BX", "CMPXCHGL", []Operand{AX, BX}, "0fb1c3"},
{"CMPXCHGQ AX,(BX)", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
{"CMPXCHGB AL,(BX)", "CMPXCHGB", []Operand{AL, Ptr(BX, 0, 1)}, "0fb003"},
{"CMPXCHGW AX,BX", "CMPXCHGW", []Operand{AX, BX}, "660fb1c3"},
{"XADDL AX,BX", "XADDL", []Operand{AX, BX}, "0fc1c3"},
{"XADDQ AX,(BX)", "XADDQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fc103"},
{"XADDB AL,(BX)", "XADDB", []Operand{AL, Ptr(BX, 0, 1)}, "0fc003"},
{"XADDW AX,BX", "XADDW", []Operand{AX, BX}, "660fc1c3"},
{"ADCXL AX,CX", "ADCXL", []Operand{AX, CX}, "660f38f6c8"},
{"ADCXQ AX,CX", "ADCXQ", []Operand{AX, CX}, "66480f38f6c8"},
{"ADOXL AX,CX", "ADOXL", []Operand{AX, CX}, "f30f38f6c8"},
{"ADOXQ AX,CX", "ADOXQ", []Operand{AX, CX}, "f3480f38f6c8"},
{"CRC32B AX,CX", "CRC32B", []Operand{AX, CX}, "f20f38f0c8"},
{"CRC32W AX,CX", "CRC32W", []Operand{AX, CX}, "66f20f38f1c8"},
{"CRC32L AX,CX", "CRC32L", []Operand{AX, CX}, "f20f38f1c8"},
{"CRC32Q AX,CX", "CRC32Q", []Operand{AX, CX}, "f2480f38f1c8"},
{"CRC32L (AX),CX", "CRC32L", []Operand{Ptr(AX, 0, 4), CX}, "f20f38f108"},
{"MOVSQ", "MOVSQ", []Operand{}, "48a5"},
{"MOVSL", "MOVSL", []Operand{}, "a5"},
{"MOVSB", "MOVSB", []Operand{}, "a4"},
{"MOVSW", "MOVSW", []Operand{}, "66a5"},
{"STOSB", "STOSB", []Operand{}, "aa"},
{"STOSQ", "STOSQ", []Operand{}, "48ab"},
{"STOSL", "STOSL", []Operand{}, "ab"},
{"STOSW", "STOSW", []Operand{}, "66ab"},
{"CLD", "CLD", []Operand{}, "fc"},
{"STD", "STD", []Operand{}, "fd"},
{"POPFQ", "POPFQ", []Operand{}, "9d"},
{"PUSHFQ", "PUSHFQ", []Operand{}, "9c"},
{"CPUID", "CPUID", []Operand{}, "0fa2"},
{"RDTSC", "RDTSC", []Operand{}, "0f31"},
{"RDTSCP", "RDTSCP", []Operand{}, "0f01f9"},
{"SYSCALL", "SYSCALL", []Operand{}, "0f05"},
{"XGETBV", "XGETBV", []Operand{}, "0f01d0"},
{"PAUSE", "PAUSE", []Operand{}, "f390"},
{"LFENCE", "LFENCE", []Operand{}, "0faee8"},
{"MFENCE", "MFENCE", []Operand{}, "0faef0"},
{"SFENCE", "SFENCE", []Operand{}, "0faef8"},
{"UNDEF", "UNDEF", []Operand{}, "0f0b"},
{"INT $3", "INT", []Operand{Imm(3)}, "cd03"},
{"LDMXCSR (AX)", "LDMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae10"},
{"STMXCSR (AX)", "STMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae18"},
{"CVTSD2SL X0,AX", "CVTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2dc0"},
{"CVTTSD2SQ X0,AX", "CVTTSD2SQ", []Operand{vreg(t, "X0"), AX}, "f2480f2cc0"},
{"CVTTSD2SL X0,AX", "CVTTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2cc0"},
{"CVTSS2SQ X0,AX", "CVTSS2SQ", []Operand{vreg(t, "X0"), AX}, "f3480f2dc0"},
{"FMOVD (AX),F0", "FMOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "F0")}, "dd00"},
{"FMOVD F0,(AX)", "FMOVD", []Operand{vreg(t, "F0"), Ptr(AX, 0, 8)}, "dd10"},
{"FMOVD F0,F1", "FMOVD", []Operand{vreg(t, "F0"), vreg(t, "F1")}, "ddd1"},
{"MOVD AX,X0", "MOVD", []Operand{AX, vreg(t, "X0")}, "66480f6ec0"},
{"MOVD X0,AX", "MOVD", []Operand{vreg(t, "X0"), AX}, "66480f7ec0"},
{"MOVD X0,X1", "MOVD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f30f7ec8"},
{"MOVD (AX),X0", "MOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "X0")}, "f30f7e00"},
{"MOVD X0,(AX)", "MOVD", []Operand{vreg(t, "X0"), Ptr(AX, 0, 8)}, "660fd600"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// LDMXCSR/STMXCSR take a memory operand only.
if _, err := Encode("LDMXCSR", AX); err == nil {
t.Errorf("LDMXCSR AX: expected an error, got none")
}
}
// TestSSEGapsGroundTruth pins the legacy SSE gap families: the scalar
// compare and square root, the Plan 9 packed spellings, the imm8-controlled
// shuffles, the lane extracts and inserts, the packed integer shifts and the
// AES/SHA round instructions, byte for byte against go tool asm (see
// testdata/verify/crypto_amd64.s and testdata/verify/sse_amd64.s).
func TestSSEGapsGroundTruth(t *testing.T) {
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"ANDNPD X0,X1", "ANDNPD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f55c8"},
{"ANDNPS X0,X1", "ANDNPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f55c8"},
{"COMISD X0,X1", "COMISD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f2fc8"},
{"SQRTSD X0,X1", "SQRTSD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f20f51c8"},
{"PSHUFL $3,X0,X1", "PSHUFL", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f70c803"},
{"PALIGNR $2,X0,X1", "PALIGNR", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0fc802"},
{"PBLENDW $3,X0,X1", "PBLENDW", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0ec803"},
{"PCMPESTRI $1,X0,X1", "PCMPESTRI", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}, "660f3a61c801"},
{"PCLMULQDQ $0,X0,X1", "PCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3a44c800"},
{"PCLMULQDQ $0,(AX),X1", "PCLMULQDQ", []Operand{Imm(0), Ptr(AX, 0, 16), vreg(t, "X1")}, "660f3a440800"},
{"PEXTRB $1,X0,AX", "PEXTRB", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a14c001"},
{"PEXTRD $1,X0,AX", "PEXTRD", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a16c001"},
{"PEXTRQ $1,X0,AX", "PEXTRQ", []Operand{Imm(1), vreg(t, "X0"), AX}, "66480f3a16c001"},
{"PEXTRW $1,X0,AX", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), AX}, "660fc5c001"},
{"PEXTRW $1,X0,(AX)", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), Ptr(AX, 0, 2)}, "660f3a150001"},
{"PINSRB $1,AX,X0", "PINSRB", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a20c001"},
{"PINSRD $1,AX,X0", "PINSRD", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a22c001"},
{"PINSRQ $1,AX,X0", "PINSRQ", []Operand{Imm(1), AX, vreg(t, "X0")}, "66480f3a22c001"},
{"PINSRW $1,AX,X0", "PINSRW", []Operand{Imm(1), AX, vreg(t, "X0")}, "660fc4c001"},
{"PINSRW $1,(AX),X0", "PINSRW", []Operand{Imm(1), Ptr(AX, 0, 2), vreg(t, "X0")}, "660fc40001"},
{"PSLLL $2,X0", "PSLLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72f002"},
{"PSRAL $2,X0", "PSRAL", []Operand{Imm(2), vreg(t, "X0")}, "660f72e002"},
{"PSRLL $2,X0", "PSRLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72d002"},
{"PSRLQ $2,X0", "PSRLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d002"},
{"PSLLQ $2,X0", "PSLLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f002"},
{"PSLLW $2,X0", "PSLLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71f002"},
{"PSRLW $2,X0", "PSRLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71d002"},
{"PSRAW $2,X0", "PSRAW", []Operand{Imm(2), vreg(t, "X0")}, "660f71e002"},
{"PSLLDQ $2,X0", "PSLLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f802"},
{"PSRLDQ $2,X0", "PSRLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d802"},
{"PSLLL X0,X1", "PSLLL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ff2c8"},
{"PSRLQ X0,X1", "PSRLQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660fd3c8"},
{"PSLLL (AX),X1", "PSLLL", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660ff208"},
{"PSUBL X0,X1", "PSUBL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffac8"},
{"PADDL X0,X1", "PADDL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffec8"},
{"PCMPEQL X0,X1", "PCMPEQL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f76c8"},
{"PUNPCKLBW X0,X1", "PUNPCKLBW", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f60c8"},
{"MOVOA X0,X1", "MOVOA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f6fc8"},
{"MOVOA (AX),X1", "MOVOA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f6f08"},
{"MOVOA X0,(AX)", "MOVOA", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "660f7f00"},
{"AESIMC X0,X1", "AESIMC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dbc8"},
{"AESIMC (AX),X1", "AESIMC", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f38db08"},
{"AESENC X0,X1", "AESENC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dcc8"},
{"AESENCLAST X0,X1", "AESENCLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38ddc8"},
{"AESDEC X0,X1", "AESDEC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dec8"},
{"AESDECLAST X0,X1", "AESDECLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dfc8"},
{"AESKEYGENASSIST $0,X0,X1", "AESKEYGENASSIST", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3adfc800"},
{"SHA1MSG1 X0,X1", "SHA1MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c9c8"},
{"SHA1MSG2 X0,X1", "SHA1MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cac8"},
{"SHA1NEXTE X0,X1", "SHA1NEXTE", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c8c8"},
{"SHA1RNDS4 $0,X0,X1", "SHA1RNDS4", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "0f3accc800"},
{"SHA256MSG1 X0,X1", "SHA256MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38ccc8"},
{"SHA256MSG2 X0,X1", "SHA256MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cdc8"},
{"SHA256RNDS2 X0,X1,X2", "SHA256RNDS2", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "0f38cbd1"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := fmt.Sprintf("%x", code); got != c.want {
t.Errorf("%s = %s, want %s", c.name, got, c.want)
}
}
// SHA256RNDS2's first operand must be the literal X0.
if _, err := Encode("SHA256RNDS2", vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")); err == nil {
t.Errorf("SHA256RNDS2 X1,...: expected an error, got none")
}
// PSLLDQ has no variable-count form.
if _, err := Encode("PSLLDQ", vreg(t, "X0"), vreg(t, "X1")); err == nil {
t.Errorf("PSLLDQ X0,X1: expected an error, got none")
}
}
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
// byte for byte (no prefix / 66 / F2 / F3 variants).
func TestSSEBinGroundTruth(t *testing.T) {
+35 -18
View File
@@ -180,10 +180,19 @@ var evexTable = map[string]evexSpec{
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
// EVEX.66.0F38, permutes (NDS form).
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2B": {2, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
// EVEX.66.0F38, population count (reg=dst, rm=src; W selects byte/word
// against dword/qword).
"VPOPCNTB": {2, 0x54, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPOPCNTD": {2, 0x55, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
"VPOPCNTQ": {2, 0x55, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
// EVEX.66.0F.W1, the qword spelling of the packed OR (VPORQ has no VEX
// form in the Go assembler: it always encodes through EVEX).
"VPORQ": {1, 0xEB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
@@ -1435,18 +1444,20 @@ var evexKOperand = map[string]bool{
}
// kmovSpec describes a KMOV width: the opcode depends on the operand
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
// prefix and W for the wider widths.
// direction, kk (k → k), kmem (k → mem), gprk (GPR/mem → k) and kgpr
// (k → GPR). Each direction group carries its own mandatory prefix and W:
// the k-destination/source forms share one pair, the GPR forms another.
type kmovSpec struct {
kk, kmem, gprk, kgpr byte
gprPP int
w int
kPP, kW int // prefix and VEX.W for the k forms
gprPP, gprW int // prefix and VEX.W for the GPR forms
}
var kmovTable = map[string]kmovSpec{
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0, 0, 0},
"KMOVB": {0x90, 0x91, 0x92, 0x93, 1, 0, 1, 0},
"KMOVD": {0x90, 0x91, 0x92, 0x93, 1, 1, 3, 0},
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 0, 1, 3, 1},
}
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
@@ -1460,14 +1471,14 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
dstReg, dstIsReg := dst.(Reg)
srcK := srcIsReg && srcReg.mask
dstK := dstIsReg && dstReg.mask
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
switch {
case srcK && dstK:
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
// k ← k: reg = dst, rm = src.
spec := vexSpec{mapSel: 1, opcode: ks.kk, w: ks.kW, pp: ks.kPP, opdigit: -1}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
case srcK && dstIsReg:
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
spec.pp = ks.gprPP
// GPR ← k: reg = dst, rm = src.
spec := vexSpec{mapSel: 1, opcode: ks.kgpr, w: ks.gprW, pp: ks.gprPP, opdigit: -1}
rBit := 0
if dstReg.idx >= 8 {
rBit = 1
@@ -1477,11 +1488,17 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
if _, ok := dst.(Mem); !ok {
return fmt.Errorf("%s: invalid destination operand", upper)
}
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
// mem ← k: reg = src, rm = dst.
spec := vexSpec{mapSel: 1, opcode: ks.kmem, w: ks.kW, pp: ks.kPP, opdigit: -1}
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
case dstK:
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
spec.pp = ks.gprPP
// k ← GPR: reg = dst, rm = src. A memory source shares the k ← k
// opcode and prefix group (the ykmovb layout the Go assembler uses).
opcode, w, pp := ks.gprk, ks.gprW, ks.gprPP
if memOperand(src) {
opcode, w, pp = ks.kk, ks.kW, ks.kPP
}
spec := vexSpec{mapSel: 1, opcode: opcode, w: w, pp: pp, opdigit: -1}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
}
return fmt.Errorf("%s requires a K register operand", upper)
+19
View File
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
// The qword OR spelling always encodes through EVEX.
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
// Byte permute and population count.
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
// Align (NDS + imm8).
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
@@ -52,6 +61,16 @@ func TestEvexGroundTruth(t *testing.T) {
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
// Moves, incl. disp8×N (64 for a 512-bit operand).
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
+630 -9
View File
@@ -14,30 +14,70 @@ var aluOp = map[string]struct {
}{
"ADD": {0x01, 0},
"OR": {0x09, 1},
"ADC": {0x11, 2},
"SBB": {0x19, 3},
"AND": {0x21, 4},
"SUB": {0x29, 5},
"XOR": {0x31, 6},
"CMP": {0x39, 7},
}
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
// mode); NEG/NOT use the 0xF6/0xF7 group.
// unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode.
// INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes
// in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4,
// DIV /6, IDIV /7; the accumulator is the implicit other operand).
var unaryOp = map[string]struct {
digit int
op byte
}{
"INC": {0, 0xFF},
"DEC": {1, 0xFF},
"NOT": {2, 0xF7},
"NEG": {3, 0xF7},
"INC": {0, 0xFF},
"DEC": {1, 0xFF},
"NOT": {2, 0xF7},
"NEG": {3, 0xF7},
"MUL": {4, 0xF7},
"DIV": {6, 0xF7},
"IDIV": {7, 0xF7},
}
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
// shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the
// 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4).
var shiftOp = map[string]int{
"SHL": 4,
"SAL": 4,
"SHR": 5,
"SAR": 7,
"ROL": 0,
"ROR": 1,
"RCL": 2,
"RCR": 3,
}
// bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form;
// the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's
// opcode row.
var bitTestOp = map[string]int{
"BT": 4,
"BTS": 5,
"BTR": 6,
"BTC": 7,
}
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
// fence names carry their opcode inside the 0F AE /digit group spelled out in
// full (E8/F0/F8), and PAUSE is F3 90.
var noOperandTable = map[string][]byte{
"CPUID": {0x0F, 0xA2},
"RDTSC": {0x0F, 0x31},
"RDTSCP": {0x0F, 0x01, 0xF9},
"SYSCALL": {0x0F, 0x05},
"XGETBV": {0x0F, 0x01, 0xD0},
"CLD": {0xFC},
"STD": {0xFD},
"PAUSE": {0xF3, 0x90},
"LFENCE": {0x0F, 0xAE, 0xE8},
"MFENCE": {0x0F, 0xAE, 0xF0},
"SFENCE": {0x0F, 0xAE, 0xF8},
"UNDEF": {0x0F, 0x0B},
}
// --- MOV --------------------------------------------------------------------
@@ -309,6 +349,13 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
if err != nil {
return err
}
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
// the destination is AL, the form the Go assembler prefers here.
if r, ok := dst.(Reg); ok && r.idx == 0 {
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
i.imm = immBytes
return e.emit(i)
}
i := newInstr(1, []byte{0x80})
if err := setRMDigit(i, digit, dst, 1); err != nil {
return err
@@ -913,6 +960,7 @@ type sseMove struct {
var sseMoveTable = map[string]sseMove{
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
"MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
@@ -1000,10 +1048,120 @@ var sseBinTable = map[string]sseBin{
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
"PCMPEQD": {0x66, 0x76, false},
"PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false},
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
"PCMPGTD": {0x66, 0x66, false},
"PSHUFB": {0x66, 0x00, true},
// Scalar compares and square root, packed adds/subtracts and the byte
// unpack, the spellings the Plan 9 table uses (COMISD orders the
// operands like every other two-operand form).
"ANDNPD": {0x66, 0x55, false},
"ANDNPS": {0x00, 0x55, false},
"COMISD": {0x66, 0x2F, false},
"SQRTSD": {0xF2, 0x51, false},
"PADDL": {0x66, 0xFE, false},
"PSUBL": {0x66, 0xFA, false},
"PUNPCKLBW": {0x66, 0x60, false},
// AES round functions (66 0F38) and the SHA message schedule helpers
// (no prefix, 0F38).
"AESENC": {0x66, 0xDC, true},
"AESENCLAST": {0x66, 0xDD, true},
"AESDEC": {0x66, 0xDE, true},
"AESDECLAST": {0x66, 0xDF, true},
"AESIMC": {0x66, 0xDB, true},
"SHA1MSG1": {0x00, 0xC9, true},
"SHA1MSG2": {0x00, 0xCA, true},
"SHA1NEXTE": {0x00, 0xC8, true},
"SHA256MSG1": {0x00, 0xCC, true},
"SHA256MSG2": {0x00, 0xCD, true},
}
// sseImm3 describes a legacy SSE instruction taking a leading imm8 and two
// further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and
// map3A select the opcode map the same way as sseBin's.
type sseImm3 struct {
prefix byte
op byte
map3A bool // opcode lives under 0F3A instead of 0F38
}
// sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3
// align/blend shuffles, the string compare, carry-less multiply and the AES
// key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings.
var sseImm3Table = map[string]sseImm3{
"PALIGNR": {0x66, 0x0F, true},
"PBLENDW": {0x66, 0x0E, true},
"PCMPESTRI": {0x66, 0x61, true},
"PCLMULQDQ": {0x66, 0x44, true},
"AESKEYGENASSIST": {0x66, 0xDF, true},
"SHA1RNDS4": {0x00, 0xCC, true},
}
// sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM
// source and rm = the destination (GPR or memory). PEXTRW's GPR destination
// uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one,
// so it carries both opcodes.
type sseExtract struct {
op []byte
opMem []byte // used when the destination is memory; nil shares op
rexW bool // PEXTRQ's REX.W
}
var sseExtractTable = map[string]sseExtract{
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
}
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
// destination and rm = the source (GPR or memory).
type sseInsert struct {
op []byte
rexW bool // PINSRQ's REX.W
}
var sseInsertTable = map[string]sseInsert{
"PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false},
"PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false},
"PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true},
"PINSRW": {[]byte{0x0F, 0xC4}, false},
}
// sseShiftImm maps the legacy packed integer shifts' immediate form:
// OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L
// (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ.
var sseShiftImm = map[string]sseShift{
"PSLLW": {0x71, 6},
"PSRLW": {0x71, 2},
"PSRAW": {0x71, 4},
"PSLLL": {0x72, 6},
"PSRLL": {0x72, 2},
"PSRAL": {0x72, 4},
"PSLLQ": {0x73, 6},
"PSRLQ": {0x73, 2},
"PSLLDQ": {0x73, 7},
"PSRLDQ": {0x73, 3},
}
// sseShiftVar maps the variable-count forms (the count comes from an XMM
// register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no
// variable form.
var sseShiftVar = map[string]byte{
"PSLLW": 0xF1,
"PSRLW": 0xD1,
"PSRAW": 0xE1,
"PSLLL": 0xF2,
"PSRLL": 0xD2,
"PSRAL": 0xE2,
"PSLLQ": 0xF3,
"PSRLQ": 0xD3,
}
// sseShift is one /digit selector in the 0F 71/72/73 immediate group.
type sseShift struct {
op byte
digit int
}
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
@@ -1016,6 +1174,7 @@ type sseShuf struct {
var sseShufTable = map[string]sseShuf{
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
"PSHUFL": {0x66, 0x70},
}
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
@@ -1090,3 +1249,465 @@ func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
}
return e.emit(i)
}
// --- carry, bit test, exchange and accumulate -------------------------------
// encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9
// order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with
// reg = index, rm = target); an immediate index uses 0F BA /digit with imm8.
func (e *enc) encodeBitTest(name string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
digit := bitTestOp[name]
index, target := ops[0], ops[1]
if reg, ok := index.(Reg); ok {
// Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC),
// the /digit base plus eight per step.
i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3})
if err := setRM(i, reg, target, size); err != nil {
return err
}
return e.emit(i)
}
imm, ok := index.(Imm)
if !ok {
return fmt.Errorf("%s index must be a register or an immediate", name)
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
i := newInstr(size, []byte{0x0F, 0xBA})
if err := setRMDigit(i, digit, target, size); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeExchange encodes XCHG. A register-to-register exchange where either
// operand is AX uses the 0x90+r accumulator form (with REX.W for the quad
// form, as the Go assembler emits it); everything else uses 0x86/0x87 with
// the register operand in ModRM.reg, the memory (or second register) in r/m.
func (e *enc) encodeExchange(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsReg := src.(Reg)
dstReg, dstIsReg := dst.(Reg)
if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) {
// 0x90+r: r is the non-AX register, whichever side it sits on.
r := dstReg
if srcReg.idx == 0 {
r = dstReg
} else {
r = srcReg
}
i := newInstr(size, []byte{0x90 + byte(r.idx&7)})
i.rexB = r.idx >= 8
return e.emit(i)
}
op := byte(0x87)
if size == 1 {
op = 0x86
}
switch {
case srcIsReg:
i := newInstr(size, []byte{op})
if err := setRM(i, srcReg, dst, size); err != nil {
return err
}
return e.emit(i)
case dstIsReg:
i := newInstr(size, []byte{op})
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("XCHG: at least one operand must be a register")
}
// encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG
// (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the
// destination writable (register or memory).
func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok {
return fmt.Errorf("%s source must be a register", name)
}
opc := op
if size == 1 {
opc = op8
}
i := newInstr(size, []byte{0x0F, opc})
if err := setRM(i, srcReg, ops[1], size); err != nil {
return err
}
return e.emit(i)
}
// encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for
// the rest; the word form carries a 0x66 operand-size prefix (66 F2, the
// prefix order the Go assembler emits) and the quad form REX.W. reg = GPR
// accumulator, rm = the data source.
func (e *enc) encodeCrc32(ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("CRC32 destination must be a general register")
}
i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1}
if size > 1 {
i.opcode[2] = 0xF1
}
i.rexW = size == 8
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg =
// destination, rm = source, the carry/overflow flag as the carry-in.
func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("ADCX/ADOX destination must be a general register")
}
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8}
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// --- string primitives, flags and INT ----------------------------------------
// encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and
// STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or
// REX.W prefix.
func (e *enc) encodeStringOp(base string, ops []Operand, size int) error {
if len(ops) != 0 {
return fmt.Errorf("%s takes no operands, got %d", base, len(ops))
}
var op byte
switch base {
case "MOVS":
op = 0xA5
if size == 1 {
op = 0xA4
}
case "STOS":
op = 0xAB
if size == 1 {
op = 0xAA
}
default:
return fmt.Errorf("unsupported string instruction %q", base)
}
return e.emit(newInstr(size, []byte{op}))
}
// encodeInt encodes INT with its single imm8 operand. The field takes the
// low byte silently inside the 32-bit span, matching the scalar convention
// (go tool asm encodes INT $256 as CD 00).
func (e *enc) encodeInt(ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("INT expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("INT operand must be an immediate")
}
if imm < -(1<<31) || imm > (1<<32)-1 {
return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm))
}
return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}})
}
// encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a
// single 32-bit memory operand.
func (e *enc) encodeMxcsr(digit int, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops))
}
m, ok := ops[0].(Mem)
if !ok {
return fmt.Errorf("MXCSR instruction requires a memory operand")
}
i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1}
if err := setMem(i, digit, m); err != nil {
return err
}
return e.emit(i)
}
// cvtIntOp maps the scalar float-to-integer conversions to their mandatory
// prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating
// CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width.
var cvtIntOp = map[string]struct {
prefix byte
op byte
}{
"CVTSD2S": {0xF2, 0x2D},
"CVTTSD2S": {0xF2, 0x2C},
"CVTSS2S": {0xF3, 0x2D},
"CVTTSS2S": {0xF3, 0x2C},
}
// encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C
// with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the
// quad spellings.
func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
}
spec := cvtIntOp[base]
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("%s destination must be a general register", base)
}
i := newInstr(size, []byte{0x0F, spec.op})
i.prefix = spec.prefix
if err := setRM(i, dstReg, src, size); err != nil {
return err
}
return e.emit(i)
}
// encodeFmov encodes the x87 double move. The memory forms are DD /0
// (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register
// move is DD C0+dst (FLD st(dst)), the form the Go assembler emits.
func (e *enc) encodeFmov(ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops))
}
src, dst := ops[0], ops[1]
srcReg, srcIsF := src.(Reg)
dstReg, dstIsF := dst.(Reg)
srcF := srcIsF && srcReg.fp
dstF := dstIsF && dstReg.fp
switch {
case srcF && dstF:
// The register form is DD /2 with rm = the destination (FST st(dst)).
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setRMDigit(i, 2, dstReg, 8); err != nil {
return err
}
return e.emit(i)
case dstF:
m, ok := src.(Mem)
if !ok {
return fmt.Errorf("FMOVD: invalid source operand")
}
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setMem(i, 0, m); err != nil {
return err
}
return e.emit(i)
case srcF:
m, ok := dst.(Mem)
if !ok {
return fmt.Errorf("FMOVD: invalid destination operand")
}
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
if err := setMem(i, 2, m); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("FMOVD needs an x87 register operand")
}
// --- legacy SSE imm8, extract, insert and packed shift families --------------
// encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src,
// dst with reg = dst, rm = src and the immediate appended last (PALIGNR,
// PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4).
func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("SSE imm8 instruction needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
src, dst := ops[1], ops[2]
dstReg, ok2 := dst.(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("SSE imm8 instruction destination must be a vector register")
}
opcode := []byte{0x0F, 0x38, m.op}
if m.map3A {
opcode = []byte{0x0F, 0x3A, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the
// XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and
// PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the
// SSE4.1 0F3A 15 one).
func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("extract needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
srcReg, srcVec := vecReg(ops[1])
if !srcVec {
return fmt.Errorf("extract source must be an XMM register")
}
opcode := m.op
if m.opMem != nil && memOperand(ops[2]) {
opcode = m.opMem
}
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
if err := setRM(i, srcReg, ops[2], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the
// XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and
// PINSRW).
func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("insert needs an immediate first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
dstReg, dstVec := vecReg(ops[2])
if !dstVec {
return fmt.Errorf("insert destination must be an XMM register")
}
i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW}
if err := setRM(i, dstReg, ops[1], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSSEShift encodes the legacy packed integer shifts. The immediate
// form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form
// OP count, dst carries the count in an XMM register (or memory) on the
// 66 0F D1-F3 opcodes. The destination is always the register written.
func (e *enc) encodeSSEShift(name string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.isVec() {
return fmt.Errorf("%s destination must be the second, vector operand", name)
}
if imm, isImm := ops[0].(Imm); isImm {
spec := sseShiftImm[name]
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
if !vecOrMem(ops[0]) {
return fmt.Errorf("%s count must be an immediate, a vector register or memory", name)
}
op, ok := sseShiftVar[name]
if !ok {
return fmt.Errorf("%s has no variable-count form", name)
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
return e.emit(i)
}
// encodeCmpsd encodes CMPSD, the scalar double compare with its predicate
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
// F2 0F C2 with reg = dst, rm = src.
func (e *enc) encodeCmpsd(ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
}
imm, ok := ops[2].(Imm)
if !ok {
return fmt.Errorf("CMPSD predicate must be an immediate")
}
immByte, err := imm8(int64(imm))
if err != nil {
return err
}
dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("CMPSD destination must be a vector register")
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
i.imm = []byte{immByte}
return e.emit(i)
}
// encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the
// literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no
// prefix, reg = dst, rm = src; X0 is implicit on the wire).
func (e *enc) encodeSha256rnds2(ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops))
}
x0, ok := ops[0].(Reg)
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
return fmt.Errorf("SHA256RNDS2 first operand must be X0")
}
dstReg, ok2 := ops[2].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("SHA256RNDS2 destination must be a vector register")
}
i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[1], 8); err != nil {
return err
}
return e.emit(i)
}
+422
View File
@@ -321,6 +321,16 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before
// the integer/FP table (their mnemonics overlap the table's 2R format
// but resolve vector-bank registers).
if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled {
if err != nil {
return nil, err
}
return code, nil
}
enc, ok := l64InstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
@@ -1490,3 +1500,415 @@ func l64Label(op *ast.Operand) string {
}
return op.Raw
}
// ---- LSX/LASX (V*/XV*) vector dispatch ----
// l64VecOperand describes a vector register operand: the 5-bit register
// number, its bank and an optional width or element suffix (V0.B16,
// V1.V[0], X3.WU[2]). The parser hands suffixed operands over verbatim
// (the element index survives only in the raw text), so the suffix is
// scanned from op.Raw.
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
unsig bool // the suffix carries the U marker (.BU[0])
hasSuf bool // any suffix present
}
// l64ParseVecOperand parses a vector register operand with an optional
// width or element suffix. ok reports whether the operand names a vector
// register at all (V or X bank, with or without a suffix).
func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
if op.Kind == ast.OpImmediate {
return v, false
}
name := strings.ReplaceAll(op.Raw, " ", "")
if name == "" || (name[0] != 'V' && name[0] != 'X') {
return v, false
}
i := 1
num := 0
for i < len(name) && name[i] >= '0' && name[i] <= '9' {
num = num*10 + int(name[i]-'0')
if num > 31 {
return v, false
}
i++
}
if i == 1 {
return v, false // no register digits
}
v.num, v.lasx = num, name[0] == 'X'
if i == len(name) {
return v, true
}
if name[i] != '.' || i+2 > len(name) {
return v, false
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
return v, false
}
v.width, v.hasSuf = w, true
i++
if i < len(name) && name[i] == 'U' {
v.unsig = true
i++
}
if i < len(name) && name[i] == '[' {
// Element form .T[i]: the closing bracket ends the operand.
if name[len(name)-1] != ']' || i+2 > len(name)-1 {
return v, false
}
idx := 0
for _, c := range name[i+1 : len(name)-1] {
if c < '0' || c > '9' {
return v, false
}
idx = idx*10 + int(c-'0')
if idx > 31 {
return v, false
}
}
v.elem, v.hasEl = idx, true
return v, true
}
// Width form .T<lanes>: the trailing digits give the lane count.
lanes := 0
if i >= len(name) {
return v, false
}
for ; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return v, false
}
lanes = lanes*10 + int(name[i]-'0')
if lanes > 64 {
return v, false
}
}
v.lanes = lanes
return v, true
}
// l64VecSuffixWidth validates a width suffix against the bank (LSX:
// B16/H8/W4/V2, LASX: B32/H16/W8/V4) and returns the encoded 2-bit width
// selector of vreplgr2vr and vldrepl.
func l64VecSuffixWidth(lasx bool, v l64VecOperand) (int, bool) {
want := map[byte]int{'B': 16, 'H': 8, 'W': 4, 'V': 2}
if lasx {
want = map[byte]int{'B': 32, 'H': 16, 'W': 8, 'V': 4}
}
lanes, ok := want[v.width]
if !ok || lanes != v.lanes {
return 0, false
}
switch v.width {
case 'B':
return 0, true
case 'H':
return 1, true
case 'W':
return 2, true
default:
return 3, true
}
}
// l64VecElementBase validates an element suffix against the bank and
// returns the encoded index field: the index rides in the rk field above a
// per-width base (vpickve2gr/vinsgr2vr give ui4 to .b, ui3 to .h, ui2 to .w
// and ui1 to .d). The LASX bank has no .b/.h element forms: the toolchain
// rejects `XVMOVQ R4, X2.B[0]` and `XVMOVQ X3.B[31], R5`.
func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) {
limit, base := 0, 0
switch v.width {
case 'B':
if lasx {
return 0, false
}
limit, base = 15, 0
case 'H':
if lasx {
return 0, false
}
limit, base = 7, 16
case 'W':
limit, base = 3, 24
if lasx {
limit, base = 7, 16
}
case 'V':
limit, base = 1, 28
if lasx {
limit, base = 3, 24
}
default:
return 0, false
}
if v.elem > limit {
return 0, false
}
return base + v.elem, true
}
// encodeLOONG64Vector encodes the LSX/LASX mnemonics the table marks as
// vector plus the VMOVQ/XVMOVQ move family. handled reports whether the
// mnemonic belongs to the vector slice; the operand shapes and opcode
// constants reproduce GOARCH=loong64 `go tool asm` exactly.
func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) {
if mnem == "VMOVQ" || mnem == "XVMOVQ" {
code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi)
return code, true, err
}
lasx, ok := l64VecBank[mnem]
if !ok {
return nil, false, nil
}
ops := instr.Operands
bank := "V"
if lasx {
bank = "X"
}
vec := func(op *ast.Operand) (int, error) {
v, isVec := l64ParseVecOperand(op)
if !isVec || v.lasx != lasx || v.hasSuf {
return -1, fmt.Errorf("%s: expected a bare %s0-%s31 vector register, got %q", mnem, bank, bank, op.Raw)
}
return v.num, nil
}
// Two-operand forms (vpcnt.v): INSTR vj, vd.
if l64Vec2R[mnem] {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[1])
if err != nil {
return nil, true, err
}
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, vd)), true, nil
}
// Immediate forms: INSTR $imm, vd or INSTR $imm, vj, vd.
if e, imm := l64VecImmInfo[mnem]; imm && len(ops) >= 2 && isImmOperand(ops[0]) {
if len(ops) > 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
imm := int(immFromOperand(ops[0]))
if imm < e.min || imm > e.max {
return nil, true, fmt.Errorf("%s: immediate out of range [%d, %d]", mnem, e.min, e.max)
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64irr(e.op, (imm+e.bias)&e.mask, vj, vd)), true, nil
}
// Vector-to-condition forms: INSTR vj, FCCn.
if l64InstrTable[mnem].format == l64Fvcf {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
if loong64RegClass(operandRegName(ops[1])) != l64ClsFCC {
return nil, true, fmt.Errorf("%s: expected an FCC condition flag, got %q", mnem, ops[1].Raw)
}
fcc := loong64RegNum(operandRegName(ops[1]))
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, fcc)), true, nil
}
// Three-register forms: INSTR vk, vj, vd or INSTR vk, vd (vj = vd).
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
vk, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64rrr(l64InstrTable[mnem].op, vk, vj, vd)), true, nil
}
// encodeLOONG64Vmovq encodes the VMOVQ/XVMOVQ move family. One mnemonic
// covers the whole LSX/LASX transfer surface, dispatched by operand shape
// exactly as the toolchain's table does:
//
// VMOVQ vd, off(rj) vst VMOVQ off(rj), vd vld
// VMOVQ vd, (rj)(rk) vstx VMOVQ (rj)(rk), vd vldx
// VMOVQ off(rj), vd.T vldrepl (load and replicate one element)
// VMOVQ vj, vd vori.b $0 (a register move)
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
if lasx {
bank = "X"
}
if len(ops) != 2 {
return nil, fmt.Errorf("VMOVQ expects 2 operands, got %d", len(ops))
}
src, srcVec := l64ParseVecOperand(ops[0])
dst, dstVec := l64ParseVecOperand(ops[1])
srcMem := isMemOperand(ops[0])
dstMem := isMemOperand(ops[1])
srcIdx := srcMem && ops[0].Addr.Index != ""
dstIdx := dstMem && ops[1].Addr.Index != ""
intReg := func(op *ast.Operand) (int, error) {
if isMemOperand(op) {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
name := operandRegName(op)
if loong64RegClass(name) != l64ClsGR {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
return loong64RegNum(name), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
if src.hasSuf || dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: a register move takes bare %s registers", bank)
}
if src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
return l64wordLE(l64rr(enc.move, src.num, dst.num)), nil
}
// Store: VMOVQ vd, off(rj) or VMOVQ vd, (rj)(rk).
if srcVec && dstMem {
if src.hasSuf || src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected a bare %s0-%s31 register as the stored value", bank, bank)
}
if dstIdx {
rj, rk := loong64RegNum(ops[1].Addr.Base), loong64RegNum(ops[1].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(enc.st, int(off), rj, src.num)), nil
}
// Load: VMOVQ off(rj), vd, the indexed VMOVQ (rj)(rk), vd, and the
// load-and-replicate form VMOVQ off(rj), vd.T.
if srcMem && dstVec {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
if srcIdx {
if dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: an indexed load takes a bare %s register", bank)
}
rj, rk := loong64RegNum(ops[0].Addr.Base), loong64RegNum(ops[0].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
op := enc.ld
if dst.hasSuf {
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
case 2:
op = enc.replW
default:
op = enc.replD
}
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}
// Element extract: VMOVQ vj.T[i], rd (vpickve2gr, signed or unsigned).
if srcVec && src.hasEl && !dstVec && !dstMem {
if src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
idx, ok := l64VecElementBase(lasx, src)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[0].Raw)
}
rd, err := intReg(ops[1])
if err != nil {
return nil, err
}
op := enc.pickS
if src.unsig {
op = enc.pickU
}
return l64wordLE(l64irr(op, idx, src.num, rd)), nil
}
// Insert and duplicate: VMOVQ rj, vd.T[i] (vinsgr2vr) and
// VMOVQ rj, vd.T (vreplgr2vr).
if !srcVec && !srcMem && dstVec && dst.hasSuf {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
rs, err := intReg(ops[0])
if err != nil {
return nil, err
}
if dst.hasEl {
idx, ok := l64VecElementBase(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.ins, idx, rs, dst.num)), nil
}
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid width suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.dup, w, rs, dst.num)), nil
}
return nil, fmt.Errorf("VMOVQ: unsupported operand combination %q, %q", ops[0].Raw, ops[1].Raw)
}
+171 -6
View File
@@ -30,7 +30,10 @@ package asm
// of the immediate and register fields), mirroring the toolchain's OP_*
// helpers, so each l64* function only ORs its fields in.
import "maps"
import (
"maps"
"strings"
)
// loong64RegNum returns the 5-bit register number for a LoongArch register
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
@@ -103,7 +106,12 @@ func loong64RegNum(name string) int {
case "R31", "S8":
return 31
}
// F0-F31, FCC0-FCC7, FCSR0-FCSR31.
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
// X0-X31) are deliberately NOT accepted here: they are a separate
// register class, and the toolchain rejects V/X names wherever an
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
// resolved only through loong64VecRegNum.
if len(name) >= 4 && name[:4] == "FCSR" {
return loong64RegSpecial(name[4:], 31)
}
@@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int {
return -1
}
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
// of their own: the toolchain accepts them only in the vector operands of the
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
// spellings never reach the integer/FP resolver.
func loong64VecRegNum(name string) int {
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
return -1
}
return loong64RegSpecial(name[1:], 31)
}
// ---- format helpers ----
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
@@ -247,7 +268,7 @@ const (
l64Firr14 // 2RI14 (ldptr/stptr)
l64Firr16 // 2RI16 (addu16i.d)
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
l64Firir // bstrins/bstrpick
l64Firrr // alsl
l64Fi15 // syscall/break/dbar
@@ -255,6 +276,8 @@ const (
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
l64Fshift // 2RI12 with a 5/6-bit shift immediate
l64Fpreld // preld (2RI12 + 5-bit hint)
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -277,10 +300,68 @@ type l64DualEnc struct {
var l64DualTable = map[string]l64DualEnc{}
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
// yet; the base integer, memory and floating-point ISA is complete.
// to their encoding.
var l64InstrTable = map[string]l64Enc{}
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
type l64Vec3Enc struct {
op uint32
lasx bool
}
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
// the opcode, the bank, the accepted immediate range, the bias the toolchain
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
type l64VecImmEnc struct {
op uint32
lasx bool
min, max int
bias int
mask int
}
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
// each accepts; presence in the map routes the mnemonic through the vector
// dispatcher rather than the integer/FP formats.
var l64VecBank = map[string]bool{}
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
var l64VecImmInfo = map[string]l64VecImmEnc{}
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
// vpcnt.v).
var l64Vec2R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
pickS, pickU uint32 // vpickve2gr.{,u} element extract
ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move
}
var l64VmovqTable = map[bool]l64VmovqEnc{
false: { // VMOVQ, the LSX (V) bank
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
},
true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
},
}
func init() {
// 3R, integer.
rrr := map[string]uint32{
@@ -360,6 +441,10 @@ func init() {
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
// LSX: convert a 64-bit integer lane to a double float. The operand
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
// so the entry stays on the 2R integer/FP format.
"FFINTDV": 0x474a << 10,
}
for m, op := range rr {
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
@@ -416,12 +501,14 @@ func init() {
// LUI is the Plan 9 spelling of lu12i.w.
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
// 4R, fused multiply-add.
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
// condition flag, the layout matches the 4R shape).
rrrr := map[string]uint32{
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
"FSEL": 0x340 << 18,
}
for m, op := range rrrr {
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
@@ -455,6 +542,10 @@ func init() {
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
// is rejected by the oracle.
am := map[string]uint32{
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
@@ -472,10 +563,84 @@ func init() {
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
// The _dbar (acquire/release) add, and, or variants: opcodes read off
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
}
for m, op := range am {
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
}
// ---- LSX/LASX (V*/XV*) ----
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
// most of them), not assumed from the LoongArch manual.
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
vec3 := map[string]l64Vec3Enc{
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
"XVADDV": {0xE817 << 15, true},
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
}
for m, e := range vec3 {
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
l64VecBank[m] = e.lasx
}
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
// range, bias and field mask are the ones the toolchain encodes: vandi.b
// stores the raw 8-bit constant, vsrai.b stores imm+8 (byte-lane bias),
// vseqi.b and vseqi.d store 5-bit and 7-bit two's-complement values.
// The mnemonics that also have a register form (VSEQB, VSEQV, VSRAB,
// VROTRW) keep their three-register entry in l64InstrTable; the
// dispatcher picks the immediate opcode from l64VecImmInfo by operand
// kind, so the immediate entries must not overwrite the table.
vecImm := map[string]l64VecImmEnc{
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -64, 63, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -64, 63, 0, 0x7F},
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
}
for m, e := range vecImm {
l64VecImmInfo[m] = e
l64VecBank[m] = e.lasx
}
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
// vsetallnez.*): the sub-op rides in the rk field.
vecCf := map[string]uint32{
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
}
for m, op := range vecCf {
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
l64VecBank[m] = strings.HasPrefix(m, "XV")
}
// Lane popcount: INSTR vj, vd (the 2R layout with the opcode extending
// over the unused vk field).
vec2r := map[string]l64Vec3Enc{
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
}
for m, e := range vec2r {
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
l64VecBank[m] = e.lasx
l64Vec2R[m] = true
}
}
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
+261
View File
@@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) {
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
}
}
// The X/V spellings name the LSX/LASX vector banks, a register class of
// their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0`
// with "unrecognized instruction" while assembling `VADDV V0, V1, V2`
// and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector
// operands resolve through loong64VecRegNum only.
vecCases := map[string]int{
"V0": 0, "V31": 31, "X0": 0, "X31": 31,
"R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1,
}
for name, want := range vecCases {
if got := loong64VecRegNum(name); got != want {
t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want)
}
}
}
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
@@ -328,3 +342,250 @@ TEXT ·f(SB), NOSPLIT, $0-0
0x4C000020, // jirl r0, r1, 0 (RET)
)
}
// TestLOONG64_vector pins the LSX/LASX slice against words read off
// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own
// loong64enc1.s): the three-register forms, the immediate forms with their
// biases, the vector-to-condition forms, lane popcount, the FP conversion,
// FSEL and the VMOVQ move family.
func TestLOONG64_vector(t *testing.T) {
t.Run("three-register and immediate forms", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VADDV V1, V2, V3
VADDW V1, V2, V3
VADDV V2, V1
VANDV V1, V2
VXORV V1, V2, V3
VSEQB V1, V2, V3
VSEQV V1, V2, V3
VSRAB V1, V2, V3
VROTRW V1, V2, V3
VANDB $0, V2, V3
VANDB $255, V2
VSEQB $3, V2, V3
VSEQV $15, V2, V3
VSEQV $-15, V2, V3
VSRAB $7, V1, V2
VROTRW $16, V1, V2
VPCNTV V1, V2
XVADDV X1, X2, X3
XVXORV X1, X2, X3
XVSEQB X1, X2, X3
XVPCNTV X1, X2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x700B8443, // vadd.v v3, v2, v1
0x700B0443, // vadd.w
0x700B8821, // vadd.v v1, v1, v2 (two-operand form)
0x71260442, // vand.v v2, v2, v1
0x71270443, // vxor.v
0x70000443, // vseq.b
0x70018443, // vseq.d
0x70EC0443, // vsra.b
0x70EF0443, // vrotr.w
0x73D00043, // vandi.b v3, v2, 0
0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form)
0x72800C43, // vseqi.b v3, v2, 3
0x7281BC43, // vseqi.d v3, v2, 15
0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement)
0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8)
0x72A0C022, // vrotri.w v2, v1, 16
0x729C2C22, // vpcnt.d v2, v1
0x740B8443, // xvadd.d x3, x2, x1
0x75270443, // xvxor.d
0x74000443, // xvseq.b
0x769C2C22, // xvpcnt.d x2, x1
0x4C000020,
)
})
t.Run("vector-to-condition", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETNEV V1, FCC0
VSETANYEQB V1, FCC0
VSETANYEQV V2, FCC0
VSETALLNEV V0, FCC0
XVSETNEV X1, FCC0
XVSETALLNEV X1, FCC0
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x729C9C20, // vsetnez.d fcc0, v1
0x729CA020, // vsetanyeqz.b
0x729CAC40, // vsetanyeqz.d
0x729CBC00, // vsetallnez.d
0x769C9C20, // xvsetnez.d
0x769CBC20, // xvsetallnez.d
0x4C000020,
)
})
t.Run("FP convert and FSEL", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
FFINTDV F0, F1
FSEL FCC0, F3, F4, F3
FSEL FCC1, F1, F2
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x011D2801, // ffint.d.v f1, f0
0x0D000C83, // fsel f3, f4, f3, fcc0
0x0D008442, // fsel f2, f2, f1, fcc1
0x4C000020,
)
})
t.Run("VMOVQ move family", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMOVQ V1, V9
VMOVQ (R4), V2
VMOVQ 16(R4), V2
VMOVQ V0, (R4)
VMOVQ V0, 32(R4)
VMOVQ (R4)(R7), V3
VMOVQ V3, (R4)(R7)
VMOVQ R6, V0.B16
VMOVQ R6, V12.W4
VMOVQ (R4), V4.W4
XVMOVQ X3, X7
XVMOVQ (R4), X2
XVMOVQ X0, (R4)
XVMOVQ (R4)(R7), X4
XVMOVQ X0, (R4)(R7)
XVMOVQ R6, X0.B32
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x732D0029, // vori.b v9, v1, 0 (register move)
0x2C000082, // vld v2, r4, 0
0x2C004082, // vld v2, r4, 16
0x2C400080, // vst v0, r4, 0
0x2C408080, // vst v0, r4, 32
0x38401C83, // vldx v3, r4, r7
0x38441C83, // vstx v3, r4, r7
0x729F00C0, // vreplgr2vr.b v0, r6
0x729F08CC, // vreplgr2vr.w v12, r6
0x30200084, // vldrepl.w v4, r4, 0
0x772D0067, // xvori.b x7, x3, 0
0x2C800082, // xvld x2, r4, 0
0x2CC00080, // xvst x0, r4, 0
0x38481C84, // xvldx x4, r4, r7
0x384C1C80, // xvstx x0, r4, r7
0x769F00C0, // xvreplgr2vr.b x0, r6
0x4C000020,
)
})
t.Run("element extract and insert", func(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VMOVQ V0.V[0], R10
VMOVQ V6.V[1], R8
VMOVQ R9, V1.V[0]
XVMOVQ X0.V[0], R10
XVMOVQ X5.W[7], R7
XVMOVQ R4, X7.V[3]
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72EFF00A, // vpickve2gr.d r10, v0, 0
0x72EFF4C8, // vpickve2gr.d r8, v6, 1
0x72EBF121, // vinsgr2vr.d v1, r9, 0
0x76EFE00A, // xvpickve2gr.d r10, x0, 0
0x76EFDCA7, // xvpickve2gr.w r7, x5, 7
0x76EBEC87, // xvinsgr2vr.d x7, r4, 3
0x4C000020,
)
})
}
// TestLOONG64_vectorErrors pins the register-class and range diagnostics of
// the vector slice; each shape is rejected by the oracle as well
// (GOARCH=loong64 go tool asm).
func TestLOONG64_vectorErrors(t *testing.T) {
cases := []string{
// Integer registers in vector positions.
`TEXT ·e(SB), NOSPLIT, $0
VADDV R4, R5, R6
RET
`,
// Crossed banks: LSX spellings take V, LASX spellings X.
`TEXT ·e(SB), NOSPLIT, $0
VADDV X1, X2, X3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVADDV V1, V2, V3
RET
`,
// The LASX bank has no .b/.h element forms.
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ R4, X2.B[0]
RET
`,
// Immediate ranges.
`TEXT ·e(SB), NOSPLIT, $0
VANDB $256, V2
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VSEQB $16, V2, V3
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VROTRW $32, V1, V2
RET
`,
// VSET* wants an FCC flag, not a vector register.
`TEXT ·e(SB), NOSPLIT, $0
VSETNEV V1, V2
RET
`,
}
for i, src := range cases {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants.
// The oracle words come from GOARCH=loong64 go tool objdump of kernels
// assembled with go tool asm, and match the toolchain's loong64enc1.s.
func TestLOONG64_dbarAtomics(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·atoms(SB), NOSPLIT, $0
AMADDDBW R14, (R13), R12
AMADDDBV R14, (R13), R12
AMANDDBW R5, (R4), R6
AMANDDBV R5, (R4), R6
AMORDBW R5, (R4), R0
AMORDBV R5, (R4), R6
AMSWAPDBW R5, (R4), R6
AMCASDBV R6, (R4), R5
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x386A39AC, // amadd_db.w r12, r13, r14
0x386AB9AC, // amadd_db.d
0x386B1486, // amand_db.w r6, r4, r5
0x386B9486, // amand_db.d
0x386C1480, // amor_db.w r0, r4, r5
0x386C9486, // amor_db.d
0x38691486, // amswap_db.w
0x385B9885, // amcas_db.w
0x4C000020,
)
}
+4 -1
View File
@@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42
}
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
// invalid register operands and operand-count mismatches.
// invalid register operands and operand-count mismatches. The X0 and
// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects
// `BEQZ X0` (the X bank is not an integer register) and the two-register
// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`).
func TestLOONG64_errors(t *testing.T) {
cases := []string{
`TEXT ·e(SB), NOSPLIT, $0
+6 -1
View File
@@ -17,12 +17,13 @@ import "strings"
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0-K7.
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register
}
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
@@ -144,6 +145,10 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ {
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
}
// x87 stack: F0..F7.
for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
}
return m
}
+567 -3
View File
@@ -224,6 +224,87 @@ func riscvInstrSize(instr *ast.Instr, fi riscvFrameInfo) int {
}
return riscvItypeImmediateSize(mnem, imm)
}
// The toolchain's synthesised instructions: some emit one word, others
// expand to a fixed sequence.
return riscvExtendedSize(mnem, ops)
}
// riscvExtendedSize returns the encoded size of the instructions the
// toolchain synthesises from other instructions (the ternary expansions and
// the vector slice); every caller keeps the layout in step with
// encodeRISCVExtended, which emits exactly these bytes.
func riscvExtendedSize(mnem string, ops []*ast.Operand) int {
switch mnem {
case "NOP":
// The toolchain drops a bare NOP entirely.
return 0
case "ANDN", "ORN":
return 8
case "MAX", "MAXU", "MIN", "MINU":
if riscvIdenticalMinMax(mnem, ops) {
rd := regFromOperand(ops[1])
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rd != 0 {
return 2 // C.MV, or C.LI when the sources are X0
}
return 4
}
return 20
case "ROR", "RORW":
if len(ops) >= 1 && isImmOperand(ops[0]) {
// SRL + [compressed] SLL of the reverse shift + OR.
return 4 + riscvRevShiftSize(mnem, ops) + 4
}
return 16 // SUB + shift + shift + OR
case "RORIW":
return 12
}
return 4
}
// riscvIdenticalMinMax reports whether a MIN/MAX sees two identical source
// registers (the toolchain folds that to ADDI $0).
func riscvIdenticalMinMax(mnem string, ops []*ast.Operand) bool {
if mnem != "MAX" && mnem != "MAXU" && mnem != "MIN" && mnem != "MINU" {
return false
}
if len(ops) != 2 && len(ops) != 3 {
return false
}
rs1 := regFromOperand(ops[1])
rs2 := regFromOperand(ops[0])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 == rd {
// The toolchain swaps the sources so the destination-identical one
// is processed first; identical sources stay identical.
rs1, rs2 = rs2, rs1
}
return rs1 >= 0 && rs1 == rs2
}
// riscvRevShiftSize returns the size of the reverse-shift instruction inside
// a ROR/RORW immediate expansion: the SLLI of the complementary amount, which
// compresses to C.SLLI only in the 64-bit form when rd == rs1, both non-zero,
// and the amount lands in 1-63. The W forms have no compressed shift.
func riscvRevShiftSize(mnem string, ops []*ast.Operand) int {
if mnem != "ROR" {
return 4 // SLLIW has no compressed form
}
imm := int(immFromOperand(ops[0]))
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
sll := (-imm) & 63
if rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 {
return 2 // C.SLLI
}
return 4
}
@@ -482,6 +563,16 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
return []byte{byte(word), byte(word >> 8), byte(word >> 16), byte(word >> 24)}, nil
}
// The toolchain's synthesised instructions and the RVV slice: expanded
// encodings the main table does not carry. FSGNJD is a plain table
// entry and stays with the FP arithmetic path.
if code, handled, err := encodeRISCVExtended(mnem, instr, pc, offsets); handled {
if err != nil {
return nil, err
}
return code, nil
}
enc, ok := riscvInstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported RISC-V instruction %q", mnem)
@@ -573,9 +664,11 @@ func encodeRISCVInstr(instr *ast.Instr, pc int, offsets map[string]int, fi riscv
}
word = riscvSType(enc, rs1, rs2, imm)
// LR (load-reserved): INSTR (addr), dst, 2 operands.
// LR (load-reserved): INSTR (addr), dst. The toolchain reads the
// operands positionally, so the base register comes from the first
// operand and the destination from the second whatever their parens.
case len(ops) == 2 && isLRInstr(mnem):
rs1, _ := memFromOperandWithFrame(ops[0], fi)
rs1 := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rd < 0 || rs1 < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
@@ -1475,6 +1568,477 @@ func extractITypeParams(instr *ast.Instr) (rd, rs1 int, imm int32) {
return
}
// ---- toolchain-synthesised instructions and the RVV slice ----
// encodeRISCVExtended encodes the instructions the Go toolchain synthesises
// from other instructions (ANDN/ORN, MIN/MAX, ROR and friends, the branch
// pseudos and FABSD), the CSR read RDTIME, and the RVV vector slice the
// compiler's kernels use. handled reports whether the mnemonic belongs to
// this group; err carries the diagnostic when it does but cannot be encoded.
// Each expansion reproduces the toolchain's instruction-for-instruction
// sequence, including its use of X31 (TMP) and its RVC compression.
func encodeRISCVExtended(mnem string, instr *ast.Instr, pc int, offsets map[string]int) ([]byte, bool, error) {
ops := instr.Operands
switch mnem {
case "NOP":
if len(ops) != 0 {
return nil, true, fmt.Errorf("NOP takes no operands")
}
// The toolchain drops a bare NOP: no bytes at all.
return nil, true, nil
case "RDTIME":
// RDTIME rd reads the time CSR through CSRRS with a zero source.
if len(ops) != 1 {
return nil, true, fmt.Errorf("RDTIME expects 1 operand, got %d", len(ops))
}
rd := regFromOperand(ops[0])
if rd < 0 {
return nil, true, fmt.Errorf("RDTIME: invalid register")
}
return wordLE(riscvIType(riscvEnc{0x73, 0x2, 0x00}, rd, 0, 0xC01)), true, nil
case "NEG", "NOT", "SEQZ":
if len(ops) != 1 && len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 1 or 2 operands, got %d", mnem, len(ops))
}
rs := regFromOperand(ops[0])
rd := rs
if len(ops) == 2 {
rd = regFromOperand(ops[1])
}
if rs < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
var word uint32
switch mnem {
case "NEG":
word = riscvRType(riscvInstrTable["SUB"], rd, 0, rs)
case "NOT":
word = riscvIType(riscvInstrTable["XORI"], rd, rs, -1)
case "SEQZ":
word = riscvIType(riscvInstrTable["SLTIU"], rd, rs, 1)
}
return wordLE(word), true, nil
case "ANDN", "ORN":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
rs2 := regFromOperand(ops[0]) // the operand to invert
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
notReg := rd
if rs1 == notReg {
notReg = 31 // TMP, when the destination would be clobbered
}
out := wordLE(riscvIType(riscvInstrTable["XORI"], notReg, rs2, -1))
op := riscvInstrTable["AND"]
if mnem == "ORN" {
op = riscvInstrTable["OR"]
}
return append(out, wordLE(riscvRType(op, rd, rs1, notReg))...), true, nil
case "MAX", "MAXU", "MIN", "MINU":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
if rs1 == rd {
// Process the destination-identical source first, as the
// toolchain does, so the sequence stays in place.
rs1, rs2 = rs2, rs1
}
if rs1 == rs2 {
// Identical inputs fold to ADDI $0 (compressed to C.MV and
// friends by the toolchain's compressor).
return riscvFoldedMove(rd, rs1), true, nil
}
slt1, slt2 := rs2, rs1
cmp := riscvInstrTable["SLT"]
if mnem == "MAX" || mnem == "MAXU" {
slt1, slt2 = slt2, slt1
}
if mnem == "MAXU" || mnem == "MINU" {
cmp = riscvInstrTable["SLTU"]
}
var out []byte
out = append(out, wordLE(riscvRType(cmp, 31, slt1, slt2))...) // the compare into TMP
out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, 31))...) // NEG TMP
out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rs2))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["AND"], rd, 31, rd))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["XOR"], rd, rs1, rd))...)
return out, true, nil
case "ROR", "RORW", "RORIW":
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
if isImmOperand(ops[0]) {
// Immediate rotate: SRLI the amount, SLLI the complement, OR.
imm := int(immFromOperand(ops[0]))
shiftW := 63
srlEnc := riscvInstrTable["SRLI"]
sllEnc := riscvInstrTable["SLLI"]
if mnem != "ROR" {
shiftW = 31
srlEnc = riscvInstrTable["SRLIW"]
sllEnc = riscvInstrTable["SLLIW"]
}
if imm < 0 || imm > shiftW {
return nil, true, fmt.Errorf("%s: shift amount out of range [0, %d]", mnem, shiftW)
}
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
var out []byte
out = append(out, wordLE(riscvRType(srlEnc, 31, rs1, imm))...)
sll := (-imm) & shiftW
if mnem == "ROR" && rd == rs1 && rd != 0 && sll >= 1 && sll <= 63 {
out = append(out, word16(rvcSLLI(uint32(rd), uint32(sll)))...) // C.SLLI
} else {
out = append(out, wordLE(riscvRType(sllEnc, rd, rs1, sll))...)
}
return append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...), true, nil
}
// Register rotate: OR of the two opposite shifts through TMP.
if mnem == "RORIW" {
return nil, true, fmt.Errorf("RORIW takes an immediate shift amount")
}
rs2 := regFromOperand(ops[0])
rs1 := regFromOperand(ops[1])
rd := rs1
if len(ops) == 3 {
rd = regFromOperand(ops[2])
}
if rs1 < 0 || rs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
sllEnc := riscvInstrTable["SLL"]
srlEnc := riscvInstrTable["SRL"]
if mnem == "RORW" {
sllEnc = riscvInstrTable["SLLW"]
srlEnc = riscvInstrTable["SRLW"]
}
var out []byte
out = append(out, wordLE(riscvRType(riscvInstrTable["SUB"], 31, 0, rs2))...) // NEG
out = append(out, wordLE(riscvRType(sllEnc, 31, rs1, 31))...)
out = append(out, wordLE(riscvRType(srlEnc, rd, rs1, rs2))...)
out = append(out, wordLE(riscvRType(riscvInstrTable["OR"], rd, 31, rd))...)
return out, true, nil
case "BGT", "BGTU", "BLE", "BLEU":
// The reversed conditional branches: BGT a, b, label is BLT b, a.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
a := regFromOperand(ops[0])
b := regFromOperand(ops[1])
if a < 0 || b < 0 {
return nil, true, fmt.Errorf("%s: invalid register", mnem)
}
target := labelFromOperand(ops[2])
targetOff, ok := offsets[target]
if !ok {
return nil, true, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
offset := int32(targetOff - pc)
if err := riscvCheckBranchOffset(target, offset); err != nil {
return nil, true, err
}
var enc riscvEnc
switch mnem {
case "BGT":
enc = riscvEnc{0x63, 0x4, 0x00} // blt b, a
case "BGTU":
enc = riscvEnc{0x63, 0x6, 0x00} // bltu b, a
case "BLE":
enc = riscvEnc{0x63, 0x5, 0x00} // bge b, a
case "BLEU":
enc = riscvEnc{0x63, 0x7, 0x00} // bgeu b, a
}
return wordLE(riscvBType(enc, b, a, offset)), true, nil
case "FABSD":
// FABSD rs, rd is FSGNJX.D (sign XOR, funct3 2) with the source in
if len(ops) != 2 {
return nil, true, fmt.Errorf("FABSD expects 2 operands, got %d", len(ops))
}
rs := regFromOperand(ops[0])
rd := regFromOperand(ops[1])
if rs < 0 || rd < 0 {
return nil, true, fmt.Errorf("FABSD: invalid register")
}
return wordLE(riscvRType(riscvEnc{0x53, 0x2, 0x11}, rd, rs, rs)), true, nil
default:
return encodeRISCVVector(mnem, ops)
}
}
// riscvFoldedMove emits the ADDI $0, rs, rd the toolchain folds identical
// MIN/MAX inputs into, with the same compression its compressor applies to
// the folded form.
func riscvFoldedMove(rd, rs int) []byte {
switch {
case rd != 0 && rs != 0:
return word16(rvcCR(0x8, uint32(rd), uint32(rs))) // C.MV
case rd == 0 && rs == 0:
return word16(0x0001) // C.NOP
case rs == 0:
return word16(rvcCI(0x2, uint32(rd), 0)) // C.LI rd, $0
default:
return wordLE(riscvIType(riscvEnc{0x13, 0x0, 0x00}, rd, rs, 0))
}
}
// encodeRISCVVector encodes the RVV slice GOROOT's kernels use. Registers
// are accepted in either spelling: the vector V registers and the integer
// registers share their 5-bit numbers, and the superset keeps hand-written
// probes simple. handled is always true: every name reaching here is one of
// the vector mnemonics.
func encodeRISCVVector(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
reg := regFromOperand
switch mnem {
case "VSETVLI", "VSETIVLI":
// INSTR avl, vsew, vlmul, vta, vma, rd.
if len(ops) != 6 {
return nil, true, fmt.Errorf("%s expects 6 operands, got %d", mnem, len(ops))
}
avl := 0
if isImmOperand(ops[0]) {
avl = int(immFromOperand(ops[0]))
if avl < 0 || avl > 31 {
return nil, true, fmt.Errorf("%s: avl immediate out of range [0, 31]", mnem)
}
} else {
avl = reg(ops[0])
if avl < 0 {
return nil, true, fmt.Errorf("%s: invalid avl register", mnem)
}
}
if mnem == "VSETIVLI" && !isImmOperand(ops[0]) {
return nil, true, fmt.Errorf("VSETIVLI expects an immediate avl")
}
vsew, err := riscvVTypeToken(operandRegName(ops[1]), "E", map[string]int{"8": 0, "16": 1, "32": 2, "64": 3})
if err != nil {
return nil, true, fmt.Errorf("%s: %w", mnem, err)
}
vlmul, err := riscvVTypeToken(operandRegName(ops[2]), "M", map[string]int{"1": 0, "2": 1, "4": 2, "8": 3, "F8": 5, "F4": 6, "F2": 7})
if err != nil {
return nil, true, fmt.Errorf("%s: %w", mnem, err)
}
vta := 0
switch operandRegName(ops[3]) {
case "TA":
vta = 1
case "TU":
default:
return nil, true, fmt.Errorf("%s: invalid tail policy %q (want TA or TU)", mnem, operandRegName(ops[3]))
}
vma := 0
switch operandRegName(ops[4]) {
case "MA":
vma = 1
case "MU":
default:
return nil, true, fmt.Errorf("%s: invalid mask policy %q (want MA or MU)", mnem, operandRegName(ops[4]))
}
rd := reg(ops[5])
if rd < 0 {
return nil, true, fmt.Errorf("%s: invalid destination register", mnem)
}
// An immediate avl always encodes as vsetivli, even under the
// VSETVLI spelling: the toolchain canonicalises the pair, and
// `VSETVLI $15` and `VSETIVLI $15` come out byte-identical
// (0xcd07f657) from GOARCH=riscv64 go tool asm.
ivli := mnem == "VSETIVLI" || isImmOperand(ops[0])
return wordLE(riscvVSetEnc(ivli, avl, riscvVType(vsew, vlmul, vta, vma), rd)), true, nil
case "VLE8V":
// Unit-stride load: INSTR (base), vd.
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rs1, ok := riscvVecMem(ops[0])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
vd := reg(ops[1])
if vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
return wordLE(riscvVLSType(0x07, 0, 0, 0, 0, rs1, vd)), true, nil
case "VSE8V", "VSE32V":
// Unit-stride store: INSTR vs3, (base).
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vs3 := reg(ops[0])
rs1, ok := riscvVecMem(ops[1])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
if vs3 < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
width := 0
if mnem == "VSE32V" {
width = 6
}
return wordLE(riscvVLSType(0x27, 0, 0, width, 0, rs1, vs3)), true, nil
case "VLSSEG4E32V", "VLSSEG8E32V":
// Constant-stride segmented load: INSTR (base), stride, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs1, ok := riscvVecMem(ops[0])
if !ok {
return nil, true, fmt.Errorf("%s: invalid memory operand", mnem)
}
rs2 := reg(ops[1])
vd := reg(ops[2])
if rs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid register operand", mnem)
}
nf := 3 // 4 fields
if mnem == "VLSSEG8E32V" {
nf = 7 // 8 fields
}
return wordLE(riscvVLSType(0x07, nf, 2, 6, int32(rs2), rs1, vd)), true, nil
case "VADDVV", "VXORVV", "VMSNEVV":
// Vector-vector: INSTR vs1, vs2, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
vs1, vs2, vd := reg(ops[0]), reg(ops[1]), reg(ops[2])
if vs1 < 0 || vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
funct6 := map[string]int{"VADDVV": 0x00, "VXORVV": 0x0B, "VMSNEVV": 0x19}[mnem]
return wordLE(riscvVVInstr(funct6, riscvVf3VV, int32(vs1), vs2, vd)), true, nil
case "VADDVX", "VMSEQVX":
// Vector-scalar: INSTR rs1, vs2, vd (the scalar in the rs1 field).
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rs1, vs2, vd := reg(ops[0]), reg(ops[1]), reg(ops[2])
if rs1 < 0 || vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid register operand", mnem)
}
funct6 := 0x00
if mnem == "VMSEQVX" {
funct6 = 0x18
}
return wordLE(riscvVVInstr(funct6, riscvVf3VX, int32(rs1), vs2, vd)), true, nil
case "VSLLVI", "VSRLVI":
// Vector-immediate shift: INSTR $uimm, vs2, vd.
if len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
imm := int(immFromOperand(ops[0]))
if imm < 0 || imm > 31 {
return nil, true, fmt.Errorf("%s: immediate out of range [0, 31]", mnem)
}
vs2, vd := reg(ops[1]), reg(ops[2])
if vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("%s: invalid vector register", mnem)
}
funct6 := 0x25 // vsll.vi
if mnem == "VSRLVI" {
funct6 = 0x28 // vsrl.vi
}
return wordLE(riscvVVInstr(funct6, riscvVf3VI, int32(imm), vs2, vd)), true, nil
case "VFIRSTM":
// vmfirst.m rd, vs2: the unmasked form carries 0x11 in the rs1 field
// and sets the mask bit (funct7 = 0x20 | 1).
if len(ops) != 2 {
return nil, true, fmt.Errorf("VFIRSTM expects 2 operands, got %d", len(ops))
}
vs2, rd := reg(ops[0]), reg(ops[1])
if vs2 < 0 || rd < 0 {
return nil, true, fmt.Errorf("VFIRSTM: invalid register operand")
}
return wordLE(riscvVUnaryInstr(0x10, riscvVf3MV, 0x11, vs2, rd)), true, nil
case "VIDV":
// vid.v vd (vs2 must be v0; the unmasked form sets the mask bit).
if len(ops) != 1 {
return nil, true, fmt.Errorf("VIDV expects 1 operand, got %d", len(ops))
}
vd := reg(ops[0])
if vd < 0 {
return nil, true, fmt.Errorf("VIDV: invalid vector register")
}
return wordLE(riscvVUnaryInstr(0x14, riscvVf3MV, 0x11, 0, vd)), true, nil
case "VMV4RV":
// vmv4r.v vd, vs2: whole-register group move.
if len(ops) != 2 {
return nil, true, fmt.Errorf("VMV4RV expects 2 operands, got %d", len(ops))
}
vs2, vd := reg(ops[0]), reg(ops[1])
if vs2 < 0 || vd < 0 {
return nil, true, fmt.Errorf("VMV4RV: invalid vector register")
}
return wordLE(riscvVUnaryInstr(0x27, 0x3, 0x3, vs2, vd)), true, nil
}
return nil, false, nil
}
// riscvVTypeToken parses a vsetvli configuration token (E8, M8, MF2 and
// friends): the letter prefix selects the field and the suffix its value
// through the given table.
func riscvVTypeToken(name, prefix string, codes map[string]int) (int, error) {
if len(name) <= len(prefix) || name[:len(prefix)] != prefix {
return 0, fmt.Errorf("invalid vtype token %q (want %s<width>)", name, prefix)
}
code, ok := codes[name[len(prefix):]]
if !ok {
return 0, fmt.Errorf("invalid vtype token %q", name)
}
return code, nil
}
// riscvVecMem reads a vector memory operand: a bare base register, the only
// addressing form the vector loads and stores carry. Frame-pseudo bases are
// rejected: the toolchain resolves no frame reference on the vector forms.
func riscvVecMem(op *ast.Operand) (rs1 int, ok bool) {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return -1, false
}
if op.Addr.Base == "" || op.Addr.Offset != 0 {
return -1, false
}
rs1 = riscvRegNum(op.Addr.Base)
return rs1, rs1 >= 0
}
// Instruction type classifiers.
func isRTypeInstr(m string) bool {
switch m {
@@ -1547,7 +2111,7 @@ func isFPArithInstr(m string) bool {
switch m {
case "FADDS", "FSUBS", "FMULS", "FDIVS",
"FADDD", "FSUBD", "FMULD", "FDIVD",
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD":
"FSQRTS", "FSQRTD", "FMINS", "FMAXS", "FMIND", "FMAXD", "FSGNJD":
return true
}
return false
+125 -27
View File
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
case "F31", "FT11":
return 31
default:
// Vector registers V0-V31 (the "V" extension). They share the
// register numbering with the integer file: a bare number 0-31.
if len(name) >= 2 && name[0] == 'V' {
if n, ok := parseRegDigits(name[1:], 31); ok {
return n
}
}
return -1
}
}
// parseRegDigits parses a decimal register suffix and reports whether it is
// within [0, max].
func parseRegDigits(digits string, max int) (int, bool) {
if digits == "" {
return 0, false
}
n := 0
for i := 0; i < len(digits); i++ {
if digits[i] < '0' || digits[i] > '9' {
return 0, false
}
n = n*10 + int(digits[i]-'0')
if n > max {
return 0, false
}
}
return n, true
}
// RISC-V instruction encoding parameters.
type riscvEnc struct {
opcode uint32 // bits [6:0]
@@ -232,25 +258,28 @@ var riscvInstrTable = map[string]riscvEnc{
"JALR": {0x67, 0x0, 0x00},
// RV64A, atomics (AMO opcode 0x2F).
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
"AMOORW": {0x2F, 0x2, 0x06 << 2},
"AMOORD": {0x2F, 0x3, 0x06 << 2},
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
// the lower two, exactly as the toolchain writes them: every AMO sets both
// aq and rl (funct7 |= 3).
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
// RV64F/D, floating-point arithmetic.
"FADDS": {0x53, 0x0, 0x00},
@@ -273,12 +302,16 @@ var riscvInstrTable = map[string]riscvEnc{
"FMAXS": {0x53, 0x1, 0x14},
"FMIND": {0x53, 0x0, 0x15},
"FMAXD": {0x53, 0x1, 0x15},
// FP sign injection (double): rs2 carries the sign source.
"FSGNJD": {0x53, 0x0, 0x11},
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
"LRW": {0x2F, 0x2, 0x02 << 2},
"LRD": {0x2F, 0x3, 0x02 << 2},
"SCW": {0x2F, 0x2, 0x03 << 2},
"SCD": {0x2F, 0x3, 0x03 << 2},
// The toolchain gives LR acquire ordering (aq = 1) and SC release
// ordering (rl = 1).
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
// FP compare, result in integer register (funct7 0x50/0x51).
"FEQS": {0x53, 0x2, 0x50},
@@ -296,11 +329,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
}
// riscvAMOType encodes an atomic (AMO) instruction.
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
// (the table stores the full field, so the word needs no reassembly).
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
}
@@ -441,6 +474,71 @@ func riscvJType(rd int, offset int32) uint32 {
0x6F // JAL opcode
}
// ---- RVV ("V" extension) encoding helpers ----
// The OP-V major opcode and its funct3 subclasses.
const (
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
riscvVf3VV = 0x0 // vector-vector
riscvVf3MV = 0x2 // vector mask
riscvVf3VI = 0x3 // vector-immediate
riscvVf3VX = 0x4 // vector-scalar
riscvVf3Cfg = 0x7 // vsetvli
)
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
// group multiplier in [2:0], the selected element width in [5:3] and the
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
func riscvVType(vsew, vlmul, vta, vma int) int {
return vlmul | vsew<<3 | vta<<6 | vma<<7
}
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
// the toolchain writes above its 10-bit vtype).
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
imm := vtype & 0x3FF
if vsetivli {
imm |= 0xC00
}
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
uint32(rd)<<7 | riscvOpV
}
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
// segment count in bits [31:29], the addressing mode in bits [28:26], the
// unmasked bit at 25 and the width in funct3. width follows the load
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
uint32(rd)<<7 | op
}
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
// the standard positions. vs1 may name an integer register for the *VX forms
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
}
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
// fields live where the second source register would be: rs1Field and vs2 are
// written verbatim (the oracle writes fixed non-zero constants there for some
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
}
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
func riscvSegNF(n int) int32 { return int32(n - 1) }
// ---- RVC (compressed) encoding helpers ----
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
+223
View File
@@ -5,6 +5,8 @@ package asm
import (
"bytes"
"encoding/binary"
"encoding/hex"
"strings"
"testing"
@@ -971,3 +973,224 @@ TEXT ·edge(SB), NOSPLIT, $0
t.Errorf("int32-span immediates must assemble: %v", err)
}
}
// riscvWants decodes code as little-endian words and pins each one; the
// expected values below were read off GOARCH=riscv64 go tool objdump of
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWants(t *testing.T, code []byte, want ...uint32) {
t.Helper()
got := make([]uint32, 0, len(code)/4)
for i := 0; i+4 <= len(code); i += 4 {
got = append(got, binary.LittleEndian.Uint32(code[i:]))
}
if len(got) < len(want) {
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
}
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// riscvWantsHex pins the exact hex encoding of a function's instruction
// bytes, including any 2-byte compressed instructions in the stream; the
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
// assembled with go tool asm (the toolchain's riscv64.s testdata
// cross-checks the same words).
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
t.Helper()
got := hex.EncodeToString(code)
if got != wantHex {
t.Errorf("code = %s, want %s", got, wantHex)
}
}
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
// the time CSR).
func TestRISCV_extendedPseudos(t *testing.T) {
t.Run("logic and minmax", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·l(SB), NOSPLIT, $0
ANDN X19, X20, X21
ANDN X19, X20
ORN X20, X19
MAX X26, X28, X29
MIN X29, X30, X5
MAX X5, X5
MAX X5, X5, X6
SEQZ X5, X6
NEG X5, X6
NOT X5
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
// then SEQZ, NEG, NOT and RDTIME.
riscvWantsHex(t, code,
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
"9682"+"1683"+
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
})
t.Run("rotate", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·r(SB), NOSPLIT, $0
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORIW $31, X13, X14
RORIW $1, X14, X15
RORIW $3, X14
RORW X15, X16, X17
RORW $31, X13
RET
`)
code := assembleRISCVHelper(t, fn)
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
riscvWantsHex(t, code,
"b30fa040b39ff50133d6a50033e6cf00"+
"b30fa040b39ff501b3d5a500b3e5bf00"+
"93dff5038605b3e5bf00"+
"9bdff6011b97160033e7ef00"+
"9b5f17009b17f701b3e7ff00"+
"9b5f37001b17d70133e7ef00"+
"b30ff040bb1ff801bb58f800b3e81f01"+
"9bdff6019b961600b3e6df00"+"67800000")
})
t.Run("fp and branches", func(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
FABSD F1, F2
FSGNJD F1, F0, F2
FMADDD F1, F2, F3, F4
FMSUBD F1, F2, F3, F4
FNMSUBD F1, F2, F3, F4
BGT X5, X6, tgt
BLE X5, X6, tgt
BGTU X5, X6, tgt
BLEU X5, X6, tgt
tgt:
RDTIME X5
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWantsHex(t, code,
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
"f32210c0"+"67800000")
})
}
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
// exactly as GOARCH=riscv64 go tool asm encodes them.
func TestRISCV_amoWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0
AMOSWAPW X5, (X6), X7
AMOSWAPD X5, (X6), X7
AMOADDW X5, (X6), X7
AMOADDD X5, (X6), X7
AMOANDW X5, (X6), X7
AMOANDD X5, (X6), X7
AMOORW X5, (X6), X7
AMOORD X5, (X6), X7
AMOXORW X5, (X6), X7
AMOXORD X5, (X6), X7
AMOMAXW X5, (X6), X7
AMOMAXD X5, (X6), X7
AMOMAXUW X5, (X6), X7
AMOMAXUD X5, (X6), X7
AMOMINUW X5, (X6), X7
AMOMINUD X5, (X6), X7
LRW (X5), X6
LRD (X5), X6
SCW X5, (X6), X7
SCD X5, (X6), X7
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0E5323AF, // amoswap.w
0x0E5333AF, // amoswap.d
0x065323AF, // amoaddd.w
0x065333AF, // amoadd.d
0x665323AF, // amoand.w
0x665333AF, // amoand.d
0x465323AF, // amoor.w
0x465333AF, // amoor.d
0x265323AF, // amoxor.w
0x265333AF, // amoxor.d
0xA65323AF, // amomax.w
0xA65333AF, // amomax.d
0xE65323AF, // amomaxu.w
0xE65333AF, // amomaxu.d
0xC65323AF, // amominu.w
0xC65333AF, // amominu.d
0x1402A32F, // lr.w (aq)
0x1402B32F, // lr.d
0x1A5323AF, // sc.w (rl)
0x1A5333AF, // sc.d
)
}
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
// toolchain canonicalises an immediate avl to vsetivli even under the
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
// identical), which is what the 0xC00 bit of the first word carries.
func TestRISCV_vectorWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·v(SB), NOSPLIT, $0
VSETVLI X5, E8, M8, TA, MA, X6
VSETIVLI $4, E32, M1, TA, MA, X0
VSETVLI $15, E32, M1, TA, MA, X12
VADDVV V1, V2, V3
VADDVX X12, V12, V12
VXORVV V8, V16, V24
VMSEQVX X12, V8, V0
VMSNEVV V8, V16, V0
VSLLVI $8, V28, V30
VSRLVI $25, V29, V29
VFIRSTM V0, X6
VIDV V12
VMV4RV V8, V24
VLE8V (X10), V8
VSE8V V24, (X10)
VSE32V V9, (X11)
VLSSEG4E32V (X14), X0, V0
VLSSEG8E32V (X10), X0, V4
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
0xCD027057, // vsetivli x0, 4
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
0x022081D7, // vadd.vv v3, v2, v1
0x02C64657, // vadd.vx v12, v12, x12
0x2F040C57, // vxor.vv v24, v16, v8
0x62864057, // vmseq.vx v0, v8, x12
0x67040057, // vmsne.vv v0, v16, v8
0x97C43F57, // vsll.vi v30, v28, 8
0xA3DCBED7, // vsrl.vi v29, v29, 25
0x4208A357, // vmfirst.m x6, v0
0x5208A657, // vid.v v12
0x9E81BC57, // vmv4r.v v24, v8
0x02050407, // vle8.v v8, (x10)
0x02050C27, // vse8.v v24, (x10)
0x0205E4A7, // vse32.v v9, (x11)
0x6A076007, // vlsseg4e32.v v0, (x14), x0
0xEA056207, // vlsseg8e32.v v4, (x10), x0
)
}
+144 -1
View File
@@ -41,7 +41,7 @@ const (
vexExtract
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
// in ModRM.reg and the destination in r/m, the layout of the EVEX
// narrowing stores (VPMOVDW, VPMOVQD).
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
vexRMRev
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
// vector length follows the source: the packed-double → dword
@@ -52,6 +52,15 @@ const (
vexRMSrcLen
// vexZero is the no-operand form (VZEROUPPER).
vexZero
// vexZeroAll is the no-operand form that zeroes the full upper state
// (VZEROALL, the L = 1 twin of VZEROUPPER).
vexZeroAll
// vexNDS3GPR is the three-operand NDS form over general-purpose
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
vexNDS3GPR
// vexImmRMGPR is the immediate form over general-purpose registers
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
vexImmRMGPR
)
// vexSpec describes one VEX instruction's encoding parameters.
@@ -125,6 +134,12 @@ var vexTable = map[string]vexSpec{
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
// Scalar fused multiply-add (NDS form). The Go assembler carries the
// same 66 prefix as the packed forms on every FMA row, and W1 on the
// double-precision spellings, so SD shares PD's prefix/W pair and the
// scalar width rides on the W bit.
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
// no vvvv).
@@ -192,6 +207,31 @@ var vexTable = map[string]vexSpec{
// VEX.128.0F.W0, no operands.
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
// falls through to the EVEX table.
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
@@ -200,6 +240,14 @@ var vexTable = map[string]vexSpec{
// rm=scalar memory; SD is 256-bit only).
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
// YMM (the encoder rejects an XMM destination, as go tool asm does).
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
// memory destination in rm).
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
// source).
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
@@ -290,6 +338,8 @@ type vexMoveSpec struct {
var vexMoveTable = map[string]vexMoveSpec{
// VEX.128/256.F3.0F.WIG, unaligned integer move.
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, aligned integer move.
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
// VEX.128/256.66.0F.WIG, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
@@ -324,6 +374,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
}
}
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
// only; an XMM destination is rejected exactly as go tool asm does.
if mnemUpper == "VBROADCASTI128" {
dstReg, ok := ops[len(ops)-1].(Reg)
if len(ops) != 2 || !ok || dstReg.size != 32 {
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
}
}
if ms, ok := vexMoveTable[mnemUpper]; ok {
return e.encodeVexMove(mnemUpper, ms, ops)
}
@@ -356,6 +414,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
case vexZero:
return e.encodeVexZero(mnemUpper, spec, ops)
case vexZeroAll:
return e.encodeVexZeroAll(mnemUpper, spec, ops)
case vexNDS3GPR:
return e.encodeVexNDS3GPR(spec, ops)
case vexImmRMGPR:
return e.encodeVexImmRMGPR(spec, ops)
case vexRMRev:
return e.encodeVexRMRev(spec, ops)
}
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
}
@@ -607,6 +673,83 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
return nil
}
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
// twin of VZEROUPPER.
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
if len(ops) != 0 {
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
}
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
return nil
}
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
// rm = src2 and L = 0.
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
}
src2, src1, dst := ops[0], ops[1], ops[2]
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
vvvvReg, ok := src1.(Reg)
if !ok || vvvvReg.isVec() {
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
}
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15-(vvvvReg.idx&15), src2)
}
// encodeVexImmRMGPR encodes the immediate form over general-purpose
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, src, dst := ops[0], ops[1], ops[2]
immVal, ok := imm.(Imm)
if !ok {
return fmt.Errorf("shift control must be an immediate")
}
dstReg, ok := dst.(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("VEX destination must be a general-purpose register")
}
immByte, err := imm8(int64(immVal))
if err != nil {
return err
}
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
return err
}
e.out = append(e.out, immByte)
return nil
}
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
// a store with no register-destination form).
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("store source must be a vector register")
}
if !memOperand(ops[1]) {
return fmt.Errorf("store destination must be memory")
}
rBit := 0
if srcReg.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
}
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
// move uses the store-form layout (reg = source, rm = destination), matching
+59
View File
@@ -19,6 +19,20 @@ func vreg(t *testing.T, name string) Reg {
return r
}
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
// go tool asm in the ground-truth tests instead.
var x86asmUnrecognised = map[string]bool{
"ANDNL": true,
"ANDNQ": true,
"MULXL": true,
"MULXQ": true,
"RORXL": true,
"RORXQ": true,
"VFMADD213SD": true,
"VFNMADD231SD": true,
}
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
// instruction and verifies it round-trips through the x86 decoder to the same
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
@@ -37,8 +51,15 @@ func TestVexNDS3(t *testing.T) {
t.Errorf("%s: Encode: %v", mnem, err)
continue
}
// The x86 decoder's table lacks a handful of rows the Go assembler
// emits (the scalar 213/231 FMA spellings among them); those are
// pinned byte for byte against go tool asm in TestVexGroundTruth
// instead of round-tripped here.
inst, err := x86asm.Decode(code, 64)
if err != nil {
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
continue
}
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
continue
}
@@ -184,6 +205,38 @@ func TestVexGroundTruth(t *testing.T) {
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
// Packed single XOR and byte compare (NDS form).
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
// Octa byte shifts (vvvv carries the destination).
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
// Two-operand test and the non-temporal and broadcast stores.
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
// Aligned integer move and the full zeroing form.
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
// BMI1/BMI2 general-register VEX forms.
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
// Two-operand reg/rm form (v̄vvv must be 1111).
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
@@ -287,6 +340,12 @@ func TestVexGroundTruth(t *testing.T) {
}
inst, err := x86asm.Decode(code, 64)
if err != nil {
// The decoder's AVX/BMI table lacks a few rows the Go
// assembler emits (the GPR VEX forms and the scalar FMA
// spellings); their bytes are the ground truth here.
if x86asmUnrecognised[c.mnem] {
continue
}
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
continue
}
+97 -15
View File
@@ -266,18 +266,62 @@ func probeShapes(a arch.Arch) []string {
// and takes R register spellings.
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
"GE, F0, F1, F2", "NE, F0, F1, $0",
// Pairs, acquire/release and exclusive atomics, LSE-AL forms.
"(R0), R1", "R0, (R1)", "R1, (R2), R3", "(R2, R3), 8(R1)",
"8(R1), (R2, R3)", "R1, R2, (R3)", "(R0)",
// System operations and their register/operand names.
"$4, R1, p2", "$35943", "$1", "$1, SPSel", "SPSel, R0",
"IVAC, R0", "(R0), PLDL1KEEP", "R1, R2, R3, R4",
// SIMD element, structure and literal-pool forms.
"(R0), [V1.B16]", "[V1.B16], (R0)", "V13.S[0], R1",
"R1, V2.B[3]", "$4, V1.B16, V2.B16", "V1.B16, (R0)",
"(R0), V1.B16", "",
// The spellings GOROOT's own kernels use, from the
// differential kernels this table was proven against.
"R0, p2", "R0, R1", "F0, F1, F2, F3", "$4, V1.B16, V2.B16, V3.B16, V4.B16",
"(R0), [V0.B8, V1.B8, V2.B8, V3.B8]", "$1, $2, V1",
"R0, R1, p2", "p2, R1", "$1234, R1", "DCZID_EL0, R1",
"$0", "R1, $4, EQ", "$33, R1, $25, R2", "$4, R1, p2",
"$4, V1.B8, V2.B8, V3.B8", "$63, V1.D2, V2.D2, V3.D2",
"V1.B16, [V2.B16], V3.B16", "V1.B8, [V2.B16, V3.B16], V4.B8",
"$4, V1.B16, V2.B16, V3.B16", "$15, V1", "V1, V2, p2",
"R0, R1, $1, $4, p2",
}
case arch.RISCV:
return []string{
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
"X5, X6, p2", "p2(SB)",
// AMO atomics: destination, base, source.
"R5, (R4), R6", "X5, (X4), X6",
// Segment stores take the first vector register aligned
// to the segment count, as the toolchain requires.
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
// The FP multiply-add family takes four registers.
"F0, F1, F2, F3",
// The RVV slice: register, vector-register and vtype forms.
"V1, V2, V3", "V1, X5, V2", "V1", "V1, (X5)", "(X5), V1",
"$15, V1", "$15", "V1, V2", "V1, X5",
"X5, X6, p2", "R5, R6, p2",
"X5, E8, M8, TA, MA, X6", "$4, E32, M1, TA, MA, X1",
"(X5), X6, V1, V2",
"",
}
case arch.LOONG64:
return []string{
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
// AMO atomics: destination, base, source.
"R5, (R4), R6", "X5, (X4), X6",
// Segment stores take the first vector register aligned
// to the segment count, as the toolchain requires.
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
// The LSX and LASX banks share the 5-bit numbering with F.
"V1, V2, V3", "X1, X2, X3", "V1, V2", "X1, X2", "V1", "X1",
// The vector compare-to-flag forms land in an FCC register.
"V1, FCC0", "X1, FCC0",
"",
}
}
return nil
@@ -376,15 +420,40 @@ func cmdAuditCorpus(args []string) error {
// corpusStats is the outcome of one corpus audit run.
type corpusStats struct {
root string
files int
generic int // files attempted for all four architectures
full int // files that assembled for every target architecture
targets []corpusTarget
tallies []*corpusTally
root string
files int
generic int // files attempted for all four architectures
otherPort int // files named for another Go port: never attempted
full int // files that assembled for every target architecture
targets []corpusTarget
tallies []*corpusTally
}
// runCorpusAudit assembles every .s file under root and returns the stats.
// goPortSuffixes lists every architecture the Go project ports to. A file
// named for one of them belongs to that port's build, not to the generic
// set, even when gasm does not support the architecture.
var goPortSuffixes = []string{
"386", "amd64", "arm", "arm64", "loong64", "mips", "mips64",
"mips64le", "mipsle", "ppc64", "ppc64le", "riscv", "riscv64",
"s390x", "wasm",
}
// otherPortFile reports whether the file's name carries a Go-architecture
// suffix gasm does not support.
func otherPortFile(path string) bool {
base := path
if i := strings.LastIndexByte(base, '/'); i >= 0 {
base = base[i+1:]
}
for _, sfx := range goPortSuffixes {
if strings.HasSuffix(base, "_"+sfx+".s") {
return true
}
}
return false
}
func runCorpusAudit(root string) (*corpusStats, error) {
files, err := asmFiles(root)
if err != nil {
@@ -403,7 +472,7 @@ func runCorpusAudit(root string) (*corpusStats, error) {
}
// full is the north-star number: a file counts when every architecture
// its name allows assembles it.
full, generic := 0, 0
full, generic, otherPort := 0, 0, 0
for _, path := range files {
src, err := readSource(path)
@@ -419,6 +488,13 @@ func runCorpusAudit(root string) (*corpusStats, error) {
wanted = append(wanted, i)
}
}
} else if otherPortFile(path) {
// A file named for a Go port gasm does not support (arm,
// 386, s390x, ...) is compiled by no supported-arch build,
// so it is neither generic nor a per-arch attempt: counting
// it as generic would make the headline unreachably low
// for reasons no supported target can fix.
otherPort++
} else {
generic++
for i := range targets {
@@ -449,19 +525,25 @@ func runCorpusAudit(root string) (*corpusStats, error) {
}
return &corpusStats{
root: root,
files: len(files),
generic: generic,
full: full,
targets: targets,
tallies: tallies,
root: root,
files: len(files),
generic: generic,
otherPort: otherPort,
full: full,
targets: targets,
tallies: tallies,
}, nil
}
// printCorpusStats renders the corpus audit report.
func printCorpusStats(s *corpusStats) {
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures)\n", s.root, s.files, s.generic)
fmt.Printf(" assemble for every target architecture: %d (%.1f%%)\n", s.full, 100*float64(s.full)/float64(max(s.files, 1)))
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d named for other Go ports, never attempted)\n", s.root, s.files, s.generic, s.otherPort)
// The rate is over the files a supported build would attempt: the
// other ports' files sit in the count for completeness but can never
// assemble, so counting them in the denominator would report the gap
// of architectures gasm deliberately does not target.
attemptable := max(s.files-s.otherPort, 1)
fmt.Printf(" assemble for every target architecture: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
for i, tg := range s.targets {
t := s.tallies[i]
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
+69
View File
@@ -0,0 +1,69 @@
// Atomics and carry-extending multi-word arithmetic: exchange,
// compare-exchange, exchange-add, ADCX/ADOX and the CRC-32 accumulator
// family. Every result is folded back so no instruction is dead.
#include "textflag.h"
// func xchg(p *uint64, v uint64) uint64
TEXT ·xchg(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
XCHGQ BX, (AX)
XCHGQ BX, CX
XCHGL BX, CX
XCHGW BX, CX
XCHGB BL, CL
MOVQ AX, ret+16(FP)
RET
// func cmpxchg(p *uint64, old, new uint64) uint8
TEXT ·cmpxchg(SB), NOSPLIT, $0-25
MOVQ p+0(FP), AX
MOVQ old+8(FP), BX
MOVQ new+16(FP), CX
CMPXCHGQ CX, (AX)
CMPXCHGL CX, BX
CMPXCHGW CX, BX
CMPXCHGB CL, BL
SETEQ AL
MOVB AL, ret+24(FP)
RET
// func xadd(p *uint64, v uint64) uint64
TEXT ·xadd(SB), NOSPLIT, $0-24
MOVQ p+0(FP), AX
MOVQ v+8(FP), BX
XADDQ BX, (AX)
XADDL BX, CX
XADDW BX, CX
XADDB BL, CL
MOVQ AX, ret+16(FP)
RET
// func adcx_adox(lo, hi, x, y uint64) uint64
TEXT ·adcx_adox(SB), NOSPLIT, $0-40
MOVQ lo+0(FP), AX
MOVQ hi+8(FP), DX
MOVQ x+16(FP), BX
MOVQ y+24(FP), CX
ADCXQ BX, AX
ADOXQ CX, DX
ADCXL BX, AX
ADOXL CX, DX
XORQ BX, BX
ADCXQ BX, AX
MOVQ AX, ret+32(FP)
RET
// func crc32(crc uint32, p *byte, n int) uint32
TEXT ·crc32(SB), NOSPLIT, $0-28
MOVL crc+0(FP), AX
MOVQ p+8(FP), SI
MOVQ n+16(FP), CX
CRC32B (SI), AX
CRC32Q (SI), CX
CRC32L (SI), AX
MOVW (SI), DX
CRC32W DX, AX
MOVL AX, ret+24(FP)
RET
+72
View File
@@ -0,0 +1,72 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 synchronisation instructions: the
// acquire/release loads and stores, the exclusive family and the LSE
// atomics with acquire and release semantics, plus the register-pair
// loads and stores. Every function is byte-compared against go tool asm.
#include "textflag.h"
// func acquireRelease()
TEXT ·acquireRelease(SB), NOSPLIT, $0-0
LDAR (R1), R2
LDARB (R3), R4
LDARH (R5), R6
LDARW (R7), R8
STLR R2, (R1)
STLRB R4, (R3)
STLRH R6, (R5)
STLRW R8, (R7)
RET
// func exclusive()
TEXT ·exclusive(SB), NOSPLIT, $0-0
LDAXR (R1), R2
LDAXRB (R3), R4
LDAXRW (R5), R6
STLXR R2, (R1), R8
STLXRB R4, (R3), R8
STLXRW R6, (R5), R8
RET
// func lseAcquireRelease()
TEXT ·lseAcquireRelease(SB), NOSPLIT, $0-0
CASALD R1, (R3), R2
CASALW R4, (R6), R5
LDADDALD R1, (R3), R2
LDADDALW R4, (R6), R5
LDCLRALB R1, (R3), R2
LDCLRALW R4, (R6), R5
LDCLRALD R1, (R3), R2
LDORALB R1, (R3), R2
LDORALW R4, (R6), R5
LDORALD R1, (R3), R2
SWPALB R1, (R3), R2
SWPALW R4, (R6), R5
SWPALD R1, (R3), R2
RET
// func lseBase()
TEXT ·lseBase(SB), NOSPLIT, $0-0
LDADDD R1, (R3), R2
LDADDW R4, (R6), R5
CASD R1, (R3), R2
CASW R4, (R6), R5
SWPD R1, (R3), R2
SWPW R4, (R6), R5
RET
// func pairs()
TEXT ·pairs(SB), NOSPLIT, $0-0
LDP (R1), (R2, R3)
LDP 8(R4), (R5, R6)
LDP -16(R1), (R2, R3)
LDPW 4(R4), (R5, R6)
STP (R2, R3), 24(R7)
STP (R2, R3),-8(R7)
STPW (R1, R2), 4(R0)
FLDPD (R8), (F1, F2)
FLDPD 8(R8), (F3, F4)
FSTPD (F3, F4),-8(R9)
RET
+49
View File
@@ -0,0 +1,49 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the loong64 atomics: the AM* family in its plain
// and _dbar (acquire/release) forms, spelled as the runtime's
// atomic_loong64.s spells them. Every AM* takes three operands:
// value, (address), result.
#include "textflag.h"
TEXT ·plain(SB), NOSPLIT, $0-0
AMSWAPB R14, (R13), R12
AMSWAPH R14, (R13), R12
AMSWAPW R5, (R4), R6
AMSWAPV R5, (R4), R0
AMCASB R14, (R13), R12
AMCASH R6, (R4), R5
AMCASW R6, (R4), R5
AMCASV R6, (R4), R5
AMADDW R5, (R4), R0
AMADDV R14, (R13), R12
AMANDW R5, (R4), R6
AMANDV R5, (R4), R6
AMORW R5, (R4), R0
AMORV R5, (R4), R6
AMXORW R5, (R4), R6
AMXORV R5, (R4), R6
AMMAXW R5, (R4), R6
AMMAXV R5, (R4), R6
AMMINW R5, (R4), R6
AMMINV R5, (R4), R6
AMMAXWU R5, (R4), R6
AMMAXVU R5, (R4), R6
AMMINWU R5, (R4), R6
AMMINVU R5, (R4), R6
RET
TEXT ·dbar(SB), NOSPLIT, $0-0
AMADDDBW R5, (R4), R6
AMADDDBV R5, (R4), R6
AMANDDBW R5, (R6), R0
AMANDDBV R5, (R4), R6
AMORDBW R5, (R6), R0
AMORDBV R5, (R4), R6
AMSWAPDBW R5, (R4), R6
AMSWAPDBV R5, (R4), R0
AMCASDBW R6, (R4), R5
AMCASDBV R6, (R4), R5
RET
+35
View File
@@ -0,0 +1,35 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 atomics: the RV64A AMO family and the
// load-reserved / store-conditional pair, in the toolchain's spelling
// (value, (address), result). Both orderings sit in the encodings: the
// table gives every AMO aq and rl, LR acquire and SC release.
#include "textflag.h"
TEXT ·amo(SB), NOSPLIT, $0-0
AMOSWAPW X5, (X6), X7
AMOSWAPD X5, (X6), X7
AMOADDW X5, (X6), X7
AMOADDD X5, (X6), X7
AMOANDW X5, (X6), X7
AMOANDD X5, (X6), X7
AMOORW X5, (X6), X7
AMOORD X5, (X6), X7
AMOXORW X5, (X6), X7
AMOXORD X5, (X6), X7
AMOMAXW X5, (X6), X7
AMOMAXD X5, (X6), X7
AMOMAXUW X5, (X6), X7
AMOMAXUD X5, (X6), X7
AMOMINUW X5, (X6), X7
AMOMINUD X5, (X6), X7
RET
TEXT ·lrsc(SB), NOSPLIT, $0-0
LRW (X5), X6
LRD (X5), X6
SCW X5, (X6), X7
SCD X5, (X6), X7
RET
+102
View File
@@ -0,0 +1,102 @@
// The AVX/AVX-512 gap families: fused scalar multiply-add, carries through
// GF(2^8) affine transforms, population counts, non-temporal stores, mask
// moves and the KMOV widths. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func avxblend(a, b []float64) float64
TEXT ·avxblend(SB), NOSPLIT, $0-56
MOVQ a_base+0(FP), SI
MOVQ b_base+24(FP), DI
VMOVUPD (SI), Y0
VMOVUPD (DI), Y1
VXORPS Y2, Y2, Y2
VSHUFPD $5, Y0, Y1, Y3
VMOVUPD Y3, (SI)
VPBLENDD $3, Y0, Y1, Y4
VPERM2F128 $1, Y4, Y0, Y0
VEXTRACTF128 $1, Y0, X1
VZEROALL
VMOVSD X1, ret+48(FP)
RET
// func avxint(p *byte, n int) uint64
TEXT ·avxint(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
VMOVDQU (SI), Y0
VPCMPEQB Y0, Y0, Y1
VPSLLDQ $2, X0, X0
VPSRLDQ $4, Y0, Y0
VPALIGNR $3, X0, X1, X1
VPCLMULQDQ $0, X0, X1, X2
VGF2P8AFFINEQB $7, X2, X0, X3
VPOPCNTB X3, X4
VPOPCNTD Y0, Y5
VPERMI2B X0, X1, X2
VPTEST X0, X0
VPMOVMSKB X1, AX
VZEROUPPER
MOVQ AX, ret+16(FP)
RET
// func avxnt(p *float64)
TEXT ·avxnt(SB), NOSPLIT, $0-8
MOVQ p+0(FP), DI
VMOVUPD (DI), Y0
VADDPD Y0, Y0, Y0
VMOVNTDQ Y0, (DI)
VMOVNTDQ X0, 16(DI)
VZEROALL
RET
// func avxmas(a, b []float64) float64
TEXT ·avxmas(SB), NOSPLIT, $0-56
MOVQ a_base+0(FP), SI
MOVQ b_base+24(FP), DI
VMOVSD (SI), X0
VMOVSD (DI), X1
VFMADD213SD X1, X0, X0
VFNMADD231SD X1, X0, X0
VADDSD X1, X0, X0
VMOVSD X0, ret+48(FP)
RET
// func avxgpr(x, y uint64) uint64
TEXT ·avxgpr(SB), NOSPLIT, $0-24
MOVQ x+0(FP), AX
MOVQ y+8(FP), BX
ANDNL BX, AX, CX
MULXQ BX, DX, SI
RORXL $3, AX, CX
RORXQ $7, BX, SI
MOVQ CX, ret+16(FP)
RET
// func avxmask(kin uint8, p *byte) uint8
TEXT ·avxmask(SB), NOSPLIT, $0-17
MOVQ p+8(FP), SI
KMOVB kin+0(FP), K1
KMOVB K1, K2
KMOVW K2, K1
KMOVD K1, K3
KMOVQ K3, K4
KMOVB K4, K1
KMOVB K1, AX
KMOVD K1, (SI)
MOVB AL, ret+8(FP)
RET
// func avx512(p *uint64, n int) uint64
TEXT ·avx512(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
VMOVDQU64 (SI), Z0
VPORQ Z0, Z0, Z1
VPOPCNTQ Z1, Z2
VPERMB Z1, Z0, Z2
VPXORD Z2, Z1, Z0
VMOVDQA64 Z0, (SI)
VZEROUPPER
XORQ AX, AX
MOVQ AX, ret+16(FP)
RET
+64
View File
@@ -0,0 +1,64 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 toolchain-synthesised instructions:
// the Zbb-style pseudos the assembler expands instruction-for-instruction
// (ANDN/ORN, MIN/MAX, ROR and friends, the reversed branches, FABSD), the
// CSR read RDTIME and the FP sign-injection and fused-multiply-add forms.
#include "textflag.h"
TEXT ·logic(SB), NOSPLIT, $0-0
ANDN X19, X20, X21
ANDN X19, X20
ANDN X21, X19, X21
ORN X20, X19
ORN X20, X19, X21
MAX X26, X28, X29
MAX X26, X28
MAXU X28, X29, X30
MAXU X28, X29
MIN X29, X30, X5
MIN X29, X30
MINU X30, X5, X6
MINU X30, X5
MAX X5, X5
MAX X5, X5, X6
SEQZ X5, X6
NEG X5, X6
NEG X5
NOT X5
NOT X5, X6
NOP
RET
TEXT ·rotate(SB), NOSPLIT, $0-0
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORIW $31, X13, X14
RORIW $1, X14, X15
RORIW $3, X14
RORW X15, X16, X17
RORW $31, X13
RET
TEXT ·fp(SB), NOSPLIT, $0-0
FABSD F1, F2
FSGNJD F1, F0, F2
FMADDD F1, F2, F3, F4
FMSUBD F1, F2, F3, F4
FNMSUBD F1, F2, F3, F4
FMADDS F1, F2, F3, F4
FNMADDS F1, F2, F3, F4
RET
TEXT ·branches(SB), NOSPLIT, $0-0
BGT X5, X6, tgt
BLE X5, X6, tgt
BGTU X5, X6, tgt
BLEU X5, X6, tgt
tgt:
RDTIME X5
RET
+57
View File
@@ -0,0 +1,57 @@
// The AES-NI, SHA and carry-less multiply round instructions as GOROOT's
// crypto kernels spell them. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func aesround(blk, rk *byte)
TEXT ·aesround(SB), NOSPLIT, $0-16
MOVQ blk+0(FP), SI
MOVQ rk+8(FP), DI
MOVOU (SI), X0
MOVOU (DI), X1
AESENC X1, X0
AESENCLAST X1, X0
AESDEC X1, X0
AESDECLAST X1, X0
AESIMC X1, X2
AESKEYGENASSIST $1, X1, X3
MOVOU X0, (SI)
MOVOU X2, (DI)
RET
// func sha1block(p *byte, n int, h *[5]uint32)
TEXT ·sha1block(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ h+16(FP), DI
MOVOU (SI), X0
MOVOU 16(SI), X1
SHA1RNDS4 $0, X1, X0
SHA1NEXTE X1, X0
SHA1MSG1 X1, X2
SHA1MSG2 X1, X2
MOVOU X0, (DI)
RET
// func sha256block(p *byte, n int, h *[8]uint32)
TEXT ·sha256block(SB), NOSPLIT, $0-24
MOVQ p+0(FP), SI
MOVQ h+16(FP), DI
MOVOU (SI), X0
MOVOU 16(SI), X1
SHA256RNDS2 X0, X1, X0
SHA256MSG1 X1, X2
SHA256MSG2 X1, X2
MOVOU X0, (DI)
RET
// func pclmul(a, b *byte)
TEXT ·pclmul(SB), NOSPLIT, $0-16
MOVQ a+0(FP), SI
MOVQ b+8(FP), DI
MOVOU (SI), X0
MOVOU (DI), X1
PCLMULQDQ $0, X1, X0
PCLMULQDQ $17, (DI), X0
MOVOU X0, (SI)
RET
+42
View File
@@ -0,0 +1,42 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 cryptographic extension: the AES round
// instructions and the SHA1, SHA256 and SHA512 families. Every function is
// byte-compared against go tool asm.
#include "textflag.h"
// func aesRound()
TEXT ·aesRound(SB), NOSPLIT, $0-0
AESE V31.B16, V29.B16
AESD V22.B16, V19.B16
AESMC V14.B16, V28.B16
AESIMC V12.B16, V27.B16
RET
// func sha1Round()
TEXT ·sha1Round(SB), NOSPLIT, $0-0
SHA1C V8.S4, V8, V2
SHA1P V3.S4, V20, V27
SHA1M V0.S4, V27, V27
SHA1H V17, V25
SHA1SU0 V17.S4, V13.S4, V16.S4
SHA1SU1 V24.S4, V23.S4
RET
// func sha256Round()
TEXT ·sha256Round(SB), NOSPLIT, $0-0
SHA256H V4.S4, V2, V11
SHA256H2 V6.S4, V16, V11
SHA256SU0 V0.S4, V16.S4
SHA256SU1 V31.S4, V3.S4, V15.S4
RET
// func sha512Round()
TEXT ·sha512Round(SB), NOSPLIT, $0-0
SHA512H V2.D2, V1, V0
SHA512H2 V4.D2, V3, V2
SHA512SU0 V9.D2, V8.D2
SHA512SU1 V7.D2, V6.D2, V5.D2
RET
+66
View File
@@ -0,0 +1,66 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 integer slice: carry-setting arithmetic,
// widening multiplies, bit manipulation, conditional compares, the compare
// and test branches, ADR and the wide-constant moves. Every function is
// byte-compared against go tool asm.
#include "textflag.h"
// func carryArith()
TEXT ·carryArith(SB), NOSPLIT, $0-0
ADC R0, R2, R12
ADCS R23, R22, R22
ADC $0, R1
SBC R25, R10, R26
SBCS R5, R9, R5
SBCS $0, R1
RET
// func wideningMul()
TEXT ·wideningMul(SB), NOSPLIT, $0-0
MUL R4, R3, R0
MSUB R19, R16, R26, R2
SMULH R24, R20, R24
UMULH R24, R20, R24
RET
// func bitManip()
TEXT ·bitManip(SB), NOSPLIT, $0-0
RBIT R11, R4
REV R1, R2
CLZ R21, R9
REVW R1, R2
CLSW R1, R2
UBFX $33, R17, $25, R5
UBFXW $4, R1, $9, R2
RET
// func condCompare()
TEXT ·condCompare(SB), NOSPLIT, $0-0
CCMP LE, R7, $19, $3
CCMP LT, R30, R6, $7
CCMN EQ, R1, R2, $3
CCMPW LE, R7, $19, $3
RET
// func branchForms()
TEXT ·branchForms(SB), NOSPLIT, $0-0
CBZ R1, target
CBNZ R7, target
CBNZW R2, target
TBZ $4, R7, target
TBNZ $33, R7, target
ADR target, R10
target:
RET
// func wideMoves()
TEXT ·wideMoves(SB), NOSPLIT, $0-0
MOVK $1234, R5
MOVK $305397760, R5
MOVKW $1234, R5
MOVK $16771847290880, R21
RET
+76
View File
@@ -0,0 +1,76 @@
// Carry arithmetic, rotates, unsigned/signed division and bit tests: the
// scalar families GOROOT's big-number and crypto kernels use. Every result
// is folded back so no instruction is dead.
#include "textflag.h"
// func carry(a, b uint64) uint64
TEXT ·carry(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
ADCQ $0, AX
MOVQ BX, CX
SBBQ $1, CX
ADCL BX, AX
ADCB AL, BL
ADCW $7, CX
MOVQ AX, ret+16(FP)
RET
// func borrow(a, b uint64) uint64
TEXT ·borrow(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
SUBQ BX, AX
SBBQ $0, AX
SBBQ BX, CX
MOVQ AX, ret+16(FP)
RET
// func rot(x uint64, n uint32) uint64
TEXT ·rot(SB), NOSPLIT, $0-24
MOVQ x+0(FP), AX
MOVL n+8(FP), CX
ROLQ CL, AX
RORQ $7, AX
ROLL $1, AX
RORL CL, AX
RCLQ $1, AX
RCRQ CL, AX
ROLW $3, AX
SALQ $2, AX
SALB $1, AX
MOVQ AX, ret+8(FP)
RET
// func muldiv(a, b uint64) uint64
TEXT ·muldiv(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
MULQ BX
MULQ (BX)
MOVL (BX), CX
MULL CX
DIVQ BX
IDIVQ BX
MOVL a+0(FP), AX
DIVL CX
IDIVL CX
MOVQ AX, ret+16(FP)
RET
// func bitfield(w *uint64) uint64
TEXT ·bitfield(SB), NOSPLIT, $0-16
MOVQ (DI), AX
MOVQ (DI), CX
BTQ AX, CX
BTQ $3, (DI)
BTL AX, CX
BTW $1, CX
BTSQ $5, AX
BTRQ AX, CX
BTCQ $7, (DI)
SETCS AL
MOVQ AX, ret+8(FP)
RET
+98
View File
@@ -0,0 +1,98 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 NEON slice: the logical and arithmetic
// three-register operations, permutations, comparisons, shifts, the crypto
// four-register group, element moves, table lookups and the structure
// loads and stores. Every function is byte-compared against go tool asm.
#include "textflag.h"
// func simdLogic()
TEXT ·simdLogic(SB), NOSPLIT, $0-0
VADD V1.B16, V2.B16, V3.B16
VADD V1.B8, V2.B8, V3.B8
VSUB V1.S4, V2.S4, V3.S4
VMUL V1.H8, V2.H8, V3.H8
VAND V4.B16, V4.B16, V9.B16
VORR V5.B16, V4.B16, V3.B16
VEOR V0.B16, V1.B16, V0.B16
VADDP V1.H8, V2.H8, V3.H8
VCMEQ V24.S4, V13.S4, V12.S4
VCMEQ $0, V2.H4, V3.H4
RET
// func simdPerm()
TEXT ·simdPerm(SB), NOSPLIT, $0-0
VZIP1 V16.H8, V3.H8, V19.H8
VZIP1 V6.D2, V9.D2, V11.D2
VZIP2 V22.D2, V25.D2, V21.D2
VREV32 V2.H8, V1.H8
VREV64 V2.S4, V3.S4
VUADDLV V31.S4, V11
VEXT $4, V2.B8, V1.B8, V3.B8
VEXT $8, V2.B16, V1.B16, V3.B16
RET
// func simdShift()
TEXT ·simdShift(SB), NOSPLIT, $0-0
VSHL $7, V22.D2, V25.D2
VSHL $24, V1.S4, V2.S4
VUSHR $6, V22.H8, V23.H8
VUSHR $56, V1.D2, V2.D2
VSRI $24, V1.S4, V2.S4
VSRI $56, V1.D2, V2.D2
RET
// func simdCrypto4()
TEXT ·simdCrypto4(SB), NOSPLIT, $0-0
VEOR3 V2.B16, V7.B16, V12.B16, V25.B16
VBCAX V1.B16, V2.B16, V26.B16, V31.B16
VXAR $63, V27.D2, V21.D2, V26.D2
VRAX1 V26.D2, V29.D2, V30.D2
VPMULL V2.D1, V1.D1, V3.Q1
VPMULL V2.B8, V1.B8, V3.H8
VPMULL2 V2.D2, V1.D2, V4.Q1
VPMULL2 V2.B16, V1.B16, V4.H8
RET
// func simdElement()
TEXT ·simdElement(SB), NOSPLIT, $0-0
VDUP V31.B[15], V18
VDUP V19.S[3], V18.S4
VDUP V1.D[1], V2.D2
VMOV V13.S[0], R20
VMOV V11.B[11], V16.B[12]
VMOV R20, V21.B[2]
VMOV V2.B16, V4.B16
RET
// func simdTable()
TEXT ·simdTable(SB), NOSPLIT, $0-0
VTBL V22.B16, [V28.B16], V11.B16
VTBL V18.B8, [V17.B16, V18.B16], V22.B8
VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8
RET
// func simdLoadStore()
TEXT ·simdLoadStore(SB), NOSPLIT, $0-0
VLD1 (R2), [V21.B16]
VLD1 (R24), [V18.D1, V19.D1, V20.D1]
VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]
VLD1.P 32(R1), [V2.B16, V3.B16]
VLD1.P 64(R4), [V5.B16, V6.B16, V7.B16, V8.B16]
VLD1R (R1), [V9.B8]
VLD1R (R0), [V0.B16]
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)
VST1 [V14.H4, V15.H4, V16.H4], (R27)
VST1.P [V2.B16], (R1)
VST1.P [V2.B16, V3.B16], 32(R1)
RET
// func simdLiteral()
TEXT ·simdLiteral(SB), NOSPLIT, $0-0
VMOVS $0x80402010, V11
VMOVD $0x8040201008040201, V20
VMOVQ $0x7040201008040201, $0x8040201008040201, V10
RET
+77
View File
@@ -0,0 +1,77 @@
// The legacy SSE gap families: scalar compares and square roots, the Plan 9
// packed spellings, shuffles, lane extracts and inserts, packed integer
// shifts and the octa moves. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func cmporder(a, b *float64) int
TEXT ·cmporder(SB), NOSPLIT, $0-24
MOVQ a+0(FP), SI
MOVQ b+8(FP), DI
MOVSD (SI), X0
MOVSD (DI), X1
ANDNPD X0, X2
ANDNPS X0, X3
COMISD X0, X1
SQRTSD X0, X2
CMPSD X0, X1, $5
MOVL SI, CX
SETPL CL
MOVL CX, ret+16(FP)
RET
// func packed(w *uint64) uint64
TEXT ·packed(SB), NOSPLIT, $0-16
MOVQ w+0(FP), SI
MOVO (SI), X0
MOVOA (SI), X1
PADDL X0, X1
PSUBL X0, X1
PCMPEQL X0, X1
PUNPCKLBW X0, X1
PSHUFL $27, X0, X2
MOVOU X2, (SI)
MOVQ (SI), AX
MOVQ AX, ret+8(FP)
RET
// func lanes(p *byte, buf *byte)
TEXT ·lanes(SB), NOSPLIT, $0-16
MOVQ p+0(FP), SI
MOVQ buf+8(FP), DI
MOVO (SI), X0
MOVQ SI, AX
PINSRB $1, AX, X0
PINSRW $2, AX, X0
PINSRD $3, AX, X0
PINSRQ $1, AX, X0
PEXTRB $1, X0, AX
PEXTRW $2, X0, AX
PEXTRD $3, X0, AX
PEXTRQ $1, X0, CX
PCMPESTRI $4, X0, X0
MOVB AL, (DI)
MOVOU X0, (SI)
RET
// func shifts(p *uint64)
TEXT ·shifts(SB), NOSPLIT, $0-8
MOVQ p+0(FP), SI
MOVO (SI), X0
MOVO X0, X1
PSLLW $3, X0
PSRLW $1, X1
PSRAW $2, X0
PSLLL $4, X0
PSRLL $5, X1
PSRAL $1, X0
PSLLQ $7, X0
PSRLQ $9, X1
PSLLL X1, X0
PSRLQ X0, X1
PSLLDQ $2, X0
PSRLDQ $4, X1
MOVOU X0, (SI)
MOVOU X1, 16(SI)
RET
+76
View File
@@ -0,0 +1,76 @@
// System, string-primitive and x87 families: flag register moves, the
// serialising instructions, MOVS/STOS, the MXCSR pair, scalar float-to-int
// conversions and FMOVD. Every result is folded back so no instruction is
// dead.
#include "textflag.h"
// func system(x uint64) uint64
TEXT ·system(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
PUSHFQ
POPFQ
CPUID
RDTSC
RDTSCP
SYSCALL
XGETBV
PAUSE
LFENCE
MFENCE
SFENCE
UNDEF
XORQ AX, BX
MOVQ BX, ret+8(FP)
RET
// func stringprim(p *byte, n int) uint64
TEXT ·stringprim(SB), NOSPLIT, $0-24
MOVQ p+0(FP), DI
MOVQ n+8(FP), CX
LEAQ buf<>(SB), AX
MOVQ AX, SI
CLD
MOVSB
MOVSW
MOVSL
MOVSQ
STOSB
STOSQ
STOSL
STOSW
MOVQ DI, ret+16(FP)
RET
DATA buf<>+0x00(SB)/8, $0
GLOBL buf<>(SB), NOPTR, $8
// func intgate(x uint64) uint64
TEXT ·intgate(SB), NOSPLIT, $0-16
MOVQ x+0(FP), AX
INT $3
MOVQ AX, ret+8(FP)
RET
// func fpmxcsr(x float64, csr *uint32) int64
TEXT ·fpmxcsr(SB), NOSPLIT, $0-24
MOVQ x+0(FP), X0
MOVQ csr+8(FP), AX
STMXCSR (AX)
LDMXCSR (AX)
CVTSD2SL X0, CX
CVTTSD2SQ X0, DX
MOVL (AX), SI
MOVQ SI, ret+8(FP)
RET
// func fmove(p *float64) float64
TEXT ·fmove(SB), NOSPLIT, $0-16
MOVQ p+0(FP), AX
FMOVD (AX), F0
FMOVD F0, F1
FMOVD F0, (AX)
MOVQ (AX), AX
MOVQ AX, ret+8(FP)
RET
+61
View File
@@ -0,0 +1,61 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 system instructions: barriers,
// cache maintenance, the system register accesses, supervisor calls,
// breakpoints and prefetches. Every function is byte-compared against
// go tool asm.
#include "textflag.h"
// func barriers()
TEXT ·barriers(SB), NOSPLIT, $0-0
DMB $15
DMB $1
DSB $15
DSB $4
ISB $15
ISB $1
RET
// func cacheOps()
TEXT ·cacheOps(SB), NOSPLIT, $0-0
DC ZVA, R4
DC IVAC, R1
DC CVAC, R2
DC CVAU, R3
DC CIVAC, R7
RET
// func sysRegs()
TEXT ·sysRegs(SB), NOSPLIT, $0-0
MRS DCZID_EL0, R3
MRS CNTVCT_EL0, R0
MRS CNTPCT_EL0, R1
MRS CNTFRQ_EL0, R2
MRS MIDR_EL1, R0
MRS ID_AA64PFR0_EL1, R0
MRS ID_AA64ISAR0_EL1, R0
MRS ID_AA64ISAR1_EL1, R0
MRS DIT, R0
MSR $3, SPSel
MSR $9, DAIFSet
MSR $6, DAIFClr
MSR $1, DIT
RET
// func exceptions()
TEXT ·exceptions(SB), NOSPLIT, $0-0
SVC $0
SVC $7165
BRK
BRK $35943
RET
// func prefetch()
TEXT ·prefetch(SB), NOSPLIT, $0-0
PRFM (R0), PLDL1KEEP
PRFM (R3), PLDL3KEEP
PRFM (R4), PSTL1KEEP
PRFM (R2), $25
RET
+85
View File
@@ -0,0 +1,85 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the loong64 LSX/LASX slice: every function pairs
// with the same instructions in the go tool asm ground truth.
#include "textflag.h"
TEXT ·threeReg(SB), NOSPLIT, $0-0
VADDV V1, V2, V3
VADDW V1, V2, V3
VADDV V2, V1
VANDV V1, V2, V3
VANDV V1, V2
VXORV V1, V2, V3
VXORV V1, V2
VSEQB V1, V2, V3
VSEQV V1, V2, V3
VSRAB V1, V2, V3
VROTRW V1, V2, V3
VPCNTV V1, V2
XVADDV X1, X2, X3
XVADDV X2, X1
XVANDV X1, X2, X3
XVXORV X1, X2, X3
XVSEQB X1, X2, X3
XVSEQV X1, X2, X3
XVPCNTV X1, X2
RET
TEXT ·immediates(SB), NOSPLIT, $0-0
VANDB $0, V2, V3
VANDB $255, V2
VSEQB $3, V2, V3
VSEQV $15, V2, V3
VSRAB $0, V1, V2
VSRAB $7, V1, V2
VSRAB $6, V1
VROTRW $0, V1, V2
VROTRW $16, V1, V2
VROTRW $16, V1
XVANDB $1, X2, X2
RET
TEXT ·conditions(SB), NOSPLIT, $0-0
VSETNEV V1, FCC0
VSETANYEQB V1, FCC0
VSETANYEQV V2, FCC0
VSETALLNEV V0, FCC0
XVSETNEV X1, FCC0
XVSETANYEQB X1, FCC0
XVSETANYEQV X1, FCC0
XVSETALLNEV X1, FCC0
RET
TEXT ·fpConvert(SB), NOSPLIT, $0-0
FFINTDV F0, F1
FSEL FCC0, F3, F4, F3
FSEL FCC1, F1, F2
RET
TEXT ·memMoves(SB), NOSPLIT, $0-0
VMOVQ V1, V9
VMOVQ (R4), V2
VMOVQ 16(R4), V2
VMOVQ V0, (R4)
VMOVQ V0, 32(R4)
VMOVQ V0,-16(R6)
VMOVQ (R4)(R7), V3
VMOVQ V3, (R4)(R7)
XVMOVQ X3, X7
XVMOVQ (R4), X2
XVMOVQ X0, (R4)
XVMOVQ (R4)(R7), X4
XVMOVQ X0, (R4)(R7)
RET
TEXT ·elements(SB), NOSPLIT, $0-0
VMOVQ R6, V0.B16
VMOVQ R6, V12.W4
XVMOVQ R6, X0.B32
VMOVQ (R4), V4.W4
VMOVQ (R10), V0.W4
XVMOVQ (R4), X0.B32
RET
+53
View File
@@ -0,0 +1,53 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the riscv64 RVV slice: the instructions GOROOT's
// vector kernels use (crypto/internal/fips140/subtle/xor_riscv64.s,
// internal/bytealg and internal/chacha8rand), spelled as they spell them.
#include "textflag.h"
TEXT ·config(SB), NOSPLIT, $0-0
VSETVLI X5, E8, M8, TA, MA, X6
VSETVLI X11, E8, M8, TA, MA, X5
VSETVLI X12, E8, M8, TA, MA, X5
VSETVLI X13, E8, M8, TU, MU, X15
VSETIVLI $4, E32, M1, TA, MA, X0
VSETIVLI $15, E32, M1, TA, MA, X12
VSETVLI $15, E32, M1, TA, MA, X12
VSETVLI X10, E16, M1, TU, MU, X12
VSETVLI X10, E32, M2, TA, MA, X12
VSETVLI X10, E64, M8, TU, MU, X12
VSETIVLI $31, E32, M1, TA, MA, X12
RET
TEXT ·loadsStores(SB), NOSPLIT, $0-0
VLE8V (X10), V8
VLE8V (X11), V16
VLE8V (X12), V16
VIDV V12
VMV4RV V8, V24
VSE8V V24, (X10)
VSE32V V0, (X11)
VSE32V V8, (X11)
VSE32V V15, (X11)
RET
TEXT ·segmented(SB), NOSPLIT, $0-0
VLSSEG4E32V (X14), X0, V0
VLSSEG8E32V (X10), X0, V4
RET
TEXT ·crypto(SB), NOSPLIT, $0-0
VADDVV V20, V4, V4
VADDVV V27, V11, V11
VADDVX X12, V12, V12
VXORVV V8, V16, V24
VXORVV V13, V13, V13
VMSEQVX X12, V8, V0
VMSNEVV V8, V16, V0
VFIRSTM V0, X6
VFIRSTM V0, X7
VSLLVI $8, V28, V30
VSRLVI $25, V29, V29
RET
+7
View File
@@ -26,6 +26,13 @@ func TestGroundTruthARM64(t *testing.T) {
"../testdata/verify/bigframe_arm64.s",
"../testdata/verify/guard_arm64.s",
"../testdata/verify/indirect_arm64.s",
"../testdata/verify/exclusive_arm64.s",
"../testdata/verify/shifts_arm64.s",
"../testdata/verify/atomics_arm64.s",
"../testdata/verify/crypto_arm64.s",
"../testdata/verify/integer_arm64.s",
"../testdata/verify/simd_arm64.s",
"../testdata/verify/system_arm64.s",
} {
t.Run(path, func(t *testing.T) {
src, err := os.ReadFile(path)
+7
View File
@@ -115,6 +115,13 @@ func TestGroundTruthAMD64(t *testing.T) {
"../testdata/verify/bigframe_amd64.s",
"../testdata/verify/guard_amd64.s",
"../testdata/verify/indirect_amd64.s",
"../testdata/verify/widen_amd64.s",
"../testdata/verify/scalar_amd64.s",
"../testdata/verify/atomics_amd64.s",
"../testdata/verify/system_amd64.s",
"../testdata/verify/crypto_amd64.s",
"../testdata/verify/sse_amd64.s",
"../testdata/verify/avx_amd64.s",
} {
t.Run(path, func(t *testing.T) {
f, errs := parser.Parse(path, mustRead(t, path))
+4
View File
@@ -25,6 +25,10 @@ func TestGroundTruthLOONG64(t *testing.T) {
"../testdata/verify/bigframe_loong64.s",
"../testdata/verify/guard_loong64.s",
"../testdata/verify/indirect_loong64.s",
"../testdata/verify/movwfp_loong64.s",
"../testdata/verify/branchu_loong64.s",
"../testdata/verify/atomics_loong64.s",
"../testdata/verify/vector_loong64.s",
"trampoline_loong64.s",
} {
t.Run(path, func(t *testing.T) {
+4
View File
@@ -29,6 +29,10 @@ func TestGroundTruthRISCV(t *testing.T) {
"../testdata/verify/guard_riscv64.s",
"../testdata/verify/indirect_riscv64.s",
"../testdata/verify/misc_riscv64.s",
"../testdata/verify/rvcstore_riscv64.s",
"../testdata/verify/atomics_riscv64.s",
"../testdata/verify/vector_riscv64.s",
"../testdata/verify/bitmanip_riscv64.s",
"trampoline_riscv64.s",
} {
t.Run(path, func(t *testing.T) {