From ca3fdce0e0cd602a9af40aa62527005560376975 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 06:44:51 +0200 Subject: [PATCH] feat(arm64): encode pairs, atomics, crypto, system and NEON slices Assisted-by: GLM 5.3 Flash --- arch/arm64.go | 27 + asm/arm64_assemble.go | 1158 ++++++++++++++++++++++++++++++- asm/arm64_encode.go | 429 +++++++++++- asm/arm64_encode_test.go | 454 +++++++++++- testdata/verify/atomics_arm64.s | 72 ++ testdata/verify/crypto_arm64.s | 42 ++ testdata/verify/integer_arm64.s | 66 ++ testdata/verify/simd_arm64.s | 98 +++ testdata/verify/system_arm64.s | 61 ++ 9 files changed, 2377 insertions(+), 30 deletions(-) create mode 100644 testdata/verify/atomics_arm64.s create mode 100644 testdata/verify/crypto_arm64.s create mode 100644 testdata/verify/integer_arm64.s create mode 100644 testdata/verify/simd_arm64.s create mode 100644 testdata/verify/system_arm64.s diff --git a/arch/arm64.go b/arch/arm64.go index 1eaf1dc..0b1d7dd 100644 --- a/arch/arm64.go +++ b/arch/arm64.go @@ -150,6 +150,33 @@ func arm64Curated() []Instr { t = append(t, i(op, "Atomic memory operation")) } + // Register-pair loads and stores. + for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} { + t = append(t, ic(op, "Register-pair load or store", 2, 2)) + } + + // Cache maintenance and prefetch. + t = append(t, i("DC", "Data cache maintenance")) + t = append(t, i("PRFM", "Memory prefetch")) + for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} { + t = append(t, i(op, "Atomic memory operation with acquire and release semantics")) + } + + // Cryptographic extensions. + for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} { + t = append(t, i(op, "AES round")) + } + for _, op := range []string{ + "SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1", + "SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1", + "SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1", + } { + t = append(t, i(op, "SHA round")) + } + for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} { + t = append(t, i(op, "Three-way XOR / rotate crypto vector operation")) + } + // Floating-point scalar. for _, op := range []string{ "FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX", diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index d754088..899ecb7 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -5,6 +5,7 @@ package asm import ( "fmt" + "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" @@ -14,7 +15,9 @@ import ( // code. Every instruction is 4 bytes; the MOV pseudo-instruction and the // immediate-arithmetic forms expand to 2-4 instructions when the immediate // does not fit, so the layout is computed in two passes (sizes, then encoding -// with resolved branch targets). +// with resolved branch targets). The returned literals carry the read-only +// constants any VMOVS/VMOVD/VMOVQ load refers to; the file assembler lays +// them out in the data section. // // The emitted bytes match the Go toolchain's arm64 assembler, which is the // ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch @@ -23,7 +26,7 @@ import ( // (the morestack check in the prologue and the call back into the runtime in // the epilogue) is not emitted, so the bytes match only for NOSPLIT functions // or zero-frame leaves, where the toolchain emits no guard either. -func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { +func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, []Arm64Literal, error) { fi := arm64ComputeFrame(t) prologue := arm64Prologue(fi) guardLen := arm64GuardLen(fi) @@ -37,6 +40,7 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ var relocs []Reloc var spadj []SpadjStep + lits := &arm64Literals{} // The prologue (3 instructions when a small frame, 4 for large) // raises the SP delta by autosize. The guard prefix shifts its PC. @@ -82,9 +86,9 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ if !ok { continue } - code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve) + code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve, lits) if err != nil { - return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) + return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) } for j := preCount; j < len(relocs); j++ { // Make the relocation offsets function-relative: each instruction @@ -110,7 +114,7 @@ func assembleARM64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, [ relocs = append(relocs, blReloc) pc += len(block) } - return out, offsets, relocs, lines, spadj, nil + return out, offsets, relocs, lines, spadj, lits.list(), nil } // arm64JumpChain precomputes jump-to-jump folding: a label whose first @@ -184,6 +188,11 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo) int { return len(arm64Return(fi)) } switch mnem { + case "VMOVS", "VMOVD", "VMOVQ": + // ADRP + ADD + wide load against a pooled literal. + return 12 + } + switch mnem { case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": return arm64MovSize(mnem, ops, fi) @@ -204,8 +213,10 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo) int { return 4 } -// encodeARM64Instr encodes a single AArch64 instruction. -func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) { +// encodeARM64Instr encodes a single AArch64 instruction. lits collects the +// read-only literals a VMOVS/VMOVD/VMOVQ constant load needs; the file +// assembler lays them out once every function is encoded. +func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64FrameInfo, relocs *[]Reloc, resolve func(string) string, lits *arm64Literals) ([]byte, error) { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands @@ -335,9 +346,106 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64Extr(mnem, enc.op, ops) } - // SIMD 3-operand (VADD, VSUB, VMUL). - if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMD3 { - return encodeARM64SIMD3(mnem, enc.op, ops) + // Acquire/release loads and stores (LDAR family, STLR family). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FAcqRel { + return encodeARM64AcqRel(mnem, enc.op, ops) + } + + // Load/store pairs (LDP, STP, LDPW, STPW, FLDPD, FSTPD). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair { + return encodeARM64Pair(mnem, enc.op, ops, fi) + } + + // Compare-and-branch and test-and-branch to a label. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBranch19 { + return encodeARM64Branch19(mnem, enc.op, ops, pc, offsets, resolve) + } + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FTestBranch { + return encodeARM64TestBranch(mnem, enc.op, ops, pc, offsets, resolve) + } + + // Data-processing (1 source): RBIT, REV, CLZ, CLS. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDP1 { + return encodeARM64DP1(mnem, enc.op, ops) + } + + // ADR/ADRP: (label, Rd) with the byte distance split into immlo and + // immhi. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FADR { + return encodeARM64ADR(mnem, enc.op, ops, pc, offsets, resolve) + } + + // Bitfield extract with wrapping immr: UBFX, SBFX. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield2 { + return encodeARM64Bitfield2(mnem, enc.op, ops) + } + + // Conditional compare: CCMP, CCMN. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCondCmp { + return encodeARM64CondCmp(mnem, enc.op, ops) + } + + // System operations: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSys { + return encodeARM64Sys(mnem, ops) + } + + // Crypto: AESD, AESE, SHA1C, SHA256H and friends. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCrypto2 { + return encodeARM64Crypto(mnem, enc.op, ops, 2) + } + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCrypto3 { + return encodeARM64Crypto(mnem, enc.op, ops, 3) + } + + // Move wide with an explicit immediate: MOVK. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMovWide { + return encodeARM64MoveWide(mnem, enc.op, ops) + } + + // SIMD element moves (VDUP, VMOV with lane indices) take precedence + // over the plain arrangement paths, which carry no index. + if (mnem == "VDUP" || mnem == "VMOV") && arm64SimdHasElement(ops) { + return encodeARM64Dup(mnem, ops) + } + + // Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1, + // VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so + // this check precedes the plain SIMD3 path below. + if spec, ok := a64SimdVTable[mnem]; ok { + return encodeARM64SimdV(mnem, spec, ops) + } + + // Arrangement-aware SIMD two-register (VREV32, VREV64, VUADDLV, VMOV). + if spec, ok := a64SimdV2Table[mnem]; ok { + return encodeARM64SimdV2(mnem, spec, ops) + } + + // SIMD four-register and immediate three-register (VEOR3, VBCAX, VXAR, + // VEXT). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMDV4 { + return encodeARM64SimdV4(mnem, enc.op, ops) + } + + // SIMD table lookup. + if mnem == "VTBL" { + return encodeARM64VTBL(ops) + } + + // SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R, + // VLD4R). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVLDST { + return encodeARM64VLDST(mnem, enc.op, ops) + } + + // SIMD shift by immediate (VSHL, VUSHR, VSRI). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FShiftImm { + return encodeARM64ShiftImm(mnem, enc.op, ops) + } + + // VMOVS/VMOVD/VMOVQ with a large constant: a literal pool load. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMoviLit { + return encodeARM64MoviLit(mnem, enc.op, ops, relocs, lits) } return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) @@ -459,6 +567,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er } return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil case 2: + // ADC family carries an immediate spelling: ADC $0, Rd reads the + // carry into Rd and takes ZR as the register operand. The + // encoding has no immediate field, so $0 is the only value. + if isImmOperand(ops[0]) { + v := arm64Imm64(ops[0]) + if v != 0 { + return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) + } + rd := arm64RegNum(operandRegName(ops[1])) + if rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | 31<<16 | uint32(rd)<<5 | uint32(rd)), nil + } if isCmp { // CMP Rm, Rn → SUBS XZR, Rn, Rm rm := arm64RegNum(operandRegName(ops[0])) @@ -1458,6 +1580,389 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil } +// encodeARM64DP1 encodes a data-processing (1 source) instruction: +// RBIT, REV, CLZ and friends take (Rn, Rd), word = base | Rn<<5 | Rd. +func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rn := arm64RegNum(operandRegName(ops[0])) + rd := arm64RegNum(operandRegName(ops[1])) + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rd)), nil +} + +// encodeARM64ADR encodes ADR/ADRP: (label, Rd), the byte distance to the +// target split into immlo (bits 30:29) and immhi (bits 23:5), with bit 31 +// selecting the page form. +func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rd := arm64RegNum(operandRegName(ops[1])) + if rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + target := resolve(arm64Label(ops[0])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + rel := int64(targetOff - pc) + if rel < -(1<<20) || rel >= 1<<20 { + return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target) + } + return a64wordLE(a64ADR(page, int32(rel>>2), uint32(rel)&3, uint32(rd))), nil +} + +// encodeARM64Bitfield2 encodes UBFX/SBFX ($lsb, Rn, $width, Rd): immr +// carries the lsb (six bits with the N flag on the X forms) and imms the +// lsb plus width minus one. A sum beyond the register width is the +// toolchain's "illegal bit number" error. +func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) { + return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem) + } + lsb := arm64Imm64(ops[0]) + width := arm64Imm64(ops[2]) + rn := arm64RegNum(operandRegName(ops[1])) + rd := arm64RegNum(operandRegName(ops[3])) + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + bits := int64(32) << (baseOp >> 31 & 1) + if lsb < 0 || lsb >= bits { + return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1) + } + if width < 1 || lsb+width > bits { + return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits) + } + immr := uint32(lsb) + n := uint32(0) + if immr >= 32 { + n = 1 << 21 + immr &^= 32 + } + return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil +} + +// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The +// third field carries Rm or a 5-bit immediate in the same bits, at the +// toolchain's choice of register or immediate operand. +func encodeARM64CondCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 4 || !isImmOperand(ops[3]) { + return nil, fmt.Errorf("%s expects 4 operands (cond, Rn, Rm|$imm, $nzcv)", mnem) + } + condName := operandRegName(ops[0]) + cond, ok := arm64CondMap[condName] + if !ok { + return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) + } + rn := arm64RegNum(operandRegName(ops[1])) + if rn < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + var v2 uint32 + if isImmOperand(ops[2]) { + v := arm64Imm64(ops[2]) + if v < 0 || v > 31 { + return nil, fmt.Errorf("%s: immediate %d out of range (0..31)", mnem, v) + } + v2 = uint32(v) + } else { + rm := arm64RegNum(operandRegName(ops[2])) + if rm < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + v2 = uint32(rm) + } + nzcv := arm64Imm64(ops[3]) + if nzcv < 0 || nzcv > 15 { + return nil, fmt.Errorf("%s: nzcv %d out of range (0..15)", mnem, nzcv) + } + // Bit 11 carries the immediate-vs-register choice for the third field. + op2 := uint32(0) + if isImmOperand(ops[2]) { + op2 = 1 << 11 + } + return a64wordLE(baseOp | v2<<16 | cond<<12 | op2 | uint32(rn)<<5 | uint32(nzcv)&0xF), nil +} + +// encodeARM64Branch19 encodes CBZ/CBNZ: (Rt, label), +// word = base | imm19<<5 | Rt with imm19 = (target - pc) >> 2. +func encodeARM64Branch19(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rt := arm64RegNum(operandRegName(ops[0])) + if rt < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + target := resolve(arm64Label(ops[1])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + rel := (targetOff - pc) >> 2 + if rel < -(1<<18) || rel >= (1<<18) { + return nil, fmt.Errorf("branch to %q too far (19-bit range)", target) + } + return a64wordLE(baseOp | uint32(rel)&0x7FFFF<<5 | uint32(rt)), nil +} + +// encodeARM64TestBranch encodes TBZ/TBNZ: ($bit, Rt, label). Bits 32 to 63 +// set the b5 flag at bit 31; there is one mnemonic per polarity, no width +// suffix. +func encodeARM64TestBranch(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { + if len(ops) != 3 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects 3 operands ($bit, Rt, label)", mnem) + } + bit := arm64Imm64(ops[0]) + if bit < 0 || bit > 63 { + return nil, fmt.Errorf("%s: bit number %d out of range (0..63)", mnem, bit) + } + rt := arm64RegNum(operandRegName(ops[1])) + if rt < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + target := resolve(arm64Label(ops[2])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + rel := (targetOff - pc) >> 2 + if rel < -(1<<13) || rel >= (1<<13) { + return nil, fmt.Errorf("branch to %q too far (14-bit range)", target) + } + return a64wordLE(baseOp | uint32(bit>>5)<<31 | uint32(bit&31)<<19 | uint32(rel)&0x3FFF<<5 | uint32(rt)), nil +} + +// encodeARM64Pair encodes load/store pair instructions. Loads spell +// (mem, (Rt1, Rt2)), stores (Rt1, Rt2), mem; the scaled immediate rides +// imm7 at bits 21:15 and must fit -64..63 after division by the access +// size (8 bytes for the D forms, 4 for the W forms). +func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64FrameInfo) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + scale := int64(8) + if strings.HasSuffix(mnem, "W") { + scale = 4 + } + load := strings.Contains(mnem, "LDP") + memOp, pairOp := ops[0], ops[1] + if !load { + memOp, pairOp = ops[1], ops[0] + } + if !isMemOperand(memOp) { + return nil, fmt.Errorf("%s: invalid memory operand", mnem) + } + rt1, rt2, ok := arm64PairOf(pairOp) + if !ok { + return nil, fmt.Errorf("%s expects a register pair (Rt1, Rt2)", mnem) + } + rn, off := arm64MemWithFrame(memOp, fi) + if rn < 0 { + return nil, fmt.Errorf("%s: invalid memory operand", mnem) + } + if off%scale != 0 || off < -64*scale || off > 63*scale { + return nil, fmt.Errorf("%s: offset %d out of pair range or not a multiple of %d", mnem, off, scale) + } + imm7 := off / scale + // The base carries the opc, V and L halves; only the scaled immediate + // and the three registers are filled in here. + return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil +} + +// encodeARM64AcqRel encodes the acquire/release loads and stores. Loads +// spell (Rn), Rt; stores Rt, (Rn). Both lay the base register at bits 9:5 +// and the data register at bits 4:0 over a base that carries no offset +// field, so a displaced operand is reported the way arm64ExclMem reports +// one for the exclusive family. +func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + memOp, regOp := ops[0], ops[1] + if !strings.HasPrefix(mnem, "LD") { + memOp, regOp = ops[1], ops[0] + } + rn, err := arm64ExclMem(mnem, memOp) + if err != nil { + return nil, err + } + rt := arm64RegNum(operandRegName(regOp)) + if rt < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil +} + +// encodeARM64Sys encodes the system operations: +// +// BRK [$imm16] SVC $imm16 +// DMB|DSB|ISB $imm4 DC , Rn +// MRS , Rd MSR $imm4, +// PRFM (Rn), $imm| +func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { + switch mnem { + case "BRK", "SVC": + base := uint32(0xd4200000) + if mnem == "SVC" { + base = 0xd4000001 + } + if len(ops) == 0 { + return a64wordLE(base), nil + } + if len(ops) != 1 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects no operand or $immediate", mnem) + } + v := arm64Imm64(ops[0]) + if v < 0 || v > 0xFFFF { + return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v) + } + return a64wordLE(base | uint32(v)<<5), nil + case "DMB", "DSB", "ISB": + if len(ops) != 1 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects $immediate", mnem) + } + v := arm64Imm64(ops[0]) + if v < 0 || v > 15 { + return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v) + } + base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem] + return a64wordLE(base | uint32(v)<<8), nil + case "DC": + if len(ops) != 2 { + return nil, fmt.Errorf("DC expects , Rn") + } + base, ok := a64DCOps[operandRegName(ops[0])] + if !ok { + return nil, fmt.Errorf("DC: unknown cache operation %q", operandRegName(ops[0])) + } + rn := arm64RegNum(operandRegName(ops[1])) + if rn < 0 { + return nil, fmt.Errorf("DC: invalid register operand") + } + return a64wordLE(base | uint32(rn)&31), nil + case "MRS": + if len(ops) != 2 { + return nil, fmt.Errorf("MRS expects , Rd") + } + base, ok := a64MRSOps[operandRegName(ops[0])] + if !ok { + return nil, fmt.Errorf("MRS: unknown system register %q", operandRegName(ops[0])) + } + rd := arm64RegNum(operandRegName(ops[1])) + if rd < 0 { + return nil, fmt.Errorf("MRS: invalid register operand") + } + return a64wordLE(base | uint32(rd)&31), nil + case "MSR": + if len(ops) != 2 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("MSR expects $immediate, ") + } + base, ok := a64MSROps[operandRegName(ops[1])] + if !ok { + return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1])) + } + v := arm64Imm64(ops[0]) + if v < 0 || v > 15 { + return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v) + } + return a64wordLE(base | uint32(v)<<8 | 31), nil + case "PRFM": + if len(ops) != 2 { + return nil, fmt.Errorf("PRFM expects (Rn), $immediate|") + } + rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{}) + if rn < 0 || off != 0 { + return nil, fmt.Errorf("PRFM: invalid memory operand") + } + var prfop int64 + if isImmOperand(ops[1]) { + prfop = arm64Imm64(ops[1]) + if prfop < 0 || prfop > 31 { + return nil, fmt.Errorf("PRFM: immediate %d out of range (0..31)", prfop) + } + } else { + p, ok := a64PRFOps[operandRegName(ops[1])] + if !ok { + return nil, fmt.Errorf("PRFM: unknown prefetch operation %q", operandRegName(ops[1])) + } + prfop = int64(p) + } + return a64wordLE(0xf9800000 | uint32(rn)<<5 | uint32(prfop)), nil + } + return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) +} + +// encodeARM64Crypto encodes the crypto instructions. Two-register forms +// spell (Rn, Rd), three-register forms (Rm, Rn, Rd); the arrangements, when +// spelled, must match the instruction's own (B16 for AES, S4 for the SHA1 +// and SHA256 families, D2 for SHA512). +func encodeARM64Crypto(mnem string, baseOp uint32, ops []*ast.Operand, n int) ([]byte, error) { + if len(ops) != n { + return nil, fmt.Errorf("%s expects %d operands, got %d", mnem, n, len(ops)) + } + vs := make([]a64Vec, n) + for i, op := range ops { + v, ok := arm64VecOf(op) + if !ok || v.hasIdx { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + if v.arr != "" && a64ArrIndex(v.arr) != a64CryptoArr[mnem] { + return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, v.arr) + } + vs[i] = v + } + if n == 2 { + return a64wordLE(baseOp | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil + } + return a64wordLE(baseOp | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil +} + +// encodeARM64MoveWide encodes a standalone MOVK: ($value, Rd) with the value +// sitting in one 16-bit chunk, the chunk's position becoming the hw field. +func encodeARM64MoveWide(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects $value, Rd", mnem) + } + rd := arm64RegNum(operandRegName(ops[1])) + if rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + // The opcode (MOVN 0, MOVZ 2, MOVK 3) and the width ride the table + // base, so MOVZ and MOVN come along for free. + opc := baseOp >> 29 & 3 + sf := baseOp >> 31 & 1 + v := arm64Imm64(ops[0]) + if v < 0 { + return nil, fmt.Errorf("%s: negative immediate %d", mnem, v) + } + hw := -1 + for i := range 4 { + if v>>(uint(i)*16)&0xFFFF != 0 { + hw = i + break + } + } + if hw < 0 { + hw = 0 // zero: every chunk is zero, hw = 0 carries it + } + for i := hw + 1; i < 4; i++ { + if v>>(uint(i)*16)&0xFFFF != 0 { + return nil, fmt.Errorf("%s: immediate %d does not fit one 16-bit chunk", mnem, v) + } + } + if sf == 0 && hw > 1 { + return nil, fmt.Errorf("%s: immediate %d out of range for the 32-bit form", mnem, v) + } + return a64wordLE(a64MoveWide(sf, opc, uint32(hw), uint32(v>>uint(hw*16)&0xFFFF), uint32(rd))), nil +} + // ---- Bitfield/EXTR encoding ---- // encodeARM64Bitfield encodes a bitfield instruction. @@ -1507,21 +2012,607 @@ func encodeARM64Extr(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er // ---- SIMD/NEON encoding ---- -// encodeARM64SIMD3 encodes a SIMD 3-operand instruction. -// VADD Vm, Vn, Vd → base | Rm<<16 | Rn<<5 | Rd (Q and size bits in base) -func encodeARM64SIMD3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { +// arm64VecOf parses a vector operand. The element suffix of V13.S[0] does +// not survive into the symbol name, so the verbatim operand text is tried +// first and the register name second. +func arm64VecOf(op *ast.Operand) (a64Vec, bool) { + if v, ok := a64VecReg(op.Raw); ok { + return v, true + } + return a64VecReg(operandRegName(op)) +} + +// arm64SimdHasElement reports whether any operand carries a lane index such +// as V13.S[0]. +func arm64SimdHasElement(ops []*ast.Operand) bool { + for _, op := range ops { + if v, ok := arm64VecOf(op); ok && v.hasIdx { + return true + } + } + return false +} + +// arm64SimdArrs validates that a SIMD operand run spells one arrangement, +// that it is the same on every operand that spells one, and that the table +// admits it. It returns the arrangement's index, with a64Arr8B for a bare +// V/F spelling. +func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) { + sel := a64Arr8B + for _, a := range arrs { + if a == "" { + continue + } + i := a64ArrIndex(a) + if i < 0 || specBit(i)&allowed == 0 { + return 0, fmt.Errorf("%s: invalid arrangement %q", mnem, a) + } + if sel != a64Arr8B && sel != i { + return 0, fmt.Errorf("%s: mixed arrangements", mnem) + } + sel = i + } + return sel, nil +} + +// specBit returns the a64SimdVSpec bitmask bit for an arrangement index. +func specBit(i int) uint16 { return 1 << uint(i) } + +// encodeARM64SimdV encodes an arrangement-aware three-register SIMD +// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a +// zero immediate takes its compare-against-zero form instead, and the +// polynomial multiplies read the arrangement from their source operands +// alone, the result spelling (H8, Q1) riding no encoding bits. +func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) { + if mnem == "VPMULL" || mnem == "VPMULL2" { + if len(ops) != 3 { + return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + vs := make([]a64Vec, 2) + arrs := make([]string, 2) + for i, op := range ops[:2] { + v, ok := arm64VecOf(op) + if !ok || v.hasIdx { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + vs[i], arrs[i] = v, v.arr + } + if _, ok := arm64VecOf(ops[2]); !ok { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + arr, err := arm64SimdArrs(mnem, arrs, spec.arrs) + if err != nil { + return nil, err + } + rd, _ := arm64VecOf(ops[2]) + return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil + } + if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) { + if arm64Imm64(ops[0]) != 0 { + return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) + } + vn, ok1 := arm64VecOf(ops[1]) + vd, ok2 := arm64VecOf(ops[2]) + if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f) + if err != nil { + return nil, err + } + return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil + } if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } - rm := arm64RegNum(operandRegName(ops[0])) - rn := arm64RegNum(operandRegName(ops[1])) - rd := arm64RegNum(operandRegName(ops[2])) - if rm < 0 || rn < 0 || rd < 0 { + vs := make([]a64Vec, 3) + arrs := make([]string, 3) + for i, op := range ops { + v, ok := arm64VecOf(op) + if !ok || v.hasIdx { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + vs[i], arrs[i] = v, v.arr + } + arr, err := arm64SimdArrs(mnem, arrs, spec.arrs) + if err != nil { + return nil, err + } + arrBits := a64ArrBits[arr] + if spec.fixed { + arrBits = 0 + } + return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil +} + +// encodeARM64SimdV2 encodes an arrangement-aware two-register SIMD +// instruction: word = base | arrBits | Rn<<5 | Rd. VMOV is the exception: +// its register pair spelling ORRs the source with itself into the +// destination (base | Rm<<16 | Rn<<5 | Rd with Rm = Rn = source). +func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + vs := make([]a64Vec, 2) + arrs := make([]string, 2) + for i, op := range ops { + v, ok := arm64VecOf(op) + if !ok || v.hasIdx { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + vs[i], arrs[i] = v, v.arr + } + if mnem == "VMOV" { + if (vs[0].arr != "" && vs[1].arr != "") && vs[0].arr != vs[1].arr { + return nil, fmt.Errorf("%s: mixed arrangements", mnem) + } + arr, err := arm64SimdArrs(mnem, arrs, spec.arrs) + if err != nil { + return nil, err + } + return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil + } + // VUADDLV spells its arrangement on the source alone; the rest take it + // on both. + vn, ok1 := arm64VecOf(ops[0]) + vd, ok2 := arm64VecOf(ops[1]) + if !ok1 || !ok2 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } - return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil + arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, spec.arrs) + if err != nil { + return nil, err + } + return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil } +// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX: +// ops ride Rm, Sa, Rn, Rd at 16, 10, 5 and 0) and its immediate relatives +// (VXAR with a 6-bit rotation at bits 15:10, VEXT with the index at bits +// 15:11 and a B16 flag at bit 30). +func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, error) { + switch mnem { + case "VEOR3", "VBCAX": + if len(ops) != 4 { + return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) + } + vs := make([]a64Vec, 4) + arrs := make([]string, 4) + for i, op := range ops { + v, ok := arm64VecOf(op) + if !ok || v.hasIdx { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + vs[i], arrs[i] = v, v.arr + } + if _, err := arm64SimdArrs(mnem, arrs, 1< 63 { + return nil, fmt.Errorf("%s: rotation %d out of range (0..63)", mnem, rot) + } + vs := make([]a64Vec, 3) + arrs := make([]string, 3) + for i, op := range ops[1:] { + v, ok := arm64VecOf(op) + if !ok || v.hasIdx { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + vs[i], arrs[i] = v, v.arr + } + if _, err := arm64SimdArrs(mnem, arrs, 1< int64(max) { + return nil, fmt.Errorf("%s: index %d out of range (0..%d)", mnem, idx, max) + } + return a64wordLE(base | b16 | uint32(idx)<<11 | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil + } + return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) +} + +// encodeARM64VTBL encodes VTBL Vidx.arr, [Vt1.arr, ...], Vdest.arr: the +// index register rides bits 19:16, the first table register bits 9:5, the +// destination bits 4:0 and the table length (registers minus one) bits +// 14:13. The table registers must be consecutive. +func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) { + if len(ops) < 3 { + return nil, fmt.Errorf("VTBL expects index, table list and destination") + } + vi, ok := arm64VecOf(ops[0]) + if !ok || vi.hasIdx { + return nil, fmt.Errorf("invalid index register in VTBL") + } + ts, end, ok := a64VecListOf(ops, 1) + if !ok || len(ts) < 1 || len(ts) > 4 { + return nil, fmt.Errorf("VTBL expects a table of one to four registers") + } + vd, ok := a64VecReg(operandRegName(ops[end+1])) + if !ok || end+2 != len(ops) || vd.hasIdx { + return nil, fmt.Errorf("invalid destination register in VTBL") + } + for i, t := range ts { + if t.hasIdx || t.reg != ts[0].reg+i { + return nil, fmt.Errorf("VTBL table registers must be consecutive") + } + } + q := uint32(0) + switch vi.arr { + case "B16": + q = 1 << 30 + case "B8", "": + default: + return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr) + } + return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil +} + +// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with +// lane indices: +// +// Vn.[i], Rd UMOV, element to general register +// Vn.[i], Vd.arr DUP, element across a vector +// Vn.[i], Vd.[j] INS, element to element +// Rs, Vd.[i] INS, general register into an element +func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + dst, dstVec := arm64VecOf(ops[1]) + dstGP := false + if !dstVec { + // A general-register spelling as destination (UMOV forms). + if rd := arm64RegNum(operandRegName(ops[1])); rd >= 0 { + dst, dstGP, dstVec = a64Vec{reg: rd}, true, true + } + } + if !dstVec { + return nil, fmt.Errorf("%s: invalid destination operand", mnem) + } + src, ok1 := arm64VecOf(ops[0]) + if (!ok1 || !src.hasIdx) && !dst.hasIdx { + return nil, fmt.Errorf("%s expects an element operand Vn.[i]", mnem) + } + if !ok1 || !src.hasIdx { + // General register into a vector element. + f, ok := a64ElemField(dst.arr, dst.idx) + if !ok { + return nil, fmt.Errorf("%s: invalid element operand", mnem) + } + rs := arm64RegNum(operandRegName(ops[0])) + if rs < 0 { + return nil, fmt.Errorf("%s: source must be a general register", mnem) + } + return a64wordLE(0x4e001c00 | f<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil + } + sf, ok := a64ElemField(src.arr, src.idx) + if !ok { + return nil, fmt.Errorf("%s: invalid element operand", mnem) + } + if dst.hasIdx { + // Element to element. + df, ok := a64ElemField(dst.arr, dst.idx) + if !ok { + return nil, fmt.Errorf("%s: invalid element operand", mnem) + } + return a64wordLE(0x6e000400 | df<<16 | sf>>1<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil + } + if dstGP { + // Element to a general register: UMOV, with the D form setting bit + // 30. + base := uint32(0x0e003c00) + if src.arr == "D" { + base = 0x4e003c00 + } + return a64wordLE(base | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil + } + if dst.arr == "" { + // Element across a bare V register. + return a64wordLE(0x5e000400 | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil + } + // Element across an arranged vector; the 128-bit arrangements set bit + // 30. + q := uint32(0) + switch dst.arr { + case "B16", "H8", "S4", "D2": + q = 1 << 30 + case "B8", "H4", "S2", "D1": + default: + return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr) + } + return a64wordLE(0x0e000400 | q | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil +} + +// encodeARM64VLDST encodes the SIMD structure loads and stores: +// +// VLD1 (Rn), [Vt.arr, ...] VST1 [Vt.arr, ...], (Rn) +// VLD1.P off(Rn), [Vt.arr, ...] VST1.P [Vt.arr, ...], off(Rn) +// VLD1R (Rn), [Vt.arr] VLD4R (Rn), [Vt.arr, Vt+1, Vt+2, Vt+3] +// +// The post-index forms set the post bit and Rm = 11111; the increment is +// implied by the register list, and a spelled offset rides along the way the +// toolchain's own encodings ignore it. +func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) { + load := strings.HasPrefix(mnem, "VLD") + listStart, memAt := 0, 1 + if load { + // Every load spells the memory operand first. + listStart, memAt = 1, 0 + } + vs, end, ok := a64VecListOf(ops, listStart) + if !ok { + return nil, fmt.Errorf("%s: invalid register list", mnem) + } + memIdx := end + 1 + if memAt == 0 { + memIdx = 0 + } + if memIdx >= len(ops) { + return nil, fmt.Errorf("%s expects a (Rn) memory operand", mnem) + } + rn, off := arm64MemWithFrame(ops[memIdx], arm64FrameInfo{}) + if rn < 0 { + return nil, fmt.Errorf("%s: invalid memory operand", mnem) + } + if off != 0 && post == 0 { + return nil, fmt.Errorf("%s: offset %d not supported, plain and list accesses take a plain (Rn) operand", mnem, off) + } + + // VLD1R loads one register and replicates; VLD4R loads four. + if strings.HasPrefix(mnem, "VLD1R") || strings.HasPrefix(mnem, "VLD4R") { + want := 1 + base := uint32(0x0d40c000) + if strings.HasPrefix(mnem, "VLD4R") { + want, base = 4, 0x0d60e000 + } + if len(vs) != want { + return nil, fmt.Errorf("%s expects a list of %d registers", mnem, want) + } + size, q, ok := a64ArrSizeQ(vs[0].arr) + if !ok { + return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr) + } + return a64wordLE(base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)), nil + } + + if len(vs) < 1 || len(vs) > 4 { + return nil, fmt.Errorf("%s expects a list of one to four registers", mnem) + } + for i, v := range vs { + if v.hasIdx || v.reg != vs[0].reg+i { + return nil, fmt.Errorf("%s: register list must be consecutive", mnem) + } + _, _, okArr := a64ArrSizeQ(v.arr) + if !okArr || (i > 0 && v.arr != vs[0].arr) { + return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, v.arr) + } + } + size, q, ok := a64ArrSizeQ(vs[0].arr) + if !ok { + return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr) + } + base := a64VLD1Base[len(vs)] + if !load { + base = a64VST1Base[len(vs)] + } + postBits := uint32(0) + if post != 0 { + postBits = 0x9f0000 + } + return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil +} + +// a64ArrSizeQ maps an arrangement to its size code (bits 11:10) and 128-bit +// flag for the structure load/store words. +func a64ArrSizeQ(arr string) (size, q uint32, ok bool) { + switch arr { + case "B8": + return 0, 0, true + case "B16": + return 0, 1, true + case "H4": + return 1, 0, true + case "H8": + return 1, 1, true + case "S2": + return 2, 0, true + case "S4": + return 2, 1, true + case "D1": + return 3, 0, true + case "D2": + return 3, 1, true + } + return 0, 0, false +} + +// encodeARM64ShiftImm encodes a SIMD shift by immediate: +// word = base | Q<<30 | immh:immb<<16 | Rn<<5 | Rd, where immh:immb is the +// element size plus the shift for a left shift (VSHL) and twice the element +// size minus the shift for right shifts (VUSHR, VSRI). +func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 3 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects 3 operands ($shift, Vn.arr, Vd.arr)", mnem) + } + sh := arm64Imm64(ops[0]) + vn, ok1 := arm64VecOf(ops[1]) + vd, ok2 := arm64VecOf(ops[2]) + if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx || vn.arr != vd.arr { + return nil, fmt.Errorf("%s: operands must share one arrangement", mnem) + } + var esize int64 + q := uint32(0) + switch vn.arr { + case "B8", "B": + esize = 8 + case "B16": + esize, q = 8, 1 + case "H4", "H": + esize = 16 + case "H8": + esize, q = 16, 1 + case "S2", "S": + esize = 32 + case "S4": + esize, q = 32, 1 + case "D2": + esize, q = 64, 1 + default: + return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vn.arr) + } + var immval int64 + switch mnem { + case "VSHL": + if sh < 0 || sh >= esize { + return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1) + } + immval = esize + sh + default: // VUSHR, VSRI + if sh < 1 || sh > esize { + return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize) + } + immval = 2*esize - sh + } + return a64wordLE(base | q<<30 | uint32(immval)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil +} + +// encodeARM64MoviLit loads a large vector constant the way the toolchain +// does: ADRP R27 and ADD materialise the literal's address, then FMOVS, +// FMOVD or the 128-bit FMOVQ form loads it, with R_ADDRARM64 relocations +// against a read-only literal the file assembler lays out. +func encodeARM64MoviLit(mnem string, ldr uint32, ops []*ast.Operand, relocs *[]Reloc, lits *arm64Literals) ([]byte, error) { + var vd a64Vec + var ok bool + var data []byte + switch mnem { + case "VMOVS", "VMOVD": + if len(ops) != 2 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects $value, Vd", mnem) + } + v := arm64Imm64(ops[0]) + if mnem == "VMOVS" { + if v < -2147483648 || v > 0xFFFFFFFF { + return nil, fmt.Errorf("%s: constant does not fit 32 bits", mnem) + } + data = a64wordLE(uint32(v)) + } else { + data = a64WordsLE(uint32(v), uint32(v>>32)) + } + vd, ok = arm64VecOf(ops[1]) + if !ok || vd.hasIdx || vd.arr != "" { + return nil, fmt.Errorf("%s: destination must be a bare V register", mnem) + } + case "VMOVQ": + if len(ops) != 3 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) { + return nil, fmt.Errorf("VMOVQ expects $lo, $hi, Vd") + } + lo, hi := arm64Imm64(ops[0]), arm64Imm64(ops[1]) + data = a64WordsLE(uint32(lo), uint32(lo>>32), uint32(hi), uint32(hi>>32)) + vd, ok = arm64VecOf(ops[2]) + if !ok || vd.hasIdx || vd.arr != "" { + return nil, fmt.Errorf("VMOVQ: destination must be a bare V register") + } + default: + return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) + } + name := lits.add(moviLitName(mnem, data), data) + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: name, Kind: RelArm64Addr}, + Reloc{Off: 4, After: 4, Name: name, Kind: RelArm64Addr}, + ) + } + return a64WordsLE( + a64ADR(1, 0, 0, 27), + a64AddSub(1, 0, 0, 0, 0, 27, 27), + ldr|0<<10|27<<5|uint32(vd.reg), + ), nil +} + +// moviLitName mirrors the toolchain's literal naming: $i32/$i64/$i128 +// followed by the constant's value in hex. +func moviLitName(mnem string, data []byte) string { + switch mnem { + case "VMOVS": + v := uint32(data[0]) | uint32(data[1])<<8 | uint32(data[2])<<16 | uint32(data[3])<<24 + return "$i32." + strconv.FormatUint(uint64(v), 16) + case "VMOVD": + v := uint64(data[0]) | uint64(data[1])<<8 | uint64(data[2])<<16 | uint64(data[3])<<24 | + uint64(data[4])<<32 | uint64(data[5])<<40 | uint64(data[6])<<48 | uint64(data[7])<<56 + return "$i64." + strconv.FormatUint(v, 16) + default: + hi := uint64(data[8]) | uint64(data[9])<<8 | uint64(data[10])<<16 | uint64(data[11])<<24 | + uint64(data[12])<<32 | uint64(data[13])<<40 | uint64(data[14])<<48 | uint64(data[15])<<56 + lo := uint64(data[0]) | uint64(data[1])<<8 | uint64(data[2])<<16 | uint64(data[3])<<24 | + uint64(data[4])<<32 | uint64(data[5])<<40 | uint64(data[6])<<48 | uint64(data[7])<<56 + return "$i128." + strings.Repeat("0", max(0, 16-len(strconv.FormatUint(hi, 16)))) + + strconv.FormatUint(hi, 16) + strings.Repeat("0", max(0, 16-len(strconv.FormatUint(lo, 16)))) + + strconv.FormatUint(lo, 16) + } +} + +// arm64Literals collects the read-only constants the VMOVS/VMOVD/VMOVQ +// loads refer to. Names follow the toolchain's $i32/$i64/$i128 spellings so +// equal constants deduplicate to one literal. +type arm64Literals struct { + order []Arm64Literal + seen map[string]bool +} + +// Arm64Literal is one pooled vector constant. +type Arm64Literal struct { + Name string + Data []byte +} + +// add registers a literal under its name and returns it. +func (l *arm64Literals) add(name string, data []byte) string { + if l.seen == nil { + l.seen = map[string]bool{} + } + if !l.seen[name] { + l.seen[name] = true + l.order = append(l.order, Arm64Literal{Name: name, Data: data}) + } + return name +} + +// list returns the literals in first-use order. +func (l *arm64Literals) list() []Arm64Literal { return l.order } + // AssembleFileARM64 assembles every TEXT function of a parsed arm64 file // and lays out its static symbols (GLOBL/DATA) in a data section behind the // code. SB references in the code are encoded as ADRP pairs with zero @@ -1534,15 +2625,26 @@ func AssembleFileARM64(f *ast.File) (*Image, error) { } img := &Image{Symbols: map[string]int{}, SourcePath: f.Path} + var pendingLits []Arm64Literal + litSeen := map[string]bool{} for _, d := range f.Decls { t, ok := d.(*ast.Text) if !ok { continue } - code, labels, relocs, lines, spadj, err := assembleARM64(t) + code, labels, relocs, lines, spadj, lits, err := assembleARM64(t) if err != nil { return nil, fmt.Errorf("%s: %w", t.Name.Name, err) } + // The literals this function's constant loads refer to join the + // data section once, deduplicated by name. + for _, lit := range lits { + if _, seen := litSeen[lit.Name]; seen { + continue + } + litSeen[lit.Name] = true + pendingLits = append(pendingLits, lit) + } fl := FuncLayout{ Name: t.Name.Name, Pkg: t.Name.Pkg, @@ -1589,6 +2691,24 @@ func AssembleFileARM64(f *ast.File) (*Image, error) { Dupok: d.dupok, }) } + // The read-only literals the VMOVS/VMOVD/VMOVQ constant loads refer to + // follow the declared data, deduplicated across the file. + for _, lit := range pendingLits { + pos := dataStart + len(img.Data) + for pos%16 != 0 { + img.Data = append(img.Data, 0) + pos++ + } + img.Symbols[lit.Name] = pos + img.Data = append(img.Data, lit.Data...) + img.DataSyms = append(img.DataSyms, DataSymbol{ + Name: lit.Name, + Offset: len(img.Data) - len(lit.Data), + Size: len(lit.Data), + Rodata: true, + Dupok: true, + }) + } markExternals(img, dataSyms) return img, nil diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index e91f77a..6c18727 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -27,7 +27,13 @@ package asm // Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET) // ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd -import "maps" +import ( + "maps" + "strconv" + "strings" + + "sourcedock.dev/petrbalvin/gasm-devkit/ast" +) // arm64RegNum returns the 5-bit register number for an AArch64 register name: // R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the @@ -288,7 +294,25 @@ const ( a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP a64FLSE // LSE atomics: LDADD, CAS, SWP - a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL + a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS + a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms + a64FCondCmp // conditional compare: CCMP, CCMN + a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms + a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms + a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD + a64FAcqRel // acquire/release: LDAR family, STLR family + a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM + a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ... + a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ... + a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ... + a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd + a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV + a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT + a64FVTBL // SIMD table lookup: VTBL + a64FDUP // SIMD element moves: VDUP, VMOV with element indices + a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R + a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI + a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool) ) // a64Enc is one instruction's encoding: its bit layout (format) and the @@ -622,10 +646,403 @@ func init() { a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10} a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10} - // ---- SIMD basics ---- - a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400} - a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400} - a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00} + // ---- SIMD: the arrangement-aware tables in this file carry VADD, + // VSUB, VMUL and every other three-register vector op. ---- + + // ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ---- + dp1 := map[string]uint32{ + "RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800, + "REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400, + "RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400, + } + for m, op := range dp1 { + a64InstrTable[m] = a64Enc{format: a64FDP1, op: op} + } + + // ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ---- + a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000} + a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000} + a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000} + a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000} + + // ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ---- + a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000} + a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000} + a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000} + a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000} + + // ---- system operations ---- + for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} { + a64InstrTable[m] = a64Enc{format: a64FSys} + } + + // ---- compare/test and branch ---- + a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000} + a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000} + a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000} + a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000} + a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000} + a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000} + + // ---- load/store pair (signed offset) ---- + a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000} + a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000} + a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000} + a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000} + a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000} + a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000} + + // ---- acquire/release loads and stores ---- + a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00} + a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00} + a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00} + a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00} + a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00} + a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00} + a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00} + a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00} + + // ---- LSE atomics with acquire and release semantics ---- + // CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/ + // LDOR/SWP families leave Rs free for the returned value. + lse := map[string]uint32{ + "CASALD": 0xc8e0fc00, + "CASALW": 0x88e0fc00, + "LDADDALD": 0xf8e00000, + "LDADDALW": 0xb8e00000, + "LDCLRALB": 0x38e01000, + "LDCLRALW": 0xb8e01000, + "LDCLRALD": 0xf8e01000, + "LDORALB": 0x38e03000, + "LDORALW": 0xb8e03000, + "LDORALD": 0xf8e03000, + "SWPALB": 0x38e08000, + "SWPALW": 0xb8e08000, + "SWPALD": 0xf8e08000, + } + for m, op := range lse { + a64InstrTable[m] = a64Enc{format: a64FLSE, op: op} + } + + // ---- carry-setting/carry-using arithmetic and widening multiply ---- + // MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate + // register preset to ZR (bits 14:10 = 11111). + dpsrExtra := map[string]uint32{ + "ADC": 0x9a000000, "ADCW": 0x1a000000, + "ADCS": 0xba000000, "ADCSW": 0x3a000000, + "SBC": 0xda000000, "SBCW": 0x5a000000, + "SBCS": 0xfa000000, "SBCSW": 0x7a000000, + "MUL": 0x9b007c00, "MULW": 0x1b007c00, + "SMULH": 0x9b407c00, "UMULH": 0x9bc07c00, + } + for m, op := range dpsrExtra { + a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op} + } + + // ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ---- + crypto2 := map[string]uint32{ + "AESD": 0x4e285800, "AESE": 0x4e284800, + "AESIMC": 0x4e287800, "AESMC": 0x4e286800, + "SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800, + "SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000, + } + for m, op := range crypto2 { + a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op} + } + crypto3 := map[string]uint32{ + "SHA1C": 0x5e000000, "SHA1P": 0x5e001000, + "SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000, + "SHA256H": 0x5e004000, "SHA256H2": 0x5e005000, + "SHA256SU1": 0x5e006000, "SHA512H": 0xce608000, + "SHA512H2": 0xce608400, "SHA512SU1": 0xce608800, + } + for m, op := range crypto3 { + a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op} + } + + // ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ---- + a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000} + a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000} + a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000} + a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000} + a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL} + a64InstrTable["VDUP"] = a64Enc{format: a64FDUP} + a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000} + a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000} + a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000} + a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10} + a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10} + a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10} + a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VST1"] = a64Enc{format: a64FVLDST} + a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST} +} + +// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word, +// the set of arrangements it accepts as a bitmask over the a64Arr index and, +// for instructions that exist at a single arrangement and carry that +// arrangement's bits inside the base already, the fixed flag. +type a64SimdVSpec struct { + base uint32 + arrs uint16 + fixed bool +} + +// a64Arr names the vector arrangements the encoders deal with, indexed by +// a64Arr. The source spellings put the element letter first: B8, H4, S2, +// D1 and the 128-bit halves B16, H8, S4, D2. +const ( + a64Arr8B = iota + a64Arr16B + a64Arr4H + a64Arr8H + a64Arr2S + a64Arr4S + a64Arr2D + a64ArrD1 + a64ArrQ1 + a64ArrCount +) + +// a64ArrNames maps an arrangement to its source spelling (element letter +// first, as the toolchain writes it). +var a64ArrNames = [a64ArrCount]string{ + a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8", + a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1", +} + +// a64ArrIndex resolves a source spelling to its a64Arr index, -1 when +// unknown. +func a64ArrIndex(s string) int { + for i, n := range a64ArrNames { + if n == s { + return i + } + } + return -1 +} + +// a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q) +// as it appears in element operands such as V13.S[0]. +func a64ElemLetter(s string) bool { + switch s { + case "B", "H", "S", "D", "Q": + return true + } + return false +} + +// a64ArrBits carries the fixed bits an arrangement contributes to the +// three-same word shape: the element size at bits 23:22 and the 128-bit +// flag at bit 30. Bit 29 belongs to the instruction's own base. +var a64ArrBits = [a64ArrCount]uint32{ + a64Arr8B: 0, + a64Arr16B: 1 << 30, + a64Arr4H: 1 << 22, + a64Arr8H: 1<<30 | 1<<22, + a64Arr2S: 1 << 23, + a64Arr4S: 1<<30 | 1<<23, + a64Arr2D: 1<<30 | 1<<23 | 1<<22, + a64ArrD1: 1<<23 | 1<<22, + a64ArrQ1: 0, +} + +// a64SimdVTable holds the arrangement-aware three-register SIMD +// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base +// word and arrangement bit was read off go tool asm. +var a64SimdVTable = map[string]a64SimdVSpec{ + "VADD": {0x0e208400, 0x7f, false}, + "VSUB": {0x2e208400, 0x7f, false}, + "VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S + "VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only + "VEOR": {0x2e201c00, 0x03, false}, + "VORR": {0x0ea01c00, 0x03, false}, + "VADDP": {0x0e20bc00, 0x7f, false}, + "VZIP1": {0x0e003800, 0x7f, false}, + "VZIP2": {0x0e007800, 0x7f, false}, + "VCMEQ": {0x2e208c00, 0x7f, false}, + "VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only + "VPMULL": {0x0e20e000, 1<= 0 { + v.arr = strings.TrimSpace(s[i+1:]) + s = s[:i] + } + if v.arr != "" { + // Element form: B[3], S[2] and friends. + if j := strings.IndexByte(v.arr, '['); j >= 0 { + k := strings.LastIndexByte(v.arr, ']') + if k < j { + return v, false + } + n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k])) + if err != nil || n < 0 { + return v, false + } + v.idx, v.hasIdx = n, true + v.arr = strings.TrimSpace(v.arr[:j]) + } + if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) { + return v, false + } + } + if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') { + return v, false + } + n := 0 + for i := 1; i < len(s); i++ { + if s[i] < '0' || s[i] > '9' { + return v, false + } + n = n*10 + int(s[i]-'0') + } + if n > 31 { + return v, false + } + v.reg = n + return v, true +} + +// a64ElemField encodes a lane index for the copy/insert group: imm5 = the +// index shifted by the element scale, with the scale's own bit set. B gets +// shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4. +func a64ElemField(arr string, idx int) (uint32, bool) { + var shift, low uint32 + switch arr { + case "B8", "B16", "B": + shift, low = 1, 1 + case "H4", "H8", "H": + shift, low = 2, 2 + case "S2", "S4", "S": + shift, low = 3, 4 + case "D1", "D2", "D": + shift, low = 4, 8 + default: + return 0, false + } + if idx < 0 || idx >= 1<<(5-shift) { + return 0, false + } + return uint32(idx)<= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") { + return nil, 0, false + } + end = start + for end < len(ops) { + if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") { + break + } + end++ + } + if end >= len(ops) { + return nil, 0, false + } + for i := start; i <= end; i++ { + s := strings.TrimSpace(ops[i].Raw) + s = strings.TrimPrefix(s, "[") + s = strings.TrimSuffix(s, "]") + if s == "" && len(ops) > start+1 { + return nil, 0, false + } + for part := range strings.SplitSeq(s, ",") { + v, ok := a64VecReg(part) + if !ok { + return nil, 0, false + } + vs = append(vs, v) + } + } + return vs, end, true } // ---- load/store helper tables ---- diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index 980a8da..3d0fc86 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -472,12 +472,456 @@ TEXT ·f(SB), NOSPLIT, $0-0 } } -// TestArm64SIMD tests SIMD encoding (via the instruction table). +// TestArm64SIMD tests SIMD encoding (via the arrangement-aware table). func TestArm64SIMD(t *testing.T) { - // Verify SIMD instructions are in the table. - for _, mnem := range []string{"VADD", "VSUB", "VMUL"} { - if _, ok := a64InstrTable[mnem]; !ok { - t.Errorf("%s not in instruction table", mnem) + // Verify SIMD instructions are in the arrangement table. + for _, mnem := range []string{"VADD", "VSUB", "VMUL", "VAND", "VEOR", "VORR", "VCMEQ", "VZIP1", "VZIP2"} { + if _, ok := a64SimdVTable[mnem]; !ok { + t.Errorf("%s not in the SIMD arrangement table", mnem) + } + } +} + +// TestArm64CarryAndBitOps pins the carry-setting arithmetic, the widening +// multiplies and the data-processing (1 source) group against go tool asm. +func TestArm64CarryAndBitOps(t *testing.T) { + got := arm64Words(t, "\tADC R0, R2, R12\n\tADCS $0, R1\n\tSBCS R5, R9, R5\n\tSBC R25, R10, R26\n"+ + "\tMUL R4, R3, R0\n\tUMULH R24, R20, R24\n\tSMULH R1, R2, R3\n\tMSUB R19, R16, R26, R2\n"+ + "\tRBIT R11, R4\n\tREV R1, R2\n\tCLZ R21, R9\n\tREVW R1, R2\n\tCLSW R1, R2\n") + want := []uint32{ + 0x9a00004c, // ADC R12, R2, R0 + 0xba1f0021, // ADCS R1, R1, ZR + 0xfa050125, // SBCS R5, R9, R5 + 0xda19015a, // SBC R26, R10, R25 + 0x9b047c60, // MUL R0, R3, R4 + 0x9bd87e98, // UMULH R24, R20, R24 + 0x9b417c43, // SMULH R3, R2, R1 + 0x9b13c342, // MSUB R2, R26, R19, R16 + 0xdac00164, // RBIT R4, R11 + 0xdac00c22, // REV R2, R1 + 0xdac012a9, // CLZ R9, R21 + 0x5ac00822, // REVW R2, R1 + 0x5ac01422, // CLSW R2, R1 + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64BitfieldExtract pins UBFX/SBFX: immr wraps to the register +// width, an out-of-range imms is an error. +func TestArm64BitfieldExtract(t *testing.T) { + got := arm64Words(t, "\tUBFX $33, R17, $25, R5\n\tUBFXW $4, R1, $9, R2\n") + want := []uint32{ + 0xd361e625, // UBFX immr=1 (33 wrapped), imms=25 + 0x53043022, // UBFXW immr=4, imms=9 + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } + for _, body := range []string{"\tUBFX $33, R17, $70, R5\n", "\tUBFX $-1, R17, $3, R5\n"} { + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if _, err := AssembleFileARM64(f); err == nil { + t.Errorf("%s: expected an error, got none", body) + } + } +} + +// TestArm64CondCompare pins CCMP/CCMN. +func TestArm64CondCompare(t *testing.T) { + got := arm64Words(t, "\tCCMP LE, R7, $19, $3\n\tCCMP LT, R30, R6, $7\n\tCCMN EQ, R1, R2, $3\n\tCCMPW LE, R7, $19, $3\n") + want := []uint32{ + 0xfa53d8e3, // CCMP imm form + 0xfa46b3c7, // CCMP register form + 0xba420023, // CCMN register form + 0x7a53d8e3, // CCMPW + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64CompareBranch pins CBZ/CBNZ/TBZ/TBNZ against a label five and +// six words ahead, matching go tool asm's own offsets. +func TestArm64CompareBranch(t *testing.T) { + // Layout: CBZ(0) TBZ(4) TBNZ(8) CBNZ(12) NOP(16) NOP(17th word...) done. + src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" + + "\tCBZ R1, done\n\tTBZ $4, R7, done\n\tTBNZ $33, R7, done\n\tCBNZW R2, done\n" + + "\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n" + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + got := leWords(img.Code) + // done sits at word 6 from each branch's own pc: CBZ rel 6, TBZ rel 5, + // TBNZ rel 4, CBNZW rel 3. + want := []uint32{ + 0xb40000c1, // CBZ R1, +6 + 0x362000a7, // TBZ $4, R7, +5 + 0xb7080087, // TBNZ $33, R7, +4 + 0x35000062, // CBNZW R2, +3 + 0xd503201f, 0xd503201f, 0xd503201f, + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64ADR pins ADR against a forward label. +func TestArm64ADR(t *testing.T) { + src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" + + "\tADR done, R10\n\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n" + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + got := leWords(img.Code) + // rel = 12 bytes: immlo 0, immhi 3. + want := []uint32{0x1000006a, 0xd503201f, 0xd503201f, 0xd503201f, 0xd65f03c0} + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64PairLoadStore pins LDP/STP/LDPW/FLDPD/FSTPD. +func TestArm64PairLoadStore(t *testing.T) { + got := arm64Words(t, "\tSTP (R2, R3), 8(R5)\n\tLDP -8(R5), (R2, R3)\n\tLDPW 4(R0), (R1, R2)\n\tSTPW (R1, R2), 4(R0)\n"+ + "\tFLDPD 8(R0), (F1, F2)\n\tFSTPD (F3, F4), -8(R5)\n") + want := []uint32{ + 0xa9008ca2, // STP (R2, R3), 8(R5) + 0xa97f8ca2, // LDP -8(R5), (R2, R3) + 0x29408801, // LDPW 4(R0), (R1, R2) + 0x29008801, // STPW (R1, R2), 4(R0) + 0x6d408801, // FLDPD 8(R0), (F1, F2) + 0x6d3f90a3, // FSTPD (F3, F4), -8(R5) + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64AcquireRelease pins LDAR/STLR and the acquire/release LSE +// families. +func TestArm64AcquireRelease(t *testing.T) { + got := arm64Words(t, "\tLDAR (R27), R22\n\tLDARB (R25), R2\n\tLDARW (R12), R29\n\tSTLR R3, (R24)\n\tSTLRB R11, (R22)\n"+ + "\tCASALD R5, (R6), R7\n\tLDADDALD R5, (R6), R7\n\tLDCLRALB R5, (R6), R7\n\tLDORALD R5, (RSP), R7\n\tSWPALW R5, (R6), R7\n") + want := []uint32{ + 0xc8dfff76, // LDAR R22, (R27) + 0x08dfff22, // LDARB R2, (R25) + 0x88dffd9d, // LDARW R29, (R12) + 0xc89fff03, // STLR R3, (R24) + 0x089ffecb, // STLRB R11, (R22) + 0xc8e5fcc7, // CASALD R7, (R6), R5 + 0xf8e500c7, // LDADDALD R7, (R6), R5 + 0x38e510c7, // LDCLRALB R7, (R6), R5 + 0xf8e533e7, // LDORALD R7, (RSP), R5 + 0xb8e580c7, // SWPALW R7, (R6), R5 + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64System pins BRK, SVC, the barriers, cache maintenance and the +// system register accesses. +func TestArm64System(t *testing.T) { + got := arm64Words(t, "\tBRK $35943\n\tBRK\n\tSVC $7165\n\tDMB $1\n\tDSB $1\n\tISB $15\n"+ + "\tDC ZVA, R4\n\tDC IVAC, R1\n\tMRS DCZID_EL0, R3\n\tMRS CNTVCT_EL0, R0\n\tMSR $9, DAIFSet\n\tMSR $3, SPSel\n"+ + "\tPRFM (R0), PLDL1KEEP\n\tPRFM (R3), PLDL3KEEP\n\tPRFM (R2), $25\n") + want := []uint32{ + 0xd4318ce0, // BRK $35943 + 0xd4200000, // BRK + 0xd4037fa1, // SVC $7165 + 0xd50331bf, // DMB $1 + 0xd503319f, // DSB $1 + 0xd5033fdf, // ISB $15 + 0xd50b7424, // DC ZVA, R4 + 0xd5087621, // DC IVAC, R1 + 0xd53b00e3, // MRS DCZID_EL0, R3 + 0xd53be040, // MRS CNTVCT_EL0, R0 + 0xd50349df, // MSR $9, DAIFSet + 0xd50043bf, // MSR $3, SPSel + 0xf9800000, // PRFM (R0), PLDL1KEEP + 0xf9800064, // PRFM (R3), PLDL3KEEP + 0xf9800059, // PRFM (R2), $25 + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64Crypto pins the AES and SHA families. +func TestArm64Crypto(t *testing.T) { + got := arm64Words(t, "\tAESE V31.B16, V29.B16\n\tAESD V22.B16, V19.B16\n\tAESIMC V12.B16, V27.B16\n\tAESMC V14.B16, V28.B16\n"+ + "\tSHA1C V8.S4, V8, V2\n\tSHA1H V17, V25\n\tSHA1P V3.S4, V20, V27\n\tSHA1SU0 V17.S4, V13.S4, V16.S4\n\tSHA1SU1 V24.S4, V23.S4\n"+ + "\tSHA256H V4.S4, V2, V11\n\tSHA256H2 V6.S4, V16, V11\n\tSHA256SU0 V0.S4, V16.S4\n\tSHA256SU1 V31.S4, V3.S4, V15.S4\n"+ + "\tSHA512H V2.D2, V1, V0\n\tSHA512H2 V4.D2, V3, V2\n\tSHA512SU0 V9.D2, V8.D2\n\tSHA512SU1 V7.D2, V6.D2, V5.D2\n") + want := []uint32{ + 0x4e284bfd, // AESE + 0x4e285ad3, // AESD + 0x4e28799b, // AESIMC + 0x4e2869dc, // AESMC + 0x5e080102, // SHA1C + 0x5e280a39, // SHA1H + 0x5e03129b, // SHA1P + 0x5e1131b0, // SHA1SU0 + 0x5e281b17, // SHA1SU1 + 0x5e04404b, // SHA256H + 0x5e06520b, // SHA256H2 + 0x5e282810, // SHA256SU0 + 0x5e1f606f, // SHA256SU1 + 0xce628020, // SHA512H + 0xce648462, // SHA512H2 + 0xcec08128, // SHA512SU0 + 0xce6788c5, // SHA512SU1 + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64SIMDLogical pins the arrangement-aware three- and two-register +// SIMD paths. +func TestArm64SIMDLogical(t *testing.T) { + got := arm64Words(t, "\tVADD V1.B16, V2.B16, V3.B16\n\tVAND V4.B16, V4.B16, V9.B16\n\tVEOR V0.B16, V1.B16, V0.B16\n"+ + "\tVORR V5.B16, V4.B16, V3.B16\n\tVADDP V1.H8, V2.H8, V3.H8\n\tVZIP1 V16.H8, V3.H8, V19.H8\n\tVZIP2 V22.D2, V25.D2, V21.D2\n"+ + "\tVCMEQ V24.S4, V13.S4, V12.S4\n\tVCMEQ $0, V2.H4, V3.H4\n\tVREV32 V2.H8, V1.H8\n\tVREV64 V2.S4, V3.S4\n\tVUADDLV V31.S4, V11\n"+ + "\tVPMULL V2.D1, V1.D1, V3.Q1\n\tVPMULL2 V2.B16, V1.B16, V4.H8\n\tVRAX1 V26.D2, V29.D2, V30.D2\n\tVMOV V2.B16, V4.B16\n") + want := []uint32{ + 0x4e218443, // VADD 16B + 0x4e241c89, // VAND + 0x6e201c20, // VEOR + 0x4ea51c83, // VORR + 0x4e61bc43, // VADDP 8H + 0x4e503873, // VZIP1 8H + 0x4ed67b35, // VZIP2 2D + 0x6eb88dac, // VCMEQ 4S + 0x0e609843, // VCMEQ $0, 4H + 0x6e600841, // VREV32 8H + 0x4ea00843, // VREV64 4S + 0x6eb03beb, // VUADDLV 4S + 0x0ee2e023, // VPMULL D1 + 0x4e22e024, // VPMULL2 16B + 0xce7a8fbe, // VRAX1 2D + 0x4ea21c44, // VMOV 16B pair + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64SIMDWide pins the four-register crypto group, VXAR, VEXT and the +// shift-by-immediate encodings. +func TestArm64SIMDWide(t *testing.T) { + got := arm64Words(t, "\tVEOR3 V2.B16, V7.B16, V12.B16, V25.B16\n\tVBCAX V1.B16, V2.B16, V26.B16, V31.B16\n"+ + "\tVXAR $63, V27.D2, V21.D2, V26.D2\n\tVEXT $4, V2.B8, V1.B8, V3.B8\n\tVEXT $8, V2.B16, V1.B16, V3.B16\n"+ + "\tVSHL $7, V22.D2, V25.D2\n\tVUSHR $6, V22.H8, V23.H8\n\tVSRI $24, V1.S4, V2.S4\n") + want := []uint32{ + 0xce070999, // VEOR3 + 0xce22075f, // VBCAX + 0xce9bfeba, // VXAR + 0x2e022023, // VEXT B8 + 0x6e024023, // VEXT B16 + 0x4f4756d9, // VSHL D2 $7 + 0x6f1a06d7, // VUSHR H8 $6 + 0x6f284422, // VSRI S4 $24 + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64SIMDElement pins VDUP and the VMOV element forms. +func TestArm64SIMDElement(t *testing.T) { + got := arm64Words(t, "\tVDUP V31.B[15], V18\n\tVDUP V19.S[3], V18.S4\n\tVDUP V1.D[1], V2.D2\n"+ + "\tVMOV V13.S[0], R20\n\tVMOV V11.B[11], V16.B[12]\n\tVMOV R20, V21.B[2]\n") + want := []uint32{ + 0x5e1f07f2, // VDUP element to register + 0x4e1c0672, // VDUP element across S4 + 0x4e180422, // VDUP element across D2 + 0x0e043db4, // VMOV element to register + 0x6e195d70, // VMOV element to element + 0x4e051e95, // VMOV register into element + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64SIMDLoadStore pins the structure loads and stores. +func TestArm64SIMDLoadStore(t *testing.T) { + got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+ + "\tVLD1.P 32(R1), [V2.B16, V3.B16]\n\tVST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)\n\tVST1.P [V2.B16], (R1)\n"+ + "\tVLD1R (R1), [V9.B8]\n\tVLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]\n") + want := []uint32{ + 0x4c407055, // VLD1 one register + 0x4c40a022, // VLD1 two registers + 0x0c402fae, // VLD1 four registers D1 + 0x4cdfa022, // VLD1.P two registers + 0x4c0029c2, // VST1 four registers S4 + 0x4c9f7022, // VST1.P one register + 0x0d40c029, // VLD1R + 0x0d60e000, // VLD4R + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64MoviLiteral pins the VMOVS/VMOVD/VMOVQ constant loads: three +// words each (ADRP, ADD, wide load) plus the pooled literal in the data +// section. +func TestArm64MoviLiteral(t *testing.T) { + src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" + + "\tVMOVS $0x80402010, V11\n\tVMOVD $0x8040201008040201, V20\n" + + "\tVMOVQ $0x7040201008040201, $0x8040201008040201, V10\n\tRET\n" + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + if img.Funcs[0].Size != 12*3+4 { + t.Errorf("func size = %d, want %d", img.Funcs[0].Size, 12*3+4) + } + want := []uint32{ + 0x9000001b, 0x9100037b, 0xbd40036b, // VMOVS: ADRP, ADD, LDR S + 0x9000001b, 0x9100037b, 0xfd400374, // VMOVD: ADRP, ADD, LDR D + 0x9000001b, 0x9100037b, 0x3dc0036a, // VMOVQ: ADRP, ADD, LDR Q + 0xd65f03c0, + } + got := leWords(img.Code) + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } + // The literals sit in the data section. + var found32, found64, found128 bool + for _, d := range img.DataSyms { + switch d.Name { + case "$i32.80402010": + found32 = d.Size == 4 + case "$i64.8040201008040201": + found64 = d.Size == 8 + case "$i128.80402010080402017040201008040201": + found128 = d.Size == 16 + } + } + if !found32 || !found64 || !found128 { + t.Errorf("literals missing: i32=%v i64=%v i128=%v", found32, found64, found128) + } +} + +// TestArm64MOVK pins standalone MOVK with the hw field derived from the +// chunk position. +func TestArm64MOVK(t *testing.T) { + got := arm64Words(t, "\tMOVK $1234, R5\n\tMOVK $305397760, R5\n\tMOVKW $1234, R5\n") + want := []uint32{ + 0xf2809a45, // MOVK hw=0 + 0xf2a24685, // MOVK hw=1 + 0x72809a45, // MOVKW hw=0 + 0xd65f03c0, + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) } } } diff --git a/testdata/verify/atomics_arm64.s b/testdata/verify/atomics_arm64.s new file mode 100644 index 0000000..9bac7b2 --- /dev/null +++ b/testdata/verify/atomics_arm64.s @@ -0,0 +1,72 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 synchronisation instructions: the +// acquire/release loads and stores, the exclusive family and the LSE +// atomics with acquire and release semantics, plus the register-pair +// loads and stores. Every function is byte-compared against go tool asm. + +#include "textflag.h" + +// func acquireRelease() +TEXT ·acquireRelease(SB), NOSPLIT, $0-0 + LDAR (R1), R2 + LDARB (R3), R4 + LDARH (R5), R6 + LDARW (R7), R8 + STLR R2, (R1) + STLRB R4, (R3) + STLRH R6, (R5) + STLRW R8, (R7) + RET + +// func exclusive() +TEXT ·exclusive(SB), NOSPLIT, $0-0 + LDAXR (R1), R2 + LDAXRB (R3), R4 + LDAXRW (R5), R6 + STLXR R2, (R1), R8 + STLXRB R4, (R3), R8 + STLXRW R6, (R5), R8 + RET + +// func lseAcquireRelease() +TEXT ·lseAcquireRelease(SB), NOSPLIT, $0-0 + CASALD R1, (R3), R2 + CASALW R4, (R6), R5 + LDADDALD R1, (R3), R2 + LDADDALW R4, (R6), R5 + LDCLRALB R1, (R3), R2 + LDCLRALW R4, (R6), R5 + LDCLRALD R1, (R3), R2 + LDORALB R1, (R3), R2 + LDORALW R4, (R6), R5 + LDORALD R1, (R3), R2 + SWPALB R1, (R3), R2 + SWPALW R4, (R6), R5 + SWPALD R1, (R3), R2 + RET + +// func lseBase() +TEXT ·lseBase(SB), NOSPLIT, $0-0 + LDADDD R1, (R3), R2 + LDADDW R4, (R6), R5 + CASD R1, (R3), R2 + CASW R4, (R6), R5 + SWPD R1, (R3), R2 + SWPW R4, (R6), R5 + RET + +// func pairs() +TEXT ·pairs(SB), NOSPLIT, $0-0 + LDP (R1), (R2, R3) + LDP 8(R4), (R5, R6) + LDP -16(R1), (R2, R3) + LDPW 4(R4), (R5, R6) + STP (R2, R3), 24(R7) + STP (R2, R3),-8(R7) + STPW (R1, R2), 4(R0) + FLDPD (R8), (F1, F2) + FLDPD 8(R8), (F3, F4) + FSTPD (F3, F4),-8(R9) + RET diff --git a/testdata/verify/crypto_arm64.s b/testdata/verify/crypto_arm64.s new file mode 100644 index 0000000..9cb8bd2 --- /dev/null +++ b/testdata/verify/crypto_arm64.s @@ -0,0 +1,42 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 cryptographic extension: the AES round +// instructions and the SHA1, SHA256 and SHA512 families. Every function is +// byte-compared against go tool asm. + +#include "textflag.h" + +// func aesRound() +TEXT ·aesRound(SB), NOSPLIT, $0-0 + AESE V31.B16, V29.B16 + AESD V22.B16, V19.B16 + AESMC V14.B16, V28.B16 + AESIMC V12.B16, V27.B16 + RET + +// func sha1Round() +TEXT ·sha1Round(SB), NOSPLIT, $0-0 + SHA1C V8.S4, V8, V2 + SHA1P V3.S4, V20, V27 + SHA1M V0.S4, V27, V27 + SHA1H V17, V25 + SHA1SU0 V17.S4, V13.S4, V16.S4 + SHA1SU1 V24.S4, V23.S4 + RET + +// func sha256Round() +TEXT ·sha256Round(SB), NOSPLIT, $0-0 + SHA256H V4.S4, V2, V11 + SHA256H2 V6.S4, V16, V11 + SHA256SU0 V0.S4, V16.S4 + SHA256SU1 V31.S4, V3.S4, V15.S4 + RET + +// func sha512Round() +TEXT ·sha512Round(SB), NOSPLIT, $0-0 + SHA512H V2.D2, V1, V0 + SHA512H2 V4.D2, V3, V2 + SHA512SU0 V9.D2, V8.D2 + SHA512SU1 V7.D2, V6.D2, V5.D2 + RET diff --git a/testdata/verify/integer_arm64.s b/testdata/verify/integer_arm64.s new file mode 100644 index 0000000..8f2139b --- /dev/null +++ b/testdata/verify/integer_arm64.s @@ -0,0 +1,66 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 integer slice: carry-setting arithmetic, +// widening multiplies, bit manipulation, conditional compares, the compare +// and test branches, ADR and the wide-constant moves. Every function is +// byte-compared against go tool asm. + +#include "textflag.h" + +// func carryArith() +TEXT ·carryArith(SB), NOSPLIT, $0-0 + ADC R0, R2, R12 + ADCS R23, R22, R22 + ADC $0, R1 + SBC R25, R10, R26 + SBCS R5, R9, R5 + SBCS $0, R1 + RET + +// func wideningMul() +TEXT ·wideningMul(SB), NOSPLIT, $0-0 + MUL R4, R3, R0 + MSUB R19, R16, R26, R2 + SMULH R24, R20, R24 + UMULH R24, R20, R24 + RET + +// func bitManip() +TEXT ·bitManip(SB), NOSPLIT, $0-0 + RBIT R11, R4 + REV R1, R2 + CLZ R21, R9 + REVW R1, R2 + CLSW R1, R2 + UBFX $33, R17, $25, R5 + UBFXW $4, R1, $9, R2 + RET + +// func condCompare() +TEXT ·condCompare(SB), NOSPLIT, $0-0 + CCMP LE, R7, $19, $3 + CCMP LT, R30, R6, $7 + CCMN EQ, R1, R2, $3 + CCMPW LE, R7, $19, $3 + RET + +// func branchForms() +TEXT ·branchForms(SB), NOSPLIT, $0-0 + CBZ R1, target + CBNZ R7, target + CBNZW R2, target + TBZ $4, R7, target + TBNZ $33, R7, target + ADR target, R10 + +target: + RET + +// func wideMoves() +TEXT ·wideMoves(SB), NOSPLIT, $0-0 + MOVK $1234, R5 + MOVK $305397760, R5 + MOVKW $1234, R5 + MOVK $16771847290880, R21 + RET diff --git a/testdata/verify/simd_arm64.s b/testdata/verify/simd_arm64.s new file mode 100644 index 0000000..73fed0c --- /dev/null +++ b/testdata/verify/simd_arm64.s @@ -0,0 +1,98 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 NEON slice: the logical and arithmetic +// three-register operations, permutations, comparisons, shifts, the crypto +// four-register group, element moves, table lookups and the structure +// loads and stores. Every function is byte-compared against go tool asm. + +#include "textflag.h" + +// func simdLogic() +TEXT ·simdLogic(SB), NOSPLIT, $0-0 + VADD V1.B16, V2.B16, V3.B16 + VADD V1.B8, V2.B8, V3.B8 + VSUB V1.S4, V2.S4, V3.S4 + VMUL V1.H8, V2.H8, V3.H8 + VAND V4.B16, V4.B16, V9.B16 + VORR V5.B16, V4.B16, V3.B16 + VEOR V0.B16, V1.B16, V0.B16 + VADDP V1.H8, V2.H8, V3.H8 + VCMEQ V24.S4, V13.S4, V12.S4 + VCMEQ $0, V2.H4, V3.H4 + RET + +// func simdPerm() +TEXT ·simdPerm(SB), NOSPLIT, $0-0 + VZIP1 V16.H8, V3.H8, V19.H8 + VZIP1 V6.D2, V9.D2, V11.D2 + VZIP2 V22.D2, V25.D2, V21.D2 + VREV32 V2.H8, V1.H8 + VREV64 V2.S4, V3.S4 + VUADDLV V31.S4, V11 + VEXT $4, V2.B8, V1.B8, V3.B8 + VEXT $8, V2.B16, V1.B16, V3.B16 + RET + +// func simdShift() +TEXT ·simdShift(SB), NOSPLIT, $0-0 + VSHL $7, V22.D2, V25.D2 + VSHL $24, V1.S4, V2.S4 + VUSHR $6, V22.H8, V23.H8 + VUSHR $56, V1.D2, V2.D2 + VSRI $24, V1.S4, V2.S4 + VSRI $56, V1.D2, V2.D2 + RET + +// func simdCrypto4() +TEXT ·simdCrypto4(SB), NOSPLIT, $0-0 + VEOR3 V2.B16, V7.B16, V12.B16, V25.B16 + VBCAX V1.B16, V2.B16, V26.B16, V31.B16 + VXAR $63, V27.D2, V21.D2, V26.D2 + VRAX1 V26.D2, V29.D2, V30.D2 + VPMULL V2.D1, V1.D1, V3.Q1 + VPMULL V2.B8, V1.B8, V3.H8 + VPMULL2 V2.D2, V1.D2, V4.Q1 + VPMULL2 V2.B16, V1.B16, V4.H8 + RET + +// func simdElement() +TEXT ·simdElement(SB), NOSPLIT, $0-0 + VDUP V31.B[15], V18 + VDUP V19.S[3], V18.S4 + VDUP V1.D[1], V2.D2 + VMOV V13.S[0], R20 + VMOV V11.B[11], V16.B[12] + VMOV R20, V21.B[2] + VMOV V2.B16, V4.B16 + RET + +// func simdTable() +TEXT ·simdTable(SB), NOSPLIT, $0-0 + VTBL V22.B16, [V28.B16], V11.B16 + VTBL V18.B8, [V17.B16, V18.B16], V22.B8 + VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8 + RET + +// func simdLoadStore() +TEXT ·simdLoadStore(SB), NOSPLIT, $0-0 + VLD1 (R2), [V21.B16] + VLD1 (R24), [V18.D1, V19.D1, V20.D1] + VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1] + VLD1.P 32(R1), [V2.B16, V3.B16] + VLD1.P 64(R4), [V5.B16, V6.B16, V7.B16, V8.B16] + VLD1R (R1), [V9.B8] + VLD1R (R0), [V0.B16] + VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8] + VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14) + VST1 [V14.H4, V15.H4, V16.H4], (R27) + VST1.P [V2.B16], (R1) + VST1.P [V2.B16, V3.B16], 32(R1) + RET + +// func simdLiteral() +TEXT ·simdLiteral(SB), NOSPLIT, $0-0 + VMOVS $0x80402010, V11 + VMOVD $0x8040201008040201, V20 + VMOVQ $0x7040201008040201, $0x8040201008040201, V10 + RET diff --git a/testdata/verify/system_arm64.s b/testdata/verify/system_arm64.s new file mode 100644 index 0000000..accafc8 --- /dev/null +++ b/testdata/verify/system_arm64.s @@ -0,0 +1,61 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 system instructions: barriers, +// cache maintenance, the system register accesses, supervisor calls, +// breakpoints and prefetches. Every function is byte-compared against +// go tool asm. + +#include "textflag.h" + +// func barriers() +TEXT ·barriers(SB), NOSPLIT, $0-0 + DMB $15 + DMB $1 + DSB $15 + DSB $4 + ISB $15 + ISB $1 + RET + +// func cacheOps() +TEXT ·cacheOps(SB), NOSPLIT, $0-0 + DC ZVA, R4 + DC IVAC, R1 + DC CVAC, R2 + DC CVAU, R3 + DC CIVAC, R7 + RET + +// func sysRegs() +TEXT ·sysRegs(SB), NOSPLIT, $0-0 + MRS DCZID_EL0, R3 + MRS CNTVCT_EL0, R0 + MRS CNTPCT_EL0, R1 + MRS CNTFRQ_EL0, R2 + MRS MIDR_EL1, R0 + MRS ID_AA64PFR0_EL1, R0 + MRS ID_AA64ISAR0_EL1, R0 + MRS ID_AA64ISAR1_EL1, R0 + MRS DIT, R0 + MSR $3, SPSel + MSR $9, DAIFSet + MSR $6, DAIFClr + MSR $1, DIT + RET + +// func exceptions() +TEXT ·exceptions(SB), NOSPLIT, $0-0 + SVC $0 + SVC $7165 + BRK + BRK $35943 + RET + +// func prefetch() +TEXT ·prefetch(SB), NOSPLIT, $0-0 + PRFM (R0), PLDL1KEEP + PRFM (R3), PLDL3KEEP + PRFM (R4), PSTL1KEEP + PRFM (R2), $25 + RET