From 6e73f59e78d7bb8dd72cc92e3042a9640cbb9128 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Thu, 20 Aug 2026 14:07:12 +0200 Subject: [PATCH] feat(asm): extend arm64 encoder with FP, conditional select, CRC32 and tests Assisted-by: MiMo V2.5 Pro --- asm/aarch64_goobj_test.go | 59 ++++++ asm/arm64_assemble.go | 311 ++++++++++++++++++++++++++++++- asm/arm64_encode.go | 145 ++++++++++++++ asm/elfarm64_test.go | 142 ++++++++++++++ testdata/verify/fp_arm64.s | 119 ++++++++++++ verify/arm64_groundtruth_test.go | 1 + 6 files changed, 768 insertions(+), 9 deletions(-) create mode 100644 asm/aarch64_goobj_test.go create mode 100644 asm/elfarm64_test.go create mode 100644 testdata/verify/fp_arm64.s diff --git a/asm/aarch64_goobj_test.go b/asm/aarch64_goobj_test.go new file mode 100644 index 0000000..6a50432 --- /dev/null +++ b/asm/aarch64_goobj_test.go @@ -0,0 +1,59 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "strings" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// TestGOObjectAARCH64Structure checks the basic structure of the emitted +// AArch64 GOOBJ: the preamble, the magic, the block offsets and the +// non-package symbol definitions. +func TestGOObjectAARCH64Structure(t *testing.T) { + f, errs := parser.Parse("k_arm64.s", ` +#include "textflag.h" + +TEXT ·add(SB), NOSPLIT, $0-24 + MOVD a+0(FP), R4 + MOVD b+8(FP), R5 + ADD R5, R4, R4 + MOVD R4, ret+16(FP) + RET +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + obj, err := img.GOObjectAARCH64("testpkg", "k_arm64.s") + if err != nil { + t.Fatalf("GOObjectAARCH64: %v", err) + } + + // Check preamble. + idx := strings.Index(string(obj), "\n!\n") + if idx < 0 { + t.Fatal("missing preamble separator") + } + preamble := string(obj[:idx]) + if !strings.HasPrefix(preamble, "go object") { + t.Errorf("preamble = %q, want 'go object ...'", preamble) + } + + // Check GOOBJ magic. + magicIdx := idx + 3 + if magicIdx+8 > len(obj) || string(obj[magicIdx:magicIdx+8]) != "\x00go120ld" { + t.Error("missing GOOBJ magic") + } + + // The object should contain the function's code. + if len(img.Code) == 0 { + t.Error("no code generated") + } +} diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 33e7d97..74d14cf 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -218,6 +218,51 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64DPSR(mnem, enc.op, ops) } + // FP 3-operand (Rm, Rn, Rd). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP3 { + return encodeARM64FP3(mnem, enc.op, ops) + } + + // FP unary (Rn, Rd). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPUnary { + return encodeARM64FPUnary(mnem, enc.op, ops) + } + + // FP 4-operand FMA (Ra, Rm, Rn, Rd). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP4 { + return encodeARM64FP4(mnem, enc.op, ops) + } + + // FP compare (Rm, Rn or #0, Rn). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCmp { + return encodeARM64FPCmp(mnem, enc.op, ops) + } + + // FP conditional compare (Rm, Rn, #nzcv, cond). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCCmp { + return encodeARM64FPCCmp(mnem, enc.op, ops) + } + + // FP conditional select (Rm, Rn, Rd, cond). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPSel { + return encodeARM64FPSel(mnem, enc.op, ops) + } + + // FP ↔ integer conversion. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCvt { + return encodeARM64FPCvt(mnem, enc.op, ops) + } + + // Conditional select (CSEL, CSINC, CSINV, CSNEG, CSET, CSETM, CINC, CINV, CNEG). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCSEL { + return encodeARM64CSEL(mnem, enc.op, ops) + } + + // CRC32. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCRC32 { + return encodeARM64CRC32(mnem, enc.op, ops) + } + return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) } @@ -621,7 +666,11 @@ func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) { return } -// encodeARM64RegMove encodes a register-to-register move as ORR Rd, ZR, Rs. +// encodeARM64RegMove encodes a register-to-register move. +// Integer → integer: ORR Rd, ZR, Rs. +// FP → FP: FMOV Fd, Fn (FP data processing). +// FP ↔ GP: FMOV general (FPCVTI encoding). +// Go Plan 9 syntax: MOV dst, src (first operand = destination). func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) { rs := arm64RegNum(operandRegName(src)) rd := arm64RegNum(operandRegName(dst)) @@ -631,21 +680,37 @@ func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) { sc := arm64RegClassOf(operandRegName(src)) dc := arm64RegClassOf(operandRegName(dst)) - // FP → FP: FMOV Rd, Rs + // FP → FP: FMOV Fd, Fn (FP data processing unary form). if sc == arm64ClsFP && dc == arm64ClsFP { - sf := uint32(1) // 64-bit - if mnem == "FMOVS" { - sf = 0 - } - // FMOV: 0x1E<<24 | type<<22 | 1<<21 | 0x10<<10 | Rm<<5 | Rd typ := uint32(1) // 64-bit double if mnem == "FMOVS" { typ = 0 // 32-bit float } - return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 0x10<<10 | uint32(rs)<<5 | uint32(rd)), nil + // FPOP1S encoding: 0x1E204000 | type<<22 | Rn<<5 | Rd + return a64wordLE(0x1E<<24 | typ<<22 | 1<<21 | 0x10<<10 | uint32(rs)<<5 | uint32(rd)), nil } - // Integer → integer: ORR Rd, ZR, Rs + // GP ↔ FP: FMOV general (FPCVTI encoding). + // Go syntax: FMOV FPdst, GPsrc or FMOV GPdst, FPsrc. + // First operand = destination, second = source. + if sc == arm64ClsFP && dc == arm64ClsGR { + // FP → GP: FMOV Wd/Xd, Sn/Dn. opcode bits[20:16]=6. + sf, typ := uint32(0), uint32(0) + if mnem == "FMOVD" { + sf, typ = 1, 1 + } + return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 6<<16 | uint32(rs)<<5 | uint32(rd)), nil + } + if sc == arm64ClsGR && dc == arm64ClsFP { + // GP → FP: FMOV Vd, Wn/Xn. opcode bits[20:16]=7. + sf, typ := uint32(0), uint32(0) + if mnem == "FMOVD" { + sf, typ = 1, 1 + } + return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil + } + + // Integer → integer: ORR Rd, ZR, Rs. sf := uint32(1) // 64-bit if mnem == "MOVW" || mnem == "MOVWU" || mnem == "MOVB" || mnem == "MOVBU" || mnem == "MOVH" || mnem == "MOVHU" { @@ -775,6 +840,234 @@ func arm64Label(op *ast.Operand) string { return op.Raw } +// ---- FP instruction encoding ---- + +// encodeARM64FP3 encodes a FP 3-operand instruction (Rm, Rn, Rd). +// FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL. +func encodeARM64FP3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 3 { + return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + rm := arm64RegNum(operandRegName(ops[0])) + rn := arm64RegNum(operandRegName(ops[1])) + rd := arm64RegNum(operandRegName(ops[2])) + if rm < 0 || rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil +} + +// encodeARM64FPUnary encodes a FP unary instruction (Rn, Rd). +// FMOV reg-reg, FABS, FNEG, FSQRT, FCVT cross-precision, FRINT*. +func encodeARM64FPUnary(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rn := arm64RegNum(operandRegName(ops[0])) + rd := arm64RegNum(operandRegName(ops[1])) + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rd)), nil +} + +// encodeARM64FP4 encodes a FP 4-operand FMA instruction (Ra, Rm, Rn, Rd). +// FMADD, FMSUB, FNMADD, FNMSUB. +func encodeARM64FP4(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + var ra, rm, rn, rd int + switch len(ops) { + case 4: + ra = arm64RegNum(operandRegName(ops[0])) + rm = arm64RegNum(operandRegName(ops[1])) + rn = arm64RegNum(operandRegName(ops[2])) + rd = arm64RegNum(operandRegName(ops[3])) + case 3: + // 3-operand form: Fa, Fm, Fd → Fd = Fa ± Fd*Fm (Rn = Rd) + ra = arm64RegNum(operandRegName(ops[0])) + rm = arm64RegNum(operandRegName(ops[1])) + rd = arm64RegNum(operandRegName(ops[2])) + rn = rd + default: + return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) + } + if ra < 0 || rm < 0 || rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(ra)<<16 | uint32(rm)<<10 | uint32(rn)<<5 | uint32(rd)), nil +} + +// encodeARM64FPCmp encodes a FP compare instruction. +// Go assembler syntax: FCMP Fn, Fm (register) or FCMP $0.0, Fn (compare with zero). +// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5]. +// Go puts first operand → Rm, second → Rn. +func encodeARM64FPCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + // Check if first operand is #0 (compare with zero): FCMP $0.0, Fn. + if isImmOperand(ops[0]) && immFromOperand(ops[0]) == 0 { + rn := arm64RegNum(operandRegName(ops[1])) + if rn < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + // For compare with zero: Rm=0, op2 bit 3 set (|= 8). + return a64wordLE((baseOp | 8) | 0<<16 | uint32(rn)<<5), nil + } + // Register compare: FCMP Fn, Fm. + // Go puts first operand in Rm field, second in Rn field. + rm := arm64RegNum(operandRegName(ops[0])) + rn := arm64RegNum(operandRegName(ops[1])) + if rm < 0 || rn < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5), nil +} + +// encodeARM64FPCCmp encodes a FP conditional compare. +// Go assembler syntax: FCCMP cond, Fn, Fm, $nzcv +// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5]. +// Go puts ops[1] in Rm field, ops[2] in Rn field. +func encodeARM64FPCCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 4 { + return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) + } + condName := operandRegName(ops[0]) + cond, ok := arm64CondMap[condName] + if !ok { + return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) + } + // Go puts ops[1] in Rm (bits 20:16), ops[2] in Rn (bits 9:5). + rm := arm64RegNum(operandRegName(ops[1])) + rn := arm64RegNum(operandRegName(ops[2])) + if rm < 0 || rn < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + nzcv := uint32(immFromOperand(ops[3])) + return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | nzcv&0xF), nil +} + +// encodeARM64FPSel encodes a FP conditional select. +// Go assembler syntax: FCSEL cond, Fn, Fm, Fd +func encodeARM64FPSel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 4 { + return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) + } + // Operand order: cond, Fn, Fm, Fd + condName := operandRegName(ops[0]) + cond, ok := arm64CondMap[condName] + if !ok { + return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) + } + rn := arm64RegNum(operandRegName(ops[1])) + rm := arm64RegNum(operandRegName(ops[2])) + rd := arm64RegNum(operandRegName(ops[3])) + if rn < 0 || rm < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil +} + +// encodeARM64FPCvt encodes a FP ↔ integer conversion instruction. +// The operand order depends on direction: FCVTZS Fd, Rn (FP→int) or SCVTF Rd, Fn (int→FP). +func encodeARM64FPCvt(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + src := arm64RegNum(operandRegName(ops[0])) + dst := arm64RegNum(operandRegName(ops[1])) + if src < 0 || dst < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(src)<<5 | uint32(dst)), nil +} + +// encodeARM64CSEL encodes a conditional select instruction. +// CSEL Rm, Rn, Rd, cond (4 operands) or CSET Rd, cond (2 operands). +func encodeARM64CSEL(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + isAlias := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW" || + mnem == "CINC" || mnem == "CINCW" || mnem == "CINV" || mnem == "CINVW" || + mnem == "CNEG" || mnem == "CNEGW" + + if isAlias { + is2op := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW" + if is2op { + // CSET cond, Rd → CSEL XZR, XZR, Rd, inverted_cond + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + condName := operandRegName(ops[0]) + cond, ok := arm64CondMap[condName] + if !ok { + return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) + } + rd := arm64RegNum(operandRegName(ops[1])) + if rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + invCond := cond ^ 1 + return a64wordLE(baseOp | 31<<16 | invCond<<12 | 31<<5 | uint32(rd)), nil + } + // CINC cond, Rn, Rd → CSINC Rn, Rn, Rd, inverted_cond + if len(ops) != 3 { + return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + condName := operandRegName(ops[0]) + cond, ok := arm64CondMap[condName] + if !ok { + return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) + } + rn := arm64RegNum(operandRegName(ops[1])) + rd := arm64RegNum(operandRegName(ops[2])) + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + invCond := cond ^ 1 + return a64wordLE(baseOp | uint32(rn)<<16 | invCond<<12 | uint32(rn)<<5 | uint32(rd)), nil + } + + // CSEL cond, Rn, Rm, Rd (4 operands) — condition first. + // Go assembler syntax: CSEL cond, Rn, Rm, Rd + // ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5], Rd in bits[4:0]. + if len(ops) != 4 { + return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) + } + condName := operandRegName(ops[0]) + cond, ok := arm64CondMap[condName] + if !ok { + return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem) + } + rn := arm64RegNum(operandRegName(ops[1])) + rm := arm64RegNum(operandRegName(ops[2])) + rd := arm64RegNum(operandRegName(ops[3])) + if rn < 0 || rm < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil +} + +// encodeARM64CRC32 encodes a CRC32 instruction. +// Go assembler syntax: CRC32B Rm, Rd (2 operands, Rn=Rd). +func encodeARM64CRC32(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) == 3 { + // 3-operand form: CRC32B Rm, Rn, Rd → use Rm and Rd, Rn=Rd. + rm := arm64RegNum(operandRegName(ops[0])) + rd := arm64RegNum(operandRegName(ops[2])) + if rm < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil + } + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + rm := arm64RegNum(operandRegName(ops[0])) + rd := arm64RegNum(operandRegName(ops[1])) + if rm < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil +} + // AssembleFileARM64 assembles every TEXT function of a parsed arm64 file // and lays out its static symbols (GLOBL/DATA) in a data section behind the // code. SB references in the code are encoded as ADRP pairs with zero diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index 852658a..86a7cb5 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -339,6 +339,16 @@ const ( a64FEXTR // EXTR a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM a64FSystem // system: NOP, BRK, etc. + a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc. + a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT* + a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc. + a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE + a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE + a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc. + a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL + a64FFMovGR // FMOV between GP and FP registers + a64FCRC32 // CRC32 + a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG ) // a64Enc is one instruction's encoding: its bit layout (format) and the @@ -517,6 +527,141 @@ func init() { a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22} a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} + + // ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ---- + fp3 := map[string]uint32{ + "FADDS": 0x1e202800, "FADDD": 0x1e602800, + "FSUBS": 0x1e203800, "FSUBD": 0x1e603800, + "FMULS": 0x1e200800, "FMULD": 0x1e600800, + "FDIVS": 0x1e201800, "FDIVD": 0x1e601800, + "FMAXS": 0x1e204800, "FMAXD": 0x1e604800, + "FMINS": 0x1e205800, "FMIND": 0x1e605800, + "FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800, + "FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800, + "FNMULS": 0x1e208800, "FNMULD": 0x1e608800, + } + for m, op := range fp3 { + a64InstrTable[m] = a64Enc{format: a64FFP3, op: op} + } + + // ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ---- + fp1 := map[string]uint32{ + "FMOVS": 0x1e204000, "FMOVD": 0x1e604000, + "FABSS": 0x1e20c000, "FABSD": 0x1e60c000, + "FNEGS": 0x1e214000, "FNEGD": 0x1e614000, + "FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000, + "FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000, + "FRINTNS": 0x1e244000, "FRINTND": 0x1e644000, + "FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000, + "FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000, + "FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000, + "FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000, + "FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000, + "FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000, + } + for m, op := range fp1 { + a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op} + } + + // ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ---- + fp4 := map[string]uint32{ + "FMADDS": 0x1f000000, "FMADDD": 0x1f400000, + "FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000, + "FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000, + "FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000, + } + for m, op := range fp4 { + a64InstrTable[m] = a64Enc{format: a64FFP4, op: op} + } + + // ---- FP compare (Rm, Rn or #0, Rn) ---- + fpcmp := map[string]uint32{ + "FCMPS": 0x1e202000, "FCMPD": 0x1e602000, + "FCMPES": 0x1e202010, "FCMPED": 0x1e602010, + } + for m, op := range fpcmp { + a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op} + } + + // ---- FP conditional compare (Rm, Rn, #nzcv, cond) ---- + fpccmp := map[string]uint32{ + "FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400, + "FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410, + } + for m, op := range fpccmp { + a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op} + } + + // ---- FP conditional select (Rm, Rn, Rd, cond) ---- + a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00} + a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00} + + // ---- FP ↔ integer conversion ---- + fpcvt := map[string]uint32{ + "FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000, + "FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000, + "FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000, + "FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000, + "SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000, + "SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000, + "UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000, + "UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000, + } + for m, op := range fpcvt { + a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op} + } + + // ---- FMOV between GP and FP registers ---- + a64InstrTable["FMOVGR"] = a64Enc{format: a64FFMovGR, op: 0x1e260000} // placeholder, actual encoding depends on direction + + // ---- conditional select: CSEL, CSINC, CSINV, CSNEG ---- + csel := map[string]uint32{ + "CSEL": 0x9a800000, "CSELW": 0x1a800000, + "CSINC": 0x9a800400, "CSINCW": 0x1a800400, + "CSINV": 0xda800000, "CSINVW": 0x5a800000, + "CSNEG": 0xda800400, "CSNEGW": 0x5a800400, + } + for m, op := range csel { + a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op} + } + // Aliases + a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400} + a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400} + a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000} + a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000} + a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400} + a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400} + a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000} + a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000} + a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400} + a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400} + + // ---- CRC32 ---- + crc32 := map[string]uint32{ + "CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400, + "CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00, + "CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400, + "CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00, + } + for m, op := range crc32 { + a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op} + } + + // ---- exclusive load/store ---- + // LDXR/STXR and variants + a64InstrTable["LDXR"] = a64Enc{format: a64FLSU, op: 0xc85f7c00} + a64InstrTable["LDXRB"] = a64Enc{format: a64FLSU, op: 0x085f7c00} + a64InstrTable["LDXRH"] = a64Enc{format: a64FLSU, op: 0x485f7c00} + a64InstrTable["LDXRW"] = a64Enc{format: a64FLSU, op: 0x885f7c00} + a64InstrTable["LDAXR"] = a64Enc{format: a64FLSU, op: 0xc85ffc00} + a64InstrTable["LDAXRB"] = a64Enc{format: a64FLSU, op: 0x085ffc00} + a64InstrTable["LDAXRH"] = a64Enc{format: a64FLSU, op: 0x485ffc00} + a64InstrTable["LDAXRW"] = a64Enc{format: a64FLSU, op: 0x885ffc00} + + // ---- SIMD basics (VADD, VSUB, VMUL, VMOV) ---- + a64InstrTable["VADD"] = a64Enc{format: a64FFP3, op: 0x0e208400} + a64InstrTable["VSUB"] = a64Enc{format: a64FFP3, op: 0x2e208400} + a64InstrTable["VMUL"] = a64Enc{format: a64FFP3, op: 0x0e209c00} } // ---- load/store helper tables ---- diff --git a/asm/elfarm64_test.go b/asm/elfarm64_test.go new file mode 100644 index 0000000..74697f4 --- /dev/null +++ b/asm/elfarm64_test.go @@ -0,0 +1,142 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "debug/elf" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64 +// relocatable object: sections, the symbol table (bindings, types, values, +// sizes) and the .rela.text relocation pair for the static-symbol load, +// parsed back with debug/elf. +func TestELFAARCH64Object(t *testing.T) { + f, errs := parser.Parse("k_arm64.s", ` +#include "textflag.h" + +TEXT ·add(SB), NOSPLIT, $0-24 + MOVD a+0(FP), R4 + MOVD b+8(FP), R5 + ADD R5, R4, R4 + MOVD R4, ret+16(FP) + RET + +TEXT ·getanswer(SB), NOSPLIT, $0-8 + MOVD answer<>(SB), R4 + MOVD R4, ret+0(FP) + RET + +GLOBL answer<>(SB), RODATA, $8 +DATA answer<>+0(SB)/8, $42 +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + obj, err := img.ELFAARCH64Object() + if err != nil { + t.Fatalf("ELFAARCH64Object: %v", err) + } + ef, err := elf.NewFile(bytes.NewReader(obj)) + if err != nil { + t.Fatalf("parse emitted object: %v", err) + } + defer ef.Close() + + if ef.Type != elf.ET_REL || ef.Machine != elf.EM_AARCH64 { + t.Errorf("type/machine = %v/%v, want ET_REL/EM_AARCH64", ef.Type, ef.Machine) + } + + text := ef.Section(".text") + data := ef.Section(".data") + if text == nil || data == nil { + t.Fatal("missing .text or .data section") + } + if text.Size == 0 { + t.Error(".text section is empty") + } + + syms, err := ef.Symbols() + if err != nil { + t.Fatalf("symbols: %v", err) + } + + foundAdd, foundGetanswer, foundAnswer := false, false, false + for _, s := range syms { + switch s.Name { + case "add": + foundAdd = true + if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL { + t.Errorf("add: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info) + } + case "getanswer": + foundGetanswer = true + if elf.SymType(s.Info&0xf) != elf.STT_FUNC || elf.SymBind(s.Info>>4) != elf.STB_GLOBAL { + t.Errorf("getanswer: info=0x%02x, want STT_FUNC|STB_GLOBAL", s.Info) + } + case "answer": + foundAnswer = true + if elf.SymType(s.Info&0xf) != elf.STT_OBJECT || elf.SymBind(s.Info>>4) != elf.STB_LOCAL { + t.Errorf("answer: info=0x%02x, want STT_OBJECT|STB_LOCAL", s.Info) + } + } + } + if !foundAdd { + t.Error("symbol 'add' not found") + } + if !foundGetanswer { + t.Error("symbol 'getanswer' not found") + } + if !foundAnswer { + t.Error("symbol 'answer' not found") + } + + // Check that .rela.text exists (getanswer has SB reference). + relaText := ef.Section(".rela.text") + if relaText == nil { + t.Error("missing .rela.text section") + } +} + +// TestELFAARCH64ObjectNoRelocations checks the ELF output when there are no +// static-symbol references (no .rela.text section). +func TestELFAARCH64ObjectNoRelocations(t *testing.T) { + f, errs := parser.Parse("k_arm64.s", ` +#include "textflag.h" + +TEXT ·add(SB), NOSPLIT, $0-24 + MOVD a+0(FP), R4 + MOVD b+8(FP), R5 + ADD R5, R4, R4 + MOVD R4, ret+16(FP) + RET +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + obj, err := img.ELFAARCH64Object() + if err != nil { + t.Fatalf("ELFAARCH64Object: %v", err) + } + ef, err := elf.NewFile(bytes.NewReader(obj)) + if err != nil { + t.Fatalf("parse emitted object: %v", err) + } + defer ef.Close() + + if ef.Section(".rela.text") != nil { + t.Error("unexpected .rela.text section when there are no relocations") + } +} diff --git a/testdata/verify/fp_arm64.s b/testdata/verify/fp_arm64.s new file mode 100644 index 0000000..fd2be60 --- /dev/null +++ b/testdata/verify/fp_arm64.s @@ -0,0 +1,119 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +// fparith exercises the FP arithmetic set. +TEXT ·fparith(SB), NOSPLIT, $0-0 + FADDD F0, F1, F2 + FSUBD F3, F4, F5 + FMULD F6, F7, F8 + FDIVD F9, F10, F11 + FADDS F12, F13, F14 + FSUBS F15, F16, F17 + FMULS F18, F19, F20 + FDIVS F21, F22, F23 + FSQRTD F24, F25 + FSQRTS F26, F27 + FNEGD F28, F29 + FNEGS F30, F31 + FABSD F0, F1 + FABSS F2, F3 + FNMULD F4, F5, F6 + FNMULS F7, F8, F9 + FMIND F10, F11, F12 + FMAXD F13, F14, F15 + FMINS F16, F17, F18 + FMAXS F19, F20, F21 + RET + +// fpfma exercises fused multiply-add. +TEXT ·fpfma(SB), NOSPLIT, $0-0 + FMADDD F0, F1, F2, F3 + FMSUBD F4, F5, F6, F7 + FNMADDD F8, F9, F10, F11 + FNMSUBD F12, F13, F14, F15 + FMADDS F16, F17, F18, F19 + FMSUBS F20, F21, F22, F23 + FNMADDS F24, F25, F26, F27 + FNMSUBS F28, F29, F30, F0 + RET + +// fpconv exercises FP↔integer conversion and cross-precision. +// Syntax: FCVTZSD Fd, Rn (float→int: FP source first, int dest second) +// SCVTFD Rn, Fd (int→float: int source first, FP dest second) +TEXT ·fpconv(SB), NOSPLIT, $0-0 + FCVTSD F0, F1 + FCVTDS F2, F3 + FCVTZSD F4, R0 + FCVTZSS F5, R1 + FCVTZUD F6, R2 + FCVTZUS F7, R3 + SCVTFD R4, F8 + SCVTFS R5, F9 + UCVTFD R6, F10 + UCVTFS R7, F11 + SCVTFWD R0, F12 + SCVTFWS R1, F13 + UCVTFWD R2, F14 + UCVTFWS R3, F15 + FMOVS F14, R20 + FMOVS R21, F15 + FMOVD F16, R22 + FMOVD R23, F17 + RET + +// fpcmp exercises FP compare and conditional compare. +// FCCMP syntax: FCCMP cond, Fn, Fm, $nzcv +// FCSEL syntax: FCSEL cond, Fn, Fm, Fd +TEXT ·fpcmp(SB), NOSPLIT, $0-0 + FCMPS F0, F1 + FCMPD F2, F3 + FCMPS $0.0, F4 + FCMPD $0.0, F5 + FCCMPS EQ, F6, F7, $0 + FCCMPD NE, F8, F9, $0 + FCSELS GE, F10, F11, F12 + FCSELD LT, F13, F14, F15 + RET + +// frint exercises FP rounding. +TEXT ·frint(SB), NOSPLIT, $0-0 + FRINTND F0, F1 + FRINTNS F2, F3 + FRINTPD F4, F5 + FRINTPS F6, F7 + FRINTMD F8, F9 + FRINTMS F10, F11 + FRINTZD F12, F13 + FRINTZS F14, F15 + FRINTAD F16, F17 + FRINTAS F18, F19 + FRINTXD F20, F21 + FRINTXS F22, F23 + FRINTID F24, F25 + FRINTIS F26, F27 + FMOVD F0, F1 + FMOVS F2, F3 + RET + +// condsel exercises conditional select and CRC32. +TEXT ·condsel(SB), NOSPLIT, $0-0 + CSEL EQ, R0, R1, R2 + CSINC NE, R3, R4, R5 + CSINV GE, R6, R7, R8 + CSNEG LT, R9, R10, R11 + CSET EQ, R12 + CSETM NE, R13 + CINC EQ, R14, R15 + CINV NE, R16, R17 + CNEG GE, R19, R20 + CRC32B R0, R2 + CRC32H R3, R5 + CRC32W R6, R8 + CRC32X R9, R11 + CRC32CB R12, R14 + CRC32CH R15, R0 + CRC32CW R1, R3 + CRC32CX R4, R6 + RET diff --git a/verify/arm64_groundtruth_test.go b/verify/arm64_groundtruth_test.go index b8fc0c9..a4213c0 100644 --- a/verify/arm64_groundtruth_test.go +++ b/verify/arm64_groundtruth_test.go @@ -19,6 +19,7 @@ import ( func TestGroundTruthARM64(t *testing.T) { for _, path := range []string{ "../testdata/verify/basic_arm64.s", + "../testdata/verify/fp_arm64.s", } { t.Run(path, func(t *testing.T) { src, err := os.ReadFile(path)