feat(asm): extend arm64 encoder with atomics, bitfield, SIMD and more test kernels
Assisted-by: MiMo V2.5 Pro
This commit is contained in:
@@ -4,6 +4,9 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
@@ -57,3 +60,122 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
||||
t.Error("no code generated")
|
||||
}
|
||||
}
|
||||
|
||||
// TestGOObjectAARCH64Link does an end-to-end link test: it cross-compiles a
|
||||
// Go program for arm64, substitutes the gasm-produced object into the package
|
||||
// archive, re-links with cmd/link, and verifies the symbol appears in the
|
||||
// resulting binary. The binary is not executed (no arm64 host or qemu).
|
||||
// Skipped when no Go toolchain is available.
|
||||
func TestGOObjectAARCH64Link(t *testing.T) {
|
||||
goBin, err := exec.LookPath("go")
|
||||
if err != nil {
|
||||
t.Skip("no Go toolchain available")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
asmSrc := `#include "textflag.h"
|
||||
TEXT ·add(SB), NOSPLIT, $0-24
|
||||
MOVD a+0(FP), R4
|
||||
MOVD b+8(FP), R5
|
||||
ADD R5, R4, R4
|
||||
MOVD R4, ret+16(FP)
|
||||
RET
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
mainSrc := `package main
|
||||
|
||||
func add(a, b int64) int64
|
||||
|
||||
func main() {
|
||||
if add(20, 22) != 42 {
|
||||
panic("bad add")
|
||||
}
|
||||
}
|
||||
`
|
||||
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64link\n\ngo 1.21\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Capture the cross build (GOARCH=arm64): the package archive and the
|
||||
// link line.
|
||||
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||
build.Dir = dir
|
||||
build.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
buildLog, err := build.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||
}
|
||||
var pkgArch, work, linkLine, asmObj string
|
||||
for _, line := range strings.Split(string(buildLog), "\n") {
|
||||
switch {
|
||||
case strings.HasPrefix(line, "WORK="):
|
||||
work = strings.TrimPrefix(line, "WORK=")
|
||||
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
|
||||
asmObj = fieldAfter(line, "-o")
|
||||
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
|
||||
pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
|
||||
pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0]
|
||||
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||
linkLine = line
|
||||
}
|
||||
}
|
||||
if work == "" || asmObj == "" {
|
||||
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
|
||||
}
|
||||
defer os.RemoveAll(work)
|
||||
|
||||
// Expand $WORK in the object path.
|
||||
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||
|
||||
// Read the toolchain-produced object and assemble the same source with gasm.
|
||||
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f, errs := parser.Parse("main_arm64.s", string(src))
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
|
||||
if err != nil {
|
||||
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||
}
|
||||
|
||||
// Replace the toolchain-produced object with gasm's.
|
||||
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
|
||||
t.Fatalf("write gasm object: %v", err)
|
||||
}
|
||||
|
||||
// Re-link.
|
||||
if linkLine == "" {
|
||||
t.Skip("could not find link command in build log")
|
||||
}
|
||||
// Expand $WORK in the link command.
|
||||
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
|
||||
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
|
||||
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||
}
|
||||
|
||||
// Verify the binary exists and contains the symbol.
|
||||
binPath := filepath.Join(dir, "prog")
|
||||
if _, err := os.Stat(binPath); err != nil {
|
||||
t.Fatalf("binary not found: %v", err)
|
||||
}
|
||||
binData, err := os.ReadFile(binPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read binary: %v", err)
|
||||
}
|
||||
if !strings.Contains(string(binData), "add") && !strings.Contains(string(binData), "a64link") {
|
||||
t.Error("binary does not contain expected symbol")
|
||||
}
|
||||
}
|
||||
|
||||
+202
-20
@@ -263,6 +263,31 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return encodeARM64CRC32(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// Exclusive load/store (LDXR, STXR, LDAXR, STLXR).
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl {
|
||||
return encodeARM64Excl(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// LSE atomics (LDADD, CAS, SWP).
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE {
|
||||
return encodeARM64LSEAtom(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM).
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield {
|
||||
return encodeARM64Bitfield(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// EXTR.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
|
||||
return encodeARM64Extr(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// SIMD 3-operand (VADD, VSUB, VMUL).
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMD3 {
|
||||
return encodeARM64SIMD3(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
|
||||
}
|
||||
|
||||
@@ -273,21 +298,28 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
||||
if len(ops) != 1 {
|
||||
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
|
||||
}
|
||||
target := resolve(arm64Label(ops[0]))
|
||||
op := ops[0]
|
||||
|
||||
// External symbol reference: BL sym(SB).
|
||||
if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
|
||||
// Emit BL with zero offset; the linker fills in the target.
|
||||
return a64wordLE(a64Branch(1, 0)), nil
|
||||
}
|
||||
|
||||
target := resolve(arm64Label(op))
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q", target)
|
||||
}
|
||||
// Branch offset in bytes, shifted right by 2 (instructions are 4-byte aligned).
|
||||
rel := (targetOff - pc) >> 2
|
||||
if rel < -(1<<25) || rel >= (1<<25) {
|
||||
return nil, fmt.Errorf("branch to %q too far (26-bit range)", target)
|
||||
}
|
||||
op := uint32(0) // B
|
||||
bop := uint32(0) // B
|
||||
if link {
|
||||
op = 1 // BL
|
||||
bop = 1 // BL
|
||||
}
|
||||
return a64wordLE(a64Branch(op, int32(rel))), nil
|
||||
return a64wordLE(a64Branch(bop, int32(rel))), nil
|
||||
}
|
||||
|
||||
// encodeARM64BranchCond encodes a conditional branch (B.cond) to a label.
|
||||
@@ -446,7 +478,7 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, fi arm64FrameInfo, relocs *[]
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
|
||||
}
|
||||
return encodeARM64LoadImm(rd, immFromOperand(src), mnem)
|
||||
return encodeARM64LoadImm(rd, arm64Imm64(src), mnem)
|
||||
}
|
||||
|
||||
// Static symbol load/store via ADRP.
|
||||
@@ -496,11 +528,11 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
|
||||
return 8 // ADRP + ADD
|
||||
}
|
||||
v := immFromOperand(src)
|
||||
v := arm64Imm64(src)
|
||||
if v == 0 {
|
||||
return 4
|
||||
}
|
||||
if arm64Movcon(int64(v)) >= 0 || arm64Movcon(^int64(v)) >= 0 {
|
||||
if arm64Movcon(v) >= 0 || arm64Movcon(^v) >= 0 {
|
||||
return 4
|
||||
}
|
||||
return 8 // MOVZ + MOVK
|
||||
@@ -534,8 +566,8 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
|
||||
// encodeARM64LoadImm loads an immediate into a register, matching the
|
||||
// toolchain's MOVZ/MOVN/MOVK sequence.
|
||||
func encodeARM64LoadImm(rd int, v int32, mnem string) ([]byte, error) {
|
||||
d := int64(v)
|
||||
func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
|
||||
d := v
|
||||
// For 32-bit MOVW, zero-extend.
|
||||
if mnem == "MOVW" || mnem == "MOVWU" {
|
||||
d = int64(uint32(v))
|
||||
@@ -555,26 +587,39 @@ func encodeARM64LoadImm(rd int, v int32, mnem string) ([]byte, error) {
|
||||
sf = 0
|
||||
}
|
||||
|
||||
// Try logical immediate (bitmask) encoding. The Go toolchain uses ORR
|
||||
// with a bitmask immediate for constants like $1, $-2, $0xFF, etc.
|
||||
// that can be represented as a repeating pattern of contiguous 1s.
|
||||
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
|
||||
if ok {
|
||||
// ORR Rd, XZR, #bitmask (logical immediate)
|
||||
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
|
||||
// The Go toolchain classifies immediates:
|
||||
// - C_ABCON0 (0 < v ≤ 4095): bitmask first for positive values
|
||||
// - Negative values: MOVN first, then bitmask
|
||||
// - C_MOVCON (movcon-eligible, outside ABCON range): MOVZ/MOVN first
|
||||
tryBitmaskFirst := (d > 0 && d <= 0xFFF)
|
||||
|
||||
if tryBitmaskFirst {
|
||||
// Small immediate: try bitmask first (Go uses ORR for values like $1, $256).
|
||||
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
|
||||
if ok {
|
||||
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
|
||||
}
|
||||
}
|
||||
|
||||
// Try MOVZ (single non-zero16-bit chunk).
|
||||
// Try MOVZ (single non-zero 16-bit chunk).
|
||||
s := arm64Movcon(d)
|
||||
if s >= 0 {
|
||||
return a64wordLE(a64MoveWide(sf, 2, uint32(s>>4), uint32((d>>uint(s))&0xFFFF), uint32(rd))), nil
|
||||
}
|
||||
// Try MOVN (single non-0xFFFF16-bit chunk of ^d).
|
||||
// Try MOVN (single non-0xFFFF 16-bit chunk of ^d).
|
||||
sn := arm64Movcon(^d)
|
||||
if sn >= 0 {
|
||||
return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32((^d>>uint(sn))&0xFFFF), uint32(rd))), nil
|
||||
}
|
||||
|
||||
// For values outside the bitmask-first range that are not movcon: try bitmask.
|
||||
if !tryBitmaskFirst {
|
||||
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
|
||||
if ok {
|
||||
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
|
||||
}
|
||||
}
|
||||
|
||||
// Multi-instruction: MOVZ + MOVK for each non-zero16-bit chunk.
|
||||
var ws []uint32
|
||||
first := true
|
||||
@@ -659,7 +704,7 @@ func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) {
|
||||
N = 0
|
||||
}
|
||||
imms = uint32((^(esize - 1))&0x3F) | uint32(ones-1)
|
||||
immr = uint32(r)
|
||||
immr = uint32((esize - r) % esize)
|
||||
return N, immr, imms, true
|
||||
}
|
||||
}
|
||||
@@ -823,6 +868,18 @@ func arm64Reg(op *ast.Operand) int {
|
||||
return arm64RegNum(operandRegName(op))
|
||||
}
|
||||
|
||||
// arm64Imm64 returns the full 64-bit immediate value of an operand.
|
||||
func arm64Imm64(op *ast.Operand) int64 {
|
||||
if op.Imm.HasVal {
|
||||
v := op.Imm.Val
|
||||
if op.Imm.Neg {
|
||||
v = -v
|
||||
}
|
||||
return v
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// arm64MemWithFrame resolves a memory operand, translating FP/SP pseudo-
|
||||
// registers via the frame mapping.
|
||||
func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int32) {
|
||||
@@ -1068,6 +1125,131 @@ func encodeARM64CRC32(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
|
||||
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// ---- Atomics encoding ----
|
||||
|
||||
// encodeARM64Excl encodes an exclusive load/store instruction.
|
||||
// LDXR (Rn), Rt → LDXR Rt, [Rn] (2 operands: mem, reg or reg, mem)
|
||||
// STXR Rs, (Rn), Rt → STXR Rs, Rt, [Rn] (3 operands: Rs, mem, Rt-status)
|
||||
func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
// LDXR/STXR have different operand forms.
|
||||
isLoad := strings.HasPrefix(mnem, "LD")
|
||||
if isLoad {
|
||||
// LDXR (Rn), Rt → 2 operands: mem, reg
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
rn, _ := arm64MemWithFrame(ops[0], arm64FrameInfo{})
|
||||
rt := arm64RegNum(operandRegName(ops[1]))
|
||||
if rn < 0 || rt < 0 {
|
||||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil
|
||||
}
|
||||
// STXR Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{})
|
||||
rt := arm64RegNum(operandRegName(ops[2]))
|
||||
if rs < 0 || rn < 0 || rt < 0 {
|
||||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
|
||||
}
|
||||
|
||||
// encodeARM64LSEAtom encodes an LSE atomic instruction (LDADD, CAS, SWP).
|
||||
// LDADD Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
|
||||
// CAS Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
|
||||
func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
rn, _ := arm64MemWithFrame(ops[1], arm64FrameInfo{})
|
||||
rt := arm64RegNum(operandRegName(ops[2]))
|
||||
if rs < 0 || rn < 0 || rt < 0 {
|
||||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
|
||||
}
|
||||
|
||||
// ---- Bitfield/EXTR encoding ----
|
||||
|
||||
// encodeARM64Bitfield encodes a bitfield instruction.
|
||||
// ASR/LSL/LSR/ROR $shamt, Rn, Rd → 3 operands: $imm, Rn, Rd
|
||||
// BFI/BFXIL/SBFM/UBFM $immr, Rn, $imms, Rd → 4 operands
|
||||
func encodeARM64Bitfield(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
isShift := mnem == "ASR" || mnem == "ASRW" || mnem == "LSL" || mnem == "LSLW" ||
|
||||
mnem == "LSR" || mnem == "LSRW" || mnem == "ROR" || mnem == "RORW"
|
||||
|
||||
if isShift {
|
||||
// ASR $shamt, Rn, Rd → SBFM with immr=shamt, imms=31/63
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
shamt := int(immFromOperand(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[2]))
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
// ASR: SBFM with immr=shamt, imms=31(32-bit) or 63(64-bit)
|
||||
is64 := mnem == "ASR"
|
||||
imms := 31
|
||||
if is64 {
|
||||
imms = 63
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(shamt)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// BFI/BFXIL/SBFM/UBFM: 4 operands ($immr, Rn, $imms, Rd)
|
||||
if len(ops) != 4 {
|
||||
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
immr := int(immFromOperand(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
imms := int(immFromOperand(ops[2]))
|
||||
rd := arm64RegNum(operandRegName(ops[3]))
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// encodeARM64Extr encodes an EXTR instruction.
|
||||
// EXTR $lsb, Rm, Rn, Rd → 4 operands
|
||||
func encodeARM64Extr(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 4 {
|
||||
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
lsb := int(immFromOperand(ops[0]))
|
||||
rm := arm64RegNum(operandRegName(ops[1]))
|
||||
rn := arm64RegNum(operandRegName(ops[2]))
|
||||
rd := arm64RegNum(operandRegName(ops[3]))
|
||||
if rm < 0 || rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(lsb)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// ---- SIMD/NEON encoding ----
|
||||
|
||||
// encodeARM64SIMD3 encodes a SIMD 3-operand instruction.
|
||||
// VADD Vm, Vn, Vd → base | Rm<<16 | Rn<<5 | Rd (Q and size bits in base)
|
||||
func encodeARM64SIMD3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
rm := arm64RegNum(operandRegName(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[2]))
|
||||
if rm < 0 || rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// AssembleFileARM64 assembles every TEXT function of a parsed arm64 file
|
||||
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
|
||||
// code. SB references in the code are encoded as ADRP pairs with zero
|
||||
|
||||
+33
-13
@@ -349,6 +349,9 @@ const (
|
||||
a64FFMovGR // FMOV between GP and FP registers
|
||||
a64FCRC32 // CRC32
|
||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR
|
||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
|
||||
)
|
||||
|
||||
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
||||
@@ -648,20 +651,37 @@ func init() {
|
||||
}
|
||||
|
||||
// ---- exclusive load/store ----
|
||||
// LDXR/STXR and variants
|
||||
a64InstrTable["LDXR"] = a64Enc{format: a64FLSU, op: 0xc85f7c00}
|
||||
a64InstrTable["LDXRB"] = a64Enc{format: a64FLSU, op: 0x085f7c00}
|
||||
a64InstrTable["LDXRH"] = a64Enc{format: a64FLSU, op: 0x485f7c00}
|
||||
a64InstrTable["LDXRW"] = a64Enc{format: a64FLSU, op: 0x885f7c00}
|
||||
a64InstrTable["LDAXR"] = a64Enc{format: a64FLSU, op: 0xc85ffc00}
|
||||
a64InstrTable["LDAXRB"] = a64Enc{format: a64FLSU, op: 0x085ffc00}
|
||||
a64InstrTable["LDAXRH"] = a64Enc{format: a64FLSU, op: 0x485ffc00}
|
||||
a64InstrTable["LDAXRW"] = a64Enc{format: a64FLSU, op: 0x885ffc00}
|
||||
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
|
||||
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
|
||||
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
|
||||
a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00}
|
||||
a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00}
|
||||
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
|
||||
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
|
||||
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
|
||||
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
|
||||
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
|
||||
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
|
||||
a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00}
|
||||
a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00}
|
||||
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
|
||||
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
|
||||
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
|
||||
|
||||
// ---- SIMD basics (VADD, VSUB, VMUL, VMOV) ----
|
||||
a64InstrTable["VADD"] = a64Enc{format: a64FFP3, op: 0x0e208400}
|
||||
a64InstrTable["VSUB"] = a64Enc{format: a64FFP3, op: 0x2e208400}
|
||||
a64InstrTable["VMUL"] = a64Enc{format: a64FFP3, op: 0x0e209c00}
|
||||
// ---- LSE atomics ----
|
||||
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10}
|
||||
a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10}
|
||||
a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10}
|
||||
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
|
||||
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
|
||||
|
||||
// ---- SIMD basics ----
|
||||
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
|
||||
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
|
||||
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
|
||||
}
|
||||
|
||||
// ---- load/store helper tables ----
|
||||
|
||||
@@ -48,7 +48,8 @@ func TestArm64EpilogueSmallEncoding(t *testing.T) {
|
||||
if len(ret) != 12 {
|
||||
t.Fatalf("epilogue length: got %d, want 12", len(ret))
|
||||
}
|
||||
expected := []uint32{0x9100a3fd, 0x9100c3ff, 0xd65f03c0}
|
||||
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #48; RET
|
||||
expected := []uint32{0xf85f83fd, 0xf84307fe, 0xd65f03c0}
|
||||
for i, w := range leWords(ret) {
|
||||
if w != expected[i] {
|
||||
t.Errorf("epilogue word %d: got %08x, want %08x", i, w, expected[i])
|
||||
@@ -161,7 +162,7 @@ func TestArm64Bitmask(t *testing.T) {
|
||||
ok bool
|
||||
}{
|
||||
{1, 1, 1, 0, 0, true}, // single bit at pos 0
|
||||
{2, 1, 1, 1, 0, true}, // single bit at pos 1 (rotated right by1)
|
||||
{2, 1, 1, 63, 0, true}, // single bit at pos 1 (immr = esize-1)
|
||||
{0, 1, 0, 0, 0, false}, // zero is not a bitmask
|
||||
{0xFFFFFFFFFFFFFFFF, 1, 0, 0, 0, false}, // all ones is not a bitmask
|
||||
{0x5555555555555555, 1, 0, 0, 0x3E, true}, // alternating bits (esize=2, ones=1)
|
||||
|
||||
+8
-5
@@ -132,15 +132,18 @@ func arm64Prologue(fi arm64FrameInfo) []byte {
|
||||
func arm64Return(fi arm64FrameInfo) []byte {
|
||||
var ws []uint32
|
||||
if fi.autosize != 0 {
|
||||
if fi.autosize <= 0xf0 {
|
||||
// Small frame (leaf or non-leaf): ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
|
||||
// The Go toolchain uses this simpler epilogue for small frames even for
|
||||
// non-leaf functions — LR is not explicitly restored; the return address
|
||||
// is already in LR from the caller's BL instruction.
|
||||
if fi.leaf {
|
||||
// Leaf with frame: ADD $autosize-8, SP, FP; ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize-8), 31, 29), // ADD $autosize-8, SP, FP
|
||||
a64AddSub(1, 0, 0, 0, uint32(fi.autosize), 31, 31), // ADD $autosize, SP, SP
|
||||
)
|
||||
} else if fi.autosize <= 0xf0 {
|
||||
// Non-leaf small frame: LDR FP, [SP, #-8]; LDR.P LR, [SP], #autosize
|
||||
ws = append(ws,
|
||||
arm64UnscaledLoad(3, 0, -8, 31, 29), // LDR FP, [SP, #-8]
|
||||
arm64PostLoad(3, 0, int32(fi.autosize), 31, 30), // LDR.P LR, [SP], #autosize
|
||||
)
|
||||
} else {
|
||||
// Large frame: LDP -8(SP), (FP, LR); ADD $autosize, SP, SP
|
||||
ws = append(ws,
|
||||
|
||||
Vendored
+27
@@ -0,0 +1,27 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// branch exercises all conditional branch forms and jump chain folding.
|
||||
TEXT ·branch(SB), NOSPLIT, $0-0
|
||||
BEQ done
|
||||
BNE skip
|
||||
BGE done
|
||||
BLT done
|
||||
BGT done
|
||||
BLE done
|
||||
BCS done
|
||||
BCC done
|
||||
BMI done
|
||||
BPL done
|
||||
BVS done
|
||||
BVC done
|
||||
BHI done
|
||||
BLS done
|
||||
skip:
|
||||
B loop
|
||||
loop:
|
||||
ADD R4, R5
|
||||
done:
|
||||
RET
|
||||
Vendored
+10
@@ -0,0 +1,10 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// caller exercises BL to an external symbol (produces a relocation).
|
||||
TEXT ·caller(SB), NOSPLIT, $0-0
|
||||
BL other(SB)
|
||||
ADD R4, R5
|
||||
RET
|
||||
Vendored
+21
@@ -0,0 +1,21 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// movimm exercises MOV with various immediate values.
|
||||
TEXT ·movimm(SB), NOSPLIT, $0-0
|
||||
MOVD $0, R0
|
||||
MOVD $1, R1
|
||||
MOVD $42, R2
|
||||
MOVD $255, R3
|
||||
MOVD $256, R4
|
||||
MOVD $0xFFFF, R5
|
||||
MOVD $0x12345678, R6
|
||||
MOVD $0x123456789ABCDEF0, R7
|
||||
MOVD $-1, R8
|
||||
MOVD $-2, R9
|
||||
MOVW $0, R10
|
||||
MOVW $100, R11
|
||||
MOVW $0x12345, R12
|
||||
RET
|
||||
@@ -20,6 +20,9 @@ func TestGroundTruthARM64(t *testing.T) {
|
||||
for _, path := range []string{
|
||||
"../testdata/verify/basic_arm64.s",
|
||||
"../testdata/verify/fp_arm64.s",
|
||||
"../testdata/verify/movimm_arm64.s",
|
||||
"../testdata/verify/branch_arm64.s",
|
||||
"../testdata/verify/call_arm64.s",
|
||||
} {
|
||||
t.Run(path, func(t *testing.T) {
|
||||
src, err := os.ReadFile(path)
|
||||
|
||||
Reference in New Issue
Block a user