feat(arm64): assemble PCALIGN padding and BYTE literal bytes
Test / test (push) Successful in 2m16s

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 11:49:05 +02:00
parent ecb203dcf5
commit 0629f5e2df
2 changed files with 1108 additions and 57 deletions
+72
View File
@@ -29,6 +29,7 @@ package asm
import (
"maps"
"math/bits"
"strconv"
"strings"
@@ -159,6 +160,67 @@ func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
}
// ---- logical immediate ----
// a64LogicalImm encodes v as the AArch64 logical (bitmask) immediate for the
// given lane width (32 or 64): it returns the N, immr and imms fields of the
// imm13 encoding. The algorithm mirrors cmd/internal/obj/arm64's
// encodeLogicalImmArrEncoding: replicate the value, shrink it to the smallest
// repeating element, find the run of ones and its rotation. ok is false when
// v is not expressible (all zeros, all ones, or not a single cyclic run).
func a64LogicalImm(v int64, width int) (n, immr, imms uint32, ok bool) {
u := uint64(v)
if width == 32 {
u &= 0xFFFFFFFF
}
size := uint64(width)
mask := ^uint64(0)
if size < 64 {
mask = uint64(1)<<size - 1
}
u &= mask
// All zeros and all ones are MOV territory, not bitmask immediates.
if u == 0 || u == mask {
return 0, 0, 0, false
}
// Shrink to the smallest repeating element.
for size > 2 {
half := size / 2
hm := uint64(1)<<half - 1
if u&hm == u>>half&hm {
size = half
u &= hm
} else {
break
}
}
ones := bits.OnesCount64(u)
// Find the right-rotation that lays the ones out contiguously at the
// bottom of the element; the hardware applies the inverse rotation.
em := uint64(1)<<size - 1
expected := uint64(1)<<ones - 1
rot := -1
for r := 0; r < int(size); r++ {
rotated := u>>r | u<<(int(size)-r)
if size < 64 {
rotated &= em
}
if rotated == expected {
rot = r
break
}
}
if rot < 0 {
return 0, 0, 0, false
}
if size == 64 {
n = 1
}
immr = uint32((int(size) - rot) % int(size))
imms = ^uint32(uint32(size*2-1))&0x3F | uint32(ones-1)
return n, immr, imms, true
}
// ---- load/store (unsigned immediate, scaled) ----
// a64LSU encodes a load/store register (unsigned immediate, scaled):
@@ -778,7 +840,9 @@ func init() {
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
@@ -904,6 +968,14 @@ var a64MRSOps = map[string]uint32{
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
"NZCV": 0xd53b4200, "FPCR": 0xd53b4400, "FPSR": 0xd53b4420,
}
// a64MSRRegOps maps the system register names GOROOT writes through the
// MSR (register) form, spelled in Go assembly as MOVD Rn, <sysreg> or
// MSR Rn, <sysreg>; the source register rides bits 4:0.
var a64MSRRegOps = map[string]uint32{
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
}
// a64MSROps maps the system register names GOROOT writes to their fixed