Assisted-by: GLM 5.3 Flash
This commit is contained in:
+1036
-57
File diff suppressed because it is too large
Load Diff
@@ -29,6 +29,7 @@ package asm
|
||||
|
||||
import (
|
||||
"maps"
|
||||
"math/bits"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
@@ -159,6 +160,67 @@ func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
|
||||
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
|
||||
}
|
||||
|
||||
// ---- logical immediate ----
|
||||
|
||||
// a64LogicalImm encodes v as the AArch64 logical (bitmask) immediate for the
|
||||
// given lane width (32 or 64): it returns the N, immr and imms fields of the
|
||||
// imm13 encoding. The algorithm mirrors cmd/internal/obj/arm64's
|
||||
// encodeLogicalImmArrEncoding: replicate the value, shrink it to the smallest
|
||||
// repeating element, find the run of ones and its rotation. ok is false when
|
||||
// v is not expressible (all zeros, all ones, or not a single cyclic run).
|
||||
func a64LogicalImm(v int64, width int) (n, immr, imms uint32, ok bool) {
|
||||
u := uint64(v)
|
||||
if width == 32 {
|
||||
u &= 0xFFFFFFFF
|
||||
}
|
||||
size := uint64(width)
|
||||
mask := ^uint64(0)
|
||||
if size < 64 {
|
||||
mask = uint64(1)<<size - 1
|
||||
}
|
||||
u &= mask
|
||||
// All zeros and all ones are MOV territory, not bitmask immediates.
|
||||
if u == 0 || u == mask {
|
||||
return 0, 0, 0, false
|
||||
}
|
||||
// Shrink to the smallest repeating element.
|
||||
for size > 2 {
|
||||
half := size / 2
|
||||
hm := uint64(1)<<half - 1
|
||||
if u&hm == u>>half&hm {
|
||||
size = half
|
||||
u &= hm
|
||||
} else {
|
||||
break
|
||||
}
|
||||
}
|
||||
ones := bits.OnesCount64(u)
|
||||
// Find the right-rotation that lays the ones out contiguously at the
|
||||
// bottom of the element; the hardware applies the inverse rotation.
|
||||
em := uint64(1)<<size - 1
|
||||
expected := uint64(1)<<ones - 1
|
||||
rot := -1
|
||||
for r := 0; r < int(size); r++ {
|
||||
rotated := u>>r | u<<(int(size)-r)
|
||||
if size < 64 {
|
||||
rotated &= em
|
||||
}
|
||||
if rotated == expected {
|
||||
rot = r
|
||||
break
|
||||
}
|
||||
}
|
||||
if rot < 0 {
|
||||
return 0, 0, 0, false
|
||||
}
|
||||
if size == 64 {
|
||||
n = 1
|
||||
}
|
||||
immr = uint32((int(size) - rot) % int(size))
|
||||
imms = ^uint32(uint32(size*2-1))&0x3F | uint32(ones-1)
|
||||
return n, immr, imms, true
|
||||
}
|
||||
|
||||
// ---- load/store (unsigned immediate, scaled) ----
|
||||
|
||||
// a64LSU encodes a load/store register (unsigned immediate, scaled):
|
||||
@@ -778,7 +840,9 @@ func init() {
|
||||
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
}
|
||||
|
||||
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
|
||||
@@ -904,6 +968,14 @@ var a64MRSOps = map[string]uint32{
|
||||
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
|
||||
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
|
||||
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
|
||||
"NZCV": 0xd53b4200, "FPCR": 0xd53b4400, "FPSR": 0xd53b4420,
|
||||
}
|
||||
|
||||
// a64MSRRegOps maps the system register names GOROOT writes through the
|
||||
// MSR (register) form, spelled in Go assembly as MOVD Rn, <sysreg> or
|
||||
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
||||
var a64MSRRegOps = map[string]uint32{
|
||||
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
||||
}
|
||||
|
||||
// a64MSROps maps the system register names GOROOT writes to their fixed
|
||||
|
||||
Reference in New Issue
Block a user