Assisted-by: GLM 5.3 Flash
This commit is contained in:
+1024
-45
File diff suppressed because it is too large
Load Diff
@@ -29,6 +29,7 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"maps"
|
"maps"
|
||||||
|
"math/bits"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
@@ -159,6 +160,67 @@ func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
|
|||||||
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
|
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- logical immediate ----
|
||||||
|
|
||||||
|
// a64LogicalImm encodes v as the AArch64 logical (bitmask) immediate for the
|
||||||
|
// given lane width (32 or 64): it returns the N, immr and imms fields of the
|
||||||
|
// imm13 encoding. The algorithm mirrors cmd/internal/obj/arm64's
|
||||||
|
// encodeLogicalImmArrEncoding: replicate the value, shrink it to the smallest
|
||||||
|
// repeating element, find the run of ones and its rotation. ok is false when
|
||||||
|
// v is not expressible (all zeros, all ones, or not a single cyclic run).
|
||||||
|
func a64LogicalImm(v int64, width int) (n, immr, imms uint32, ok bool) {
|
||||||
|
u := uint64(v)
|
||||||
|
if width == 32 {
|
||||||
|
u &= 0xFFFFFFFF
|
||||||
|
}
|
||||||
|
size := uint64(width)
|
||||||
|
mask := ^uint64(0)
|
||||||
|
if size < 64 {
|
||||||
|
mask = uint64(1)<<size - 1
|
||||||
|
}
|
||||||
|
u &= mask
|
||||||
|
// All zeros and all ones are MOV territory, not bitmask immediates.
|
||||||
|
if u == 0 || u == mask {
|
||||||
|
return 0, 0, 0, false
|
||||||
|
}
|
||||||
|
// Shrink to the smallest repeating element.
|
||||||
|
for size > 2 {
|
||||||
|
half := size / 2
|
||||||
|
hm := uint64(1)<<half - 1
|
||||||
|
if u&hm == u>>half&hm {
|
||||||
|
size = half
|
||||||
|
u &= hm
|
||||||
|
} else {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ones := bits.OnesCount64(u)
|
||||||
|
// Find the right-rotation that lays the ones out contiguously at the
|
||||||
|
// bottom of the element; the hardware applies the inverse rotation.
|
||||||
|
em := uint64(1)<<size - 1
|
||||||
|
expected := uint64(1)<<ones - 1
|
||||||
|
rot := -1
|
||||||
|
for r := 0; r < int(size); r++ {
|
||||||
|
rotated := u>>r | u<<(int(size)-r)
|
||||||
|
if size < 64 {
|
||||||
|
rotated &= em
|
||||||
|
}
|
||||||
|
if rotated == expected {
|
||||||
|
rot = r
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if rot < 0 {
|
||||||
|
return 0, 0, 0, false
|
||||||
|
}
|
||||||
|
if size == 64 {
|
||||||
|
n = 1
|
||||||
|
}
|
||||||
|
immr = uint32((int(size) - rot) % int(size))
|
||||||
|
imms = ^uint32(uint32(size*2-1))&0x3F | uint32(ones-1)
|
||||||
|
return n, immr, imms, true
|
||||||
|
}
|
||||||
|
|
||||||
// ---- load/store (unsigned immediate, scaled) ----
|
// ---- load/store (unsigned immediate, scaled) ----
|
||||||
|
|
||||||
// a64LSU encodes a load/store register (unsigned immediate, scaled):
|
// a64LSU encodes a load/store register (unsigned immediate, scaled):
|
||||||
@@ -778,7 +840,9 @@ func init() {
|
|||||||
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
||||||
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
|
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
|
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
}
|
}
|
||||||
|
|
||||||
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
|
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
|
||||||
@@ -904,6 +968,14 @@ var a64MRSOps = map[string]uint32{
|
|||||||
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
|
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
|
||||||
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
|
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
|
||||||
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
|
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
|
||||||
|
"NZCV": 0xd53b4200, "FPCR": 0xd53b4400, "FPSR": 0xd53b4420,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64MSRRegOps maps the system register names GOROOT writes through the
|
||||||
|
// MSR (register) form, spelled in Go assembly as MOVD Rn, <sysreg> or
|
||||||
|
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
||||||
|
var a64MSRRegOps = map[string]uint32{
|
||||||
|
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
||||||
}
|
}
|
||||||
|
|
||||||
// a64MSROps maps the system register names GOROOT writes to their fixed
|
// a64MSROps maps the system register names GOROOT writes to their fixed
|
||||||
|
|||||||
Reference in New Issue
Block a user