86 lines
2.1 KiB
ArmAsm
86 lines
2.1 KiB
ArmAsm
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|||
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
||
|
|
|
||
|
|
// Shifts, division and multiply-accumulate, byte-parity-checked against
|
||
|
|
// go tool asm: the immediate shift aliases (SBFM/UBFM, ROR via EXTR), the
|
||
|
|
// register two-source forms (LSLV/LSRV/ASRV/RORV), SDIV/UDIV and the
|
||
|
|
// four-operand MADD/MSUB. Every function body is padded to a whole number
|
||
|
|
// of 16 bytes so the toolchain's object adds no trailing alignment.
|
||
|
|
|
||
|
|
#include "textflag.h"
|
||
|
|
|
||
|
|
// shiftimm exercises the immediate shift aliases in both widths.
|
||
|
|
TEXT ·shiftimm(SB), NOSPLIT, $0-48
|
||
|
|
MOVD x+0(FP), R0
|
||
|
|
LSL $4, R0, R1
|
||
|
|
LSR $8, R0, R2
|
||
|
|
ASR $4, R0, R3
|
||
|
|
ROR $12, R0, R4
|
||
|
|
LSLW $4, R0, R5
|
||
|
|
LSRW $8, R0, R6
|
||
|
|
ASRW $4, R0, R7
|
||
|
|
RORW $12, R0, R8
|
||
|
|
MOVD R1, r1+0(FP)
|
||
|
|
MOVD R2, r2+8(FP)
|
||
|
|
MOVD R3, r3+16(FP)
|
||
|
|
MOVD R4, r4+24(FP)
|
||
|
|
MOVW R5, r5+32(FP)
|
||
|
|
MOVW R6, r6+40(FP)
|
||
|
|
RET
|
||
|
|
|
||
|
|
// shiftreg exercises the register shift forms in both widths: the count
|
||
|
|
// comes from a register, encoding as LSLV/LSRV/ASRV/RORV.
|
||
|
|
TEXT ·shiftreg(SB), NOSPLIT, $0-40
|
||
|
|
MOVD x+0(FP), R0
|
||
|
|
MOVD c+8(FP), R20
|
||
|
|
LSL R20, R0, R1
|
||
|
|
LSR R20, R0, R2
|
||
|
|
ASR R20, R0, R3
|
||
|
|
ROR R20, R0, R4
|
||
|
|
LSLW R20, R0, R5
|
||
|
|
LSRW R20, R0, R6
|
||
|
|
ASRW R20, R0, R7
|
||
|
|
RORW R20, R0, R8
|
||
|
|
MOVD R1, r1+0(FP)
|
||
|
|
MOVD R2, r2+8(FP)
|
||
|
|
MOVD R3, r3+16(FP)
|
||
|
|
MOVD R4, r4+24(FP)
|
||
|
|
MOVW R5, r5+32(FP)
|
||
|
|
RET
|
||
|
|
|
||
|
|
// shift2op exercises the two-operand spellings, which fold to Rn = Rd.
|
||
|
|
TEXT ·shift2op(SB), NOSPLIT, $0-24
|
||
|
|
MOVD x+0(FP), R1
|
||
|
|
MOVD c+8(FP), R20
|
||
|
|
LSL $4, R1
|
||
|
|
LSR R20, R1
|
||
|
|
LSL $12, R2
|
||
|
|
ROR $8, R2
|
||
|
|
LSLW $4, R3
|
||
|
|
RORW $8, R3
|
||
|
|
MOVD R1, r1+0(FP)
|
||
|
|
MOVD R2, r2+8(FP)
|
||
|
|
MOVD R3, r3+16(FP)
|
||
|
|
RET
|
||
|
|
|
||
|
|
// divmul exercises SDIV/UDIV in both widths and the four-operand
|
||
|
|
// MADD/MSUB, whose accumulate register is the second operand (Rm, Ra, Rn,
|
||
|
|
// Rd) and rides bits 14:10.
|
||
|
|
TEXT ·divmul(SB), NOSPLIT, $0-40
|
||
|
|
MOVD x+0(FP), R0
|
||
|
|
MOVD y+8(FP), R1
|
||
|
|
SDIV R1, R0, R2
|
||
|
|
UDIV R1, R0, R3
|
||
|
|
SDIVW R1, R0, R4
|
||
|
|
UDIVW R1, R0, R5
|
||
|
|
MADD R1, R0, R2, R6
|
||
|
|
MSUB R1, R0, R2, R7
|
||
|
|
MADDW R1, R0, R4, R8
|
||
|
|
MSUBW R1, R0, R4, R9
|
||
|
|
MOVD R2, r2+0(FP)
|
||
|
|
MOVD R3, r3+8(FP)
|
||
|
|
MOVD R6, r6+16(FP)
|
||
|
|
MOVD R7, r7+24(FP)
|
||
|
|
MOVD R8, r8+32(FP)
|
||
|
|
RET
|