Files

86 lines
2.1 KiB
ArmAsm
Raw Permalink Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Shifts, division and multiply-accumulate, byte-parity-checked against
// go tool asm: the immediate shift aliases (SBFM/UBFM, ROR via EXTR), the
// register two-source forms (LSLV/LSRV/ASRV/RORV), SDIV/UDIV and the
// four-operand MADD/MSUB. Every function body is padded to a whole number
// of 16 bytes so the toolchain's object adds no trailing alignment.
#include "textflag.h"
// shiftimm exercises the immediate shift aliases in both widths.
TEXT ·shiftimm(SB), NOSPLIT, $0-48
MOVD x+0(FP), R0
LSL $4, R0, R1
LSR $8, R0, R2
ASR $4, R0, R3
ROR $12, R0, R4
LSLW $4, R0, R5
LSRW $8, R0, R6
ASRW $4, R0, R7
RORW $12, R0, R8
MOVD R1, r1+0(FP)
MOVD R2, r2+8(FP)
MOVD R3, r3+16(FP)
MOVD R4, r4+24(FP)
MOVW R5, r5+32(FP)
MOVW R6, r6+40(FP)
RET
// shiftreg exercises the register shift forms in both widths: the count
// comes from a register, encoding as LSLV/LSRV/ASRV/RORV.
TEXT ·shiftreg(SB), NOSPLIT, $0-40
MOVD x+0(FP), R0
MOVD c+8(FP), R20
LSL R20, R0, R1
LSR R20, R0, R2
ASR R20, R0, R3
ROR R20, R0, R4
LSLW R20, R0, R5
LSRW R20, R0, R6
ASRW R20, R0, R7
RORW R20, R0, R8
MOVD R1, r1+0(FP)
MOVD R2, r2+8(FP)
MOVD R3, r3+16(FP)
MOVD R4, r4+24(FP)
MOVW R5, r5+32(FP)
RET
// shift2op exercises the two-operand spellings, which fold to Rn = Rd.
TEXT ·shift2op(SB), NOSPLIT, $0-24
MOVD x+0(FP), R1
MOVD c+8(FP), R20
LSL $4, R1
LSR R20, R1
LSL $12, R2
ROR $8, R2
LSLW $4, R3
RORW $8, R3
MOVD R1, r1+0(FP)
MOVD R2, r2+8(FP)
MOVD R3, r3+16(FP)
RET
// divmul exercises SDIV/UDIV in both widths and the four-operand
// MADD/MSUB, whose accumulate register is the second operand (Rm, Ra, Rn,
// Rd) and rides bits 14:10.
TEXT ·divmul(SB), NOSPLIT, $0-40
MOVD x+0(FP), R0
MOVD y+8(FP), R1
SDIV R1, R0, R2
UDIV R1, R0, R3
SDIVW R1, R0, R4
UDIVW R1, R0, R5
MADD R1, R0, R2, R6
MSUB R1, R0, R2, R7
MADDW R1, R0, R4, R8
MSUBW R1, R0, R4, R9
MOVD R2, r2+0(FP)
MOVD R3, r3+8(FP)
MOVD R6, r6+16(FP)
MOVD R7, r7+24(FP)
MOVD R8, r8+32(FP)
RET