fix(arm64): encode shifts, divides and multiplies and align sizes with emission
Assisted-by: GLM 5.3
This commit is contained in:
Vendored
+28
@@ -0,0 +1,28 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Wide MOV immediates that expand past one instruction, each followed by a
|
||||
// branch to a later label: the displacement is what the two passes must agree
|
||||
// on, so a size pass that disagrees with the encoder corrupts the branch.
|
||||
// Byte-parity-checked against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
TEXT ·movimm2(SB), NOSPLIT, $0-24
|
||||
MOVW $-1, R0
|
||||
B after1
|
||||
MOVD $0x0001000200030004, R1
|
||||
B after2
|
||||
MOVD $0x0001000200030000, R2
|
||||
B after3
|
||||
MOVW $0xFFFFFFFF, R3
|
||||
|
||||
after1:
|
||||
MOVD R0, r0+0(FP)
|
||||
|
||||
after2:
|
||||
MOVD R1, r1+8(FP)
|
||||
|
||||
after3:
|
||||
MOVD R2, r2+16(FP)
|
||||
RET
|
||||
Vendored
+85
@@ -0,0 +1,85 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Shifts, division and multiply-accumulate, byte-parity-checked against
|
||||
// go tool asm: the immediate shift aliases (SBFM/UBFM, ROR via EXTR), the
|
||||
// register two-source forms (LSLV/LSRV/ASRV/RORV), SDIV/UDIV and the
|
||||
// four-operand MADD/MSUB. Every function body is padded to a whole number
|
||||
// of 16 bytes so the toolchain's object adds no trailing alignment.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// shiftimm exercises the immediate shift aliases in both widths.
|
||||
TEXT ·shiftimm(SB), NOSPLIT, $0-48
|
||||
MOVD x+0(FP), R0
|
||||
LSL $4, R0, R1
|
||||
LSR $8, R0, R2
|
||||
ASR $4, R0, R3
|
||||
ROR $12, R0, R4
|
||||
LSLW $4, R0, R5
|
||||
LSRW $8, R0, R6
|
||||
ASRW $4, R0, R7
|
||||
RORW $12, R0, R8
|
||||
MOVD R1, r1+0(FP)
|
||||
MOVD R2, r2+8(FP)
|
||||
MOVD R3, r3+16(FP)
|
||||
MOVD R4, r4+24(FP)
|
||||
MOVW R5, r5+32(FP)
|
||||
MOVW R6, r6+40(FP)
|
||||
RET
|
||||
|
||||
// shiftreg exercises the register shift forms in both widths: the count
|
||||
// comes from a register, encoding as LSLV/LSRV/ASRV/RORV.
|
||||
TEXT ·shiftreg(SB), NOSPLIT, $0-40
|
||||
MOVD x+0(FP), R0
|
||||
MOVD c+8(FP), R20
|
||||
LSL R20, R0, R1
|
||||
LSR R20, R0, R2
|
||||
ASR R20, R0, R3
|
||||
ROR R20, R0, R4
|
||||
LSLW R20, R0, R5
|
||||
LSRW R20, R0, R6
|
||||
ASRW R20, R0, R7
|
||||
RORW R20, R0, R8
|
||||
MOVD R1, r1+0(FP)
|
||||
MOVD R2, r2+8(FP)
|
||||
MOVD R3, r3+16(FP)
|
||||
MOVD R4, r4+24(FP)
|
||||
MOVW R5, r5+32(FP)
|
||||
RET
|
||||
|
||||
// shift2op exercises the two-operand spellings, which fold to Rn = Rd.
|
||||
TEXT ·shift2op(SB), NOSPLIT, $0-24
|
||||
MOVD x+0(FP), R1
|
||||
MOVD c+8(FP), R20
|
||||
LSL $4, R1
|
||||
LSR R20, R1
|
||||
LSL $12, R2
|
||||
ROR $8, R2
|
||||
LSLW $4, R3
|
||||
RORW $8, R3
|
||||
MOVD R1, r1+0(FP)
|
||||
MOVD R2, r2+8(FP)
|
||||
MOVD R3, r3+16(FP)
|
||||
RET
|
||||
|
||||
// divmul exercises SDIV/UDIV in both widths and the four-operand
|
||||
// MADD/MSUB, whose accumulate register is the second operand (Rm, Ra, Rn,
|
||||
// Rd) and rides bits 14:10.
|
||||
TEXT ·divmul(SB), NOSPLIT, $0-40
|
||||
MOVD x+0(FP), R0
|
||||
MOVD y+8(FP), R1
|
||||
SDIV R1, R0, R2
|
||||
UDIV R1, R0, R3
|
||||
SDIVW R1, R0, R4
|
||||
UDIVW R1, R0, R5
|
||||
MADD R1, R0, R2, R6
|
||||
MSUB R1, R0, R2, R7
|
||||
MADDW R1, R0, R4, R8
|
||||
MSUBW R1, R0, R4, R9
|
||||
MOVD R2, r2+0(FP)
|
||||
MOVD R3, r3+8(FP)
|
||||
MOVD R6, r6+16(FP)
|
||||
MOVD R7, r7+24(FP)
|
||||
MOVD R8, r8+32(FP)
|
||||
RET
|
||||
Reference in New Issue
Block a user