fix(arm64): encode shifts, divides and multiplies and align sizes with emission

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-19 23:49:07 +02:00
parent 4258131a3a
commit 401386956c
8 changed files with 742 additions and 160 deletions
+28
View File
@@ -0,0 +1,28 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Wide MOV immediates that expand past one instruction, each followed by a
// branch to a later label: the displacement is what the two passes must agree
// on, so a size pass that disagrees with the encoder corrupts the branch.
// Byte-parity-checked against go tool asm.
#include "textflag.h"
TEXT ·movimm2(SB), NOSPLIT, $0-24
MOVW $-1, R0
B after1
MOVD $0x0001000200030004, R1
B after2
MOVD $0x0001000200030000, R2
B after3
MOVW $0xFFFFFFFF, R3
after1:
MOVD R0, r0+0(FP)
after2:
MOVD R1, r1+8(FP)
after3:
MOVD R2, r2+16(FP)
RET
+85
View File
@@ -0,0 +1,85 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Shifts, division and multiply-accumulate, byte-parity-checked against
// go tool asm: the immediate shift aliases (SBFM/UBFM, ROR via EXTR), the
// register two-source forms (LSLV/LSRV/ASRV/RORV), SDIV/UDIV and the
// four-operand MADD/MSUB. Every function body is padded to a whole number
// of 16 bytes so the toolchain's object adds no trailing alignment.
#include "textflag.h"
// shiftimm exercises the immediate shift aliases in both widths.
TEXT ·shiftimm(SB), NOSPLIT, $0-48
MOVD x+0(FP), R0
LSL $4, R0, R1
LSR $8, R0, R2
ASR $4, R0, R3
ROR $12, R0, R4
LSLW $4, R0, R5
LSRW $8, R0, R6
ASRW $4, R0, R7
RORW $12, R0, R8
MOVD R1, r1+0(FP)
MOVD R2, r2+8(FP)
MOVD R3, r3+16(FP)
MOVD R4, r4+24(FP)
MOVW R5, r5+32(FP)
MOVW R6, r6+40(FP)
RET
// shiftreg exercises the register shift forms in both widths: the count
// comes from a register, encoding as LSLV/LSRV/ASRV/RORV.
TEXT ·shiftreg(SB), NOSPLIT, $0-40
MOVD x+0(FP), R0
MOVD c+8(FP), R20
LSL R20, R0, R1
LSR R20, R0, R2
ASR R20, R0, R3
ROR R20, R0, R4
LSLW R20, R0, R5
LSRW R20, R0, R6
ASRW R20, R0, R7
RORW R20, R0, R8
MOVD R1, r1+0(FP)
MOVD R2, r2+8(FP)
MOVD R3, r3+16(FP)
MOVD R4, r4+24(FP)
MOVW R5, r5+32(FP)
RET
// shift2op exercises the two-operand spellings, which fold to Rn = Rd.
TEXT ·shift2op(SB), NOSPLIT, $0-24
MOVD x+0(FP), R1
MOVD c+8(FP), R20
LSL $4, R1
LSR R20, R1
LSL $12, R2
ROR $8, R2
LSLW $4, R3
RORW $8, R3
MOVD R1, r1+0(FP)
MOVD R2, r2+8(FP)
MOVD R3, r3+16(FP)
RET
// divmul exercises SDIV/UDIV in both widths and the four-operand
// MADD/MSUB, whose accumulate register is the second operand (Rm, Ra, Rn,
// Rd) and rides bits 14:10.
TEXT ·divmul(SB), NOSPLIT, $0-40
MOVD x+0(FP), R0
MOVD y+8(FP), R1
SDIV R1, R0, R2
UDIV R1, R0, R3
SDIVW R1, R0, R4
UDIVW R1, R0, R5
MADD R1, R0, R2, R6
MSUB R1, R0, R2, R7
MADDW R1, R0, R4, R8
MSUBW R1, R0, R4, R9
MOVD R2, r2+0(FP)
MOVD R3, r3+8(FP)
MOVD R6, r6+16(FP)
MOVD R7, r7+24(FP)
MOVD R8, r8+32(FP)
RET