feat(arm64): wide immediates, SIMD compare and system operand forms

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 14:25:47 +02:00
parent ad82aac663
commit 9b238a525a
5 changed files with 1143 additions and 176 deletions
+48
View File
@@ -0,0 +1,48 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Carry arithmetic, logical shifts, register aliases with element selectors
// and the ADC/SBC immediate spellings: the shapes nat_arm64.s, p256 and
// gcm_arm64.s exercise. Byte-for-byte against go tool asm.
#include "textflag.h"
#define acc0 V8
#define acc1 V9
#define const0 R15
#define POLY V15
// carry pins the ADC/SBC family: the $0 spellings in two and three
// operands, and the register-carry forms.
TEXT ·carry(SB), NOSPLIT, $0-0
ADC $0, R20
ADC $0, R20, R4
SBCS $0, R4
SBCS $0, R4, R12
SBCS R15, R4, R12
SBC $0, R1
ADCSW $0, R2, R3
RET
// shift pins the shifted-register forms including ROR, which only the
// logical family accepts.
TEXT ·shift(SB), NOSPLIT, $0-0
ANDW R9@>7, R19, R26
AND R1@>33, R2, R3
ADD R1<<11, R2, R3
SUB R1->33, R2
ORR R5<<2, R6, R7
RET
// vecalias pins the vector aliases with element selectors and the
// structure loads with aliased members.
TEXT ·vecalias(SB), NOSPLIT, $0-0
MOVD $0xC2, R1
VMOV R1, POLY.D[0]
VMOV R0, POLY.D[1]
VEOR POLY.B16, POLY.B16, POLY.B16
VLD1 (R0), [acc0.B16]
VLD1.P (R0), [acc0.B16, acc1.B16]
VST1 [acc0.B16, acc1.B16], (R1)
VST1.P [acc0.B16, acc1.B16], 32(R1)
RET
+70
View File
@@ -0,0 +1,70 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Wide-immediate arithmetic: every classification band of the ADD/SUB
// immediate family (single imm12, the ADDCON2 split, bitmask and MOVZ/MOVN/
// MOVK materialisations into REGTMP) plus the logical bitmask immediates and
// their materialised fallback. Byte-for-byte against go tool asm.
#include "textflag.h"
// imm12 covers the plain and shifted-by-12 imm12 forms.
TEXT ·imm12(SB), NOSPLIT, $0-0
ADD $1, R2, R3
ADD $0x000aaa, R2, R3
ADD $0xaaa000, R2
SUB $0x000aaa, R2, R3
SUB $0xaaa000, R2
ADDW $40960, R0
CMP $40960, R0
CMPW $40960, R0
RET
// split pins the ADDCON2 band: two imm12 instructions, low half first.
TEXT ·split(SB), NOSPLIT, $0-0
ADD $0xaaaaaa, R2, R3
SUB $0xaaaaaa, R2
ADD $0x186a0, R2, R5
SUB $0x186a0, R2, R3
ADDW $0x60060, R2
RET
// regtmp covers the single-word materialisations: MOVZ for a movcon value,
// MOVN for the complement form, the bitmask ORR otherwise.
TEXT ·regtmp(SB), NOSPLIT, $0-0
ADD $0x1ffe00, R2, R3
ADD $0x3fffffffc000, R5
ADD $-2048, R2, R3
ADD $-100000, R2, R3
CMP $0x1000000, R2
CMP $0x100000000, R0
SUB $-0x100000000, R0, R1
RET
// movseq covers the omovlconst sequences: MOVZ/MOVN ladders and the
// compare forms that never split.
TEXT ·movseq(SB), NOSPLIT, $0-0
ADD $0x12345678, R2, R3
SUB $0xe7791f700, R3, R1
CMP $0xaaaaaa, R2
CMP $0xffffffffffa0, R3
CMPW $27745, R2
CMPW $0x60060, R2
ADDS $0xaaaaaa, R2, R3
CMN $0x1000000, R2
ADDW $0x12345678, R2, R3
RET
// logical covers the bitmask immediates of the logical family and the
// materialised fallback for the values a bitmask cannot carry.
TEXT ·logical(SB), NOSPLIT, $0-0
AND $0x3ff00000, R2, R3
BIC $0x22220000, R3, R4
ORR $0x3ff00000, R2
EOR $0x3ff00000, R2, R3
ANDS $0x3ff00000, R2
ORNW $0x3ff00000, R2
EONW $0x3ff00000, R2
BICSW $0x6006000060060, R5
TST $0x4900000049, R0
RET