feat(amd64): assemble the double-shift and static-SB operand shapes

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 11:40:39 +02:00
parent cc6e416c59
commit 6c672567f3
7 changed files with 196 additions and 6 deletions
+33
View File
@@ -0,0 +1,33 @@
// The three-operand SHL/SHR forms, which go tool asm encodes as SHLD/SHRD:
// immediate and CL (or its CX spelling) counts at the Q and W widths, next
// to the two-operand CX-count spelling GOROOT's bignum kernels use. Every
// result is folded back so no instruction is dead.
#include "textflag.h"
// func dblshift(x, y uint64) uint64
TEXT ·dblshift(SB), NOSPLIT, $0-24
MOVQ x+0(FP), SI
MOVQ y+8(FP), DI
MOVQ $12, CX
SHLQ $13, SI, DI
SHRQ $7, DI, SI
SHLQ CX, SI, DI
SHRQ CX, DI, SI
SHLQ CX, SI
SHLQ $9, DI
SHLW $1, SI, DI
SHRW $3, DI, SI
XORQ DI, SI
MOVQ SI, ret+16(FP)
RET
// func dblshift32(a, b uint32) uint32
TEXT ·dblshift32(SB), NOSPLIT, $0-12
MOVL a+0(FP), SI
MOVL b+4(FP), DI
SHLL $5, SI, DI
SHRL $2, DI, SI
XORL SI, DI
MOVL DI, ret+8(FP)
RET
+27
View File
@@ -0,0 +1,27 @@
// Legacy SSE octa moves against static (SB) symbols: the load and store
// shapes GOROOT's AES-CTR, AES-GCM and P-256 kernels spell (MOVOU
// bswapMask<>+0(SB), X0 and the reverse), including offsets into the symbol
// and the aligned MOVO pair. Every result is folded back so no instruction
// is dead.
#include "textflag.h"
// func ssestatic() uint64
TEXT ·ssestatic(SB), NOSPLIT, $0-8
MOVOU bswapMask<>+0(SB), X0
MOVOU bswapMask<>+8(SB), X1
MOVO rodataMask<>+0(SB), X2
PXOR X1, X0
PXOR X2, X0
MOVOU X0, sink<>+0(SB)
MOVOU sink<>+0(SB), X3
PXOR X3, X0
MOVQ X0, AX
MOVQ AX, ret+0(FP)
RET
GLOBL bswapMask<>(SB), RODATA|NOPTR, $16
GLOBL rodataMask<>(SB), RODATA|NOPTR, $16
GLOBL sink<>(SB), NOPTR, $16