feat(amd64): encode the mixed-width extend family and PMOVMSKB
Assisted-by: GLM 5.3
This commit is contained in:
Vendored
+39
@@ -0,0 +1,39 @@
|
||||
// Mixed-width sign- and zero-extending moves plus PMOVMSKB, the spellings
|
||||
// GOROOT's runtime and bytealg kernels use. Every result is folded back so
|
||||
// no instruction is dead.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func widen(p *byte) uint64
|
||||
TEXT ·widen(SB), NOSPLIT, $0-16
|
||||
MOVBQZX 0(DI), AX
|
||||
MOVWQZX 2(DI), CX
|
||||
ADDQ CX, AX
|
||||
MOVLQZX 4(DI), DX
|
||||
ADDQ DX, AX
|
||||
MOVBQSX 8(DI), R8
|
||||
ADDQ R8, AX
|
||||
MOVWQSX 12(DI), R9
|
||||
ADDQ R9, AX
|
||||
MOVBLSX 16(DI), R10
|
||||
ADDL R10, AX
|
||||
MOVLQSX 20(DI), R11
|
||||
ADDQ R11, AX
|
||||
MOVQ AX, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func widenw(p *byte) int32
|
||||
TEXT ·widenw(SB), NOSPLIT, $0-16
|
||||
MOVBWZX 0(DI), AX
|
||||
MOVBWSX 1(DI), CX
|
||||
ADDL CX, AX
|
||||
MOVLQZX AX, DX
|
||||
MOVL DX, ret+8(FP)
|
||||
RET
|
||||
|
||||
// func mask(x *XMM) int
|
||||
TEXT ·mask(SB), NOSPLIT, $0-16
|
||||
MOVOU 0(DI), X1
|
||||
PMOVMSKB X1, AX
|
||||
MOVQ AX, ret+8(FP)
|
||||
RET
|
||||
Reference in New Issue
Block a user