mirror of
https://github.com/BLAKE3-team/BLAKE3
synced 2024-04-27 16:55:04 +02:00
assembly implementations
This commit is contained in:
parent
1c5d4eea6a
commit
b6b3c27824
|
@ -0,0 +1,1800 @@
|
|||
.intel_syntax noprefix
|
||||
.global _blake3_hash_many_avx2
|
||||
.global blake3_hash_many_avx2
|
||||
#ifdef __APPLE__
|
||||
.text
|
||||
#else
|
||||
.section .text
|
||||
#endif
|
||||
.p2align 6
|
||||
_blake3_hash_many_avx2:
|
||||
blake3_hash_many_avx2:
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 680
|
||||
and rsp, 0xFFFFFFFFFFFFFFC0
|
||||
neg r9d
|
||||
vmovd xmm0, r9d
|
||||
vpbroadcastd ymm0, xmm0
|
||||
vmovdqa ymmword ptr [rsp+0x280], ymm0
|
||||
vpand ymm1, ymm0, ymmword ptr [ADD0+rip]
|
||||
vpand ymm2, ymm0, ymmword ptr [ADD1+rip]
|
||||
vmovdqa ymmword ptr [rsp+0x220], ymm2
|
||||
vmovd xmm2, r8d
|
||||
vpbroadcastd ymm2, xmm2
|
||||
vpaddd ymm2, ymm2, ymm1
|
||||
vmovdqa ymmword ptr [rsp+0x240], ymm2
|
||||
vpxor ymm1, ymm1, ymmword ptr [CMP_MSB_MASK+rip]
|
||||
vpxor ymm2, ymm2, ymmword ptr [CMP_MSB_MASK+rip]
|
||||
vpcmpgtd ymm2, ymm1, ymm2
|
||||
shr r8, 32
|
||||
vmovd xmm3, r8d
|
||||
vpbroadcastd ymm3, xmm3
|
||||
vpsubd ymm3, ymm3, ymm2
|
||||
vmovdqa ymmword ptr [rsp+0x260], ymm3
|
||||
shl rdx, 6
|
||||
mov qword ptr [rsp+0x2A0], rdx
|
||||
cmp rsi, 8
|
||||
jc 3f
|
||||
2:
|
||||
vpbroadcastd ymm0, dword ptr [rcx]
|
||||
vpbroadcastd ymm1, dword ptr [rcx+0x4]
|
||||
vpbroadcastd ymm2, dword ptr [rcx+0x8]
|
||||
vpbroadcastd ymm3, dword ptr [rcx+0xC]
|
||||
vpbroadcastd ymm4, dword ptr [rcx+0x10]
|
||||
vpbroadcastd ymm5, dword ptr [rcx+0x14]
|
||||
vpbroadcastd ymm6, dword ptr [rcx+0x18]
|
||||
vpbroadcastd ymm7, dword ptr [rcx+0x1C]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov r12, qword ptr [rdi+0x20]
|
||||
mov r13, qword ptr [rdi+0x28]
|
||||
mov r14, qword ptr [rdi+0x30]
|
||||
mov r15, qword ptr [rdi+0x38]
|
||||
movzx eax, byte ptr [rbp+0x38]
|
||||
movzx ebx, byte ptr [rbp+0x40]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
1:
|
||||
movzx ebx, byte ptr [rbp+0x48]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+0x2A0]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+0x200], eax
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x40], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x40]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x40], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x40]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x40], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x40]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x40], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0x20], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0x40], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0x60], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x30], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x30]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x30], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x30]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x30], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x30]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x30], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+0x80], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0xA0], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0xC0], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0xE0], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x20], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x20]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x20], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x20]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x20], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x20]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x20], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+0x100], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0x120], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0x140], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0x160], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x10], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x10]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x10], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x10]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x10], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x10]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x10], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+0x180], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0x1A0], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0x1C0], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0x1E0], ymm11
|
||||
vpbroadcastd ymm15, dword ptr [rsp+0x200]
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r12+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r13+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r14+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
prefetcht0 [r15+rdx+0x80]
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm0, ymmword ptr [rsp+0x240]
|
||||
vpxor ymm13, ymm1, ymmword ptr [rsp+0x260]
|
||||
vpxor ymm14, ymm2, ymmword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vpxor ymm15, ymm3, ymm15
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [BLAKE3_IV_0+rip]
|
||||
vpaddd ymm9, ymm13, ymmword ptr [BLAKE3_IV_1+rip]
|
||||
vpaddd ymm10, ymm14, ymmword ptr [BLAKE3_IV_2+rip]
|
||||
vpaddd ymm11, ymm15, ymmword ptr [BLAKE3_IV_3+rip]
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vpxor ymm0, ymm0, ymm8
|
||||
vpxor ymm1, ymm1, ymm9
|
||||
vpxor ymm2, ymm2, ymm10
|
||||
vpxor ymm3, ymm3, ymm11
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpxor ymm4, ymm4, ymm12
|
||||
vpxor ymm5, ymm5, ymm13
|
||||
vpxor ymm6, ymm6, ymm14
|
||||
vpxor ymm7, ymm7, ymm15
|
||||
movzx eax, byte ptr [rbp+0x38]
|
||||
jne 1b
|
||||
mov rbx, qword ptr [rbp+0x50]
|
||||
vunpcklps ymm8, ymm0, ymm1
|
||||
vunpcklps ymm9, ymm2, ymm3
|
||||
vunpckhps ymm10, ymm0, ymm1
|
||||
vunpcklps ymm11, ymm4, ymm5
|
||||
vunpcklps ymm0, ymm6, ymm7
|
||||
vshufps ymm12, ymm8, ymm9, 78
|
||||
vblendps ymm1, ymm8, ymm12, 0xCC
|
||||
vshufps ymm8, ymm11, ymm0, 78
|
||||
vunpckhps ymm13, ymm2, ymm3
|
||||
vblendps ymm2, ymm11, ymm8, 0xCC
|
||||
vblendps ymm3, ymm12, ymm9, 0xCC
|
||||
vperm2f128 ymm12, ymm1, ymm2, 0x20
|
||||
vmovups ymmword ptr [rbx], ymm12
|
||||
vunpckhps ymm14, ymm4, ymm5
|
||||
vblendps ymm4, ymm8, ymm0, 0xCC
|
||||
vunpckhps ymm15, ymm6, ymm7
|
||||
vperm2f128 ymm7, ymm3, ymm4, 0x20
|
||||
vmovups ymmword ptr [rbx+0x20], ymm7
|
||||
vshufps ymm5, ymm10, ymm13, 78
|
||||
vblendps ymm6, ymm5, ymm13, 0xCC
|
||||
vshufps ymm13, ymm14, ymm15, 78
|
||||
vblendps ymm10, ymm10, ymm5, 0xCC
|
||||
vblendps ymm14, ymm14, ymm13, 0xCC
|
||||
vperm2f128 ymm8, ymm10, ymm14, 0x20
|
||||
vmovups ymmword ptr [rbx+0x40], ymm8
|
||||
vblendps ymm15, ymm13, ymm15, 0xCC
|
||||
vperm2f128 ymm13, ymm6, ymm15, 0x20
|
||||
vmovups ymmword ptr [rbx+0x60], ymm13
|
||||
vperm2f128 ymm9, ymm1, ymm2, 0x31
|
||||
vperm2f128 ymm11, ymm3, ymm4, 0x31
|
||||
vmovups ymmword ptr [rbx+0x80], ymm9
|
||||
vperm2f128 ymm14, ymm10, ymm14, 0x31
|
||||
vperm2f128 ymm15, ymm6, ymm15, 0x31
|
||||
vmovups ymmword ptr [rbx+0xA0], ymm11
|
||||
vmovups ymmword ptr [rbx+0xC0], ymm14
|
||||
vmovups ymmword ptr [rbx+0xE0], ymm15
|
||||
vmovdqa ymm0, ymmword ptr [rsp+0x220]
|
||||
vpaddd ymm1, ymm0, ymmword ptr [rsp+0x240]
|
||||
vmovdqa ymmword ptr [rsp+0x240], ymm1
|
||||
vpxor ymm0, ymm0, ymmword ptr [CMP_MSB_MASK+rip]
|
||||
vpxor ymm2, ymm1, ymmword ptr [CMP_MSB_MASK+rip]
|
||||
vpcmpgtd ymm2, ymm0, ymm2
|
||||
vmovdqa ymm0, ymmword ptr [rsp+0x260]
|
||||
vpsubd ymm2, ymm0, ymm2
|
||||
vmovdqa ymmword ptr [rsp+0x260], ymm2
|
||||
add rdi, 64
|
||||
add rbx, 256
|
||||
mov qword ptr [rbp+0x50], rbx
|
||||
sub rsi, 8
|
||||
cmp rsi, 8
|
||||
jnc 2b
|
||||
test rsi, rsi
|
||||
jnz 3f
|
||||
4:
|
||||
vzeroupper
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
.p2align 5
|
||||
3:
|
||||
mov rbx, qword ptr [rbp+0x50]
|
||||
mov r15, qword ptr [rsp+0x2A0]
|
||||
movzx r13d, byte ptr [rbp+0x38]
|
||||
movzx r12d, byte ptr [rbp+0x48]
|
||||
test rsi, 0x4
|
||||
je 3f
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+0x10]
|
||||
vmovdqa ymm8, ymm0
|
||||
vmovdqa ymm9, ymm1
|
||||
vbroadcasti128 ymm12, xmmword ptr [rsp+0x240]
|
||||
vbroadcasti128 ymm13, xmmword ptr [rsp+0x260]
|
||||
vpunpckldq ymm14, ymm12, ymm13
|
||||
vpunpckhdq ymm15, ymm12, ymm13
|
||||
vpermq ymm14, ymm14, 0x50
|
||||
vpermq ymm15, ymm15, 0x50
|
||||
vbroadcasti128 ymm12, xmmword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vpblendd ymm14, ymm14, ymm12, 0x44
|
||||
vpblendd ymm15, ymm15, ymm12, 0x44
|
||||
vmovdqa ymmword ptr [rsp], ymm14
|
||||
vmovdqa ymmword ptr [rsp+0x20], ymm15
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+0x200], eax
|
||||
vmovups ymm2, ymmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm2, ymm2, xmmword ptr [r9+rdx-0x40], 0x01
|
||||
vmovups ymm3, ymmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm3, ymm3, xmmword ptr [r9+rdx-0x30], 0x01
|
||||
vshufps ymm4, ymm2, ymm3, 136
|
||||
vshufps ymm5, ymm2, ymm3, 221
|
||||
vmovups ymm2, ymmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm2, ymm2, xmmword ptr [r9+rdx-0x20], 0x01
|
||||
vmovups ymm3, ymmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm3, ymm3, xmmword ptr [r9+rdx-0x10], 0x01
|
||||
vshufps ymm6, ymm2, ymm3, 136
|
||||
vshufps ymm7, ymm2, ymm3, 221
|
||||
vpshufd ymm6, ymm6, 0x93
|
||||
vpshufd ymm7, ymm7, 0x93
|
||||
vmovups ymm10, ymmword ptr [r10+rdx-0x40]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r11+rdx-0x40], 0x01
|
||||
vmovups ymm11, ymmword ptr [r10+rdx-0x30]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r11+rdx-0x30], 0x01
|
||||
vshufps ymm12, ymm10, ymm11, 136
|
||||
vshufps ymm13, ymm10, ymm11, 221
|
||||
vmovups ymm10, ymmword ptr [r10+rdx-0x20]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r11+rdx-0x20], 0x01
|
||||
vmovups ymm11, ymmword ptr [r10+rdx-0x10]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r11+rdx-0x10], 0x01
|
||||
vshufps ymm14, ymm10, ymm11, 136
|
||||
vshufps ymm15, ymm10, ymm11, 221
|
||||
vpshufd ymm14, ymm14, 0x93
|
||||
vpshufd ymm15, ymm15, 0x93
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
vpbroadcastd ymm2, dword ptr [rsp+0x200]
|
||||
vmovdqa ymm3, ymmword ptr [rsp]
|
||||
vmovdqa ymm11, ymmword ptr [rsp+0x20]
|
||||
vpblendd ymm3, ymm3, ymm2, 0x88
|
||||
vpblendd ymm11, ymm11, ymm2, 0x88
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vmovdqa ymm10, ymm2
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vmovdqa ymmword ptr [rsp+0x40], ymm4
|
||||
nop
|
||||
vmovdqa ymmword ptr [rsp+0x60], ymm12
|
||||
nop
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 12
|
||||
vpslld ymm9, ymm9, 20
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x80], ymm5
|
||||
vmovdqa ymmword ptr [rsp+0xA0], ymm13
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 7
|
||||
vpslld ymm9, ymm9, 25
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpshufd ymm0, ymm0, 0x93
|
||||
vpshufd ymm8, ymm8, 0x93
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm11, ymm11, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x39
|
||||
vpshufd ymm10, ymm10, 0x39
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm8, ymm8, ymm14
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 12
|
||||
vpslld ymm9, ymm9, 20
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm8, ymm8, ymm15
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 7
|
||||
vpslld ymm9, ymm9, 25
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpshufd ymm0, ymm0, 0x39
|
||||
vpshufd ymm8, ymm8, 0x39
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm11, ymm11, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x93
|
||||
vpshufd ymm10, ymm10, 0x93
|
||||
dec al
|
||||
je 1f
|
||||
vmovdqa ymm4, ymmword ptr [rsp+0x40]
|
||||
vmovdqa ymm5, ymmword ptr [rsp+0x80]
|
||||
vshufps ymm12, ymm4, ymm5, 214
|
||||
vpshufd ymm13, ymm4, 0x0F
|
||||
vpshufd ymm4, ymm12, 0x39
|
||||
vshufps ymm12, ymm6, ymm7, 250
|
||||
vpblendd ymm13, ymm13, ymm12, 0xAA
|
||||
vpunpcklqdq ymm12, ymm7, ymm5
|
||||
vpblendd ymm12, ymm12, ymm6, 0x88
|
||||
vpshufd ymm12, ymm12, 0x78
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 0x1E
|
||||
vmovdqa ymmword ptr [rsp+0x40], ymm13
|
||||
vmovdqa ymmword ptr [rsp+0x80], ymm12
|
||||
vmovdqa ymm12, ymmword ptr [rsp+0x60]
|
||||
vmovdqa ymm13, ymmword ptr [rsp+0xA0]
|
||||
vshufps ymm5, ymm12, ymm13, 214
|
||||
vpshufd ymm6, ymm12, 0x0F
|
||||
vpshufd ymm12, ymm5, 0x39
|
||||
vshufps ymm5, ymm14, ymm15, 250
|
||||
vpblendd ymm6, ymm6, ymm5, 0xAA
|
||||
vpunpcklqdq ymm5, ymm15, ymm13
|
||||
vpblendd ymm5, ymm5, ymm14, 0x88
|
||||
vpshufd ymm5, ymm5, 0x78
|
||||
vpunpckhdq ymm13, ymm13, ymm15
|
||||
vpunpckldq ymm14, ymm14, ymm13
|
||||
vpshufd ymm15, ymm14, 0x1E
|
||||
vmovdqa ymm13, ymm6
|
||||
vmovdqa ymm14, ymm5
|
||||
vmovdqa ymm5, ymmword ptr [rsp+0x40]
|
||||
vmovdqa ymm6, ymmword ptr [rsp+0x80]
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
vpxor ymm8, ymm8, ymm10
|
||||
vpxor ymm9, ymm9, ymm11
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
vextracti128 xmmword ptr [rbx+0x20], ymm0, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x30], ymm1, 0x01
|
||||
vmovdqu xmmword ptr [rbx+0x40], xmm8
|
||||
vmovdqu xmmword ptr [rbx+0x50], xmm9
|
||||
vextracti128 xmmword ptr [rbx+0x60], ymm8, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x70], ymm9, 0x01
|
||||
vmovaps xmm8, xmmword ptr [rsp+0x280]
|
||||
vmovaps xmm0, xmmword ptr [rsp+0x240]
|
||||
vmovaps xmm1, xmmword ptr [rsp+0x250]
|
||||
vmovaps xmm2, xmmword ptr [rsp+0x260]
|
||||
vmovaps xmm3, xmmword ptr [rsp+0x270]
|
||||
vblendvps xmm0, xmm0, xmm1, xmm8
|
||||
vblendvps xmm2, xmm2, xmm3, xmm8
|
||||
vmovaps xmmword ptr [rsp+0x240], xmm0
|
||||
vmovaps xmmword ptr [rsp+0x260], xmm2
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
3:
|
||||
test rsi, 0x2
|
||||
je 3f
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+0x10]
|
||||
vmovd xmm13, dword ptr [rsp+0x240]
|
||||
vpinsrd xmm13, xmm13, dword ptr [rsp+0x260], 1
|
||||
vpinsrd xmm13, xmm13, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vmovd xmm14, dword ptr [rsp+0x244]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+0x264], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vinserti128 ymm13, ymm13, xmm14, 0x01
|
||||
vbroadcasti128 ymm14, xmmword ptr [ROT16+rip]
|
||||
vbroadcasti128 ymm15, xmmword ptr [ROT8+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+0x200], eax
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vpbroadcastd ymm8, dword ptr [rsp+0x200]
|
||||
vpblendd ymm3, ymm13, ymm8, 0x88
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-0x40], 0x01
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-0x30], 0x01
|
||||
vshufps ymm4, ymm8, ymm9, 136
|
||||
vshufps ymm5, ymm8, ymm9, 221
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-0x20], 0x01
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-0x10], 0x01
|
||||
vshufps ymm6, ymm8, ymm9, 136
|
||||
vshufps ymm7, ymm8, ymm9, 221
|
||||
vpshufd ymm6, ymm6, 0x93
|
||||
vpshufd ymm7, ymm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm14
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm15
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpshufd ymm0, ymm0, 0x93
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x39
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm14
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm15
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpshufd ymm0, ymm0, 0x39
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps ymm8, ymm4, ymm5, 214
|
||||
vpshufd ymm9, ymm4, 0x0F
|
||||
vpshufd ymm4, ymm8, 0x39
|
||||
vshufps ymm8, ymm6, ymm7, 250
|
||||
vpblendd ymm9, ymm9, ymm8, 0xAA
|
||||
vpunpcklqdq ymm8, ymm7, ymm5
|
||||
vpblendd ymm8, ymm8, ymm6, 0x88
|
||||
vpshufd ymm8, ymm8, 0x78
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 0x1E
|
||||
vmovdqa ymm5, ymm9
|
||||
vmovdqa ymm6, ymm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
vextracti128 xmmword ptr [rbx+0x20], ymm0, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x30], ymm1, 0x01
|
||||
vmovaps ymm8, ymmword ptr [rsp+0x280]
|
||||
vmovaps ymm0, ymmword ptr [rsp+0x240]
|
||||
vmovups ymm1, ymmword ptr [rsp+0x248]
|
||||
vmovaps ymm2, ymmword ptr [rsp+0x260]
|
||||
vmovups ymm3, ymmword ptr [rsp+0x268]
|
||||
vblendvps ymm0, ymm0, ymm1, ymm8
|
||||
vblendvps ymm2, ymm2, ymm3, ymm8
|
||||
vmovaps ymmword ptr [rsp+0x240], ymm0
|
||||
vmovaps ymmword ptr [rsp+0x260], ymm2
|
||||
add rbx, 64
|
||||
add rdi, 16
|
||||
sub rsi, 2
|
||||
3:
|
||||
test rsi, 0x1
|
||||
je 4b
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+0x10]
|
||||
vmovd xmm3, dword ptr [rsp+0x240]
|
||||
vpinsrd xmm3, xmm3, dword ptr [rsp+0x260], 1
|
||||
vpinsrd xmm13, xmm3, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vmovdqa xmm14, xmmword ptr [ROT16+rip]
|
||||
vmovdqa xmm15, xmmword ptr [ROT8+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
vmovdqa xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vmovdqa xmm3, xmm13
|
||||
vpinsrd xmm3, xmm3, eax, 3
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-0x30]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-0x10]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 0x93
|
||||
vpshufd xmm7, xmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm14
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 12
|
||||
vpslld xmm1, xmm1, 20
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm15
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 7
|
||||
vpslld xmm1, xmm1, 25
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpshufd xmm0, xmm0, 0x93
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x39
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm14
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 12
|
||||
vpslld xmm1, xmm1, 20
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm15
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 7
|
||||
vpslld xmm1, xmm1, 25
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpshufd xmm0, xmm0, 0x39
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0x0F
|
||||
vpshufd xmm4, xmm8, 0x39
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0xAA
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 0x88
|
||||
vpshufd xmm8, xmm8, 0x78
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 0x1E
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
jmp 4b
|
||||
|
||||
|
||||
#ifdef __APPLE__
|
||||
.static_data
|
||||
#else
|
||||
.section .rodata
|
||||
#endif
|
||||
.p2align 6
|
||||
ADD0:
|
||||
.long 0, 1, 2, 3, 4, 5, 6, 7
|
||||
ADD1:
|
||||
.long 8, 8, 8, 8, 8, 8, 8, 8
|
||||
BLAKE3_IV_0:
|
||||
.long 0x6A09E667, 0x6A09E667, 0x6A09E667, 0x6A09E667
|
||||
.long 0x6A09E667, 0x6A09E667, 0x6A09E667, 0x6A09E667
|
||||
BLAKE3_IV_1:
|
||||
.long 0xBB67AE85, 0xBB67AE85, 0xBB67AE85, 0xBB67AE85
|
||||
.long 0xBB67AE85, 0xBB67AE85, 0xBB67AE85, 0xBB67AE85
|
||||
BLAKE3_IV_2:
|
||||
.long 0x3C6EF372, 0x3C6EF372, 0x3C6EF372, 0x3C6EF372
|
||||
.long 0x3C6EF372, 0x3C6EF372, 0x3C6EF372, 0x3C6EF372
|
||||
BLAKE3_IV_3:
|
||||
.long 0xA54FF53A, 0xA54FF53A, 0xA54FF53A, 0xA54FF53A
|
||||
.long 0xA54FF53A, 0xA54FF53A, 0xA54FF53A, 0xA54FF53A
|
||||
BLAKE3_BLOCK_LEN:
|
||||
.long 0x00000040, 0x00000040, 0x00000040, 0x00000040
|
||||
.long 0x00000040, 0x00000040, 0x00000040, 0x00000040
|
||||
ROT16:
|
||||
.byte 2, 3, 0, 1, 6, 7, 4, 5, 10, 11, 8, 9, 14, 15, 12, 13
|
||||
ROT8:
|
||||
.byte 1, 2, 3, 0, 5, 6, 7, 4, 9, 10, 11, 8, 13, 14, 15, 12
|
||||
CMP_MSB_MASK:
|
||||
.long 0x80000000, 0x80000000, 0x80000000, 0x80000000
|
||||
.long 0x80000000, 0x80000000, 0x80000000, 0x80000000
|
||||
BLAKE3_IV:
|
||||
.long 0x6A09E667, 0xBB67AE85, 0x3C6EF372, 0xA54FF53A
|
||||
|
|
@ -0,0 +1,1817 @@
|
|||
.intel_syntax noprefix
|
||||
.global _blake3_hash_many_avx2
|
||||
.global blake3_hash_many_avx2
|
||||
.section .text
|
||||
.p2align 6
|
||||
_blake3_hash_many_avx2:
|
||||
blake3_hash_many_avx2:
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rsi
|
||||
push rdi
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 880
|
||||
and rsp, 0xFFFFFFFFFFFFFFC0
|
||||
vmovdqa xmmword ptr [rsp+0x2D0], xmm6
|
||||
vmovdqa xmmword ptr [rsp+0x2E0], xmm7
|
||||
vmovdqa xmmword ptr [rsp+0x2F0], xmm8
|
||||
vmovdqa xmmword ptr [rsp+0x300], xmm9
|
||||
vmovdqa xmmword ptr [rsp+0x310], xmm10
|
||||
vmovdqa xmmword ptr [rsp+0x320], xmm11
|
||||
vmovdqa xmmword ptr [rsp+0x330], xmm12
|
||||
vmovdqa xmmword ptr [rsp+0x340], xmm13
|
||||
vmovdqa xmmword ptr [rsp+0x350], xmm14
|
||||
vmovdqa xmmword ptr [rsp+0x360], xmm15
|
||||
mov rdi, rcx
|
||||
mov rsi, rdx
|
||||
mov rdx, r8
|
||||
mov rcx, r9
|
||||
mov r8, qword ptr [rbp+0x68]
|
||||
movzx r9, byte ptr [rbp+0x70]
|
||||
neg r9d
|
||||
vmovd xmm0, r9d
|
||||
vpbroadcastd ymm0, xmm0
|
||||
vmovdqa ymmword ptr [rsp+0x260], ymm0
|
||||
vpand ymm1, ymm0, ymmword ptr [ADD0+rip]
|
||||
vpand ymm2, ymm0, ymmword ptr [ADD1+rip]
|
||||
vmovdqa ymmword ptr [rsp+0x2A0], ymm2
|
||||
vmovd xmm2, r8d
|
||||
vpbroadcastd ymm2, xmm2
|
||||
vpaddd ymm2, ymm2, ymm1
|
||||
vmovdqa ymmword ptr [rsp+0x220], ymm2
|
||||
vpxor ymm1, ymm1, ymmword ptr [CMP_MSB_MASK+rip]
|
||||
vpxor ymm2, ymm2, ymmword ptr [CMP_MSB_MASK+rip]
|
||||
vpcmpgtd ymm2, ymm1, ymm2
|
||||
shr r8, 32
|
||||
vmovd xmm3, r8d
|
||||
vpbroadcastd ymm3, xmm3
|
||||
vpsubd ymm3, ymm3, ymm2
|
||||
vmovdqa ymmword ptr [rsp+0x240], ymm3
|
||||
shl rdx, 6
|
||||
mov qword ptr [rsp+0x2C0], rdx
|
||||
cmp rsi, 8
|
||||
jc 3f
|
||||
2:
|
||||
vpbroadcastd ymm0, dword ptr [rcx]
|
||||
vpbroadcastd ymm1, dword ptr [rcx+0x4]
|
||||
vpbroadcastd ymm2, dword ptr [rcx+0x8]
|
||||
vpbroadcastd ymm3, dword ptr [rcx+0xC]
|
||||
vpbroadcastd ymm4, dword ptr [rcx+0x10]
|
||||
vpbroadcastd ymm5, dword ptr [rcx+0x14]
|
||||
vpbroadcastd ymm6, dword ptr [rcx+0x18]
|
||||
vpbroadcastd ymm7, dword ptr [rcx+0x1C]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov r12, qword ptr [rdi+0x20]
|
||||
mov r13, qword ptr [rdi+0x28]
|
||||
mov r14, qword ptr [rdi+0x30]
|
||||
mov r15, qword ptr [rdi+0x38]
|
||||
movzx eax, byte ptr [rbp+0x78]
|
||||
movzx ebx, byte ptr [rbp+0x80]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
1:
|
||||
movzx ebx, byte ptr [rbp+0x88]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+0x2C0]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+0x200], eax
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x40], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x40]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x40], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x40]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x40], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x40]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x40], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0x20], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0x40], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0x60], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x30], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x30]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x30], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x30]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x30], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x30]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x30], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+0x80], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0xA0], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0xC0], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0xE0], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x20], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x20]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x20], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x20]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x20], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x20]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x20], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+0x100], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0x120], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0x140], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0x160], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x10], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x10]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x10], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x10]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x10], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x10]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x10], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+0x180], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0x1A0], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0x1C0], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0x1E0], ymm11
|
||||
vpbroadcastd ymm15, dword ptr [rsp+0x200]
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r12+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r13+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r14+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
prefetcht0 [r15+rdx+0x80]
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm0, ymmword ptr [rsp+0x220]
|
||||
vpxor ymm13, ymm1, ymmword ptr [rsp+0x240]
|
||||
vpxor ymm14, ymm2, ymmword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vpxor ymm15, ymm3, ymm15
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [BLAKE3_IV_0+rip]
|
||||
vpaddd ymm9, ymm13, ymmword ptr [BLAKE3_IV_1+rip]
|
||||
vpaddd ymm10, ymm14, ymmword ptr [BLAKE3_IV_2+rip]
|
||||
vpaddd ymm11, ymm15, ymmword ptr [BLAKE3_IV_3+rip]
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x160]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0xA0]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x20]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x100]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1E0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x120]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xC0]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x1C0]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x40]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x60]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0xE0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x200], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0x140]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0x180]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0x80]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0x1A0]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+0x200]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vpxor ymm0, ymm0, ymm8
|
||||
vpxor ymm1, ymm1, ymm9
|
||||
vpxor ymm2, ymm2, ymm10
|
||||
vpxor ymm3, ymm3, ymm11
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpxor ymm4, ymm4, ymm12
|
||||
vpxor ymm5, ymm5, ymm13
|
||||
vpxor ymm6, ymm6, ymm14
|
||||
vpxor ymm7, ymm7, ymm15
|
||||
movzx eax, byte ptr [rbp+0x78]
|
||||
jne 1b
|
||||
mov rbx, qword ptr [rbp+0x90]
|
||||
vunpcklps ymm8, ymm0, ymm1
|
||||
vunpcklps ymm9, ymm2, ymm3
|
||||
vunpckhps ymm10, ymm0, ymm1
|
||||
vunpcklps ymm11, ymm4, ymm5
|
||||
vunpcklps ymm0, ymm6, ymm7
|
||||
vshufps ymm12, ymm8, ymm9, 78
|
||||
vblendps ymm1, ymm8, ymm12, 0xCC
|
||||
vshufps ymm8, ymm11, ymm0, 78
|
||||
vunpckhps ymm13, ymm2, ymm3
|
||||
vblendps ymm2, ymm11, ymm8, 0xCC
|
||||
vblendps ymm3, ymm12, ymm9, 0xCC
|
||||
vperm2f128 ymm12, ymm1, ymm2, 0x20
|
||||
vmovups ymmword ptr [rbx], ymm12
|
||||
vunpckhps ymm14, ymm4, ymm5
|
||||
vblendps ymm4, ymm8, ymm0, 0xCC
|
||||
vunpckhps ymm15, ymm6, ymm7
|
||||
vperm2f128 ymm7, ymm3, ymm4, 0x20
|
||||
vmovups ymmword ptr [rbx+0x20], ymm7
|
||||
vshufps ymm5, ymm10, ymm13, 78
|
||||
vblendps ymm6, ymm5, ymm13, 0xCC
|
||||
vshufps ymm13, ymm14, ymm15, 78
|
||||
vblendps ymm10, ymm10, ymm5, 0xCC
|
||||
vblendps ymm14, ymm14, ymm13, 0xCC
|
||||
vperm2f128 ymm8, ymm10, ymm14, 0x20
|
||||
vmovups ymmword ptr [rbx+0x40], ymm8
|
||||
vblendps ymm15, ymm13, ymm15, 0xCC
|
||||
vperm2f128 ymm13, ymm6, ymm15, 0x20
|
||||
vmovups ymmword ptr [rbx+0x60], ymm13
|
||||
vperm2f128 ymm9, ymm1, ymm2, 0x31
|
||||
vperm2f128 ymm11, ymm3, ymm4, 0x31
|
||||
vmovups ymmword ptr [rbx+0x80], ymm9
|
||||
vperm2f128 ymm14, ymm10, ymm14, 0x31
|
||||
vperm2f128 ymm15, ymm6, ymm15, 0x31
|
||||
vmovups ymmword ptr [rbx+0xA0], ymm11
|
||||
vmovups ymmword ptr [rbx+0xC0], ymm14
|
||||
vmovups ymmword ptr [rbx+0xE0], ymm15
|
||||
vmovdqa ymm0, ymmword ptr [rsp+0x2A0]
|
||||
vpaddd ymm1, ymm0, ymmword ptr [rsp+0x220]
|
||||
vmovdqa ymmword ptr [rsp+0x220], ymm1
|
||||
vpxor ymm0, ymm0, ymmword ptr [CMP_MSB_MASK+rip]
|
||||
vpxor ymm2, ymm1, ymmword ptr [CMP_MSB_MASK+rip]
|
||||
vpcmpgtd ymm2, ymm0, ymm2
|
||||
vmovdqa ymm0, ymmword ptr [rsp+0x240]
|
||||
vpsubd ymm2, ymm0, ymm2
|
||||
vmovdqa ymmword ptr [rsp+0x240], ymm2
|
||||
add rdi, 64
|
||||
add rbx, 256
|
||||
mov qword ptr [rbp+0x90], rbx
|
||||
sub rsi, 8
|
||||
cmp rsi, 8
|
||||
jnc 2b
|
||||
test rsi, rsi
|
||||
jnz 3f
|
||||
4:
|
||||
vzeroupper
|
||||
vmovdqa xmm6, xmmword ptr [rsp+0x2D0]
|
||||
vmovdqa xmm7, xmmword ptr [rsp+0x2E0]
|
||||
vmovdqa xmm8, xmmword ptr [rsp+0x2F0]
|
||||
vmovdqa xmm9, xmmword ptr [rsp+0x300]
|
||||
vmovdqa xmm10, xmmword ptr [rsp+0x310]
|
||||
vmovdqa xmm11, xmmword ptr [rsp+0x320]
|
||||
vmovdqa xmm12, xmmword ptr [rsp+0x330]
|
||||
vmovdqa xmm13, xmmword ptr [rsp+0x340]
|
||||
vmovdqa xmm14, xmmword ptr [rsp+0x350]
|
||||
vmovdqa xmm15, xmmword ptr [rsp+0x360]
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop rdi
|
||||
pop rsi
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
.p2align 5
|
||||
3:
|
||||
mov rbx, qword ptr [rbp+0x90]
|
||||
mov r15, qword ptr [rsp+0x2C0]
|
||||
movzx r13d, byte ptr [rbp+0x78]
|
||||
movzx r12d, byte ptr [rbp+0x88]
|
||||
test rsi, 0x4
|
||||
je 3f
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+0x10]
|
||||
vmovdqa ymm8, ymm0
|
||||
vmovdqa ymm9, ymm1
|
||||
vbroadcasti128 ymm12, xmmword ptr [rsp+0x220]
|
||||
vbroadcasti128 ymm13, xmmword ptr [rsp+0x240]
|
||||
vpunpckldq ymm14, ymm12, ymm13
|
||||
vpunpckhdq ymm15, ymm12, ymm13
|
||||
vpermq ymm14, ymm14, 0x50
|
||||
vpermq ymm15, ymm15, 0x50
|
||||
vbroadcasti128 ymm12, xmmword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vpblendd ymm14, ymm14, ymm12, 0x44
|
||||
vpblendd ymm15, ymm15, ymm12, 0x44
|
||||
vmovdqa ymmword ptr [rsp], ymm14
|
||||
vmovdqa ymmword ptr [rsp+0x20], ymm15
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+0x200], eax
|
||||
vmovups ymm2, ymmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm2, ymm2, xmmword ptr [r9+rdx-0x40], 0x01
|
||||
vmovups ymm3, ymmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm3, ymm3, xmmword ptr [r9+rdx-0x30], 0x01
|
||||
vshufps ymm4, ymm2, ymm3, 136
|
||||
vshufps ymm5, ymm2, ymm3, 221
|
||||
vmovups ymm2, ymmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm2, ymm2, xmmword ptr [r9+rdx-0x20], 0x01
|
||||
vmovups ymm3, ymmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm3, ymm3, xmmword ptr [r9+rdx-0x10], 0x01
|
||||
vshufps ymm6, ymm2, ymm3, 136
|
||||
vshufps ymm7, ymm2, ymm3, 221
|
||||
vpshufd ymm6, ymm6, 0x93
|
||||
vpshufd ymm7, ymm7, 0x93
|
||||
vmovups ymm10, ymmword ptr [r10+rdx-0x40]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r11+rdx-0x40], 0x01
|
||||
vmovups ymm11, ymmword ptr [r10+rdx-0x30]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r11+rdx-0x30], 0x01
|
||||
vshufps ymm12, ymm10, ymm11, 136
|
||||
vshufps ymm13, ymm10, ymm11, 221
|
||||
vmovups ymm10, ymmword ptr [r10+rdx-0x20]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r11+rdx-0x20], 0x01
|
||||
vmovups ymm11, ymmword ptr [r10+rdx-0x10]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r11+rdx-0x10], 0x01
|
||||
vshufps ymm14, ymm10, ymm11, 136
|
||||
vshufps ymm15, ymm10, ymm11, 221
|
||||
vpshufd ymm14, ymm14, 0x93
|
||||
vpshufd ymm15, ymm15, 0x93
|
||||
vpbroadcastd ymm2, dword ptr [rsp+0x200]
|
||||
vmovdqa ymm3, ymmword ptr [rsp]
|
||||
vmovdqa ymm11, ymmword ptr [rsp+0x20]
|
||||
vpblendd ymm3, ymm3, ymm2, 0x88
|
||||
vpblendd ymm11, ymm11, ymm2, 0x88
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vmovdqa ymm10, ymm2
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vmovdqa ymmword ptr [rsp+0x40], ymm4
|
||||
nop
|
||||
vmovdqa ymmword ptr [rsp+0x60], ymm12
|
||||
nop
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 12
|
||||
vpslld ymm9, ymm9, 20
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vmovdqa ymmword ptr [rsp+0x80], ymm5
|
||||
vmovdqa ymmword ptr [rsp+0xA0], ymm13
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 7
|
||||
vpslld ymm9, ymm9, 25
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpshufd ymm0, ymm0, 0x93
|
||||
vpshufd ymm8, ymm8, 0x93
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm11, ymm11, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x39
|
||||
vpshufd ymm10, ymm10, 0x39
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm8, ymm8, ymm14
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT16+rip]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 12
|
||||
vpslld ymm9, ymm9, 20
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm8, ymm8, ymm15
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT8+rip]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 7
|
||||
vpslld ymm9, ymm9, 25
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpshufd ymm0, ymm0, 0x39
|
||||
vpshufd ymm8, ymm8, 0x39
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm11, ymm11, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x93
|
||||
vpshufd ymm10, ymm10, 0x93
|
||||
dec al
|
||||
je 1f
|
||||
vmovdqa ymm4, ymmword ptr [rsp+0x40]
|
||||
vmovdqa ymm5, ymmword ptr [rsp+0x80]
|
||||
vshufps ymm12, ymm4, ymm5, 214
|
||||
vpshufd ymm13, ymm4, 0x0F
|
||||
vpshufd ymm4, ymm12, 0x39
|
||||
vshufps ymm12, ymm6, ymm7, 250
|
||||
vpblendd ymm13, ymm13, ymm12, 0xAA
|
||||
vpunpcklqdq ymm12, ymm7, ymm5
|
||||
vpblendd ymm12, ymm12, ymm6, 0x88
|
||||
vpshufd ymm12, ymm12, 0x78
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 0x1E
|
||||
vmovdqa ymmword ptr [rsp+0x40], ymm13
|
||||
vmovdqa ymmword ptr [rsp+0x80], ymm12
|
||||
vmovdqa ymm12, ymmword ptr [rsp+0x60]
|
||||
vmovdqa ymm13, ymmword ptr [rsp+0xA0]
|
||||
vshufps ymm5, ymm12, ymm13, 214
|
||||
vpshufd ymm6, ymm12, 0x0F
|
||||
vpshufd ymm12, ymm5, 0x39
|
||||
vshufps ymm5, ymm14, ymm15, 250
|
||||
vpblendd ymm6, ymm6, ymm5, 0xAA
|
||||
vpunpcklqdq ymm5, ymm15, ymm13
|
||||
vpblendd ymm5, ymm5, ymm14, 0x88
|
||||
vpshufd ymm5, ymm5, 0x78
|
||||
vpunpckhdq ymm13, ymm13, ymm15
|
||||
vpunpckldq ymm14, ymm14, ymm13
|
||||
vpshufd ymm15, ymm14, 0x1E
|
||||
vmovdqa ymm13, ymm6
|
||||
vmovdqa ymm14, ymm5
|
||||
vmovdqa ymm5, ymmword ptr [rsp+0x40]
|
||||
vmovdqa ymm6, ymmword ptr [rsp+0x80]
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
vpxor ymm8, ymm8, ymm10
|
||||
vpxor ymm9, ymm9, ymm11
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
vextracti128 xmmword ptr [rbx+0x20], ymm0, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x30], ymm1, 0x01
|
||||
vmovdqu xmmword ptr [rbx+0x40], xmm8
|
||||
vmovdqu xmmword ptr [rbx+0x50], xmm9
|
||||
vextracti128 xmmword ptr [rbx+0x60], ymm8, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x70], ymm9, 0x01
|
||||
vmovaps xmm8, xmmword ptr [rsp+0x260]
|
||||
vmovaps xmm0, xmmword ptr [rsp+0x220]
|
||||
vmovaps xmm1, xmmword ptr [rsp+0x230]
|
||||
vmovaps xmm2, xmmword ptr [rsp+0x240]
|
||||
vmovaps xmm3, xmmword ptr [rsp+0x250]
|
||||
vblendvps xmm0, xmm0, xmm1, xmm8
|
||||
vblendvps xmm2, xmm2, xmm3, xmm8
|
||||
vmovaps xmmword ptr [rsp+0x220], xmm0
|
||||
vmovaps xmmword ptr [rsp+0x240], xmm2
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
3:
|
||||
test rsi, 0x2
|
||||
je 3f
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+0x10]
|
||||
vmovd xmm13, dword ptr [rsp+0x220]
|
||||
vpinsrd xmm13, xmm13, dword ptr [rsp+0x240], 1
|
||||
vpinsrd xmm13, xmm13, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vmovd xmm14, dword ptr [rsp+0x224]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+0x244], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vinserti128 ymm13, ymm13, xmm14, 0x01
|
||||
vbroadcasti128 ymm14, xmmword ptr [ROT16+rip]
|
||||
vbroadcasti128 ymm15, xmmword ptr [ROT8+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+0x200], eax
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vpbroadcastd ymm8, dword ptr [rsp+0x200]
|
||||
vpblendd ymm3, ymm13, ymm8, 0x88
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-0x40], 0x01
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-0x30], 0x01
|
||||
vshufps ymm4, ymm8, ymm9, 136
|
||||
vshufps ymm5, ymm8, ymm9, 221
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-0x20], 0x01
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-0x10], 0x01
|
||||
vshufps ymm6, ymm8, ymm9, 136
|
||||
vshufps ymm7, ymm8, ymm9, 221
|
||||
vpshufd ymm6, ymm6, 0x93
|
||||
vpshufd ymm7, ymm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm14
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm15
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpshufd ymm0, ymm0, 0x93
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x39
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm14
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm15
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpshufd ymm0, ymm0, 0x39
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps ymm8, ymm4, ymm5, 214
|
||||
vpshufd ymm9, ymm4, 0x0F
|
||||
vpshufd ymm4, ymm8, 0x39
|
||||
vshufps ymm8, ymm6, ymm7, 250
|
||||
vpblendd ymm9, ymm9, ymm8, 0xAA
|
||||
vpunpcklqdq ymm8, ymm7, ymm5
|
||||
vpblendd ymm8, ymm8, ymm6, 0x88
|
||||
vpshufd ymm8, ymm8, 0x78
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 0x1E
|
||||
vmovdqa ymm5, ymm9
|
||||
vmovdqa ymm6, ymm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
vextracti128 xmmword ptr [rbx+0x20], ymm0, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x30], ymm1, 0x01
|
||||
vmovaps ymm8, ymmword ptr [rsp+0x260]
|
||||
vmovaps ymm0, ymmword ptr [rsp+0x220]
|
||||
vmovups ymm1, ymmword ptr [rsp+0x228]
|
||||
vmovaps ymm2, ymmword ptr [rsp+0x240]
|
||||
vmovups ymm3, ymmword ptr [rsp+0x248]
|
||||
vblendvps ymm0, ymm0, ymm1, ymm8
|
||||
vblendvps ymm2, ymm2, ymm3, ymm8
|
||||
vmovaps ymmword ptr [rsp+0x220], ymm0
|
||||
vmovaps ymmword ptr [rsp+0x240], ymm2
|
||||
add rbx, 64
|
||||
add rdi, 16
|
||||
sub rsi, 2
|
||||
3:
|
||||
test rsi, 0x1
|
||||
je 4b
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+0x10]
|
||||
vmovd xmm3, dword ptr [rsp+0x220]
|
||||
vpinsrd xmm3, xmm3, dword ptr [rsp+0x240], 1
|
||||
vpinsrd xmm13, xmm3, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vmovdqa xmm14, xmmword ptr [ROT16+rip]
|
||||
vmovdqa xmm15, xmmword ptr [ROT8+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
vmovdqa xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vmovdqa xmm3, xmm13
|
||||
vpinsrd xmm3, xmm3, eax, 3
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-0x30]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-0x10]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 0x93
|
||||
vpshufd xmm7, xmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm14
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 12
|
||||
vpslld xmm1, xmm1, 20
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm15
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 7
|
||||
vpslld xmm1, xmm1, 25
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpshufd xmm0, xmm0, 0x93
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x39
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm14
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 12
|
||||
vpslld xmm1, xmm1, 20
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm15
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 7
|
||||
vpslld xmm1, xmm1, 25
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpshufd xmm0, xmm0, 0x39
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0x0F
|
||||
vpshufd xmm4, xmm8, 0x39
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0xAA
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 0x88
|
||||
vpshufd xmm8, xmm8, 0x78
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 0x1E
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
jmp 4b
|
||||
|
||||
.section .rodata
|
||||
.p2align 6
|
||||
ADD0:
|
||||
.long 0, 1, 2, 3, 4, 5, 6, 7
|
||||
ADD1:
|
||||
.long 8, 8, 8, 8, 8, 8, 8, 8
|
||||
BLAKE3_IV_0:
|
||||
.long 0x6A09E667, 0x6A09E667, 0x6A09E667, 0x6A09E667
|
||||
.long 0x6A09E667, 0x6A09E667, 0x6A09E667, 0x6A09E667
|
||||
BLAKE3_IV_1:
|
||||
.long 0xBB67AE85, 0xBB67AE85, 0xBB67AE85, 0xBB67AE85
|
||||
.long 0xBB67AE85, 0xBB67AE85, 0xBB67AE85, 0xBB67AE85
|
||||
BLAKE3_IV_2:
|
||||
.long 0x3C6EF372, 0x3C6EF372, 0x3C6EF372, 0x3C6EF372
|
||||
.long 0x3C6EF372, 0x3C6EF372, 0x3C6EF372, 0x3C6EF372
|
||||
BLAKE3_IV_3:
|
||||
.long 0xA54FF53A, 0xA54FF53A, 0xA54FF53A, 0xA54FF53A
|
||||
.long 0xA54FF53A, 0xA54FF53A, 0xA54FF53A, 0xA54FF53A
|
||||
BLAKE3_BLOCK_LEN:
|
||||
.long 0x00000040, 0x00000040, 0x00000040, 0x00000040
|
||||
.long 0x00000040, 0x00000040, 0x00000040, 0x00000040
|
||||
ROT16:
|
||||
.byte 2, 3, 0, 1, 6, 7, 4, 5, 10, 11, 8, 9, 14, 15, 12, 13
|
||||
ROT8:
|
||||
.byte 1, 2, 3, 0, 5, 6, 7, 4, 9, 10, 11, 8, 13, 14, 15, 12
|
||||
CMP_MSB_MASK:
|
||||
.long 0x80000000, 0x80000000, 0x80000000, 0x80000000
|
||||
.long 0x80000000, 0x80000000, 0x80000000, 0x80000000
|
||||
BLAKE3_IV:
|
||||
.long 0x6A09E667, 0xBB67AE85, 0x3C6EF372, 0xA54FF53A
|
||||
|
|
@ -0,0 +1,1828 @@
|
|||
public _blake3_hash_many_avx2
|
||||
public blake3_hash_many_avx2
|
||||
|
||||
_TEXT SEGMENT ALIGN(16) 'CODE'
|
||||
|
||||
ALIGN 16
|
||||
blake3_hash_many_avx2 PROC
|
||||
_blake3_hash_many_avx2 PROC
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rsi
|
||||
push rdi
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 880
|
||||
and rsp, 0FFFFFFFFFFFFFFC0H
|
||||
vmovdqa xmmword ptr [rsp+2D0H], xmm6
|
||||
vmovdqa xmmword ptr [rsp+2E0H], xmm7
|
||||
vmovdqa xmmword ptr [rsp+2F0H], xmm8
|
||||
vmovdqa xmmword ptr [rsp+300H], xmm9
|
||||
vmovdqa xmmword ptr [rsp+310H], xmm10
|
||||
vmovdqa xmmword ptr [rsp+320H], xmm11
|
||||
vmovdqa xmmword ptr [rsp+330H], xmm12
|
||||
vmovdqa xmmword ptr [rsp+340H], xmm13
|
||||
vmovdqa xmmword ptr [rsp+350H], xmm14
|
||||
vmovdqa xmmword ptr [rsp+360H], xmm15
|
||||
mov rdi, rcx
|
||||
mov rsi, rdx
|
||||
mov rdx, r8
|
||||
mov rcx, r9
|
||||
mov r8, qword ptr [rbp+68H]
|
||||
movzx r9, byte ptr [rbp+70H]
|
||||
neg r9d
|
||||
vmovd xmm0, r9d
|
||||
vpbroadcastd ymm0, xmm0
|
||||
vmovdqa ymmword ptr [rsp+260H], ymm0
|
||||
vpand ymm1, ymm0, ymmword ptr [ADD0]
|
||||
vpand ymm2, ymm0, ymmword ptr [ADD1]
|
||||
vmovdqa ymmword ptr [rsp+2A0H], ymm2
|
||||
vmovd xmm2, r8d
|
||||
vpbroadcastd ymm2, xmm2
|
||||
vpaddd ymm2, ymm2, ymm1
|
||||
vmovdqa ymmword ptr [rsp+220H], ymm2
|
||||
vpxor ymm1, ymm1, ymmword ptr [CMP_MSB_MASK]
|
||||
vpxor ymm2, ymm2, ymmword ptr [CMP_MSB_MASK]
|
||||
vpcmpgtd ymm2, ymm1, ymm2
|
||||
shr r8, 32
|
||||
vmovd xmm3, r8d
|
||||
vpbroadcastd ymm3, xmm3
|
||||
vpsubd ymm3, ymm3, ymm2
|
||||
vmovdqa ymmword ptr [rsp+240H], ymm3
|
||||
shl rdx, 6
|
||||
mov qword ptr [rsp+2C0H], rdx
|
||||
cmp rsi, 8
|
||||
jc final7blocks
|
||||
outerloop8:
|
||||
vpbroadcastd ymm0, dword ptr [rcx]
|
||||
vpbroadcastd ymm1, dword ptr [rcx+4H]
|
||||
vpbroadcastd ymm2, dword ptr [rcx+8H]
|
||||
vpbroadcastd ymm3, dword ptr [rcx+0CH]
|
||||
vpbroadcastd ymm4, dword ptr [rcx+10H]
|
||||
vpbroadcastd ymm5, dword ptr [rcx+14H]
|
||||
vpbroadcastd ymm6, dword ptr [rcx+18H]
|
||||
vpbroadcastd ymm7, dword ptr [rcx+1CH]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
mov r10, qword ptr [rdi+10H]
|
||||
mov r11, qword ptr [rdi+18H]
|
||||
mov r12, qword ptr [rdi+20H]
|
||||
mov r13, qword ptr [rdi+28H]
|
||||
mov r14, qword ptr [rdi+30H]
|
||||
mov r15, qword ptr [rdi+38H]
|
||||
movzx eax, byte ptr [rbp+78H]
|
||||
movzx ebx, byte ptr [rbp+80H]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
ALIGN 16
|
||||
innerloop8:
|
||||
movzx ebx, byte ptr [rbp+88H]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+2C0H]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+200H], eax
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-40H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-40H], 01H
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-40H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-40H], 01H
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-40H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-40H], 01H
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-40H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-40H], 01H
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+20H], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+40H], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+60H], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-30H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-30H], 01H
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-30H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-30H], 01H
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-30H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-30H], 01H
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-30H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-30H], 01H
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+80H], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+0A0H], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+0C0H], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+0E0H], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-20H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-20H], 01H
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-20H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-20H], 01H
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-20H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-20H], 01H
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-20H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-20H], 01H
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+100H], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+120H], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+140H], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+160H], ymm11
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-10H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-10H], 01H
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-10H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-10H], 01H
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-10H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-10H], 01H
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-10H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-10H], 01H
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm8, ymm12, ymm14, 136
|
||||
vmovaps ymmword ptr [rsp+180H], ymm8
|
||||
vshufps ymm9, ymm12, ymm14, 221
|
||||
vmovaps ymmword ptr [rsp+1A0H], ymm9
|
||||
vshufps ymm10, ymm13, ymm15, 136
|
||||
vmovaps ymmword ptr [rsp+1C0H], ymm10
|
||||
vshufps ymm11, ymm13, ymm15, 221
|
||||
vmovaps ymmword ptr [rsp+1E0H], ymm11
|
||||
vpbroadcastd ymm15, dword ptr [rsp+200H]
|
||||
prefetcht0 byte ptr [r8+rdx+80H]
|
||||
prefetcht0 byte ptr [r12+rdx+80H]
|
||||
prefetcht0 byte ptr [r9+rdx+80H]
|
||||
prefetcht0 byte ptr [r13+rdx+80H]
|
||||
prefetcht0 byte ptr [r10+rdx+80H]
|
||||
prefetcht0 byte ptr [r14+rdx+80H]
|
||||
prefetcht0 byte ptr [r11+rdx+80H]
|
||||
prefetcht0 byte ptr [r15+rdx+80H]
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+40H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+80H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0C0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm0, ymmword ptr [rsp+220H]
|
||||
vpxor ymm13, ymm1, ymmword ptr [rsp+240H]
|
||||
vpxor ymm14, ymm2, ymmword ptr [BLAKE3_BLOCK_LEN]
|
||||
vpxor ymm15, ymm3, ymm15
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [BLAKE3_IV_0]
|
||||
vpaddd ymm9, ymm13, ymmword ptr [BLAKE3_IV_1]
|
||||
vpaddd ymm10, ymm14, ymmword ptr [BLAKE3_IV_2]
|
||||
vpaddd ymm11, ymm15, ymmword ptr [BLAKE3_IV_3]
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+20H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+60H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0A0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0E0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+100H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+140H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+180H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1C0H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+120H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+160H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+1A0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1E0H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+40H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+60H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0E0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+80H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0C0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+140H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1A0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+20H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+180H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+120H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1E0H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+160H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0A0H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+1C0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+100H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+60H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+140H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+1A0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0E0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+80H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+180H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+40H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1C0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0C0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+120H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+160H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+100H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0A0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+1E0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+20H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+140H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+180H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+1C0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1A0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0E0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+120H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+60H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1E0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+80H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+160H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0A0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+20H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+40H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+100H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0C0H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+180H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+120H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+1E0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1C0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+1A0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+160H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+140H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+100H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+0E0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0A0H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0C0H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+40H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+60H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+20H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+80H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+120H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+160H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+100H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1E0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+1C0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0A0H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+180H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+20H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+1A0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+40H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+80H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+60H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+140H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+0C0H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0E0H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+160H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+0A0H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+20H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+100H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+1E0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+120H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0C0H]
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxor ymm12, ymm12, ymm0
|
||||
vpxor ymm13, ymm13, ymm1
|
||||
vpxor ymm14, ymm14, ymm2
|
||||
vpxor ymm15, ymm15, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpaddd ymm8, ymm12, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxor ymm4, ymm4, ymm8
|
||||
vpxor ymm5, ymm5, ymm9
|
||||
vpxor ymm6, ymm6, ymm10
|
||||
vpxor ymm7, ymm7, ymm11
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+1C0H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+40H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+60H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+0E0H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT16]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vmovdqa ymmword ptr [rsp+200H], ymm8
|
||||
vpsrld ymm8, ymm5, 12
|
||||
vpslld ymm5, ymm5, 20
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 12
|
||||
vpslld ymm6, ymm6, 20
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 12
|
||||
vpslld ymm7, ymm7, 20
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 12
|
||||
vpslld ymm4, ymm4, 20
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpaddd ymm0, ymm0, ymmword ptr [rsp+140H]
|
||||
vpaddd ymm1, ymm1, ymmword ptr [rsp+180H]
|
||||
vpaddd ymm2, ymm2, ymmword ptr [rsp+80H]
|
||||
vpaddd ymm3, ymm3, ymmword ptr [rsp+1A0H]
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxor ymm15, ymm15, ymm0
|
||||
vpxor ymm12, ymm12, ymm1
|
||||
vpxor ymm13, ymm13, ymm2
|
||||
vpxor ymm14, ymm14, ymm3
|
||||
vbroadcasti128 ymm8, xmmword ptr [ROT8]
|
||||
vpshufb ymm15, ymm15, ymm8
|
||||
vpshufb ymm12, ymm12, ymm8
|
||||
vpshufb ymm13, ymm13, ymm8
|
||||
vpshufb ymm14, ymm14, ymm8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm13, ymmword ptr [rsp+200H]
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxor ymm5, ymm5, ymm10
|
||||
vpxor ymm6, ymm6, ymm11
|
||||
vpxor ymm7, ymm7, ymm8
|
||||
vpxor ymm4, ymm4, ymm9
|
||||
vpxor ymm0, ymm0, ymm8
|
||||
vpxor ymm1, ymm1, ymm9
|
||||
vpxor ymm2, ymm2, ymm10
|
||||
vpxor ymm3, ymm3, ymm11
|
||||
vpsrld ymm8, ymm5, 7
|
||||
vpslld ymm5, ymm5, 25
|
||||
vpor ymm5, ymm5, ymm8
|
||||
vpsrld ymm8, ymm6, 7
|
||||
vpslld ymm6, ymm6, 25
|
||||
vpor ymm6, ymm6, ymm8
|
||||
vpsrld ymm8, ymm7, 7
|
||||
vpslld ymm7, ymm7, 25
|
||||
vpor ymm7, ymm7, ymm8
|
||||
vpsrld ymm8, ymm4, 7
|
||||
vpslld ymm4, ymm4, 25
|
||||
vpor ymm4, ymm4, ymm8
|
||||
vpxor ymm4, ymm4, ymm12
|
||||
vpxor ymm5, ymm5, ymm13
|
||||
vpxor ymm6, ymm6, ymm14
|
||||
vpxor ymm7, ymm7, ymm15
|
||||
movzx eax, byte ptr [rbp+78H]
|
||||
jne innerloop8
|
||||
mov rbx, qword ptr [rbp+90H]
|
||||
vunpcklps ymm8, ymm0, ymm1
|
||||
vunpcklps ymm9, ymm2, ymm3
|
||||
vunpckhps ymm10, ymm0, ymm1
|
||||
vunpcklps ymm11, ymm4, ymm5
|
||||
vunpcklps ymm0, ymm6, ymm7
|
||||
vshufps ymm12, ymm8, ymm9, 78
|
||||
vblendps ymm1, ymm8, ymm12, 0CCH
|
||||
vshufps ymm8, ymm11, ymm0, 78
|
||||
vunpckhps ymm13, ymm2, ymm3
|
||||
vblendps ymm2, ymm11, ymm8, 0CCH
|
||||
vblendps ymm3, ymm12, ymm9, 0CCH
|
||||
vperm2f128 ymm12, ymm1, ymm2, 20H
|
||||
vmovups ymmword ptr [rbx], ymm12
|
||||
vunpckhps ymm14, ymm4, ymm5
|
||||
vblendps ymm4, ymm8, ymm0, 0CCH
|
||||
vunpckhps ymm15, ymm6, ymm7
|
||||
vperm2f128 ymm7, ymm3, ymm4, 20H
|
||||
vmovups ymmword ptr [rbx+20H], ymm7
|
||||
vshufps ymm5, ymm10, ymm13, 78
|
||||
vblendps ymm6, ymm5, ymm13, 0CCH
|
||||
vshufps ymm13, ymm14, ymm15, 78
|
||||
vblendps ymm10, ymm10, ymm5, 0CCH
|
||||
vblendps ymm14, ymm14, ymm13, 0CCH
|
||||
vperm2f128 ymm8, ymm10, ymm14, 20H
|
||||
vmovups ymmword ptr [rbx+40H], ymm8
|
||||
vblendps ymm15, ymm13, ymm15, 0CCH
|
||||
vperm2f128 ymm13, ymm6, ymm15, 20H
|
||||
vmovups ymmword ptr [rbx+60H], ymm13
|
||||
vperm2f128 ymm9, ymm1, ymm2, 31H
|
||||
vperm2f128 ymm11, ymm3, ymm4, 31H
|
||||
vmovups ymmword ptr [rbx+80H], ymm9
|
||||
vperm2f128 ymm14, ymm10, ymm14, 31H
|
||||
vperm2f128 ymm15, ymm6, ymm15, 31H
|
||||
vmovups ymmword ptr [rbx+0A0H], ymm11
|
||||
vmovups ymmword ptr [rbx+0C0H], ymm14
|
||||
vmovups ymmword ptr [rbx+0E0H], ymm15
|
||||
vmovdqa ymm0, ymmword ptr [rsp+2A0H]
|
||||
vpaddd ymm1, ymm0, ymmword ptr [rsp+220H]
|
||||
vmovdqa ymmword ptr [rsp+220H], ymm1
|
||||
vpxor ymm0, ymm0, ymmword ptr [CMP_MSB_MASK]
|
||||
vpxor ymm2, ymm1, ymmword ptr [CMP_MSB_MASK]
|
||||
vpcmpgtd ymm2, ymm0, ymm2
|
||||
vmovdqa ymm0, ymmword ptr [rsp+240H]
|
||||
vpsubd ymm2, ymm0, ymm2
|
||||
vmovdqa ymmword ptr [rsp+240H], ymm2
|
||||
add rdi, 64
|
||||
add rbx, 256
|
||||
mov qword ptr [rbp+90H], rbx
|
||||
sub rsi, 8
|
||||
cmp rsi, 8
|
||||
jnc outerloop8
|
||||
test rsi, rsi
|
||||
jnz final7blocks
|
||||
unwind:
|
||||
vzeroupper
|
||||
vmovdqa xmm6, xmmword ptr [rsp+2D0H]
|
||||
vmovdqa xmm7, xmmword ptr [rsp+2E0H]
|
||||
vmovdqa xmm8, xmmword ptr [rsp+2F0H]
|
||||
vmovdqa xmm9, xmmword ptr [rsp+300H]
|
||||
vmovdqa xmm10, xmmword ptr [rsp+310H]
|
||||
vmovdqa xmm11, xmmword ptr [rsp+320H]
|
||||
vmovdqa xmm12, xmmword ptr [rsp+330H]
|
||||
vmovdqa xmm13, xmmword ptr [rsp+340H]
|
||||
vmovdqa xmm14, xmmword ptr [rsp+350H]
|
||||
vmovdqa xmm15, xmmword ptr [rsp+360H]
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop rdi
|
||||
pop rsi
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
ALIGN 16
|
||||
final7blocks:
|
||||
mov rbx, qword ptr [rbp+90H]
|
||||
mov r15, qword ptr [rsp+2C0H]
|
||||
movzx r13d, byte ptr [rbp+78H]
|
||||
movzx r12d, byte ptr [rbp+88H]
|
||||
test rsi, 4H
|
||||
je final3blocks
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+10H]
|
||||
vmovdqa ymm8, ymm0
|
||||
vmovdqa ymm9, ymm1
|
||||
vbroadcasti128 ymm12, xmmword ptr [rsp+220H]
|
||||
vbroadcasti128 ymm13, xmmword ptr [rsp+240H]
|
||||
vpunpckldq ymm14, ymm12, ymm13
|
||||
vpunpckhdq ymm15, ymm12, ymm13
|
||||
vpermq ymm14, ymm14, 50H
|
||||
vpermq ymm15, ymm15, 50H
|
||||
vbroadcasti128 ymm12, xmmword ptr [BLAKE3_BLOCK_LEN]
|
||||
vpblendd ymm14, ymm14, ymm12, 44H
|
||||
vpblendd ymm15, ymm15, ymm12, 44H
|
||||
vmovdqa ymmword ptr [rsp], ymm14
|
||||
vmovdqa ymmword ptr [rsp+20H], ymm15
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
mov r10, qword ptr [rdi+10H]
|
||||
mov r11, qword ptr [rdi+18H]
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
ALIGN 16
|
||||
innerloop4:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+200H], eax
|
||||
vmovups ymm2, ymmword ptr [r8+rdx-40H]
|
||||
vinsertf128 ymm2, ymm2, xmmword ptr [r9+rdx-40H], 01H
|
||||
vmovups ymm3, ymmword ptr [r8+rdx-30H]
|
||||
vinsertf128 ymm3, ymm3, xmmword ptr [r9+rdx-30H], 01H
|
||||
vshufps ymm4, ymm2, ymm3, 136
|
||||
vshufps ymm5, ymm2, ymm3, 221
|
||||
vmovups ymm2, ymmword ptr [r8+rdx-20H]
|
||||
vinsertf128 ymm2, ymm2, xmmword ptr [r9+rdx-20H], 01H
|
||||
vmovups ymm3, ymmword ptr [r8+rdx-10H]
|
||||
vinsertf128 ymm3, ymm3, xmmword ptr [r9+rdx-10H], 01H
|
||||
vshufps ymm6, ymm2, ymm3, 136
|
||||
vshufps ymm7, ymm2, ymm3, 221
|
||||
vpshufd ymm6, ymm6, 93H
|
||||
vpshufd ymm7, ymm7, 93H
|
||||
vmovups ymm10, ymmword ptr [r10+rdx-40H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r11+rdx-40H], 01H
|
||||
vmovups ymm11, ymmword ptr [r10+rdx-30H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r11+rdx-30H], 01H
|
||||
vshufps ymm12, ymm10, ymm11, 136
|
||||
vshufps ymm13, ymm10, ymm11, 221
|
||||
vmovups ymm10, ymmword ptr [r10+rdx-20H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r11+rdx-20H], 01H
|
||||
vmovups ymm11, ymmword ptr [r10+rdx-10H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r11+rdx-10H], 01H
|
||||
vshufps ymm14, ymm10, ymm11, 136
|
||||
vshufps ymm15, ymm10, ymm11, 221
|
||||
vpshufd ymm14, ymm14, 93H
|
||||
vpshufd ymm15, ymm15, 93H
|
||||
vpbroadcastd ymm2, dword ptr [rsp+200H]
|
||||
vmovdqa ymm3, ymmword ptr [rsp]
|
||||
vmovdqa ymm11, ymmword ptr [rsp+20H]
|
||||
vpblendd ymm3, ymm3, ymm2, 88H
|
||||
vpblendd ymm11, ymm11, ymm2, 88H
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV]
|
||||
vmovdqa ymm10, ymm2
|
||||
mov al, 7
|
||||
roundloop4:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vmovdqa ymmword ptr [rsp+40H], ymm4
|
||||
nop
|
||||
vmovdqa ymmword ptr [rsp+60H], ymm12
|
||||
nop
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT16]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 12
|
||||
vpslld ymm9, ymm9, 20
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vmovdqa ymmword ptr [rsp+80H], ymm5
|
||||
vmovdqa ymmword ptr [rsp+0A0H], ymm13
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT8]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 7
|
||||
vpslld ymm9, ymm9, 25
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpshufd ymm0, ymm0, 93H
|
||||
vpshufd ymm8, ymm8, 93H
|
||||
vpshufd ymm3, ymm3, 4EH
|
||||
vpshufd ymm11, ymm11, 4EH
|
||||
vpshufd ymm2, ymm2, 39H
|
||||
vpshufd ymm10, ymm10, 39H
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm8, ymm8, ymm14
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT16]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 12
|
||||
vpslld ymm9, ymm9, 20
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm8, ymm8, ymm15
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpaddd ymm8, ymm8, ymm9
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpxor ymm11, ymm11, ymm8
|
||||
vbroadcasti128 ymm4, xmmword ptr [ROT8]
|
||||
vpshufb ymm3, ymm3, ymm4
|
||||
vpshufb ymm11, ymm11, ymm4
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpaddd ymm10, ymm10, ymm11
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpxor ymm9, ymm9, ymm10
|
||||
vpsrld ymm4, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm4
|
||||
vpsrld ymm4, ymm9, 7
|
||||
vpslld ymm9, ymm9, 25
|
||||
vpor ymm9, ymm9, ymm4
|
||||
vpshufd ymm0, ymm0, 39H
|
||||
vpshufd ymm8, ymm8, 39H
|
||||
vpshufd ymm3, ymm3, 4EH
|
||||
vpshufd ymm11, ymm11, 4EH
|
||||
vpshufd ymm2, ymm2, 93H
|
||||
vpshufd ymm10, ymm10, 93H
|
||||
dec al
|
||||
je endroundloop4
|
||||
vmovdqa ymm4, ymmword ptr [rsp+40H]
|
||||
vmovdqa ymm5, ymmword ptr [rsp+80H]
|
||||
vshufps ymm12, ymm4, ymm5, 214
|
||||
vpshufd ymm13, ymm4, 0FH
|
||||
vpshufd ymm4, ymm12, 39H
|
||||
vshufps ymm12, ymm6, ymm7, 250
|
||||
vpblendd ymm13, ymm13, ymm12, 0AAH
|
||||
vpunpcklqdq ymm12, ymm7, ymm5
|
||||
vpblendd ymm12, ymm12, ymm6, 88H
|
||||
vpshufd ymm12, ymm12, 78H
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 1EH
|
||||
vmovdqa ymmword ptr [rsp+40H], ymm13
|
||||
vmovdqa ymmword ptr [rsp+80H], ymm12
|
||||
vmovdqa ymm12, ymmword ptr [rsp+60H]
|
||||
vmovdqa ymm13, ymmword ptr [rsp+0A0H]
|
||||
vshufps ymm5, ymm12, ymm13, 214
|
||||
vpshufd ymm6, ymm12, 0FH
|
||||
vpshufd ymm12, ymm5, 39H
|
||||
vshufps ymm5, ymm14, ymm15, 250
|
||||
vpblendd ymm6, ymm6, ymm5, 0AAH
|
||||
vpunpcklqdq ymm5, ymm15, ymm13
|
||||
vpblendd ymm5, ymm5, ymm14, 88H
|
||||
vpshufd ymm5, ymm5, 78H
|
||||
vpunpckhdq ymm13, ymm13, ymm15
|
||||
vpunpckldq ymm14, ymm14, ymm13
|
||||
vpshufd ymm15, ymm14, 1EH
|
||||
vmovdqa ymm13, ymm6
|
||||
vmovdqa ymm14, ymm5
|
||||
vmovdqa ymm5, ymmword ptr [rsp+40H]
|
||||
vmovdqa ymm6, ymmword ptr [rsp+80H]
|
||||
jmp roundloop4
|
||||
endroundloop4:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
vpxor ymm8, ymm8, ymm10
|
||||
vpxor ymm9, ymm9, ymm11
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne innerloop4
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+10H], xmm1
|
||||
vextracti128 xmmword ptr [rbx+20H], ymm0, 01H
|
||||
vextracti128 xmmword ptr [rbx+30H], ymm1, 01H
|
||||
vmovdqu xmmword ptr [rbx+40H], xmm8
|
||||
vmovdqu xmmword ptr [rbx+50H], xmm9
|
||||
vextracti128 xmmword ptr [rbx+60H], ymm8, 01H
|
||||
vextracti128 xmmword ptr [rbx+70H], ymm9, 01H
|
||||
vmovaps xmm8, xmmword ptr [rsp+260H]
|
||||
vmovaps xmm0, xmmword ptr [rsp+220H]
|
||||
vmovaps xmm1, xmmword ptr [rsp+230H]
|
||||
vmovaps xmm2, xmmword ptr [rsp+240H]
|
||||
vmovaps xmm3, xmmword ptr [rsp+250H]
|
||||
vblendvps xmm0, xmm0, xmm1, xmm8
|
||||
vblendvps xmm2, xmm2, xmm3, xmm8
|
||||
vmovaps xmmword ptr [rsp+220H], xmm0
|
||||
vmovaps xmmword ptr [rsp+240H], xmm2
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
final3blocks:
|
||||
test rsi, 2H
|
||||
je final1blocks
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+10H]
|
||||
vmovd xmm13, dword ptr [rsp+220H]
|
||||
vpinsrd xmm13, xmm13, dword ptr [rsp+240H], 1
|
||||
vpinsrd xmm13, xmm13, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
vmovd xmm14, dword ptr [rsp+224H]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+244H], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
vinserti128 ymm13, ymm13, xmm14, 01H
|
||||
vbroadcasti128 ymm14, xmmword ptr [ROT16]
|
||||
vbroadcasti128 ymm15, xmmword ptr [ROT8]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
ALIGN 16
|
||||
innerloop2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+200H], eax
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV]
|
||||
vpbroadcastd ymm8, dword ptr [rsp+200H]
|
||||
vpblendd ymm3, ymm13, ymm8, 88H
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-40H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-40H], 01H
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-30H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-30H], 01H
|
||||
vshufps ymm4, ymm8, ymm9, 136
|
||||
vshufps ymm5, ymm8, ymm9, 221
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-20H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-20H], 01H
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-10H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-10H], 01H
|
||||
vshufps ymm6, ymm8, ymm9, 136
|
||||
vshufps ymm7, ymm8, ymm9, 221
|
||||
vpshufd ymm6, ymm6, 93H
|
||||
vpshufd ymm7, ymm7, 93H
|
||||
mov al, 7
|
||||
roundloop2:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm14
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm15
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpshufd ymm0, ymm0, 93H
|
||||
vpshufd ymm3, ymm3, 4EH
|
||||
vpshufd ymm2, ymm2, 39H
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm14
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 12
|
||||
vpslld ymm1, ymm1, 20
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxor ymm3, ymm3, ymm0
|
||||
vpshufb ymm3, ymm3, ymm15
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxor ymm1, ymm1, ymm2
|
||||
vpsrld ymm8, ymm1, 7
|
||||
vpslld ymm1, ymm1, 25
|
||||
vpor ymm1, ymm1, ymm8
|
||||
vpshufd ymm0, ymm0, 39H
|
||||
vpshufd ymm3, ymm3, 4EH
|
||||
vpshufd ymm2, ymm2, 93H
|
||||
dec al
|
||||
jz endroundloop2
|
||||
vshufps ymm8, ymm4, ymm5, 214
|
||||
vpshufd ymm9, ymm4, 0FH
|
||||
vpshufd ymm4, ymm8, 39H
|
||||
vshufps ymm8, ymm6, ymm7, 250
|
||||
vpblendd ymm9, ymm9, ymm8, 0AAH
|
||||
vpunpcklqdq ymm8, ymm7, ymm5
|
||||
vpblendd ymm8, ymm8, ymm6, 88H
|
||||
vpshufd ymm8, ymm8, 78H
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 1EH
|
||||
vmovdqa ymm5, ymm9
|
||||
vmovdqa ymm6, ymm8
|
||||
jmp roundloop2
|
||||
endroundloop2:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne innerloop2
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+10H], xmm1
|
||||
vextracti128 xmmword ptr [rbx+20H], ymm0, 01H
|
||||
vextracti128 xmmword ptr [rbx+30H], ymm1, 01H
|
||||
vmovaps ymm8, ymmword ptr [rsp+260H]
|
||||
vmovaps ymm0, ymmword ptr [rsp+220H]
|
||||
vmovups ymm1, ymmword ptr [rsp+228H]
|
||||
vmovaps ymm2, ymmword ptr [rsp+240H]
|
||||
vmovups ymm3, ymmword ptr [rsp+248H]
|
||||
vblendvps ymm0, ymm0, ymm1, ymm8
|
||||
vblendvps ymm2, ymm2, ymm3, ymm8
|
||||
vmovaps ymmword ptr [rsp+220H], ymm0
|
||||
vmovaps ymmword ptr [rsp+240H], ymm2
|
||||
add rbx, 64
|
||||
add rdi, 16
|
||||
sub rsi, 2
|
||||
final1blocks:
|
||||
test rsi, 1H
|
||||
je unwind
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+10H]
|
||||
vmovd xmm3, dword ptr [rsp+220H]
|
||||
vpinsrd xmm3, xmm3, dword ptr [rsp+240H], 1
|
||||
vpinsrd xmm13, xmm3, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
vmovdqa xmm14, xmmword ptr [ROT16]
|
||||
vmovdqa xmm15, xmmword ptr [ROT8]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
ALIGN 16
|
||||
innerloop1:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
vmovdqa xmm2, xmmword ptr [BLAKE3_IV]
|
||||
vmovdqa xmm3, xmm13
|
||||
vpinsrd xmm3, xmm3, eax, 3
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-40H]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-30H]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-20H]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-10H]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 93H
|
||||
vpshufd xmm7, xmm7, 93H
|
||||
mov al, 7
|
||||
roundloop1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm14
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 12
|
||||
vpslld xmm1, xmm1, 20
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm15
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 7
|
||||
vpslld xmm1, xmm1, 25
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpshufd xmm0, xmm0, 93H
|
||||
vpshufd xmm3, xmm3, 4EH
|
||||
vpshufd xmm2, xmm2, 39H
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm14
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 12
|
||||
vpslld xmm1, xmm1, 20
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxor xmm3, xmm3, xmm0
|
||||
vpshufb xmm3, xmm3, xmm15
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxor xmm1, xmm1, xmm2
|
||||
vpsrld xmm8, xmm1, 7
|
||||
vpslld xmm1, xmm1, 25
|
||||
vpor xmm1, xmm1, xmm8
|
||||
vpshufd xmm0, xmm0, 39H
|
||||
vpshufd xmm3, xmm3, 4EH
|
||||
vpshufd xmm2, xmm2, 93H
|
||||
dec al
|
||||
jz endroundloop1
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0FH
|
||||
vpshufd xmm4, xmm8, 39H
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0AAH
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 88H
|
||||
vpshufd xmm8, xmm8, 78H
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 1EH
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp roundloop1
|
||||
endroundloop1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne innerloop1
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+10H], xmm1
|
||||
jmp unwind
|
||||
|
||||
_blake3_hash_many_avx2 ENDP
|
||||
blake3_hash_many_avx2 ENDP
|
||||
_TEXT ENDS
|
||||
|
||||
_RDATA SEGMENT READONLY PAGE ALIAS(".rdata") 'CONST'
|
||||
ALIGN 64
|
||||
ADD0:
|
||||
dd 0, 1, 2, 3, 4, 5, 6, 7
|
||||
|
||||
ADD1:
|
||||
dd 8 dup (8)
|
||||
|
||||
BLAKE3_IV_0:
|
||||
dd 8 dup (6A09E667H)
|
||||
|
||||
BLAKE3_IV_1:
|
||||
dd 8 dup (0BB67AE85H)
|
||||
|
||||
BLAKE3_IV_2:
|
||||
dd 8 dup (3C6EF372H)
|
||||
|
||||
BLAKE3_IV_3:
|
||||
dd 8 dup (0A54FF53AH)
|
||||
|
||||
BLAKE3_BLOCK_LEN:
|
||||
dd 8 dup (64)
|
||||
|
||||
ROT16:
|
||||
db 2, 3, 0, 1, 6, 7, 4, 5, 10, 11, 8, 9, 14, 15, 12, 13
|
||||
|
||||
ROT8:
|
||||
db 1, 2, 3, 0, 5, 6, 7, 4, 9, 10, 11, 8, 13, 14, 15, 12
|
||||
|
||||
CMP_MSB_MASK:
|
||||
dd 8 dup(80000000H)
|
||||
|
||||
BLAKE3_IV:
|
||||
dd 6A09E667H, 0BB67AE85H, 3C6EF372H, 0A54FF53AH
|
||||
|
||||
_RDATA ENDS
|
||||
END
|
|
@ -0,0 +1,2569 @@
|
|||
.intel_syntax noprefix
|
||||
|
||||
.global _blake3_hash_many_avx512
|
||||
.global blake3_hash_many_avx512
|
||||
.global blake3_compress_in_place_avx512
|
||||
.global _blake3_compress_in_place_avx512
|
||||
.global blake3_compress_xof_avx512
|
||||
.global _blake3_compress_xof_avx512
|
||||
|
||||
#ifdef __APPLE__
|
||||
.text
|
||||
#else
|
||||
.section .text
|
||||
#endif
|
||||
.p2align 6
|
||||
_blake3_hash_many_avx512:
|
||||
blake3_hash_many_avx512:
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 144
|
||||
and rsp, 0xFFFFFFFFFFFFFFC0
|
||||
neg r9
|
||||
kmovw k1, r9d
|
||||
vmovd xmm0, r8d
|
||||
vpbroadcastd ymm0, xmm0
|
||||
shr r8, 32
|
||||
vmovd xmm1, r8d
|
||||
vpbroadcastd ymm1, xmm1
|
||||
vmovdqa ymm4, ymm1
|
||||
vmovdqa ymm5, ymm1
|
||||
vpaddd ymm2, ymm0, ymmword ptr [ADD0+rip]
|
||||
vpaddd ymm3, ymm0, ymmword ptr [ADD0+32+rip]
|
||||
vpcmpltud k2, ymm2, ymm0
|
||||
vpcmpltud k3, ymm3, ymm0
|
||||
vpaddd ymm4 {k2}, ymm4, dword ptr [ADD1+rip] {1to8}
|
||||
vpaddd ymm5 {k3}, ymm5, dword ptr [ADD1+rip] {1to8}
|
||||
knotw k2, k1
|
||||
vmovdqa32 ymm2 {k2}, ymm0
|
||||
vmovdqa32 ymm3 {k2}, ymm0
|
||||
vmovdqa32 ymm4 {k2}, ymm1
|
||||
vmovdqa32 ymm5 {k2}, ymm1
|
||||
vmovdqa ymmword ptr [rsp], ymm2
|
||||
vmovdqa ymmword ptr [rsp+0x1*0x20], ymm3
|
||||
vmovdqa ymmword ptr [rsp+0x2*0x20], ymm4
|
||||
vmovdqa ymmword ptr [rsp+0x3*0x20], ymm5
|
||||
shl rdx, 6
|
||||
mov qword ptr [rsp+0x80], rdx
|
||||
cmp rsi, 16
|
||||
jc 3f
|
||||
2:
|
||||
vpbroadcastd zmm0, dword ptr [rcx]
|
||||
vpbroadcastd zmm1, dword ptr [rcx+0x1*0x4]
|
||||
vpbroadcastd zmm2, dword ptr [rcx+0x2*0x4]
|
||||
vpbroadcastd zmm3, dword ptr [rcx+0x3*0x4]
|
||||
vpbroadcastd zmm4, dword ptr [rcx+0x4*0x4]
|
||||
vpbroadcastd zmm5, dword ptr [rcx+0x5*0x4]
|
||||
vpbroadcastd zmm6, dword ptr [rcx+0x6*0x4]
|
||||
vpbroadcastd zmm7, dword ptr [rcx+0x7*0x4]
|
||||
movzx eax, byte ptr [rbp+0x38]
|
||||
movzx ebx, byte ptr [rbp+0x40]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
1:
|
||||
movzx ebx, byte ptr [rbp+0x48]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+0x80]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+0x88], eax
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov r12, qword ptr [rdi+0x40]
|
||||
mov r13, qword ptr [rdi+0x48]
|
||||
mov r14, qword ptr [rdi+0x50]
|
||||
mov r15, qword ptr [rdi+0x58]
|
||||
vmovdqu32 ymm16, ymmword ptr [rdx+r8-0x2*0x20]
|
||||
vinserti32x8 zmm16, zmm16, ymmword ptr [rdx+r12-0x2*0x20], 0x01
|
||||
vmovdqu32 ymm17, ymmword ptr [rdx+r9-0x2*0x20]
|
||||
vinserti32x8 zmm17, zmm17, ymmword ptr [rdx+r13-0x2*0x20], 0x01
|
||||
vpunpcklqdq zmm8, zmm16, zmm17
|
||||
vpunpckhqdq zmm9, zmm16, zmm17
|
||||
vmovdqu32 ymm18, ymmword ptr [rdx+r10-0x2*0x20]
|
||||
vinserti32x8 zmm18, zmm18, ymmword ptr [rdx+r14-0x2*0x20], 0x01
|
||||
vmovdqu32 ymm19, ymmword ptr [rdx+r11-0x2*0x20]
|
||||
vinserti32x8 zmm19, zmm19, ymmword ptr [rdx+r15-0x2*0x20], 0x01
|
||||
vpunpcklqdq zmm10, zmm18, zmm19
|
||||
vpunpckhqdq zmm11, zmm18, zmm19
|
||||
mov r8, qword ptr [rdi+0x20]
|
||||
mov r9, qword ptr [rdi+0x28]
|
||||
mov r10, qword ptr [rdi+0x30]
|
||||
mov r11, qword ptr [rdi+0x38]
|
||||
mov r12, qword ptr [rdi+0x60]
|
||||
mov r13, qword ptr [rdi+0x68]
|
||||
mov r14, qword ptr [rdi+0x70]
|
||||
mov r15, qword ptr [rdi+0x78]
|
||||
vmovdqu32 ymm16, ymmword ptr [rdx+r8-0x2*0x20]
|
||||
vinserti32x8 zmm16, zmm16, ymmword ptr [rdx+r12-0x2*0x20], 0x01
|
||||
vmovdqu32 ymm17, ymmword ptr [rdx+r9-0x2*0x20]
|
||||
vinserti32x8 zmm17, zmm17, ymmword ptr [rdx+r13-0x2*0x20], 0x01
|
||||
vpunpcklqdq zmm12, zmm16, zmm17
|
||||
vpunpckhqdq zmm13, zmm16, zmm17
|
||||
vmovdqu32 ymm18, ymmword ptr [rdx+r10-0x2*0x20]
|
||||
vinserti32x8 zmm18, zmm18, ymmword ptr [rdx+r14-0x2*0x20], 0x01
|
||||
vmovdqu32 ymm19, ymmword ptr [rdx+r11-0x2*0x20]
|
||||
vinserti32x8 zmm19, zmm19, ymmword ptr [rdx+r15-0x2*0x20], 0x01
|
||||
vpunpcklqdq zmm14, zmm18, zmm19
|
||||
vpunpckhqdq zmm15, zmm18, zmm19
|
||||
vmovdqa32 zmm27, zmmword ptr [INDEX0+rip]
|
||||
vmovdqa32 zmm31, zmmword ptr [INDEX1+rip]
|
||||
vshufps zmm16, zmm8, zmm10, 136
|
||||
vshufps zmm17, zmm12, zmm14, 136
|
||||
vmovdqa32 zmm20, zmm16
|
||||
vpermt2d zmm16, zmm27, zmm17
|
||||
vpermt2d zmm20, zmm31, zmm17
|
||||
vshufps zmm17, zmm8, zmm10, 221
|
||||
vshufps zmm30, zmm12, zmm14, 221
|
||||
vmovdqa32 zmm21, zmm17
|
||||
vpermt2d zmm17, zmm27, zmm30
|
||||
vpermt2d zmm21, zmm31, zmm30
|
||||
vshufps zmm18, zmm9, zmm11, 136
|
||||
vshufps zmm8, zmm13, zmm15, 136
|
||||
vmovdqa32 zmm22, zmm18
|
||||
vpermt2d zmm18, zmm27, zmm8
|
||||
vpermt2d zmm22, zmm31, zmm8
|
||||
vshufps zmm19, zmm9, zmm11, 221
|
||||
vshufps zmm8, zmm13, zmm15, 221
|
||||
vmovdqa32 zmm23, zmm19
|
||||
vpermt2d zmm19, zmm27, zmm8
|
||||
vpermt2d zmm23, zmm31, zmm8
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov r12, qword ptr [rdi+0x40]
|
||||
mov r13, qword ptr [rdi+0x48]
|
||||
mov r14, qword ptr [rdi+0x50]
|
||||
mov r15, qword ptr [rdi+0x58]
|
||||
vmovdqu32 ymm24, ymmword ptr [r8+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r12+rdx-0x1*0x20], 0x01
|
||||
vmovdqu32 ymm25, ymmword ptr [r9+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r13+rdx-0x1*0x20], 0x01
|
||||
vpunpcklqdq zmm8, zmm24, zmm25
|
||||
vpunpckhqdq zmm9, zmm24, zmm25
|
||||
vmovdqu32 ymm24, ymmword ptr [r10+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r14+rdx-0x1*0x20], 0x01
|
||||
vmovdqu32 ymm25, ymmword ptr [r11+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r15+rdx-0x1*0x20], 0x01
|
||||
vpunpcklqdq zmm10, zmm24, zmm25
|
||||
vpunpckhqdq zmm11, zmm24, zmm25
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r12+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r13+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r14+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
prefetcht0 [r15+rdx+0x80]
|
||||
mov r8, qword ptr [rdi+0x20]
|
||||
mov r9, qword ptr [rdi+0x28]
|
||||
mov r10, qword ptr [rdi+0x30]
|
||||
mov r11, qword ptr [rdi+0x38]
|
||||
mov r12, qword ptr [rdi+0x60]
|
||||
mov r13, qword ptr [rdi+0x68]
|
||||
mov r14, qword ptr [rdi+0x70]
|
||||
mov r15, qword ptr [rdi+0x78]
|
||||
vmovdqu32 ymm24, ymmword ptr [r8+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r12+rdx-0x1*0x20], 0x01
|
||||
vmovdqu32 ymm25, ymmword ptr [r9+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r13+rdx-0x1*0x20], 0x01
|
||||
vpunpcklqdq zmm12, zmm24, zmm25
|
||||
vpunpckhqdq zmm13, zmm24, zmm25
|
||||
vmovdqu32 ymm24, ymmword ptr [r10+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r14+rdx-0x1*0x20], 0x01
|
||||
vmovdqu32 ymm25, ymmword ptr [r11+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r15+rdx-0x1*0x20], 0x01
|
||||
vpunpcklqdq zmm14, zmm24, zmm25
|
||||
vpunpckhqdq zmm15, zmm24, zmm25
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r12+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r13+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r14+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
prefetcht0 [r15+rdx+0x80]
|
||||
vshufps zmm24, zmm8, zmm10, 136
|
||||
vshufps zmm30, zmm12, zmm14, 136
|
||||
vmovdqa32 zmm28, zmm24
|
||||
vpermt2d zmm24, zmm27, zmm30
|
||||
vpermt2d zmm28, zmm31, zmm30
|
||||
vshufps zmm25, zmm8, zmm10, 221
|
||||
vshufps zmm30, zmm12, zmm14, 221
|
||||
vmovdqa32 zmm29, zmm25
|
||||
vpermt2d zmm25, zmm27, zmm30
|
||||
vpermt2d zmm29, zmm31, zmm30
|
||||
vshufps zmm26, zmm9, zmm11, 136
|
||||
vshufps zmm8, zmm13, zmm15, 136
|
||||
vmovdqa32 zmm30, zmm26
|
||||
vpermt2d zmm26, zmm27, zmm8
|
||||
vpermt2d zmm30, zmm31, zmm8
|
||||
vshufps zmm8, zmm9, zmm11, 221
|
||||
vshufps zmm10, zmm13, zmm15, 221
|
||||
vpermi2d zmm27, zmm8, zmm10
|
||||
vpermi2d zmm31, zmm8, zmm10
|
||||
vpbroadcastd zmm8, dword ptr [BLAKE3_IV_0+rip]
|
||||
vpbroadcastd zmm9, dword ptr [BLAKE3_IV_1+rip]
|
||||
vpbroadcastd zmm10, dword ptr [BLAKE3_IV_2+rip]
|
||||
vpbroadcastd zmm11, dword ptr [BLAKE3_IV_3+rip]
|
||||
vmovdqa32 zmm12, zmmword ptr [rsp]
|
||||
vmovdqa32 zmm13, zmmword ptr [rsp+0x1*0x40]
|
||||
vpbroadcastd zmm14, dword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vpbroadcastd zmm15, dword ptr [rsp+0x22*0x4]
|
||||
vpaddd zmm0, zmm0, zmm16
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm20
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm17
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm21
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm24
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm28
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm25
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm29
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm18
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm23
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm22
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm16
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm17
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm25
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm27
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm30
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm19
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm29
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm20
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm18
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm22
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm27
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm21
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm31
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm26
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm30
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm23
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm19
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm20
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm21
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm16
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm24
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm28
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm31
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm29
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm26
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm23
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm16
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm18
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm17
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm25
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm24
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm30
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm28
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm29
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm18
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm19
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm22
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm27
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm17
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm31
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm25
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm30
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm19
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm26
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm20
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpxord zmm0, zmm0, zmm8
|
||||
vpxord zmm1, zmm1, zmm9
|
||||
vpxord zmm2, zmm2, zmm10
|
||||
vpxord zmm3, zmm3, zmm11
|
||||
vpxord zmm4, zmm4, zmm12
|
||||
vpxord zmm5, zmm5, zmm13
|
||||
vpxord zmm6, zmm6, zmm14
|
||||
vpxord zmm7, zmm7, zmm15
|
||||
movzx eax, byte ptr [rbp+0x38]
|
||||
jne 1b
|
||||
mov rbx, qword ptr [rbp+0x50]
|
||||
vpunpckldq zmm16, zmm0, zmm1
|
||||
vpunpckhdq zmm17, zmm0, zmm1
|
||||
vpunpckldq zmm18, zmm2, zmm3
|
||||
vpunpckhdq zmm19, zmm2, zmm3
|
||||
vpunpckldq zmm20, zmm4, zmm5
|
||||
vpunpckhdq zmm21, zmm4, zmm5
|
||||
vpunpckldq zmm22, zmm6, zmm7
|
||||
vpunpckhdq zmm23, zmm6, zmm7
|
||||
vpunpcklqdq zmm0, zmm16, zmm18
|
||||
vpunpckhqdq zmm1, zmm16, zmm18
|
||||
vpunpcklqdq zmm2, zmm17, zmm19
|
||||
vpunpckhqdq zmm3, zmm17, zmm19
|
||||
vpunpcklqdq zmm4, zmm20, zmm22
|
||||
vpunpckhqdq zmm5, zmm20, zmm22
|
||||
vpunpcklqdq zmm6, zmm21, zmm23
|
||||
vpunpckhqdq zmm7, zmm21, zmm23
|
||||
vshufi32x4 zmm16, zmm0, zmm4, 0x88
|
||||
vshufi32x4 zmm17, zmm1, zmm5, 0x88
|
||||
vshufi32x4 zmm18, zmm2, zmm6, 0x88
|
||||
vshufi32x4 zmm19, zmm3, zmm7, 0x88
|
||||
vshufi32x4 zmm20, zmm0, zmm4, 0xDD
|
||||
vshufi32x4 zmm21, zmm1, zmm5, 0xDD
|
||||
vshufi32x4 zmm22, zmm2, zmm6, 0xDD
|
||||
vshufi32x4 zmm23, zmm3, zmm7, 0xDD
|
||||
vshufi32x4 zmm0, zmm16, zmm17, 0x88
|
||||
vshufi32x4 zmm1, zmm18, zmm19, 0x88
|
||||
vshufi32x4 zmm2, zmm20, zmm21, 0x88
|
||||
vshufi32x4 zmm3, zmm22, zmm23, 0x88
|
||||
vshufi32x4 zmm4, zmm16, zmm17, 0xDD
|
||||
vshufi32x4 zmm5, zmm18, zmm19, 0xDD
|
||||
vshufi32x4 zmm6, zmm20, zmm21, 0xDD
|
||||
vshufi32x4 zmm7, zmm22, zmm23, 0xDD
|
||||
vmovdqu32 zmmword ptr [rbx], zmm0
|
||||
vmovdqu32 zmmword ptr [rbx+0x1*0x40], zmm1
|
||||
vmovdqu32 zmmword ptr [rbx+0x2*0x40], zmm2
|
||||
vmovdqu32 zmmword ptr [rbx+0x3*0x40], zmm3
|
||||
vmovdqu32 zmmword ptr [rbx+0x4*0x40], zmm4
|
||||
vmovdqu32 zmmword ptr [rbx+0x5*0x40], zmm5
|
||||
vmovdqu32 zmmword ptr [rbx+0x6*0x40], zmm6
|
||||
vmovdqu32 zmmword ptr [rbx+0x7*0x40], zmm7
|
||||
vmovdqa32 zmm0, zmmword ptr [rsp]
|
||||
vmovdqa32 zmm1, zmmword ptr [rsp+0x1*0x40]
|
||||
vmovdqa32 zmm2, zmm0
|
||||
vpaddd zmm2{k1}, zmm0, dword ptr [ADD16+rip] {1to16}
|
||||
vpcmpltud k2, zmm2, zmm0
|
||||
vpaddd zmm1 {k2}, zmm1, dword ptr [ADD1+rip] {1to16}
|
||||
vmovdqa32 zmmword ptr [rsp], zmm2
|
||||
vmovdqa32 zmmword ptr [rsp+0x1*0x40], zmm1
|
||||
add rdi, 128
|
||||
add rbx, 512
|
||||
mov qword ptr [rbp+0x50], rbx
|
||||
sub rsi, 16
|
||||
cmp rsi, 16
|
||||
jnc 2b
|
||||
test rsi, rsi
|
||||
jnz 3f
|
||||
4:
|
||||
vzeroupper
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
.p2align 6
|
||||
3:
|
||||
test esi, 0x8
|
||||
je 3f
|
||||
vpbroadcastd ymm0, dword ptr [rcx]
|
||||
vpbroadcastd ymm1, dword ptr [rcx+0x4]
|
||||
vpbroadcastd ymm2, dword ptr [rcx+0x8]
|
||||
vpbroadcastd ymm3, dword ptr [rcx+0xC]
|
||||
vpbroadcastd ymm4, dword ptr [rcx+0x10]
|
||||
vpbroadcastd ymm5, dword ptr [rcx+0x14]
|
||||
vpbroadcastd ymm6, dword ptr [rcx+0x18]
|
||||
vpbroadcastd ymm7, dword ptr [rcx+0x1C]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov r12, qword ptr [rdi+0x20]
|
||||
mov r13, qword ptr [rdi+0x28]
|
||||
mov r14, qword ptr [rdi+0x30]
|
||||
mov r15, qword ptr [rdi+0x38]
|
||||
movzx eax, byte ptr [rbp+0x38]
|
||||
movzx ebx, byte ptr [rbp+0x40]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
2:
|
||||
movzx ebx, byte ptr [rbp+0x48]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+0x80]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+0x88], eax
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x40], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x40]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x40], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x40]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x40], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x40]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x40], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm16, ymm12, ymm14, 136
|
||||
vshufps ymm17, ymm12, ymm14, 221
|
||||
vshufps ymm18, ymm13, ymm15, 136
|
||||
vshufps ymm19, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x30], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x30]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x30], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x30]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x30], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x30]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x30], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm20, ymm12, ymm14, 136
|
||||
vshufps ymm21, ymm12, ymm14, 221
|
||||
vshufps ymm22, ymm13, ymm15, 136
|
||||
vshufps ymm23, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x20], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x20]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x20], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x20]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x20], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x20]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x20], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm24, ymm12, ymm14, 136
|
||||
vshufps ymm25, ymm12, ymm14, 221
|
||||
vshufps ymm26, ymm13, ymm15, 136
|
||||
vshufps ymm27, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x10], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x10]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x10], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x10]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x10], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x10]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x10], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm28, ymm12, ymm14, 136
|
||||
vshufps ymm29, ymm12, ymm14, 221
|
||||
vshufps ymm30, ymm13, ymm15, 136
|
||||
vshufps ymm31, ymm13, ymm15, 221
|
||||
vpbroadcastd ymm8, dword ptr [BLAKE3_IV_0+rip]
|
||||
vpbroadcastd ymm9, dword ptr [BLAKE3_IV_1+rip]
|
||||
vpbroadcastd ymm10, dword ptr [BLAKE3_IV_2+rip]
|
||||
vpbroadcastd ymm11, dword ptr [BLAKE3_IV_3+rip]
|
||||
vmovdqa ymm12, ymmword ptr [rsp]
|
||||
vmovdqa ymm13, ymmword ptr [rsp+0x40]
|
||||
vpbroadcastd ymm14, dword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vpbroadcastd ymm15, dword ptr [rsp+0x88]
|
||||
vpaddd ymm0, ymm0, ymm16
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm20
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm17
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm21
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm24
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm28
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm25
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm29
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm18
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm23
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm22
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm16
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm17
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm25
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm27
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm30
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm19
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm29
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm20
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm18
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm22
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm27
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm21
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm31
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm26
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm30
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm23
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm19
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm20
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm21
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm16
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm24
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm28
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm31
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm29
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm26
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm23
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm16
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm18
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm17
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm25
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm24
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm30
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm28
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm29
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm18
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm19
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm22
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm27
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm17
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm31
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm25
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm30
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm19
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm26
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm20
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpxor ymm0, ymm0, ymm8
|
||||
vpxor ymm1, ymm1, ymm9
|
||||
vpxor ymm2, ymm2, ymm10
|
||||
vpxor ymm3, ymm3, ymm11
|
||||
vpxor ymm4, ymm4, ymm12
|
||||
vpxor ymm5, ymm5, ymm13
|
||||
vpxor ymm6, ymm6, ymm14
|
||||
vpxor ymm7, ymm7, ymm15
|
||||
movzx eax, byte ptr [rbp+0x38]
|
||||
jne 2b
|
||||
mov rbx, qword ptr [rbp+0x50]
|
||||
vunpcklps ymm8, ymm0, ymm1
|
||||
vunpcklps ymm9, ymm2, ymm3
|
||||
vunpckhps ymm10, ymm0, ymm1
|
||||
vunpcklps ymm11, ymm4, ymm5
|
||||
vunpcklps ymm0, ymm6, ymm7
|
||||
vshufps ymm12, ymm8, ymm9, 78
|
||||
vblendps ymm1, ymm8, ymm12, 0xCC
|
||||
vshufps ymm8, ymm11, ymm0, 78
|
||||
vunpckhps ymm13, ymm2, ymm3
|
||||
vblendps ymm2, ymm11, ymm8, 0xCC
|
||||
vblendps ymm3, ymm12, ymm9, 0xCC
|
||||
vperm2f128 ymm12, ymm1, ymm2, 0x20
|
||||
vmovups ymmword ptr [rbx], ymm12
|
||||
vunpckhps ymm14, ymm4, ymm5
|
||||
vblendps ymm4, ymm8, ymm0, 0xCC
|
||||
vunpckhps ymm15, ymm6, ymm7
|
||||
vperm2f128 ymm7, ymm3, ymm4, 0x20
|
||||
vmovups ymmword ptr [rbx+0x20], ymm7
|
||||
vshufps ymm5, ymm10, ymm13, 78
|
||||
vblendps ymm6, ymm5, ymm13, 0xCC
|
||||
vshufps ymm13, ymm14, ymm15, 78
|
||||
vblendps ymm10, ymm10, ymm5, 0xCC
|
||||
vblendps ymm14, ymm14, ymm13, 0xCC
|
||||
vperm2f128 ymm8, ymm10, ymm14, 0x20
|
||||
vmovups ymmword ptr [rbx+0x40], ymm8
|
||||
vblendps ymm15, ymm13, ymm15, 0xCC
|
||||
vperm2f128 ymm13, ymm6, ymm15, 0x20
|
||||
vmovups ymmword ptr [rbx+0x60], ymm13
|
||||
vperm2f128 ymm9, ymm1, ymm2, 0x31
|
||||
vperm2f128 ymm11, ymm3, ymm4, 0x31
|
||||
vmovups ymmword ptr [rbx+0x80], ymm9
|
||||
vperm2f128 ymm14, ymm10, ymm14, 0x31
|
||||
vperm2f128 ymm15, ymm6, ymm15, 0x31
|
||||
vmovups ymmword ptr [rbx+0xA0], ymm11
|
||||
vmovups ymmword ptr [rbx+0xC0], ymm14
|
||||
vmovups ymmword ptr [rbx+0xE0], ymm15
|
||||
vmovdqa ymm0, ymmword ptr [rsp]
|
||||
vmovdqa ymm2, ymmword ptr [rsp+0x2*0x20]
|
||||
vmovdqa32 ymm0 {k1}, ymmword ptr [rsp+0x1*0x20]
|
||||
vmovdqa32 ymm2 {k1}, ymmword ptr [rsp+0x3*0x20]
|
||||
vmovdqa ymmword ptr [rsp], ymm0
|
||||
vmovdqa ymmword ptr [rsp+0x2*0x20], ymm2
|
||||
add rbx, 256
|
||||
mov qword ptr [rbp+0x50], rbx
|
||||
add rdi, 64
|
||||
sub rsi, 8
|
||||
3:
|
||||
mov rbx, qword ptr [rbp+0x50]
|
||||
mov r15, qword ptr [rsp+0x80]
|
||||
movzx r13, byte ptr [rbp+0x38]
|
||||
movzx r12, byte ptr [rbp+0x48]
|
||||
test esi, 0x4
|
||||
je 3f
|
||||
vbroadcasti32x4 zmm0, xmmword ptr [rcx]
|
||||
vbroadcasti32x4 zmm1, xmmword ptr [rcx+0x1*0x10]
|
||||
vmovdqa xmm12, xmmword ptr [rsp]
|
||||
vmovdqa xmm13, xmmword ptr [rsp+0x4*0x10]
|
||||
vpunpckldq xmm14, xmm12, xmm13
|
||||
vpunpckhdq xmm15, xmm12, xmm13
|
||||
vpermq ymm14, ymm14, 0xDC
|
||||
vpermq ymm15, ymm15, 0xDC
|
||||
vpbroadcastd zmm12, dword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vinserti32x8 zmm13, zmm14, ymm15, 0x01
|
||||
mov eax, 17476
|
||||
kmovw k2, eax
|
||||
vpblendmd zmm13 {k2}, zmm13, zmm12
|
||||
vbroadcasti32x4 zmm15, xmmword ptr [BLAKE3_IV+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov eax, 43690
|
||||
kmovw k3, eax
|
||||
mov eax, 34952
|
||||
kmovw k4, eax
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+0x88], eax
|
||||
vmovdqa32 zmm2, zmm15
|
||||
vpbroadcastd zmm8, dword ptr [rsp+0x22*0x4]
|
||||
vpblendmd zmm3 {k4}, zmm13, zmm8
|
||||
vmovups zmm8, zmmword ptr [r8+rdx-0x1*0x40]
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r9+rdx-0x4*0x10], 0x01
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r10+rdx-0x4*0x10], 0x02
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r11+rdx-0x4*0x10], 0x03
|
||||
vmovups zmm9, zmmword ptr [r8+rdx-0x30]
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r9+rdx-0x3*0x10], 0x01
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r10+rdx-0x3*0x10], 0x02
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r11+rdx-0x3*0x10], 0x03
|
||||
vshufps zmm4, zmm8, zmm9, 136
|
||||
vshufps zmm5, zmm8, zmm9, 221
|
||||
vmovups zmm8, zmmword ptr [r8+rdx-0x20]
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r9+rdx-0x2*0x10], 0x01
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r10+rdx-0x2*0x10], 0x02
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r11+rdx-0x2*0x10], 0x03
|
||||
vmovups zmm9, zmmword ptr [r8+rdx-0x10]
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r9+rdx-0x1*0x10], 0x01
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r10+rdx-0x1*0x10], 0x02
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r11+rdx-0x1*0x10], 0x03
|
||||
vshufps zmm6, zmm8, zmm9, 136
|
||||
vshufps zmm7, zmm8, zmm9, 221
|
||||
vpshufd zmm6, zmm6, 0x93
|
||||
vpshufd zmm7, zmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 16
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 12
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 8
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 7
|
||||
vpshufd zmm0, zmm0, 0x93
|
||||
vpshufd zmm3, zmm3, 0x4E
|
||||
vpshufd zmm2, zmm2, 0x39
|
||||
vpaddd zmm0, zmm0, zmm6
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 16
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 12
|
||||
vpaddd zmm0, zmm0, zmm7
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 8
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 7
|
||||
vpshufd zmm0, zmm0, 0x39
|
||||
vpshufd zmm3, zmm3, 0x4E
|
||||
vpshufd zmm2, zmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps zmm8, zmm4, zmm5, 214
|
||||
vpshufd zmm9, zmm4, 0x0F
|
||||
vpshufd zmm4, zmm8, 0x39
|
||||
vshufps zmm8, zmm6, zmm7, 250
|
||||
vpblendmd zmm9 {k3}, zmm9, zmm8
|
||||
vpunpcklqdq zmm8, zmm7, zmm5
|
||||
vpblendmd zmm8 {k4}, zmm8, zmm6
|
||||
vpshufd zmm8, zmm8, 0x78
|
||||
vpunpckhdq zmm5, zmm5, zmm7
|
||||
vpunpckldq zmm6, zmm6, zmm5
|
||||
vpshufd zmm7, zmm6, 0x1E
|
||||
vmovdqa32 zmm5, zmm9
|
||||
vmovdqa32 zmm6, zmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxord zmm0, zmm0, zmm2
|
||||
vpxord zmm1, zmm1, zmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
vextracti128 xmmword ptr [rbx+0x20], ymm0, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x30], ymm1, 0x01
|
||||
vextracti32x4 xmmword ptr [rbx+0x4*0x10], zmm0, 0x02
|
||||
vextracti32x4 xmmword ptr [rbx+0x5*0x10], zmm1, 0x02
|
||||
vextracti32x4 xmmword ptr [rbx+0x6*0x10], zmm0, 0x03
|
||||
vextracti32x4 xmmword ptr [rbx+0x7*0x10], zmm1, 0x03
|
||||
vmovdqa xmm0, xmmword ptr [rsp]
|
||||
vmovdqa xmm2, xmmword ptr [rsp+0x40]
|
||||
vmovdqa32 xmm0 {k1}, xmmword ptr [rsp+0x1*0x10]
|
||||
vmovdqa32 xmm2 {k1}, xmmword ptr [rsp+0x5*0x10]
|
||||
vmovdqa xmmword ptr [rsp], xmm0
|
||||
vmovdqa xmmword ptr [rsp+0x40], xmm2
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
3:
|
||||
test esi, 0x2
|
||||
je 3f
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+0x10]
|
||||
vmovd xmm13, dword ptr [rsp]
|
||||
vpinsrd xmm13, xmm13, dword ptr [rsp+0x40], 1
|
||||
vpinsrd xmm13, xmm13, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vmovd xmm14, dword ptr [rsp+0x4]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+0x44], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vinserti128 ymm13, ymm13, xmm14, 0x01
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+0x88], eax
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vpbroadcastd ymm8, dword ptr [rsp+0x88]
|
||||
vpblendd ymm3, ymm13, ymm8, 0x88
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-0x40], 0x01
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-0x30], 0x01
|
||||
vshufps ymm4, ymm8, ymm9, 136
|
||||
vshufps ymm5, ymm8, ymm9, 221
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-0x20], 0x01
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-0x10], 0x01
|
||||
vshufps ymm6, ymm8, ymm9, 136
|
||||
vshufps ymm7, ymm8, ymm9, 221
|
||||
vpshufd ymm6, ymm6, 0x93
|
||||
vpshufd ymm7, ymm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 16
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 12
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 8
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 7
|
||||
vpshufd ymm0, ymm0, 0x93
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x39
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 16
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 12
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 8
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 7
|
||||
vpshufd ymm0, ymm0, 0x39
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps ymm8, ymm4, ymm5, 214
|
||||
vpshufd ymm9, ymm4, 0x0F
|
||||
vpshufd ymm4, ymm8, 0x39
|
||||
vshufps ymm8, ymm6, ymm7, 250
|
||||
vpblendd ymm9, ymm9, ymm8, 0xAA
|
||||
vpunpcklqdq ymm8, ymm7, ymm5
|
||||
vpblendd ymm8, ymm8, ymm6, 0x88
|
||||
vpshufd ymm8, ymm8, 0x78
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 0x1E
|
||||
vmovdqa ymm5, ymm9
|
||||
vmovdqa ymm6, ymm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
vextracti128 xmmword ptr [rbx+0x20], ymm0, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x30], ymm1, 0x01
|
||||
vmovdqa xmm0, xmmword ptr [rsp]
|
||||
vmovdqa xmm2, xmmword ptr [rsp+0x4*0x10]
|
||||
vmovdqu32 xmm0 {k1}, xmmword ptr [rsp+0x8]
|
||||
vmovdqu32 xmm2 {k1}, xmmword ptr [rsp+0x48]
|
||||
vmovdqa xmmword ptr [rsp], xmm0
|
||||
vmovdqa xmmword ptr [rsp+0x4*0x10], xmm2
|
||||
add rbx, 64
|
||||
add rdi, 16
|
||||
sub rsi, 2
|
||||
3:
|
||||
test esi, 0x1
|
||||
je 4b
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+0x10]
|
||||
vmovd xmm14, dword ptr [rsp]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+0x40], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vmovdqa xmm15, xmmword ptr [BLAKE3_IV+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
vpinsrd xmm3, xmm14, eax, 3
|
||||
vmovdqa xmm2, xmm15
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-0x30]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-0x10]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 0x93
|
||||
vpshufd xmm7, xmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x93
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x39
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x39
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0x0F
|
||||
vpshufd xmm4, xmm8, 0x39
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0xAA
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 0x88
|
||||
vpshufd xmm8, xmm8, 0x78
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 0x1E
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
jmp 4b
|
||||
.p2align 6
|
||||
_blake3_compress_in_place_avx512:
|
||||
blake3_compress_in_place_avx512:
|
||||
vmovdqu xmm0, xmmword ptr [rdi]
|
||||
vmovdqu xmm1, xmmword ptr [rdi+0x10]
|
||||
movzx eax, r8b
|
||||
movzx edx, dl
|
||||
shl rax, 32
|
||||
add rdx, rax
|
||||
vmovq xmm3, rcx
|
||||
vmovq xmm4, rdx
|
||||
vpunpcklqdq xmm3, xmm3, xmm4
|
||||
vmovaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vmovups xmm8, xmmword ptr [rsi]
|
||||
vmovups xmm9, xmmword ptr [rsi+0x10]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [rsi+0x20]
|
||||
vmovups xmm9, xmmword ptr [rsi+0x30]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 0x93
|
||||
vpshufd xmm7, xmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x93
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x39
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x39
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0x0F
|
||||
vpshufd xmm4, xmm8, 0x39
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0xAA
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 0x88
|
||||
vpshufd xmm8, xmm8, 0x78
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 0x1E
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
vmovdqu xmmword ptr [rdi], xmm0
|
||||
vmovdqu xmmword ptr [rdi+0x10], xmm1
|
||||
ret
|
||||
|
||||
.p2align 6
|
||||
_blake3_compress_xof_avx512:
|
||||
blake3_compress_xof_avx512:
|
||||
vmovdqu xmm0, xmmword ptr [rdi]
|
||||
vmovdqu xmm1, xmmword ptr [rdi+0x10]
|
||||
movzx eax, r8b
|
||||
movzx edx, dl
|
||||
shl rax, 32
|
||||
add rdx, rax
|
||||
vmovq xmm3, rcx
|
||||
vmovq xmm4, rdx
|
||||
vpunpcklqdq xmm3, xmm3, xmm4
|
||||
vmovaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vmovups xmm8, xmmword ptr [rsi]
|
||||
vmovups xmm9, xmmword ptr [rsi+0x10]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [rsi+0x20]
|
||||
vmovups xmm9, xmmword ptr [rsi+0x30]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 0x93
|
||||
vpshufd xmm7, xmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x93
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x39
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x39
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0x0F
|
||||
vpshufd xmm4, xmm8, 0x39
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0xAA
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 0x88
|
||||
vpshufd xmm8, xmm8, 0x78
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 0x1E
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
vpxor xmm2, xmm2, [rdi]
|
||||
vpxor xmm3, xmm3, [rdi+0x10]
|
||||
vmovdqu xmmword ptr [r9], xmm0
|
||||
vmovdqu xmmword ptr [r9+0x10], xmm1
|
||||
vmovdqu xmmword ptr [r9+0x20], xmm2
|
||||
vmovdqu xmmword ptr [r9+0x30], xmm3
|
||||
ret
|
||||
|
||||
#ifdef __APPLE__
|
||||
.static_data
|
||||
#else
|
||||
.section .rodata
|
||||
#endif
|
||||
.p2align 6
|
||||
INDEX0:
|
||||
.long 0, 1, 2, 3, 16, 17, 18, 19
|
||||
.long 8, 9, 10, 11, 24, 25, 26, 27
|
||||
INDEX1:
|
||||
.long 4, 5, 6, 7, 20, 21, 22, 23
|
||||
.long 12, 13, 14, 15, 28, 29, 30, 31
|
||||
ADD0:
|
||||
.long 0, 1, 2, 3, 4, 5, 6, 7
|
||||
.long 8, 9, 10, 11, 12, 13, 14, 15
|
||||
ADD1: .long 1
|
||||
|
||||
ADD16: .long 16
|
||||
BLAKE3_BLOCK_LEN:
|
||||
.long 64
|
||||
.p2align 6
|
||||
BLAKE3_IV:
|
||||
BLAKE3_IV_0:
|
||||
.long 0x6A09E667
|
||||
BLAKE3_IV_1:
|
||||
.long 0xBB67AE85
|
||||
BLAKE3_IV_2:
|
||||
.long 0x3C6EF372
|
||||
BLAKE3_IV_3:
|
||||
.long 0xA54FF53A
|
|
@ -0,0 +1,2615 @@
|
|||
.intel_syntax noprefix
|
||||
|
||||
.global _blake3_hash_many_avx512
|
||||
.global blake3_hash_many_avx512
|
||||
.global blake3_compress_in_place_avx512
|
||||
.global _blake3_compress_in_place_avx512
|
||||
.global blake3_compress_xof_avx512
|
||||
.global _blake3_compress_xof_avx512
|
||||
|
||||
.section .text
|
||||
.p2align 6
|
||||
_blake3_hash_many_avx512:
|
||||
blake3_hash_many_avx512:
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rdi
|
||||
push rsi
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 304
|
||||
and rsp, 0xFFFFFFFFFFFFFFC0
|
||||
vmovdqa xmmword ptr [rsp+0x90], xmm6
|
||||
vmovdqa xmmword ptr [rsp+0xA0], xmm7
|
||||
vmovdqa xmmword ptr [rsp+0xB0], xmm8
|
||||
vmovdqa xmmword ptr [rsp+0xC0], xmm9
|
||||
vmovdqa xmmword ptr [rsp+0xD0], xmm10
|
||||
vmovdqa xmmword ptr [rsp+0xE0], xmm11
|
||||
vmovdqa xmmword ptr [rsp+0xF0], xmm12
|
||||
vmovdqa xmmword ptr [rsp+0x100], xmm13
|
||||
vmovdqa xmmword ptr [rsp+0x110], xmm14
|
||||
vmovdqa xmmword ptr [rsp+0x120], xmm15
|
||||
mov rdi, rcx
|
||||
mov rsi, rdx
|
||||
mov rdx, r8
|
||||
mov rcx, r9
|
||||
mov r8, qword ptr [rbp+0x68]
|
||||
movzx r9, byte ptr [rbp+0x70]
|
||||
neg r9
|
||||
kmovw k1, r9d
|
||||
vmovd xmm0, r8d
|
||||
vpbroadcastd ymm0, xmm0
|
||||
shr r8, 32
|
||||
vmovd xmm1, r8d
|
||||
vpbroadcastd ymm1, xmm1
|
||||
vmovdqa ymm4, ymm1
|
||||
vmovdqa ymm5, ymm1
|
||||
vpaddd ymm2, ymm0, ymmword ptr [ADD0+rip]
|
||||
vpaddd ymm3, ymm0, ymmword ptr [ADD0+32+rip]
|
||||
vpcmpltud k2, ymm2, ymm0
|
||||
vpcmpltud k3, ymm3, ymm0
|
||||
vpaddd ymm4 {k2}, ymm4, dword ptr [ADD1+rip] {1to8}
|
||||
vpaddd ymm5 {k3}, ymm5, dword ptr [ADD1+rip] {1to8}
|
||||
knotw k2, k1
|
||||
vmovdqa32 ymm2 {k2}, ymm0
|
||||
vmovdqa32 ymm3 {k2}, ymm0
|
||||
vmovdqa32 ymm4 {k2}, ymm1
|
||||
vmovdqa32 ymm5 {k2}, ymm1
|
||||
vmovdqa ymmword ptr [rsp], ymm2
|
||||
vmovdqa ymmword ptr [rsp+0x20], ymm3
|
||||
vmovdqa ymmword ptr [rsp+0x40], ymm4
|
||||
vmovdqa ymmword ptr [rsp+0x60], ymm5
|
||||
shl rdx, 6
|
||||
mov qword ptr [rsp+0x80], rdx
|
||||
cmp rsi, 16
|
||||
jc 3f
|
||||
2:
|
||||
vpbroadcastd zmm0, dword ptr [rcx]
|
||||
vpbroadcastd zmm1, dword ptr [rcx+0x1*0x4]
|
||||
vpbroadcastd zmm2, dword ptr [rcx+0x2*0x4]
|
||||
vpbroadcastd zmm3, dword ptr [rcx+0x3*0x4]
|
||||
vpbroadcastd zmm4, dword ptr [rcx+0x4*0x4]
|
||||
vpbroadcastd zmm5, dword ptr [rcx+0x5*0x4]
|
||||
vpbroadcastd zmm6, dword ptr [rcx+0x6*0x4]
|
||||
vpbroadcastd zmm7, dword ptr [rcx+0x7*0x4]
|
||||
movzx eax, byte ptr [rbp+0x78]
|
||||
movzx ebx, byte ptr [rbp+0x80]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
1:
|
||||
movzx ebx, byte ptr [rbp+0x88]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+0x80]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+0x88], eax
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov r12, qword ptr [rdi+0x40]
|
||||
mov r13, qword ptr [rdi+0x48]
|
||||
mov r14, qword ptr [rdi+0x50]
|
||||
mov r15, qword ptr [rdi+0x58]
|
||||
vmovdqu32 ymm16, ymmword ptr [rdx+r8-0x2*0x20]
|
||||
vinserti32x8 zmm16, zmm16, ymmword ptr [rdx+r12-0x2*0x20], 0x01
|
||||
vmovdqu32 ymm17, ymmword ptr [rdx+r9-0x2*0x20]
|
||||
vinserti32x8 zmm17, zmm17, ymmword ptr [rdx+r13-0x2*0x20], 0x01
|
||||
vpunpcklqdq zmm8, zmm16, zmm17
|
||||
vpunpckhqdq zmm9, zmm16, zmm17
|
||||
vmovdqu32 ymm18, ymmword ptr [rdx+r10-0x2*0x20]
|
||||
vinserti32x8 zmm18, zmm18, ymmword ptr [rdx+r14-0x2*0x20], 0x01
|
||||
vmovdqu32 ymm19, ymmword ptr [rdx+r11-0x2*0x20]
|
||||
vinserti32x8 zmm19, zmm19, ymmword ptr [rdx+r15-0x2*0x20], 0x01
|
||||
vpunpcklqdq zmm10, zmm18, zmm19
|
||||
vpunpckhqdq zmm11, zmm18, zmm19
|
||||
mov r8, qword ptr [rdi+0x20]
|
||||
mov r9, qword ptr [rdi+0x28]
|
||||
mov r10, qword ptr [rdi+0x30]
|
||||
mov r11, qword ptr [rdi+0x38]
|
||||
mov r12, qword ptr [rdi+0x60]
|
||||
mov r13, qword ptr [rdi+0x68]
|
||||
mov r14, qword ptr [rdi+0x70]
|
||||
mov r15, qword ptr [rdi+0x78]
|
||||
vmovdqu32 ymm16, ymmword ptr [rdx+r8-0x2*0x20]
|
||||
vinserti32x8 zmm16, zmm16, ymmword ptr [rdx+r12-0x2*0x20], 0x01
|
||||
vmovdqu32 ymm17, ymmword ptr [rdx+r9-0x2*0x20]
|
||||
vinserti32x8 zmm17, zmm17, ymmword ptr [rdx+r13-0x2*0x20], 0x01
|
||||
vpunpcklqdq zmm12, zmm16, zmm17
|
||||
vpunpckhqdq zmm13, zmm16, zmm17
|
||||
vmovdqu32 ymm18, ymmword ptr [rdx+r10-0x2*0x20]
|
||||
vinserti32x8 zmm18, zmm18, ymmword ptr [rdx+r14-0x2*0x20], 0x01
|
||||
vmovdqu32 ymm19, ymmword ptr [rdx+r11-0x2*0x20]
|
||||
vinserti32x8 zmm19, zmm19, ymmword ptr [rdx+r15-0x2*0x20], 0x01
|
||||
vpunpcklqdq zmm14, zmm18, zmm19
|
||||
vpunpckhqdq zmm15, zmm18, zmm19
|
||||
vmovdqa32 zmm27, zmmword ptr [INDEX0+rip]
|
||||
vmovdqa32 zmm31, zmmword ptr [INDEX1+rip]
|
||||
vshufps zmm16, zmm8, zmm10, 136
|
||||
vshufps zmm17, zmm12, zmm14, 136
|
||||
vmovdqa32 zmm20, zmm16
|
||||
vpermt2d zmm16, zmm27, zmm17
|
||||
vpermt2d zmm20, zmm31, zmm17
|
||||
vshufps zmm17, zmm8, zmm10, 221
|
||||
vshufps zmm30, zmm12, zmm14, 221
|
||||
vmovdqa32 zmm21, zmm17
|
||||
vpermt2d zmm17, zmm27, zmm30
|
||||
vpermt2d zmm21, zmm31, zmm30
|
||||
vshufps zmm18, zmm9, zmm11, 136
|
||||
vshufps zmm8, zmm13, zmm15, 136
|
||||
vmovdqa32 zmm22, zmm18
|
||||
vpermt2d zmm18, zmm27, zmm8
|
||||
vpermt2d zmm22, zmm31, zmm8
|
||||
vshufps zmm19, zmm9, zmm11, 221
|
||||
vshufps zmm8, zmm13, zmm15, 221
|
||||
vmovdqa32 zmm23, zmm19
|
||||
vpermt2d zmm19, zmm27, zmm8
|
||||
vpermt2d zmm23, zmm31, zmm8
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov r12, qword ptr [rdi+0x40]
|
||||
mov r13, qword ptr [rdi+0x48]
|
||||
mov r14, qword ptr [rdi+0x50]
|
||||
mov r15, qword ptr [rdi+0x58]
|
||||
vmovdqu32 ymm24, ymmword ptr [r8+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r12+rdx-0x1*0x20], 0x01
|
||||
vmovdqu32 ymm25, ymmword ptr [r9+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r13+rdx-0x1*0x20], 0x01
|
||||
vpunpcklqdq zmm8, zmm24, zmm25
|
||||
vpunpckhqdq zmm9, zmm24, zmm25
|
||||
vmovdqu32 ymm24, ymmword ptr [r10+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r14+rdx-0x1*0x20], 0x01
|
||||
vmovdqu32 ymm25, ymmword ptr [r11+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r15+rdx-0x1*0x20], 0x01
|
||||
vpunpcklqdq zmm10, zmm24, zmm25
|
||||
vpunpckhqdq zmm11, zmm24, zmm25
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r12+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r13+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r14+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
prefetcht0 [r15+rdx+0x80]
|
||||
mov r8, qword ptr [rdi+0x20]
|
||||
mov r9, qword ptr [rdi+0x28]
|
||||
mov r10, qword ptr [rdi+0x30]
|
||||
mov r11, qword ptr [rdi+0x38]
|
||||
mov r12, qword ptr [rdi+0x60]
|
||||
mov r13, qword ptr [rdi+0x68]
|
||||
mov r14, qword ptr [rdi+0x70]
|
||||
mov r15, qword ptr [rdi+0x78]
|
||||
vmovdqu32 ymm24, ymmword ptr [r8+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r12+rdx-0x1*0x20], 0x01
|
||||
vmovdqu32 ymm25, ymmword ptr [r9+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r13+rdx-0x1*0x20], 0x01
|
||||
vpunpcklqdq zmm12, zmm24, zmm25
|
||||
vpunpckhqdq zmm13, zmm24, zmm25
|
||||
vmovdqu32 ymm24, ymmword ptr [r10+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r14+rdx-0x1*0x20], 0x01
|
||||
vmovdqu32 ymm25, ymmword ptr [r11+rdx-0x1*0x20]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r15+rdx-0x1*0x20], 0x01
|
||||
vpunpcklqdq zmm14, zmm24, zmm25
|
||||
vpunpckhqdq zmm15, zmm24, zmm25
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r12+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r13+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r14+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
prefetcht0 [r15+rdx+0x80]
|
||||
vshufps zmm24, zmm8, zmm10, 136
|
||||
vshufps zmm30, zmm12, zmm14, 136
|
||||
vmovdqa32 zmm28, zmm24
|
||||
vpermt2d zmm24, zmm27, zmm30
|
||||
vpermt2d zmm28, zmm31, zmm30
|
||||
vshufps zmm25, zmm8, zmm10, 221
|
||||
vshufps zmm30, zmm12, zmm14, 221
|
||||
vmovdqa32 zmm29, zmm25
|
||||
vpermt2d zmm25, zmm27, zmm30
|
||||
vpermt2d zmm29, zmm31, zmm30
|
||||
vshufps zmm26, zmm9, zmm11, 136
|
||||
vshufps zmm8, zmm13, zmm15, 136
|
||||
vmovdqa32 zmm30, zmm26
|
||||
vpermt2d zmm26, zmm27, zmm8
|
||||
vpermt2d zmm30, zmm31, zmm8
|
||||
vshufps zmm8, zmm9, zmm11, 221
|
||||
vshufps zmm10, zmm13, zmm15, 221
|
||||
vpermi2d zmm27, zmm8, zmm10
|
||||
vpermi2d zmm31, zmm8, zmm10
|
||||
vpbroadcastd zmm8, dword ptr [BLAKE3_IV_0+rip]
|
||||
vpbroadcastd zmm9, dword ptr [BLAKE3_IV_1+rip]
|
||||
vpbroadcastd zmm10, dword ptr [BLAKE3_IV_2+rip]
|
||||
vpbroadcastd zmm11, dword ptr [BLAKE3_IV_3+rip]
|
||||
vmovdqa32 zmm12, zmmword ptr [rsp]
|
||||
vmovdqa32 zmm13, zmmword ptr [rsp+0x1*0x40]
|
||||
vpbroadcastd zmm14, dword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vpbroadcastd zmm15, dword ptr [rsp+0x22*0x4]
|
||||
vpaddd zmm0, zmm0, zmm16
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm20
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm17
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm21
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm24
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm28
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm25
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm29
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm18
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm23
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm22
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm16
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm17
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm25
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm27
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm30
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm19
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm29
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm20
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm18
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm22
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm27
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm21
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm31
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm26
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm30
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm23
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm19
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm20
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm21
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm16
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm24
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm28
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm31
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm29
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm26
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm23
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm16
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm18
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm17
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm25
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm24
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm30
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm28
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm29
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm18
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm19
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm22
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm27
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm17
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm31
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm25
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm30
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm19
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm26
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm20
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpxord zmm0, zmm0, zmm8
|
||||
vpxord zmm1, zmm1, zmm9
|
||||
vpxord zmm2, zmm2, zmm10
|
||||
vpxord zmm3, zmm3, zmm11
|
||||
vpxord zmm4, zmm4, zmm12
|
||||
vpxord zmm5, zmm5, zmm13
|
||||
vpxord zmm6, zmm6, zmm14
|
||||
vpxord zmm7, zmm7, zmm15
|
||||
movzx eax, byte ptr [rbp+0x78]
|
||||
jne 1b
|
||||
mov rbx, qword ptr [rbp+0x90]
|
||||
vpunpckldq zmm16, zmm0, zmm1
|
||||
vpunpckhdq zmm17, zmm0, zmm1
|
||||
vpunpckldq zmm18, zmm2, zmm3
|
||||
vpunpckhdq zmm19, zmm2, zmm3
|
||||
vpunpckldq zmm20, zmm4, zmm5
|
||||
vpunpckhdq zmm21, zmm4, zmm5
|
||||
vpunpckldq zmm22, zmm6, zmm7
|
||||
vpunpckhdq zmm23, zmm6, zmm7
|
||||
vpunpcklqdq zmm0, zmm16, zmm18
|
||||
vpunpckhqdq zmm1, zmm16, zmm18
|
||||
vpunpcklqdq zmm2, zmm17, zmm19
|
||||
vpunpckhqdq zmm3, zmm17, zmm19
|
||||
vpunpcklqdq zmm4, zmm20, zmm22
|
||||
vpunpckhqdq zmm5, zmm20, zmm22
|
||||
vpunpcklqdq zmm6, zmm21, zmm23
|
||||
vpunpckhqdq zmm7, zmm21, zmm23
|
||||
vshufi32x4 zmm16, zmm0, zmm4, 0x88
|
||||
vshufi32x4 zmm17, zmm1, zmm5, 0x88
|
||||
vshufi32x4 zmm18, zmm2, zmm6, 0x88
|
||||
vshufi32x4 zmm19, zmm3, zmm7, 0x88
|
||||
vshufi32x4 zmm20, zmm0, zmm4, 0xDD
|
||||
vshufi32x4 zmm21, zmm1, zmm5, 0xDD
|
||||
vshufi32x4 zmm22, zmm2, zmm6, 0xDD
|
||||
vshufi32x4 zmm23, zmm3, zmm7, 0xDD
|
||||
vshufi32x4 zmm0, zmm16, zmm17, 0x88
|
||||
vshufi32x4 zmm1, zmm18, zmm19, 0x88
|
||||
vshufi32x4 zmm2, zmm20, zmm21, 0x88
|
||||
vshufi32x4 zmm3, zmm22, zmm23, 0x88
|
||||
vshufi32x4 zmm4, zmm16, zmm17, 0xDD
|
||||
vshufi32x4 zmm5, zmm18, zmm19, 0xDD
|
||||
vshufi32x4 zmm6, zmm20, zmm21, 0xDD
|
||||
vshufi32x4 zmm7, zmm22, zmm23, 0xDD
|
||||
vmovdqu32 zmmword ptr [rbx], zmm0
|
||||
vmovdqu32 zmmword ptr [rbx+0x1*0x40], zmm1
|
||||
vmovdqu32 zmmword ptr [rbx+0x2*0x40], zmm2
|
||||
vmovdqu32 zmmword ptr [rbx+0x3*0x40], zmm3
|
||||
vmovdqu32 zmmword ptr [rbx+0x4*0x40], zmm4
|
||||
vmovdqu32 zmmword ptr [rbx+0x5*0x40], zmm5
|
||||
vmovdqu32 zmmword ptr [rbx+0x6*0x40], zmm6
|
||||
vmovdqu32 zmmword ptr [rbx+0x7*0x40], zmm7
|
||||
vmovdqa32 zmm0, zmmword ptr [rsp]
|
||||
vmovdqa32 zmm1, zmmword ptr [rsp+0x1*0x40]
|
||||
vmovdqa32 zmm2, zmm0
|
||||
vpaddd zmm2{k1}, zmm0, dword ptr [ADD16+rip] {1to16}
|
||||
vpcmpltud k2, zmm2, zmm0
|
||||
vpaddd zmm1 {k2}, zmm1, dword ptr [ADD1+rip] {1to16}
|
||||
vmovdqa32 zmmword ptr [rsp], zmm2
|
||||
vmovdqa32 zmmword ptr [rsp+0x1*0x40], zmm1
|
||||
add rdi, 128
|
||||
add rbx, 512
|
||||
mov qword ptr [rbp+0x90], rbx
|
||||
sub rsi, 16
|
||||
cmp rsi, 16
|
||||
jnc 2b
|
||||
test rsi, rsi
|
||||
jne 3f
|
||||
4:
|
||||
vzeroupper
|
||||
vmovdqa xmm6, xmmword ptr [rsp+0x90]
|
||||
vmovdqa xmm7, xmmword ptr [rsp+0xA0]
|
||||
vmovdqa xmm8, xmmword ptr [rsp+0xB0]
|
||||
vmovdqa xmm9, xmmword ptr [rsp+0xC0]
|
||||
vmovdqa xmm10, xmmword ptr [rsp+0xD0]
|
||||
vmovdqa xmm11, xmmword ptr [rsp+0xE0]
|
||||
vmovdqa xmm12, xmmword ptr [rsp+0xF0]
|
||||
vmovdqa xmm13, xmmword ptr [rsp+0x100]
|
||||
vmovdqa xmm14, xmmword ptr [rsp+0x110]
|
||||
vmovdqa xmm15, xmmword ptr [rsp+0x120]
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop rsi
|
||||
pop rdi
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
.p2align 6
|
||||
3:
|
||||
test esi, 0x8
|
||||
je 3f
|
||||
vpbroadcastd ymm0, dword ptr [rcx]
|
||||
vpbroadcastd ymm1, dword ptr [rcx+0x4]
|
||||
vpbroadcastd ymm2, dword ptr [rcx+0x8]
|
||||
vpbroadcastd ymm3, dword ptr [rcx+0xC]
|
||||
vpbroadcastd ymm4, dword ptr [rcx+0x10]
|
||||
vpbroadcastd ymm5, dword ptr [rcx+0x14]
|
||||
vpbroadcastd ymm6, dword ptr [rcx+0x18]
|
||||
vpbroadcastd ymm7, dword ptr [rcx+0x1C]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov r12, qword ptr [rdi+0x20]
|
||||
mov r13, qword ptr [rdi+0x28]
|
||||
mov r14, qword ptr [rdi+0x30]
|
||||
mov r15, qword ptr [rdi+0x38]
|
||||
movzx eax, byte ptr [rbp+0x78]
|
||||
movzx ebx, byte ptr [rbp+0x80]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
2:
|
||||
movzx ebx, byte ptr [rbp+0x88]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+0x80]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+0x88], eax
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x40], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x40]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x40], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x40]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x40], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x40]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x40], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm16, ymm12, ymm14, 136
|
||||
vshufps ymm17, ymm12, ymm14, 221
|
||||
vshufps ymm18, ymm13, ymm15, 136
|
||||
vshufps ymm19, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x30], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x30]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x30], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x30]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x30], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x30]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x30], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm20, ymm12, ymm14, 136
|
||||
vshufps ymm21, ymm12, ymm14, 221
|
||||
vshufps ymm22, ymm13, ymm15, 136
|
||||
vshufps ymm23, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x20], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x20]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x20], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x20]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x20], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x20]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x20], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm24, ymm12, ymm14, 136
|
||||
vshufps ymm25, ymm12, ymm14, 221
|
||||
vshufps ymm26, ymm13, ymm15, 136
|
||||
vshufps ymm27, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-0x10], 0x01
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-0x10]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-0x10], 0x01
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-0x10]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-0x10], 0x01
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-0x10]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-0x10], 0x01
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm28, ymm12, ymm14, 136
|
||||
vshufps ymm29, ymm12, ymm14, 221
|
||||
vshufps ymm30, ymm13, ymm15, 136
|
||||
vshufps ymm31, ymm13, ymm15, 221
|
||||
vpbroadcastd ymm8, dword ptr [BLAKE3_IV_0+rip]
|
||||
vpbroadcastd ymm9, dword ptr [BLAKE3_IV_1+rip]
|
||||
vpbroadcastd ymm10, dword ptr [BLAKE3_IV_2+rip]
|
||||
vpbroadcastd ymm11, dword ptr [BLAKE3_IV_3+rip]
|
||||
vmovdqa ymm12, ymmword ptr [rsp]
|
||||
vmovdqa ymm13, ymmword ptr [rsp+0x40]
|
||||
vpbroadcastd ymm14, dword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vpbroadcastd ymm15, dword ptr [rsp+0x88]
|
||||
vpaddd ymm0, ymm0, ymm16
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm20
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm17
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm21
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm24
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm28
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm25
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm29
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm18
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm23
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm22
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm16
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm17
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm25
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm27
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm30
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm19
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm29
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm20
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm18
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm22
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm27
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm21
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm31
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm26
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm30
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm23
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm19
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm20
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm21
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm16
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm24
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm28
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm31
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm29
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm26
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm23
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm16
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm18
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm17
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm25
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm24
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm30
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm28
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm29
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm18
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm19
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm22
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm27
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm17
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm31
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm25
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm30
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm19
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm26
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm20
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpxor ymm0, ymm0, ymm8
|
||||
vpxor ymm1, ymm1, ymm9
|
||||
vpxor ymm2, ymm2, ymm10
|
||||
vpxor ymm3, ymm3, ymm11
|
||||
vpxor ymm4, ymm4, ymm12
|
||||
vpxor ymm5, ymm5, ymm13
|
||||
vpxor ymm6, ymm6, ymm14
|
||||
vpxor ymm7, ymm7, ymm15
|
||||
movzx eax, byte ptr [rbp+0x78]
|
||||
jne 2b
|
||||
mov rbx, qword ptr [rbp+0x90]
|
||||
vunpcklps ymm8, ymm0, ymm1
|
||||
vunpcklps ymm9, ymm2, ymm3
|
||||
vunpckhps ymm10, ymm0, ymm1
|
||||
vunpcklps ymm11, ymm4, ymm5
|
||||
vunpcklps ymm0, ymm6, ymm7
|
||||
vshufps ymm12, ymm8, ymm9, 78
|
||||
vblendps ymm1, ymm8, ymm12, 0xCC
|
||||
vshufps ymm8, ymm11, ymm0, 78
|
||||
vunpckhps ymm13, ymm2, ymm3
|
||||
vblendps ymm2, ymm11, ymm8, 0xCC
|
||||
vblendps ymm3, ymm12, ymm9, 0xCC
|
||||
vperm2f128 ymm12, ymm1, ymm2, 0x20
|
||||
vmovups ymmword ptr [rbx], ymm12
|
||||
vunpckhps ymm14, ymm4, ymm5
|
||||
vblendps ymm4, ymm8, ymm0, 0xCC
|
||||
vunpckhps ymm15, ymm6, ymm7
|
||||
vperm2f128 ymm7, ymm3, ymm4, 0x20
|
||||
vmovups ymmword ptr [rbx+0x20], ymm7
|
||||
vshufps ymm5, ymm10, ymm13, 78
|
||||
vblendps ymm6, ymm5, ymm13, 0xCC
|
||||
vshufps ymm13, ymm14, ymm15, 78
|
||||
vblendps ymm10, ymm10, ymm5, 0xCC
|
||||
vblendps ymm14, ymm14, ymm13, 0xCC
|
||||
vperm2f128 ymm8, ymm10, ymm14, 0x20
|
||||
vmovups ymmword ptr [rbx+0x40], ymm8
|
||||
vblendps ymm15, ymm13, ymm15, 0xCC
|
||||
vperm2f128 ymm13, ymm6, ymm15, 0x20
|
||||
vmovups ymmword ptr [rbx+0x60], ymm13
|
||||
vperm2f128 ymm9, ymm1, ymm2, 0x31
|
||||
vperm2f128 ymm11, ymm3, ymm4, 0x31
|
||||
vmovups ymmword ptr [rbx+0x80], ymm9
|
||||
vperm2f128 ymm14, ymm10, ymm14, 0x31
|
||||
vperm2f128 ymm15, ymm6, ymm15, 0x31
|
||||
vmovups ymmword ptr [rbx+0xA0], ymm11
|
||||
vmovups ymmword ptr [rbx+0xC0], ymm14
|
||||
vmovups ymmword ptr [rbx+0xE0], ymm15
|
||||
vmovdqa ymm0, ymmword ptr [rsp]
|
||||
vmovdqa ymm2, ymmword ptr [rsp+0x40]
|
||||
vmovdqa32 ymm0 {k1}, ymmword ptr [rsp+0x1*0x20]
|
||||
vmovdqa32 ymm2 {k1}, ymmword ptr [rsp+0x3*0x20]
|
||||
vmovdqa ymmword ptr [rsp], ymm0
|
||||
vmovdqa ymmword ptr [rsp+0x40], ymm2
|
||||
add rbx, 256
|
||||
mov qword ptr [rbp+0x90], rbx
|
||||
add rdi, 64
|
||||
sub rsi, 8
|
||||
3:
|
||||
mov rbx, qword ptr [rbp+0x90]
|
||||
mov r15, qword ptr [rsp+0x80]
|
||||
movzx r13, byte ptr [rbp+0x78]
|
||||
movzx r12, byte ptr [rbp+0x88]
|
||||
test esi, 0x4
|
||||
je 3f
|
||||
vbroadcasti32x4 zmm0, xmmword ptr [rcx]
|
||||
vbroadcasti32x4 zmm1, xmmword ptr [rcx+0x1*0x10]
|
||||
vmovdqa xmm12, xmmword ptr [rsp]
|
||||
vmovdqa xmm13, xmmword ptr [rsp+0x40]
|
||||
vpunpckldq xmm14, xmm12, xmm13
|
||||
vpunpckhdq xmm15, xmm12, xmm13
|
||||
vpermq ymm14, ymm14, 0xDC
|
||||
vpermq ymm15, ymm15, 0xDC
|
||||
vpbroadcastd zmm12, dword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
vinserti32x8 zmm13, zmm14, ymm15, 0x01
|
||||
mov eax, 17476
|
||||
kmovw k2, eax
|
||||
vpblendmd zmm13 {k2}, zmm13, zmm12
|
||||
vbroadcasti32x4 zmm15, xmmword ptr [BLAKE3_IV+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
mov eax, 43690
|
||||
kmovw k3, eax
|
||||
mov eax, 34952
|
||||
kmovw k4, eax
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+0x88], eax
|
||||
vmovdqa32 zmm2, zmm15
|
||||
vpbroadcastd zmm8, dword ptr [rsp+0x22*0x4]
|
||||
vpblendmd zmm3 {k4}, zmm13, zmm8
|
||||
vmovups zmm8, zmmword ptr [r8+rdx-0x1*0x40]
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r9+rdx-0x4*0x10], 0x01
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r10+rdx-0x4*0x10], 0x02
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r11+rdx-0x4*0x10], 0x03
|
||||
vmovups zmm9, zmmword ptr [r8+rdx-0x30]
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r9+rdx-0x3*0x10], 0x01
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r10+rdx-0x3*0x10], 0x02
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r11+rdx-0x3*0x10], 0x03
|
||||
vshufps zmm4, zmm8, zmm9, 136
|
||||
vshufps zmm5, zmm8, zmm9, 221
|
||||
vmovups zmm8, zmmword ptr [r8+rdx-0x20]
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r9+rdx-0x2*0x10], 0x01
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r10+rdx-0x2*0x10], 0x02
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r11+rdx-0x2*0x10], 0x03
|
||||
vmovups zmm9, zmmword ptr [r8+rdx-0x10]
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r9+rdx-0x1*0x10], 0x01
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r10+rdx-0x1*0x10], 0x02
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r11+rdx-0x1*0x10], 0x03
|
||||
vshufps zmm6, zmm8, zmm9, 136
|
||||
vshufps zmm7, zmm8, zmm9, 221
|
||||
vpshufd zmm6, zmm6, 0x93
|
||||
vpshufd zmm7, zmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 16
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 12
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 8
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 7
|
||||
vpshufd zmm0, zmm0, 0x93
|
||||
vpshufd zmm3, zmm3, 0x4E
|
||||
vpshufd zmm2, zmm2, 0x39
|
||||
vpaddd zmm0, zmm0, zmm6
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 16
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 12
|
||||
vpaddd zmm0, zmm0, zmm7
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 8
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 7
|
||||
vpshufd zmm0, zmm0, 0x39
|
||||
vpshufd zmm3, zmm3, 0x4E
|
||||
vpshufd zmm2, zmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps zmm8, zmm4, zmm5, 214
|
||||
vpshufd zmm9, zmm4, 0x0F
|
||||
vpshufd zmm4, zmm8, 0x39
|
||||
vshufps zmm8, zmm6, zmm7, 250
|
||||
vpblendmd zmm9 {k3}, zmm9, zmm8
|
||||
vpunpcklqdq zmm8, zmm7, zmm5
|
||||
vpblendmd zmm8 {k4}, zmm8, zmm6
|
||||
vpshufd zmm8, zmm8, 0x78
|
||||
vpunpckhdq zmm5, zmm5, zmm7
|
||||
vpunpckldq zmm6, zmm6, zmm5
|
||||
vpshufd zmm7, zmm6, 0x1E
|
||||
vmovdqa32 zmm5, zmm9
|
||||
vmovdqa32 zmm6, zmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxord zmm0, zmm0, zmm2
|
||||
vpxord zmm1, zmm1, zmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
vextracti128 xmmword ptr [rbx+0x20], ymm0, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x30], ymm1, 0x01
|
||||
vextracti32x4 xmmword ptr [rbx+0x4*0x10], zmm0, 0x02
|
||||
vextracti32x4 xmmword ptr [rbx+0x5*0x10], zmm1, 0x02
|
||||
vextracti32x4 xmmword ptr [rbx+0x6*0x10], zmm0, 0x03
|
||||
vextracti32x4 xmmword ptr [rbx+0x7*0x10], zmm1, 0x03
|
||||
vmovdqa xmm0, xmmword ptr [rsp]
|
||||
vmovdqa xmm2, xmmword ptr [rsp+0x40]
|
||||
vmovdqa32 xmm0 {k1}, xmmword ptr [rsp+0x1*0x10]
|
||||
vmovdqa32 xmm2 {k1}, xmmword ptr [rsp+0x5*0x10]
|
||||
vmovdqa xmmword ptr [rsp], xmm0
|
||||
vmovdqa xmmword ptr [rsp+0x40], xmm2
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
3:
|
||||
test esi, 0x2
|
||||
je 3f
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+0x10]
|
||||
vmovd xmm13, dword ptr [rsp]
|
||||
vpinsrd xmm13, xmm13, dword ptr [rsp+0x40], 1
|
||||
vpinsrd xmm13, xmm13, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vmovd xmm14, dword ptr [rsp+0x4]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+0x44], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vinserti128 ymm13, ymm13, xmm14, 0x01
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+0x88], eax
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vpbroadcastd ymm8, dword ptr [rsp+0x88]
|
||||
vpblendd ymm3, ymm13, ymm8, 0x88
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-0x40]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-0x40], 0x01
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-0x30]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-0x30], 0x01
|
||||
vshufps ymm4, ymm8, ymm9, 136
|
||||
vshufps ymm5, ymm8, ymm9, 221
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-0x20]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-0x20], 0x01
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-0x10]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-0x10], 0x01
|
||||
vshufps ymm6, ymm8, ymm9, 136
|
||||
vshufps ymm7, ymm8, ymm9, 221
|
||||
vpshufd ymm6, ymm6, 0x93
|
||||
vpshufd ymm7, ymm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 16
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 12
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 8
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 7
|
||||
vpshufd ymm0, ymm0, 0x93
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x39
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 16
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 12
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 8
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 7
|
||||
vpshufd ymm0, ymm0, 0x39
|
||||
vpshufd ymm3, ymm3, 0x4E
|
||||
vpshufd ymm2, ymm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps ymm8, ymm4, ymm5, 214
|
||||
vpshufd ymm9, ymm4, 0x0F
|
||||
vpshufd ymm4, ymm8, 0x39
|
||||
vshufps ymm8, ymm6, ymm7, 250
|
||||
vpblendd ymm9, ymm9, ymm8, 0xAA
|
||||
vpunpcklqdq ymm8, ymm7, ymm5
|
||||
vpblendd ymm8, ymm8, ymm6, 0x88
|
||||
vpshufd ymm8, ymm8, 0x78
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 0x1E
|
||||
vmovdqa ymm5, ymm9
|
||||
vmovdqa ymm6, ymm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
vextracti128 xmmword ptr [rbx+0x20], ymm0, 0x01
|
||||
vextracti128 xmmword ptr [rbx+0x30], ymm1, 0x01
|
||||
vmovdqa xmm0, xmmword ptr [rsp]
|
||||
vmovdqa xmm2, xmmword ptr [rsp+0x40]
|
||||
vmovdqu32 xmm0 {k1}, xmmword ptr [rsp+0x8]
|
||||
vmovdqu32 xmm2 {k1}, xmmword ptr [rsp+0x48]
|
||||
vmovdqa xmmword ptr [rsp], xmm0
|
||||
vmovdqa xmmword ptr [rsp+0x40], xmm2
|
||||
add rbx, 64
|
||||
add rdi, 16
|
||||
sub rsi, 2
|
||||
3:
|
||||
test esi, 0x1
|
||||
je 4b
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+0x10]
|
||||
vmovd xmm14, dword ptr [rsp]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+0x40], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
vmovdqa xmm15, xmmword ptr [BLAKE3_IV+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
.p2align 5
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
vpinsrd xmm3, xmm14, eax, 3
|
||||
vmovdqa xmm2, xmm15
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-0x30]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-0x10]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 0x93
|
||||
vpshufd xmm7, xmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x93
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x39
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x39
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0x0F
|
||||
vpshufd xmm4, xmm8, 0x39
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0xAA
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 0x88
|
||||
vpshufd xmm8, xmm8, 0x78
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 0x1E
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+0x10], xmm1
|
||||
jmp 4b
|
||||
|
||||
|
||||
.p2align 6
|
||||
_blake3_compress_in_place_avx512:
|
||||
blake3_compress_in_place_avx512:
|
||||
sub rsp, 72
|
||||
vmovdqa xmmword ptr [rsp], xmm6
|
||||
vmovdqa xmmword ptr [rsp+0x10], xmm7
|
||||
vmovdqa xmmword ptr [rsp+0x20], xmm8
|
||||
vmovdqa xmmword ptr [rsp+0x30], xmm9
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+0x10]
|
||||
movzx eax, byte ptr [rsp+0x70]
|
||||
movzx r8d, r8b
|
||||
shl rax, 32
|
||||
add r8, rax
|
||||
vmovq xmm3, r9
|
||||
vmovq xmm4, r8
|
||||
vpunpcklqdq xmm3, xmm3, xmm4
|
||||
vmovaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vmovups xmm8, xmmword ptr [rdx]
|
||||
vmovups xmm9, xmmword ptr [rdx+0x10]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [rdx+0x20]
|
||||
vmovups xmm9, xmmword ptr [rdx+0x30]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 0x93
|
||||
vpshufd xmm7, xmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x93
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x39
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x39
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0x0F
|
||||
vpshufd xmm4, xmm8, 0x39
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0xAA
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 0x88
|
||||
vpshufd xmm8, xmm8, 0x78
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 0x1E
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
vmovdqu xmmword ptr [rcx], xmm0
|
||||
vmovdqu xmmword ptr [rcx+0x10], xmm1
|
||||
vmovdqa xmm6, xmmword ptr [rsp]
|
||||
vmovdqa xmm7, xmmword ptr [rsp+0x10]
|
||||
vmovdqa xmm8, xmmword ptr [rsp+0x20]
|
||||
vmovdqa xmm9, xmmword ptr [rsp+0x30]
|
||||
add rsp, 72
|
||||
ret
|
||||
|
||||
|
||||
.p2align 6
|
||||
_blake3_compress_xof_avx512:
|
||||
blake3_compress_xof_avx512:
|
||||
sub rsp, 72
|
||||
vmovdqa xmmword ptr [rsp], xmm6
|
||||
vmovdqa xmmword ptr [rsp+0x10], xmm7
|
||||
vmovdqa xmmword ptr [rsp+0x20], xmm8
|
||||
vmovdqa xmmword ptr [rsp+0x30], xmm9
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+0x10]
|
||||
movzx eax, byte ptr [rsp+0x70]
|
||||
movzx r8d, r8b
|
||||
mov r10, qword ptr [rsp+0x78]
|
||||
shl rax, 32
|
||||
add r8, rax
|
||||
vmovq xmm3, r9
|
||||
vmovq xmm4, r8
|
||||
vpunpcklqdq xmm3, xmm3, xmm4
|
||||
vmovaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
vmovups xmm8, xmmword ptr [rdx]
|
||||
vmovups xmm9, xmmword ptr [rdx+0x10]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [rdx+0x20]
|
||||
vmovups xmm9, xmmword ptr [rdx+0x30]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 0x93
|
||||
vpshufd xmm7, xmm7, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x93
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x39
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 0x39
|
||||
vpshufd xmm3, xmm3, 0x4E
|
||||
vpshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0x0F
|
||||
vpshufd xmm4, xmm8, 0x39
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0xAA
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 0x88
|
||||
vpshufd xmm8, xmm8, 0x78
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 0x1E
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
vpxor xmm2, xmm2, xmmword ptr [rcx]
|
||||
vpxor xmm3, xmm3, xmmword ptr [rcx+0x10]
|
||||
vmovdqu xmmword ptr [r10], xmm0
|
||||
vmovdqu xmmword ptr [r10+0x10], xmm1
|
||||
vmovdqu xmmword ptr [r10+0x20], xmm2
|
||||
vmovdqu xmmword ptr [r10+0x30], xmm3
|
||||
vmovdqa xmm6, xmmword ptr [rsp]
|
||||
vmovdqa xmm7, xmmword ptr [rsp+0x10]
|
||||
vmovdqa xmm8, xmmword ptr [rsp+0x20]
|
||||
vmovdqa xmm9, xmmword ptr [rsp+0x30]
|
||||
add rsp, 72
|
||||
ret
|
||||
|
||||
.section .rodata
|
||||
.p2align 6
|
||||
INDEX0:
|
||||
.long 0, 1, 2, 3, 16, 17, 18, 19
|
||||
.long 8, 9, 10, 11, 24, 25, 26, 27
|
||||
INDEX1:
|
||||
.long 4, 5, 6, 7, 20, 21, 22, 23
|
||||
.long 12, 13, 14, 15, 28, 29, 30, 31
|
||||
ADD0:
|
||||
.long 0, 1, 2, 3, 4, 5, 6, 7
|
||||
.long 8, 9, 10, 11, 12, 13, 14, 15
|
||||
ADD1: .long 1
|
||||
|
||||
ADD16: .long 16
|
||||
BLAKE3_BLOCK_LEN:
|
||||
.long 64
|
||||
.p2align 6
|
||||
BLAKE3_IV:
|
||||
BLAKE3_IV_0:
|
||||
.long 0x6A09E667
|
||||
BLAKE3_IV_1:
|
||||
.long 0xBB67AE85
|
||||
BLAKE3_IV_2:
|
||||
.long 0x3C6EF372
|
||||
BLAKE3_IV_3:
|
||||
.long 0xA54FF53A
|
|
@ -0,0 +1,2634 @@
|
|||
public _blake3_hash_many_avx512
|
||||
public blake3_hash_many_avx512
|
||||
public blake3_compress_in_place_avx512
|
||||
public _blake3_compress_in_place_avx512
|
||||
public blake3_compress_xof_avx512
|
||||
public _blake3_compress_xof_avx512
|
||||
|
||||
_TEXT SEGMENT ALIGN(16) 'CODE'
|
||||
|
||||
ALIGN 16
|
||||
blake3_hash_many_avx512 PROC
|
||||
_blake3_hash_many_avx512 PROC
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rdi
|
||||
push rsi
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 304
|
||||
and rsp, 0FFFFFFFFFFFFFFC0H
|
||||
vmovdqa xmmword ptr [rsp+90H], xmm6
|
||||
vmovdqa xmmword ptr [rsp+0A0H], xmm7
|
||||
vmovdqa xmmword ptr [rsp+0B0H], xmm8
|
||||
vmovdqa xmmword ptr [rsp+0C0H], xmm9
|
||||
vmovdqa xmmword ptr [rsp+0D0H], xmm10
|
||||
vmovdqa xmmword ptr [rsp+0E0H], xmm11
|
||||
vmovdqa xmmword ptr [rsp+0F0H], xmm12
|
||||
vmovdqa xmmword ptr [rsp+100H], xmm13
|
||||
vmovdqa xmmword ptr [rsp+110H], xmm14
|
||||
vmovdqa xmmword ptr [rsp+120H], xmm15
|
||||
mov rdi, rcx
|
||||
mov rsi, rdx
|
||||
mov rdx, r8
|
||||
mov rcx, r9
|
||||
mov r8, qword ptr [rbp+68H]
|
||||
movzx r9, byte ptr [rbp+70H]
|
||||
neg r9
|
||||
kmovw k1, r9d
|
||||
vmovd xmm0, r8d
|
||||
vpbroadcastd ymm0, xmm0
|
||||
shr r8, 32
|
||||
vmovd xmm1, r8d
|
||||
vpbroadcastd ymm1, xmm1
|
||||
vmovdqa ymm4, ymm1
|
||||
vmovdqa ymm5, ymm1
|
||||
vpaddd ymm2, ymm0, ymmword ptr [ADD0]
|
||||
vpaddd ymm3, ymm0, ymmword ptr [ADD0+32]
|
||||
vpcmpud k2, ymm2, ymm0, 1
|
||||
vpcmpud k3, ymm3, ymm0, 1
|
||||
; XXX: ml64.exe does not currently understand the syntax. We use a workaround.
|
||||
vpbroadcastd ymm6, dword ptr [ADD1]
|
||||
vpaddd ymm4 {k2}, ymm4, ymm6
|
||||
vpaddd ymm5 {k3}, ymm5, ymm6
|
||||
; vpaddd ymm4 {k2}, ymm4, dword ptr [ADD1] {1to8}
|
||||
; vpaddd ymm5 {k3}, ymm5, dword ptr [ADD1] {1to8}
|
||||
knotw k2, k1
|
||||
vmovdqa32 ymm2 {k2}, ymm0
|
||||
vmovdqa32 ymm3 {k2}, ymm0
|
||||
vmovdqa32 ymm4 {k2}, ymm1
|
||||
vmovdqa32 ymm5 {k2}, ymm1
|
||||
vmovdqa ymmword ptr [rsp], ymm2
|
||||
vmovdqa ymmword ptr [rsp+20H], ymm3
|
||||
vmovdqa ymmword ptr [rsp+40H], ymm4
|
||||
vmovdqa ymmword ptr [rsp+60H], ymm5
|
||||
shl rdx, 6
|
||||
mov qword ptr [rsp+80H], rdx
|
||||
cmp rsi, 16
|
||||
jc final15blocks
|
||||
outerloop16:
|
||||
vpbroadcastd zmm0, dword ptr [rcx]
|
||||
vpbroadcastd zmm1, dword ptr [rcx+1H*4H]
|
||||
vpbroadcastd zmm2, dword ptr [rcx+2H*4H]
|
||||
vpbroadcastd zmm3, dword ptr [rcx+3H*4H]
|
||||
vpbroadcastd zmm4, dword ptr [rcx+4H*4H]
|
||||
vpbroadcastd zmm5, dword ptr [rcx+5H*4H]
|
||||
vpbroadcastd zmm6, dword ptr [rcx+6H*4H]
|
||||
vpbroadcastd zmm7, dword ptr [rcx+7H*4H]
|
||||
movzx eax, byte ptr [rbp+78H]
|
||||
movzx ebx, byte ptr [rbp+80H]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
ALIGN 16
|
||||
innerloop16:
|
||||
movzx ebx, byte ptr [rbp+88H]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+80H]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+88H], eax
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
mov r10, qword ptr [rdi+10H]
|
||||
mov r11, qword ptr [rdi+18H]
|
||||
mov r12, qword ptr [rdi+40H]
|
||||
mov r13, qword ptr [rdi+48H]
|
||||
mov r14, qword ptr [rdi+50H]
|
||||
mov r15, qword ptr [rdi+58H]
|
||||
vmovdqu32 ymm16, ymmword ptr [rdx+r8-2H*20H]
|
||||
vinserti32x8 zmm16, zmm16, ymmword ptr [rdx+r12-2H*20H], 01H
|
||||
vmovdqu32 ymm17, ymmword ptr [rdx+r9-2H*20H]
|
||||
vinserti32x8 zmm17, zmm17, ymmword ptr [rdx+r13-2H*20H], 01H
|
||||
vpunpcklqdq zmm8, zmm16, zmm17
|
||||
vpunpckhqdq zmm9, zmm16, zmm17
|
||||
vmovdqu32 ymm18, ymmword ptr [rdx+r10-2H*20H]
|
||||
vinserti32x8 zmm18, zmm18, ymmword ptr [rdx+r14-2H*20H], 01H
|
||||
vmovdqu32 ymm19, ymmword ptr [rdx+r11-2H*20H]
|
||||
vinserti32x8 zmm19, zmm19, ymmword ptr [rdx+r15-2H*20H], 01H
|
||||
vpunpcklqdq zmm10, zmm18, zmm19
|
||||
vpunpckhqdq zmm11, zmm18, zmm19
|
||||
mov r8, qword ptr [rdi+20H]
|
||||
mov r9, qword ptr [rdi+28H]
|
||||
mov r10, qword ptr [rdi+30H]
|
||||
mov r11, qword ptr [rdi+38H]
|
||||
mov r12, qword ptr [rdi+60H]
|
||||
mov r13, qword ptr [rdi+68H]
|
||||
mov r14, qword ptr [rdi+70H]
|
||||
mov r15, qword ptr [rdi+78H]
|
||||
vmovdqu32 ymm16, ymmword ptr [rdx+r8-2H*20H]
|
||||
vinserti32x8 zmm16, zmm16, ymmword ptr [rdx+r12-2H*20H], 01H
|
||||
vmovdqu32 ymm17, ymmword ptr [rdx+r9-2H*20H]
|
||||
vinserti32x8 zmm17, zmm17, ymmword ptr [rdx+r13-2H*20H], 01H
|
||||
vpunpcklqdq zmm12, zmm16, zmm17
|
||||
vpunpckhqdq zmm13, zmm16, zmm17
|
||||
vmovdqu32 ymm18, ymmword ptr [rdx+r10-2H*20H]
|
||||
vinserti32x8 zmm18, zmm18, ymmword ptr [rdx+r14-2H*20H], 01H
|
||||
vmovdqu32 ymm19, ymmword ptr [rdx+r11-2H*20H]
|
||||
vinserti32x8 zmm19, zmm19, ymmword ptr [rdx+r15-2H*20H], 01H
|
||||
vpunpcklqdq zmm14, zmm18, zmm19
|
||||
vpunpckhqdq zmm15, zmm18, zmm19
|
||||
vmovdqa32 zmm27, zmmword ptr [INDEX0]
|
||||
vmovdqa32 zmm31, zmmword ptr [INDEX1]
|
||||
vshufps zmm16, zmm8, zmm10, 136
|
||||
vshufps zmm17, zmm12, zmm14, 136
|
||||
vmovdqa32 zmm20, zmm16
|
||||
vpermt2d zmm16, zmm27, zmm17
|
||||
vpermt2d zmm20, zmm31, zmm17
|
||||
vshufps zmm17, zmm8, zmm10, 221
|
||||
vshufps zmm30, zmm12, zmm14, 221
|
||||
vmovdqa32 zmm21, zmm17
|
||||
vpermt2d zmm17, zmm27, zmm30
|
||||
vpermt2d zmm21, zmm31, zmm30
|
||||
vshufps zmm18, zmm9, zmm11, 136
|
||||
vshufps zmm8, zmm13, zmm15, 136
|
||||
vmovdqa32 zmm22, zmm18
|
||||
vpermt2d zmm18, zmm27, zmm8
|
||||
vpermt2d zmm22, zmm31, zmm8
|
||||
vshufps zmm19, zmm9, zmm11, 221
|
||||
vshufps zmm8, zmm13, zmm15, 221
|
||||
vmovdqa32 zmm23, zmm19
|
||||
vpermt2d zmm19, zmm27, zmm8
|
||||
vpermt2d zmm23, zmm31, zmm8
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
mov r10, qword ptr [rdi+10H]
|
||||
mov r11, qword ptr [rdi+18H]
|
||||
mov r12, qword ptr [rdi+40H]
|
||||
mov r13, qword ptr [rdi+48H]
|
||||
mov r14, qword ptr [rdi+50H]
|
||||
mov r15, qword ptr [rdi+58H]
|
||||
vmovdqu32 ymm24, ymmword ptr [r8+rdx-1H*20H]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r12+rdx-1H*20H], 01H
|
||||
vmovdqu32 ymm25, ymmword ptr [r9+rdx-1H*20H]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r13+rdx-1H*20H], 01H
|
||||
vpunpcklqdq zmm8, zmm24, zmm25
|
||||
vpunpckhqdq zmm9, zmm24, zmm25
|
||||
vmovdqu32 ymm24, ymmword ptr [r10+rdx-1H*20H]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r14+rdx-1H*20H], 01H
|
||||
vmovdqu32 ymm25, ymmword ptr [r11+rdx-1H*20H]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r15+rdx-1H*20H], 01H
|
||||
vpunpcklqdq zmm10, zmm24, zmm25
|
||||
vpunpckhqdq zmm11, zmm24, zmm25
|
||||
prefetcht0 byte ptr [r8+rdx+80H]
|
||||
prefetcht0 byte ptr [r12+rdx+80H]
|
||||
prefetcht0 byte ptr [r9+rdx+80H]
|
||||
prefetcht0 byte ptr [r13+rdx+80H]
|
||||
prefetcht0 byte ptr [r10+rdx+80H]
|
||||
prefetcht0 byte ptr [r14+rdx+80H]
|
||||
prefetcht0 byte ptr [r11+rdx+80H]
|
||||
prefetcht0 byte ptr [r15+rdx+80H]
|
||||
mov r8, qword ptr [rdi+20H]
|
||||
mov r9, qword ptr [rdi+28H]
|
||||
mov r10, qword ptr [rdi+30H]
|
||||
mov r11, qword ptr [rdi+38H]
|
||||
mov r12, qword ptr [rdi+60H]
|
||||
mov r13, qword ptr [rdi+68H]
|
||||
mov r14, qword ptr [rdi+70H]
|
||||
mov r15, qword ptr [rdi+78H]
|
||||
vmovdqu32 ymm24, ymmword ptr [r8+rdx-1H*20H]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r12+rdx-1H*20H], 01H
|
||||
vmovdqu32 ymm25, ymmword ptr [r9+rdx-1H*20H]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r13+rdx-1H*20H], 01H
|
||||
vpunpcklqdq zmm12, zmm24, zmm25
|
||||
vpunpckhqdq zmm13, zmm24, zmm25
|
||||
vmovdqu32 ymm24, ymmword ptr [r10+rdx-1H*20H]
|
||||
vinserti32x8 zmm24, zmm24, ymmword ptr [r14+rdx-1H*20H], 01H
|
||||
vmovdqu32 ymm25, ymmword ptr [r11+rdx-1H*20H]
|
||||
vinserti32x8 zmm25, zmm25, ymmword ptr [r15+rdx-1H*20H], 01H
|
||||
vpunpcklqdq zmm14, zmm24, zmm25
|
||||
vpunpckhqdq zmm15, zmm24, zmm25
|
||||
prefetcht0 byte ptr [r8+rdx+80H]
|
||||
prefetcht0 byte ptr [r12+rdx+80H]
|
||||
prefetcht0 byte ptr [r9+rdx+80H]
|
||||
prefetcht0 byte ptr [r13+rdx+80H]
|
||||
prefetcht0 byte ptr [r10+rdx+80H]
|
||||
prefetcht0 byte ptr [r14+rdx+80H]
|
||||
prefetcht0 byte ptr [r11+rdx+80H]
|
||||
prefetcht0 byte ptr [r15+rdx+80H]
|
||||
vshufps zmm24, zmm8, zmm10, 136
|
||||
vshufps zmm30, zmm12, zmm14, 136
|
||||
vmovdqa32 zmm28, zmm24
|
||||
vpermt2d zmm24, zmm27, zmm30
|
||||
vpermt2d zmm28, zmm31, zmm30
|
||||
vshufps zmm25, zmm8, zmm10, 221
|
||||
vshufps zmm30, zmm12, zmm14, 221
|
||||
vmovdqa32 zmm29, zmm25
|
||||
vpermt2d zmm25, zmm27, zmm30
|
||||
vpermt2d zmm29, zmm31, zmm30
|
||||
vshufps zmm26, zmm9, zmm11, 136
|
||||
vshufps zmm8, zmm13, zmm15, 136
|
||||
vmovdqa32 zmm30, zmm26
|
||||
vpermt2d zmm26, zmm27, zmm8
|
||||
vpermt2d zmm30, zmm31, zmm8
|
||||
vshufps zmm8, zmm9, zmm11, 221
|
||||
vshufps zmm10, zmm13, zmm15, 221
|
||||
vpermi2d zmm27, zmm8, zmm10
|
||||
vpermi2d zmm31, zmm8, zmm10
|
||||
vpbroadcastd zmm8, dword ptr [BLAKE3_IV_0]
|
||||
vpbroadcastd zmm9, dword ptr [BLAKE3_IV_1]
|
||||
vpbroadcastd zmm10, dword ptr [BLAKE3_IV_2]
|
||||
vpbroadcastd zmm11, dword ptr [BLAKE3_IV_3]
|
||||
vmovdqa32 zmm12, zmmword ptr [rsp]
|
||||
vmovdqa32 zmm13, zmmword ptr [rsp+1H*40H]
|
||||
vpbroadcastd zmm14, dword ptr [BLAKE3_BLOCK_LEN]
|
||||
vpbroadcastd zmm15, dword ptr [rsp+22H*4H]
|
||||
vpaddd zmm0, zmm0, zmm16
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm20
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm17
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm21
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm24
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm28
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm25
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm29
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm18
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm23
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm22
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm16
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm17
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm25
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm27
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm30
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm19
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm29
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm20
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm18
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm22
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm27
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm21
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm31
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm26
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm30
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm23
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm19
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm20
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm21
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm16
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm24
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm28
|
||||
vpaddd zmm1, zmm1, zmm25
|
||||
vpaddd zmm2, zmm2, zmm31
|
||||
vpaddd zmm3, zmm3, zmm30
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm29
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm26
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm23
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm16
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm18
|
||||
vpaddd zmm1, zmm1, zmm19
|
||||
vpaddd zmm2, zmm2, zmm17
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm25
|
||||
vpaddd zmm1, zmm1, zmm27
|
||||
vpaddd zmm2, zmm2, zmm24
|
||||
vpaddd zmm3, zmm3, zmm31
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm30
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm28
|
||||
vpaddd zmm3, zmm3, zmm17
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm29
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm18
|
||||
vpaddd zmm3, zmm3, zmm20
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm19
|
||||
vpaddd zmm1, zmm1, zmm26
|
||||
vpaddd zmm2, zmm2, zmm22
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpaddd zmm0, zmm0, zmm27
|
||||
vpaddd zmm1, zmm1, zmm21
|
||||
vpaddd zmm2, zmm2, zmm17
|
||||
vpaddd zmm3, zmm3, zmm24
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vprord zmm15, zmm15, 16
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 12
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vpaddd zmm0, zmm0, zmm31
|
||||
vpaddd zmm1, zmm1, zmm16
|
||||
vpaddd zmm2, zmm2, zmm25
|
||||
vpaddd zmm3, zmm3, zmm22
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm1, zmm1, zmm5
|
||||
vpaddd zmm2, zmm2, zmm6
|
||||
vpaddd zmm3, zmm3, zmm7
|
||||
vpxord zmm12, zmm12, zmm0
|
||||
vpxord zmm13, zmm13, zmm1
|
||||
vpxord zmm14, zmm14, zmm2
|
||||
vpxord zmm15, zmm15, zmm3
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vprord zmm15, zmm15, 8
|
||||
vpaddd zmm8, zmm8, zmm12
|
||||
vpaddd zmm9, zmm9, zmm13
|
||||
vpaddd zmm10, zmm10, zmm14
|
||||
vpaddd zmm11, zmm11, zmm15
|
||||
vpxord zmm4, zmm4, zmm8
|
||||
vpxord zmm5, zmm5, zmm9
|
||||
vpxord zmm6, zmm6, zmm10
|
||||
vpxord zmm7, zmm7, zmm11
|
||||
vprord zmm4, zmm4, 7
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vpaddd zmm0, zmm0, zmm30
|
||||
vpaddd zmm1, zmm1, zmm18
|
||||
vpaddd zmm2, zmm2, zmm19
|
||||
vpaddd zmm3, zmm3, zmm23
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 16
|
||||
vprord zmm12, zmm12, 16
|
||||
vprord zmm13, zmm13, 16
|
||||
vprord zmm14, zmm14, 16
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 12
|
||||
vprord zmm6, zmm6, 12
|
||||
vprord zmm7, zmm7, 12
|
||||
vprord zmm4, zmm4, 12
|
||||
vpaddd zmm0, zmm0, zmm26
|
||||
vpaddd zmm1, zmm1, zmm28
|
||||
vpaddd zmm2, zmm2, zmm20
|
||||
vpaddd zmm3, zmm3, zmm29
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm1, zmm1, zmm6
|
||||
vpaddd zmm2, zmm2, zmm7
|
||||
vpaddd zmm3, zmm3, zmm4
|
||||
vpxord zmm15, zmm15, zmm0
|
||||
vpxord zmm12, zmm12, zmm1
|
||||
vpxord zmm13, zmm13, zmm2
|
||||
vpxord zmm14, zmm14, zmm3
|
||||
vprord zmm15, zmm15, 8
|
||||
vprord zmm12, zmm12, 8
|
||||
vprord zmm13, zmm13, 8
|
||||
vprord zmm14, zmm14, 8
|
||||
vpaddd zmm10, zmm10, zmm15
|
||||
vpaddd zmm11, zmm11, zmm12
|
||||
vpaddd zmm8, zmm8, zmm13
|
||||
vpaddd zmm9, zmm9, zmm14
|
||||
vpxord zmm5, zmm5, zmm10
|
||||
vpxord zmm6, zmm6, zmm11
|
||||
vpxord zmm7, zmm7, zmm8
|
||||
vpxord zmm4, zmm4, zmm9
|
||||
vprord zmm5, zmm5, 7
|
||||
vprord zmm6, zmm6, 7
|
||||
vprord zmm7, zmm7, 7
|
||||
vprord zmm4, zmm4, 7
|
||||
vpxord zmm0, zmm0, zmm8
|
||||
vpxord zmm1, zmm1, zmm9
|
||||
vpxord zmm2, zmm2, zmm10
|
||||
vpxord zmm3, zmm3, zmm11
|
||||
vpxord zmm4, zmm4, zmm12
|
||||
vpxord zmm5, zmm5, zmm13
|
||||
vpxord zmm6, zmm6, zmm14
|
||||
vpxord zmm7, zmm7, zmm15
|
||||
movzx eax, byte ptr [rbp+78H]
|
||||
jne innerloop16
|
||||
mov rbx, qword ptr [rbp+90H]
|
||||
vpunpckldq zmm16, zmm0, zmm1
|
||||
vpunpckhdq zmm17, zmm0, zmm1
|
||||
vpunpckldq zmm18, zmm2, zmm3
|
||||
vpunpckhdq zmm19, zmm2, zmm3
|
||||
vpunpckldq zmm20, zmm4, zmm5
|
||||
vpunpckhdq zmm21, zmm4, zmm5
|
||||
vpunpckldq zmm22, zmm6, zmm7
|
||||
vpunpckhdq zmm23, zmm6, zmm7
|
||||
vpunpcklqdq zmm0, zmm16, zmm18
|
||||
vpunpckhqdq zmm1, zmm16, zmm18
|
||||
vpunpcklqdq zmm2, zmm17, zmm19
|
||||
vpunpckhqdq zmm3, zmm17, zmm19
|
||||
vpunpcklqdq zmm4, zmm20, zmm22
|
||||
vpunpckhqdq zmm5, zmm20, zmm22
|
||||
vpunpcklqdq zmm6, zmm21, zmm23
|
||||
vpunpckhqdq zmm7, zmm21, zmm23
|
||||
vshufi32x4 zmm16, zmm0, zmm4, 88H
|
||||
vshufi32x4 zmm17, zmm1, zmm5, 88H
|
||||
vshufi32x4 zmm18, zmm2, zmm6, 88H
|
||||
vshufi32x4 zmm19, zmm3, zmm7, 88H
|
||||
vshufi32x4 zmm20, zmm0, zmm4, 0DDH
|
||||
vshufi32x4 zmm21, zmm1, zmm5, 0DDH
|
||||
vshufi32x4 zmm22, zmm2, zmm6, 0DDH
|
||||
vshufi32x4 zmm23, zmm3, zmm7, 0DDH
|
||||
vshufi32x4 zmm0, zmm16, zmm17, 88H
|
||||
vshufi32x4 zmm1, zmm18, zmm19, 88H
|
||||
vshufi32x4 zmm2, zmm20, zmm21, 88H
|
||||
vshufi32x4 zmm3, zmm22, zmm23, 88H
|
||||
vshufi32x4 zmm4, zmm16, zmm17, 0DDH
|
||||
vshufi32x4 zmm5, zmm18, zmm19, 0DDH
|
||||
vshufi32x4 zmm6, zmm20, zmm21, 0DDH
|
||||
vshufi32x4 zmm7, zmm22, zmm23, 0DDH
|
||||
vmovdqu32 zmmword ptr [rbx], zmm0
|
||||
vmovdqu32 zmmword ptr [rbx+1H*40H], zmm1
|
||||
vmovdqu32 zmmword ptr [rbx+2H*40H], zmm2
|
||||
vmovdqu32 zmmword ptr [rbx+3H*40H], zmm3
|
||||
vmovdqu32 zmmword ptr [rbx+4H*40H], zmm4
|
||||
vmovdqu32 zmmword ptr [rbx+5H*40H], zmm5
|
||||
vmovdqu32 zmmword ptr [rbx+6H*40H], zmm6
|
||||
vmovdqu32 zmmword ptr [rbx+7H*40H], zmm7
|
||||
vmovdqa32 zmm0, zmmword ptr [rsp]
|
||||
vmovdqa32 zmm1, zmmword ptr [rsp+1H*40H]
|
||||
vmovdqa32 zmm2, zmm0
|
||||
; XXX: ml64.exe does not currently understand the syntax. We use a workaround.
|
||||
vpbroadcastd zmm4, dword ptr [ADD16]
|
||||
vpbroadcastd zmm5, dword ptr [ADD1]
|
||||
vpaddd zmm2{k1}, zmm0, zmm4
|
||||
; vpaddd zmm2{k1}, zmm0, dword ptr [ADD16] ; {1to16}
|
||||
vpcmpud k2, zmm2, zmm0, 1
|
||||
vpaddd zmm1 {k2}, zmm1, zmm5
|
||||
; vpaddd zmm1 {k2}, zmm1, dword ptr [ADD1] ; {1to16}
|
||||
vmovdqa32 zmmword ptr [rsp], zmm2
|
||||
vmovdqa32 zmmword ptr [rsp+1H*40H], zmm1
|
||||
add rdi, 128
|
||||
add rbx, 512
|
||||
mov qword ptr [rbp+90H], rbx
|
||||
sub rsi, 16
|
||||
cmp rsi, 16
|
||||
jnc outerloop16
|
||||
test rsi, rsi
|
||||
jne final15blocks
|
||||
unwind:
|
||||
vzeroupper
|
||||
vmovdqa xmm6, xmmword ptr [rsp+90H]
|
||||
vmovdqa xmm7, xmmword ptr [rsp+0A0H]
|
||||
vmovdqa xmm8, xmmword ptr [rsp+0B0H]
|
||||
vmovdqa xmm9, xmmword ptr [rsp+0C0H]
|
||||
vmovdqa xmm10, xmmword ptr [rsp+0D0H]
|
||||
vmovdqa xmm11, xmmword ptr [rsp+0E0H]
|
||||
vmovdqa xmm12, xmmword ptr [rsp+0F0H]
|
||||
vmovdqa xmm13, xmmword ptr [rsp+100H]
|
||||
vmovdqa xmm14, xmmword ptr [rsp+110H]
|
||||
vmovdqa xmm15, xmmword ptr [rsp+120H]
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop rsi
|
||||
pop rdi
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
ALIGN 16
|
||||
final15blocks:
|
||||
test esi, 8H
|
||||
je final7blocks
|
||||
vpbroadcastd ymm0, dword ptr [rcx]
|
||||
vpbroadcastd ymm1, dword ptr [rcx+4H]
|
||||
vpbroadcastd ymm2, dword ptr [rcx+8H]
|
||||
vpbroadcastd ymm3, dword ptr [rcx+0CH]
|
||||
vpbroadcastd ymm4, dword ptr [rcx+10H]
|
||||
vpbroadcastd ymm5, dword ptr [rcx+14H]
|
||||
vpbroadcastd ymm6, dword ptr [rcx+18H]
|
||||
vpbroadcastd ymm7, dword ptr [rcx+1CH]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
mov r10, qword ptr [rdi+10H]
|
||||
mov r11, qword ptr [rdi+18H]
|
||||
mov r12, qword ptr [rdi+20H]
|
||||
mov r13, qword ptr [rdi+28H]
|
||||
mov r14, qword ptr [rdi+30H]
|
||||
mov r15, qword ptr [rdi+38H]
|
||||
movzx eax, byte ptr [rbp+78H]
|
||||
movzx ebx, byte ptr [rbp+80H]
|
||||
or eax, ebx
|
||||
xor edx, edx
|
||||
innerloop8:
|
||||
movzx ebx, byte ptr [rbp+88H]
|
||||
or ebx, eax
|
||||
add rdx, 64
|
||||
cmp rdx, qword ptr [rsp+80H]
|
||||
cmove eax, ebx
|
||||
mov dword ptr [rsp+88H], eax
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-40H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-40H], 01H
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-40H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-40H], 01H
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-40H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-40H], 01H
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-40H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-40H], 01H
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm16, ymm12, ymm14, 136
|
||||
vshufps ymm17, ymm12, ymm14, 221
|
||||
vshufps ymm18, ymm13, ymm15, 136
|
||||
vshufps ymm19, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-30H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-30H], 01H
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-30H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-30H], 01H
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-30H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-30H], 01H
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-30H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-30H], 01H
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm20, ymm12, ymm14, 136
|
||||
vshufps ymm21, ymm12, ymm14, 221
|
||||
vshufps ymm22, ymm13, ymm15, 136
|
||||
vshufps ymm23, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-20H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-20H], 01H
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-20H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-20H], 01H
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-20H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-20H], 01H
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-20H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-20H], 01H
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm24, ymm12, ymm14, 136
|
||||
vshufps ymm25, ymm12, ymm14, 221
|
||||
vshufps ymm26, ymm13, ymm15, 136
|
||||
vshufps ymm27, ymm13, ymm15, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-10H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r12+rdx-10H], 01H
|
||||
vmovups xmm9, xmmword ptr [r9+rdx-10H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r13+rdx-10H], 01H
|
||||
vunpcklpd ymm12, ymm8, ymm9
|
||||
vunpckhpd ymm13, ymm8, ymm9
|
||||
vmovups xmm10, xmmword ptr [r10+rdx-10H]
|
||||
vinsertf128 ymm10, ymm10, xmmword ptr [r14+rdx-10H], 01H
|
||||
vmovups xmm11, xmmword ptr [r11+rdx-10H]
|
||||
vinsertf128 ymm11, ymm11, xmmword ptr [r15+rdx-10H], 01H
|
||||
vunpcklpd ymm14, ymm10, ymm11
|
||||
vunpckhpd ymm15, ymm10, ymm11
|
||||
vshufps ymm28, ymm12, ymm14, 136
|
||||
vshufps ymm29, ymm12, ymm14, 221
|
||||
vshufps ymm30, ymm13, ymm15, 136
|
||||
vshufps ymm31, ymm13, ymm15, 221
|
||||
vpbroadcastd ymm8, dword ptr [BLAKE3_IV_0]
|
||||
vpbroadcastd ymm9, dword ptr [BLAKE3_IV_1]
|
||||
vpbroadcastd ymm10, dword ptr [BLAKE3_IV_2]
|
||||
vpbroadcastd ymm11, dword ptr [BLAKE3_IV_3]
|
||||
vmovdqa ymm12, ymmword ptr [rsp]
|
||||
vmovdqa ymm13, ymmword ptr [rsp+40H]
|
||||
vpbroadcastd ymm14, dword ptr [BLAKE3_BLOCK_LEN]
|
||||
vpbroadcastd ymm15, dword ptr [rsp+88H]
|
||||
vpaddd ymm0, ymm0, ymm16
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm20
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm17
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm21
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm24
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm28
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm25
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm29
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm18
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm23
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm22
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm16
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm17
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm25
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm27
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm30
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm19
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm29
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm20
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm18
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm22
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm27
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm21
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm31
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm26
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm30
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm23
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm19
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm20
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm21
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm16
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm24
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm28
|
||||
vpaddd ymm1, ymm1, ymm25
|
||||
vpaddd ymm2, ymm2, ymm31
|
||||
vpaddd ymm3, ymm3, ymm30
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm29
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm26
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm23
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm16
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm18
|
||||
vpaddd ymm1, ymm1, ymm19
|
||||
vpaddd ymm2, ymm2, ymm17
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm25
|
||||
vpaddd ymm1, ymm1, ymm27
|
||||
vpaddd ymm2, ymm2, ymm24
|
||||
vpaddd ymm3, ymm3, ymm31
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm30
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm28
|
||||
vpaddd ymm3, ymm3, ymm17
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm29
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm18
|
||||
vpaddd ymm3, ymm3, ymm20
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm19
|
||||
vpaddd ymm1, ymm1, ymm26
|
||||
vpaddd ymm2, ymm2, ymm22
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpaddd ymm0, ymm0, ymm27
|
||||
vpaddd ymm1, ymm1, ymm21
|
||||
vpaddd ymm2, ymm2, ymm17
|
||||
vpaddd ymm3, ymm3, ymm24
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vprord ymm15, ymm15, 16
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 12
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vpaddd ymm0, ymm0, ymm31
|
||||
vpaddd ymm1, ymm1, ymm16
|
||||
vpaddd ymm2, ymm2, ymm25
|
||||
vpaddd ymm3, ymm3, ymm22
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm1, ymm1, ymm5
|
||||
vpaddd ymm2, ymm2, ymm6
|
||||
vpaddd ymm3, ymm3, ymm7
|
||||
vpxord ymm12, ymm12, ymm0
|
||||
vpxord ymm13, ymm13, ymm1
|
||||
vpxord ymm14, ymm14, ymm2
|
||||
vpxord ymm15, ymm15, ymm3
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vprord ymm15, ymm15, 8
|
||||
vpaddd ymm8, ymm8, ymm12
|
||||
vpaddd ymm9, ymm9, ymm13
|
||||
vpaddd ymm10, ymm10, ymm14
|
||||
vpaddd ymm11, ymm11, ymm15
|
||||
vpxord ymm4, ymm4, ymm8
|
||||
vpxord ymm5, ymm5, ymm9
|
||||
vpxord ymm6, ymm6, ymm10
|
||||
vpxord ymm7, ymm7, ymm11
|
||||
vprord ymm4, ymm4, 7
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vpaddd ymm0, ymm0, ymm30
|
||||
vpaddd ymm1, ymm1, ymm18
|
||||
vpaddd ymm2, ymm2, ymm19
|
||||
vpaddd ymm3, ymm3, ymm23
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 16
|
||||
vprord ymm12, ymm12, 16
|
||||
vprord ymm13, ymm13, 16
|
||||
vprord ymm14, ymm14, 16
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 12
|
||||
vprord ymm6, ymm6, 12
|
||||
vprord ymm7, ymm7, 12
|
||||
vprord ymm4, ymm4, 12
|
||||
vpaddd ymm0, ymm0, ymm26
|
||||
vpaddd ymm1, ymm1, ymm28
|
||||
vpaddd ymm2, ymm2, ymm20
|
||||
vpaddd ymm3, ymm3, ymm29
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm1, ymm1, ymm6
|
||||
vpaddd ymm2, ymm2, ymm7
|
||||
vpaddd ymm3, ymm3, ymm4
|
||||
vpxord ymm15, ymm15, ymm0
|
||||
vpxord ymm12, ymm12, ymm1
|
||||
vpxord ymm13, ymm13, ymm2
|
||||
vpxord ymm14, ymm14, ymm3
|
||||
vprord ymm15, ymm15, 8
|
||||
vprord ymm12, ymm12, 8
|
||||
vprord ymm13, ymm13, 8
|
||||
vprord ymm14, ymm14, 8
|
||||
vpaddd ymm10, ymm10, ymm15
|
||||
vpaddd ymm11, ymm11, ymm12
|
||||
vpaddd ymm8, ymm8, ymm13
|
||||
vpaddd ymm9, ymm9, ymm14
|
||||
vpxord ymm5, ymm5, ymm10
|
||||
vpxord ymm6, ymm6, ymm11
|
||||
vpxord ymm7, ymm7, ymm8
|
||||
vpxord ymm4, ymm4, ymm9
|
||||
vprord ymm5, ymm5, 7
|
||||
vprord ymm6, ymm6, 7
|
||||
vprord ymm7, ymm7, 7
|
||||
vprord ymm4, ymm4, 7
|
||||
vpxor ymm0, ymm0, ymm8
|
||||
vpxor ymm1, ymm1, ymm9
|
||||
vpxor ymm2, ymm2, ymm10
|
||||
vpxor ymm3, ymm3, ymm11
|
||||
vpxor ymm4, ymm4, ymm12
|
||||
vpxor ymm5, ymm5, ymm13
|
||||
vpxor ymm6, ymm6, ymm14
|
||||
vpxor ymm7, ymm7, ymm15
|
||||
movzx eax, byte ptr [rbp+78H]
|
||||
jne innerloop8
|
||||
mov rbx, qword ptr [rbp+90H]
|
||||
vunpcklps ymm8, ymm0, ymm1
|
||||
vunpcklps ymm9, ymm2, ymm3
|
||||
vunpckhps ymm10, ymm0, ymm1
|
||||
vunpcklps ymm11, ymm4, ymm5
|
||||
vunpcklps ymm0, ymm6, ymm7
|
||||
vshufps ymm12, ymm8, ymm9, 78
|
||||
vblendps ymm1, ymm8, ymm12, 0CCH
|
||||
vshufps ymm8, ymm11, ymm0, 78
|
||||
vunpckhps ymm13, ymm2, ymm3
|
||||
vblendps ymm2, ymm11, ymm8, 0CCH
|
||||
vblendps ymm3, ymm12, ymm9, 0CCH
|
||||
vperm2f128 ymm12, ymm1, ymm2, 20H
|
||||
vmovups ymmword ptr [rbx], ymm12
|
||||
vunpckhps ymm14, ymm4, ymm5
|
||||
vblendps ymm4, ymm8, ymm0, 0CCH
|
||||
vunpckhps ymm15, ymm6, ymm7
|
||||
vperm2f128 ymm7, ymm3, ymm4, 20H
|
||||
vmovups ymmword ptr [rbx+20H], ymm7
|
||||
vshufps ymm5, ymm10, ymm13, 78
|
||||
vblendps ymm6, ymm5, ymm13, 0CCH
|
||||
vshufps ymm13, ymm14, ymm15, 78
|
||||
vblendps ymm10, ymm10, ymm5, 0CCH
|
||||
vblendps ymm14, ymm14, ymm13, 0CCH
|
||||
vperm2f128 ymm8, ymm10, ymm14, 20H
|
||||
vmovups ymmword ptr [rbx+40H], ymm8
|
||||
vblendps ymm15, ymm13, ymm15, 0CCH
|
||||
vperm2f128 ymm13, ymm6, ymm15, 20H
|
||||
vmovups ymmword ptr [rbx+60H], ymm13
|
||||
vperm2f128 ymm9, ymm1, ymm2, 31H
|
||||
vperm2f128 ymm11, ymm3, ymm4, 31H
|
||||
vmovups ymmword ptr [rbx+80H], ymm9
|
||||
vperm2f128 ymm14, ymm10, ymm14, 31H
|
||||
vperm2f128 ymm15, ymm6, ymm15, 31H
|
||||
vmovups ymmword ptr [rbx+0A0H], ymm11
|
||||
vmovups ymmword ptr [rbx+0C0H], ymm14
|
||||
vmovups ymmword ptr [rbx+0E0H], ymm15
|
||||
vmovdqa ymm0, ymmword ptr [rsp]
|
||||
vmovdqa ymm2, ymmword ptr [rsp+40H]
|
||||
vmovdqa32 ymm0 {k1}, ymmword ptr [rsp+1H*20H]
|
||||
vmovdqa32 ymm2 {k1}, ymmword ptr [rsp+3H*20H]
|
||||
vmovdqa ymmword ptr [rsp], ymm0
|
||||
vmovdqa ymmword ptr [rsp+40H], ymm2
|
||||
add rbx, 256
|
||||
mov qword ptr [rbp+90H], rbx
|
||||
add rdi, 64
|
||||
sub rsi, 8
|
||||
final7blocks:
|
||||
mov rbx, qword ptr [rbp+90H]
|
||||
mov r15, qword ptr [rsp+80H]
|
||||
movzx r13, byte ptr [rbp+78H]
|
||||
movzx r12, byte ptr [rbp+88H]
|
||||
test esi, 4H
|
||||
je final3blocks
|
||||
vbroadcasti32x4 zmm0, xmmword ptr [rcx]
|
||||
vbroadcasti32x4 zmm1, xmmword ptr [rcx+1H*10H]
|
||||
vmovdqa xmm12, xmmword ptr [rsp]
|
||||
vmovdqa xmm13, xmmword ptr [rsp+40H]
|
||||
vpunpckldq xmm14, xmm12, xmm13
|
||||
vpunpckhdq xmm15, xmm12, xmm13
|
||||
vpermq ymm14, ymm14, 0DCH
|
||||
vpermq ymm15, ymm15, 0DCH
|
||||
vpbroadcastd zmm12, dword ptr [BLAKE3_BLOCK_LEN]
|
||||
vinserti32x8 zmm13, zmm14, ymm15, 01H
|
||||
mov eax, 17476
|
||||
kmovw k2, eax
|
||||
vpblendmd zmm13 {k2}, zmm13, zmm12
|
||||
vbroadcasti32x4 zmm15, xmmword ptr [BLAKE3_IV]
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
mov r10, qword ptr [rdi+10H]
|
||||
mov r11, qword ptr [rdi+18H]
|
||||
mov eax, 43690
|
||||
kmovw k3, eax
|
||||
mov eax, 34952
|
||||
kmovw k4, eax
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
ALIGN 16
|
||||
innerloop4:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+88H], eax
|
||||
vmovdqa32 zmm2, zmm15
|
||||
vpbroadcastd zmm8, dword ptr [rsp+22H*4H]
|
||||
vpblendmd zmm3 {k4}, zmm13, zmm8
|
||||
vmovups zmm8, zmmword ptr [r8+rdx-1H*40H]
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r9+rdx-4H*10H], 01H
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r10+rdx-4H*10H], 02H
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r11+rdx-4H*10H], 03H
|
||||
vmovups zmm9, zmmword ptr [r8+rdx-30H]
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r9+rdx-3H*10H], 01H
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r10+rdx-3H*10H], 02H
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r11+rdx-3H*10H], 03H
|
||||
vshufps zmm4, zmm8, zmm9, 136
|
||||
vshufps zmm5, zmm8, zmm9, 221
|
||||
vmovups zmm8, zmmword ptr [r8+rdx-20H]
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r9+rdx-2H*10H], 01H
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r10+rdx-2H*10H], 02H
|
||||
vinserti32x4 zmm8, zmm8, xmmword ptr [r11+rdx-2H*10H], 03H
|
||||
vmovups zmm9, zmmword ptr [r8+rdx-10H]
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r9+rdx-1H*10H], 01H
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r10+rdx-1H*10H], 02H
|
||||
vinserti32x4 zmm9, zmm9, xmmword ptr [r11+rdx-1H*10H], 03H
|
||||
vshufps zmm6, zmm8, zmm9, 136
|
||||
vshufps zmm7, zmm8, zmm9, 221
|
||||
vpshufd zmm6, zmm6, 93H
|
||||
vpshufd zmm7, zmm7, 93H
|
||||
mov al, 7
|
||||
roundloop4:
|
||||
vpaddd zmm0, zmm0, zmm4
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 16
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 12
|
||||
vpaddd zmm0, zmm0, zmm5
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 8
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 7
|
||||
vpshufd zmm0, zmm0, 93H
|
||||
vpshufd zmm3, zmm3, 4EH
|
||||
vpshufd zmm2, zmm2, 39H
|
||||
vpaddd zmm0, zmm0, zmm6
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 16
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 12
|
||||
vpaddd zmm0, zmm0, zmm7
|
||||
vpaddd zmm0, zmm0, zmm1
|
||||
vpxord zmm3, zmm3, zmm0
|
||||
vprord zmm3, zmm3, 8
|
||||
vpaddd zmm2, zmm2, zmm3
|
||||
vpxord zmm1, zmm1, zmm2
|
||||
vprord zmm1, zmm1, 7
|
||||
vpshufd zmm0, zmm0, 39H
|
||||
vpshufd zmm3, zmm3, 4EH
|
||||
vpshufd zmm2, zmm2, 93H
|
||||
dec al
|
||||
jz endroundloop4
|
||||
vshufps zmm8, zmm4, zmm5, 214
|
||||
vpshufd zmm9, zmm4, 0FH
|
||||
vpshufd zmm4, zmm8, 39H
|
||||
vshufps zmm8, zmm6, zmm7, 250
|
||||
vpblendmd zmm9 {k3}, zmm9, zmm8
|
||||
vpunpcklqdq zmm8, zmm7, zmm5
|
||||
vpblendmd zmm8 {k4}, zmm8, zmm6
|
||||
vpshufd zmm8, zmm8, 78H
|
||||
vpunpckhdq zmm5, zmm5, zmm7
|
||||
vpunpckldq zmm6, zmm6, zmm5
|
||||
vpshufd zmm7, zmm6, 1EH
|
||||
vmovdqa32 zmm5, zmm9
|
||||
vmovdqa32 zmm6, zmm8
|
||||
jmp roundloop4
|
||||
endroundloop4:
|
||||
vpxord zmm0, zmm0, zmm2
|
||||
vpxord zmm1, zmm1, zmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne innerloop4
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+10H], xmm1
|
||||
vextracti128 xmmword ptr [rbx+20H], ymm0, 01H
|
||||
vextracti128 xmmword ptr [rbx+30H], ymm1, 01H
|
||||
vextracti32x4 xmmword ptr [rbx+4H*10H], zmm0, 02H
|
||||
vextracti32x4 xmmword ptr [rbx+5H*10H], zmm1, 02H
|
||||
vextracti32x4 xmmword ptr [rbx+6H*10H], zmm0, 03H
|
||||
vextracti32x4 xmmword ptr [rbx+7H*10H], zmm1, 03H
|
||||
vmovdqa xmm0, xmmword ptr [rsp]
|
||||
vmovdqa xmm2, xmmword ptr [rsp+40H]
|
||||
vmovdqa32 xmm0 {k1}, xmmword ptr [rsp+1H*10H]
|
||||
vmovdqa32 xmm2 {k1}, xmmword ptr [rsp+5H*10H]
|
||||
vmovdqa xmmword ptr [rsp], xmm0
|
||||
vmovdqa xmmword ptr [rsp+40H], xmm2
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
final3blocks:
|
||||
test esi, 2H
|
||||
je final1block
|
||||
vbroadcasti128 ymm0, xmmword ptr [rcx]
|
||||
vbroadcasti128 ymm1, xmmword ptr [rcx+10H]
|
||||
vmovd xmm13, dword ptr [rsp]
|
||||
vpinsrd xmm13, xmm13, dword ptr [rsp+40H], 1
|
||||
vpinsrd xmm13, xmm13, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
vmovd xmm14, dword ptr [rsp+4H]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+44H], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
vinserti128 ymm13, ymm13, xmm14, 01H
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
ALIGN 16
|
||||
innerloop2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
mov dword ptr [rsp+88H], eax
|
||||
vbroadcasti128 ymm2, xmmword ptr [BLAKE3_IV]
|
||||
vpbroadcastd ymm8, dword ptr [rsp+88H]
|
||||
vpblendd ymm3, ymm13, ymm8, 88H
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-40H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-40H], 01H
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-30H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-30H], 01H
|
||||
vshufps ymm4, ymm8, ymm9, 136
|
||||
vshufps ymm5, ymm8, ymm9, 221
|
||||
vmovups ymm8, ymmword ptr [r8+rdx-20H]
|
||||
vinsertf128 ymm8, ymm8, xmmword ptr [r9+rdx-20H], 01H
|
||||
vmovups ymm9, ymmword ptr [r8+rdx-10H]
|
||||
vinsertf128 ymm9, ymm9, xmmword ptr [r9+rdx-10H], 01H
|
||||
vshufps ymm6, ymm8, ymm9, 136
|
||||
vshufps ymm7, ymm8, ymm9, 221
|
||||
vpshufd ymm6, ymm6, 93H
|
||||
vpshufd ymm7, ymm7, 93H
|
||||
mov al, 7
|
||||
roundloop2:
|
||||
vpaddd ymm0, ymm0, ymm4
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 16
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 12
|
||||
vpaddd ymm0, ymm0, ymm5
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 8
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 7
|
||||
vpshufd ymm0, ymm0, 93H
|
||||
vpshufd ymm3, ymm3, 4EH
|
||||
vpshufd ymm2, ymm2, 39H
|
||||
vpaddd ymm0, ymm0, ymm6
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 16
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 12
|
||||
vpaddd ymm0, ymm0, ymm7
|
||||
vpaddd ymm0, ymm0, ymm1
|
||||
vpxord ymm3, ymm3, ymm0
|
||||
vprord ymm3, ymm3, 8
|
||||
vpaddd ymm2, ymm2, ymm3
|
||||
vpxord ymm1, ymm1, ymm2
|
||||
vprord ymm1, ymm1, 7
|
||||
vpshufd ymm0, ymm0, 39H
|
||||
vpshufd ymm3, ymm3, 4EH
|
||||
vpshufd ymm2, ymm2, 93H
|
||||
dec al
|
||||
jz endroundloop2
|
||||
vshufps ymm8, ymm4, ymm5, 214
|
||||
vpshufd ymm9, ymm4, 0FH
|
||||
vpshufd ymm4, ymm8, 39H
|
||||
vshufps ymm8, ymm6, ymm7, 250
|
||||
vpblendd ymm9, ymm9, ymm8, 0AAH
|
||||
vpunpcklqdq ymm8, ymm7, ymm5
|
||||
vpblendd ymm8, ymm8, ymm6, 88H
|
||||
vpshufd ymm8, ymm8, 78H
|
||||
vpunpckhdq ymm5, ymm5, ymm7
|
||||
vpunpckldq ymm6, ymm6, ymm5
|
||||
vpshufd ymm7, ymm6, 1EH
|
||||
vmovdqa ymm5, ymm9
|
||||
vmovdqa ymm6, ymm8
|
||||
jmp roundloop2
|
||||
endroundloop2:
|
||||
vpxor ymm0, ymm0, ymm2
|
||||
vpxor ymm1, ymm1, ymm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne innerloop2
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+10H], xmm1
|
||||
vextracti128 xmmword ptr [rbx+20H], ymm0, 01H
|
||||
vextracti128 xmmword ptr [rbx+30H], ymm1, 01H
|
||||
vmovdqa xmm0, xmmword ptr [rsp]
|
||||
vmovdqa xmm2, xmmword ptr [rsp+40H]
|
||||
vmovdqu32 xmm0 {k1}, xmmword ptr [rsp+8H]
|
||||
vmovdqu32 xmm2 {k1}, xmmword ptr [rsp+48H]
|
||||
vmovdqa xmmword ptr [rsp], xmm0
|
||||
vmovdqa xmmword ptr [rsp+40H], xmm2
|
||||
add rbx, 64
|
||||
add rdi, 16
|
||||
sub rsi, 2
|
||||
final1block:
|
||||
test esi, 1H
|
||||
je unwind
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+10H]
|
||||
vmovd xmm14, dword ptr [rsp]
|
||||
vpinsrd xmm14, xmm14, dword ptr [rsp+40H], 1
|
||||
vpinsrd xmm14, xmm14, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
vmovdqa xmm15, xmmword ptr [BLAKE3_IV]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
ALIGN 16
|
||||
innerloop1:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
vpinsrd xmm3, xmm14, eax, 3
|
||||
vmovdqa xmm2, xmm15
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-40H]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-30H]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [r8+rdx-20H]
|
||||
vmovups xmm9, xmmword ptr [r8+rdx-10H]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 93H
|
||||
vpshufd xmm7, xmm7, 93H
|
||||
mov al, 7
|
||||
roundloop1:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 93H
|
||||
vpshufd xmm3, xmm3, 4EH
|
||||
vpshufd xmm2, xmm2, 39H
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 39H
|
||||
vpshufd xmm3, xmm3, 4EH
|
||||
vpshufd xmm2, xmm2, 93H
|
||||
dec al
|
||||
jz endroundloop1
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0FH
|
||||
vpshufd xmm4, xmm8, 39H
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0AAH
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 88H
|
||||
vpshufd xmm8, xmm8, 78H
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 1EH
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp roundloop1
|
||||
endroundloop1:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne innerloop1
|
||||
vmovdqu xmmword ptr [rbx], xmm0
|
||||
vmovdqu xmmword ptr [rbx+10H], xmm1
|
||||
jmp unwind
|
||||
|
||||
_blake3_hash_many_avx512 ENDP
|
||||
blake3_hash_many_avx512 ENDP
|
||||
|
||||
ALIGN 16
|
||||
blake3_compress_in_place_avx512 PROC
|
||||
_blake3_compress_in_place_avx512 PROC
|
||||
sub rsp, 72
|
||||
vmovdqa xmmword ptr [rsp], xmm6
|
||||
vmovdqa xmmword ptr [rsp+10H], xmm7
|
||||
vmovdqa xmmword ptr [rsp+20H], xmm8
|
||||
vmovdqa xmmword ptr [rsp+30H], xmm9
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+10H]
|
||||
movzx eax, byte ptr [rsp+70H]
|
||||
movzx r8d, r8b
|
||||
shl rax, 32
|
||||
add r8, rax
|
||||
vmovq xmm3, r9
|
||||
vmovq xmm4, r8
|
||||
vpunpcklqdq xmm3, xmm3, xmm4
|
||||
vmovaps xmm2, xmmword ptr [BLAKE3_IV]
|
||||
vmovups xmm8, xmmword ptr [rdx]
|
||||
vmovups xmm9, xmmword ptr [rdx+10H]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [rdx+20H]
|
||||
vmovups xmm9, xmmword ptr [rdx+30H]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 93H
|
||||
vpshufd xmm7, xmm7, 93H
|
||||
mov al, 7
|
||||
@@:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 93H
|
||||
vpshufd xmm3, xmm3, 4EH
|
||||
vpshufd xmm2, xmm2, 39H
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 39H
|
||||
vpshufd xmm3, xmm3, 4EH
|
||||
vpshufd xmm2, xmm2, 93H
|
||||
dec al
|
||||
jz @F
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0FH
|
||||
vpshufd xmm4, xmm8, 39H
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0AAH
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 88H
|
||||
vpshufd xmm8, xmm8, 78H
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 1EH
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp @B
|
||||
@@:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
vmovdqu xmmword ptr [rcx], xmm0
|
||||
vmovdqu xmmword ptr [rcx+10H], xmm1
|
||||
vmovdqa xmm6, xmmword ptr [rsp]
|
||||
vmovdqa xmm7, xmmword ptr [rsp+10H]
|
||||
vmovdqa xmm8, xmmword ptr [rsp+20H]
|
||||
vmovdqa xmm9, xmmword ptr [rsp+30H]
|
||||
add rsp, 72
|
||||
ret
|
||||
_blake3_compress_in_place_avx512 ENDP
|
||||
blake3_compress_in_place_avx512 ENDP
|
||||
|
||||
ALIGN 16
|
||||
blake3_compress_xof_avx512 PROC
|
||||
_blake3_compress_xof_avx512 PROC
|
||||
sub rsp, 72
|
||||
vmovdqa xmmword ptr [rsp], xmm6
|
||||
vmovdqa xmmword ptr [rsp+10H], xmm7
|
||||
vmovdqa xmmword ptr [rsp+20H], xmm8
|
||||
vmovdqa xmmword ptr [rsp+30H], xmm9
|
||||
vmovdqu xmm0, xmmword ptr [rcx]
|
||||
vmovdqu xmm1, xmmword ptr [rcx+10H]
|
||||
movzx eax, byte ptr [rsp+70H]
|
||||
movzx r8d, r8b
|
||||
mov r10, qword ptr [rsp+78H]
|
||||
shl rax, 32
|
||||
add r8, rax
|
||||
vmovq xmm3, r9
|
||||
vmovq xmm4, r8
|
||||
vpunpcklqdq xmm3, xmm3, xmm4
|
||||
vmovaps xmm2, xmmword ptr [BLAKE3_IV]
|
||||
vmovups xmm8, xmmword ptr [rdx]
|
||||
vmovups xmm9, xmmword ptr [rdx+10H]
|
||||
vshufps xmm4, xmm8, xmm9, 136
|
||||
vshufps xmm5, xmm8, xmm9, 221
|
||||
vmovups xmm8, xmmword ptr [rdx+20H]
|
||||
vmovups xmm9, xmmword ptr [rdx+30H]
|
||||
vshufps xmm6, xmm8, xmm9, 136
|
||||
vshufps xmm7, xmm8, xmm9, 221
|
||||
vpshufd xmm6, xmm6, 93H
|
||||
vpshufd xmm7, xmm7, 93H
|
||||
mov al, 7
|
||||
@@:
|
||||
vpaddd xmm0, xmm0, xmm4
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm5
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 93H
|
||||
vpshufd xmm3, xmm3, 4EH
|
||||
vpshufd xmm2, xmm2, 39H
|
||||
vpaddd xmm0, xmm0, xmm6
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 16
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 12
|
||||
vpaddd xmm0, xmm0, xmm7
|
||||
vpaddd xmm0, xmm0, xmm1
|
||||
vpxord xmm3, xmm3, xmm0
|
||||
vprord xmm3, xmm3, 8
|
||||
vpaddd xmm2, xmm2, xmm3
|
||||
vpxord xmm1, xmm1, xmm2
|
||||
vprord xmm1, xmm1, 7
|
||||
vpshufd xmm0, xmm0, 39H
|
||||
vpshufd xmm3, xmm3, 4EH
|
||||
vpshufd xmm2, xmm2, 93H
|
||||
dec al
|
||||
jz @F
|
||||
vshufps xmm8, xmm4, xmm5, 214
|
||||
vpshufd xmm9, xmm4, 0FH
|
||||
vpshufd xmm4, xmm8, 39H
|
||||
vshufps xmm8, xmm6, xmm7, 250
|
||||
vpblendd xmm9, xmm9, xmm8, 0AAH
|
||||
vpunpcklqdq xmm8, xmm7, xmm5
|
||||
vpblendd xmm8, xmm8, xmm6, 88H
|
||||
vpshufd xmm8, xmm8, 78H
|
||||
vpunpckhdq xmm5, xmm5, xmm7
|
||||
vpunpckldq xmm6, xmm6, xmm5
|
||||
vpshufd xmm7, xmm6, 1EH
|
||||
vmovdqa xmm5, xmm9
|
||||
vmovdqa xmm6, xmm8
|
||||
jmp @B
|
||||
@@:
|
||||
vpxor xmm0, xmm0, xmm2
|
||||
vpxor xmm1, xmm1, xmm3
|
||||
vpxor xmm2, xmm2, xmmword ptr [rcx]
|
||||
vpxor xmm3, xmm3, xmmword ptr [rcx+10H]
|
||||
vmovdqu xmmword ptr [r10], xmm0
|
||||
vmovdqu xmmword ptr [r10+10H], xmm1
|
||||
vmovdqu xmmword ptr [r10+20H], xmm2
|
||||
vmovdqu xmmword ptr [r10+30H], xmm3
|
||||
vmovdqa xmm6, xmmword ptr [rsp]
|
||||
vmovdqa xmm7, xmmword ptr [rsp+10H]
|
||||
vmovdqa xmm8, xmmword ptr [rsp+20H]
|
||||
vmovdqa xmm9, xmmword ptr [rsp+30H]
|
||||
add rsp, 72
|
||||
ret
|
||||
_blake3_compress_xof_avx512 ENDP
|
||||
blake3_compress_xof_avx512 ENDP
|
||||
|
||||
_TEXT ENDS
|
||||
|
||||
_RDATA SEGMENT READONLY PAGE ALIAS(".rdata") 'CONST'
|
||||
ALIGN 64
|
||||
INDEX0:
|
||||
dd 0, 1, 2, 3, 16, 17, 18, 19
|
||||
dd 8, 9, 10, 11, 24, 25, 26, 27
|
||||
INDEX1:
|
||||
dd 4, 5, 6, 7, 20, 21, 22, 23
|
||||
dd 12, 13, 14, 15, 28, 29, 30, 31
|
||||
ADD0:
|
||||
dd 0, 1, 2, 3, 4, 5, 6, 7
|
||||
dd 8, 9, 10, 11, 12, 13, 14, 15
|
||||
ADD1:
|
||||
dd 1
|
||||
ADD16:
|
||||
dd 16
|
||||
BLAKE3_BLOCK_LEN:
|
||||
dd 64
|
||||
ALIGN 64
|
||||
BLAKE3_IV:
|
||||
BLAKE3_IV_0:
|
||||
dd 06A09E667H
|
||||
BLAKE3_IV_1:
|
||||
dd 0BB67AE85H
|
||||
BLAKE3_IV_2:
|
||||
dd 03C6EF372H
|
||||
BLAKE3_IV_3:
|
||||
dd 0A54FF53AH
|
||||
|
||||
_RDATA ENDS
|
||||
END
|
|
@ -0,0 +1,2011 @@
|
|||
.intel_syntax noprefix
|
||||
.global blake3_hash_many_sse41
|
||||
.global _blake3_hash_many_sse41
|
||||
.global blake3_compress_in_place_sse41
|
||||
.global _blake3_compress_in_place_sse41
|
||||
.global blake3_compress_xof_sse41
|
||||
.global _blake3_compress_xof_sse41
|
||||
#ifdef __APPLE__
|
||||
.text
|
||||
#else
|
||||
.section .text
|
||||
#endif
|
||||
.p2align 6
|
||||
_blake3_hash_many_sse41:
|
||||
blake3_hash_many_sse41:
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 360
|
||||
and rsp, 0xFFFFFFFFFFFFFFC0
|
||||
neg r9d
|
||||
movd xmm0, r9d
|
||||
pshufd xmm0, xmm0, 0x00
|
||||
movdqa xmmword ptr [rsp+0x130], xmm0
|
||||
movdqa xmm1, xmm0
|
||||
pand xmm1, xmmword ptr [ADD0+rip]
|
||||
pand xmm0, xmmword ptr [ADD1+rip]
|
||||
movdqa xmmword ptr [rsp+0x150], xmm0
|
||||
movd xmm0, r8d
|
||||
pshufd xmm0, xmm0, 0x00
|
||||
paddd xmm0, xmm1
|
||||
movdqa xmmword ptr [rsp+0x110], xmm0
|
||||
pxor xmm0, xmmword ptr [CMP_MSB_MASK+rip]
|
||||
pxor xmm1, xmmword ptr [CMP_MSB_MASK+rip]
|
||||
pcmpgtd xmm1, xmm0
|
||||
shr r8, 32
|
||||
movd xmm2, r8d
|
||||
pshufd xmm2, xmm2, 0x00
|
||||
psubd xmm2, xmm1
|
||||
movdqa xmmword ptr [rsp+0x120], xmm2
|
||||
mov rbx, qword ptr [rbp+0x50]
|
||||
mov r15, rdx
|
||||
shl r15, 6
|
||||
movzx r13d, byte ptr [rbp+0x38]
|
||||
movzx r12d, byte ptr [rbp+0x48]
|
||||
cmp rsi, 4
|
||||
jc 3f
|
||||
2:
|
||||
movdqu xmm3, xmmword ptr [rcx]
|
||||
pshufd xmm0, xmm3, 0x00
|
||||
pshufd xmm1, xmm3, 0x55
|
||||
pshufd xmm2, xmm3, 0xAA
|
||||
pshufd xmm3, xmm3, 0xFF
|
||||
movdqu xmm7, xmmword ptr [rcx+0x10]
|
||||
pshufd xmm4, xmm7, 0x00
|
||||
pshufd xmm5, xmm7, 0x55
|
||||
pshufd xmm6, xmm7, 0xAA
|
||||
pshufd xmm7, xmm7, 0xFF
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
1:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-0x40]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-0x40]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-0x40]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp], xmm8
|
||||
movdqa xmmword ptr [rsp+0x10], xmm9
|
||||
movdqa xmmword ptr [rsp+0x20], xmm12
|
||||
movdqa xmmword ptr [rsp+0x30], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-0x30]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-0x30]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-0x30]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-0x30]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+0x40], xmm8
|
||||
movdqa xmmword ptr [rsp+0x50], xmm9
|
||||
movdqa xmmword ptr [rsp+0x60], xmm12
|
||||
movdqa xmmword ptr [rsp+0x70], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-0x20]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-0x20]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-0x20]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+0x80], xmm8
|
||||
movdqa xmmword ptr [rsp+0x90], xmm9
|
||||
movdqa xmmword ptr [rsp+0xA0], xmm12
|
||||
movdqa xmmword ptr [rsp+0xB0], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-0x10]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-0x10]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-0x10]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-0x10]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+0xC0], xmm8
|
||||
movdqa xmmword ptr [rsp+0xD0], xmm9
|
||||
movdqa xmmword ptr [rsp+0xE0], xmm12
|
||||
movdqa xmmword ptr [rsp+0xF0], xmm13
|
||||
movdqa xmm9, xmmword ptr [BLAKE3_IV_1+rip]
|
||||
movdqa xmm10, xmmword ptr [BLAKE3_IV_2+rip]
|
||||
movdqa xmm11, xmmword ptr [BLAKE3_IV_3+rip]
|
||||
movdqa xmm12, xmmword ptr [rsp+0x110]
|
||||
movdqa xmm13, xmmword ptr [rsp+0x120]
|
||||
movdqa xmm14, xmmword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
movd xmm15, eax
|
||||
pshufd xmm15, xmm15, 0x00
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
paddd xmm0, xmmword ptr [rsp]
|
||||
paddd xmm1, xmmword ptr [rsp+0x20]
|
||||
paddd xmm2, xmmword ptr [rsp+0x40]
|
||||
paddd xmm3, xmmword ptr [rsp+0x60]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [BLAKE3_IV_0+rip]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x10]
|
||||
paddd xmm1, xmmword ptr [rsp+0x30]
|
||||
paddd xmm2, xmmword ptr [rsp+0x50]
|
||||
paddd xmm3, xmmword ptr [rsp+0x70]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x80]
|
||||
paddd xmm1, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm3, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x90]
|
||||
paddd xmm1, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm3, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x20]
|
||||
paddd xmm1, xmmword ptr [rsp+0x30]
|
||||
paddd xmm2, xmmword ptr [rsp+0x70]
|
||||
paddd xmm3, xmmword ptr [rsp+0x40]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x60]
|
||||
paddd xmm1, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm2, xmmword ptr [rsp]
|
||||
paddd xmm3, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x10]
|
||||
paddd xmm1, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x90]
|
||||
paddd xmm3, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x50]
|
||||
paddd xmm2, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x80]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x30]
|
||||
paddd xmm1, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x70]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x40]
|
||||
paddd xmm1, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x20]
|
||||
paddd xmm3, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x60]
|
||||
paddd xmm1, xmmword ptr [rsp+0x90]
|
||||
paddd xmm2, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x80]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x50]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x10]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm1, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm3, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x70]
|
||||
paddd xmm1, xmmword ptr [rsp+0x90]
|
||||
paddd xmm2, xmmword ptr [rsp+0x30]
|
||||
paddd xmm3, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x40]
|
||||
paddd xmm1, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x50]
|
||||
paddd xmm3, xmmword ptr [rsp+0x10]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp]
|
||||
paddd xmm1, xmmword ptr [rsp+0x20]
|
||||
paddd xmm2, xmmword ptr [rsp+0x80]
|
||||
paddd xmm3, xmmword ptr [rsp+0x60]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x90]
|
||||
paddd xmm2, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm3, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm1, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x80]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x70]
|
||||
paddd xmm1, xmmword ptr [rsp+0x50]
|
||||
paddd xmm2, xmmword ptr [rsp]
|
||||
paddd xmm3, xmmword ptr [rsp+0x60]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x20]
|
||||
paddd xmm1, xmmword ptr [rsp+0x30]
|
||||
paddd xmm2, xmmword ptr [rsp+0x10]
|
||||
paddd xmm3, xmmword ptr [rsp+0x40]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x90]
|
||||
paddd xmm1, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x80]
|
||||
paddd xmm3, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x50]
|
||||
paddd xmm2, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x10]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+0x20]
|
||||
paddd xmm3, xmmword ptr [rsp+0x40]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x30]
|
||||
paddd xmm1, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x60]
|
||||
paddd xmm3, xmmword ptr [rsp+0x70]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x50]
|
||||
paddd xmm2, xmmword ptr [rsp+0x10]
|
||||
paddd xmm3, xmmword ptr [rsp+0x80]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+0x90]
|
||||
paddd xmm3, xmmword ptr [rsp+0x60]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x20]
|
||||
paddd xmm2, xmmword ptr [rsp+0x30]
|
||||
paddd xmm3, xmmword ptr [rsp+0x70]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm1, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x40]
|
||||
paddd xmm3, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
pxor xmm0, xmm8
|
||||
pxor xmm1, xmm9
|
||||
pxor xmm2, xmm10
|
||||
pxor xmm3, xmm11
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
pxor xmm4, xmm12
|
||||
pxor xmm5, xmm13
|
||||
pxor xmm6, xmm14
|
||||
pxor xmm7, xmm15
|
||||
mov eax, r13d
|
||||
jne 1b
|
||||
movdqa xmm9, xmm0
|
||||
punpckldq xmm0, xmm1
|
||||
punpckhdq xmm9, xmm1
|
||||
movdqa xmm11, xmm2
|
||||
punpckldq xmm2, xmm3
|
||||
punpckhdq xmm11, xmm3
|
||||
movdqa xmm1, xmm0
|
||||
punpcklqdq xmm0, xmm2
|
||||
punpckhqdq xmm1, xmm2
|
||||
movdqa xmm3, xmm9
|
||||
punpcklqdq xmm9, xmm11
|
||||
punpckhqdq xmm3, xmm11
|
||||
movdqu xmmword ptr [rbx], xmm0
|
||||
movdqu xmmword ptr [rbx+0x20], xmm1
|
||||
movdqu xmmword ptr [rbx+0x40], xmm9
|
||||
movdqu xmmword ptr [rbx+0x60], xmm3
|
||||
movdqa xmm9, xmm4
|
||||
punpckldq xmm4, xmm5
|
||||
punpckhdq xmm9, xmm5
|
||||
movdqa xmm11, xmm6
|
||||
punpckldq xmm6, xmm7
|
||||
punpckhdq xmm11, xmm7
|
||||
movdqa xmm5, xmm4
|
||||
punpcklqdq xmm4, xmm6
|
||||
punpckhqdq xmm5, xmm6
|
||||
movdqa xmm7, xmm9
|
||||
punpcklqdq xmm9, xmm11
|
||||
punpckhqdq xmm7, xmm11
|
||||
movdqu xmmword ptr [rbx+0x10], xmm4
|
||||
movdqu xmmword ptr [rbx+0x30], xmm5
|
||||
movdqu xmmword ptr [rbx+0x50], xmm9
|
||||
movdqu xmmword ptr [rbx+0x70], xmm7
|
||||
movdqa xmm1, xmmword ptr [rsp+0x110]
|
||||
movdqa xmm0, xmm1
|
||||
paddd xmm1, xmmword ptr [rsp+0x150]
|
||||
movdqa xmmword ptr [rsp+0x110], xmm1
|
||||
pxor xmm0, xmmword ptr [CMP_MSB_MASK+rip]
|
||||
pxor xmm1, xmmword ptr [CMP_MSB_MASK+rip]
|
||||
pcmpgtd xmm0, xmm1
|
||||
movdqa xmm1, xmmword ptr [rsp+0x120]
|
||||
psubd xmm1, xmm0
|
||||
movdqa xmmword ptr [rsp+0x120], xmm1
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
cmp rsi, 4
|
||||
jnc 2b
|
||||
test rsi, rsi
|
||||
jnz 3f
|
||||
4:
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
.p2align 5
|
||||
3:
|
||||
test esi, 0x2
|
||||
je 3f
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+0x10]
|
||||
movaps xmm8, xmm0
|
||||
movaps xmm9, xmm1
|
||||
movd xmm13, dword ptr [rsp+0x110]
|
||||
pinsrd xmm13, dword ptr [rsp+0x120], 1
|
||||
pinsrd xmm13, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
movaps xmmword ptr [rsp], xmm13
|
||||
movd xmm14, dword ptr [rsp+0x114]
|
||||
pinsrd xmm14, dword ptr [rsp+0x124], 1
|
||||
pinsrd xmm14, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
movaps xmmword ptr [rsp+0x10], xmm14
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
movaps xmm10, xmm2
|
||||
movups xmm4, xmmword ptr [r8+rdx-0x40]
|
||||
movups xmm5, xmmword ptr [r8+rdx-0x30]
|
||||
movaps xmm3, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm3, xmm5, 221
|
||||
movaps xmm5, xmm3
|
||||
movups xmm6, xmmword ptr [r8+rdx-0x20]
|
||||
movups xmm7, xmmword ptr [r8+rdx-0x10]
|
||||
movaps xmm3, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 0x93
|
||||
shufps xmm3, xmm7, 221
|
||||
pshufd xmm7, xmm3, 0x93
|
||||
movups xmm12, xmmword ptr [r9+rdx-0x40]
|
||||
movups xmm13, xmmword ptr [r9+rdx-0x30]
|
||||
movaps xmm11, xmm12
|
||||
shufps xmm12, xmm13, 136
|
||||
shufps xmm11, xmm13, 221
|
||||
movaps xmm13, xmm11
|
||||
movups xmm14, xmmword ptr [r9+rdx-0x20]
|
||||
movups xmm15, xmmword ptr [r9+rdx-0x10]
|
||||
movaps xmm11, xmm14
|
||||
shufps xmm14, xmm15, 136
|
||||
pshufd xmm14, xmm14, 0x93
|
||||
shufps xmm11, xmm15, 221
|
||||
pshufd xmm15, xmm11, 0x93
|
||||
movaps xmm3, xmmword ptr [rsp]
|
||||
movaps xmm11, xmmword ptr [rsp+0x10]
|
||||
pinsrd xmm3, eax, 3
|
||||
pinsrd xmm11, eax, 3
|
||||
mov al, 7
|
||||
1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm8, xmm12
|
||||
movaps xmmword ptr [rsp+0x20], xmm4
|
||||
movaps xmmword ptr [rsp+0x30], xmm12
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
movaps xmm12, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm3, xmm12
|
||||
pshufb xmm11, xmm12
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm4, 12
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 20
|
||||
psrld xmm4, 12
|
||||
por xmm9, xmm4
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm8, xmm13
|
||||
movaps xmmword ptr [rsp+0x40], xmm5
|
||||
movaps xmmword ptr [rsp+0x50], xmm13
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
movaps xmm13, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm3, xmm13
|
||||
pshufb xmm11, xmm13
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm4, 7
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 25
|
||||
psrld xmm4, 7
|
||||
por xmm9, xmm4
|
||||
pshufd xmm0, xmm0, 0x93
|
||||
pshufd xmm8, xmm8, 0x93
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm11, xmm11, 0x4E
|
||||
pshufd xmm2, xmm2, 0x39
|
||||
pshufd xmm10, xmm10, 0x39
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm8, xmm14
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
pshufb xmm3, xmm12
|
||||
pshufb xmm11, xmm12
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm4, 12
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 20
|
||||
psrld xmm4, 12
|
||||
por xmm9, xmm4
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm8, xmm15
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
pshufb xmm3, xmm13
|
||||
pshufb xmm11, xmm13
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm4, 7
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 25
|
||||
psrld xmm4, 7
|
||||
por xmm9, xmm4
|
||||
pshufd xmm0, xmm0, 0x39
|
||||
pshufd xmm8, xmm8, 0x39
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm11, xmm11, 0x4E
|
||||
pshufd xmm2, xmm2, 0x93
|
||||
pshufd xmm10, xmm10, 0x93
|
||||
dec al
|
||||
je 1f
|
||||
movdqa xmm12, xmmword ptr [rsp+0x20]
|
||||
movdqa xmm5, xmmword ptr [rsp+0x40]
|
||||
pshufd xmm13, xmm12, 0x0F
|
||||
shufps xmm12, xmm5, 214
|
||||
pshufd xmm4, xmm12, 0x39
|
||||
movdqa xmm12, xmm6
|
||||
shufps xmm12, xmm7, 250
|
||||
pblendw xmm13, xmm12, 0xCC
|
||||
movdqa xmm12, xmm7
|
||||
punpcklqdq xmm12, xmm5
|
||||
pblendw xmm12, xmm6, 0xC0
|
||||
pshufd xmm12, xmm12, 0x78
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 0x1E
|
||||
movdqa xmmword ptr [rsp+0x20], xmm13
|
||||
movdqa xmmword ptr [rsp+0x40], xmm12
|
||||
movdqa xmm5, xmmword ptr [rsp+0x30]
|
||||
movdqa xmm13, xmmword ptr [rsp+0x50]
|
||||
pshufd xmm6, xmm5, 0x0F
|
||||
shufps xmm5, xmm13, 214
|
||||
pshufd xmm12, xmm5, 0x39
|
||||
movdqa xmm5, xmm14
|
||||
shufps xmm5, xmm15, 250
|
||||
pblendw xmm6, xmm5, 0xCC
|
||||
movdqa xmm5, xmm15
|
||||
punpcklqdq xmm5, xmm13
|
||||
pblendw xmm5, xmm14, 0xC0
|
||||
pshufd xmm5, xmm5, 0x78
|
||||
punpckhdq xmm13, xmm15
|
||||
punpckldq xmm14, xmm13
|
||||
pshufd xmm15, xmm14, 0x1E
|
||||
movdqa xmm13, xmm6
|
||||
movdqa xmm14, xmm5
|
||||
movdqa xmm5, xmmword ptr [rsp+0x20]
|
||||
movdqa xmm6, xmmword ptr [rsp+0x40]
|
||||
jmp 1b
|
||||
1:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
pxor xmm8, xmm10
|
||||
pxor xmm9, xmm11
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
movups xmmword ptr [rbx], xmm0
|
||||
movups xmmword ptr [rbx+0x10], xmm1
|
||||
movups xmmword ptr [rbx+0x20], xmm8
|
||||
movups xmmword ptr [rbx+0x30], xmm9
|
||||
movdqa xmm0, xmmword ptr [rsp+0x130]
|
||||
movdqa xmm1, xmmword ptr [rsp+0x110]
|
||||
movdqa xmm2, xmmword ptr [rsp+0x120]
|
||||
movdqu xmm3, xmmword ptr [rsp+0x118]
|
||||
movdqu xmm4, xmmword ptr [rsp+0x128]
|
||||
blendvps xmm1, xmm3, xmm0
|
||||
blendvps xmm2, xmm4, xmm0
|
||||
movdqa xmmword ptr [rsp+0x110], xmm1
|
||||
movdqa xmmword ptr [rsp+0x120], xmm2
|
||||
add rdi, 16
|
||||
add rbx, 64
|
||||
sub rsi, 2
|
||||
3:
|
||||
test esi, 0x1
|
||||
je 4b
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+0x10]
|
||||
movd xmm13, dword ptr [rsp+0x110]
|
||||
pinsrd xmm13, dword ptr [rsp+0x120], 1
|
||||
pinsrd xmm13, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
movaps xmm14, xmmword ptr [ROT8+rip]
|
||||
movaps xmm15, xmmword ptr [ROT16+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+0x40]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
movaps xmm3, xmm13
|
||||
pinsrd xmm3, eax, 3
|
||||
movups xmm4, xmmword ptr [r8+rdx-0x40]
|
||||
movups xmm5, xmmword ptr [r8+rdx-0x30]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [r8+rdx-0x20]
|
||||
movups xmm7, xmmword ptr [r8+rdx-0x10]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 0x93
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x93
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x39
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x39
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0x0F
|
||||
pshufd xmm4, xmm8, 0x39
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0xCC
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0xC0
|
||||
pshufd xmm8, xmm8, 0x78
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 0x1E
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
movups xmmword ptr [rbx], xmm0
|
||||
movups xmmword ptr [rbx+0x10], xmm1
|
||||
jmp 4b
|
||||
|
||||
.p2align 6
|
||||
blake3_compress_in_place_sse41:
|
||||
_blake3_compress_in_place_sse41:
|
||||
movups xmm0, xmmword ptr [rdi]
|
||||
movups xmm1, xmmword ptr [rdi+0x10]
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
shl r8, 32
|
||||
add rdx, r8
|
||||
movq xmm3, rcx
|
||||
movq xmm4, rdx
|
||||
punpcklqdq xmm3, xmm4
|
||||
movups xmm4, xmmword ptr [rsi]
|
||||
movups xmm5, xmmword ptr [rsi+0x10]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [rsi+0x20]
|
||||
movups xmm7, xmmword ptr [rsi+0x30]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 0x93
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 0x93
|
||||
movaps xmm14, xmmword ptr [ROT8+rip]
|
||||
movaps xmm15, xmmword ptr [ROT16+rip]
|
||||
mov al, 7
|
||||
1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x93
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x39
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x39
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0x0F
|
||||
pshufd xmm4, xmm8, 0x39
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0xCC
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0xC0
|
||||
pshufd xmm8, xmm8, 0x78
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 0x1E
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
movups xmmword ptr [rdi], xmm0
|
||||
movups xmmword ptr [rdi+0x10], xmm1
|
||||
ret
|
||||
|
||||
.p2align 6
|
||||
blake3_compress_xof_sse41:
|
||||
_blake3_compress_xof_sse41:
|
||||
movups xmm0, xmmword ptr [rdi]
|
||||
movups xmm1, xmmword ptr [rdi+0x10]
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
movzx eax, r8b
|
||||
movzx edx, dl
|
||||
shl rax, 32
|
||||
add rdx, rax
|
||||
movq xmm3, rcx
|
||||
movq xmm4, rdx
|
||||
punpcklqdq xmm3, xmm4
|
||||
movups xmm4, xmmword ptr [rsi]
|
||||
movups xmm5, xmmword ptr [rsi+0x10]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [rsi+0x20]
|
||||
movups xmm7, xmmword ptr [rsi+0x30]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 0x93
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 0x93
|
||||
movaps xmm14, xmmword ptr [ROT8+rip]
|
||||
movaps xmm15, xmmword ptr [ROT16+rip]
|
||||
mov al, 7
|
||||
1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x93
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x39
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x39
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0x0F
|
||||
pshufd xmm4, xmm8, 0x39
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0xCC
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0xC0
|
||||
pshufd xmm8, xmm8, 0x78
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 0x1E
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
movdqu xmm4, xmmword ptr [rdi]
|
||||
movdqu xmm5, xmmword ptr [rdi+0x10]
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
pxor xmm2, xmm4
|
||||
pxor xmm3, xmm5
|
||||
movups xmmword ptr [r9], xmm0
|
||||
movups xmmword ptr [r9+0x10], xmm1
|
||||
movups xmmword ptr [r9+0x20], xmm2
|
||||
movups xmmword ptr [r9+0x30], xmm3
|
||||
ret
|
||||
|
||||
|
||||
#ifdef __APPLE__
|
||||
.static_data
|
||||
#else
|
||||
.section .rodata
|
||||
#endif
|
||||
.p2align 6
|
||||
BLAKE3_IV:
|
||||
.long 0x6A09E667, 0xBB67AE85
|
||||
.long 0x3C6EF372, 0xA54FF53A
|
||||
ROT16:
|
||||
.byte 2, 3, 0, 1, 6, 7, 4, 5, 10, 11, 8, 9, 14, 15, 12, 13
|
||||
ROT8:
|
||||
.byte 1, 2, 3, 0, 5, 6, 7, 4, 9, 10, 11, 8, 13, 14, 15, 12
|
||||
ADD0:
|
||||
.long 0, 1, 2, 3
|
||||
ADD1:
|
||||
.long 4, 4, 4, 4
|
||||
BLAKE3_IV_0:
|
||||
.long 0x6A09E667, 0x6A09E667, 0x6A09E667, 0x6A09E667
|
||||
BLAKE3_IV_1:
|
||||
.long 0xBB67AE85, 0xBB67AE85, 0xBB67AE85, 0xBB67AE85
|
||||
BLAKE3_IV_2:
|
||||
.long 0x3C6EF372, 0x3C6EF372, 0x3C6EF372, 0x3C6EF372
|
||||
BLAKE3_IV_3:
|
||||
.long 0xA54FF53A, 0xA54FF53A, 0xA54FF53A, 0xA54FF53A
|
||||
BLAKE3_BLOCK_LEN:
|
||||
.long 64, 64, 64, 64
|
||||
CMP_MSB_MASK:
|
||||
.long 0x80000000, 0x80000000, 0x80000000, 0x80000000
|
|
@ -0,0 +1,2057 @@
|
|||
.intel_syntax noprefix
|
||||
.global blake3_hash_many_sse41
|
||||
.global _blake3_hash_many_sse41
|
||||
.global blake3_compress_in_place_sse41
|
||||
.global _blake3_compress_in_place_sse41
|
||||
.global blake3_compress_xof_sse41
|
||||
.global _blake3_compress_xof_sse41
|
||||
.section .text
|
||||
.p2align 6
|
||||
_blake3_hash_many_sse41:
|
||||
blake3_hash_many_sse41:
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rsi
|
||||
push rdi
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 528
|
||||
and rsp, 0xFFFFFFFFFFFFFFC0
|
||||
movdqa xmmword ptr [rsp+0x170], xmm6
|
||||
movdqa xmmword ptr [rsp+0x180], xmm7
|
||||
movdqa xmmword ptr [rsp+0x190], xmm8
|
||||
movdqa xmmword ptr [rsp+0x1A0], xmm9
|
||||
movdqa xmmword ptr [rsp+0x1B0], xmm10
|
||||
movdqa xmmword ptr [rsp+0x1C0], xmm11
|
||||
movdqa xmmword ptr [rsp+0x1D0], xmm12
|
||||
movdqa xmmword ptr [rsp+0x1E0], xmm13
|
||||
movdqa xmmword ptr [rsp+0x1F0], xmm14
|
||||
movdqa xmmword ptr [rsp+0x200], xmm15
|
||||
mov rdi, rcx
|
||||
mov rsi, rdx
|
||||
mov rdx, r8
|
||||
mov rcx, r9
|
||||
mov r8, qword ptr [rbp+0x68]
|
||||
movzx r9, byte ptr [rbp+0x70]
|
||||
neg r9d
|
||||
movd xmm0, r9d
|
||||
pshufd xmm0, xmm0, 0x00
|
||||
movdqa xmmword ptr [rsp+0x130], xmm0
|
||||
movdqa xmm1, xmm0
|
||||
pand xmm1, xmmword ptr [ADD0+rip]
|
||||
pand xmm0, xmmword ptr [ADD1+rip]
|
||||
movdqa xmmword ptr [rsp+0x150], xmm0
|
||||
movd xmm0, r8d
|
||||
pshufd xmm0, xmm0, 0x00
|
||||
paddd xmm0, xmm1
|
||||
movdqa xmmword ptr [rsp+0x110], xmm0
|
||||
pxor xmm0, xmmword ptr [CMP_MSB_MASK+rip]
|
||||
pxor xmm1, xmmword ptr [CMP_MSB_MASK+rip]
|
||||
pcmpgtd xmm1, xmm0
|
||||
shr r8, 32
|
||||
movd xmm2, r8d
|
||||
pshufd xmm2, xmm2, 0x00
|
||||
psubd xmm2, xmm1
|
||||
movdqa xmmword ptr [rsp+0x120], xmm2
|
||||
mov rbx, qword ptr [rbp+0x90]
|
||||
mov r15, rdx
|
||||
shl r15, 6
|
||||
movzx r13d, byte ptr [rbp+0x78]
|
||||
movzx r12d, byte ptr [rbp+0x88]
|
||||
cmp rsi, 4
|
||||
jc 3f
|
||||
2:
|
||||
movdqu xmm3, xmmword ptr [rcx]
|
||||
pshufd xmm0, xmm3, 0x00
|
||||
pshufd xmm1, xmm3, 0x55
|
||||
pshufd xmm2, xmm3, 0xAA
|
||||
pshufd xmm3, xmm3, 0xFF
|
||||
movdqu xmm7, xmmword ptr [rcx+0x10]
|
||||
pshufd xmm4, xmm7, 0x00
|
||||
pshufd xmm5, xmm7, 0x55
|
||||
pshufd xmm6, xmm7, 0xAA
|
||||
pshufd xmm7, xmm7, 0xFF
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
mov r10, qword ptr [rdi+0x10]
|
||||
mov r11, qword ptr [rdi+0x18]
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
1:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-0x40]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-0x40]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-0x40]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-0x40]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp], xmm8
|
||||
movdqa xmmword ptr [rsp+0x10], xmm9
|
||||
movdqa xmmword ptr [rsp+0x20], xmm12
|
||||
movdqa xmmword ptr [rsp+0x30], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-0x30]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-0x30]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-0x30]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-0x30]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+0x40], xmm8
|
||||
movdqa xmmword ptr [rsp+0x50], xmm9
|
||||
movdqa xmmword ptr [rsp+0x60], xmm12
|
||||
movdqa xmmword ptr [rsp+0x70], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-0x20]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-0x20]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-0x20]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-0x20]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+0x80], xmm8
|
||||
movdqa xmmword ptr [rsp+0x90], xmm9
|
||||
movdqa xmmword ptr [rsp+0xA0], xmm12
|
||||
movdqa xmmword ptr [rsp+0xB0], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-0x10]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-0x10]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-0x10]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-0x10]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+0xC0], xmm8
|
||||
movdqa xmmword ptr [rsp+0xD0], xmm9
|
||||
movdqa xmmword ptr [rsp+0xE0], xmm12
|
||||
movdqa xmmword ptr [rsp+0xF0], xmm13
|
||||
movdqa xmm9, xmmword ptr [BLAKE3_IV_1+rip]
|
||||
movdqa xmm10, xmmword ptr [BLAKE3_IV_2+rip]
|
||||
movdqa xmm11, xmmword ptr [BLAKE3_IV_3+rip]
|
||||
movdqa xmm12, xmmword ptr [rsp+0x110]
|
||||
movdqa xmm13, xmmword ptr [rsp+0x120]
|
||||
movdqa xmm14, xmmword ptr [BLAKE3_BLOCK_LEN+rip]
|
||||
movd xmm15, eax
|
||||
pshufd xmm15, xmm15, 0x00
|
||||
prefetcht0 [r8+rdx+0x80]
|
||||
prefetcht0 [r9+rdx+0x80]
|
||||
prefetcht0 [r10+rdx+0x80]
|
||||
prefetcht0 [r11+rdx+0x80]
|
||||
paddd xmm0, xmmword ptr [rsp]
|
||||
paddd xmm1, xmmword ptr [rsp+0x20]
|
||||
paddd xmm2, xmmword ptr [rsp+0x40]
|
||||
paddd xmm3, xmmword ptr [rsp+0x60]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [BLAKE3_IV_0+rip]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x10]
|
||||
paddd xmm1, xmmword ptr [rsp+0x30]
|
||||
paddd xmm2, xmmword ptr [rsp+0x50]
|
||||
paddd xmm3, xmmword ptr [rsp+0x70]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x80]
|
||||
paddd xmm1, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm3, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x90]
|
||||
paddd xmm1, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm3, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x20]
|
||||
paddd xmm1, xmmword ptr [rsp+0x30]
|
||||
paddd xmm2, xmmword ptr [rsp+0x70]
|
||||
paddd xmm3, xmmword ptr [rsp+0x40]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x60]
|
||||
paddd xmm1, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm2, xmmword ptr [rsp]
|
||||
paddd xmm3, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x10]
|
||||
paddd xmm1, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x90]
|
||||
paddd xmm3, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x50]
|
||||
paddd xmm2, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x80]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x30]
|
||||
paddd xmm1, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x70]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x40]
|
||||
paddd xmm1, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x20]
|
||||
paddd xmm3, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x60]
|
||||
paddd xmm1, xmmword ptr [rsp+0x90]
|
||||
paddd xmm2, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x80]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x50]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x10]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm1, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm3, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x70]
|
||||
paddd xmm1, xmmword ptr [rsp+0x90]
|
||||
paddd xmm2, xmmword ptr [rsp+0x30]
|
||||
paddd xmm3, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x40]
|
||||
paddd xmm1, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x50]
|
||||
paddd xmm3, xmmword ptr [rsp+0x10]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp]
|
||||
paddd xmm1, xmmword ptr [rsp+0x20]
|
||||
paddd xmm2, xmmword ptr [rsp+0x80]
|
||||
paddd xmm3, xmmword ptr [rsp+0x60]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x90]
|
||||
paddd xmm2, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm3, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm1, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm2, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x80]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x70]
|
||||
paddd xmm1, xmmword ptr [rsp+0x50]
|
||||
paddd xmm2, xmmword ptr [rsp]
|
||||
paddd xmm3, xmmword ptr [rsp+0x60]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x20]
|
||||
paddd xmm1, xmmword ptr [rsp+0x30]
|
||||
paddd xmm2, xmmword ptr [rsp+0x10]
|
||||
paddd xmm3, xmmword ptr [rsp+0x40]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x90]
|
||||
paddd xmm1, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x80]
|
||||
paddd xmm3, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x50]
|
||||
paddd xmm2, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm3, xmmword ptr [rsp+0x10]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+0x20]
|
||||
paddd xmm3, xmmword ptr [rsp+0x40]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0x30]
|
||||
paddd xmm1, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x60]
|
||||
paddd xmm3, xmmword ptr [rsp+0x70]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xB0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x50]
|
||||
paddd xmm2, xmmword ptr [rsp+0x10]
|
||||
paddd xmm3, xmmword ptr [rsp+0x80]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xF0]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+0x90]
|
||||
paddd xmm3, xmmword ptr [rsp+0x60]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xE0]
|
||||
paddd xmm1, xmmword ptr [rsp+0x20]
|
||||
paddd xmm2, xmmword ptr [rsp+0x30]
|
||||
paddd xmm3, xmmword ptr [rsp+0x70]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+0x100], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0xA0]
|
||||
paddd xmm1, xmmword ptr [rsp+0xC0]
|
||||
paddd xmm2, xmmword ptr [rsp+0x40]
|
||||
paddd xmm3, xmmword ptr [rsp+0xD0]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+0x100]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
pxor xmm0, xmm8
|
||||
pxor xmm1, xmm9
|
||||
pxor xmm2, xmm10
|
||||
pxor xmm3, xmm11
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
pxor xmm4, xmm12
|
||||
pxor xmm5, xmm13
|
||||
pxor xmm6, xmm14
|
||||
pxor xmm7, xmm15
|
||||
mov eax, r13d
|
||||
jne 1b
|
||||
movdqa xmm9, xmm0
|
||||
punpckldq xmm0, xmm1
|
||||
punpckhdq xmm9, xmm1
|
||||
movdqa xmm11, xmm2
|
||||
punpckldq xmm2, xmm3
|
||||
punpckhdq xmm11, xmm3
|
||||
movdqa xmm1, xmm0
|
||||
punpcklqdq xmm0, xmm2
|
||||
punpckhqdq xmm1, xmm2
|
||||
movdqa xmm3, xmm9
|
||||
punpcklqdq xmm9, xmm11
|
||||
punpckhqdq xmm3, xmm11
|
||||
movdqu xmmword ptr [rbx], xmm0
|
||||
movdqu xmmword ptr [rbx+0x20], xmm1
|
||||
movdqu xmmword ptr [rbx+0x40], xmm9
|
||||
movdqu xmmword ptr [rbx+0x60], xmm3
|
||||
movdqa xmm9, xmm4
|
||||
punpckldq xmm4, xmm5
|
||||
punpckhdq xmm9, xmm5
|
||||
movdqa xmm11, xmm6
|
||||
punpckldq xmm6, xmm7
|
||||
punpckhdq xmm11, xmm7
|
||||
movdqa xmm5, xmm4
|
||||
punpcklqdq xmm4, xmm6
|
||||
punpckhqdq xmm5, xmm6
|
||||
movdqa xmm7, xmm9
|
||||
punpcklqdq xmm9, xmm11
|
||||
punpckhqdq xmm7, xmm11
|
||||
movdqu xmmword ptr [rbx+0x10], xmm4
|
||||
movdqu xmmword ptr [rbx+0x30], xmm5
|
||||
movdqu xmmword ptr [rbx+0x50], xmm9
|
||||
movdqu xmmword ptr [rbx+0x70], xmm7
|
||||
movdqa xmm1, xmmword ptr [rsp+0x110]
|
||||
movdqa xmm0, xmm1
|
||||
paddd xmm1, xmmword ptr [rsp+0x150]
|
||||
movdqa xmmword ptr [rsp+0x110], xmm1
|
||||
pxor xmm0, xmmword ptr [CMP_MSB_MASK+rip]
|
||||
pxor xmm1, xmmword ptr [CMP_MSB_MASK+rip]
|
||||
pcmpgtd xmm0, xmm1
|
||||
movdqa xmm1, xmmword ptr [rsp+0x120]
|
||||
psubd xmm1, xmm0
|
||||
movdqa xmmword ptr [rsp+0x120], xmm1
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
cmp rsi, 4
|
||||
jnc 2b
|
||||
test rsi, rsi
|
||||
jne 3f
|
||||
4:
|
||||
movdqa xmm6, xmmword ptr [rsp+0x170]
|
||||
movdqa xmm7, xmmword ptr [rsp+0x180]
|
||||
movdqa xmm8, xmmword ptr [rsp+0x190]
|
||||
movdqa xmm9, xmmword ptr [rsp+0x1A0]
|
||||
movdqa xmm10, xmmword ptr [rsp+0x1B0]
|
||||
movdqa xmm11, xmmword ptr [rsp+0x1C0]
|
||||
movdqa xmm12, xmmword ptr [rsp+0x1D0]
|
||||
movdqa xmm13, xmmword ptr [rsp+0x1E0]
|
||||
movdqa xmm14, xmmword ptr [rsp+0x1F0]
|
||||
movdqa xmm15, xmmword ptr [rsp+0x200]
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop rdi
|
||||
pop rsi
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
.p2align 5
|
||||
3:
|
||||
test esi, 0x2
|
||||
je 3f
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+0x10]
|
||||
movaps xmm8, xmm0
|
||||
movaps xmm9, xmm1
|
||||
movd xmm13, dword ptr [rsp+0x110]
|
||||
pinsrd xmm13, dword ptr [rsp+0x120], 1
|
||||
pinsrd xmm13, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
movaps xmmword ptr [rsp], xmm13
|
||||
movd xmm14, dword ptr [rsp+0x114]
|
||||
pinsrd xmm14, dword ptr [rsp+0x124], 1
|
||||
pinsrd xmm14, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
movaps xmmword ptr [rsp+0x10], xmm14
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+0x8]
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
movaps xmm10, xmm2
|
||||
movups xmm4, xmmword ptr [r8+rdx-0x40]
|
||||
movups xmm5, xmmword ptr [r8+rdx-0x30]
|
||||
movaps xmm3, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm3, xmm5, 221
|
||||
movaps xmm5, xmm3
|
||||
movups xmm6, xmmword ptr [r8+rdx-0x20]
|
||||
movups xmm7, xmmword ptr [r8+rdx-0x10]
|
||||
movaps xmm3, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 0x93
|
||||
shufps xmm3, xmm7, 221
|
||||
pshufd xmm7, xmm3, 0x93
|
||||
movups xmm12, xmmword ptr [r9+rdx-0x40]
|
||||
movups xmm13, xmmword ptr [r9+rdx-0x30]
|
||||
movaps xmm11, xmm12
|
||||
shufps xmm12, xmm13, 136
|
||||
shufps xmm11, xmm13, 221
|
||||
movaps xmm13, xmm11
|
||||
movups xmm14, xmmword ptr [r9+rdx-0x20]
|
||||
movups xmm15, xmmword ptr [r9+rdx-0x10]
|
||||
movaps xmm11, xmm14
|
||||
shufps xmm14, xmm15, 136
|
||||
pshufd xmm14, xmm14, 0x93
|
||||
shufps xmm11, xmm15, 221
|
||||
pshufd xmm15, xmm11, 0x93
|
||||
movaps xmm3, xmmword ptr [rsp]
|
||||
movaps xmm11, xmmword ptr [rsp+0x10]
|
||||
pinsrd xmm3, eax, 3
|
||||
pinsrd xmm11, eax, 3
|
||||
mov al, 7
|
||||
1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm8, xmm12
|
||||
movaps xmmword ptr [rsp+0x20], xmm4
|
||||
movaps xmmword ptr [rsp+0x30], xmm12
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
movaps xmm12, xmmword ptr [ROT16+rip]
|
||||
pshufb xmm3, xmm12
|
||||
pshufb xmm11, xmm12
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm4, 12
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 20
|
||||
psrld xmm4, 12
|
||||
por xmm9, xmm4
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm8, xmm13
|
||||
movaps xmmword ptr [rsp+0x40], xmm5
|
||||
movaps xmmword ptr [rsp+0x50], xmm13
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
movaps xmm13, xmmword ptr [ROT8+rip]
|
||||
pshufb xmm3, xmm13
|
||||
pshufb xmm11, xmm13
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm4, 7
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 25
|
||||
psrld xmm4, 7
|
||||
por xmm9, xmm4
|
||||
pshufd xmm0, xmm0, 0x93
|
||||
pshufd xmm8, xmm8, 0x93
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm11, xmm11, 0x4E
|
||||
pshufd xmm2, xmm2, 0x39
|
||||
pshufd xmm10, xmm10, 0x39
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm8, xmm14
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
pshufb xmm3, xmm12
|
||||
pshufb xmm11, xmm12
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm4, 12
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 20
|
||||
psrld xmm4, 12
|
||||
por xmm9, xmm4
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm8, xmm15
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
pshufb xmm3, xmm13
|
||||
pshufb xmm11, xmm13
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm4, 7
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 25
|
||||
psrld xmm4, 7
|
||||
por xmm9, xmm4
|
||||
pshufd xmm0, xmm0, 0x39
|
||||
pshufd xmm8, xmm8, 0x39
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm11, xmm11, 0x4E
|
||||
pshufd xmm2, xmm2, 0x93
|
||||
pshufd xmm10, xmm10, 0x93
|
||||
dec al
|
||||
je 1f
|
||||
movdqa xmm12, xmmword ptr [rsp+0x20]
|
||||
movdqa xmm5, xmmword ptr [rsp+0x40]
|
||||
pshufd xmm13, xmm12, 0x0F
|
||||
shufps xmm12, xmm5, 214
|
||||
pshufd xmm4, xmm12, 0x39
|
||||
movdqa xmm12, xmm6
|
||||
shufps xmm12, xmm7, 250
|
||||
pblendw xmm13, xmm12, 0xCC
|
||||
movdqa xmm12, xmm7
|
||||
punpcklqdq xmm12, xmm5
|
||||
pblendw xmm12, xmm6, 0xC0
|
||||
pshufd xmm12, xmm12, 0x78
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 0x1E
|
||||
movdqa xmmword ptr [rsp+0x20], xmm13
|
||||
movdqa xmmword ptr [rsp+0x40], xmm12
|
||||
movdqa xmm5, xmmword ptr [rsp+0x30]
|
||||
movdqa xmm13, xmmword ptr [rsp+0x50]
|
||||
pshufd xmm6, xmm5, 0x0F
|
||||
shufps xmm5, xmm13, 214
|
||||
pshufd xmm12, xmm5, 0x39
|
||||
movdqa xmm5, xmm14
|
||||
shufps xmm5, xmm15, 250
|
||||
pblendw xmm6, xmm5, 0xCC
|
||||
movdqa xmm5, xmm15
|
||||
punpcklqdq xmm5, xmm13
|
||||
pblendw xmm5, xmm14, 0xC0
|
||||
pshufd xmm5, xmm5, 0x78
|
||||
punpckhdq xmm13, xmm15
|
||||
punpckldq xmm14, xmm13
|
||||
pshufd xmm15, xmm14, 0x1E
|
||||
movdqa xmm13, xmm6
|
||||
movdqa xmm14, xmm5
|
||||
movdqa xmm5, xmmword ptr [rsp+0x20]
|
||||
movdqa xmm6, xmmword ptr [rsp+0x40]
|
||||
jmp 1b
|
||||
1:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
pxor xmm8, xmm10
|
||||
pxor xmm9, xmm11
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
movups xmmword ptr [rbx], xmm0
|
||||
movups xmmword ptr [rbx+0x10], xmm1
|
||||
movups xmmword ptr [rbx+0x20], xmm8
|
||||
movups xmmword ptr [rbx+0x30], xmm9
|
||||
movdqa xmm0, xmmword ptr [rsp+0x130]
|
||||
movdqa xmm1, xmmword ptr [rsp+0x110]
|
||||
movdqa xmm2, xmmword ptr [rsp+0x120]
|
||||
movdqu xmm3, xmmword ptr [rsp+0x118]
|
||||
movdqu xmm4, xmmword ptr [rsp+0x128]
|
||||
blendvps xmm1, xmm3, xmm0
|
||||
blendvps xmm2, xmm4, xmm0
|
||||
movdqa xmmword ptr [rsp+0x110], xmm1
|
||||
movdqa xmmword ptr [rsp+0x120], xmm2
|
||||
add rdi, 16
|
||||
add rbx, 64
|
||||
sub rsi, 2
|
||||
3:
|
||||
test esi, 0x1
|
||||
je 4b
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+0x10]
|
||||
movd xmm13, dword ptr [rsp+0x110]
|
||||
pinsrd xmm13, dword ptr [rsp+0x120], 1
|
||||
pinsrd xmm13, dword ptr [BLAKE3_BLOCK_LEN+rip], 2
|
||||
movaps xmm14, xmmword ptr [ROT8+rip]
|
||||
movaps xmm15, xmmword ptr [ROT16+rip]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+0x80]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
movaps xmm3, xmm13
|
||||
pinsrd xmm3, eax, 3
|
||||
movups xmm4, xmmword ptr [r8+rdx-0x40]
|
||||
movups xmm5, xmmword ptr [r8+rdx-0x30]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [r8+rdx-0x20]
|
||||
movups xmm7, xmmword ptr [r8+rdx-0x10]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 0x93
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 0x93
|
||||
mov al, 7
|
||||
1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x93
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x39
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x39
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0x0F
|
||||
pshufd xmm4, xmm8, 0x39
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0xCC
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0xC0
|
||||
pshufd xmm8, xmm8, 0x78
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 0x1E
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne 2b
|
||||
movups xmmword ptr [rbx], xmm0
|
||||
movups xmmword ptr [rbx+0x10], xmm1
|
||||
jmp 4b
|
||||
|
||||
.p2align 6
|
||||
blake3_compress_in_place_sse41:
|
||||
_blake3_compress_in_place_sse41:
|
||||
sub rsp, 72
|
||||
movdqa xmmword ptr [rsp], xmm6
|
||||
movdqa xmmword ptr [rsp+0x10], xmm7
|
||||
movdqa xmmword ptr [rsp+0x20], xmm8
|
||||
movdqa xmmword ptr [rsp+0x30], xmm9
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+0x10]
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
movzx eax, byte ptr [rsp+0x70]
|
||||
movzx r8d, r8b
|
||||
shl rax, 32
|
||||
add r8, rax
|
||||
movq xmm3, r9
|
||||
movq xmm4, r8
|
||||
punpcklqdq xmm3, xmm4
|
||||
movups xmm4, xmmword ptr [rdx]
|
||||
movups xmm5, xmmword ptr [rdx+0x10]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [rdx+0x20]
|
||||
movups xmm7, xmmword ptr [rdx+0x30]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 0x93
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 0x93
|
||||
movaps xmm14, xmmword ptr [ROT8+rip]
|
||||
movaps xmm15, xmmword ptr [ROT16+rip]
|
||||
mov al, 7
|
||||
1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x93
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x39
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x39
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0x0F
|
||||
pshufd xmm4, xmm8, 0x39
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0xCC
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0xC0
|
||||
pshufd xmm8, xmm8, 0x78
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 0x1E
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
movups xmmword ptr [rcx], xmm0
|
||||
movups xmmword ptr [rcx+0x10], xmm1
|
||||
movdqa xmm6, xmmword ptr [rsp]
|
||||
movdqa xmm7, xmmword ptr [rsp+0x10]
|
||||
movdqa xmm8, xmmword ptr [rsp+0x20]
|
||||
movdqa xmm9, xmmword ptr [rsp+0x30]
|
||||
add rsp, 72
|
||||
ret
|
||||
|
||||
|
||||
.p2align 6
|
||||
_blake3_compress_xof_sse41:
|
||||
blake3_compress_xof_sse41:
|
||||
sub rsp, 72
|
||||
movdqa xmmword ptr [rsp], xmm6
|
||||
movdqa xmmword ptr [rsp+0x10], xmm7
|
||||
movdqa xmmword ptr [rsp+0x20], xmm8
|
||||
movdqa xmmword ptr [rsp+0x30], xmm9
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+0x10]
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV+rip]
|
||||
movzx eax, byte ptr [rsp+0x70]
|
||||
movzx r8d, r8b
|
||||
mov r10, qword ptr [rsp+0x78]
|
||||
shl rax, 32
|
||||
add r8, rax
|
||||
movq xmm3, r9
|
||||
movq xmm4, r8
|
||||
punpcklqdq xmm3, xmm4
|
||||
movups xmm4, xmmword ptr [rdx]
|
||||
movups xmm5, xmmword ptr [rdx+0x10]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [rdx+0x20]
|
||||
movups xmm7, xmmword ptr [rdx+0x30]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 0x93
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 0x93
|
||||
movaps xmm14, xmmword ptr [ROT8+rip]
|
||||
movaps xmm15, xmmword ptr [ROT16+rip]
|
||||
mov al, 7
|
||||
1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x93
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x39
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 0x39
|
||||
pshufd xmm3, xmm3, 0x4E
|
||||
pshufd xmm2, xmm2, 0x93
|
||||
dec al
|
||||
jz 1f
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0x0F
|
||||
pshufd xmm4, xmm8, 0x39
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0xCC
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0xC0
|
||||
pshufd xmm8, xmm8, 0x78
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 0x1E
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp 1b
|
||||
1:
|
||||
movdqu xmm4, xmmword ptr [rcx]
|
||||
movdqu xmm5, xmmword ptr [rcx+0x10]
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
pxor xmm2, xmm4
|
||||
pxor xmm3, xmm5
|
||||
movups xmmword ptr [r10], xmm0
|
||||
movups xmmword ptr [r10+0x10], xmm1
|
||||
movups xmmword ptr [r10+0x20], xmm2
|
||||
movups xmmword ptr [r10+0x30], xmm3
|
||||
movdqa xmm6, xmmword ptr [rsp]
|
||||
movdqa xmm7, xmmword ptr [rsp+0x10]
|
||||
movdqa xmm8, xmmword ptr [rsp+0x20]
|
||||
movdqa xmm9, xmmword ptr [rsp+0x30]
|
||||
add rsp, 72
|
||||
ret
|
||||
|
||||
|
||||
.section .rodata
|
||||
.p2align 6
|
||||
BLAKE3_IV:
|
||||
.long 0x6A09E667, 0xBB67AE85
|
||||
.long 0x3C6EF372, 0xA54FF53A
|
||||
ROT16:
|
||||
.byte 2, 3, 0, 1, 6, 7, 4, 5, 10, 11, 8, 9, 14, 15, 12, 13
|
||||
ROT8:
|
||||
.byte 1, 2, 3, 0, 5, 6, 7, 4, 9, 10, 11, 8, 13, 14, 15, 12
|
||||
ADD0:
|
||||
.long 0, 1, 2, 3
|
||||
ADD1:
|
||||
.long 4, 4, 4, 4
|
||||
BLAKE3_IV_0:
|
||||
.long 0x6A09E667, 0x6A09E667, 0x6A09E667, 0x6A09E667
|
||||
BLAKE3_IV_1:
|
||||
.long 0xBB67AE85, 0xBB67AE85, 0xBB67AE85, 0xBB67AE85
|
||||
BLAKE3_IV_2:
|
||||
.long 0x3C6EF372, 0x3C6EF372, 0x3C6EF372, 0x3C6EF372
|
||||
BLAKE3_IV_3:
|
||||
.long 0xA54FF53A, 0xA54FF53A, 0xA54FF53A, 0xA54FF53A
|
||||
BLAKE3_BLOCK_LEN:
|
||||
.long 64, 64, 64, 64
|
||||
CMP_MSB_MASK:
|
||||
.long 0x80000000, 0x80000000, 0x80000000, 0x80000000
|
|
@ -0,0 +1,2077 @@
|
|||
public _blake3_hash_many_sse41
|
||||
public blake3_hash_many_sse41
|
||||
public blake3_compress_in_place_sse41
|
||||
public _blake3_compress_in_place_sse41
|
||||
public blake3_compress_xof_sse41
|
||||
public _blake3_compress_xof_sse41
|
||||
|
||||
_TEXT SEGMENT ALIGN(16) 'CODE'
|
||||
|
||||
ALIGN 16
|
||||
blake3_hash_many_sse41 PROC
|
||||
_blake3_hash_many_sse41 PROC
|
||||
push r15
|
||||
push r14
|
||||
push r13
|
||||
push r12
|
||||
push rsi
|
||||
push rdi
|
||||
push rbx
|
||||
push rbp
|
||||
mov rbp, rsp
|
||||
sub rsp, 528
|
||||
and rsp, 0FFFFFFFFFFFFFFC0H
|
||||
movdqa xmmword ptr [rsp+170H], xmm6
|
||||
movdqa xmmword ptr [rsp+180H], xmm7
|
||||
movdqa xmmword ptr [rsp+190H], xmm8
|
||||
movdqa xmmword ptr [rsp+1A0H], xmm9
|
||||
movdqa xmmword ptr [rsp+1B0H], xmm10
|
||||
movdqa xmmword ptr [rsp+1C0H], xmm11
|
||||
movdqa xmmword ptr [rsp+1D0H], xmm12
|
||||
movdqa xmmword ptr [rsp+1E0H], xmm13
|
||||
movdqa xmmword ptr [rsp+1F0H], xmm14
|
||||
movdqa xmmword ptr [rsp+200H], xmm15
|
||||
mov rdi, rcx
|
||||
mov rsi, rdx
|
||||
mov rdx, r8
|
||||
mov rcx, r9
|
||||
mov r8, qword ptr [rbp+68H]
|
||||
movzx r9, byte ptr [rbp+70H]
|
||||
neg r9d
|
||||
movd xmm0, r9d
|
||||
pshufd xmm0, xmm0, 00H
|
||||
movdqa xmmword ptr [rsp+130H], xmm0
|
||||
movdqa xmm1, xmm0
|
||||
pand xmm1, xmmword ptr [ADD0]
|
||||
pand xmm0, xmmword ptr [ADD1]
|
||||
movdqa xmmword ptr [rsp+150H], xmm0
|
||||
movd xmm0, r8d
|
||||
pshufd xmm0, xmm0, 00H
|
||||
paddd xmm0, xmm1
|
||||
movdqa xmmword ptr [rsp+110H], xmm0
|
||||
pxor xmm0, xmmword ptr [CMP_MSB_MASK]
|
||||
pxor xmm1, xmmword ptr [CMP_MSB_MASK]
|
||||
pcmpgtd xmm1, xmm0
|
||||
shr r8, 32
|
||||
movd xmm2, r8d
|
||||
pshufd xmm2, xmm2, 00H
|
||||
psubd xmm2, xmm1
|
||||
movdqa xmmword ptr [rsp+120H], xmm2
|
||||
mov rbx, qword ptr [rbp+90H]
|
||||
mov r15, rdx
|
||||
shl r15, 6
|
||||
movzx r13d, byte ptr [rbp+78H]
|
||||
movzx r12d, byte ptr [rbp+88H]
|
||||
cmp rsi, 4
|
||||
jc final3blocks
|
||||
outerloop4:
|
||||
movdqu xmm3, xmmword ptr [rcx]
|
||||
pshufd xmm0, xmm3, 00H
|
||||
pshufd xmm1, xmm3, 55H
|
||||
pshufd xmm2, xmm3, 0AAH
|
||||
pshufd xmm3, xmm3, 0FFH
|
||||
movdqu xmm7, xmmword ptr [rcx+10H]
|
||||
pshufd xmm4, xmm7, 00H
|
||||
pshufd xmm5, xmm7, 55H
|
||||
pshufd xmm6, xmm7, 0AAH
|
||||
pshufd xmm7, xmm7, 0FFH
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
mov r10, qword ptr [rdi+10H]
|
||||
mov r11, qword ptr [rdi+18H]
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
innerloop4:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-40H]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-40H]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-40H]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-40H]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp], xmm8
|
||||
movdqa xmmword ptr [rsp+10H], xmm9
|
||||
movdqa xmmword ptr [rsp+20H], xmm12
|
||||
movdqa xmmword ptr [rsp+30H], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-30H]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-30H]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-30H]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-30H]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+40H], xmm8
|
||||
movdqa xmmword ptr [rsp+50H], xmm9
|
||||
movdqa xmmword ptr [rsp+60H], xmm12
|
||||
movdqa xmmword ptr [rsp+70H], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-20H]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-20H]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-20H]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-20H]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+80H], xmm8
|
||||
movdqa xmmword ptr [rsp+90H], xmm9
|
||||
movdqa xmmword ptr [rsp+0A0H], xmm12
|
||||
movdqa xmmword ptr [rsp+0B0H], xmm13
|
||||
movdqu xmm8, xmmword ptr [r8+rdx-10H]
|
||||
movdqu xmm9, xmmword ptr [r9+rdx-10H]
|
||||
movdqu xmm10, xmmword ptr [r10+rdx-10H]
|
||||
movdqu xmm11, xmmword ptr [r11+rdx-10H]
|
||||
movdqa xmm12, xmm8
|
||||
punpckldq xmm8, xmm9
|
||||
punpckhdq xmm12, xmm9
|
||||
movdqa xmm14, xmm10
|
||||
punpckldq xmm10, xmm11
|
||||
punpckhdq xmm14, xmm11
|
||||
movdqa xmm9, xmm8
|
||||
punpcklqdq xmm8, xmm10
|
||||
punpckhqdq xmm9, xmm10
|
||||
movdqa xmm13, xmm12
|
||||
punpcklqdq xmm12, xmm14
|
||||
punpckhqdq xmm13, xmm14
|
||||
movdqa xmmword ptr [rsp+0C0H], xmm8
|
||||
movdqa xmmword ptr [rsp+0D0H], xmm9
|
||||
movdqa xmmword ptr [rsp+0E0H], xmm12
|
||||
movdqa xmmword ptr [rsp+0F0H], xmm13
|
||||
movdqa xmm9, xmmword ptr [BLAKE3_IV_1]
|
||||
movdqa xmm10, xmmword ptr [BLAKE3_IV_2]
|
||||
movdqa xmm11, xmmword ptr [BLAKE3_IV_3]
|
||||
movdqa xmm12, xmmword ptr [rsp+110H]
|
||||
movdqa xmm13, xmmword ptr [rsp+120H]
|
||||
movdqa xmm14, xmmword ptr [BLAKE3_BLOCK_LEN]
|
||||
movd xmm15, eax
|
||||
pshufd xmm15, xmm15, 00H
|
||||
prefetcht0 byte ptr [r8+rdx+80H]
|
||||
prefetcht0 byte ptr [r9+rdx+80H]
|
||||
prefetcht0 byte ptr [r10+rdx+80H]
|
||||
prefetcht0 byte ptr [r11+rdx+80H]
|
||||
paddd xmm0, xmmword ptr [rsp]
|
||||
paddd xmm1, xmmword ptr [rsp+20H]
|
||||
paddd xmm2, xmmword ptr [rsp+40H]
|
||||
paddd xmm3, xmmword ptr [rsp+60H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [BLAKE3_IV_0]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+10H]
|
||||
paddd xmm1, xmmword ptr [rsp+30H]
|
||||
paddd xmm2, xmmword ptr [rsp+50H]
|
||||
paddd xmm3, xmmword ptr [rsp+70H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+80H]
|
||||
paddd xmm1, xmmword ptr [rsp+0A0H]
|
||||
paddd xmm2, xmmword ptr [rsp+0C0H]
|
||||
paddd xmm3, xmmword ptr [rsp+0E0H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+90H]
|
||||
paddd xmm1, xmmword ptr [rsp+0B0H]
|
||||
paddd xmm2, xmmword ptr [rsp+0D0H]
|
||||
paddd xmm3, xmmword ptr [rsp+0F0H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+20H]
|
||||
paddd xmm1, xmmword ptr [rsp+30H]
|
||||
paddd xmm2, xmmword ptr [rsp+70H]
|
||||
paddd xmm3, xmmword ptr [rsp+40H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+60H]
|
||||
paddd xmm1, xmmword ptr [rsp+0A0H]
|
||||
paddd xmm2, xmmword ptr [rsp]
|
||||
paddd xmm3, xmmword ptr [rsp+0D0H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+10H]
|
||||
paddd xmm1, xmmword ptr [rsp+0C0H]
|
||||
paddd xmm2, xmmword ptr [rsp+90H]
|
||||
paddd xmm3, xmmword ptr [rsp+0F0H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0B0H]
|
||||
paddd xmm1, xmmword ptr [rsp+50H]
|
||||
paddd xmm2, xmmword ptr [rsp+0E0H]
|
||||
paddd xmm3, xmmword ptr [rsp+80H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+30H]
|
||||
paddd xmm1, xmmword ptr [rsp+0A0H]
|
||||
paddd xmm2, xmmword ptr [rsp+0D0H]
|
||||
paddd xmm3, xmmword ptr [rsp+70H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+40H]
|
||||
paddd xmm1, xmmword ptr [rsp+0C0H]
|
||||
paddd xmm2, xmmword ptr [rsp+20H]
|
||||
paddd xmm3, xmmword ptr [rsp+0E0H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+60H]
|
||||
paddd xmm1, xmmword ptr [rsp+90H]
|
||||
paddd xmm2, xmmword ptr [rsp+0B0H]
|
||||
paddd xmm3, xmmword ptr [rsp+80H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+50H]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+0F0H]
|
||||
paddd xmm3, xmmword ptr [rsp+10H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0A0H]
|
||||
paddd xmm1, xmmword ptr [rsp+0C0H]
|
||||
paddd xmm2, xmmword ptr [rsp+0E0H]
|
||||
paddd xmm3, xmmword ptr [rsp+0D0H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+70H]
|
||||
paddd xmm1, xmmword ptr [rsp+90H]
|
||||
paddd xmm2, xmmword ptr [rsp+30H]
|
||||
paddd xmm3, xmmword ptr [rsp+0F0H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+40H]
|
||||
paddd xmm1, xmmword ptr [rsp+0B0H]
|
||||
paddd xmm2, xmmword ptr [rsp+50H]
|
||||
paddd xmm3, xmmword ptr [rsp+10H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp]
|
||||
paddd xmm1, xmmword ptr [rsp+20H]
|
||||
paddd xmm2, xmmword ptr [rsp+80H]
|
||||
paddd xmm3, xmmword ptr [rsp+60H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0C0H]
|
||||
paddd xmm1, xmmword ptr [rsp+90H]
|
||||
paddd xmm2, xmmword ptr [rsp+0F0H]
|
||||
paddd xmm3, xmmword ptr [rsp+0E0H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0D0H]
|
||||
paddd xmm1, xmmword ptr [rsp+0B0H]
|
||||
paddd xmm2, xmmword ptr [rsp+0A0H]
|
||||
paddd xmm3, xmmword ptr [rsp+80H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+70H]
|
||||
paddd xmm1, xmmword ptr [rsp+50H]
|
||||
paddd xmm2, xmmword ptr [rsp]
|
||||
paddd xmm3, xmmword ptr [rsp+60H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+20H]
|
||||
paddd xmm1, xmmword ptr [rsp+30H]
|
||||
paddd xmm2, xmmword ptr [rsp+10H]
|
||||
paddd xmm3, xmmword ptr [rsp+40H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+90H]
|
||||
paddd xmm1, xmmword ptr [rsp+0B0H]
|
||||
paddd xmm2, xmmword ptr [rsp+80H]
|
||||
paddd xmm3, xmmword ptr [rsp+0F0H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0E0H]
|
||||
paddd xmm1, xmmword ptr [rsp+50H]
|
||||
paddd xmm2, xmmword ptr [rsp+0C0H]
|
||||
paddd xmm3, xmmword ptr [rsp+10H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0D0H]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+20H]
|
||||
paddd xmm3, xmmword ptr [rsp+40H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+30H]
|
||||
paddd xmm1, xmmword ptr [rsp+0A0H]
|
||||
paddd xmm2, xmmword ptr [rsp+60H]
|
||||
paddd xmm3, xmmword ptr [rsp+70H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0B0H]
|
||||
paddd xmm1, xmmword ptr [rsp+50H]
|
||||
paddd xmm2, xmmword ptr [rsp+10H]
|
||||
paddd xmm3, xmmword ptr [rsp+80H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0F0H]
|
||||
paddd xmm1, xmmword ptr [rsp]
|
||||
paddd xmm2, xmmword ptr [rsp+90H]
|
||||
paddd xmm3, xmmword ptr [rsp+60H]
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm1, xmm5
|
||||
paddd xmm2, xmm6
|
||||
paddd xmm3, xmm7
|
||||
pxor xmm12, xmm0
|
||||
pxor xmm13, xmm1
|
||||
pxor xmm14, xmm2
|
||||
pxor xmm15, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
pshufb xmm15, xmm8
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm12
|
||||
paddd xmm9, xmm13
|
||||
paddd xmm10, xmm14
|
||||
paddd xmm11, xmm15
|
||||
pxor xmm4, xmm8
|
||||
pxor xmm5, xmm9
|
||||
pxor xmm6, xmm10
|
||||
pxor xmm7, xmm11
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0E0H]
|
||||
paddd xmm1, xmmword ptr [rsp+20H]
|
||||
paddd xmm2, xmmword ptr [rsp+30H]
|
||||
paddd xmm3, xmmword ptr [rsp+70H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT16]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
movdqa xmmword ptr [rsp+100H], xmm8
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 12
|
||||
pslld xmm5, 20
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 12
|
||||
pslld xmm6, 20
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 12
|
||||
pslld xmm7, 20
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 12
|
||||
pslld xmm4, 20
|
||||
por xmm4, xmm8
|
||||
paddd xmm0, xmmword ptr [rsp+0A0H]
|
||||
paddd xmm1, xmmword ptr [rsp+0C0H]
|
||||
paddd xmm2, xmmword ptr [rsp+40H]
|
||||
paddd xmm3, xmmword ptr [rsp+0D0H]
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm1, xmm6
|
||||
paddd xmm2, xmm7
|
||||
paddd xmm3, xmm4
|
||||
pxor xmm15, xmm0
|
||||
pxor xmm12, xmm1
|
||||
pxor xmm13, xmm2
|
||||
pxor xmm14, xmm3
|
||||
movdqa xmm8, xmmword ptr [ROT8]
|
||||
pshufb xmm15, xmm8
|
||||
pshufb xmm12, xmm8
|
||||
pshufb xmm13, xmm8
|
||||
pshufb xmm14, xmm8
|
||||
paddd xmm10, xmm15
|
||||
paddd xmm11, xmm12
|
||||
movdqa xmm8, xmmword ptr [rsp+100H]
|
||||
paddd xmm8, xmm13
|
||||
paddd xmm9, xmm14
|
||||
pxor xmm5, xmm10
|
||||
pxor xmm6, xmm11
|
||||
pxor xmm7, xmm8
|
||||
pxor xmm4, xmm9
|
||||
pxor xmm0, xmm8
|
||||
pxor xmm1, xmm9
|
||||
pxor xmm2, xmm10
|
||||
pxor xmm3, xmm11
|
||||
movdqa xmm8, xmm5
|
||||
psrld xmm8, 7
|
||||
pslld xmm5, 25
|
||||
por xmm5, xmm8
|
||||
movdqa xmm8, xmm6
|
||||
psrld xmm8, 7
|
||||
pslld xmm6, 25
|
||||
por xmm6, xmm8
|
||||
movdqa xmm8, xmm7
|
||||
psrld xmm8, 7
|
||||
pslld xmm7, 25
|
||||
por xmm7, xmm8
|
||||
movdqa xmm8, xmm4
|
||||
psrld xmm8, 7
|
||||
pslld xmm4, 25
|
||||
por xmm4, xmm8
|
||||
pxor xmm4, xmm12
|
||||
pxor xmm5, xmm13
|
||||
pxor xmm6, xmm14
|
||||
pxor xmm7, xmm15
|
||||
mov eax, r13d
|
||||
jne innerloop4
|
||||
movdqa xmm9, xmm0
|
||||
punpckldq xmm0, xmm1
|
||||
punpckhdq xmm9, xmm1
|
||||
movdqa xmm11, xmm2
|
||||
punpckldq xmm2, xmm3
|
||||
punpckhdq xmm11, xmm3
|
||||
movdqa xmm1, xmm0
|
||||
punpcklqdq xmm0, xmm2
|
||||
punpckhqdq xmm1, xmm2
|
||||
movdqa xmm3, xmm9
|
||||
punpcklqdq xmm9, xmm11
|
||||
punpckhqdq xmm3, xmm11
|
||||
movdqu xmmword ptr [rbx], xmm0
|
||||
movdqu xmmword ptr [rbx+20H], xmm1
|
||||
movdqu xmmword ptr [rbx+40H], xmm9
|
||||
movdqu xmmword ptr [rbx+60H], xmm3
|
||||
movdqa xmm9, xmm4
|
||||
punpckldq xmm4, xmm5
|
||||
punpckhdq xmm9, xmm5
|
||||
movdqa xmm11, xmm6
|
||||
punpckldq xmm6, xmm7
|
||||
punpckhdq xmm11, xmm7
|
||||
movdqa xmm5, xmm4
|
||||
punpcklqdq xmm4, xmm6
|
||||
punpckhqdq xmm5, xmm6
|
||||
movdqa xmm7, xmm9
|
||||
punpcklqdq xmm9, xmm11
|
||||
punpckhqdq xmm7, xmm11
|
||||
movdqu xmmword ptr [rbx+10H], xmm4
|
||||
movdqu xmmword ptr [rbx+30H], xmm5
|
||||
movdqu xmmword ptr [rbx+50H], xmm9
|
||||
movdqu xmmword ptr [rbx+70H], xmm7
|
||||
movdqa xmm1, xmmword ptr [rsp+110H]
|
||||
movdqa xmm0, xmm1
|
||||
paddd xmm1, xmmword ptr [rsp+150H]
|
||||
movdqa xmmword ptr [rsp+110H], xmm1
|
||||
pxor xmm0, xmmword ptr [CMP_MSB_MASK]
|
||||
pxor xmm1, xmmword ptr [CMP_MSB_MASK]
|
||||
pcmpgtd xmm0, xmm1
|
||||
movdqa xmm1, xmmword ptr [rsp+120H]
|
||||
psubd xmm1, xmm0
|
||||
movdqa xmmword ptr [rsp+120H], xmm1
|
||||
add rbx, 128
|
||||
add rdi, 32
|
||||
sub rsi, 4
|
||||
cmp rsi, 4
|
||||
jnc outerloop4
|
||||
test rsi, rsi
|
||||
jne final3blocks
|
||||
unwind:
|
||||
movdqa xmm6, xmmword ptr [rsp+170H]
|
||||
movdqa xmm7, xmmword ptr [rsp+180H]
|
||||
movdqa xmm8, xmmword ptr [rsp+190H]
|
||||
movdqa xmm9, xmmword ptr [rsp+1A0H]
|
||||
movdqa xmm10, xmmword ptr [rsp+1B0H]
|
||||
movdqa xmm11, xmmword ptr [rsp+1C0H]
|
||||
movdqa xmm12, xmmword ptr [rsp+1D0H]
|
||||
movdqa xmm13, xmmword ptr [rsp+1E0H]
|
||||
movdqa xmm14, xmmword ptr [rsp+1F0H]
|
||||
movdqa xmm15, xmmword ptr [rsp+200H]
|
||||
mov rsp, rbp
|
||||
pop rbp
|
||||
pop rbx
|
||||
pop rdi
|
||||
pop rsi
|
||||
pop r12
|
||||
pop r13
|
||||
pop r14
|
||||
pop r15
|
||||
ret
|
||||
ALIGN 16
|
||||
final3blocks:
|
||||
test esi, 2H
|
||||
je final1block
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+10H]
|
||||
movaps xmm8, xmm0
|
||||
movaps xmm9, xmm1
|
||||
movd xmm13, dword ptr [rsp+110H]
|
||||
pinsrd xmm13, dword ptr [rsp+120H], 1
|
||||
pinsrd xmm13, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
movaps xmmword ptr [rsp], xmm13
|
||||
movd xmm14, dword ptr [rsp+114H]
|
||||
pinsrd xmm14, dword ptr [rsp+124H], 1
|
||||
pinsrd xmm14, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
movaps xmmword ptr [rsp+10H], xmm14
|
||||
mov r8, qword ptr [rdi]
|
||||
mov r9, qword ptr [rdi+8H]
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
innerloop2:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV]
|
||||
movaps xmm10, xmm2
|
||||
movups xmm4, xmmword ptr [r8+rdx-40H]
|
||||
movups xmm5, xmmword ptr [r8+rdx-30H]
|
||||
movaps xmm3, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm3, xmm5, 221
|
||||
movaps xmm5, xmm3
|
||||
movups xmm6, xmmword ptr [r8+rdx-20H]
|
||||
movups xmm7, xmmword ptr [r8+rdx-10H]
|
||||
movaps xmm3, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 93H
|
||||
shufps xmm3, xmm7, 221
|
||||
pshufd xmm7, xmm3, 93H
|
||||
movups xmm12, xmmword ptr [r9+rdx-40H]
|
||||
movups xmm13, xmmword ptr [r9+rdx-30H]
|
||||
movaps xmm11, xmm12
|
||||
shufps xmm12, xmm13, 136
|
||||
shufps xmm11, xmm13, 221
|
||||
movaps xmm13, xmm11
|
||||
movups xmm14, xmmword ptr [r9+rdx-20H]
|
||||
movups xmm15, xmmword ptr [r9+rdx-10H]
|
||||
movaps xmm11, xmm14
|
||||
shufps xmm14, xmm15, 136
|
||||
pshufd xmm14, xmm14, 93H
|
||||
shufps xmm11, xmm15, 221
|
||||
pshufd xmm15, xmm11, 93H
|
||||
movaps xmm3, xmmword ptr [rsp]
|
||||
movaps xmm11, xmmword ptr [rsp+10H]
|
||||
pinsrd xmm3, eax, 3
|
||||
pinsrd xmm11, eax, 3
|
||||
mov al, 7
|
||||
roundloop2:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm8, xmm12
|
||||
movaps xmmword ptr [rsp+20H], xmm4
|
||||
movaps xmmword ptr [rsp+30H], xmm12
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
movaps xmm12, xmmword ptr [ROT16]
|
||||
pshufb xmm3, xmm12
|
||||
pshufb xmm11, xmm12
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm4, 12
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 20
|
||||
psrld xmm4, 12
|
||||
por xmm9, xmm4
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm8, xmm13
|
||||
movaps xmmword ptr [rsp+40H], xmm5
|
||||
movaps xmmword ptr [rsp+50H], xmm13
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
movaps xmm13, xmmword ptr [ROT8]
|
||||
pshufb xmm3, xmm13
|
||||
pshufb xmm11, xmm13
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm4, 7
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 25
|
||||
psrld xmm4, 7
|
||||
por xmm9, xmm4
|
||||
pshufd xmm0, xmm0, 93H
|
||||
pshufd xmm8, xmm8, 93H
|
||||
pshufd xmm3, xmm3, 4EH
|
||||
pshufd xmm11, xmm11, 4EH
|
||||
pshufd xmm2, xmm2, 39H
|
||||
pshufd xmm10, xmm10, 39H
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm8, xmm14
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
pshufb xmm3, xmm12
|
||||
pshufb xmm11, xmm12
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm4, 12
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 20
|
||||
psrld xmm4, 12
|
||||
por xmm9, xmm4
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm8, xmm15
|
||||
paddd xmm0, xmm1
|
||||
paddd xmm8, xmm9
|
||||
pxor xmm3, xmm0
|
||||
pxor xmm11, xmm8
|
||||
pshufb xmm3, xmm13
|
||||
pshufb xmm11, xmm13
|
||||
paddd xmm2, xmm3
|
||||
paddd xmm10, xmm11
|
||||
pxor xmm1, xmm2
|
||||
pxor xmm9, xmm10
|
||||
movdqa xmm4, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm4, 7
|
||||
por xmm1, xmm4
|
||||
movdqa xmm4, xmm9
|
||||
pslld xmm9, 25
|
||||
psrld xmm4, 7
|
||||
por xmm9, xmm4
|
||||
pshufd xmm0, xmm0, 39H
|
||||
pshufd xmm8, xmm8, 39H
|
||||
pshufd xmm3, xmm3, 4EH
|
||||
pshufd xmm11, xmm11, 4EH
|
||||
pshufd xmm2, xmm2, 93H
|
||||
pshufd xmm10, xmm10, 93H
|
||||
dec al
|
||||
je endroundloop2
|
||||
movdqa xmm12, xmmword ptr [rsp+20H]
|
||||
movdqa xmm5, xmmword ptr [rsp+40H]
|
||||
pshufd xmm13, xmm12, 0FH
|
||||
shufps xmm12, xmm5, 214
|
||||
pshufd xmm4, xmm12, 39H
|
||||
movdqa xmm12, xmm6
|
||||
shufps xmm12, xmm7, 250
|
||||
pblendw xmm13, xmm12, 0CCH
|
||||
movdqa xmm12, xmm7
|
||||
punpcklqdq xmm12, xmm5
|
||||
pblendw xmm12, xmm6, 0C0H
|
||||
pshufd xmm12, xmm12, 78H
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 1EH
|
||||
movdqa xmmword ptr [rsp+20H], xmm13
|
||||
movdqa xmmword ptr [rsp+40H], xmm12
|
||||
movdqa xmm5, xmmword ptr [rsp+30H]
|
||||
movdqa xmm13, xmmword ptr [rsp+50H]
|
||||
pshufd xmm6, xmm5, 0FH
|
||||
shufps xmm5, xmm13, 214
|
||||
pshufd xmm12, xmm5, 39H
|
||||
movdqa xmm5, xmm14
|
||||
shufps xmm5, xmm15, 250
|
||||
pblendw xmm6, xmm5, 0CCH
|
||||
movdqa xmm5, xmm15
|
||||
punpcklqdq xmm5, xmm13
|
||||
pblendw xmm5, xmm14, 0C0H
|
||||
pshufd xmm5, xmm5, 78H
|
||||
punpckhdq xmm13, xmm15
|
||||
punpckldq xmm14, xmm13
|
||||
pshufd xmm15, xmm14, 1EH
|
||||
movdqa xmm13, xmm6
|
||||
movdqa xmm14, xmm5
|
||||
movdqa xmm5, xmmword ptr [rsp+20H]
|
||||
movdqa xmm6, xmmword ptr [rsp+40H]
|
||||
jmp roundloop2
|
||||
endroundloop2:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
pxor xmm8, xmm10
|
||||
pxor xmm9, xmm11
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne innerloop2
|
||||
movups xmmword ptr [rbx], xmm0
|
||||
movups xmmword ptr [rbx+10H], xmm1
|
||||
movups xmmword ptr [rbx+20H], xmm8
|
||||
movups xmmword ptr [rbx+30H], xmm9
|
||||
movdqa xmm0, xmmword ptr [rsp+130H]
|
||||
movdqa xmm1, xmmword ptr [rsp+110H]
|
||||
movdqa xmm2, xmmword ptr [rsp+120H]
|
||||
movdqu xmm3, xmmword ptr [rsp+118H]
|
||||
movdqu xmm4, xmmword ptr [rsp+128H]
|
||||
blendvps xmm1, xmm3, xmm0
|
||||
blendvps xmm2, xmm4, xmm0
|
||||
movdqa xmmword ptr [rsp+110H], xmm1
|
||||
movdqa xmmword ptr [rsp+120H], xmm2
|
||||
add rdi, 16
|
||||
add rbx, 64
|
||||
sub rsi, 2
|
||||
final1block:
|
||||
test esi, 1H
|
||||
je unwind
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+10H]
|
||||
movd xmm13, dword ptr [rsp+110H]
|
||||
pinsrd xmm13, dword ptr [rsp+120H], 1
|
||||
pinsrd xmm13, dword ptr [BLAKE3_BLOCK_LEN], 2
|
||||
movaps xmm14, xmmword ptr [ROT8]
|
||||
movaps xmm15, xmmword ptr [ROT16]
|
||||
mov r8, qword ptr [rdi]
|
||||
movzx eax, byte ptr [rbp+80H]
|
||||
or eax, r13d
|
||||
xor edx, edx
|
||||
innerloop1:
|
||||
mov r14d, eax
|
||||
or eax, r12d
|
||||
add rdx, 64
|
||||
cmp rdx, r15
|
||||
cmovne eax, r14d
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV]
|
||||
movaps xmm3, xmm13
|
||||
pinsrd xmm3, eax, 3
|
||||
movups xmm4, xmmword ptr [r8+rdx-40H]
|
||||
movups xmm5, xmmword ptr [r8+rdx-30H]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [r8+rdx-20H]
|
||||
movups xmm7, xmmword ptr [r8+rdx-10H]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 93H
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 93H
|
||||
mov al, 7
|
||||
roundloop1:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 93H
|
||||
pshufd xmm3, xmm3, 4EH
|
||||
pshufd xmm2, xmm2, 39H
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 39H
|
||||
pshufd xmm3, xmm3, 4EH
|
||||
pshufd xmm2, xmm2, 93H
|
||||
dec al
|
||||
jz endroundloop1
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0FH
|
||||
pshufd xmm4, xmm8, 39H
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0CCH
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0C0H
|
||||
pshufd xmm8, xmm8, 78H
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 1EH
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp roundloop1
|
||||
endroundloop1:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
mov eax, r13d
|
||||
cmp rdx, r15
|
||||
jne innerloop1
|
||||
movups xmmword ptr [rbx], xmm0
|
||||
movups xmmword ptr [rbx+10H], xmm1
|
||||
jmp unwind
|
||||
_blake3_hash_many_sse41 ENDP
|
||||
blake3_hash_many_sse41 ENDP
|
||||
|
||||
blake3_compress_in_place_sse41 PROC
|
||||
_blake3_compress_in_place_sse41 PROC
|
||||
sub rsp, 72
|
||||
movdqa xmmword ptr [rsp], xmm6
|
||||
movdqa xmmword ptr [rsp+10H], xmm7
|
||||
movdqa xmmword ptr [rsp+20H], xmm8
|
||||
movdqa xmmword ptr [rsp+30H], xmm9
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+10H]
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV]
|
||||
movzx eax, byte ptr [rsp+70H]
|
||||
movzx r8d, r8b
|
||||
shl rax, 32
|
||||
add r8, rax
|
||||
movq xmm3, r9
|
||||
movq xmm4, r8
|
||||
punpcklqdq xmm3, xmm4
|
||||
movups xmm4, xmmword ptr [rdx]
|
||||
movups xmm5, xmmword ptr [rdx+10H]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [rdx+20H]
|
||||
movups xmm7, xmmword ptr [rdx+30H]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 93H
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 93H
|
||||
movaps xmm14, xmmword ptr [ROT8]
|
||||
movaps xmm15, xmmword ptr [ROT16]
|
||||
mov al, 7
|
||||
@@:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 93H
|
||||
pshufd xmm3, xmm3, 4EH
|
||||
pshufd xmm2, xmm2, 39H
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 39H
|
||||
pshufd xmm3, xmm3, 4EH
|
||||
pshufd xmm2, xmm2, 93H
|
||||
dec al
|
||||
jz @F
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0FH
|
||||
pshufd xmm4, xmm8, 39H
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0CCH
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0C0H
|
||||
pshufd xmm8, xmm8, 78H
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 1EH
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp @B
|
||||
@@:
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
movups xmmword ptr [rcx], xmm0
|
||||
movups xmmword ptr [rcx+10H], xmm1
|
||||
movdqa xmm6, xmmword ptr [rsp]
|
||||
movdqa xmm7, xmmword ptr [rsp+10H]
|
||||
movdqa xmm8, xmmword ptr [rsp+20H]
|
||||
movdqa xmm9, xmmword ptr [rsp+30H]
|
||||
add rsp, 72
|
||||
ret
|
||||
_blake3_compress_in_place_sse41 ENDP
|
||||
blake3_compress_in_place_sse41 ENDP
|
||||
|
||||
ALIGN 16
|
||||
blake3_compress_xof_sse41 PROC
|
||||
_blake3_compress_xof_sse41 PROC
|
||||
sub rsp, 72
|
||||
movdqa xmmword ptr [rsp], xmm6
|
||||
movdqa xmmword ptr [rsp+10H], xmm7
|
||||
movdqa xmmword ptr [rsp+20H], xmm8
|
||||
movdqa xmmword ptr [rsp+30H], xmm9
|
||||
movups xmm0, xmmword ptr [rcx]
|
||||
movups xmm1, xmmword ptr [rcx+10H]
|
||||
movaps xmm2, xmmword ptr [BLAKE3_IV]
|
||||
movzx eax, byte ptr [rsp+70H]
|
||||
movzx r8d, r8b
|
||||
mov r10, qword ptr [rsp+78H]
|
||||
shl rax, 32
|
||||
add r8, rax
|
||||
movq xmm3, r9
|
||||
movq xmm4, r8
|
||||
punpcklqdq xmm3, xmm4
|
||||
movups xmm4, xmmword ptr [rdx]
|
||||
movups xmm5, xmmword ptr [rdx+10H]
|
||||
movaps xmm8, xmm4
|
||||
shufps xmm4, xmm5, 136
|
||||
shufps xmm8, xmm5, 221
|
||||
movaps xmm5, xmm8
|
||||
movups xmm6, xmmword ptr [rdx+20H]
|
||||
movups xmm7, xmmword ptr [rdx+30H]
|
||||
movaps xmm8, xmm6
|
||||
shufps xmm6, xmm7, 136
|
||||
pshufd xmm6, xmm6, 93H
|
||||
shufps xmm8, xmm7, 221
|
||||
pshufd xmm7, xmm8, 93H
|
||||
movaps xmm14, xmmword ptr [ROT8]
|
||||
movaps xmm15, xmmword ptr [ROT16]
|
||||
mov al, 7
|
||||
@@:
|
||||
paddd xmm0, xmm4
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm5
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 93H
|
||||
pshufd xmm3, xmm3, 4EH
|
||||
pshufd xmm2, xmm2, 39H
|
||||
paddd xmm0, xmm6
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm15
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 20
|
||||
psrld xmm11, 12
|
||||
por xmm1, xmm11
|
||||
paddd xmm0, xmm7
|
||||
paddd xmm0, xmm1
|
||||
pxor xmm3, xmm0
|
||||
pshufb xmm3, xmm14
|
||||
paddd xmm2, xmm3
|
||||
pxor xmm1, xmm2
|
||||
movdqa xmm11, xmm1
|
||||
pslld xmm1, 25
|
||||
psrld xmm11, 7
|
||||
por xmm1, xmm11
|
||||
pshufd xmm0, xmm0, 39H
|
||||
pshufd xmm3, xmm3, 4EH
|
||||
pshufd xmm2, xmm2, 93H
|
||||
dec al
|
||||
jz @F
|
||||
movdqa xmm8, xmm4
|
||||
shufps xmm8, xmm5, 214
|
||||
pshufd xmm9, xmm4, 0FH
|
||||
pshufd xmm4, xmm8, 39H
|
||||
movdqa xmm8, xmm6
|
||||
shufps xmm8, xmm7, 250
|
||||
pblendw xmm9, xmm8, 0CCH
|
||||
movdqa xmm8, xmm7
|
||||
punpcklqdq xmm8, xmm5
|
||||
pblendw xmm8, xmm6, 0C0H
|
||||
pshufd xmm8, xmm8, 78H
|
||||
punpckhdq xmm5, xmm7
|
||||
punpckldq xmm6, xmm5
|
||||
pshufd xmm7, xmm6, 1EH
|
||||
movdqa xmm5, xmm9
|
||||
movdqa xmm6, xmm8
|
||||
jmp @B
|
||||
@@:
|
||||
movdqu xmm4, xmmword ptr [rcx]
|
||||
movdqu xmm5, xmmword ptr [rcx+10H]
|
||||
pxor xmm0, xmm2
|
||||
pxor xmm1, xmm3
|
||||
pxor xmm2, xmm4
|
||||
pxor xmm3, xmm5
|
||||
movups xmmword ptr [r10], xmm0
|
||||
movups xmmword ptr [r10+10H], xmm1
|
||||
movups xmmword ptr [r10+20H], xmm2
|
||||
movups xmmword ptr [r10+30H], xmm3
|
||||
movdqa xmm6, xmmword ptr [rsp]
|
||||
movdqa xmm7, xmmword ptr [rsp+10H]
|
||||
movdqa xmm8, xmmword ptr [rsp+20H]
|
||||
movdqa xmm9, xmmword ptr [rsp+30H]
|
||||
add rsp, 72
|
||||
ret
|
||||
_blake3_compress_xof_sse41 ENDP
|
||||
blake3_compress_xof_sse41 ENDP
|
||||
|
||||
_TEXT ENDS
|
||||
|
||||
|
||||
_RDATA SEGMENT READONLY PAGE ALIAS(".rdata") 'CONST'
|
||||
ALIGN 64
|
||||
BLAKE3_IV:
|
||||
dd 6A09E667H, 0BB67AE85H, 3C6EF372H, 0A54FF53AH
|
||||
|
||||
ADD0:
|
||||
dd 0, 1, 2, 3
|
||||
|
||||
ADD1:
|
||||
dd 4 dup (4)
|
||||
|
||||
BLAKE3_IV_0:
|
||||
dd 4 dup (6A09E667H)
|
||||
|
||||
BLAKE3_IV_1:
|
||||
dd 4 dup (0BB67AE85H)
|
||||
|
||||
BLAKE3_IV_2:
|
||||
dd 4 dup (3C6EF372H)
|
||||
|
||||
BLAKE3_IV_3:
|
||||
dd 4 dup (0A54FF53AH)
|
||||
|
||||
BLAKE3_BLOCK_LEN:
|
||||
dd 4 dup (64)
|
||||
|
||||
ROT16:
|
||||
db 2, 3, 0, 1, 6, 7, 4, 5, 10, 11, 8, 9, 14, 15, 12, 13
|
||||
|
||||
ROT8:
|
||||
db 1, 2, 3, 0, 5, 6, 7, 4, 9, 10, 11, 8, 13, 14, 15, 12
|
||||
|
||||
CMP_MSB_MASK:
|
||||
dd 8 dup(80000000H)
|
||||
|
||||
_RDATA ENDS
|
||||
END
|
||||
|
Loading…
Reference in New Issue