diff --git a/apple-x86/crypto/fipsmodule/sha256-586-apple.S b/apple-x86/crypto/fipsmodule/sha256-586-apple.S index d43510a49..53421104c 100644 --- a/apple-x86/crypto/fipsmodule/sha256-586-apple.S +++ b/apple-x86/crypto/fipsmodule/sha256-586-apple.S @@ -33,26 +33,24 @@ L000pic_point: movl L_OPENSSL_ia32cap_P$non_lazy_ptr-L001K256(%ebp),%edx movl (%edx),%ecx movl 4(%edx),%ebx - testl $1048576,%ecx - jnz L002loop movl 8(%edx),%edx testl $16777216,%ecx - jz L003no_xmm + jz L002no_xmm andl $1073741824,%ecx andl $268435968,%ebx orl %ebx,%ecx andl $1342177280,%ecx cmpl $1342177280,%ecx - je L004AVX + je L003AVX testl $512,%ebx - jnz L005SSSE3 -L003no_xmm: + jnz L004SSSE3 +L002no_xmm: subl %edi,%eax cmpl $256,%eax - jae L006unrolled - jmp L002loop + jae L005unrolled + jmp L006loop .align 4,0x90 -L002loop: +L006loop: movl (%edi),%eax movl 4(%edi),%ebx movl 8(%edi),%ecx @@ -245,7 +243,7 @@ L00816_63: leal 356(%esp),%esp subl $256,%ebp cmpl 8(%esp),%edi - jb L002loop + jb L006loop movl 12(%esp),%esp popl %edi popl %esi @@ -262,7 +260,7 @@ L001K256: .byte 112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103 .byte 62,0 .align 4,0x90 -L006unrolled: +L005unrolled: leal -96(%esp),%esp movl (%esi),%eax movl 4(%esi),%ebp @@ -3169,7 +3167,7 @@ L009grand_loop: popl %ebp ret .align 5,0x90 -L005SSSE3: +L004SSSE3: leal -96(%esp),%esp movl (%esi),%eax movl 4(%esi),%ebx @@ -4380,7 +4378,7 @@ L011ssse3_00_47: popl %ebp ret .align 5,0x90 -L004AVX: +L003AVX: leal -96(%esp),%esp vzeroall movl (%esi),%eax diff --git a/linux-x86/crypto/fipsmodule/sha256-586-linux.S b/linux-x86/crypto/fipsmodule/sha256-586-linux.S index ee41b78cb..7da9d1eb4 100644 --- a/linux-x86/crypto/fipsmodule/sha256-586-linux.S +++ b/linux-x86/crypto/fipsmodule/sha256-586-linux.S @@ -34,26 +34,24 @@ sha256_block_data_order: leal OPENSSL_ia32cap_P-.L001K256(%ebp),%edx movl (%edx),%ecx movl 4(%edx),%ebx - testl $1048576,%ecx - jnz .L002loop movl 8(%edx),%edx testl $16777216,%ecx - jz .L003no_xmm + jz .L002no_xmm andl $1073741824,%ecx andl $268435968,%ebx orl %ebx,%ecx andl $1342177280,%ecx cmpl $1342177280,%ecx - je .L004AVX + je .L003AVX testl $512,%ebx - jnz .L005SSSE3 -.L003no_xmm: + jnz .L004SSSE3 +.L002no_xmm: subl %edi,%eax cmpl $256,%eax - jae .L006unrolled - jmp .L002loop + jae .L005unrolled + jmp .L006loop .align 16 -.L002loop: +.L006loop: movl (%edi),%eax movl 4(%edi),%ebx movl 8(%edi),%ecx @@ -246,7 +244,7 @@ sha256_block_data_order: leal 356(%esp),%esp subl $256,%ebp cmpl 8(%esp),%edi - jb .L002loop + jb .L006loop movl 12(%esp),%esp popl %edi popl %esi @@ -263,7 +261,7 @@ sha256_block_data_order: .byte 112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103 .byte 62,0 .align 16 -.L006unrolled: +.L005unrolled: leal -96(%esp),%esp movl (%esi),%eax movl 4(%esi),%ebp @@ -3170,7 +3168,7 @@ sha256_block_data_order: popl %ebp ret .align 32 -.L005SSSE3: +.L004SSSE3: leal -96(%esp),%esp movl (%esi),%eax movl 4(%esi),%ebx @@ -4381,7 +4379,7 @@ sha256_block_data_order: popl %ebp ret .align 32 -.L004AVX: +.L003AVX: leal -96(%esp),%esp vzeroall movl (%esi),%eax diff --git a/src/crypto/fipsmodule/sha/asm/sha256-586.pl b/src/crypto/fipsmodule/sha/asm/sha256-586.pl index d23d81fbf..ab821e745 100644 --- a/src/crypto/fipsmodule/sha/asm/sha256-586.pl +++ b/src/crypto/fipsmodule/sha/asm/sha256-586.pl @@ -211,8 +211,6 @@ sub BODY_00_15() { &picmeup("edx","OPENSSL_ia32cap_P",$K256,&label("K256")); &mov ("ecx",&DWP(0,"edx")); &mov ("ebx",&DWP(4,"edx")); - &test ("ecx",1<<20); # check for P4 - &jnz (&label("loop")); &mov ("edx",&DWP(8,"edx")) if ($xmm); &test ("ecx",1<<24); # check for FXSR &jz ($unroll_after?&label("no_xmm"):&label("loop")); diff --git a/win-x86/crypto/fipsmodule/sha256-586-win.asm b/win-x86/crypto/fipsmodule/sha256-586-win.asm index 4e0278b0a..65f3b549a 100644 --- a/win-x86/crypto/fipsmodule/sha256-586-win.asm +++ b/win-x86/crypto/fipsmodule/sha256-586-win.asm @@ -41,26 +41,24 @@ L$000pic_point: lea edx,[_OPENSSL_ia32cap_P] mov ecx,DWORD [edx] mov ebx,DWORD [4+edx] - test ecx,1048576 - jnz NEAR L$002loop mov edx,DWORD [8+edx] test ecx,16777216 - jz NEAR L$003no_xmm + jz NEAR L$002no_xmm and ecx,1073741824 and ebx,268435968 or ecx,ebx and ecx,1342177280 cmp ecx,1342177280 - je NEAR L$004AVX + je NEAR L$003AVX test ebx,512 - jnz NEAR L$005SSSE3 -L$003no_xmm: + jnz NEAR L$004SSSE3 +L$002no_xmm: sub eax,edi cmp eax,256 - jae NEAR L$006unrolled - jmp NEAR L$002loop + jae NEAR L$005unrolled + jmp NEAR L$006loop align 16 -L$002loop: +L$006loop: mov eax,DWORD [edi] mov ebx,DWORD [4+edi] mov ecx,DWORD [8+edi] @@ -253,7 +251,7 @@ L$00816_63: lea esp,[356+esp] sub ebp,256 cmp edi,DWORD [8+esp] - jb NEAR L$002loop + jb NEAR L$006loop mov esp,DWORD [12+esp] pop edi pop esi @@ -270,7 +268,7 @@ db 67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97 db 112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103 db 62,0 align 16 -L$006unrolled: +L$005unrolled: lea esp,[esp-96] mov eax,DWORD [esi] mov ebp,DWORD [4+esi] @@ -3177,7 +3175,7 @@ L$009grand_loop: pop ebp ret align 32 -L$005SSSE3: +L$004SSSE3: lea esp,[esp-96] mov eax,DWORD [esi] mov ebx,DWORD [4+esi] @@ -4388,7 +4386,7 @@ db 102,15,58,15,249,4 pop ebp ret align 32 -L$004AVX: +L$003AVX: lea esp,[esp-96] vzeroall mov eax,DWORD [esi]