update main-with-bazel from master branch
This commit is contained in:
@@ -14,27 +14,14 @@
|
||||
.code 32
|
||||
#endif
|
||||
|
||||
.globl _sha1_block_data_order
|
||||
.private_extern _sha1_block_data_order
|
||||
.globl _sha1_block_data_order_nohw
|
||||
.private_extern _sha1_block_data_order_nohw
|
||||
#ifdef __thumb2__
|
||||
.thumb_func _sha1_block_data_order
|
||||
.thumb_func _sha1_block_data_order_nohw
|
||||
#endif
|
||||
|
||||
.align 5
|
||||
_sha1_block_data_order:
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
Lsha1_block:
|
||||
adr r3,Lsha1_block
|
||||
ldr r12,LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV8_SHA1
|
||||
bne LARMv8
|
||||
tst r12,#ARMV7_NEON
|
||||
bne LNEON
|
||||
#endif
|
||||
_sha1_block_data_order_nohw:
|
||||
stmdb sp!,{r4,r5,r6,r7,r8,r9,r10,r11,r12,lr}
|
||||
add r2,r1,r2,lsl#6 @ r2 to point at the end of r1
|
||||
ldmia r0,{r3,r4,r5,r6,r7}
|
||||
@@ -492,10 +479,6 @@ LK_00_19:.word 0x5a827999
|
||||
LK_20_39:.word 0x6ed9eba1
|
||||
LK_40_59:.word 0x8f1bbcdc
|
||||
LK_60_79:.word 0xca62c1d6
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-Lsha1_block
|
||||
#endif
|
||||
.byte 83,72,65,49,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,52,47,78,69,79,78,47,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
|
||||
.align 2
|
||||
.align 5
|
||||
@@ -503,12 +486,13 @@ LOPENSSL_armcap:
|
||||
|
||||
|
||||
|
||||
.globl _sha1_block_data_order_neon
|
||||
.private_extern _sha1_block_data_order_neon
|
||||
#ifdef __thumb2__
|
||||
.thumb_func sha1_block_data_order_neon
|
||||
.thumb_func _sha1_block_data_order_neon
|
||||
#endif
|
||||
.align 4
|
||||
sha1_block_data_order_neon:
|
||||
LNEON:
|
||||
_sha1_block_data_order_neon:
|
||||
stmdb sp!,{r4,r5,r6,r7,r8,r9,r10,r11,r12,lr}
|
||||
add r2,r1,r2,lsl#6 @ r2 to point at the end of r1
|
||||
@ dmb @ errata #451034 on early Cortex A8
|
||||
@@ -1364,12 +1348,13 @@ Loop_neon:
|
||||
# define INST(a,b,c,d) .byte a,b,c,d|0x10
|
||||
# endif
|
||||
|
||||
.globl _sha1_block_data_order_hw
|
||||
.private_extern _sha1_block_data_order_hw
|
||||
#ifdef __thumb2__
|
||||
.thumb_func sha1_block_data_order_armv8
|
||||
.thumb_func _sha1_block_data_order_hw
|
||||
#endif
|
||||
.align 5
|
||||
sha1_block_data_order_armv8:
|
||||
LARMv8:
|
||||
_sha1_block_data_order_hw:
|
||||
vstmdb sp!,{d8,d9,d10,d11,d12,d13,d14,d15} @ ABI specification says so
|
||||
|
||||
veor q1,q1,q1
|
||||
@@ -1498,13 +1483,5 @@ Loop_v8:
|
||||
vldmia sp!,{d8,d9,d10,d11,d12,d13,d14,d15}
|
||||
bx lr @ bx lr
|
||||
|
||||
#endif
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.comm _OPENSSL_armcap_P,4
|
||||
.non_lazy_symbol_pointer
|
||||
OPENSSL_armcap_P:
|
||||
.indirect_symbol _OPENSSL_armcap_P
|
||||
.long 0
|
||||
.private_extern _OPENSSL_armcap_P
|
||||
#endif
|
||||
#endif // !OPENSSL_NO_ASM && defined(OPENSSL_ARM) && defined(__APPLE__)
|
||||
|
||||
@@ -90,37 +90,18 @@ K256:
|
||||
.word 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2
|
||||
|
||||
.word 0 @ terminator
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-Lsha256_block_data_order
|
||||
#endif
|
||||
.align 5
|
||||
|
||||
.globl _sha256_block_data_order
|
||||
.private_extern _sha256_block_data_order
|
||||
.globl _sha256_block_data_order_nohw
|
||||
.private_extern _sha256_block_data_order_nohw
|
||||
#ifdef __thumb2__
|
||||
.thumb_func _sha256_block_data_order
|
||||
#endif
|
||||
_sha256_block_data_order:
|
||||
Lsha256_block_data_order:
|
||||
adr r3,Lsha256_block_data_order
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
ldr r12,LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV8_SHA256
|
||||
bne LARMv8
|
||||
tst r12,#ARMV7_NEON
|
||||
bne LNEON
|
||||
.thumb_func _sha256_block_data_order_nohw
|
||||
#endif
|
||||
_sha256_block_data_order_nohw:
|
||||
add r2,r1,r2,lsl#6 @ len to point at the end of inp
|
||||
stmdb sp!,{r0,r1,r2,r4-r11,lr}
|
||||
ldmia r0,{r4,r5,r6,r7,r8,r9,r10,r11}
|
||||
@ TODO(davidben): When the OPENSSL_armcap logic above is removed,
|
||||
@ replace this with a simple ADR.
|
||||
sub r14,r3,#256+32 @ K256
|
||||
adr r14,K256
|
||||
sub sp,sp,#16*4 @ alloca(X[16])
|
||||
Loop:
|
||||
# if __ARM_ARCH>=7
|
||||
@@ -1895,10 +1876,12 @@ Lrounds_16_xx:
|
||||
.align 5
|
||||
.skip 16
|
||||
_sha256_block_data_order_neon:
|
||||
LNEON:
|
||||
stmdb sp!,{r4,r5,r6,r7,r8,r9,r10,r11,r12,lr}
|
||||
|
||||
sub r11,sp,#16*4+16
|
||||
@ In Arm mode, the following ADR runs up against the limits of encodable
|
||||
@ offsets. It only fits because the offset, when the ADR is placed here,
|
||||
@ is a multiple of 16.
|
||||
adr r14,K256
|
||||
bic r11,r11,#15 @ align for 128-bit stores
|
||||
mov r12,sp
|
||||
@@ -2681,14 +2664,29 @@ L_00_48:
|
||||
# define INST(a,b,c,d) .byte a,b,c,d
|
||||
# endif
|
||||
|
||||
LK256_shortcut:
|
||||
@ PC is 8 bytes ahead in Arm mode and 4 bytes ahead in Thumb mode.
|
||||
#if defined(__thumb2__)
|
||||
.word K256-(LK256_add+4)
|
||||
#else
|
||||
.word K256-(LK256_add+8)
|
||||
#endif
|
||||
|
||||
.globl _sha256_block_data_order_hw
|
||||
.private_extern _sha256_block_data_order_hw
|
||||
#ifdef __thumb2__
|
||||
.thumb_func sha256_block_data_order_armv8
|
||||
.thumb_func _sha256_block_data_order_hw
|
||||
#endif
|
||||
.align 5
|
||||
sha256_block_data_order_armv8:
|
||||
LARMv8:
|
||||
_sha256_block_data_order_hw:
|
||||
@ K256 is too far to reference from one ADR command in Thumb mode. In
|
||||
@ Arm mode, we could make it fit by aligning the ADR offset to a 64-byte
|
||||
@ boundary. For simplicity, just load the offset from .LK256_shortcut.
|
||||
ldr r3,LK256_shortcut
|
||||
LK256_add:
|
||||
add r3,pc,r3
|
||||
|
||||
vld1.32 {q0,q1},[r0]
|
||||
sub r3,r3,#256+32
|
||||
add r2,r1,r2,lsl#6 @ len to point at the end of inp
|
||||
b Loop_v8
|
||||
|
||||
@@ -2825,12 +2823,4 @@ Loop_v8:
|
||||
.byte 83,72,65,50,53,54,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,52,47,78,69,79,78,47,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
|
||||
.align 2
|
||||
.align 2
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.comm _OPENSSL_armcap_P,4
|
||||
.non_lazy_symbol_pointer
|
||||
OPENSSL_armcap_P:
|
||||
.indirect_symbol _OPENSSL_armcap_P
|
||||
.long 0
|
||||
.private_extern _OPENSSL_armcap_P
|
||||
#endif
|
||||
#endif // !OPENSSL_NO_ASM && defined(OPENSSL_ARM) && defined(__APPLE__)
|
||||
|
||||
@@ -135,36 +135,16 @@ K512:
|
||||
WORD64(0x4cc5d4be,0xcb3e42b6, 0x597f299c,0xfc657e2a)
|
||||
WORD64(0x5fcb6fab,0x3ad6faec, 0x6c44198c,0x4a475817)
|
||||
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-Lsha512_block_data_order
|
||||
.skip 32-4
|
||||
#else
|
||||
.skip 32
|
||||
#endif
|
||||
|
||||
.globl _sha512_block_data_order
|
||||
.private_extern _sha512_block_data_order
|
||||
.globl _sha512_block_data_order_nohw
|
||||
.private_extern _sha512_block_data_order_nohw
|
||||
#ifdef __thumb2__
|
||||
.thumb_func _sha512_block_data_order
|
||||
#endif
|
||||
_sha512_block_data_order:
|
||||
Lsha512_block_data_order:
|
||||
adr r3,Lsha512_block_data_order
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
ldr r12,LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV7_NEON
|
||||
bne LNEON
|
||||
.thumb_func _sha512_block_data_order_nohw
|
||||
#endif
|
||||
_sha512_block_data_order_nohw:
|
||||
add r2,r1,r2,lsl#7 @ len to point at the end of inp
|
||||
stmdb sp!,{r4,r5,r6,r7,r8,r9,r10,r11,r12,lr}
|
||||
@ TODO(davidben): When the OPENSSL_armcap logic above is removed,
|
||||
@ replace this with a simple ADR.
|
||||
sub r14,r3,#672 @ K512
|
||||
adr r14,K512
|
||||
sub sp,sp,#9*8
|
||||
|
||||
ldr r7,[r0,#32+LO]
|
||||
@@ -551,7 +531,6 @@ L16_79:
|
||||
#endif
|
||||
.align 4
|
||||
_sha512_block_data_order_neon:
|
||||
LNEON:
|
||||
dmb @ errata #451034 on early Cortex A8
|
||||
add r2,r1,r2,lsl#7 @ len to point at the end of inp
|
||||
adr r3,K512
|
||||
@@ -1877,12 +1856,4 @@ L16_79_neon:
|
||||
.byte 83,72,65,53,49,50,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,52,47,78,69,79,78,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
|
||||
.align 2
|
||||
.align 2
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.comm _OPENSSL_armcap_P,4
|
||||
.non_lazy_symbol_pointer
|
||||
OPENSSL_armcap_P:
|
||||
.indirect_symbol _OPENSSL_armcap_P
|
||||
.long 0
|
||||
.private_extern _OPENSSL_armcap_P
|
||||
#endif
|
||||
#endif // !OPENSSL_NO_ASM && defined(OPENSSL_ARM) && defined(__APPLE__)
|
||||
|
||||
@@ -14,25 +14,12 @@
|
||||
.code 32
|
||||
#endif
|
||||
|
||||
.globl sha1_block_data_order
|
||||
.hidden sha1_block_data_order
|
||||
.type sha1_block_data_order,%function
|
||||
.globl sha1_block_data_order_nohw
|
||||
.hidden sha1_block_data_order_nohw
|
||||
.type sha1_block_data_order_nohw,%function
|
||||
|
||||
.align 5
|
||||
sha1_block_data_order:
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.Lsha1_block:
|
||||
adr r3,.Lsha1_block
|
||||
ldr r12,.LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV8_SHA1
|
||||
bne .LARMv8
|
||||
tst r12,#ARMV7_NEON
|
||||
bne .LNEON
|
||||
#endif
|
||||
sha1_block_data_order_nohw:
|
||||
stmdb sp!,{r4,r5,r6,r7,r8,r9,r10,r11,r12,lr}
|
||||
add r2,r1,r2,lsl#6 @ r2 to point at the end of r1
|
||||
ldmia r0,{r3,r4,r5,r6,r7}
|
||||
@@ -483,17 +470,13 @@ sha1_block_data_order:
|
||||
moveq pc,lr @ be binary compatible with V4, yet
|
||||
.word 0xe12fff1e @ interoperable with Thumb ISA:-)
|
||||
#endif
|
||||
.size sha1_block_data_order,.-sha1_block_data_order
|
||||
.size sha1_block_data_order_nohw,.-sha1_block_data_order_nohw
|
||||
|
||||
.align 5
|
||||
.LK_00_19:.word 0x5a827999
|
||||
.LK_20_39:.word 0x6ed9eba1
|
||||
.LK_40_59:.word 0x8f1bbcdc
|
||||
.LK_60_79:.word 0xca62c1d6
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-.Lsha1_block
|
||||
#endif
|
||||
.byte 83,72,65,49,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,52,47,78,69,79,78,47,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
|
||||
.align 2
|
||||
.align 5
|
||||
@@ -501,10 +484,11 @@ sha1_block_data_order:
|
||||
.arch armv7-a
|
||||
.fpu neon
|
||||
|
||||
.globl sha1_block_data_order_neon
|
||||
.hidden sha1_block_data_order_neon
|
||||
.type sha1_block_data_order_neon,%function
|
||||
.align 4
|
||||
sha1_block_data_order_neon:
|
||||
.LNEON:
|
||||
stmdb sp!,{r4,r5,r6,r7,r8,r9,r10,r11,r12,lr}
|
||||
add r2,r1,r2,lsl#6 @ r2 to point at the end of r1
|
||||
@ dmb @ errata #451034 on early Cortex A8
|
||||
@@ -1360,10 +1344,11 @@ sha1_block_data_order_neon:
|
||||
# define INST(a,b,c,d) .byte a,b,c,d|0x10
|
||||
# endif
|
||||
|
||||
.type sha1_block_data_order_armv8,%function
|
||||
.globl sha1_block_data_order_hw
|
||||
.hidden sha1_block_data_order_hw
|
||||
.type sha1_block_data_order_hw,%function
|
||||
.align 5
|
||||
sha1_block_data_order_armv8:
|
||||
.LARMv8:
|
||||
sha1_block_data_order_hw:
|
||||
vstmdb sp!,{d8,d9,d10,d11,d12,d13,d14,d15} @ ABI specification says so
|
||||
|
||||
veor q1,q1,q1
|
||||
@@ -1491,10 +1476,6 @@ sha1_block_data_order_armv8:
|
||||
|
||||
vldmia sp!,{d8,d9,d10,d11,d12,d13,d14,d15}
|
||||
bx lr @ bx lr
|
||||
.size sha1_block_data_order_armv8,.-sha1_block_data_order_armv8
|
||||
#endif
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.comm OPENSSL_armcap_P,4,4
|
||||
.hidden OPENSSL_armcap_P
|
||||
.size sha1_block_data_order_hw,.-sha1_block_data_order_hw
|
||||
#endif
|
||||
#endif // !OPENSSL_NO_ASM && defined(OPENSSL_ARM) && defined(__ELF__)
|
||||
|
||||
@@ -90,35 +90,16 @@ K256:
|
||||
.word 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2
|
||||
.size K256,.-K256
|
||||
.word 0 @ terminator
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-.Lsha256_block_data_order
|
||||
#endif
|
||||
.align 5
|
||||
|
||||
.globl sha256_block_data_order
|
||||
.hidden sha256_block_data_order
|
||||
.type sha256_block_data_order,%function
|
||||
sha256_block_data_order:
|
||||
.Lsha256_block_data_order:
|
||||
adr r3,.Lsha256_block_data_order
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
ldr r12,.LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV8_SHA256
|
||||
bne .LARMv8
|
||||
tst r12,#ARMV7_NEON
|
||||
bne .LNEON
|
||||
#endif
|
||||
.globl sha256_block_data_order_nohw
|
||||
.hidden sha256_block_data_order_nohw
|
||||
.type sha256_block_data_order_nohw,%function
|
||||
sha256_block_data_order_nohw:
|
||||
add r2,r1,r2,lsl#6 @ len to point at the end of inp
|
||||
stmdb sp!,{r0,r1,r2,r4-r11,lr}
|
||||
ldmia r0,{r4,r5,r6,r7,r8,r9,r10,r11}
|
||||
@ TODO(davidben): When the OPENSSL_armcap logic above is removed,
|
||||
@ replace this with a simple ADR.
|
||||
sub r14,r3,#256+32 @ K256
|
||||
adr r14,K256
|
||||
sub sp,sp,#16*4 @ alloca(X[16])
|
||||
.Loop:
|
||||
# if __ARM_ARCH>=7
|
||||
@@ -1880,7 +1861,7 @@ sha256_block_data_order:
|
||||
moveq pc,lr @ be binary compatible with V4, yet
|
||||
.word 0xe12fff1e @ interoperable with Thumb ISA:-)
|
||||
#endif
|
||||
.size sha256_block_data_order,.-sha256_block_data_order
|
||||
.size sha256_block_data_order_nohw,.-sha256_block_data_order_nohw
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.arch armv7-a
|
||||
.fpu neon
|
||||
@@ -1891,10 +1872,12 @@ sha256_block_data_order:
|
||||
.align 5
|
||||
.skip 16
|
||||
sha256_block_data_order_neon:
|
||||
.LNEON:
|
||||
stmdb sp!,{r4,r5,r6,r7,r8,r9,r10,r11,r12,lr}
|
||||
|
||||
sub r11,sp,#16*4+16
|
||||
@ In Arm mode, the following ADR runs up against the limits of encodable
|
||||
@ offsets. It only fits because the offset, when the ADR is placed here,
|
||||
@ is a multiple of 16.
|
||||
adr r14,K256
|
||||
bic r11,r11,#15 @ align for 128-bit stores
|
||||
mov r12,sp
|
||||
@@ -2677,12 +2660,27 @@ sha256_block_data_order_neon:
|
||||
# define INST(a,b,c,d) .byte a,b,c,d
|
||||
# endif
|
||||
|
||||
.type sha256_block_data_order_armv8,%function
|
||||
.LK256_shortcut:
|
||||
@ PC is 8 bytes ahead in Arm mode and 4 bytes ahead in Thumb mode.
|
||||
#if defined(__thumb2__)
|
||||
.word K256-(.LK256_add+4)
|
||||
#else
|
||||
.word K256-(.LK256_add+8)
|
||||
#endif
|
||||
|
||||
.globl sha256_block_data_order_hw
|
||||
.hidden sha256_block_data_order_hw
|
||||
.type sha256_block_data_order_hw,%function
|
||||
.align 5
|
||||
sha256_block_data_order_armv8:
|
||||
.LARMv8:
|
||||
sha256_block_data_order_hw:
|
||||
@ K256 is too far to reference from one ADR command in Thumb mode. In
|
||||
@ Arm mode, we could make it fit by aligning the ADR offset to a 64-byte
|
||||
@ boundary. For simplicity, just load the offset from .LK256_shortcut.
|
||||
ldr r3,.LK256_shortcut
|
||||
.LK256_add:
|
||||
add r3,pc,r3
|
||||
|
||||
vld1.32 {q0,q1},[r0]
|
||||
sub r3,r3,#256+32
|
||||
add r2,r1,r2,lsl#6 @ len to point at the end of inp
|
||||
b .Loop_v8
|
||||
|
||||
@@ -2814,13 +2812,9 @@ sha256_block_data_order_armv8:
|
||||
vst1.32 {q0,q1},[r0]
|
||||
|
||||
bx lr @ bx lr
|
||||
.size sha256_block_data_order_armv8,.-sha256_block_data_order_armv8
|
||||
.size sha256_block_data_order_hw,.-sha256_block_data_order_hw
|
||||
#endif
|
||||
.byte 83,72,65,50,53,54,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,52,47,78,69,79,78,47,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
|
||||
.align 2
|
||||
.align 2
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.comm OPENSSL_armcap_P,4,4
|
||||
.hidden OPENSSL_armcap_P
|
||||
#endif
|
||||
#endif // !OPENSSL_NO_ASM && defined(OPENSSL_ARM) && defined(__ELF__)
|
||||
|
||||
@@ -135,34 +135,14 @@ K512:
|
||||
WORD64(0x4cc5d4be,0xcb3e42b6, 0x597f299c,0xfc657e2a)
|
||||
WORD64(0x5fcb6fab,0x3ad6faec, 0x6c44198c,0x4a475817)
|
||||
.size K512,.-K512
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-.Lsha512_block_data_order
|
||||
.skip 32-4
|
||||
#else
|
||||
.skip 32
|
||||
#endif
|
||||
|
||||
.globl sha512_block_data_order
|
||||
.hidden sha512_block_data_order
|
||||
.type sha512_block_data_order,%function
|
||||
sha512_block_data_order:
|
||||
.Lsha512_block_data_order:
|
||||
adr r3,.Lsha512_block_data_order
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
ldr r12,.LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV7_NEON
|
||||
bne .LNEON
|
||||
#endif
|
||||
.globl sha512_block_data_order_nohw
|
||||
.hidden sha512_block_data_order_nohw
|
||||
.type sha512_block_data_order_nohw,%function
|
||||
sha512_block_data_order_nohw:
|
||||
add r2,r1,r2,lsl#7 @ len to point at the end of inp
|
||||
stmdb sp!,{r4,r5,r6,r7,r8,r9,r10,r11,r12,lr}
|
||||
@ TODO(davidben): When the OPENSSL_armcap logic above is removed,
|
||||
@ replace this with a simple ADR.
|
||||
sub r14,r3,#672 @ K512
|
||||
adr r14,K512
|
||||
sub sp,sp,#9*8
|
||||
|
||||
ldr r7,[r0,#32+LO]
|
||||
@@ -537,7 +517,7 @@ sha512_block_data_order:
|
||||
moveq pc,lr @ be binary compatible with V4, yet
|
||||
.word 0xe12fff1e @ interoperable with Thumb ISA:-)
|
||||
#endif
|
||||
.size sha512_block_data_order,.-sha512_block_data_order
|
||||
.size sha512_block_data_order_nohw,.-sha512_block_data_order_nohw
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.arch armv7-a
|
||||
.fpu neon
|
||||
@@ -547,7 +527,6 @@ sha512_block_data_order:
|
||||
.type sha512_block_data_order_neon,%function
|
||||
.align 4
|
||||
sha512_block_data_order_neon:
|
||||
.LNEON:
|
||||
dmb @ errata #451034 on early Cortex A8
|
||||
add r2,r1,r2,lsl#7 @ len to point at the end of inp
|
||||
adr r3,K512
|
||||
@@ -1873,8 +1852,4 @@ sha512_block_data_order_neon:
|
||||
.byte 83,72,65,53,49,50,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,52,47,78,69,79,78,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
|
||||
.align 2
|
||||
.align 2
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.comm OPENSSL_armcap_P,4,4
|
||||
.hidden OPENSSL_armcap_P
|
||||
#endif
|
||||
#endif // !OPENSSL_NO_ASM && defined(OPENSSL_ARM) && defined(__ELF__)
|
||||
|
||||
@@ -197,24 +197,11 @@ $code=<<___;
|
||||
.code 32
|
||||
#endif
|
||||
|
||||
.global sha1_block_data_order
|
||||
.type sha1_block_data_order,%function
|
||||
.global sha1_block_data_order_nohw
|
||||
.type sha1_block_data_order_nohw,%function
|
||||
|
||||
.align 5
|
||||
sha1_block_data_order:
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.Lsha1_block:
|
||||
adr r3,.Lsha1_block
|
||||
ldr r12,.LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV8_SHA1
|
||||
bne .LARMv8
|
||||
tst r12,#ARMV7_NEON
|
||||
bne .LNEON
|
||||
#endif
|
||||
sha1_block_data_order_nohw:
|
||||
stmdb sp!,{r4-r12,lr}
|
||||
add $len,$inp,$len,lsl#6 @ $len to point at the end of $inp
|
||||
ldmia $ctx,{$a,$b,$c,$d,$e}
|
||||
@@ -304,17 +291,13 @@ $code.=<<___;
|
||||
moveq pc,lr @ be binary compatible with V4, yet
|
||||
bx lr @ interoperable with Thumb ISA:-)
|
||||
#endif
|
||||
.size sha1_block_data_order,.-sha1_block_data_order
|
||||
.size sha1_block_data_order_nohw,.-sha1_block_data_order_nohw
|
||||
|
||||
.align 5
|
||||
.LK_00_19: .word 0x5a827999
|
||||
.LK_20_39: .word 0x6ed9eba1
|
||||
.LK_40_59: .word 0x8f1bbcdc
|
||||
.LK_60_79: .word 0xca62c1d6
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-.Lsha1_block
|
||||
#endif
|
||||
.asciz "SHA1 block transform for ARMv4/NEON/ARMv8, CRYPTOGAMS by <appro\@openssl.org>"
|
||||
.align 5
|
||||
___
|
||||
@@ -530,10 +513,10 @@ $code.=<<___;
|
||||
.arch armv7-a
|
||||
.fpu neon
|
||||
|
||||
.global sha1_block_data_order_neon
|
||||
.type sha1_block_data_order_neon,%function
|
||||
.align 4
|
||||
sha1_block_data_order_neon:
|
||||
.LNEON:
|
||||
stmdb sp!,{r4-r12,lr}
|
||||
add $len,$inp,$len,lsl#6 @ $len to point at the end of $inp
|
||||
@ dmb @ errata #451034 on early Cortex A8
|
||||
@@ -625,10 +608,10 @@ $code.=<<___;
|
||||
# define INST(a,b,c,d) .byte a,b,c,d|0x10
|
||||
# endif
|
||||
|
||||
.type sha1_block_data_order_armv8,%function
|
||||
.global sha1_block_data_order_hw
|
||||
.type sha1_block_data_order_hw,%function
|
||||
.align 5
|
||||
sha1_block_data_order_armv8:
|
||||
.LARMv8:
|
||||
sha1_block_data_order_hw:
|
||||
vstmdb sp!,{d8-d15} @ ABI specification says so
|
||||
|
||||
veor $E,$E,$E
|
||||
@@ -693,16 +676,10 @@ $code.=<<___;
|
||||
|
||||
vldmia sp!,{d8-d15}
|
||||
ret @ bx lr
|
||||
.size sha1_block_data_order_armv8,.-sha1_block_data_order_armv8
|
||||
.size sha1_block_data_order_hw,.-sha1_block_data_order_hw
|
||||
#endif
|
||||
___
|
||||
}}}
|
||||
$code.=<<___;
|
||||
#if __ARM_MAX_ARCH__>=7
|
||||
.comm OPENSSL_armcap_P,4,4
|
||||
.hidden OPENSSL_armcap_P
|
||||
#endif
|
||||
___
|
||||
|
||||
{ my %opcode = (
|
||||
"sha1c" => 0xf2000c40, "sha1p" => 0xf2100c40,
|
||||
|
||||
@@ -217,34 +217,15 @@ K256:
|
||||
.word 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2
|
||||
.size K256,.-K256
|
||||
.word 0 @ terminator
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-.Lsha256_block_data_order
|
||||
#endif
|
||||
.align 5
|
||||
|
||||
.global sha256_block_data_order
|
||||
.type sha256_block_data_order,%function
|
||||
sha256_block_data_order:
|
||||
.Lsha256_block_data_order:
|
||||
adr r3,.Lsha256_block_data_order
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
ldr r12,.LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV8_SHA256
|
||||
bne .LARMv8
|
||||
tst r12,#ARMV7_NEON
|
||||
bne .LNEON
|
||||
#endif
|
||||
.global sha256_block_data_order_nohw
|
||||
.type sha256_block_data_order_nohw,%function
|
||||
sha256_block_data_order_nohw:
|
||||
add $len,$inp,$len,lsl#6 @ len to point at the end of inp
|
||||
stmdb sp!,{$ctx,$inp,$len,r4-r11,lr}
|
||||
ldmia $ctx,{$A,$B,$C,$D,$E,$F,$G,$H}
|
||||
@ TODO(davidben): When the OPENSSL_armcap logic above is removed,
|
||||
@ replace this with a simple ADR.
|
||||
sub $Ktbl,r3,#256+32 @ K256
|
||||
adr $Ktbl,K256
|
||||
sub sp,sp,#16*4 @ alloca(X[16])
|
||||
.Loop:
|
||||
# if __ARM_ARCH>=7
|
||||
@@ -298,7 +279,7 @@ $code.=<<___;
|
||||
moveq pc,lr @ be binary compatible with V4, yet
|
||||
bx lr @ interoperable with Thumb ISA:-)
|
||||
#endif
|
||||
.size sha256_block_data_order,.-sha256_block_data_order
|
||||
.size sha256_block_data_order_nohw,.-sha256_block_data_order_nohw
|
||||
___
|
||||
######################################################################
|
||||
# NEON stuff
|
||||
@@ -483,10 +464,12 @@ $code.=<<___;
|
||||
.align 5
|
||||
.skip 16
|
||||
sha256_block_data_order_neon:
|
||||
.LNEON:
|
||||
stmdb sp!,{r4-r12,lr}
|
||||
|
||||
sub $H,sp,#16*4+16
|
||||
@ In Arm mode, the following ADR runs up against the limits of encodable
|
||||
@ offsets. It only fits because the offset, when the ADR is placed here,
|
||||
@ is a multiple of 16.
|
||||
adr $Ktbl,K256
|
||||
bic $H,$H,#15 @ align for 128-bit stores
|
||||
mov $t2,sp
|
||||
@@ -613,12 +596,26 @@ $code.=<<___;
|
||||
# define INST(a,b,c,d) .byte a,b,c,d
|
||||
# endif
|
||||
|
||||
.type sha256_block_data_order_armv8,%function
|
||||
.LK256_shortcut:
|
||||
@ PC is 8 bytes ahead in Arm mode and 4 bytes ahead in Thumb mode.
|
||||
#if defined(__thumb2__)
|
||||
.word K256-(.LK256_add+4)
|
||||
#else
|
||||
.word K256-(.LK256_add+8)
|
||||
#endif
|
||||
|
||||
.global sha256_block_data_order_hw
|
||||
.type sha256_block_data_order_hw,%function
|
||||
.align 5
|
||||
sha256_block_data_order_armv8:
|
||||
.LARMv8:
|
||||
sha256_block_data_order_hw:
|
||||
@ K256 is too far to reference from one ADR command in Thumb mode. In
|
||||
@ Arm mode, we could make it fit by aligning the ADR offset to a 64-byte
|
||||
@ boundary. For simplicity, just load the offset from .LK256_shortcut.
|
||||
ldr $Ktbl,.LK256_shortcut
|
||||
.LK256_add:
|
||||
add $Ktbl,pc,$Ktbl
|
||||
|
||||
vld1.32 {$ABCD,$EFGH},[$ctx]
|
||||
sub $Ktbl,$Ktbl,#256+32
|
||||
add $len,$inp,$len,lsl#6 @ len to point at the end of inp
|
||||
b .Loop_v8
|
||||
|
||||
@@ -680,17 +677,13 @@ $code.=<<___;
|
||||
vst1.32 {$ABCD,$EFGH},[$ctx]
|
||||
|
||||
ret @ bx lr
|
||||
.size sha256_block_data_order_armv8,.-sha256_block_data_order_armv8
|
||||
.size sha256_block_data_order_hw,.-sha256_block_data_order_hw
|
||||
#endif
|
||||
___
|
||||
}}}
|
||||
$code.=<<___;
|
||||
.asciz "SHA256 block transform for ARMv4/NEON/ARMv8, CRYPTOGAMS by <appro\@openssl.org>"
|
||||
.align 2
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.comm OPENSSL_armcap_P,4,4
|
||||
.hidden OPENSSL_armcap_P
|
||||
#endif
|
||||
___
|
||||
|
||||
open SELF,$0;
|
||||
|
||||
@@ -276,33 +276,13 @@ WORD64(0x3c9ebe0a,0x15c9bebc, 0x431d67c4,0x9c100d4c)
|
||||
WORD64(0x4cc5d4be,0xcb3e42b6, 0x597f299c,0xfc657e2a)
|
||||
WORD64(0x5fcb6fab,0x3ad6faec, 0x6c44198c,0x4a475817)
|
||||
.size K512,.-K512
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.LOPENSSL_armcap:
|
||||
.word OPENSSL_armcap_P-.Lsha512_block_data_order
|
||||
.skip 32-4
|
||||
#else
|
||||
.skip 32
|
||||
#endif
|
||||
|
||||
.global sha512_block_data_order
|
||||
.type sha512_block_data_order,%function
|
||||
sha512_block_data_order:
|
||||
.Lsha512_block_data_order:
|
||||
adr r3,.Lsha512_block_data_order
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
ldr r12,.LOPENSSL_armcap
|
||||
ldr r12,[r3,r12] @ OPENSSL_armcap_P
|
||||
#ifdef __APPLE__
|
||||
ldr r12,[r12]
|
||||
#endif
|
||||
tst r12,#ARMV7_NEON
|
||||
bne .LNEON
|
||||
#endif
|
||||
.global sha512_block_data_order_nohw
|
||||
.type sha512_block_data_order_nohw,%function
|
||||
sha512_block_data_order_nohw:
|
||||
add $len,$inp,$len,lsl#7 @ len to point at the end of inp
|
||||
stmdb sp!,{r4-r12,lr}
|
||||
@ TODO(davidben): When the OPENSSL_armcap logic above is removed,
|
||||
@ replace this with a simple ADR.
|
||||
sub $Ktbl,r3,#672 @ K512
|
||||
adr $Ktbl,K512
|
||||
sub sp,sp,#9*8
|
||||
|
||||
ldr $Elo,[$ctx,#$Eoff+$lo]
|
||||
@@ -501,7 +481,7 @@ $code.=<<___;
|
||||
moveq pc,lr @ be binary compatible with V4, yet
|
||||
bx lr @ interoperable with Thumb ISA:-)
|
||||
#endif
|
||||
.size sha512_block_data_order,.-sha512_block_data_order
|
||||
.size sha512_block_data_order_nohw,.-sha512_block_data_order_nohw
|
||||
___
|
||||
|
||||
{
|
||||
@@ -612,7 +592,6 @@ $code.=<<___;
|
||||
.type sha512_block_data_order_neon,%function
|
||||
.align 4
|
||||
sha512_block_data_order_neon:
|
||||
.LNEON:
|
||||
dmb @ errata #451034 on early Cortex A8
|
||||
add $len,$inp,$len,lsl#7 @ len to point at the end of inp
|
||||
adr $Ktbl,K512
|
||||
@@ -650,10 +629,6 @@ ___
|
||||
$code.=<<___;
|
||||
.asciz "SHA512 block transform for ARMv4/NEON, CRYPTOGAMS by <appro\@openssl.org>"
|
||||
.align 2
|
||||
#if __ARM_MAX_ARCH__>=7 && !defined(__KERNEL__)
|
||||
.comm OPENSSL_armcap_P,4,4
|
||||
.hidden OPENSSL_armcap_P
|
||||
#endif
|
||||
___
|
||||
|
||||
$code =~ s/\`([^\`]*)\`/eval $1/gem;
|
||||
|
||||
@@ -26,7 +26,7 @@ extern "C" {
|
||||
// Define SHA{n}[_{variant}]_ASM if sha{n}_block_data_order[_{variant}] is
|
||||
// defined in assembly.
|
||||
|
||||
#if !defined(OPENSSL_NO_ASM) && (defined(OPENSSL_X86) || defined(OPENSSL_ARM))
|
||||
#if !defined(OPENSSL_NO_ASM) && defined(OPENSSL_X86)
|
||||
|
||||
#define SHA1_ASM
|
||||
#define SHA256_ASM
|
||||
@@ -39,6 +39,35 @@ void sha256_block_data_order(uint32_t *state, const uint8_t *data,
|
||||
void sha512_block_data_order(uint64_t *state, const uint8_t *data,
|
||||
size_t num_blocks);
|
||||
|
||||
#elif !defined(OPENSSL_NO_ASM) && defined(OPENSSL_ARM)
|
||||
|
||||
#define SHA1_ASM_NOHW
|
||||
#define SHA256_ASM_NOHW
|
||||
#define SHA512_ASM_NOHW
|
||||
|
||||
#define SHA1_ASM_HW
|
||||
OPENSSL_INLINE int sha1_hw_capable(void) {
|
||||
return CRYPTO_is_ARMv8_SHA1_capable();
|
||||
}
|
||||
|
||||
#define SHA1_ASM_NEON
|
||||
void sha1_block_data_order_neon(uint32_t *state, const uint8_t *data,
|
||||
size_t num);
|
||||
|
||||
#define SHA256_ASM_HW
|
||||
OPENSSL_INLINE int sha256_hw_capable(void) {
|
||||
return CRYPTO_is_ARMv8_SHA256_capable();
|
||||
}
|
||||
|
||||
#define SHA256_ASM_NEON
|
||||
void sha256_block_data_order_neon(uint32_t *state, const uint8_t *data,
|
||||
size_t num);
|
||||
|
||||
// Armv8.2 SHA-512 instructions are not available in 32-bit.
|
||||
#define SHA512_ASM_NEON
|
||||
void sha512_block_data_order_neon(uint64_t *state, const uint8_t *data,
|
||||
size_t num);
|
||||
|
||||
#elif !defined(OPENSSL_NO_ASM) && defined(OPENSSL_AARCH64)
|
||||
|
||||
#define SHA1_ASM_NOHW
|
||||
@@ -148,6 +177,7 @@ void sha256_block_data_order_nohw(uint32_t *state, const uint8_t *data,
|
||||
void sha512_block_data_order_hw(uint64_t *state, const uint8_t *data,
|
||||
size_t num);
|
||||
#endif
|
||||
|
||||
#if defined(SHA512_ASM_NOHW)
|
||||
void sha512_block_data_order_nohw(uint64_t *state, const uint8_t *data,
|
||||
size_t num);
|
||||
|
||||
@@ -409,6 +409,12 @@ static void sha1_block_data_order(uint32_t *state, const uint8_t *data,
|
||||
sha1_block_data_order_ssse3(state, data, num);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA1_ASM_NEON)
|
||||
if (CRYPTO_is_NEON_capable()) {
|
||||
sha1_block_data_order_neon(state, data, num);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
sha1_block_data_order_nohw(state, data, num);
|
||||
}
|
||||
|
||||
@@ -331,6 +331,12 @@ static void sha256_block_data_order(uint32_t *state, const uint8_t *data,
|
||||
sha256_block_data_order_ssse3(state, data, num);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA256_ASM_NEON)
|
||||
if (CRYPTO_is_NEON_capable()) {
|
||||
sha256_block_data_order_neon(state, data, num);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
sha256_block_data_order_nohw(state, data, num);
|
||||
}
|
||||
|
||||
@@ -515,6 +515,12 @@ static void sha512_block_data_order(uint64_t *state, const uint8_t *data,
|
||||
sha512_block_data_order_avx(state, data, num);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA512_ASM_NEON)
|
||||
if (CRYPTO_is_NEON_capable()) {
|
||||
sha512_block_data_order_neon(state, data, num);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
sha512_block_data_order_nohw(state, data, num);
|
||||
}
|
||||
|
||||
@@ -75,6 +75,11 @@ TEST(SHATest, SHA1ABI) {
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA1_ASM_NEON)
|
||||
if (CRYPTO_is_NEON_capable()) {
|
||||
CHECK_ABI(sha1_block_data_order_neon, ctx.h, kBuf, blocks);
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA1_ASM_NOHW)
|
||||
CHECK_ABI(sha1_block_data_order_nohw, ctx.h, kBuf, blocks);
|
||||
#endif
|
||||
@@ -107,6 +112,11 @@ TEST(SHATest, SHA256ABI) {
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA256_ASM_NEON)
|
||||
if (CRYPTO_is_NEON_capable()) {
|
||||
CHECK_ABI(sha256_block_data_order_neon, ctx.h, kBuf, blocks);
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA256_ASM_NOHW)
|
||||
CHECK_ABI(sha256_block_data_order_nohw, ctx.h, kBuf, blocks);
|
||||
#endif
|
||||
@@ -132,6 +142,11 @@ TEST(SHATest, SHA512ABI) {
|
||||
CHECK_ABI(sha512_block_data_order_avx, ctx.h, kBuf, blocks);
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA512_ASM_NEON)
|
||||
if (CRYPTO_is_NEON_capable()) {
|
||||
CHECK_ABI(sha512_block_data_order_neon, ctx.h, kBuf, blocks);
|
||||
}
|
||||
#endif
|
||||
#if defined(SHA512_ASM_NOHW)
|
||||
CHECK_ABI(sha512_block_data_order_nohw, ctx.h, kBuf, blocks);
|
||||
#endif
|
||||
|
||||
Reference in New Issue
Block a user