From 16d2d4b45408d19291ead8bfa8b5afa43742d033 Mon Sep 17 00:00:00 2001 From: Alexandre Dumont Date: Tue, 27 Mar 2012 00:20:58 +0200 Subject: [PATCH 1/5] bionic: Add NASTY_PTHREAD_CREATE_HACK This is kanged from commit 678ffd0 (Change-Id: Ice2da9c02c3c84b9ed417e623b248a87b8f9678e) by Steve Kondik. This is a workaround for bad linkage of proprietary libs on certain devices. Thanks to Zinx for tracking it down. --- libc/Android.mk | 4 ++++ libc/bionic/pthread.c | 7 +++++++ libc/include/pthread.h | 6 ++++++ 3 files changed, 17 insertions(+) diff --git a/libc/Android.mk b/libc/Android.mk index cb0c336973..35176978d9 100644 --- a/libc/Android.mk +++ b/libc/Android.mk @@ -563,6 +563,10 @@ ifeq ($(TARGET_ARCH),arm) libc_crt_target_cflags += -DCRT_LEGACY_WORKAROUND endif +ifeq ($(BOARD_USE_NASTY_PTHREAD_CREATE_HACK),true) + libc_common_cflags += -DNASTY_PTHREAD_CREATE_HACK +endif + # Define some common includes # ======================================================== libc_common_c_includes := \ diff --git a/libc/bionic/pthread.c b/libc/bionic/pthread.c index 3435d219cb..f6210ecd07 100644 --- a/libc/bionic/pthread.c +++ b/libc/bionic/pthread.c @@ -386,6 +386,13 @@ int pthread_create(pthread_t *thread_out, pthread_attr_t const * attr, return 0; } +#ifdef NASTY_PTHREAD_CREATE_HACK +int _debug_pthread_create(void *debug0, void *debug1, pthread_t *thread, + const pthread_attr_t *attr, void *(*start_routine) (void *), void *arg) +{ + return pthread_create(thread, attr, start_routine, arg); +} +#endif int pthread_attr_init(pthread_attr_t * attr) { diff --git a/libc/include/pthread.h b/libc/include/pthread.h index 56c48eafca..eeed476da6 100644 --- a/libc/include/pthread.h +++ b/libc/include/pthread.h @@ -138,6 +138,12 @@ int pthread_getattr_np(pthread_t thid, pthread_attr_t * attr); int pthread_create(pthread_t *thread, pthread_attr_t const * attr, void *(*start_routine)(void *), void * arg); + +#ifdef NASTY_PTHREAD_CREATE_HACK +int _debug_pthread_create(void *debug0, void *debug1, pthread_t *thread, + const pthread_attr_t *attr, void *(*start_routine) (void *), void *arg); +#endif + void pthread_exit(void * retval); int pthread_join(pthread_t thid, void ** ret_val); int pthread_detach(pthread_t thid); From 08bda34a5b3113f4f04bc2a257f884d3b8f4f72d Mon Sep 17 00:00:00 2001 From: podxboq Date: Fri, 11 May 2012 00:59:15 +0200 Subject: [PATCH 2/5] Add support for omx/mm-video qcom open source --- libc/kernel/common/linux/msm_q6vdec.h | 49 ++++++++++++++++++++++++++- 1 file changed, 48 insertions(+), 1 deletion(-) diff --git a/libc/kernel/common/linux/msm_q6vdec.h b/libc/kernel/common/linux/msm_q6vdec.h index 0182bfbf71..6dbd0e7971 100644 --- a/libc/kernel/common/linux/msm_q6vdec.h +++ b/libc/kernel/common/linux/msm_q6vdec.h @@ -26,6 +26,10 @@ #define VDEC_IOCTL_CLOSE _IO(VDEC_IOCTL_MAGIC, 8) #define VDEC_IOCTL_FREEBUFFERS _IOW(VDEC_IOCTL_MAGIC, 9, struct vdec_buf_info) #define VDEC_IOCTL_GETDECATTRIBUTES _IOR(VDEC_IOCTL_MAGIC, 10, struct vdec_dec_attributes) +#define VDEC_IOCTL_GETVERSION _IOR(VDEC_IOCTL_MAGIC, 11, struct vdec_version) +#define VDEC_IOCTL_SETPROPERTY _IOW(VDEC_IOCTL_MAGIC, 12, struct vdec_property_info) +#define VDEC_IOCTL_GETPROPERTY _IOR(VDEC_IOCTL_MAGIC, 13, struct vdec_property_info) +#define VDEC_IOCTL_PERFORMANCE_CHANGE_REQ _IOW(VDEC_IOCTL_MAGIC, 14, unsigned int) enum { VDEC_FRAME_DECODE_OK, @@ -99,6 +103,7 @@ struct vdec_config { u32 h264_nal_len_size; u32 postproc_flag; u32 fruc_enable; + u32 color_format; u32 reserved; }; @@ -208,5 +213,47 @@ struct vdec_dec_attributes { struct vdec_buf_desc dec_req2; }; -#endif +struct vdec_version { + u32 major; + u32 minor; +}; + +struct dal_vdec_rectangle { + u32 width; + u32 height; +}; + +struct stride_type { + u32 luma; + u32 chroma; +}; + +struct frame_alignment_type { + u32 luma_width; + u32 luma_height; + u32 chroma_width; + u32 chroma_height; + u32 chroma_offset; +}; +union vdec_property { + u32 fourcc; + u32 profile; + u32 level; + struct dal_vdec_rectangle dim; + struct vdec_cropping_window cw; + struct vdec_buf_desc input_req; + struct vdec_buf_desc output_req; + struct stride_type stride; + u32 num_dal_ports; + u32 priority; + struct frame_alignment_type frame_alignment; + u32 def_type; +}; + +struct vdec_property_info { + enum vdec_property_id id; + union vdec_property property; +}; + +#endif From 3d89fa4a0f1e00622bceb3b033476e943576b96b Mon Sep 17 00:00:00 2001 From: Shawe Date: Fri, 11 May 2012 08:12:21 +0200 Subject: [PATCH 3/5] Added missing enum Change-Id: Idb2df21e3b81b6bace2d1c2e66ef323cf3a66180 --- libc/kernel/common/linux/msm_q6vdec.h | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/libc/kernel/common/linux/msm_q6vdec.h b/libc/kernel/common/linux/msm_q6vdec.h index 6dbd0e7971..bbb76e62a3 100644 --- a/libc/kernel/common/linux/msm_q6vdec.h +++ b/libc/kernel/common/linux/msm_q6vdec.h @@ -62,6 +62,20 @@ enum { VDEC_QUEUE_BADSTATE, }; +enum vdec_property_id { + VDEC_FOURCC, + VDEC_PROFILE, + VDEC_LEVEL, + VDEC_DIMENSIONS, + VDEC_CWIN, + VDEC_INPUT_BUF_REQ, + VDEC_OUTPUT_BUF_REQ, + VDEC_LUMA_CHROMA_STRIDE, + VDEC_NUM_DAL_PORTS, + VDEC_PRIORITY, + VDEC_FRAME_ALIGNMENT +}; + struct vdec_input_buf_info { u32 offset; u32 data; From 419836b77dbac6b8cab87a1bf5ca0b7e0b69d67d Mon Sep 17 00:00:00 2001 From: cmartinezlozano Date: Tue, 28 Aug 2012 20:48:01 +0200 Subject: [PATCH 4/5] NEON Optimizations Change-Id: Ie4473b416f60b387594fb4a42a76b7b0ed8febfd --- libc/Android.mk | 65 +++-- libc/arch-arm/bionic/armv7/bzero.S | 102 +++++++ libc/arch-arm/bionic/armv7/memchr.S | 150 ++++++++++ libc/arch-arm/bionic/armv7/memcpy.S | 374 ++++++++++++++++++++++++ libc/arch-arm/bionic/armv7/memset.S | 119 ++++++++ libc/arch-arm/bionic/armv7/strchr.S | 77 +++++ libc/arch-arm/bionic/armv7/strcpy.c | 179 ++++++++++++ libc/arch-arm/bionic/memcpy.S | 159 +++++++++- libc/arch-arm/bionic/memmove.S | 174 ++++++++++- libc/arch-arm/bionic/memset.S | 63 +++- libc/bionic/md5.c | 2 +- libc/bionic/md5.h | 5 +- libc/bionic/sha1.c | 186 +++++++----- libc/bionic/stubs.c | 37 ++- libc/include/sha1.h | 13 +- libc/string/bcopy.c | 167 +++++++++++ libc/string/strcat.c | 6 + libc/string/strncat.c | 4 + libc/unistd/getopt_long.c | 12 +- libm/Android.mk | 13 +- libm/arm/e_pow.S | 432 ++++++++++++++++++++++++++++ libm/src/e_pow.c | 8 + linker/linker.c | 16 +- 23 files changed, 2250 insertions(+), 113 deletions(-) create mode 100644 libc/arch-arm/bionic/armv7/bzero.S create mode 100644 libc/arch-arm/bionic/armv7/memchr.S create mode 100644 libc/arch-arm/bionic/armv7/memcpy.S create mode 100644 libc/arch-arm/bionic/armv7/memset.S create mode 100644 libc/arch-arm/bionic/armv7/strchr.S create mode 100644 libc/arch-arm/bionic/armv7/strcpy.c create mode 100644 libm/arm/e_pow.S diff --git a/libc/Android.mk b/libc/Android.mk index 35176978d9..2f7404764b 100644 --- a/libc/Android.mk +++ b/libc/Android.mk @@ -178,14 +178,12 @@ libc_common_src_files := \ stdlib/wchar.c \ string/index.c \ string/memccpy.c \ - string/memchr.c \ string/memmem.c \ string/memrchr.c \ string/memswap.c \ string/strcasecmp.c \ string/strcasestr.c \ string/strcat.c \ - string/strchr.c \ string/strcoll.c \ string/strcspn.c \ string/strdup.c \ @@ -269,7 +267,6 @@ libc_common_src_files := \ bionic/libc_init_common.c \ bionic/logd_write.c \ bionic/md5.c \ - bionic/memmove_words.c \ bionic/pututline.c \ bionic/realpath.c \ bionic/sched_getaffinity.c \ @@ -355,30 +352,29 @@ libc_common_src_files += \ arch-arm/bionic/tkill.S \ arch-arm/bionic/memcmp.S \ arch-arm/bionic/memcmp16.S \ - arch-arm/bionic/memcpy.S \ - arch-arm/bionic/memset.S \ arch-arm/bionic/setjmp.S \ arch-arm/bionic/sigsetjmp.S \ - arch-arm/bionic/strcpy.S \ arch-arm/bionic/strcmp.S \ arch-arm/bionic/syscall.S \ string/strncmp.c \ unistd/socketcalls.c -ifeq ($(ARCH_ARM_HAVE_ARMV7A),true) -libc_common_src_files += arch-arm/bionic/strlen-armv7.S -else -libc_common_src_files += arch-arm/bionic/strlen.c.arm -endif # Check if we want a neonized version of memmove instead of the # current ARM version ifeq ($(TARGET_USE_SCORPION_BIONIC_OPTIMIZATION),true) libc_common_src_files += \ + arch-arm/bionic/memmove.S \ + bionic/memmove_words.c +else + ifneq (, $(filter true,$(TARGET_USE_KRAIT_BIONIC_OPTIMIZATION) $(TARGET_USE_SPARROW_BIONIC_OPTIMIZATION))) + libc_common_src_files += \ arch-arm/bionic/memmove.S -else # Non-Scorpion-based ARM -libc_common_src_files += \ + else # Other ARM + libc_common_src_files += \ string/bcopy.c \ - string/memmove.c.arm + string/memmove.c.arm \ + bionic/memmove_words.c + endif # !TARGET_USE_KRAIT_BIONIC_OPTIMIZATION endif # !TARGET_USE_SCORPION_BIONIC_OPTIMIZATION # These files need to be arm so that gdbserver @@ -400,8 +396,32 @@ libc_arch_static_src_files := \ libc_arch_dynamic_src_files := \ arch-arm/bionic/exidx_dynamic.c + +ifeq ($(ARCH_ARM_HAVE_ARMV7A),true) +libc_common_src_files += \ + arch-arm/bionic/armv7/memchr.S \ + arch-arm/bionic/armv7/memcpy.S \ + arch-arm/bionic/armv7/memset.S \ + arch-arm/bionic/armv7/bzero.S \ + arch-arm/bionic/armv7/strchr.S \ + arch-arm/bionic/armv7/strcpy.c \ + arch-arm/bionic/strlen-armv7.S +else +libc_common_src_files += \ + string/memchr.c \ + arch-arm/bionic/memcpy.S \ + arch-arm/bionic/memset.S \ + string/strchr.c \ + arch-arm/bionic/strcpy.S \ + arch-arm/bionic/strlen.c.arm +endif + else # !arm +libc_common_src_files += \ + string/memchr.c \ + string/strchr.c + ifeq ($(TARGET_ARCH),x86) libc_common_src_files += \ arch-x86/bionic/__get_sp.S \ @@ -537,6 +557,19 @@ ifeq ($(TARGET_ARCH),arm) libc_common_cflags += -DPLDSIZE=$(TARGET_SCORPION_BIONIC_PLDSIZE) endif endif + # Add in defines to activate KRAIT_NEON_OPTIMIZATION + ifeq ($(TARGET_USE_KRAIT_BIONIC_OPTIMIZATION),true) + libc_common_cflags += -DKRAIT_NEON_OPTIMIZATION + ifeq ($(TARGET_USE_KRAIT_PLD_SET),true) + libc_common_cflags += -DPLDOFFS=$(TARGET_KRAIT_BIONIC_PLDOFFS) + libc_common_cflags += -DPLDTHRESH=$(TARGET_KRAIT_BIONIC_PLDTHRESH) + libc_common_cflags += -DPLDSIZE=$(TARGET_KRAIT_BIONIC_PLDSIZE) + libc_common_cflags += -DBBTHRESH=$(TARGET_KRAIT_BIONIC_BBTHRESH) + endif + endif + ifeq ($(TARGET_USE_SPARROW_BIONIC_OPTIMIZATION),true) + libc_common_cflags += -DSPARROW_NEON_OPTIMIZATION + endif ifeq ($(TARGET_CORTEX_CACHE_LINE_32),true) libc_common_cflags += -DCORTEX_CACHE_LINE_32 endif @@ -563,10 +596,6 @@ ifeq ($(TARGET_ARCH),arm) libc_crt_target_cflags += -DCRT_LEGACY_WORKAROUND endif -ifeq ($(BOARD_USE_NASTY_PTHREAD_CREATE_HACK),true) - libc_common_cflags += -DNASTY_PTHREAD_CREATE_HACK -endif - # Define some common includes # ======================================================== libc_common_c_includes := \ diff --git a/libc/arch-arm/bionic/armv7/bzero.S b/libc/arch-arm/bionic/armv7/bzero.S new file mode 100644 index 0000000000..5230ba5380 --- /dev/null +++ b/libc/arch-arm/bionic/armv7/bzero.S @@ -0,0 +1,102 @@ +/* Copyright (c) 2010-2011, Linaro Limited + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + * Neither the name of Linaro Limited nor the names of its + contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + memset Written by Dave Gilbert + Adapted to bzero and Bionic by Bernhard Rosenkraenzer + + This memset routine is optimised on a Cortex-A9 and should work on + all ARMv7 processors. */ + +#include + + .syntax unified + .arch armv7-a + .text + .thumb + +@ --------------------------------------------------------------------------- + .thumb_func + .p2align 4,,15 +ENTRY(bzero) + @ r0 = address + @ r1 = count + @ Doesn't return anything + + cbz r1, 10f @ Exit if 0 length + mov r2, #0 + + tst r0, #7 + beq 2f @ Already aligned + + @ Ok, so we're misaligned here +1: + strb r2, [r0], #1 + subs r1,r1,#1 + tst r0, #7 + cbz r1, 10f @ Exit if we hit the end + bne 1b @ go round again if still misaligned + +2: + @ OK, so we're aligned + push {r4,r5,r6,r7} + bics r4, r1, #15 @ if less than 16 bytes then need to finish it off + beq 5f + +3: + mov r5,r2 + mov r6,r2 + mov r7,r2 + +4: + subs r4,r4,#16 + stmia r0!,{r2,r5,r6,r7} + bne 4b + and r1,r1,#15 + + @ At this point we're still aligned and we have upto align-1 bytes left to right + @ we can avoid some of the byte-at-a time now by testing for some big chunks + tst r1,#8 + itt ne + subne r1,r1,#8 + stmiane r0!,{r2,r5} + +5: + pop {r4,r5,r6,r7} + cbz r1, 10f + + @ Got to do any last < alignment bytes +6: + subs r1,r1,#1 + strb r2,[r0],#1 + bne 6b + +10: + bx lr @ goodbye +END(bzero) diff --git a/libc/arch-arm/bionic/armv7/memchr.S b/libc/arch-arm/bionic/armv7/memchr.S new file mode 100644 index 0000000000..02e59b98d2 --- /dev/null +++ b/libc/arch-arm/bionic/armv7/memchr.S @@ -0,0 +1,150 @@ +/* Copyright (c) 2010-2011, Linaro Limited + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + * Neither the name of Linaro Limited nor the names of its + contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + Written by Dave Gilbert + Adapted to Bionic by Bernhard Rosenkraenzer + + This memchr routine is optimised on a Cortex-A9 and should work on + all ARMv7 processors. It has a fast past for short sizes, and has + an optimised path for large data sets; the worst case is finding the + match early in a large data set. */ + +#include + +@ 2011-02-07 david.gilbert@linaro.org +@ Extracted from local git a5b438d861 +@ 2011-07-14 david.gilbert@linaro.org +@ Import endianness fix from local git ea786f1b +@ 2011-12-07 david.gilbert@linaro.org +@ Removed unneeded cbz from align loop + + .syntax unified + .arch armv7-a + +@ this lets us check a flag in a 00/ff byte easily in either endianness +#ifdef __ARMEB__ +#define CHARTSTMASK(c) 1<<(31-(c*8)) +#else +#define CHARTSTMASK(c) 1<<(c*8) +#endif + +@ --------------------------------------------------------------------------- + .thumb_func + .p2align 4,,15 +ENTRY(memchr) + @ r0 = start of memory to scan + @ r1 = character to look for + @ r2 = length + @ returns r0 = pointer to character or NULL if not found + and r1,r1,#0xff @ Don't think we can trust the caller to actually pass a char + + cmp r2,#16 @ If it's short don't bother with anything clever + blt 20f + + tst r0, #7 @ If it's already aligned skip the next bit + beq 10f + + @ Work up to an aligned point +5: + ldrb r3, [r0],#1 + subs r2, r2, #1 + cmp r3, r1 + beq 50f @ If it matches exit found + tst r0, #7 + bne 5b @ If not aligned yet then do next byte + +10: + @ At this point, we are aligned, we know we have at least 8 bytes to work with + push {r4,r5,r6,r7} + orr r1, r1, r1, lsl #8 @ expand the match word across to all bytes + orr r1, r1, r1, lsl #16 + bic r4, r2, #7 @ Number of double words to work with + mvns r7, #0 @ all F's + movs r3, #0 + +15: + ldmia r0!,{r5,r6} + subs r4, r4, #8 + eor r5,r5, r1 @ Get it so that r5,r6 have 00's where the bytes match the target + eor r6,r6, r1 + uadd8 r5, r5, r7 @ Parallel add 0xff - sets the GE bits for anything that wasn't 0 + sel r5, r3, r7 @ bytes are 00 for none-00 bytes, or ff for 00 bytes - NOTE INVERSION + uadd8 r6, r6, r7 @ Parallel add 0xff - sets the GE bits for anything that wasn't 0 + sel r6, r5, r7 @ chained....bytes are 00 for none-00 bytes, or ff for 00 bytes - NOTE INVERSION + cbnz r6, 60f + bne 15b @ (Flags from the subs above) If not run out of bytes then go around again + + pop {r4,r5,r6,r7} + and r1,r1,#0xff @ Get r1 back to a single character from the expansion above + and r2,r2,#7 @ Leave the count remaining as the number after the double words have been done + +20: + cbz r2, 40f @ 0 length or hit the end already then not found + +21: @ Post aligned section, or just a short call + ldrb r3,[r0],#1 + subs r2,r2,#1 + eor r3,r3,r1 @ r3 = 0 if match - doesn't break flags from sub + cbz r3, 50f + bne 21b @ on r2 flags + +40: + movs r0,#0 @ not found + bx lr + +50: + subs r0,r0,#1 @ found + bx lr + +60: @ We're here because the fast path found a hit - now we have to track down exactly which word it was + @ r0 points to the start of the double word after the one that was tested + @ r5 has the 00/ff pattern for the first word, r6 has the chained value + cmp r5, #0 + itte eq + moveq r5, r6 @ the end is in the 2nd word + subeq r0,r0,#3 @ Points to 2nd byte of 2nd word + subne r0,r0,#7 @ or 2nd byte of 1st word + + @ r0 currently points to the 3rd byte of the word containing the hit + tst r5, # CHARTSTMASK(0) @ 1st character + bne 61f + adds r0,r0,#1 + tst r5, # CHARTSTMASK(1) @ 2nd character + ittt eq + addeq r0,r0,#1 + tsteq r5, # (3<<15) @ 2nd & 3rd character + @ If not the 3rd must be the last one + addeq r0,r0,#1 + +61: + pop {r4,r5,r6,r7} + subs r0,r0,#1 + bx lr +END(memchr) diff --git a/libc/arch-arm/bionic/armv7/memcpy.S b/libc/arch-arm/bionic/armv7/memcpy.S new file mode 100644 index 0000000000..f906c78885 --- /dev/null +++ b/libc/arch-arm/bionic/armv7/memcpy.S @@ -0,0 +1,374 @@ +/* Copyright (c) 2010-2011, Linaro Limited + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + * Neither the name of Linaro Limited nor the names of its + contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + Written by Dave Gilbert + Adapted to Bionic by Bernhard Rosenkraenzer + + This memcpy routine is optimised on a Cortex-A9 and should work on + all ARMv7 processors. */ + +#include + +@ 2011-09-01 david.gilbert@linaro.org +@ Extracted from local git 2f11b436 + + .syntax unified + .arch armv7-a + +@ this lets us check a flag in a 00/ff byte easily in either endianness +#ifdef __ARMEB__ +#define CHARTSTMASK(c) 1<<(31-(c*8)) +#else +#define CHARTSTMASK(c) 1<<(c*8) +#endif + .thumb + +#if defined(__ARM_NEON__) +@ --------------------------------------------------------------------------- + .thumb_func + .p2align 4,,15 +ENTRY(memcpy) + @ r0 = dest + @ r1 = source + @ r2 = count + @ returns dest in r0 + @ Overlaps of source/dest not allowed according to spec + @ Note this routine relies on v7 misaligned loads/stores + pld [r1] + mov r12, r0 @ stash original r0 + cmp r2,#32 + blt 10f @ take the small copy case separately + + @ test for either source or destination being misaligned + @ (We only rely on word align) + tst r0,#3 + it eq + tsteq r1,#3 + bne 30f @ misaligned case + +4: + @ at this point we are word (or better) aligned and have at least + @ 32 bytes to play with + + @ If it's a huge copy, try Neon + cmp r2, #128*1024 + bge 35f @ Sharing general non-aligned case here, aligned could be faster + + push {r3,r4,r5,r6,r7,r8,r10,r11} +5: + ldmia r1!,{r3,r4,r5,r6,r7,r8,r10,r11} + sub r2,r2,#32 + pld [r1,#96] + cmp r2,#32 + stmia r0!,{r3,r4,r5,r6,r7,r8,r10,r11} + bge 5b + + pop {r3,r4,r5,r6,r7,r8,r10,r11} + @ We are now down to less than 32 bytes + cbz r2,15f @ quick exit for the case where we copied a multiple of 32 + +10: @ small copies (not necessarily aligned - note might be slightly more than 32bytes) + cmp r2,#4 + blt 12f +11: + sub r2,r2,#4 + cmp r2,#4 + ldr r3, [r1],#4 + str r3, [r0],#4 + bge 11b +12: + tst r2,#2 + itt ne + ldrhne r3, [r1],#2 + strhne r3, [r0],#2 + + tst r2,#1 + itt ne + ldrbne r3, [r1],#1 + strbne r3, [r0],#1 + +15: @ exit + mov r0,r12 @ restore r0 + bx lr + + .align 2 + .p2align 4,,15 +30: @ non-aligned - at least 32 bytes to play with + @ Test for co-misalignment + eor r3, r0, r1 + tst r3,#3 + beq 50f + + @ Use Neon for misaligned +35: + vld1.8 {d0,d1,d2,d3}, [r1]! + sub r2,r2,#32 + cmp r2,#32 + pld [r1,#96] + vst1.8 {d0,d1,d2,d3}, [r0]! + bge 35b + b 10b @ TODO: Probably a bad idea to switch to ARM at this point + + .align 2 + .p2align 4,,15 +50: @ Co-misaligned + @ At this point we've got at least 32 bytes +51: + ldrb r3,[r1],#1 + sub r2,r2,#1 + strb r3,[r0],#1 + tst r0,#7 + bne 51b + + cmp r2,#32 + blt 10b + b 4b +END(memcpy) +#else /* __ARM_NEON__ */ + + .thumb + +@ --------------------------------------------------------------------------- + .thumb_func + .p2align 4,,15 +ENTRY(memcpy) + @ r0 = dest + @ r1 = source + @ r2 = count + @ returns dest in r0 + @ Overlaps of source/dest not allowed according to spec + @ Note this routine relies on v7 misaligned loads/stores + pld [r1] + mov r12, r0 @ stash original r0 + cmp r2,#32 + blt 10f @ take the small copy case separately + + @ test for either source or destination being misaligned + @ (We only rely on word align) + @ TODO: Test for co-misalignment + tst r0,#3 + it eq + tsteq r1,#3 + bne 30f @ misaligned case + +4: + @ at this point we are word (or better) aligned and have at least + @ 32 bytes to play with + push {r3,r4,r5,r6,r7,r8,r10,r11} +5: + ldmia r1!,{r3,r4,r5,r6,r7,r8,r10,r11} + pld [r1,#96] + sub r2,r2,#32 + cmp r2,#32 + stmia r0!,{r3,r4,r5,r6,r7,r8,r10,r11} + bge 5b + + pop {r3,r4,r5,r6,r7,r8,r10,r11} + @ We are now down to less than 32 bytes + cbz r2,15f @ quick exit for the case where we copied a multiple of 32 + +10: @ small copies (not necessarily aligned - note might be slightly more than 32bytes) + cmp r2,#4 + blt 12f +11: + sub r2,r2,#4 + cmp r2,#4 + ldr r3, [r1],#4 + str r3, [r0],#4 + bge 11b +12: + tst r2,#2 + itt ne + ldrhne r3, [r1],#2 + strhne r3, [r0],#2 + + tst r2,#1 + itt ne + ldrbne r3, [r1],#1 + strbne r3, [r0],#1 + +15: @ exit + mov r0,r12 @ restore r0 + bx lr + +30: @ non-aligned - at least 32 bytes to play with + @ On v7 we're allowed to do ldr's and str's from arbitrary alignments + @ but not ldrd/strd or ldm/stm + @ Note Neon is often a better choice misaligned using vld1 + + @ copy a byte at a time until the point where we have an aligned destination + @ we know we have enough bytes to go to know we won't run out in this phase + tst r0,#7 + beq 35f + +31: + ldrb r3,[r1],#1 + sub r2,r2,#1 + strb r3,[r0],#1 + tst r0,#7 + bne 31b + + cmp r2,#32 @ Lets get back to knowing we have 32 bytes to play with + blt 11b + + @ Now the store address is aligned +35: + push {r3,r4,r5,r6,r7,r8,r10,r11,r12,r14} + and r6,r1,#3 @ how misaligned we are + cmp r6,#2 + cbz r6, 100f @ Go there if we're actually aligned + bge 120f @ And here if it's aligned on 2 or 3 byte + @ Note might be worth splitting to bgt and a separate beq + @ if the branches are well separated + + @ At this point dest is aligned, source is 1 byte forward +110: + ldr r3,[r1] @ Misaligned load - but it gives the first 4 bytes to store + sub r2,r2,#3 @ Number of bytes left in whole words we can load + add r1,r1,#3 @ To aligned load address + bic r3,r3,#0xff000000 + +112: + ldmia r1!,{r5,r6,r7,r8} + sub r2,r2,#32 + cmp r2,#32 + pld [r1,#96] + + orr r3,r3,r5,lsl#24 + mov r4,r5,lsr#8 + mov r5,r6,lsr#8 + orr r4,r4,r6,lsl#24 + mov r6,r7,lsr#8 + ldmia r1!,{r10,r11,r12,r14} + orr r5,r5,r7,lsl#24 + mov r7,r8,lsr#8 + orr r6,r6,r8,lsl#24 + mov r8,r10,lsr#8 + orr r7,r7,r10,lsl#24 + mov r10,r11,lsr#8 + orr r8,r8,r11,lsl#24 + orr r10,r10,r12,lsl#24 + mov r11,r12,lsr#8 + orr r11,r11,r14,lsl#24 + stmia r0!,{r3,r4,r5,r6,r7,r8,r10,r11} + mov r3,r14,lsr#8 + + bge 112b + + @ Deal with the stragglers + add r2,r2,#3 + sub r1,r1,#3 + pop {r3,r4,r5,r6,r7,r8,r10,r11,r12,r14} + b 10b + +100: @ Dest and source aligned - must have been originally co-misaligned + @ Fallback to main aligned case if still big enough + pop {r3,r4,r5,r6,r7,r8,r10,r11,r12,r14} + b 4b @ Big copies (32 bytes or more) + +120: @ Dest is aligned, source is align+2 or 3 + bgt 130f @ Now split off for 3 byte offset + + ldrh r3,[r1] + sub r2,r2,#2 @ Number of bytes left in whole words we can load + add r1,r1,#2 @ To aligned load address + +122: + ldmia r1!,{r5,r6,r7,r8} + sub r2,r2,#32 + cmp r2,#32 + pld [r1,#96] + + orr r3,r3,r5,lsl#16 + mov r4,r5,lsr#16 + mov r5,r6,lsr#16 + orr r4,r4,r6,lsl#16 + mov r6,r7,lsr#16 + ldmia r1!,{r10,r11,r12,r14} + orr r5,r5,r7,lsl#16 + orr r6,r6,r8,lsl#16 + mov r7,r8,lsr#16 + orr r7,r7,r10,lsl#16 + mov r8,r10,lsr#16 + orr r8,r8,r11,lsl#16 + mov r10,r11,lsr#16 + orr r10,r10,r12,lsl#16 + mov r11,r12,lsr#16 + orr r11,r11,r14,lsl#16 + stmia r0!,{r3,r4,r5,r6,r7,r8,r10,r11} + mov r3,r14,lsr#16 + + bge 122b + + @ Deal with the stragglers + add r2,r2,#2 + sub r1,r1,#2 + pop {r3,r4,r5,r6,r7,r8,r10,r11,r12,r14} + b 10b + +130: @ Dest is aligned, source is align+3 + ldrb r3,[r1] + sub r2,r2,#1 @ Number of bytes left in whole words we can load + add r1,r1,#1 @ To aligned load address + +132: + ldmia r1!,{r5,r6,r7,r8} + sub r2,r2,#32 + cmp r2,#32 + pld [r1,#96] + + orr r3,r3,r5,lsl#8 + mov r4,r5,lsr#24 + mov r5,r6,lsr#24 + orr r4,r4,r6,lsl#8 + mov r6,r7,lsr#24 + ldmia r1!,{r10,r11,r12,r14} + orr r5,r5,r7,lsl#8 + mov r7,r8,lsr#24 + orr r6,r6,r8,lsl#8 + mov r8,r10,lsr#24 + orr r7,r7,r10,lsl#8 + orr r8,r8,r11,lsl#8 + mov r10,r11,lsr#24 + orr r10,r10,r12,lsl#8 + mov r11,r12,lsr#24 + orr r11,r11,r14,lsl#8 + stmia r0!,{r3,r4,r5,r6,r7,r8,r10,r11} + mov r3,r14,lsr#24 + + bge 132b + + @ Deal with the stragglers + add r2,r2,#1 + sub r1,r1,#1 + pop {r3,r4,r5,r6,r7,r8,r10,r11,r12,r14} + b 10b +END(memcpy) +#endif diff --git a/libc/arch-arm/bionic/armv7/memset.S b/libc/arch-arm/bionic/armv7/memset.S new file mode 100644 index 0000000000..d1bcbe8dbc --- /dev/null +++ b/libc/arch-arm/bionic/armv7/memset.S @@ -0,0 +1,119 @@ +/* Copyright (c) 2010-2011, Linaro Limited + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + * Neither the name of Linaro Limited nor the names of its + contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + Written by Dave Gilbert + Adapted to Bionic by Bernhard Rosenkraenzer + + This memset routine is optimised on a Cortex-A9 and should work on + all ARMv7 processors. */ + +#include + + .syntax unified + .arch armv7-a + +@ 2011-08-30 david.gilbert@linaro.org +@ Extracted from local git 2f11b436 + +@ this lets us check a flag in a 00/ff byte easily in either endianness +#ifdef __ARMEB__ +#define CHARTSTMASK(c) 1<<(31-(c*8)) +#else +#define CHARTSTMASK(c) 1<<(c*8) +#endif + .text + .thumb + +@ --------------------------------------------------------------------------- + .thumb_func + .p2align 4,,15 +ENTRY(memset) + @ r0 = address + @ r1 = character + @ r2 = count + @ returns original address in r0 + + mov r3, r0 @ Leave r0 alone + cbz r2, 10f @ Exit if 0 length + + tst r0, #7 + beq 2f @ Already aligned + + @ Ok, so we're misaligned here +1: + strb r1, [r3], #1 + subs r2,r2,#1 + tst r3, #7 + cbz r2, 10f @ Exit if we hit the end + bne 1b @ go round again if still misaligned + +2: + @ OK, so we're aligned + push {r4,r5,r6,r7} + bics r4, r2, #15 @ if less than 16 bytes then need to finish it off + beq 5f + +3: + @ POSIX says that ch is cast to an unsigned char. A uxtb is one + @ byte and takes two cycles, where an AND is four bytes but one + @ cycle. + and r1, #0xFF + orr r1, r1, r1, lsl#8 @ Same character into all bytes + orr r1, r1, r1, lsl#16 + mov r5,r1 + mov r6,r1 + mov r7,r1 + +4: + subs r4,r4,#16 + stmia r3!,{r1,r5,r6,r7} + bne 4b + and r2,r2,#15 + + @ At this point we're still aligned and we have upto align-1 bytes left to right + @ we can avoid some of the byte-at-a time now by testing for some big chunks + tst r2,#8 + itt ne + subne r2,r2,#8 + stmiane r3!,{r1,r5} + +5: + pop {r4,r5,r6,r7} + cbz r2, 10f + + @ Got to do any last < alignment bytes +6: + subs r2,r2,#1 + strb r1,[r3],#1 + bne 6b + +10: + bx lr @ goodbye +END(memset) diff --git a/libc/arch-arm/bionic/armv7/strchr.S b/libc/arch-arm/bionic/armv7/strchr.S new file mode 100644 index 0000000000..370aac9e85 --- /dev/null +++ b/libc/arch-arm/bionic/armv7/strchr.S @@ -0,0 +1,77 @@ +/* Copyright (c) 2010-2011, Linaro Limited + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + * Neither the name of Linaro Limited nor the names of its + contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + Written by Dave Gilbert + Adapted to Bionic by Bernhard Rosenkraenzer + + A very simple strchr routine, from benchmarks on A9 it's a bit faster than + the current version in eglibc (2.12.1-0ubuntu14 package) + I don't think doing a word at a time version is worth it since a lot + of strchr cases are very short anyway */ + +#include + +@ 2011-02-07 david.gilbert@linaro.org +@ Extracted from local git a5b438d861 + + .syntax unified + .arch armv7-a + + .text + .thumb + +@ --------------------------------------------------------------------------- + + .thumb_func + .p2align 4,,15 +ENTRY(strchr) + @ r0 = start of string + @ r1 = character to match + @ returns NULL for no match, or a pointer to the match + and r1,r1, #255 + +1: + ldrb r2,[r0],#1 + cmp r2,r1 + cbz r2,10f + bne 1b + + @ We're here if it matched +5: + subs r0,r0,#1 + bx lr + +10: + @ We're here if we ran off the end + cmp r1, #0 @ Corner case - you're allowed to search for the nil and get a pointer to it + beq 5b @ A bit messy, if it's common we should branch at the start to a special loop + mov r0,#0 + bx lr +END(strchr) diff --git a/libc/arch-arm/bionic/armv7/strcpy.c b/libc/arch-arm/bionic/armv7/strcpy.c new file mode 100644 index 0000000000..ce74de663a --- /dev/null +++ b/libc/arch-arm/bionic/armv7/strcpy.c @@ -0,0 +1,179 @@ +/* + * Copyright (c) 2008 ARM Ltd + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions + * are met: + * 1. Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * 2. Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * 3. The name of the company may not be used to endorse or promote + * products derived from this software without specific prior written + * permission. + * + * THIS SOFTWARE IS PROVIDED BY ARM LTD ``AS IS'' AND ANY EXPRESS OR IMPLIED + * WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF + * MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. + * IN NO EVENT SHALL ARM LTD BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + * SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED + * TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + * PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + * LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + * NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ + +#include + +#ifdef __thumb2__ +#define magic1(REG) "#0x01010101" +#define magic2(REG) "#0x80808080" +#else +#define magic1(REG) #REG +#define magic2(REG) #REG ", lsl #7" +#endif + +#pragma GCC diagnostic push +/* gcc fails to see the fat that the assembly code + * takes care of a return value, causing a + * "control reaches end of non-void function" + * warning (and, of course, error when building + * with -Werror). + * Let's disable that warning just for this + * function, where we know it's bogus. */ +#pragma GCC diagnostic ignored "-Wreturn-type" + +char* __attribute__((naked)) +strcpy (char* dst, const char* src) +{ + asm ( +#if !(defined(__OPTIMIZE_SIZE__) || defined (PREFER_SIZE_OVER_SPEED) || \ + (defined (__thumb__) && !defined (__thumb2__))) + "pld [r1, #0]\n\t" + "eor r2, r0, r1\n\t" + "mov ip, r0\n\t" + "tst r2, #3\n\t" + "bne 4f\n\t" + "tst r1, #3\n\t" + "bne 3f\n" + "5:\n\t" +#ifndef __thumb2__ + "str r5, [sp, #-4]!\n\t" + "mov r5, #0x01\n\t" + "orr r5, r5, r5, lsl #8\n\t" + "orr r5, r5, r5, lsl #16\n\t" +#endif + + "str r4, [sp, #-4]!\n\t" + "tst r1, #4\n\t" + "ldr r3, [r1], #4\n\t" + "beq 2f\n\t" + "sub r2, r3, "magic1(r5)"\n\t" + "bics r2, r2, r3\n\t" + "tst r2, "magic2(r5)"\n\t" + "itt eq\n\t" + "streq r3, [ip], #4\n\t" + "ldreq r3, [r1], #4\n" + "bne 1f\n\t" + /* Inner loop. We now know that r1 is 64-bit aligned, so we + can safely fetch up to two words. This allows us to avoid + load stalls. */ + ".p2align 2\n" + "2:\n\t" + "pld [r1, #8]\n\t" + "ldr r4, [r1], #4\n\t" + "sub r2, r3, "magic1(r5)"\n\t" + "bics r2, r2, r3\n\t" + "tst r2, "magic2(r5)"\n\t" + "sub r2, r4, "magic1(r5)"\n\t" + "bne 1f\n\t" + "str r3, [ip], #4\n\t" + "bics r2, r2, r4\n\t" + "tst r2, "magic2(r5)"\n\t" + "itt eq\n\t" + "ldreq r3, [r1], #4\n\t" + "streq r4, [ip], #4\n\t" + "beq 2b\n\t" + "mov r3, r4\n" + "1:\n\t" +#ifdef __ARMEB__ + "rors r3, r3, #24\n\t" +#endif + "strb r3, [ip], #1\n\t" + "tst r3, #0xff\n\t" +#ifdef __ARMEL__ + "ror r3, r3, #8\n\t" +#endif + "bne 1b\n\t" + "ldr r4, [sp], #4\n\t" +#ifndef __thumb2__ + "ldr r5, [sp], #4\n\t" +#endif + "BX LR\n" + + /* Strings have the same offset from word alignment, but it's + not zero. */ + "3:\n\t" + "tst r1, #1\n\t" + "beq 1f\n\t" + "ldrb r2, [r1], #1\n\t" + "strb r2, [ip], #1\n\t" + "cmp r2, #0\n\t" + "it eq\n" + "BXEQ LR\n" + "1:\n\t" + "tst r1, #2\n\t" + "beq 5b\n\t" + "ldrh r2, [r1], #2\n\t" +#ifdef __ARMEB__ + "tst r2, #0xff00\n\t" + "iteet ne\n\t" + "strneh r2, [ip], #2\n\t" + "lsreq r2, r2, #8\n\t" + "streqb r2, [ip]\n\t" + "tstne r2, #0xff\n\t" +#else + "tst r2, #0xff\n\t" + "itet ne\n\t" + "strneh r2, [ip], #2\n\t" + "streqb r2, [ip]\n\t" + "tstne r2, #0xff00\n\t" +#endif + "bne 5b\n\t" + "BX LR\n" + + /* src and dst do not have a common word-alignement. Fall back to + byte copying. */ + "4:\n\t" + "ldrb r2, [r1], #1\n\t" + "strb r2, [ip], #1\n\t" + "cmp r2, #0\n\t" + "bne 4b\n\t" + "BX LR" + +#elif !defined (__thumb__) || defined (__thumb2__) + "mov r3, r0\n\t" + "1:\n\t" + "ldrb r2, [r1], #1\n\t" + "strb r2, [r3], #1\n\t" + "cmp r2, #0\n\t" + "bne 1b\n\t" + "BX LR" +#else + "mov r3, r0\n\t" + "1:\n\t" + "ldrb r2, [r1]\n\t" + "add r1, r1, #1\n\t" + "strb r2, [r3]\n\t" + "add r3, r3, #1\n\t" + "cmp r2, #0\n\t" + "bne 1b\n\t" + "BX LR" +#endif + ); +} + +#pragma GCC diagnostic pop diff --git a/libc/arch-arm/bionic/memcpy.S b/libc/arch-arm/bionic/memcpy.S index d13c605a05..69561b579e 100644 --- a/libc/arch-arm/bionic/memcpy.S +++ b/libc/arch-arm/bionic/memcpy.S @@ -31,8 +31,163 @@ #include #include -#if defined(__ARM_NEON__) -#if defined(SCORPION_NEON_OPTIMIZATION) +#if defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY) +#if defined(KRAIT_NEON_OPTIMIZATION) + /* + * These can be overridden in: + * device///BoardConfig.mk + * by setting the following: + * TARGET_USE_KRAIT_BIONIC_OPTIMIZATION := true + * TARGET_USE_KRAIT_PLD_SET := true + * TARGET_KRAIT_BIONIC_PLDOFFS := + * TARGET_KRAIT_BIONIC_PLDSIZE := + * TARGET_KRAIT_BIONIC_PLDTHRESH := + * TARGET_KRAIT_BIONIC_BBTHRESH := + */ +#ifndef PLDOFFS +#define PLDOFFS (10) +#endif +#ifndef PLDTHRESH +#define PLDTHRESH (PLDOFFS) +#endif +#ifndef BBTHRESH +#define BBTHRESH (4096/64) +#endif +#if (PLDOFFS < 1) +#error Routine does not support offsets less than 1 +#endif +#if (PLDTHRESH < PLDOFFS) +#error PLD threshold must be greater than or equal to the PLD offset +#endif +#ifndef PLDSIZE +#define PLDSIZE (64) +#endif +#define NOP_OPCODE (0xe320f000) + + .text + .fpu neon + .global memcpy + .type memcpy, %function + .align 5 +memcpy: + stmfd sp!, {r0, r9, r10, lr} + cmp r2, #4 + blt .Lneon_lt4 + cmp r2, #16 + blt .Lneon_lt16 + cmp r2, #32 + blt .Lneon_16 + cmp r2, #64 + blt .Lneon_copy_32_a + + mov r12, r2, lsr #6 + cmp r12, #PLDTHRESH + ble .Lneon_copy_64_loop_nopld + + cmp r12, #BBTHRESH + ble .Lneon_prime_pump + + add lr, r0, #0x400 + add r9, r1, #(PLDOFFS*PLDSIZE) + sub lr, lr, r9 + lsl lr, lr, #21 + lsr lr, lr, #21 + add lr, lr, #(PLDOFFS*PLDSIZE) + cmp r12, lr, lsr #6 + movle lr, #(PLDOFFS*PLDSIZE) + + movgt r9, #(PLDOFFS) + rsbgts r9, r9, lr, lsr #6 + ble .Lneon_prime_pump + + add r10, r1, lr + bic r10, #0x3F + + sub r12, lr, lsr #6 + cmp r9, r12 + suble r12, r12, r9 + movgt r9, r12 + movgt r12, #0 + + pld [r1, #((PLDOFFS-1)*PLDSIZE)] + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_copy_64_loop_outer_doublepld: + pld [r1, #((PLDOFFS)*PLDSIZE)] + vld1.32 {q0, q1}, [r1]! + vld1.32 {q2, q3}, [r1]! + ldr r3, [r10] + subs r9, r9, #1 + vst1.32 {q0, q1}, [r0]! + vst1.32 {q2, q3}, [r0]! + add r10, #64 + bne .Lneon_copy_64_loop_outer_doublepld + cmp r12, #0 + bne .Lneon_copy_64_loop_outer + mov r12, lr, lsr #6 + b .Lneon_copy_64_loop_nopld + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_prime_pump: + mov lr, #(PLDOFFS*PLDSIZE) + add r10, r1, #(PLDOFFS*PLDSIZE) + bic r10, #0x3F + sub r12, r12, #PLDOFFS + pld [r10, #(-1*PLDSIZE)] + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_copy_64_loop_outer: + vld1.32 {q0, q1}, [r1]! + vld1.32 {q2, q3}, [r1]! + ldr r3, [r10] + subs r12, r12, #1 + vst1.32 {q0, q1}, [r0]! + vst1.32 {q2, q3}, [r0]! + add r10, #64 + bne .Lneon_copy_64_loop_outer + mov r12, lr, lsr #6 + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_copy_64_loop_nopld: + vld1.32 {q8, q9}, [r1]! + vld1.32 {q10, q11}, [r1]! + subs r12, r12, #1 + vst1.32 {q8, q9}, [r0]! + vst1.32 {q10, q11}, [r0]! + bne .Lneon_copy_64_loop_nopld + ands r2, r2, #0x3f + beq .Lneon_exit + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_copy_32_a: + movs r12, r2, lsl #27 + bcc .Lneon_16 + vld1.32 {q0,q1}, [r1]! + vst1.32 {q0,q1}, [r0]! + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_16: + bpl .Lneon_lt16 + vld1.32 {q8}, [r1]! + vst1.32 {q8}, [r0]! + ands r2, r2, #0x0f + beq .Lneon_exit + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_lt16: + movs r12, r2, lsl #29 + ldrcs r3, [r1], #4 + ldrcs r12, [r1], #4 + strcs r3, [r0], #4 + strcs r12, [r0], #4 + ldrmi r3, [r1], #4 + strmi r3, [r0], #4 + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_lt4: + movs r2, r2, lsl #31 + ldrcsh r3, [r1], #2 + strcsh r3, [r0], #2 + ldrmib r12, [r1] + strmib r12, [r0] + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_exit: + ldmfd sp!, {r0, r9, r10, lr} + bx lr + .end +#elif defined(SCORPION_NEON_OPTIMIZATION) /* * These can be overridden in: * device///BoardConfig.mk diff --git a/libc/arch-arm/bionic/memmove.S b/libc/arch-arm/bionic/memmove.S index 123419584f..937d14bfe7 100644 --- a/libc/arch-arm/bionic/memmove.S +++ b/libc/arch-arm/bionic/memmove.S @@ -1,5 +1,5 @@ /*************************************************************************** - Copyright (c) 2009-2011 Code Aurora Forum. All rights reserved. + Copyright (c) 2009-2012 Code Aurora Forum. All rights reserved. Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met: @@ -37,7 +37,177 @@ #include -#if defined(SCORPION_NEON_OPTIMIZATION) +#if defined(KRAIT_NEON_OPTIMIZATION) || defined(SPARROW_NEON_OPTIMIZATION) + /* + * These can be overridden in: + * device///BoardConfig.mk + * by setting the following: + * TARGET_USE_KRAIT_BIONIC_OPTIMIZATION := true + * TARGET_USE_KRAIT_PLD_SET := true + * TARGET_KRAIT_BIONIC_PLDOFFS := + * TARGET_KRAIT_BIONIC_PLDSIZE := + * TARGET_KRAIT_BIONIC_PLDTHRESH := + */ +#ifndef PLDOFFS +#define PLDOFFS (10) +#endif +#ifndef PLDTHRESH +#define PLDTHRESH (PLDOFFS) +#endif +#if (PLDOFFS < 5) +#error Routine does not support offsets less than 5 +#endif +#if (PLDTHRESH < PLDOFFS) +#error PLD threshold must be greater than or equal to the PLD offset +#endif +#ifndef PLDSIZE +#define PLDSIZE (64) +#endif +#define NOP_OPCODE (0xe320f000) + + .code 32 + .align 5 + .global memmove + .type memmove, %function + + .global _memmove_words + .type _memmove_words, %function + + .global bcopy + .type bcopy, %function + +bcopy: + mov r12, r0 + mov r0, r1 + mov r1, r12 + .balignl 64, NOP_OPCODE, 4*2 +memmove: +_memmove_words: +.Lneon_memmove_cmf: + subs r12, r0, r1 + bxeq lr + cmphi r2, r12 + bls memcpy /* Use memcpy for non-overlapping areas */ + + push {r0} + +.Lneon_back_to_front_copy: + add r0, r0, r2 + add r1, r1, r2 + cmp r2, #4 + bgt .Lneon_b2f_gt4 + cmp r2, #0 +.Lneon_b2f_smallcopy_loop: + beq .Lneon_memmove_done + ldrb r12, [r1, #-1]! + subs r2, r2, #1 + strb r12, [r0, #-1]! + b .Lneon_b2f_smallcopy_loop +.Lneon_b2f_gt4: + sub r3, r0, r1 + cmp r2, r3 + movle r12, r2 + movgt r12, r3 + cmp r12, #64 + bge .Lneon_b2f_copy_64 + cmp r12, #32 + bge .Lneon_b2f_copy_32 + cmp r12, #8 + bge .Lneon_b2f_copy_8 + cmp r12, #4 + bge .Lneon_b2f_copy_4 + b .Lneon_b2f_copy_1 +.Lneon_b2f_copy_64: + sub r1, r1, #64 /* Predecrement */ + sub r0, r0, #64 + movs r12, r2, lsr #6 + cmp r12, #PLDTHRESH + ble .Lneon_b2f_copy_64_loop_nopld + sub r12, #PLDOFFS + pld [r1, #-(PLDOFFS-5)*PLDSIZE] + pld [r1, #-(PLDOFFS-4)*PLDSIZE] + pld [r1, #-(PLDOFFS-3)*PLDSIZE] + pld [r1, #-(PLDOFFS-2)*PLDSIZE] + pld [r1, #-(PLDOFFS-1)*PLDSIZE] + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_b2f_copy_64_loop_outer: + pld [r1, #-(PLDOFFS)*PLDSIZE] + vld1.32 {q0, q1}, [r1]! + vld1.32 {q2, q3}, [r1] + subs r12, r12, #1 + vst1.32 {q0, q1}, [r0]! + sub r1, r1, #96 /* Post-fixup and predecrement */ + vst1.32 {q2, q3}, [r0] + sub r0, r0, #96 + bne .Lneon_b2f_copy_64_loop_outer + mov r12, #PLDOFFS + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_b2f_copy_64_loop_nopld: + vld1.32 {q8, q9}, [r1]! + vld1.32 {q10, q11}, [r1] + subs r12, r12, #1 + vst1.32 {q8, q9}, [r0]! + sub r1, r1, #96 /* Post-fixup and predecrement */ + vst1.32 {q10, q11}, [r0] + sub r0, r0, #96 + bne .Lneon_b2f_copy_64_loop_nopld + ands r2, r2, #0x3f + beq .Lneon_memmove_done + add r1, r1, #64 /* Post-fixup */ + add r0, r0, #64 + cmp r2, #32 + blt .Lneon_b2f_copy_finish +.Lneon_b2f_copy_32: + mov r12, r2, lsr #5 +.Lneon_b2f_copy_32_loop: + sub r1, r1, #32 /* Predecrement */ + sub r0, r0, #32 + vld1.32 {q0,q1}, [r1] + subs r12, r12, #1 + vst1.32 {q0,q1}, [r0] + bne .Lneon_b2f_copy_32_loop + ands r2, r2, #0x1f + beq .Lneon_memmove_done +.Lneon_b2f_copy_finish: +.Lneon_b2f_copy_8: + movs r12, r2, lsr #0x3 + beq .Lneon_b2f_copy_4 + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_b2f_copy_8_loop: + sub r1, r1, #8 /* Predecrement */ + sub r0, r0, #8 + vld1.32 {d0}, [r1] + subs r12, r12, #1 + vst1.32 {d0}, [r0] + bne .Lneon_b2f_copy_8_loop + ands r2, r2, #0x7 + beq .Lneon_memmove_done +.Lneon_b2f_copy_4: + movs r12, r2, lsr #0x2 + beq .Lneon_b2f_copy_1 +.Lneon_b2f_copy_4_loop: + ldr r3, [r1, #-4]! + subs r12, r12, #1 + str r3, [r0, #-4]! + bne .Lneon_b2f_copy_4_loop + ands r2, r2, #0x3 +.Lneon_b2f_copy_1: + cmp r2, #0 + beq .Lneon_memmove_done + .balignl 64, NOP_OPCODE, 4*2 +.Lneon_b2f_copy_1_loop: + ldrb r12, [r1, #-1]! + subs r2, r2, #1 + strb r12, [r0, #-1]! + bne .Lneon_b2f_copy_1_loop + +.Lneon_memmove_done: + pop {r0} + bx lr + + .end + +#elif defined(SCORPION_NEON_OPTIMIZATION) /* * These can be overridden in: * device///BoardConfig.mk diff --git a/libc/arch-arm/bionic/memset.S b/libc/arch-arm/bionic/memset.S index c386e7e6d4..bf0b30f87b 100644 --- a/libc/arch-arm/bionic/memset.S +++ b/libc/arch-arm/bionic/memset.S @@ -113,7 +113,66 @@ memset: .end #else /* !(SCORPION_NEON_OPTIMIZATION || CORTEX_CACHE_LINE_32) */ - +#if defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY) +@ this lets us check a flag in a 00/ff byte easily in either endianness +#ifdef __ARMEB__ +#define CHARTSTMASK(c) 1<<(31-(c*8)) +#else +#define CHARTSTMASK(c) 1<<(c*8) +#endif + .text + .thumb + +@ --------------------------------------------------------------------------- + .code 32 + .p2align 4,,15 +ENTRY(memset) + and r1, r1, #0xff + cmp r2, #0 + bxeq lr + orr r1, r1, r1, lsl #8 + tst r0, #7 + mov r3, r0 + orr r1, r1, r1, lsl #16 + beq .Lmemset_align8 +.Lmemset_make_align: + strb r1, [r3], #1 + subs r2, r2, #1 + bxeq lr + tst r3, #7 + bne .Lmemset_make_align + +.Lmemset_align8: + cmp r2, #16 + mov r12, r1 + blt .Lmemset_less16 + push {r4, lr} + mov r4, r1 + mov lr, r1 +.Lmemset_loop32: + subs r2, r2, #32 + stmhsia r3!, {r1, r4, r12, lr} + stmhsia r3!, {r1, r4, r12, lr} + bhs .Lmemset_loop32 + adds r2, r2, #32 + popeq {r4, pc} + tst r2, #16 + stmneia r3!, {r1, r4, r12, lr} + pop {r4, lr} + subs r2, #16 + bxeq lr +.Lmemset_less16: + movs r2, r2, lsl #29 + stmcsia r3!, {r1, r12} + strmi r1, [r3], #4 + movs r2, r2, lsl #2 + strhcs r1, [r3], #2 + strbmi r1, [r3], #1 + bx lr + +END(memset) + +#else /* * Optimized memset() for ARM. * @@ -193,5 +252,5 @@ ENTRY(memset) ldmfd sp!, {r0, r4-r7, lr} bx lr END(memset) - +#endif #endif /* SCORPION_NEON_OPTIMIZATION */ diff --git a/libc/bionic/md5.c b/libc/bionic/md5.c index ba4aaed220..02785bdaa3 100644 --- a/libc/bionic/md5.c +++ b/libc/bionic/md5.c @@ -231,7 +231,7 @@ MD5_Update (struct md5 *m, const void *v, size_t len) } calc(m, current); #else - calc(m, (u_int32_t*)m->save); + calc(m, m->save32); #endif offset = 0; } diff --git a/libc/bionic/md5.h b/libc/bionic/md5.h index a381994efa..3b0a2511bd 100644 --- a/libc/bionic/md5.h +++ b/libc/bionic/md5.h @@ -40,7 +40,10 @@ struct md5 { unsigned int sz[2]; u_int32_t counter[4]; - unsigned char save[64]; + union { + unsigned char save[64]; + u_int32_t save32[16]; + }; }; typedef struct md5 MD5_CTX; diff --git a/libc/bionic/sha1.c b/libc/bionic/sha1.c index a4fbd673bb..efa95a55c7 100644 --- a/libc/bionic/sha1.c +++ b/libc/bionic/sha1.c @@ -22,7 +22,10 @@ #include #include #include -#include + +#if HAVE_NBTOOL_CONFIG_H +#include "nbtool_config.h" +#endif #if !HAVE_SHA1_H @@ -33,7 +36,8 @@ * I got the idea of expanding during the round function from SSLeay */ #if BYTE_ORDER == LITTLE_ENDIAN -# define blk0(i) swap32(block->l[i]) +# define blk0(i) (block->l[i] = (rol(block->l[i],24)&0xFF00FF00) \ + |(rol(block->l[i],8)&0x00FF00FF)) #else # define blk0(i) block->l[i] #endif @@ -50,17 +54,77 @@ #define R4(v,w,x,y,z,i) z+=(w^x^y)+blk(i)+0xCA62C1D6+rol(v,5);w=rol(w,30); typedef union { - uint8_t c[SHA1_BLOCK_SIZE]; - uint32_t l[SHA1_BLOCK_SIZE/4]; + u_char c[64]; + u_int l[16]; } CHAR64LONG16; +/* old sparc64 gcc could not compile this */ +#undef SPARC64_GCC_WORKAROUND +#if defined(__sparc64__) && defined(__GNUC__) && __GNUC__ < 3 +#define SPARC64_GCC_WORKAROUND +#endif + +#ifdef SPARC64_GCC_WORKAROUND +void do_R01(u_int32_t *a, u_int32_t *b, u_int32_t *c, u_int32_t *d, u_int32_t *e, CHAR64LONG16 *); +void do_R2(u_int32_t *a, u_int32_t *b, u_int32_t *c, u_int32_t *d, u_int32_t *e, CHAR64LONG16 *); +void do_R3(u_int32_t *a, u_int32_t *b, u_int32_t *c, u_int32_t *d, u_int32_t *e, CHAR64LONG16 *); +void do_R4(u_int32_t *a, u_int32_t *b, u_int32_t *c, u_int32_t *d, u_int32_t *e, CHAR64LONG16 *); + +#define nR0(v,w,x,y,z,i) R0(*v,*w,*x,*y,*z,i) +#define nR1(v,w,x,y,z,i) R1(*v,*w,*x,*y,*z,i) +#define nR2(v,w,x,y,z,i) R2(*v,*w,*x,*y,*z,i) +#define nR3(v,w,x,y,z,i) R3(*v,*w,*x,*y,*z,i) +#define nR4(v,w,x,y,z,i) R4(*v,*w,*x,*y,*z,i) + +void +do_R01(u_int32_t *a, u_int32_t *b, u_int32_t *c, u_int32_t *d, u_int32_t *e, CHAR64LONG16 *block) +{ + nR0(a,b,c,d,e, 0); nR0(e,a,b,c,d, 1); nR0(d,e,a,b,c, 2); nR0(c,d,e,a,b, 3); + nR0(b,c,d,e,a, 4); nR0(a,b,c,d,e, 5); nR0(e,a,b,c,d, 6); nR0(d,e,a,b,c, 7); + nR0(c,d,e,a,b, 8); nR0(b,c,d,e,a, 9); nR0(a,b,c,d,e,10); nR0(e,a,b,c,d,11); + nR0(d,e,a,b,c,12); nR0(c,d,e,a,b,13); nR0(b,c,d,e,a,14); nR0(a,b,c,d,e,15); + nR1(e,a,b,c,d,16); nR1(d,e,a,b,c,17); nR1(c,d,e,a,b,18); nR1(b,c,d,e,a,19); +} + +void +do_R2(u_int32_t *a, u_int32_t *b, u_int32_t *c, u_int32_t *d, u_int32_t *e, CHAR64LONG16 *block) +{ + nR2(a,b,c,d,e,20); nR2(e,a,b,c,d,21); nR2(d,e,a,b,c,22); nR2(c,d,e,a,b,23); + nR2(b,c,d,e,a,24); nR2(a,b,c,d,e,25); nR2(e,a,b,c,d,26); nR2(d,e,a,b,c,27); + nR2(c,d,e,a,b,28); nR2(b,c,d,e,a,29); nR2(a,b,c,d,e,30); nR2(e,a,b,c,d,31); + nR2(d,e,a,b,c,32); nR2(c,d,e,a,b,33); nR2(b,c,d,e,a,34); nR2(a,b,c,d,e,35); + nR2(e,a,b,c,d,36); nR2(d,e,a,b,c,37); nR2(c,d,e,a,b,38); nR2(b,c,d,e,a,39); +} + +void +do_R3(u_int32_t *a, u_int32_t *b, u_int32_t *c, u_int32_t *d, u_int32_t *e, CHAR64LONG16 *block) +{ + nR3(a,b,c,d,e,40); nR3(e,a,b,c,d,41); nR3(d,e,a,b,c,42); nR3(c,d,e,a,b,43); + nR3(b,c,d,e,a,44); nR3(a,b,c,d,e,45); nR3(e,a,b,c,d,46); nR3(d,e,a,b,c,47); + nR3(c,d,e,a,b,48); nR3(b,c,d,e,a,49); nR3(a,b,c,d,e,50); nR3(e,a,b,c,d,51); + nR3(d,e,a,b,c,52); nR3(c,d,e,a,b,53); nR3(b,c,d,e,a,54); nR3(a,b,c,d,e,55); + nR3(e,a,b,c,d,56); nR3(d,e,a,b,c,57); nR3(c,d,e,a,b,58); nR3(b,c,d,e,a,59); +} + +void +do_R4(u_int32_t *a, u_int32_t *b, u_int32_t *c, u_int32_t *d, u_int32_t *e, CHAR64LONG16 *block) +{ + nR4(a,b,c,d,e,60); nR4(e,a,b,c,d,61); nR4(d,e,a,b,c,62); nR4(c,d,e,a,b,63); + nR4(b,c,d,e,a,64); nR4(a,b,c,d,e,65); nR4(e,a,b,c,d,66); nR4(d,e,a,b,c,67); + nR4(c,d,e,a,b,68); nR4(b,c,d,e,a,69); nR4(a,b,c,d,e,70); nR4(e,a,b,c,d,71); + nR4(d,e,a,b,c,72); nR4(c,d,e,a,b,73); nR4(b,c,d,e,a,74); nR4(a,b,c,d,e,75); + nR4(e,a,b,c,d,76); nR4(d,e,a,b,c,77); nR4(c,d,e,a,b,78); nR4(b,c,d,e,a,79); +} +#endif + /* * Hash a single 512-bit block. This is the core of the algorithm. */ -void SHA1Transform(uint32_t state[SHA1_DIGEST_LENGTH/4], - const uint8_t buffer[SHA1_BLOCK_SIZE]) +void SHA1Transform(state, buffer) + u_int32_t state[5]; + const u_char buffer[64]; { - uint32_t a, b, c, d, e; + u_int32_t a, b, c, d, e; CHAR64LONG16 *block; #ifdef SHA1HANDSOFF @@ -72,7 +136,7 @@ void SHA1Transform(uint32_t state[SHA1_DIGEST_LENGTH/4], #ifdef SHA1HANDSOFF block = &workspace; - (void)memcpy(block, buffer, SHA1_BLOCK_SIZE); + (void)memcpy(block, buffer, 64); #else block = (CHAR64LONG16 *)(void *)buffer; #endif @@ -84,6 +148,12 @@ void SHA1Transform(uint32_t state[SHA1_DIGEST_LENGTH/4], d = state[3]; e = state[4]; +#ifdef SPARC64_GCC_WORKAROUND + do_R01(&a, &b, &c, &d, &e, block); + do_R2(&a, &b, &c, &d, &e, block); + do_R3(&a, &b, &c, &d, &e, block); + do_R4(&a, &b, &c, &d, &e, block); +#else /* 4 rounds of 20 operations each. Loop unrolled. */ R0(a,b,c,d,e, 0); R0(e,a,b,c,d, 1); R0(d,e,a,b,c, 2); R0(c,d,e,a,b, 3); R0(b,c,d,e,a, 4); R0(a,b,c,d,e, 5); R0(e,a,b,c,d, 6); R0(d,e,a,b,c, 7); @@ -105,6 +175,7 @@ void SHA1Transform(uint32_t state[SHA1_DIGEST_LENGTH/4], R4(c,d,e,a,b,68); R4(b,c,d,e,a,69); R4(a,b,c,d,e,70); R4(e,a,b,c,d,71); R4(d,e,a,b,c,72); R4(c,d,e,a,b,73); R4(b,c,d,e,a,74); R4(a,b,c,d,e,75); R4(e,a,b,c,d,76); R4(d,e,a,b,c,77); R4(c,d,e,a,b,78); R4(b,c,d,e,a,79); +#endif /* Add the working vars back into context.state[] */ state[0] += a; @@ -121,91 +192,78 @@ void SHA1Transform(uint32_t state[SHA1_DIGEST_LENGTH/4], /* * SHA1Init - Initialize new context */ -void SHA1Init(SHA1_CTX *context) +void SHA1Init(context) + SHA1_CTX *context; { + assert(context != 0); /* SHA1 initialization constants */ - *context = (SHA1_CTX) { - .state = { - 0x67452301, - 0xEFCDAB89, - 0x98BADCFE, - 0x10325476, - 0xC3D2E1F0, - }, - .count = 0, - }; + context->state[0] = 0x67452301; + context->state[1] = 0xEFCDAB89; + context->state[2] = 0x98BADCFE; + context->state[3] = 0x10325476; + context->state[4] = 0xC3D2E1F0; + context->count[0] = context->count[1] = 0; } /* * Run your data through this. */ -void SHA1Update(SHA1_CTX *context, const uint8_t *data, unsigned int len) +void SHA1Update(context, data, len) + SHA1_CTX *context; + const u_char *data; + u_int len; { - unsigned int i, j; - unsigned int partial, done; - const uint8_t *src; + u_int i, j; assert(context != 0); assert(data != 0); - partial = context->count % SHA1_BLOCK_SIZE; - context->count += len; - done = 0; - src = data; - - if ((partial + len) >= SHA1_BLOCK_SIZE) { - if (partial) { - done = -partial; - memcpy(context->buffer + partial, data, done + SHA1_BLOCK_SIZE); - src = context->buffer; - } - do { - SHA1Transform(context->state, src); - done += SHA1_BLOCK_SIZE; - src = data + done; - } while (done + SHA1_BLOCK_SIZE <= len); - partial = 0; + j = context->count[0]; + if ((context->count[0] += len << 3) < j) + context->count[1] += (len>>29)+1; + j = (j >> 3) & 63; + if ((j + len) > 63) { + (void)memcpy(&context->buffer[j], data, (i = 64-j)); + SHA1Transform(context->state, context->buffer); + for ( ; i + 63 < len; i += 64) + SHA1Transform(context->state, &data[i]); + j = 0; + } else { + i = 0; } - memcpy(context->buffer + partial, src, len - done); + (void)memcpy(&context->buffer[j], &data[i], len - i); } /* * Add padding and return the message digest. */ -void SHA1Final(uint8_t digest[SHA1_DIGEST_LENGTH], SHA1_CTX *context) +void SHA1Final(digest, context) + u_char digest[20]; + SHA1_CTX* context; { - uint32_t i, index, pad_len; - uint64_t bits; - static const uint8_t padding[SHA1_BLOCK_SIZE] = { 0x80, }; + u_int i; + u_char finalcount[8]; assert(digest != 0); assert(context != 0); -#if BYTE_ORDER == LITTLE_ENDIAN - bits = swap64(context->count << 3); -#else - bits = context->count << 3; -#endif - - /* Pad out to 56 mod 64 */ - index = context->count & 0x3f; - pad_len = (index < 56) ? (56 - index) : ((64 + 56) - index); - SHA1Update(context, padding, pad_len); - - /* Append length */ - SHA1Update(context, (const uint8_t *)&bits, sizeof(bits)); + for (i = 0; i < 8; i++) { + finalcount[i] = (u_char)((context->count[(i >= 4 ? 0 : 1)] + >> ((3-(i & 3)) * 8) ) & 255); /* Endian independent */ + } + SHA1Update(context, (const u_char *)"\200", 1); + while ((context->count[0] & 504) != 448) + SHA1Update(context, (const u_char *)"\0", 1); + SHA1Update(context, finalcount, 8); /* Should cause a SHA1Transform() */ if (digest) { - for (i = 0; i < SHA1_DIGEST_LENGTH/4; i++) -#if BYTE_ORDER == LITTLE_ENDIAN - ((uint32_t *)digest)[i] = swap32(context->state[i]); -#else - ((uint32_t *)digest)[i] = context->state[i]; -#endif + for (i = 0; i < 20; i++) + digest[i] = (u_char) + ((context->state[i>>2] >> ((3-(i & 3)) * 8) ) & 255); } } diff --git a/libc/bionic/stubs.c b/libc/bionic/stubs.c index 5f63427229..9e48d903b1 100644 --- a/libc/bionic/stubs.c +++ b/libc/bionic/stubs.c @@ -413,7 +413,7 @@ getgrnam(const char *name) struct netent* getnetbyname(const char *name) { - fprintf(stderr, "FIX ME! implement getgrnam() %s:%d\n", __FILE__, __LINE__); + fprintf(stderr, "FIX ME! implement getnetbyname() %s:%d\n", __FILE__, __LINE__); return NULL; } @@ -445,15 +445,46 @@ struct netent *getnetbyaddr(uint32_t net, int type) return NULL; } +// Android doesn't have /etc/protocols. Use this minimal list. +struct protoent protocols[] = { + {"ip", {"IP", NULL}, 0}, + {"icmp", {"ICMP", NULL}, 1}, + {"tcp", {"TCP", NULL}, 6}, + {"udp", {"UDP", NULL}, 17}, + {NULL, {NULL}, 0} +}; + struct protoent *getprotobyname(const char *name) { - fprintf(stderr, "FIX ME! implement %s() %s:%d\n", __FUNCTION__, __FILE__, __LINE__); + int i = 0; + + while (name && protocols[i].p_name != 0) + { + if (strcmp(protocols[i].p_name, name) == 0) + { + return &protocols[i]; + } + + i++; + } + return NULL; } struct protoent *getprotobynumber(int proto) { - fprintf(stderr, "FIX ME! implement %s() %s:%d\n", __FUNCTION__, __FILE__, __LINE__); + int i = 0; + + while (protocols[i].p_name != 0) + { + if (protocols[i].p_proto == proto) + { + return &protocols[i]; + } + + i++; + } + return NULL; } diff --git a/libc/include/sha1.h b/libc/include/sha1.h index bc51ac0c2b..522436e769 100644 --- a/libc/include/sha1.h +++ b/libc/include/sha1.h @@ -16,17 +16,16 @@ #define SHA1_BLOCK_SIZE 64 typedef struct { - uint64_t count; - uint32_t state[SHA1_DIGEST_LENGTH / 4]; - uint8_t buffer[SHA1_BLOCK_SIZE]; + uint32_t state[5]; + uint32_t count[2]; + u_char buffer[64]; } SHA1_CTX; __BEGIN_DECLS -void SHA1Transform(uint32_t[SHA1_DIGEST_LENGTH/4], - const uint8_t[SHA1_BLOCK_SIZE]); +void SHA1Transform(uint32_t[5], const u_char[64]); void SHA1Init(SHA1_CTX *); -void SHA1Update(SHA1_CTX *, const uint8_t *, unsigned int); -void SHA1Final(uint8_t[SHA1_DIGEST_LENGTH], SHA1_CTX *); +void SHA1Update(SHA1_CTX *, const u_char *, u_int); +void SHA1Final(u_char[SHA1_DIGEST_LENGTH], SHA1_CTX *); __END_DECLS #endif /* _SYS_SHA1_H_ */ diff --git a/libc/string/bcopy.c b/libc/string/bcopy.c index 4308c6484a..1a8c1db450 100644 --- a/libc/string/bcopy.c +++ b/libc/string/bcopy.c @@ -74,6 +74,7 @@ bcopy(const void *src0, void *dst0, size_t length) #define TLOOP1(s) do { s; } while (--t) if ((unsigned long)dst < (unsigned long)src) { + #if defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY) /* * Copy forward. */ @@ -97,7 +98,11 @@ bcopy(const void *src0, void *dst0, size_t length) TLOOP(*(word *)dst = *(word *)src; src += wsize; dst += wsize); t = length & wmask; TLOOP(*dst++ = *src++); + #else + memcpy(dst, src, length); + #endif } else { + #if !(defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY)) /* * Copy backwards. Otherwise essentially the same. * Alignment works as before, except that it takes @@ -118,6 +123,168 @@ bcopy(const void *src0, void *dst0, size_t length) TLOOP(src -= wsize; dst -= wsize; *(word *)dst = *(word *)src); t = length & wmask; TLOOP(*--dst = *--src); + #else + src += length; + dst += length; + if (!(((unsigned long)dst ^ (unsigned long)src) & 0x03)) { + // can be aligned + asm volatile ( + "pld [%[src], #-64] \n" + "tst %[src], #0x03 \n" + "beq .Lbbcopy_aligned \n" + + ".Lbbcopy_make_align: \n" + "ldrb r12, [%[src], #-1]! \n" + "subs %[length], %[length], #1 \n" + "strb r12, [%[dst], #-1]! \n" + "beq .Lbbcopy_out \n" + "tst %[src], #0x03 \n" + "bne .Lbbcopy_make_align \n" + + ".Lbbcopy_aligned: \n" + "cmp %[length], #64 \n" + "blt .Lbbcopy_align_less_64 \n" + ".Lbbcopy_align_loop64: \n" + "vldmdb %[src]!, {q0 - q3} \n" + "sub %[length], %[length], #64 \n" + "cmp %[length], #64 \n" + "pld [%[src], #-64] \n" + "pld [%[src], #-96] \n" + "vstmdb %[dst]!, {q0 - q3} \n" + "bge .Lbbcopy_align_loop64 \n" + "cmp %[length], #0 \n" + "beq .Lbbcopy_out \n" + + ".Lbbcopy_align_less_64: \n" + "cmp %[length], #32 \n" + "blt .Lbbcopy_align_less_32 \n" + "vldmdb %[src]!, {q0 - q1} \n" + "subs %[length], %[length], #32 \n" + "vstmdb %[dst]!, {q0 - q1} \n" + "beq .Lbbcopy_out \n" + + ".Lbbcopy_align_less_32: \n" + "cmp %[length], #16 \n" + "blt .Lbbcopy_align_less_16 \n" + "vldmdb %[src]!, {q0} \n" + "subs %[length], %[length], #16 \n" + "vstmdb %[dst]!, {q0} \n" + "beq .Lbbcopy_out \n" + + ".Lbbcopy_align_less_16: \n" + "cmp %[length], #8 \n" + "blt .Lbbcopy_align_less_8 \n" + "vldmdb %[src]!, {d0} \n" + "subs %[length], %[length], #8 \n" + "vstmdb %[dst]!, {d0} \n" + "beq .Lbbcopy_out \n" + + ".Lbbcopy_align_less_8: \n" + "cmp %[length], #4 \n" + "blt .Lbbcopy_align_less_4 \n" + "ldr r12, [%[src], #-4]! \n" + "subs %[length], %[length], #4 \n" + "str r12, [%[dst], #-4]! \n" + "beq .Lbbcopy_out \n" + + ".Lbbcopy_align_less_4: \n" + "cmp %[length], #2 \n" + "blt .Lbbcopy_align_less_2 \n" + "ldrh r12, [%[src], #-2]! \n" + "subs %[length], %[length], #2 \n" + "strh r12, [%[dst], #-2]! \n" + "beq .Lbbcopy_out \n" + + ".Lbbcopy_align_less_2: \n" + "ldrb r12, [%[src], #-1]! \n" + "strb r12, [%[dst], #-1]! \n" + + ".Lbbcopy_out: \n" + : + : [src] "r" (src), [dst] "r" (dst), [length] "r" (length) + : "memory", "cc", "r12" + ); + } else { + // can not be aligned + asm volatile ( + "cmp %[length], #64 \n" + "pld [%[src], #-32] \n" + "blt .Lbbcopy___less_64 \n" + "mov r12, #-32 \n" + "sub %[src], %[src], #32 \n" + "sub %[dst], %[dst], #32 \n" + ".Lbbcopy___loop64: \n" + "vld1.8 {q0 - q1}, [%[src]], r12 \n" + "vld1.8 {q2 - q3}, [%[src]], r12 \n" + "sub %[length], %[length], #64 \n" + "cmp %[length], #64 \n" + "pld [%[src], #-64] \n" + "pld [%[src], #-96] \n" + "vst1.8 {q0 - q1}, [%[dst]], r12 \n" + "vst1.8 {q2 - q3}, [%[dst]], r12 \n" + "bge .Lbbcopy___loop64 \n" + "cmp %[length], #0 \n" + "beq .Lbcopy_out \n" + "add %[src], %[src], #32 \n" + "add %[dst], %[dst], #32 \n" + + ".Lbbcopy___less_64: \n" + "cmp %[length], #32 \n" + "blt .Lbbcopy___less_32 \n" + "sub %[src], %[src], #32 \n" + "sub %[dst], %[dst], #32 \n" + "vld1.8 {q0 - q1}, [%[src]] \n" + "subs %[length], %[length], #32 \n" + "vst1.8 {q0 - q1}, [%[dst]] \n" + "beq .Lbcopy_out \n" + + ".Lbbcopy___less_32: \n" + "cmp %[length], #16 \n" + "blt .Lbbcopy___less_16 \n" + "sub %[src], %[src], #16 \n" + "sub %[dst], %[dst], #16 \n" + "vld1.8 {q0}, [%[src]] \n" + "subs %[length], %[length], #16 \n" + "vst1.8 {q0}, [%[dst]] \n" + "beq .Lbcopy_out \n" + + ".Lbbcopy___less_16: \n" + "cmp %[length], #8 \n" + "blt .Lbbcopy___less_8 \n" + "sub %[src], %[src], #8 \n" + "sub %[dst], %[dst], #8 \n" + "vld1.8 {d0}, [%[src]] \n" + "subs %[length], %[length], #8 \n" + "vst1.8 {d0}, [%[dst]] \n" + "beq .Lbcopy_out \n" + + ".Lbbcopy___less_8: \n" + "cmp %[length], #4 \n" + "blt .Lbbcopy___less_4 \n" + "ldr r12, [%[src], #-4]! \n" + "subs %[length], %[length], #4 \n" + "str r12, [%[dst], #-4]! \n" + "beq .Lbcopy_out \n" + + ".Lbbcopy___less_4: \n" + "cmp %[length], #2 \n" + "blt .Lbbcopy___less_2 \n" + "ldrh r12, [%[src], #-2]! \n" + "subs %[length], %[length], #2 \n" + "strh r12, [%[dst], #-2]! \n" + "beq .Lbcopy_out \n" + + ".Lbbcopy___less_2: \n" + "ldrb r12, [%[src], #-1]! \n" + "strb r12, [%[dst], #-1]! \n" + + ".Lbcopy_out: \n" + : + : [src] "r" (src), [dst] "r" (dst), [length] "r" (length) + : "memory", "cc", "r12" + ); + } + #endif } done: #if defined(MEMCOPY) || defined(MEMMOVE) diff --git a/libc/string/strcat.c b/libc/string/strcat.c index 7cea5229fb..ab80157fde 100644 --- a/libc/string/strcat.c +++ b/libc/string/strcat.c @@ -45,7 +45,13 @@ strcat(char *s, const char *append) { char *save = s; +#if !(defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY)) for (; *s; ++s); while ((*s++ = *append++) != '\0'); return(save); +#else + s += strlen(s); + strcpy(s, append); + return save; +#endif } diff --git a/libc/string/strncat.c b/libc/string/strncat.c index c4df4f2fad..cb8924c4d7 100644 --- a/libc/string/strncat.c +++ b/libc/string/strncat.c @@ -44,8 +44,12 @@ strncat(char *dst, const char *src, size_t n) char *d = dst; const char *s = src; + #if !(defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY)) while (*d != 0) d++; + #else + d += strlen(d); + #endif do { if ((*d = *s++) == 0) break; diff --git a/libc/unistd/getopt_long.c b/libc/unistd/getopt_long.c index dbdf01a81d..0b8181a296 100644 --- a/libc/unistd/getopt_long.c +++ b/libc/unistd/getopt_long.c @@ -100,12 +100,12 @@ static int nonopt_start = -1; /* first non option argument (for permute) */ static int nonopt_end = -1; /* first option after non options (for permute) */ /* Error messages */ -static const char recargchar[] = "option requires an argument -- %c"; -static const char recargstring[] = "option requires an argument -- %s"; -static const char ambig[] = "ambiguous option -- %.*s"; -static const char noarg[] = "option doesn't take an argument -- %.*s"; -static const char illoptchar[] = "unknown option -- %c"; -static const char illoptstring[] = "unknown option -- %s"; +static const char recargchar[] = "option requires an argument -- %c\n"; +static const char recargstring[] = "option requires an argument -- %s\n"; +static const char ambig[] = "ambiguous option -- %.*s\n"; +static const char noarg[] = "option doesn't take an argument -- %.*s\n"; +static const char illoptchar[] = "unknown option -- %c\n"; +static const char illoptstring[] = "unknown option -- %s\n"; /* * Compute the greatest common divisor of a and b. diff --git a/libm/Android.mk b/libm/Android.mk index 29c75a2bb4..ab71ed8822 100644 --- a/libm/Android.mk +++ b/libm/Android.mk @@ -152,7 +152,6 @@ libm_common_src_files:= \ src/s_isnan.c \ src/s_modf.c - ifeq ($(TARGET_ARCH),arm) libm_common_src_files += \ arm/fenv.c \ @@ -161,6 +160,18 @@ ifeq ($(TARGET_ARCH),arm) src/s_scalbn.c \ src/s_scalbnf.c + ifeq ($(TARGET_USE_KRAIT_BIONIC_OPTIMIZATION),true) + libm_common_src_files += \ + arm/e_pow.S + libm_common_cflags += -DKRAIT_NEON_OPTIMIZATION + endif + + ifeq ($(TARGET_USE_SPARROW_BIONIC_OPTIMIZATION),true) + libm_common_src_files += \ + arm/e_pow.S + libm_common_cflags += -DSPARROW_NEON_OPTIMIZATION + endif + libm_common_includes = $(LOCAL_PATH)/arm else diff --git a/libm/arm/e_pow.S b/libm/arm/e_pow.S new file mode 100644 index 0000000000..a984d4ce2b --- /dev/null +++ b/libm/arm/e_pow.S @@ -0,0 +1,432 @@ +@ Copyright (c) 2012, Code Aurora Forum. All rights reserved. +@ +@ Redistribution and use in source and binary forms, with or without +@ modification, are permitted provided that the following conditions are +@ met: +@ * Redistributions of source code must retain the above copyright +@ notice, this list of conditions and the following disclaimer. +@ * Redistributions in binary form must reproduce the above +@ copyright notice, this list of conditions and the following +@ disclaimer in the documentation and/or other materials provided +@ with the distribution. +@ * Neither the name of Code Aurora Forum, Inc. nor the names of its +@ contributors may be used to endorse or promote products derived +@ from this software without specific prior written permission. +@ +@ THIS SOFTWARE IS PROVIDED "AS IS" AND ANY EXPRESS OR IMPLIED +@ WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF +@ MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT +@ ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS +@ BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR +@ CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF +@ SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR +@ BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, +@ WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE +@ OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN +@ IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + +#include +#include + +@ Values which exist the program lifetime: +#define HIGH_WORD_MASK d31 +#define EXPONENT_MASK d30 +#define int_1 d29 +#define double_1 d28 +@ sign and 2^int_n fixup: +#define expadjustment d7 +#define literals r10 +@ Values which exist within both polynomial implementations: +#define int_n d2 +#define int_n_low s4 +#define int_n_high s5 +#define double_n d3 +#define k1 d27 +#define k2 d26 +#define k3 d25 +#define k4 d24 +#define k5 d23 +#define k6 d22 +@ Values which cross the boundaries between polynomial implementations: +#define ss d16 +#define ss2 d17 +#define ss4 d18 +#define Result d0 +#define Return_hw r1 +#define Return_lw r0 +#define ylg2x d0 +@ Intermediate values only needed sometimes: +@ initial (sorted in approximate order of availability for overwriting): +#define x_hw r1 +#define x_lw r0 +#define y_hw r3 +#define y_lw r2 +#define x d0 +#define bp d4 +#define y d1 +#define ln2 d5 +@ log series: +#define u d19 +#define v d20 +#define lg2coeff d21 +#define bpa d5 +#define bpb d3 +#define lg2const d6 +#define xmantissa r8 +#define twoto1o5 r4 +#define twoto3o5 r5 +#define ix r6 +#define iEXP_MASK r7 +@ exp input setup: +#define twoto1o8mask d3 +#define twoto1o4mask d4 +#define twoto1o2mask d1 +#define ylg2x_round_offset d16 +#define ylg2x_temp d17 +#define yn_temp d18 +#define yn_round_offset d19 +@ Careful, overwriting HIGH_WORD_MASK, reset it if you need it again ... +#define rounded_exponent d31 +@ exp series: +#define k7 d21 +#define k8 d20 +#define ss3 d19 +#define k0 d20 +#define twoto1o4 d6 + +@instructions that gas doesn't like to encode correctly: +#define vmov_f64 fconstd +#define vmov_f32 fconsts +#define vmovne_f64 fconstdne + +ENTRY(pow_neon) + vmov x, x_lw, x_hw + push {r4, r5, r6, r7, r8, r9, r10, lr} + + @ pre-staged bp values + vldr bpa, .LbpA + vldr bpb, .LbpB + @ load two fifths into constant term in case we need it due to offsets + vldr lg2const, .Ltwofifths + + @ bp is initially 1.0, may adjust later based on x value + vmov_f64 bp, #0x70 + + @ extract the mantissa from x for scaled value comparisons + lsl xmantissa, x_hw, #12 + + @ twoto1o5 = 2^(1/5) (input bracketing) + movw twoto1o5, #0x186c + movt twoto1o5, #0x2611 + @ twoto3o5 = 2^(3/5) (input bracketing) + movw twoto3o5, #0x003b + movt twoto3o5, #0x8406 + + @ finish extracting xmantissa + orr xmantissa, xmantissa, x_lw, lsr #20 + + @ begin preparing a mask for normalization + vmov.i64 HIGH_WORD_MASK, #0xffffffff00000000 + + @ double_1 = (double) 1.0 + vmov_f64 double_1, #0x70 + + vmov y, y_lw, y_hw + + cmp xmantissa, twoto1o5 + + vshl.i64 EXPONENT_MASK, HIGH_WORD_MASK, #20 + vshr.u64 int_1, HIGH_WORD_MASK, #63 + + adr literals, .LliteralTable + + movw iEXP_MASK, #0xfff0 + movt iEXP_MASK, #0x0000 + + bic ix, x_hw, iEXP_MASK + + @ if normalized x > 2^(1/5), bp = 1 + (2^(2/5)-1) = 2^(2/5) + vaddhi.f64 bp, bp, bpa + @ zero out lg2 constant term if don't offset our input + vsubls.f64 lg2const, lg2const, lg2const + + @ will need ln2 for various things + vldr ln2, .Lln2 + + cmp xmantissa, twoto3o5 +@@@@ X Value Normalization @@@@ + + @ ss = abs(x) 2^(-1024) + vbic.i64 ss, x, EXPONENT_MASK + + @ N = (floor(log2(x)) + 0x3ff) * 2^52 + vand.i64 int_n, x, EXPONENT_MASK + + @ if normalized x > 2^(3/5), bp = 2^(2/5) + (2^(4/5) - 2^(2/5) = 2^(4/5) + vaddhi.f64 bp, bp, bpb + vaddhi.f64 lg2const, lg2const, lg2const + + @ load log2 polynomial series constants + vldm literals!, {k4, k3, k2, k1} + + @ s = abs(x) 2^(-floor(log2(x))) (normalize abs(x) to around 1) + vorr.i64 ss, ss, double_1 + +@@@@ 3/2 (Log(bp(1+s)/(1-s))) input computation (s = (x-bp)/(x+bp)) @@@@ + + vsub.f64 u, ss, bp + vadd.f64 v, ss, bp + + @ s = (x-1)/(x+1) + vdiv.f64 ss, u, v + + @ load 2/(3log2) into lg2coeff + vldr lg2coeff, .Ltwooverthreeln2 + + @ N = floor(log2(x)) * 2^52 + vsub.i64 int_n, int_n, double_1 + +@@@@ 3/2 (Log(bp(1+s)/(1-s))) polynomial series @@@@ + + @ ss2 = ((x-dp)/(x+dp))^2 + vmul.f64 ss2, ss, ss + @ ylg2x = 3.0 + vmov_f64 ylg2x, #8 + vmul.f64 ss4, ss2, ss2 + + @ todo: useful later for two-way clamp + vmul.f64 lg2coeff, lg2coeff, y + + @ N = floor(log2(x)) + vshr.s64 int_n, int_n, #52 + + @ k3 = ss^2 * L4 + L3 + vmla.f64 k3, ss2, k4 + + @ k1 = ss^2 * L2 + L1 + vmla.f64 k1, ss2, k2 + + @ scale ss by 2/(3 ln 2) + vmul.f64 lg2coeff, ss, lg2coeff + + @ ylg2x = 3.0 + s^2 + vadd.f64 ylg2x, ylg2x, ss2 + + vcvt.f64.s32 double_n, int_n_low + + @ k1 = s^4 (s^2 L4 + L3) + s^2 L2 + L1 + vmla.f64 k1, ss4, k3 + + @ add in constant term + vadd.f64 double_n, lg2const + + @ ylg2x = 3.0 + s^2 + s^4 (s^4 (s^2 L4 + L3) + s^2 L2 + L1) + vmla.f64 ylg2x, ss4, k1 + + @ ylg2x = y 2 s / (3 ln(2)) (3.0 + s^2 + s^4 (s^4(s^2 L4 + L3) + s^2 L2 + L1) + vmul.f64 ylg2x, lg2coeff, ylg2x + +@@@@ Compute input to Exp(s) (s = y(n + log2(x)) - (floor(8 yn + 1)/8 + floor(8 ylog2(x) + 1)/8) @@@@@ + + @ mask to extract bit 1 (2^-2 from our fixed-point representation) + vshl.u64 twoto1o4mask, int_1, #1 + + @ double_n = y * n + vmul.f64 double_n, double_n, y + + @ Load 2^(1/4) for later computations + vldr twoto1o4, .Ltwoto1o4 + + @ move unmodified y*lg2x into temp space + vmov ylg2x_temp, ylg2x + @ move unmodified y*n into temp space + vmov yn_temp, double_n + + @ either add or subtract one based on the sign of double_n and ylg2x + vshr.s64 ylg2x_round_offset, ylg2x, #62 + vshr.s64 yn_round_offset, double_n, #62 + + @ load exp polynomial series constants + vldm literals!, {k8, k7, k6, k5, k4, k3, k2, k1} + + @ mask to extract bit 2 (2^-1 from our fixed-point representation) + vshl.u64 twoto1o2mask, int_1, #2 + + @ make rounding offsets either 1 or -1 instead of 0 or -2 + vorr.u64 ylg2x_round_offset, ylg2x_round_offset, int_1 + vorr.u64 yn_round_offset, yn_round_offset, int_1 + + @ compute floor(8 y * n + 1)/8 + @ and floor(8 y (log2(x)) + 1)/8 + vcvt.s32.f64 ylg2x, ylg2x, #3 + + vcvt.s32.f64 double_n, double_n, #3 + @ round up to the nearest 1/8th + vadd.s32 ylg2x, ylg2x, ylg2x_round_offset + vadd.s32 double_n, double_n, yn_round_offset + + @ clear out round-up bit for y log2(x) + vbic.s32 ylg2x, ylg2x, int_1 + @ clear out round-up bit for yn + vbic.s32 double_n, double_n, int_1 + @ add together the (fixed precision) rounded parts + vadd.s64 rounded_exponent, double_n, ylg2x + @ turn int_n into a double with value 2^int_n + vshl.i64 int_n, rounded_exponent, #49 + @ compute masks for 2^(1/4) and 2^(1/2) fixups for fractional part of fixed-precision rounded values: + vand.u64 twoto1o4mask, twoto1o4mask, rounded_exponent + vand.u64 twoto1o2mask, twoto1o2mask, rounded_exponent + + @ convert back into floating point, double_n now holds (double) floor(8 y * n + 1)/8 + @ ylg2x now holds (double) floor(8 y * log2(x) + 1)/8 + vcvt.f64.s32 ylg2x, ylg2x, #3 + vcvt.f64.s32 double_n, double_n, #3 + + @ put the 2 bit (0.5) through the roof of twoto1o2mask (make it 0x0 or 0xffffffffffffffff) + vqshl.u64 twoto1o2mask, twoto1o2mask, #62 + @ put the 1 bit (0.25) through the roof of twoto1o4mask (make it 0x0 or 0xffffffffffffffff) + vqshl.u64 twoto1o4mask, twoto1o4mask, #63 + + @ center y*log2(x) fractional part between -0.125 and 0.125 by subtracting (double) floor(8 y * log2(x) + 1)/8 + vsub.f64 ylg2x_temp, ylg2x_temp, ylg2x + @ center y*n fractional part between -0.125 and 0.125 by subtracting (double) floor(8 y * n + 1)/8 + vsub.f64 yn_temp, yn_temp, double_n + + @ Add fractional parts of yn and y log2(x) together + vadd.f64 ss, ylg2x_temp, yn_temp + + @ Result = 1.0 (offset for exp(s) series) + vmov_f64 Result, #0x70 + + @ multiply fractional part of y * log2(x) by ln(2) + vmul.f64 ss, ln2, ss + +@@@@ 10th order polynomial series for Exp(s) @@@@ + + @ ss2 = (ss)^2 + vmul.f64 ss2, ss, ss + + @ twoto1o2mask = twoto1o2mask & twoto1o4 + vand.u64 twoto1o2mask, twoto1o2mask, twoto1o4 + @ twoto1o2mask = twoto1o2mask & twoto1o4 + vand.u64 twoto1o4mask, twoto1o4mask, twoto1o4 + + @ Result = 1.0 + ss + vadd.f64 Result, Result, ss + + @ k7 = ss k8 + k7 + vmla.f64 k7, ss, k8 + + @ ss4 = (ss*ss) * (ss*ss) + vmul.f64 ss4, ss2, ss2 + + @ twoto1o2mask = twoto1o2mask | (double) 1.0 - results in either 1.0 or 2^(1/4) in twoto1o2mask + vorr.u64 twoto1o2mask, twoto1o2mask, double_1 + @ twoto1o2mask = twoto1o4mask | (double) 1.0 - results in either 1.0 or 2^(1/4) in twoto1o4mask + vorr.u64 twoto1o4mask, twoto1o4mask, double_1 + + @ TODO: should setup sign here, expadjustment = 1.0 + vmov_f64 expadjustment, #0x70 + + @ ss3 = (ss*ss) * ss + vmul.f64 ss3, ss2, ss + + @ k0 = 1/2 (first non-unity coefficient) + vmov_f64 k0, #0x60 + + @ Mask out non-exponent bits to make sure we have just 2^int_n + vand.i64 int_n, int_n, EXPONENT_MASK + + @ square twoto1o2mask to get 1.0 or 2^(1/2) + vmul.f64 twoto1o2mask, twoto1o2mask, twoto1o2mask + @ multiply twoto2o4mask into the exponent output adjustment value + vmul.f64 expadjustment, expadjustment, twoto1o4mask + + @ k5 = ss k6 + k5 + vmla.f64 k5, ss, k6 + + @ k3 = ss k4 + k3 + vmla.f64 k3, ss, k4 + + @ k1 = ss k2 + k1 + vmla.f64 k1, ss, k2 + + @ multiply twoto1o2mask into exponent output adjustment value + vmul.f64 expadjustment, expadjustment, twoto1o2mask + + @ k5 = ss^2 ( ss k8 + k7 ) + ss k6 + k5 + vmla.f64 k5, ss2, k7 + + @ k1 = ss^2 ( ss k4 + k3 ) + ss k2 + k1 + vmla.f64 k1, ss2, k3 + + @ Result = 1.0 + ss + 1/2 ss^2 + vmla.f64 Result, ss2, k0 + + @ Adjust int_n so that it's a double precision value that can be multiplied by Result + vadd.i64 expadjustment, int_n, expadjustment + + @ k1 = ss^4 ( ss^2 ( ss k8 + k7 ) + ss k6 + k5 ) + ss^2 ( ss k4 + k3 ) + ss k2 + k1 + vmla.f64 k1, ss4, k5 + + @ Result = 1.0 + ss + 1/2 ss^2 + ss^3 ( ss^4 ( ss^2 ( ss k8 + k7 ) + ss k6 + k5 ) + ss^2 ( ss k4 + k3 ) + ss k2 + k1 ) + vmla.f64 Result, ss3, k1 + + @ multiply by adjustment (sign*(rounding ? sqrt(2) : 1) * 2^int_n) + vmul.f64 Result, expadjustment, Result + +.LleavePow: + @ return Result (FP) + vmov Return_lw, Return_hw, Result +.LleavePowDirect: + @ leave directly returning whatever is in Return_lw and Return_hw + pop {r4, r5, r6, r7, r8, r9, r10, pc} + +.align 6 +.LliteralTable: +@ Least-sqares tuned constants for 11th order (log2((1+s)/(1-s)): +.LL4: @ ~3/11 + .long 0x53a79915, 0x3fd1b108 +.LL3: @ ~1/3 + .long 0x9ca0567a, 0x3fd554fa +.LL2: @ ~3/7 + .long 0x1408e660, 0x3fdb6db7 +.LL1: @ ~3/5 + .long 0x332D4313, 0x3fe33333 + +@ Least-squares tuned constants for 10th order exp(s): +.LE10: @ ~1/3628800 + .long 0x25c7ba0a, 0x3e92819b +.LE9: @ ~1/362880 + .long 0x9499b49c, 0x3ec72294 +.LE8: @ ~1/40320 + .long 0xabb79d95, 0x3efa019f +.LE7: @ ~1/5040 + .long 0x8723aeaa, 0x3f2a019f +.LE6: @ ~1/720 + .long 0x16c76a94, 0x3f56c16c +.LE5: @ ~1/120 + .long 0x11185da8, 0x3f811111 +.LE4: @ ~1/24 + .long 0x5555551c, 0x3fa55555 +.LE3: @ ~1/6 + .long 0x555554db, 0x3fc55555 + +.LbpA: @ (2^(2/5) - 1) + .long 0x4ee54db1, 0x3fd472d1 + +.LbpB: @ (2^(4/5) - 2^(2/5)) + .long 0x1c8a36cf, 0x3fdafb62 + +.Ltwofifths: @ + .long 0x9999999a, 0x3fd99999 + +.Ltwooverthreeln2: + .long 0xDC3A03FD, 0x3FEEC709 + +.Lln2: @ ln(2) + .long 0xFEFA39EF, 0x3FE62E42 + +.Ltwoto1o4: @ 2^1/4 + .long 0x0a31b715, 0x3ff306fe +END(pow) diff --git a/libm/src/e_pow.c b/libm/src/e_pow.c index d213132534..69f2713dad 100644 --- a/libm/src/e_pow.c +++ b/libm/src/e_pow.c @@ -13,6 +13,10 @@ static char rcsid[] = "$FreeBSD: src/lib/msun/src/e_pow.c,v 1.11 2005/02/04 18:26:06 das Exp $"; #endif +#if defined(KRAIT_NEON_OPTIMIZATION) || defined(SPARROW_NEON_OPTIMIZATION) +double pow_neon(double x, double y); +#endif + /* __ieee754_pow(x,y) return x**y * * n @@ -201,6 +205,10 @@ __ieee754_pow(double x, double y) t1 = u+v; SET_LOW_WORD(t1,0); t2 = v-(t1-u); +#if defined(KRAIT_NEON_OPTIMIZATION) || defined(SPARROW_NEON_OPTIMIZATION) + } else if (ix <= 0x40100000 && iy <= 0x40100000 && hy > 0 && hx > 0) { + return pow_neon(x,y); +#endif } else { double ss,s2,s_h,s_l,t_h,t_l; n = 0; diff --git a/linker/linker.c b/linker/linker.c index a4bb25bac6..6413b6a66f 100644 --- a/linker/linker.c +++ b/linker/linker.c @@ -108,7 +108,10 @@ static const char *ldpreload_names[LDPRELOAD_MAX + 1]; static soinfo *preloads[LDPRELOAD_MAX + 1]; +#if LINKER_DEBUG int debug_verbosity; +#endif + static int pid; /* This boolean is set if the program being loaded is setuid */ @@ -908,10 +911,10 @@ load_segments(int fd, void *header, soinfo *si) { Elf32_Ehdr *ehdr = (Elf32_Ehdr *)header; Elf32_Phdr *phdr = (Elf32_Phdr *)((unsigned char *)header + ehdr->e_phoff); - unsigned char *base = (unsigned char *)si->base; + Elf32_Addr base = (Elf32_Addr) si->base; int cnt; unsigned len; - unsigned char *tmp; + Elf32_Addr tmp; unsigned char *pbase; unsigned char *extra_base; unsigned extra_len; @@ -935,7 +938,7 @@ load_segments(int fd, void *header, soinfo *si) TRACE("[ %d - Trying to load segment from '%s' @ 0x%08x " "(0x%08x). p_vaddr=0x%08x p_offset=0x%08x ]\n", pid, si->name, (unsigned)tmp, len, phdr->p_vaddr, phdr->p_offset); - pbase = mmap(tmp, len, PFLAGS_TO_PROT(phdr->p_flags), + pbase = mmap((void *)tmp, len, PFLAGS_TO_PROT(phdr->p_flags), MAP_PRIVATE | MAP_FIXED, fd, phdr->p_offset & (~PAGE_MASK)); if (pbase == MAP_FAILED) { @@ -977,7 +980,7 @@ load_segments(int fd, void *header, soinfo *si) * | | * _+---------------------+ page boundary */ - tmp = (unsigned char *)(((unsigned)pbase + len + PAGE_SIZE - 1) & + tmp = (Elf32_Addr)(((unsigned)pbase + len + PAGE_SIZE - 1) & (~PAGE_MASK)); if (tmp < (base + phdr->p_vaddr + phdr->p_memsz)) { extra_len = base + phdr->p_vaddr + phdr->p_memsz - tmp; @@ -1650,7 +1653,6 @@ static void call_constructors(soinfo *si) } } - static void call_destructors(soinfo *si) { if (si->fini_array) { @@ -1969,7 +1971,7 @@ static int link_image(soinfo *si, unsigned wr_offset) of the DT_NEEDED entry itself, so that we can retrieve the soinfo directly later from the dynamic segment. This is a hack, but it allows us to map from DT_NEEDED to soinfo efficiently - later on when we resolve relocations, trying to look up a symgol + later on when we resolve relocations, trying to look up a symbol with dlsym(). */ d[1] = (unsigned)lsi; @@ -2152,10 +2154,12 @@ unsigned __linker_init(unsigned **elfdata) /* Get a few environment variables */ { +#if LINKER_DEBUG const char* env; env = linker_env_get("DEBUG"); /* XXX: TODO: Change to LD_DEBUG */ if (env) debug_verbosity = atoi(env); +#endif /* Normally, these are cleaned by linker_env_secure, but the test * against program_is_setuid doesn't cost us anything */ From b9980c76e71436f536906bf2e7f078fe0476aa38 Mon Sep 17 00:00:00 2001 From: cmartinezlozano Date: Fri, 31 Aug 2012 00:06:01 +0200 Subject: [PATCH 5/5] Fixed wrong ifdef for NEON Change-Id: I32814217e7ab0390bf9799c2e6da22ee857567f3 --- libc/string/bcopy.c | 50 +++++++++++++++++++++---------------------- libc/string/strcat.c | 10 ++++----- libc/string/strncat.c | 6 +++--- 3 files changed, 33 insertions(+), 33 deletions(-) diff --git a/libc/string/bcopy.c b/libc/string/bcopy.c index 1a8c1db450..a1ac4e469d 100644 --- a/libc/string/bcopy.c +++ b/libc/string/bcopy.c @@ -74,7 +74,9 @@ bcopy(const void *src0, void *dst0, size_t length) #define TLOOP1(s) do { s; } while (--t) if ((unsigned long)dst < (unsigned long)src) { - #if defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY) + #if defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY) + memcpy(dst, src, length); + #else /* * Copy forward. */ @@ -98,32 +100,9 @@ bcopy(const void *src0, void *dst0, size_t length) TLOOP(*(word *)dst = *(word *)src; src += wsize; dst += wsize); t = length & wmask; TLOOP(*dst++ = *src++); - #else - memcpy(dst, src, length); #endif } else { - #if !(defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY)) - /* - * Copy backwards. Otherwise essentially the same. - * Alignment works as before, except that it takes - * (t&wmask) bytes to align, not wsize-(t&wmask). - */ - src += length; - dst += length; - t = (long)src; - if ((t | (long)dst) & wmask) { - if ((t ^ (long)dst) & wmask || length <= wsize) - t = length; - else - t &= wmask; - length -= t; - TLOOP1(*--dst = *--src); - } - t = length / wsize; - TLOOP(src -= wsize; dst -= wsize; *(word *)dst = *(word *)src); - t = length & wmask; - TLOOP(*--dst = *--src); - #else + #if defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY) src += length; dst += length; if (!(((unsigned long)dst ^ (unsigned long)src) & 0x03)) { @@ -284,6 +263,27 @@ bcopy(const void *src0, void *dst0, size_t length) : "memory", "cc", "r12" ); } + #else + /* + * Copy backwards. Otherwise essentially the same. + * Alignment works as before, except that it takes + * (t&wmask) bytes to align, not wsize-(t&wmask). + */ + src += length; + dst += length; + t = (long)src; + if ((t | (long)dst) & wmask) { + if ((t ^ (long)dst) & wmask || length <= wsize) + t = length; + else + t &= wmask; + length -= t; + TLOOP1(*--dst = *--src); + } + t = length / wsize; + TLOOP(src -= wsize; dst -= wsize; *(word *)dst = *(word *)src); + t = length & wmask; + TLOOP(*--dst = *--src); #endif } done: diff --git a/libc/string/strcat.c b/libc/string/strcat.c index ab80157fde..5f3c696d71 100644 --- a/libc/string/strcat.c +++ b/libc/string/strcat.c @@ -45,13 +45,13 @@ strcat(char *s, const char *append) { char *save = s; -#if !(defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY)) - for (; *s; ++s); - while ((*s++ = *append++) != '\0'); - return(save); -#else +#if defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY) s += strlen(s); strcpy(s, append); return save; +#else + for (; *s; ++s); + while ((*s++ = *append++) != '\0'); + return(save); #endif } diff --git a/libc/string/strncat.c b/libc/string/strncat.c index cb8924c4d7..241c2579f8 100644 --- a/libc/string/strncat.c +++ b/libc/string/strncat.c @@ -44,11 +44,11 @@ strncat(char *dst, const char *src, size_t n) char *d = dst; const char *s = src; - #if !(defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY)) + #if defined(__ARM_NEON__) && !defined(ARCH_ARM_USE_NON_NEON_MEMCPY) + d += strlen(d); + #else while (*d != 0) d++; - #else - d += strlen(d); #endif do { if ((*d = *s++) == 0)