[PATCH v2 5/6] RISC-V: memmove() speed optimized: Call memcpy()
m fally <[email protected]>
| Newsgroups | gmane.comp.lib.newlib |
|---|---|
| Message-ID | <[email protected]> |
Redirect to memcpy() if the memory areas of source and destination do not overlap. Only redirect if length is > SZREG in order to reduce overhead on very short copies. Signed-off-by: m fally <[email protected]> --- newlib/libc/machine/riscv/memmove.c | 160 ++++++++++++++++------------ 1 file changed, 93 insertions(+), 67 deletions(-) diff --git a/newlib/libc/machine/riscv/memmove.c b/newlib/libc/machine/riscv/memmove.c index 12010f20f..209a75c69 100644 --- a/newlib/libc/machine/riscv/memmove.c +++ b/newlib/libc/machine/riscv/memmove.c @@ -76,6 +76,16 @@ __libc_aligned_copy_unrolled (uintxlen_t *aligned_dst, *aligned_dst = dst8; } +static inline void +__libc_memmove_bytewise_forward_copy (unsigned char *dst, + const unsigned char *src, size_t length) +{ + while (length--) + { + *dst++ = *src++; + } +} + void *__inhibit_loop_to_libcall memmove (void *dst_void, const void *src_void, size_t length) { @@ -84,82 +94,87 @@ memmove (void *dst_void, const void *src_void, size_t length) uintxlen_t *aligned_dst; const uintxlen_t *aligned_src; - if (src < dst && dst < src + length) + if (src <= dst) { - /* Destructive overlap...have to copy backwards */ - src += length; - dst += length; - - if (length >= SZREG) + if (dst < src + length) /* Memory areas overlap destructively, have to + copy backwards. */ { - if (__libc_fast_xlen_aligned (dst, src)) - { - aligned_dst = (uintxlen_t *)dst; - aligned_src = (uintxlen_t *)src; + src += length; + dst += length; - /* If possible, unroll the word-copy loop by a factor 9 to - match memcpy. This speeds up the copying process for longer - lengths while barely degrading performance for lengths < - SZREG*9. Since we are copying backwards, decrement the - addresses before copying. - */ - while (length >= SZREG * 9) + if (length >= SZREG) + { + if (__libc_fast_xlen_aligned (dst, src)) { - aligned_dst -= 9; - aligned_src -= 9; - __libc_aligned_copy_unrolled (aligned_dst, aligned_src); - length -= (SZREG * 9); - } + aligned_dst = (uintxlen_t *)dst; + aligned_src = (uintxlen_t *)src; - while (length >= SZREG) - { - *--aligned_dst = *--aligned_src; - length -= SZREG; - } + /* If possible, unroll the word-copy loop by a factor 9 to + match memcpy. This speeds up the copying process for + longer lengths while barely degrading performance for + lengths < SZREG*9. Since we are copying backwards, + decrement the addresses before copying. + */ + while (length >= SZREG * 9) + { + aligned_dst -= 9; + aligned_src -= 9; + __libc_aligned_copy_unrolled (aligned_dst, aligned_src); + length -= (SZREG * 9); + } - /* Pick up any residual with a byte copier. */ - dst = (unsigned char *)aligned_dst; - src = (unsigned char *)aligned_src; - } + while (length >= SZREG) + { + *--aligned_dst = *--aligned_src; + length -= SZREG; + } + + /* Pick up any residual with a byte copier. */ + dst = (unsigned char *)aligned_dst; + src = (unsigned char *)aligned_src; + } #if !defined(__riscv_misaligned_fast) - else if (length > (SZREG * 2)) - { - /* At least one address is not xlen-aligned. If - misaligned accesses are slow or prohibited, - align the src so we can load SZREG bytes at a time. - This reduces the amount of memory accesses made - and therefore improves performance. - */ - while ((uintxlen_t)src & (SZREG - 1)) + else if (length > (SZREG * 2)) { - *--dst = *--src; - length--; - } + /* At least one address is not xlen-aligned. If + misaligned accesses are slow or prohibited, + align the src so we can load SZREG bytes at a time. + This reduces the amount of memory accesses made + and therefore improves performance. + */ + while ((uintxlen_t)src & (SZREG - 1)) + { + *--dst = *--src; + length--; + } - aligned_src = (uintxlen_t *)src; + aligned_src = (uintxlen_t *)src; - /* Decrement the addresses before copying since - we are copying backwards. */ - do - { - aligned_src--; - dst -= SZREG; - __libc_memmove_misaligned_copy (dst, aligned_src); - length -= SZREG; - } - while (length >= SZREG); + /* Decrement the addresses before copying since + we are copying backwards. */ + do + { + aligned_src--; + dst -= SZREG; + __libc_memmove_misaligned_copy (dst, aligned_src); + length -= SZREG; + } + while (length >= SZREG); - /* Pick up any residual with a byte copier. */ - src = (unsigned char *)aligned_src; - } + /* Pick up any residual with a byte copier. */ + src = (unsigned char *)aligned_src; + } #endif - } - while (length--) - { - *--dst = *--src; + } + while (length--) + { + *--dst = *--src; + } + + return dst_void; } } - else /* Memory areas overlap non-destructively or not at all. */ + else if (src < dst + length) /* Memory areas overlap non-destructively. */ { if (length >= SZREG) { @@ -222,12 +237,23 @@ memmove (void *dst_void, const void *src_void, size_t length) } #endif } - while (length--) - { - *dst++ = *src++; - } + + __libc_memmove_bytewise_forward_copy (dst, src, length); + return dst_void; } - return dst_void; + /* Memory areas do not overlap, redirect to memcpy. + Copy byte-by-byte for lengths <= SZREG to reduce + overhead on very short copies. + */ + if (length > SZREG) + { + return memcpy (dst_void, src_void, length); + } + else + { + __libc_memmove_bytewise_forward_copy (dst, src, length); + return dst_void; + } } #endif -- 2.49.0