[PATCH v2 3/6] RISC-V: memmove() speed optimized: Add loop-unrolling

m fally marlene.fally@gmail.com
Tue Jun 17 15:07:32 GMT 2025


Add loop-unrolling for the case where both source and destination
address are aligned in the case of a destructive overlap, and
increase the unroll factor from 4 to 9 for the word-by-word
copy loop in the non-destructive case.
This matches the loop-unrolling done in memcpy() and increases
performance for lenghts >= SZREG*9 while almost not at all
degrading performance for shorter lengths.

Reviewed-by: Christian Herber <christian.herber@oss.nxp.com>
Signed-off-by: m fally <marlene.fally@gmail.com>
---
 newlib/libc/machine/riscv/memmove.c | 57 ++++++++++++++++++++++++-----
 1 file changed, 48 insertions(+), 9 deletions(-)

diff --git a/newlib/libc/machine/riscv/memmove.c b/newlib/libc/machine/riscv/memmove.c
index 2e5c6ca9b..f8937c2a2 100644
--- a/newlib/libc/machine/riscv/memmove.c
+++ b/newlib/libc/machine/riscv/memmove.c
@@ -30,6 +30,31 @@ __libc_fast_xlen_aligned (void *dst, const void *src)
 #endif
 }
 
+static inline void
+__libc_aligned_copy_unrolled (uintxlen_t *aligned_dst,
+                              const uintxlen_t *aligned_src)
+{
+  uintxlen_t dst0 = *aligned_src++;
+  uintxlen_t dst1 = *aligned_src++;
+  uintxlen_t dst2 = *aligned_src++;
+  uintxlen_t dst3 = *aligned_src++;
+  uintxlen_t dst4 = *aligned_src++;
+  uintxlen_t dst5 = *aligned_src++;
+  uintxlen_t dst6 = *aligned_src++;
+  uintxlen_t dst7 = *aligned_src++;
+  uintxlen_t dst8 = *aligned_src;
+
+  *aligned_dst++ = dst0;
+  *aligned_dst++ = dst1;
+  *aligned_dst++ = dst2;
+  *aligned_dst++ = dst3;
+  *aligned_dst++ = dst4;
+  *aligned_dst++ = dst5;
+  *aligned_dst++ = dst6;
+  *aligned_dst++ = dst7;
+  *aligned_dst = dst8;
+}
+
 void *__inhibit_loop_to_libcall
 memmove (void *dst_void, const void *src_void, size_t length)
 {
@@ -49,7 +74,20 @@ memmove (void *dst_void, const void *src_void, size_t length)
           aligned_dst = (uintxlen_t *)dst;
           aligned_src = (uintxlen_t *)src;
 
-          /* Copy one uintxlen_t word at a time if possible.  */
+          /* If possible, unroll the word-copy loop by a factor 9 to
+             match memcpy. This speeds up the copying process for longer
+             lengths while barely degrading performance for lengths < SZREG*9.
+             Since we are copying backwards, decrement the addresses
+             before copying.
+           */
+          while (length >= SZREG * 9)
+            {
+              aligned_dst -= 9;
+              aligned_src -= 9;
+              __libc_aligned_copy_unrolled (aligned_dst, aligned_src);
+              length -= (SZREG * 9);
+            }
+
           while (length >= SZREG)
             {
               *--aligned_dst = *--aligned_src;
@@ -76,17 +114,18 @@ memmove (void *dst_void, const void *src_void, size_t length)
           aligned_dst = (uintxlen_t *)dst;
           aligned_src = (uintxlen_t *)src;
 
-          /* Copy 4X uintxlen_t words at a time if possible.  */
-          while (length >= (SZREG * 4))
+          /* If possible, unroll the word-copy loop by a factor 9 to
+             match memcpy. This speeds up the copying process for longer
+             lengths while barely degrading performance for lengths < SZREG*9.
+           */
+          while (length >= SZREG * 9)
             {
-              *aligned_dst++ = *aligned_src++;
-              *aligned_dst++ = *aligned_src++;
-              *aligned_dst++ = *aligned_src++;
-              *aligned_dst++ = *aligned_src++;
-              length -= SZREG * 4;
+              __libc_aligned_copy_unrolled (aligned_dst, aligned_src);
+              aligned_dst += 9;
+              aligned_src += 9;
+              length -= (SZREG * 9);
             }
 
-          /* Copy one uintxlen_t word at a time if possible.  */
           while (length >= SZREG)
             {
               *aligned_dst++ = *aligned_src++;
-- 
2.49.0



More information about the Newlib mailing list