[PATCH 3/3] RISC-V: memcpy() align dest when misaligned access is prohibited

Sebastian Huber sebastian.huber@embedded-brains.de
Mon Jun 30 10:01:33 GMT 2025


Hello Christian,

I don't want to touch the Newlib configuration to get a working library. GCC 13 is still maintained, so it should work out of the box. Optimizations which work only for a subset of machines should be only enabled if they are supported (this is the case, when __riscv_misaligned_fast is defined).

----- Am 30. Jun 2025 um 11:52 schrieb Christian Herber christian.herber@oss.nxp.com:

> I believe there is a config parameter to newlib, which you can use if your
> toolchain does not support the conventional macros.
> I do not know them off by hard, but maybe this pointer helps you already.
> 
>> -----Original Message-----
>> From: Sebastian Huber <sebastian.huber@embedded-brains.de>
>> Sent: Monday, 30 June 2025 11:41
>> To: ma mandourr <ma.mandourr@gmail.com>
>> Cc: newlib <newlib@sourceware.org>; Christian Herber (OSS)
>> <christian.herber@oss.nxp.com>
>> Subject: Re: [PATCH 3/3] RISC-V: memcpy() align dest when misaligned access is
>> prohibited
>> 
>> Hello Mahmoud Abumandour,
>> 
>> I use Newlib on a Microchip PolarFire SoC which doesn't support misaligned
>> access. I get exceptions (mcause is 4 or 6) with the current Newlib. I use the
>> following machine options with GCC 13:
>> 
>> -march=rv64imac -mabi=lp64 -mcmodel=medany -mstrict-align
>> 
>> It seems that neither __riscv_misaligned_slow nor __riscv_misaligned_avoid is
>> defined by GCC 13, so the workaround is not activated. This define is only
>> available in GCC 14 and later.
>> 
>> Would it make sense to change this check to
>> 
>> #if !defined(__riscv_misaligned_fast)
>> 
>> instead of
>> 
>> #if defined(__riscv_misaligned_slow) || defined(__riscv_misaligned_avoid)
>> 
>> ?
>> 
>> ----- Am 27. Apr 2025 um 13:41 schrieb ma mandourr ma.mandourr@gmail.com:
>> 
>> > From: Mahmoud Abumandour <ma.mandourr@gmail.com>
>> >
>> > Add a code path for when source and dest are differently aligned.
>> >
>> > If misaligned access is slow or prohibited, and the alignments of the
>> > source and destination are different, we align the destination to do
>> > XLEN stores. This uses only one aligned store for every four (or eight
>> > for XLEN == 64) bytes of data.
>> >
>> > Reviewed-by: Christian Herber <christian.herber@oss.nxp.com>
>> > Signed-off-by: Mahmoud Abumandour <ma.mandourr@gmail.com>
>> > ---
>> > newlib/libc/machine/riscv/memcpy.c | 72 +++++++++++++++++++++++++-----
>> > 1 file changed, 60 insertions(+), 12 deletions(-)
>> >
>> > diff --git a/newlib/libc/machine/riscv/memcpy.c
>> > b/newlib/libc/machine/riscv/memcpy.c
>> > index 427005efd..5d6b2f301 100644
>> > --- a/newlib/libc/machine/riscv/memcpy.c
>> > +++ b/newlib/libc/machine/riscv/memcpy.c
>> > @@ -16,7 +16,6 @@
>> >
>> > #include "../../string/local.h"
>> > #include "xlenint.h"
>> > -#include <stdint.h>
>> > #include <string.h>
>> > #include <sys/asm.h>
>> >
>> > @@ -31,6 +30,30 @@ __libc_memcpy_bytewise (unsigned char *dst, const
>> > unsigned char *src,
>> >     *dst++ = *src++;
>> > }
>> >
>> > +#if defined(__riscv_misaligned_slow) ||
>> > +defined(__riscv_misaligned_avoid) static uintxlen_t __libc_load_xlen
>> > +(const void *src) {
>> > +  const unsigned char *p = (const unsigned char *)src;
>> > +  uintxlen_t ret = 0;
>> > +  unsigned char b0 = *p++;
>> > +  unsigned char b1 = *p++;
>> > +  unsigned char b2 = *p++;
>> > +  unsigned char b3 = *p++;
>> > +  ret = (uintxlen_t)b0 | ((uintxlen_t)b1 << 8) | ((uintxlen_t)b2 << 16)
>> > +        | ((uintxlen_t)b3 << 24);
>> > +#if __riscv_xlen == 64
>> > +  unsigned char b4 = *p++;
>> > +  unsigned char b5 = *p++;
>> > +  unsigned char b6 = *p++;
>> > +  unsigned char b7 = *p++;
>> > +  ret |= ((uintxlen_t)b4 << 32) | ((uintxlen_t)b5 << 40)
>> > +         | ((uintxlen_t)b6 << 48) | ((uintxlen_t)b7 << 56); #endif
>> > +  return ret;
>> > +}
>> > +#endif
>> > +
>> > void *
>> > __inhibit_loop_to_libcall
>> > memcpy (void *__restrict aa, const void *__restrict bb, size_t n) @@
>> > -39,23 +62,51 @@ memcpy (void *__restrict aa, const void *__restrict
>> > bb, size_t n)
>> >   const unsigned char *b = (const unsigned char *)bb;
>> >   unsigned char *end = a + n;
>> >   uintptr_t msk = SZREG - 1;
>> > -#if __riscv_misaligned_slow || __riscv_misaligned_fast
>> >   if (n < SZREG)
>> > -#else
>> > -  if (unlikely ((((uintptr_t)a & msk) != ((uintptr_t)b & msk)) || n <
>> > SZREG)) -#endif
>> >     {
>> >       if (__builtin_expect (a < end, 1))
>> >         __libc_memcpy_bytewise (a, b, n);
>> >       return aa;
>> >     }
>> >
>> > +/*
>> > + * If misaligned access is slow or prohibited, and the alignments of
>> > +the source
>> > + * and destination are different, we align the destination to do XLEN stores.
>> > + * This uses only one aligned store for every four (or eight for XLEN
>> > +== 64)
>> > + * bytes of data.
>> > + */
>> > +#if defined(__riscv_misaligned_slow) ||
>> > +defined(__riscv_misaligned_avoid)
>> > +  if (unlikely ((((uintptr_t)a & msk) != ((uintptr_t)b & msk))))
>> > +    {
>> > +      size_t dst_pad = (uintptr_t)a & msk;
>> > +      dst_pad = (SZREG - dst_pad) & msk;
>> > +      __libc_memcpy_bytewise (a, b, dst_pad);
>> > +      a += dst_pad;
>> > +      b += dst_pad;
>> > +
>> > +      uintxlen_t *la = (uintxlen_t *)a;
>> > +      const unsigned char *cb = (const unsigned char *)b;
>> > +      uintxlen_t *lend = (uintxlen_t *)((uintptr_t)end & ~msk);
>> > +
>> > +      while (la < lend)
>> > +        {
>> > +          *la++ = __libc_load_xlen (cb);
>> > +          cb += SZREG;
>> > +        }
>> > +      a = (unsigned char *)la;
>> > +      b = (const unsigned char *)cb;
>> > +      if (unlikely (a < end))
>> > +        __libc_memcpy_bytewise (a, b, end - a);
>> > +      return aa;
>> > +    }
>> > +#endif
>> > +
>> >   if (unlikely (((uintptr_t)a & msk) != 0))
>> >     {
>> > -      size_t rem = SZREG - ((uintptr_t)a & msk);
>> > -      __libc_memcpy_bytewise (a, b, rem);
>> > -      a += rem;
>> > -      b += rem;
>> > +      size_t pad = SZREG - ((uintptr_t)a & msk);
>> > +      __libc_memcpy_bytewise (a, b, pad);
>> > +      a += pad;
>> > +      b += pad;
>> >     }
>> >
>> >   uintxlen_t *la = (uintxlen_t *)a;
>> > @@ -87,9 +138,6 @@ memcpy (void *__restrict aa, const void *__restrict
>> > bb, size_t n)
>> >         }
>> >     }
>> >
>> > -  while (la < lend)
>> > -      *la++ = *lb++;
>> > -
>> >   a = (unsigned char *)la;
>> >   b = (const unsigned char *)lb;
>> >   if (unlikely (a < end))
>> > --
>> > 2.43.0
>> 
>> --
>> embedded brains GmbH & Co. KG
>> Herr Sebastian HUBER
>> Dornierstr. 4
>> 82178 Puchheim
>> Germany
>> email: sebastian.huber@embedded-brains.de
>> phone: +49-89-18 94 741 - 16
>> fax:   +49-89-18 94 741 - 08
>> 
>> Registergericht: Amtsgericht München
>> Registernummer: HRB 157899
>> Vertretungsberechtigte Geschäftsführer: Peter Rasmussen, Thomas Dörfler
>> Unsere Datenschutzerklärung finden Sie hier:
> > https://embedded-brains.de/datenschutzerklaerung/

-- 
embedded brains GmbH & Co. KG
Herr Sebastian HUBER
Dornierstr. 4
82178 Puchheim
Germany
email: sebastian.huber@embedded-brains.de
phone: +49-89-18 94 741 - 16
fax:   +49-89-18 94 741 - 08

Registergericht: Amtsgericht München
Registernummer: HRB 157899
Vertretungsberechtigte Geschäftsführer: Peter Rasmussen, Thomas Dörfler
Unsere Datenschutzerklärung finden Sie hier:
https://embedded-brains.de/datenschutzerklaerung/


More information about the Newlib mailing list