[PATCH] Vectorise special cases for SVE log1p(f)

Adhemerval Zanella Netto adhemerval.zanella@linaro.org
Thu May 21 12:16:31 GMT 2026



On 18/05/26 13:20, Thomas Daubney wrote:
> This patch adds vectorised special cases for the SVE log functions
> log1p and log1pf.
> 
> When built with GCC-15 and executed on a Neoverse V2 platform, the
> following benchmarking throughput uplifts were measured:
> 
> log1pf -> 285% speed-up (4.85 ns/element to 1.26 ns/element)
> log1p  -> 117% speed-up (8.25 ns/element to 3.80 ns/element)
> 
> Note that the numbers here are for the special case path only and that
> the fast path performance has been maintained. These changes have also
> maintained the same level of accuracy as before.

LGTM, thanks.

Reviewed-by: Adhemerval Zanella  <adhemerval.zanella@linaro.org>

> ---
>  sysdeps/aarch64/fpu/log1p_sve.c        | 12 ++++++++----
>  sysdeps/aarch64/fpu/log1pf_sve.c       | 23 ++++++++++++++++-------
>  sysdeps/aarch64/fpu/sv_log1pf_inline.h |  5 ++---
>  3 files changed, 26 insertions(+), 14 deletions(-)
> 
> diff --git a/sysdeps/aarch64/fpu/log1p_sve.c b/sysdeps/aarch64/fpu/log1p_sve.c
> index bde0e8a7d1..331cfb34d3 100644
> --- a/sysdeps/aarch64/fpu/log1p_sve.c
> +++ b/sysdeps/aarch64/fpu/log1p_sve.c
> @@ -63,9 +63,13 @@ static const struct data
>  #define BottomMask 0xffffffff
>  
>  static svfloat64_t NOINLINE
> -special_case (svfloat64_t x, svfloat64_t y, svbool_t special)
> +special_case (svfloat64_t x, svfloat64_t y, svbool_t special, svbool_t pg)
>  {
> -  return sv_call_f64 (log1p, x, y, special);
> +  y = svsel (special, sv_f64 (NAN), y);
> +  svbool_t ret_pinf = svcmpeq (pg, x, INFINITY);
> +  svbool_t ret_minf = svcmpeq (pg, x, -1.0);
> +  y = svsel (ret_pinf, sv_f64 (INFINITY), y);
> +  return svsel (ret_minf, sv_f64 (-INFINITY), y);
>  }
>  
>  /* Vector approximation for log1p using polynomial on reduced interval. Maximum
> @@ -159,8 +163,8 @@ svfloat64_t SV_NAME_D1 (log1p) (svfloat64_t x, svbool_t pg)
>    if (__glibc_unlikely (svptest_any (pg, special)))
>      return special_case (
>  	x, svmla_x (svptrue_b64 (), svadd_x (svptrue_b64 (), ylo, yhi), f2, p),
> -	special);
> +	special, pg);
>    return svmla_x (svptrue_b64 (), svadd_x (svptrue_b64 (), ylo, yhi), f2, p);
>  }
>  
> -strong_alias (SV_NAME_D1 (log1p), SV_NAME_D1 (logp1))
> +strong_alias (SV_NAME_D1 (log1p), SV_NAME_D1 (logp1))
> \ No newline at end of file
> diff --git a/sysdeps/aarch64/fpu/log1pf_sve.c b/sysdeps/aarch64/fpu/log1pf_sve.c
> index f4d409eb50..297f5374ca 100644
> --- a/sysdeps/aarch64/fpu/log1pf_sve.c
> +++ b/sysdeps/aarch64/fpu/log1pf_sve.c
> @@ -17,14 +17,21 @@
>     License along with the GNU C Library; if not, see
>     <https://www.gnu.org/licenses/>.  */
>  
> +#include "../../ieee754/flt-32/math_config.h"
>  #include "sv_math.h"
>  #include "sv_log1pf_inline.h"
>  
>  static svfloat32_t NOINLINE
> -special_case (svfloat32_t x, svbool_t special)
> +special_case (svfloat32_t x, svfloat32_t y, svbool_t pg, svbool_t special,
> +	      const struct sv_log1pf_data *d)
>  {
> -  return sv_call_f32 (log1pf, x, sv_log1pf_inline (x, svptrue_b32 ()),
> -		      special);
> +  y = svsel_f32 (special, svreinterpret_f32 (sv_u32 (d->nan)), y);
> +
> +  svbool_t ret_pinf = svcmpeq (pg, x, asfloat (d->inf));
> +  svbool_t ret_minf = svcmpeq (pg, x, -1.0f);
> +
> +  y = svsel_f32 (ret_pinf, svreinterpret_f32 (sv_u32 (d->inf)), y);
> +  return svsel_f32 (ret_minf, sv_f32 (-d->inf), y);
>  }
>  
>  /* Vector log1pf approximation using polynomial on reduced interval. Worst-case
> @@ -33,14 +40,16 @@ special_case (svfloat32_t x, svbool_t special)
>  				 want 0x1.9f323ep-2.  */
>  svfloat32_t SV_NAME_F1 (log1p) (svfloat32_t x, svbool_t pg)
>  {
> +
> +  const struct sv_log1pf_data *d = ptr_barrier (&sv_log1pf_data);
>    /* x < -1, Inf/Nan.  */
> -  svbool_t special = svcmpeq (pg, svreinterpret_u32 (x), 0x7f800000);
> -  special = svorn_z (pg, special, svcmpge (pg, x, -1));
> +  svbool_t special = svcmpeq (pg, svreinterpret_u32 (x), d->inf);
> +  special = svorn_z (pg, special, svcmpge (pg, x, -1.0f));
>  
>    if (__glibc_unlikely (svptest_any (pg, special)))
> -    return special_case (x, special);
> +    return special_case (x, sv_log1pf_inline (x, pg), pg, special, d);
>  
>    return sv_log1pf_inline (x, pg);
>  }
>  
> -strong_alias (SV_NAME_F1 (log1p), SV_NAME_F1 (logp1))
> +strong_alias (SV_NAME_F1 (log1p), SV_NAME_F1 (logp1))
> \ No newline at end of file
> diff --git a/sysdeps/aarch64/fpu/sv_log1pf_inline.h b/sysdeps/aarch64/fpu/sv_log1pf_inline.h
> index 494fa279f8..3677578c59 100644
> --- a/sysdeps/aarch64/fpu/sv_log1pf_inline.h
> +++ b/sysdeps/aarch64/fpu/sv_log1pf_inline.h
> @@ -21,7 +21,6 @@
>  #define AARCH64_FPU_SV_LOG1PF_INLINE_H
>  
>  #include "sv_math.h"
> -#include "vecmath_config.h"
>  
>  #define SignExponentMask 0xff800000
>  
> @@ -31,6 +30,7 @@ static const struct sv_log1pf_data
>    float c1, c3, c5, c7;
>    float ln2, exp_bias, quarter;
>    uint32_t four, three_quarters;
> +  uint32_t inf, nan;
>  } sv_log1pf_data = {
>    /* Do not store first term of polynomial, which is -0.5, as
>       this can be fmov-ed directly instead of including it in
> @@ -39,7 +39,7 @@ static const struct sv_log1pf_data
>    .c3 = -0x1.54ef78p-3f,	.c4 = 0x1.28a1f4p-3f,  .c5 = -0x1.0da91p-3f,
>    .c6 = 0x1.abcb6p-4f,		.c7 = -0x1.6f0d5ep-5f, .ln2 = 0x1.62e43p-1f,
>    .exp_bias = 0x1p-23f,		.quarter = 0x1p-2f,    .four = 0x40800000,
> -  .three_quarters = 0x3f400000,
> +  .three_quarters = 0x3f400000, .inf = 0x7f800000,     .nan = 0x7fc00000,
>  };
>  
>  static inline svfloat32_t
> @@ -93,5 +93,4 @@ sv_log1pf_inline (svfloat32_t x, svbool_t pg)
>    svfloat32_t scale_back = svmul_lane_f32 (svcvt_f32_x (pg, k), fconst, 1);
>    return svmla_lane_f32 (p, scale_back, fconst, 0);
>  }
> -
>  #endif



More information about the Libc-alpha mailing list