[PATCH] riscv: add RVV-optimized memchr
Kito Cheng
kito.cheng@gmail.com
Fri Mar 27 02:35:55 GMT 2026
> @@ -0,0 +1,129 @@
> +/* RISC-V RVV based memchr.
> + Copyright (C) 2026 Free Software Foundation, Inc.
> + This file is part of the GNU C Library.
> +
> + The GNU C Library is free software; you can redistribute it and/or
> + modify it under the terms of the GNU Lesser General Public
> + License as published by the Free Software Foundation; either
> + version 2.1 of the License, or (at your option) any later version.
> +
> + The GNU C Library is distributed in the hope that it will be useful,
> + but WITHOUT ANY WARRANTY; without even the implied warranty of
> + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
> + Lesser General Public License for more details.
> +
> + You should have received a copy of the GNU Lesser General Public
> + License along with the GNU C Library; if not, see
> + <https://www.gnu.org/licenses/>. */
> +
> +#include <sysdep.h>
> +#include <sys/asm.h>
> +
> +#ifndef MEMCHR
> +# define MEMCHR __memchr
> +#endif
> +
> +#define srcin a0
> +#define chrin a1
> +#define cntin a2
> +#define result a0
> +
> +#define src a0
> +#define cntrem a2
> +#define cntrem_4 a5
> +
> +#define vdata v0
> +#define vmask v16
> +#define first_index a4
> +
> +#define tmp t0
> +#define vset_vl t6
> +
> +#define src_2 a3
> +#define src_3 a6
> +#define src_4 a7
> +
> +#define vdata_1 v0
> +#define vdata_2 v4
> +#define vdata_3 v8
> +#define vdata_4 v12
> +
> +#define vmask_1 v16
> +#define vmask_2 v20
> +#define vmask_3 v24
> +#define vmask_4 v28
> +
> +#define first_index_1 a4
> +#define first_index_2 t2
> +#define first_index_3 t3
> +#define first_index_4 t4
> +
> +ENTRY (MEMCHR)
> +.option push
> +.option arch, +v
> +.option arch, +zba
^^^ do we really need zba here? I don't see any instruction from zba
in the implementation.
> + beqz cntin, L(ret)
> + and tmp, cntin, 0x3F
> + beqz tmp, L(loop_pre)
> + vsetvli vset_vl, tmp, e8, m8, ta, ma
> + vle8.v vdata, (srcin)
> + vmseq.vx vmask, vdata, chrin
> + vfirst.m first_index, vmask
> + bgez first_index, L(found)
> +
> + beq vset_vl, cntin, L(ret)
> +
> + add src, srcin, vset_vl
> + sub cntrem, cntin, vset_vl
> +
> +L(loop_pre):
> + srli cntrem_4, cntrem, 2
> +L(loop):
> + vsetvli vset_vl, cntrem_4, e8, m4, ta, ma
> +
> + vle8.v vdata_1, (src)
> + add src_2, src, vset_vl
> + vle8.v vdata_2, (src_2)
> + add src_3, src_2, vset_vl
> + vle8.v vdata_3, (src_3)
> + add src_4, src_3, vset_vl
> + vle8.v vdata_4, (src_4)
> +
> + vmseq.vx vmask_1, vdata_1, chrin
> + vmseq.vx vmask_2, vdata_2, chrin
> + vmseq.vx vmask_3, vdata_3, chrin
> + vmseq.vx vmask_4, vdata_4, chrin
> +
> + vfirst.m first_index, vmask_1
> + vfirst.m first_index_2, vmask_2
> + vfirst.m first_index_3, vmask_3
> + vfirst.m first_index_4, vmask_4
> +
> + bgez first_index_1, L(found1)
> + bgez first_index_2, L(found2)
> + bgez first_index_3, L(found3)
> + bgez first_index_4, L(found4)
> +
> + add src, src_4, vset_vl
> + sub cntrem_4, cntrem_4, vset_vl
> + bnez cntrem_4, L(loop)
> +L(ret):
> + li result, 0
> + ret
> +L(found4):
> + add result, src_4, first_index_4
> + ret
> +L(found3):
> + add result, src_3, first_index_3
> + ret
> +L(found2):
> + add result, src_2, first_index_2
> + ret
> +L(found1):
> +L(found):
> + add result, src, first_index
> + ret
> +.option pop
> +END (MEMCHR)
> +weak_alias (MEMCHR, memchr)
> +libc_hidden_builtin_def (memchr)
More information about the Libc-alpha
mailing list