[PATCH] riscv: add RVV-optimized memchr

daichengrong daichengrong@iscas.ac.cn
Wed Apr 22 01:11:08 GMT 2026



On 3/27/26 10:35, Kito Cheng wrote:
>> @@ -0,0 +1,129 @@
>> +/* RISC-V RVV based memchr.
>> +   Copyright (C) 2026 Free Software Foundation, Inc.
>> +   This file is part of the GNU C Library.
>> +
>> +   The GNU C Library is free software; you can redistribute it and/or
>> +   modify it under the terms of the GNU Lesser General Public
>> +   License as published by the Free Software Foundation; either
>> +   version 2.1 of the License, or (at your option) any later version.
>> +
>> +   The GNU C Library is distributed in the hope that it will be useful,
>> +   but WITHOUT ANY WARRANTY; without even the implied warranty of
>> +   MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
>> +   Lesser General Public License for more details.
>> +
>> +   You should have received a copy of the GNU Lesser General Public
>> +   License along with the GNU C Library; if not, see
>> +   <https://www.gnu.org/licenses/>.  */
>> +
>> +#include <sysdep.h>
>> +#include <sys/asm.h>
>> +
>> +#ifndef MEMCHR
>> +# define MEMCHR __memchr
>> +#endif
>> +
>> +#define srcin          a0
>> +#define chrin          a1
>> +#define cntin          a2
>> +#define result         a0
>> +
>> +#define src                a0
>> +#define cntrem         a2
>> +#define cntrem_4    a5
>> +
>> +#define vdata          v0
>> +#define vmask          v16
>> +#define first_index a4
>> +
>> +#define        tmp         t0
>> +#define        vset_vl         t6
>> +
>> +#define src_2          a3
>> +#define src_3          a6
>> +#define src_4          a7
>> +
>> +#define vdata_1                v0
>> +#define vdata_2                v4
>> +#define vdata_3                v8
>> +#define vdata_4                v12
>> +
>> +#define vmask_1                v16
>> +#define vmask_2                v20
>> +#define vmask_3                v24
>> +#define vmask_4                v28
>> +
>> +#define first_index_1  a4
>> +#define first_index_2  t2
>> +#define first_index_3  t3
>> +#define first_index_4  t4
>> +
>> +ENTRY (MEMCHR)
>> +.option push
>> +.option arch, +v
>> +.option arch, +zba
> 
> ^^^ do we really need zba here? I don't see any instruction from zba
> in the implementation.
> 
Good catch, thanks for pointing that out.

The current implementation does not actually make use of any Zba-specific instructions, so the .option arch, +zba is unnecessary here. I'll drop it in the next revision to keep the requirements minimal and consistent with the code.
>> +    beqz        cntin, L(ret)
>> +    and         tmp, cntin, 0x3F
>> +    beqz        tmp, L(loop_pre)
>> +    vsetvli     vset_vl, tmp, e8, m8, ta, ma
>> +    vle8.v      vdata, (srcin)
>> +    vmseq.vx    vmask, vdata, chrin
>> +    vfirst.m    first_index, vmask
>> +    bgez        first_index, L(found)
>> +
>> +    beq         vset_vl, cntin, L(ret)
>> +
>> +    add         src, srcin, vset_vl
>> +    sub         cntrem, cntin, vset_vl
>> +
>> +L(loop_pre):
>> +    srli        cntrem_4, cntrem, 2
>> +L(loop):
>> +    vsetvli     vset_vl, cntrem_4, e8, m4, ta, ma
>> +
>> +    vle8.v      vdata_1, (src)
>> +    add         src_2, src, vset_vl
>> +    vle8.v      vdata_2, (src_2)
>> +    add         src_3, src_2, vset_vl
>> +    vle8.v      vdata_3, (src_3)
>> +    add         src_4, src_3, vset_vl
>> +    vle8.v      vdata_4, (src_4)
>> +
>> +    vmseq.vx    vmask_1, vdata_1, chrin
>> +    vmseq.vx    vmask_2, vdata_2, chrin
>> +    vmseq.vx    vmask_3, vdata_3, chrin
>> +    vmseq.vx    vmask_4, vdata_4, chrin
>> +
>> +    vfirst.m    first_index, vmask_1
>> +    vfirst.m    first_index_2, vmask_2
>> +    vfirst.m    first_index_3, vmask_3
>> +    vfirst.m    first_index_4, vmask_4
>> +
>> +    bgez        first_index_1, L(found1)
>> +    bgez        first_index_2, L(found2)
>> +    bgez        first_index_3, L(found3)
>> +    bgez        first_index_4, L(found4)
>> +
>> +    add         src, src_4, vset_vl
>> +    sub         cntrem_4, cntrem_4, vset_vl
>> +    bnez        cntrem_4, L(loop)
>> +L(ret):
>> +    li          result, 0
>> +    ret
>> +L(found4):
>> +    add result, src_4, first_index_4
>> +    ret
>> +L(found3):
>> +    add result, src_3, first_index_3
>> +    ret
>> +L(found2):
>> +    add result, src_2, first_index_2
>> +    ret
>> +L(found1):
>> +L(found):
>> +    add result, src, first_index
>> +    ret
>> +.option pop
>> +END (MEMCHR)
>> +weak_alias (MEMCHR, memchr)
>> +libc_hidden_builtin_def (memchr)



More information about the Libc-alpha mailing list