[PATCH v2 4/4] RISC-V: strcmp [speed optimized]: optimize mismatch logic for targets with Zb* extension support
puranikvinit
kvp933.vinit@gmail.com
Wed May 28 08:28:56 GMT 2025
Reworks the mismatch handling to use Zbb's ctz/clz instructions for
faster byte difference detection, significantly improving performance on
Zbb-capable targets. Non-Zbb targets retain the original logic for
compatibility.
Signed-off-by: puranikvinit <kvp933.vinit@gmail.com>
Reviewed-by: Christian Herber <christian.herber@oss.nxp.com>
---
newlib/libc/machine/riscv/strcmp.S | 157 +++++++++++++++++------------
1 file changed, 94 insertions(+), 63 deletions(-)
diff --git a/newlib/libc/machine/riscv/strcmp.S b/newlib/libc/machine/riscv/strcmp.S
index 42c870c10..0b1dfc4b1 100644
--- a/newlib/libc/machine/riscv/strcmp.S
+++ b/newlib/libc/machine/riscv/strcmp.S
@@ -99,78 +99,109 @@ strcmp:
.Lmismatch:
# words don't match, but a2 has no null byte.
+ #if __riscv_zbb
+ xor a4, a2, a3 # find differing bits
-#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+ # Check system endianness
+ # If little-endian, use Count Trailing Zeros (ctz)
+ # If big-endian, use Count Leading Zeros (clz)
+ # This helps identify the position of the first differing byte between a2 and a3.
-#if __riscv_xlen == 64
- sll a4, a2, 48
- sll a5, a3, 48
- bne a4, a5, .Lmismatch_upper
- sll a4, a2, 32
- sll a5, a3, 32
- bne a4, a5, .Lmismatch_upper
-#endif
- sll a4, a2, 16
- sll a5, a3, 16
- bne a4, a5, .Lmismatch_upper
-
- srl a4, a2, 8*SZREG-16
- srl a5, a3, 8*SZREG-16
- sub a0, a4, a5
- and a1, a0, 0xff
- bnez a1, .Lfinal_upper_diff
- ret
+ # For example, in little-endian, least significant byte comes first.
+ # So trailing zeros help find which byte position differs.
-.Lmismatch_upper:
- srl a4, a4, 8*SZREG-16
- srl a5, a5, 8*SZREG-16
- sub a0, a4, a5
- and a1, a0, 0xff
- bnez a1, .Lfinal_upper_diff
- ret
+ # In big-endian, most significant byte comes first, so leading zeros are used.
+ # The position will then be used to extract the differing byte.
-.Lfinal_upper_diff:
- and a4, a4, 0xff
- and a5, a5, 0xff
- sub a0, a4, a5
- ret
+ #if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+ ctz a5, a4
+ #else
+ clz a5, a4
+ #endif
-#else
+ andi a5, a5, -8 # find position of bit offset to the start of the byte where the first difference occurs
-#if __riscv_xlen == 64
- srl a4, a2, 48
- srl a5, a3, 48
- bne a4, a5, .Lmismatch_lower
- srl a4, a2, 32
- srl a5, a3, 32
- bne a4, a5, .Lmismatch_lower
-#endif
- srl a4, a2, 16
- srl a5, a3, 16
- bne a4, a5, .Lmismatch_lower
-
- srl a4, a2, 8
- srl a5, a3, 8
- bne a4, a5, .Lbyte_diff
- and a4, a2, 0xff
- and a5, a3, 0xff
-
-.Lbyte_diff:
- sub a0, a4, a5
- ret
-.Lmismatch_lower:
- srl a2, a4, 8
- srl a3, a5, 8
- bne a2, a3, .Lfinal_lower_diff
- and a2, a4, 0xff
- and a3, a5, 0xff
+ # Shift a2 and a3 right by a5 bits to bring the target byte to the LSB, and isolate the byte of interest
+ srl a2, a2, a5
+ and a2, a2, 0xff
-.Lfinal_lower_diff:
- sub a0, a2, a3
- ret
+ srl a3, a3, a5
+ and a3, a3, 0xff
-#endif
+
+ sub a0, a2, a3 # Calculate and return the difference in the isolated bytes
+ ret
+
+ #else
+ #if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+ #if __riscv_xlen == 64
+ sll a4, a2, 48
+ sll a5, a3, 48
+ bne a4, a5, .Lmismatch_upper
+ sll a4, a2, 32
+ sll a5, a3, 32
+ bne a4, a5, .Lmismatch_upper
+ #endif
+ sll a4, a2, 16
+ sll a5, a3, 16
+ bne a4, a5, .Lmismatch_upper
+
+ srl a4, a2, 8*SZREG-16
+ srl a5, a3, 8*SZREG-16
+ sub a0, a4, a5
+ and a1, a0, 0xff
+ bnez a1, .Lfinal_upper_diff
+ ret
+
+ .Lmismatch_upper:
+ srl a4, a4, 8*SZREG-16
+ srl a5, a5, 8*SZREG-16
+ sub a0, a4, a5
+ and a1, a0, 0xff
+ bnez a1, .Lfinal_upper_diff
+ ret
+
+ .Lfinal_upper_diff:
+ and a4, a4, 0xff
+ and a5, a5, 0xff
+ sub a0, a4, a5
+ ret
+ #else
+ #if __riscv_xlen == 64
+ srl a4, a2, 48
+ srl a5, a3, 48
+ bne a4, a5, .Lmismatch_lower
+ srl a4, a2, 32
+ srl a5, a3, 32
+ bne a4, a5, .Lmismatch_lower
+ #endif
+ srl a4, a2, 16
+ srl a5, a3, 16
+ bne a4, a5, .Lmismatch_lower
+
+ srl a4, a2, 8
+ srl a5, a3, 8
+ bne a4, a5, .Lbyte_diff
+ and a4, a2, 0xff
+ and a5, a3, 0xff
+
+ .Lbyte_diff:
+ sub a0, a4, a5
+ ret
+
+ .Lmismatch_lower:
+ srl a2, a4, 8
+ srl a3, a5, 8
+ bne a2, a3, .Lfinal_lower_diff
+ and a2, a4, 0xff
+ and a3, a5, 0xff
+
+ .Lfinal_lower_diff:
+ sub a0, a2, a3
+ ret
+ #endif
+ #endif
.Lmisaligned:
# misaligned
--
2.49.0
More information about the Newlib
mailing list