[PATCH v1 2/2] riscv: optmize ucopy for has_fast_unaligned_accesses

Andy Chiu tchiu at tenstorrent.com
Fri Sep 11 11:24:00 PDT 2026


Skipping the software aligning code increases the ucopy bandwidth by up
to 11.4%. Commit b94cec5761d2 ("riscv: skip software algning code for
HAVE_EFFICIENT_UNALIGNED_ACCESS") provides this optimization at compile
time, which depends on NONPORTABLE. This patch extends it such that the
optimization can be enabled at runtime, after the unaligned access speed
is resolved.

To do so, we utilize the assembly version of static key and connect it
to fast_unaligned_access_speed_key. The kernel jumps to the software
aligning code as the default behavior, then the jump is optimized to an
nop if the platform supports native misaligned access.

Signed-off-by: Andy Chiu <tchiu at tenstorrent.com>
---
 arch/riscv/lib/uaccess.S | 61 ++++++++++++++++++++++------------------
 1 file changed, 34 insertions(+), 27 deletions(-)

diff --git a/arch/riscv/lib/uaccess.S b/arch/riscv/lib/uaccess.S
index cf8586a937de..ad1457e789fb 100644
--- a/arch/riscv/lib/uaccess.S
+++ b/arch/riscv/lib/uaccess.S
@@ -5,6 +5,7 @@
 #include <asm/csr.h>
 #include <asm/hwcap.h>
 #include <asm/alternative-macros.h>
+#include <asm/jump_label.h>
 
 	.macro fixup op reg addr lbl
 100:
@@ -77,33 +78,11 @@ SYM_FUNC_START(fallback_scalar_usercopy_sum_enabled)
 	bltu	a2, a3, .Lbyte_copy_tail
 
 #if !defined(CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS)
-	/*
-	 * Copy first bytes until dst is aligned to word boundary.
-	 * a0 - start of dst
-	 * t1 - start of aligned dst
-	 */
-	addi	t1, a0, SZREG-1
-	andi	t1, t1, ~(SZREG-1)
-	/* dst is already aligned, skip */
-	beq	a0, t1, .Lskip_align_dst
-1:
-	/* a5 - one byte for copying data */
-	fixup lb      a5, 0(a1), 10f
-	addi	a1, a1, 1	/* src */
-	fixup sb      a5, 0(a0), 10f
-	addi	a0, a0, 1	/* dst */
-	bltu	a0, t1, 1b	/* t1 - start of aligned dst */
-
-.Lskip_align_dst:
-	/*
-	 * Now dst is aligned.
-	 * Use shift-copy if src is misaligned.
-	 * Use word-copy if both src and dst are aligned because
-	 * can not use shift-copy which do not require shifting
-	 */
-	/* a1 - start of src */
-	andi	a3, a1, SZREG-1
-	bnez	a3, .Lshift_copy
+#if defined(CONFIG_RISCV_PROBE_UNALIGNED_ACCESS) && defined(CONFIG_JUMP_LABEL)
+	ARCH_STATIC_BRANCH_JUMP_ASM(fast_unaligned_access_speed_key + 1, .Lalign_dst)
+#else
+	j	.Lalign_dst
+#endif
 #endif
 .Lword_copy:
         /*
@@ -139,6 +118,34 @@ SYM_FUNC_START(fallback_scalar_usercopy_sum_enabled)
 	j	.Lbyte_copy_tail
 
 #if !defined(CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS)
+.Lalign_dst:
+	/*
+	 * Copy first bytes until dst is aligned to word boundary.
+	 * a0 - start of dst
+	 * t1 - start of aligned dst
+	 */
+	addi	t1, a0, SZREG-1
+	andi	t1, t1, ~(SZREG-1)
+	/* dst is already aligned, skip */
+	beq	a0, t1, .Lskip_align_dst
+1:
+	/* a5 - one byte for copying data */
+	fixup lb      a5, 0(a1), 10f
+	addi	a1, a1, 1	/* src */
+	fixup sb      a5, 0(a0), 10f
+	addi	a0, a0, 1	/* dst */
+	bltu	a0, t1, 1b	/* t1 - start of aligned dst */
+
+.Lskip_align_dst:
+	/*
+	 * Now dst is aligned.
+	 * Use shift-copy if src is misaligned.
+	 * Use word-copy if both src and dst are aligned because
+	 * can not use shift-copy which do not require shifting
+	 */
+	/* a1 - start of src */
+	andi	a3, a1, SZREG-1
+	beqz	a3, .Lword_copy
 .Lshift_copy:
 
 	/*
-- 
2.43.0




More information about the linux-riscv mailing list