[PATCH v2 04/20] arm64: cmpxchg128: LSE: Remove redundant operands

Mark Rutland mark.rutland at arm.com
Tue Aug 4 10:04:47 PDT 2026


The LSE assembly for cmpxhg128*() has redundant operands which cause
unnecessary register pressure. These operands can be removed without
adverse effects, as described below.

It is necessary to force specific registers for the 64-bit halves of
'old' and 'new', as the encoding of CASP[A][L] requires that these
halves are allocated into even/odd register pairs, and contemporary
versions of LLVM don't support a mechanism to allocate a value into an
even/odd register pair for inline assembly. We manually allocate these
into x0/x1/x2/x3.

It is not necessary to force the address into a specific register, as
the encoding of CASP[A][L] can accept this in any GPR or SP. Hence, it
is not necessary to manually allocate the address into x4 for the
'[ptr]' operand. The assembly uses the '[v]' operand, and contemporary
compilers happen to allocate '[v]' into a separate register from
'[ptr]', meaning that '[ptr]' only serves to create register pressure.

It is not necessary to allocate the '[oldval1]' and '[oldval2]'
operands, as these are not used by the assembly. The assembly uses the
'[old1]' and '[old2]' operands for both input and output. The
'[oldval1]' and '[oldval2]' operands only serve to create register
pressure.

Remove the redundant operands. This saves on register pressure, as
demonstrated by the test case below. There's still some unfortunate
register shuffling due to the manual allocation of 'old' and 'new', but
this should be less prominent within a larger function.

Test case:

| u128 outline_cmpxchg128(u128 *p, u128 old, u128 new)
| {
| 	return cmpxchg128(p, old, new);
| }

Generated code before this patch:

| <outline_cmpxchg128>:
|        mov     x6, x0
|        mov     x1, x3
|        mov     x0, x2
|        b       1f
|        mov     x2, x4
|        mov     x3, x5
|        mov     x4, x6
|        mov     x5, x0
|        mov     x7, x1
|        caspal  x0, x1, x2, x3, [x6]
|        ret
| 1:     prfm    pstl1strm, [x6]
| 2:     ldxp    x8, x7, [x6]
|        cmp     x8, x0
|        ccmp    x7, x3, #0x0, eq
|        b.ne    3f
|        stlxp   w2, x4, x5, [x6]
|        cbnz    w2, 2b
|        dmb     ish
| 3:     mov     x0, x8
|        mov     x1, x7
|        ret

Generated code after this patch:

| <outline_cmpxchg128>:
|        mov     x6, x0
|        mov     x0, x2
|        b       1f
|        mov     x1, x3
|        mov     x2, x4
|        mov     x3, x5
|        caspal  x0, x1, x2, x3, [x6]
|        ret
| 1:     prfm    pstl1strm, [x6]
| 2:     ldxp    x8, x7, [x6]
|        cmp     x8, x0
|        ccmp    x7, x3, #0x0, eq
|        3f
|        stlxp   w2, x4, x5, [x6]
|        cbnz    w2, 2b
|        dmb     ish
| 3:     mov     x0, x8
|        mov     x1, x7
|        ret

Signed-off-by: Mark Rutland <mark.rutland at arm.com>
Cc: Ada Couprie Diaz <ada.coupriediaz at arm.com>
Cc: Ard Biesheuvel <ardb at kernel.org>
Cc: Catalin Marinas <catalin.marinas at arm.com>
Cc: James Morse <james.morse at arm.com>
Cc: Jinjie Ruan <ruanjinjie at huawei.com>
Cc: Marc Zyngier <maz at kernel.org>
Cc: Peter Zijlstra <peterz at infradead.org>
Cc: Vladimir Murzin <vladimir.murzin at arm.com>
Cc: Will Deacon <will at kernel.org>
Cc: Yang Shi <yang at os.amperecomputing.com>
---
 arch/arm64/include/asm/atomic_lse.h | 4 +---
 1 file changed, 1 insertion(+), 3 deletions(-)

diff --git a/arch/arm64/include/asm/atomic_lse.h b/arch/arm64/include/asm/atomic_lse.h
index afad1849c4cf5..d588af0565331 100644
--- a/arch/arm64/include/asm/atomic_lse.h
+++ b/arch/arm64/include/asm/atomic_lse.h
@@ -291,15 +291,13 @@ __lse__cmpxchg128##name(volatile u128 *ptr, u128 old, u128 new)		\
 	register unsigned long x1 asm ("x1") = o.high;			\
 	register unsigned long x2 asm ("x2") = n.low;			\
 	register unsigned long x3 asm ("x3") = n.high;			\
-	register unsigned long x4 asm ("x4") = (unsigned long)ptr;	\
 									\
 	asm volatile(							\
 	__LSE_PREAMBLE							\
 	"	casp" #mb "\t%[old1], %[old2], %[new1], %[new2], %[v]\n"\
 	: [old1] "+&r" (x0), [old2] "+&r" (x1),				\
 	  [v] "+Q" (*(u128 *)ptr)					\
-	: [new1] "r" (x2), [new2] "r" (x3), [ptr] "r" (x4),		\
-	  [oldval1] "r" (o.low), [oldval2] "r" (o.high)			\
+	: [new1] "r" (x2), [new2] "r" (x3)				\
 	: cl);								\
 									\
 	r.low = x0; r.high = x1;					\
-- 
2.30.2




More information about the linux-arm-kernel mailing list