[PATCH v2 04/20] arm64: cmpxchg128: LSE: Remove redundant operands

Mark Rutland <[email protected]> Tue, 4 Aug 2026 18:04:47 +0100
Newsgroups gmane.linux.kernel.stable,gmane.linux.ports.arm.kernel
Message-ID <[email protected]>
The LSE assembly for cmpxhg128*() has redundant operands which cause
unnecessary register pressure. These operands can be removed without
adverse effects, as described below.

It is necessary to force specific registers for the 64-bit halves of
'old' and 'new', as the encoding of CASP[A][L] requires that these
halves are allocated into even/odd register pairs, and contemporary
versions of LLVM don't support a mechanism to allocate a value into an
even/odd register pair for inline assembly. We manually allocate these
into x0/x1/x2/x3.

It is not necessary to force the address into a specific register, as
the encoding of CASP[A][L] can accept this in any GPR or SP. Hence, it
is not necessary to manually allocate the address into x4 for the
'[ptr]' operand. The assembly uses the '[v]' operand, and contemporary
compilers happen to allocate '[v]' into a separate register from
'[ptr]', meaning that '[ptr]' only serves to create register pressure.

It is not necessary to allocate the '[oldval1]' and '[oldval2]'
operands, as these are not used by the assembly. The assembly uses the
'[old1]' and '[old2]' operands for both input and output. The
'[oldval1]' and '[oldval2]' operands only serve to create register
pressure.

Remove the redundant operands. This saves on register pressure, as
demonstrated by the test case below. There's still some unfortunate
register shuffling due to the manual allocation of 'old' and 'new', but
this should be less prominent within a larger function.

Test case:

| u128 outline_cmpxchg128(u128 *p, u128 old, u128 new)
| {
| 	return cmpxchg128(p, old, new);
| }

Generated code before this patch:

| <outline_cmpxchg128>:
|        mov     x6, x0
|        mov     x1, x3
|        mov     x0, x2
|        b       1f
|        mov     x2, x4
|        mov     x3, x5
|        mov     x4, x6
|        mov     x5, x0
|        mov     x7, x1
|        caspal  x0, x1, x2, x3, [x6]
|        ret
| 1:     prfm    pstl1strm, [x6]
| 2:     ldxp    x8, x7, [x6]
|        cmp     x8, x0
|        ccmp    x7, x3, #0x0, eq
|        b.ne    3f
|        stlxp   w2, x4, x5, [x6]
|        cbnz    w2, 2b
|        dmb     ish
| 3:     mov     x0, x8
|        mov     x1, x7
|        ret

Generated code after this patch:

| <outline_cmpxchg128>:
|        mov     x6, x0
|        mov     x0, x2
|        b       1f
|        mov     x1, x3
|        mov     x2, x4
|        mov     x3, x5
|        caspal  x0, x1, x2, x3, [x6]
|        ret
| 1:     prfm    pstl1strm, [x6]
| 2:     ldxp    x8, x7, [x6]
|        cmp     x8, x0
|        ccmp    x7, x3, #0x0, eq
|        3f
|        stlxp   w2, x4, x5, [x6]
|        cbnz    w2, 2b
|        dmb     ish
| 3:     mov     x0, x8
|        mov     x1, x7
|        ret

Signed-off-by: Mark Rutland <[email protected]>
Cc: Ada Couprie Diaz <[email protected]>
Cc: Ard Biesheuvel <[email protected]>
Cc: Catalin Marinas <[email protected]>
Cc: James Morse <[email protected]>
Cc: Jinjie Ruan <[email protected]>
Cc: Marc Zyngier <[email protected]>
Cc: Peter Zijlstra <[email protected]>
Cc: Vladimir Murzin <[email protected]>
Cc: Will Deacon <[email protected]>
Cc: Yang Shi <[email protected]>
---
 arch/arm64/include/asm/atomic_lse.h | 4 +---
 1 file changed, 1 insertion(+), 3 deletions(-)

diff --git a/arch/arm64/include/asm/atomic_lse.h b/arch/arm64/include/asm/atomic_lse.h
index afad1849c4cf5..d588af0565331 100644
--- a/arch/arm64/include/asm/atomic_lse.h
+++ b/arch/arm64/include/asm/atomic_lse.h
@@ -291,15 +291,13 @@ __lse__cmpxchg128##name(volatile u128 *ptr, u128 old, u128 new)		\
 	register unsigned long x1 asm ("x1") = o.high;			\
 	register unsigned long x2 asm ("x2") = n.low;			\
 	register unsigned long x3 asm ("x3") = n.high;			\
-	register unsigned long x4 asm ("x4") = (unsigned long)ptr;	\
 									\
 	asm volatile(							\
 	__LSE_PREAMBLE							\
 	"	casp" #mb "\t%[old1], %[old2], %[new1], %[new2], %[v]\n"\
 	: [old1] "+&r" (x0), [old2] "+&r" (x1),				\
 	  [v] "+Q" (*(u128 *)ptr)					\
-	: [new1] "r" (x2), [new2] "r" (x3), [ptr] "r" (x4),		\
-	  [oldval1] "r" (o.low), [oldval2] "r" (o.high)			\
+	: [new1] "r" (x2), [new2] "r" (x3)				\
 	: cl);								\
 									\
 	r.low = x0; r.high = x1;					\
-- 
2.30.2