[PATCH v2 19/20] arm64: percpu: Implement preemptible CMPXCHG128 ops
Mark Rutland <[email protected]> Tue, 4 Aug 2026 18:05:02 +0100
| Newsgroups | gmane.linux.kernel.stable,gmane.linux.ports.arm.kernel |
|---|---|
| Message-ID | <[email protected]> |
Use the PCPU GPR infrastructure to implement this_cpu_cmpxchg128().
Note that before this patch, the LL/SC and LSE implementation was chosen
with an alternative branch. After this patch the implementations are
patched inline, matching the style of the other percpu ops.
The sub-optimal register shuffling before and after this patch occurs
because we hard-code specific registers, as LLVM (currently) lacks a way
to place a 128-bit type in a compiler-allocated even-odd pair of
registers for inline assembly. We should be able to improve that in
future, given compiler support.
Test case:
| u128 outline_this_cpu_cmpxchg128(u128 __percpu *p, u128 old, u128 new)
| {
| return this_cpu_cmpxchg128(*p, old, new);
| }
Generated code before this patch (v7.2-rc4):
| <outline_this_cpu_cmpxchg128>:
| paciasp
| stp x29, x30, [sp, #-32]!
| mov x6, x0
| mov x0, x2
| mov x29, sp
| mov x1, x3
| mrs x2, sp_el0
| ldr w7, [x2, #8]
| add w7, w7, #0x1
| str w7, [x2, #8]
| mrs x2, tpidr_el1
| add x6, x6, x2
| b 4f
| mov x2, x4
| mov x3, x5
| mov x4, x6
| mov x5, x0
| mov x7, x1
| casp x0, x1, x2, x3, [x6]
| 1: mrs x3, sp_el0
| ldr x2, [x3, #8]
| sub x2, x2, #0x1
| str w2, [x3, #8]
| cbz x2, 2f
| ldr x2, [x3, #8]
| cbnz x2, 3f
| 2: stp x0, x1, [sp, #16]
| bl preempt_schedule_notrace
| ldp x0, x1, [sp, #16]
| 3: ldp x29, x30, [sp], #32
| autiasp
| ret
| 4: prfm pstl1strm, [x6]
| 5: ldxp x8, x7, [x6]
| cmp x8, x0
| ccmp x7, x3, #0x0, eq
| b.ne 6f
| stxp w2, x4, x5, [x6]
| cbnz w2, 5b
| 6: mov x0, x8
| mov x1, x7
| b 1b
Generated code after this patch:
| <outline_this_cpu_cmpxchg128>:
| mov x6, x0
| mov x1, x3
| mov x0, x2
| mov x3, x5
| mrs x7, sp_el0
| mov x2, x4
| mov w5, #0x10a6
| strh w5, [x7, #20]
| mrs x5, tpidr_el1
| add x4, x6, x5
| prfm pstl1strm, [x4]
| 1: ldxp x9, x10, [x4]
| cmp x9, x0
| ccmp x10, x1, #0x0, eq
| b.ne 2f
| stxp w8, x2, x3, [x4]
| cbnz w8, 1b
| 2: strh wzr, [x7, #20]
| mov x0, x9
| mov x1, x10
| ret
Signed-off-by: Mark Rutland <[email protected]>
Cc: Ada Couprie Diaz <[email protected]>
Cc: Ard Biesheuvel <[email protected]>
Cc: Catalin Marinas <[email protected]>
Cc: James Morse <[email protected]>
Cc: Jinjie Ruan <[email protected]>
Cc: Marc Zyngier <[email protected]>
Cc: Peter Zijlstra <[email protected]>
Cc: Vladimir Murzin <[email protected]>
Cc: Will Deacon <[email protected]>
Cc: Yang Shi <[email protected]>
---
arch/arm64/include/asm/percpu.h | 70 +++++++++++++++++++++++++++------
1 file changed, 57 insertions(+), 13 deletions(-)
diff --git a/arch/arm64/include/asm/percpu.h b/arch/arm64/include/asm/percpu.h
index af1d4ca5c85ef..8087ed6e12241 100644
--- a/arch/arm64/include/asm/percpu.h
+++ b/arch/arm64/include/asm/percpu.h
@@ -490,19 +490,63 @@ PERCPU_CMPXCHG_OP(x, , 64)
#define this_cpu_cmpxchg64(pcp, o, n) this_cpu_cmpxchg_8(pcp, o, n)
-#define this_cpu_cmpxchg128(pcp, o, n) \
-({ \
- typedef typeof(pcp) pcp_op_T__; \
- u128 old__, new__, ret__; \
- pcp_op_T__ *ptr__; \
- old__ = o; \
- new__ = n; \
- preempt_disable_notrace(); \
- ptr__ = raw_cpu_ptr(&(pcp)); \
- ret__ = cmpxchg128_local((void *)ptr__, old__, new__); \
- preempt_enable_notrace(); \
- ret__; \
-})
+static inline u128
+__percpu_cmpxchg_128(void __percpu *pcp, u128 old, u128 new)
+{
+ u16 *gprs = ¤t_thread_info()->pcpu_gprs;
+ unsigned long addr;
+ unsigned long off;
+ union __u128_halves r, o = { .full = (old) },
+ n = { .full = (new) };
+ register unsigned long ol asm ("x0") = o.low;
+ register unsigned long oh asm ("x1") = o.high;
+ register unsigned long nl asm ("x2") = n.low;
+ register unsigned long nh asm ("x3") = n.high;
+ unsigned long rl, rh;
+ unsigned long tmp;
+
+ asm volatile (
+ __PCPU_GPRS_BEGIN("%[gprs]", "%[pcp]", "%[off]", "%[addr]")
+ ARM64_LSE_ATOMIC_INSN(
+ /* LL/SC */
+ " prfm pstl1strm, [%[addr]]\n"
+ "1: ldxp %[rl], %[rh], [%[addr]]\n"
+ " cmp %[rl], %[ol]\n"
+ " ccmp %[rh], %[oh], 0, eq\n"
+ " b.ne 2f\n"
+ " stxp %w[tmp], %[nl], %[nh], [%[addr]]\n"
+ " cbnz %w[tmp], 1b\n"
+ "2:\n"
+ ,
+ /* LSE atomics */
+ " casp %[ol], %[oh], %[nl], %[nh], [%[addr]]\n"
+ " mov %[rl], %[ol]\n"
+ " mov %[rh], %[oh]\n"
+ __nops(4)
+ )
+ __PCPU_GPRS_END("%[gprs]")
+ : [gprs] "=Qo" (*gprs),
+ [addr] "=&r" (addr),
+ [off] "=&r" (off),
+ [tmp] "=&r" (tmp),
+ [ol] "+&r" (ol),
+ [oh] "+&r" (oh),
+ [rl] "=&r" (rl),
+ [rh] "=&r" (rh)
+ : [pcp] "r" (pcp),
+ [nl] "r" (nl),
+ [nh] "r" (nh)
+ : "memory", "cc"
+ );
+
+ r.low = rl;
+ r.high = rh;
+
+ return r.full;
+}
+
+#define this_cpu_cmpxchg128(pcp, o, n) \
+ _pcp_wrap_return(__percpu_cmpxchg_128, pcp, o, n)
#ifdef __KVM_NVHE_HYPERVISOR__
extern unsigned long __hyp_per_cpu_offset(unsigned int cpu);
--
2.30.2