[PATCH v2 18/20] arm64: percpu: Implement preemptible CMPXCHG ops

Mark Rutland <[email protected]>
Newsgroups org.infradead.lists.linux-arm-kernel,org.kernel.vger.stable
Message-ID <[email protected]>
Use the PCPU GPR infrastructure to implement all of the {8,16,32,64}-bit
cmpxchg ops.

Note that before this patch, the LL/SC and LSE implementation was chosen
with an alternative branch. After this patch the implementations are
patched inline, matching the style of the other percpu ops.

Test case:

| u64 outline_this_cpu_cmpxchg_u64(u64 __percpu *p, u64 o, u64 n)
| {
| 	return this_cpu_cmpxchg(*p, o, n);
| }

Generated code before this patch (v7.2-rc4):

| <outline_this_cpu_cmpxchg_u64>:
|        paciasp
|        stp     x29, x30, [sp, #-32]!
|        mrs     x4, sp_el0
|        mov     x29, sp
|        ldr     w3, [x4, #8]
|        add     w3, w3, #0x1
|        str     w3, [x4, #8]
|        mrs     x3, tpidr_el1
|        add     x3, x0, x3
|        b       4f		// alternative branch
|        cas     x1, x2, [x3]
|        mov     x0, x1
| 1:     mrs     x2, sp_el0
|        ldr     x1, [x2, #8]
|        sub     x1, x1, #0x1
|        str     w1, [x2, #8]
|        cbz     x1, 2f
|        ldr     x1, [x2, #8]
|        cbnz    x1, 3f
| 2:     str     x0, [sp, #24]
|        bl      preempt_schedule_notrace
|        ldr     x0, [sp, #24]
| 3:     ldp     x29, x30, [sp], #32
|        autiasp
|        ret
| 4:     prfm    pstl1strm, [x3]
| 5:     ldxr    x0, [x3]
|        eor     x4, x0, x1
|        cbnz    x4, 6f
|        stxr    w4, x2, [x3]
|        cbnz    w4, 5b
| 6:     b       1b

Generated code after this patch:

| <outline_this_cpu_cmpxchg_u64>:
|        mrs     x4, sp_el0
|        mov     w6, #0x14c0
|        strh    w6, [x4, #20]
|        mrs     x6, tpidr_el1
|        add     x5, x0, x6
|        prfm    pstl1strm, [x5]
| 1:     ldxr    x3, [x5]
|        eor     x7, x3, x1
|        cbnz    x7, 2f
|        stxr    w7, x2, [x5]
|        cbnz    w7, 1b
| 2:     strh    wzr, [x4, #20]
|        mov     x0, x3
|        ret

Signed-off-by: Mark Rutland <[email protected]>
Cc: Ada Couprie Diaz <[email protected]>
Cc: Ard Biesheuvel <[email protected]>
Cc: Catalin Marinas <[email protected]>
Cc: James Morse <[email protected]>
Cc: Jinjie Ruan <[email protected]>
Cc: Marc Zyngier <[email protected]>
Cc: Peter Zijlstra <[email protected]>
Cc: Vladimir Murzin <[email protected]>
Cc: Will Deacon <[email protected]>
Cc: Yang Shi <[email protected]>

WIP: improve extension in this_cpu_cmpxchg*()

Signed-off-by: Mark Rutland <[email protected]>
---
 arch/arm64/include/asm/percpu.h | 69 +++++++++++++++++++++++++++++++--
 1 file changed, 65 insertions(+), 4 deletions(-)

diff --git a/arch/arm64/include/asm/percpu.h b/arch/arm64/include/asm/percpu.h
index dfae0aecdb895..af1d4ca5c85ef 100644
--- a/arch/arm64/include/asm/percpu.h
+++ b/arch/arm64/include/asm/percpu.h
@@ -306,12 +306,67 @@ __percpu_xchg_case_##sz(void __percpu *pcp, u##sz val)			\
 	return ret;							\
 }
 
+#define PERCPU_CMPXCHG_OP(w, sfx, sz)					\
+static inline unsigned long						\
+__percpu_cmpxchg_case_##sz(void __percpu *pcp,				\
+			   u##sz old,					\
+			   u##sz new)					\
+{									\
+	/*                                                              \
+	 * Sub-word sizes require zero extension so that EOR+CBNZ don't	\
+	 * consume non-zero upper bits of the register containing "old".\
+	 */								\
+	xwreg_t(w) cmpval = xwreg_zero_extend(old, w, sz);		\
+	u16 *gprs = &current_thread_info()->pcpu_gprs;			\
+	unsigned long addr;						\
+	unsigned long off;						\
+	unsigned long tmp;						\
+	unsigned long oldval;						\
+									\
+	asm volatile (							\
+	__PCPU_GPRS_BEGIN("%[gprs]", "%[pcp]", "%[off]", "%[addr]")	\
+	ARM64_LSE_ATOMIC_INSN(						\
+	/* LL/SC */							\
+	"	prfm	pstl1strm, [%[addr]]\n"				\
+	"1:	ldxr" #sfx "\t%" #w "[oldval], [%[addr]]\n"		\
+	"	eor	%" #w "[tmp], %" #w "[oldval], %" #w "[old]\n"	\
+	"	cbnz	%" #w "[tmp], 2f\n"				\
+	"	stxr" #sfx "\t%w[tmp], %" #w "[new], [%[addr]]\n"	\
+	"	cbnz	%w[tmp], 1b\n"					\
+	"2:\n"								\
+	,								\
+	/* LSE atomics */						\
+	"	mov	%" #w "[oldval], %" #w "[old]\n"		\
+	"	cas" #sfx "\t%" #w "[oldval], %" #w "[new], [%[addr]]\n"\
+		__nops(4)						\
+	)								\
+	__PCPU_GPRS_END("%[gprs]")					\
+	: [gprs] "=Qo" (*gprs),						\
+	  [addr] "=&r" (addr),						\
+	  [off] "=&r" (off),						\
+	  [tmp] "=&r" (tmp),						\
+	  [oldval] "=&r" (oldval)					\
+	: [pcp] "r" (pcp),						\
+	  [old] "r" (cmpval),						\
+	  [new] "r" (new)						\
+	: "memory"							\
+	);								\
+									\
+	return oldval;							\
+}
+
 PERCPU_XCHG_OP(w, b, 8)
 PERCPU_XCHG_OP(w, h, 16)
 PERCPU_XCHG_OP(w,  , 32)
 PERCPU_XCHG_OP(x,  , 64)
 
+PERCPU_CMPXCHG_OP(w, b, 8)
+PERCPU_CMPXCHG_OP(w, h, 16)
+PERCPU_CMPXCHG_OP(w,  , 32)
+PERCPU_CMPXCHG_OP(x,  , 64)
+
 #undef PERCPU_XCHG_OP
+#undef PERCPU_CMPXCHG_OP
 
 /*
  * It would be nice to avoid the conditional call into the scheduler when
@@ -355,6 +410,12 @@ PERCPU_XCHG_OP(x,  , 64)
 	(typeof(pcp))op(&(pcp), (unsigned long)(val));			\
 })
 
+#define _pcp_wrap_cmpxchg(op, pcp, old, new)				\
+({									\
+	(typeof(pcp))op(&(pcp), (unsigned long)(old),			\
+			(unsigned long)(new));				\
+})
+
 #define this_cpu_read_1(pcp)		\
 	_pcp_wrap_return(__percpu_read_8, pcp)
 #define this_cpu_read_2(pcp)		\
@@ -419,13 +480,13 @@ PERCPU_XCHG_OP(x,  , 64)
 	_pcp_wrap_xchg(__percpu_xchg_case_64, pcp, val)
 
 #define this_cpu_cmpxchg_1(pcp, o, n)	\
-	_pcp_protect_return(cmpxchg_relaxed, pcp, o, n)
+	_pcp_wrap_cmpxchg(__percpu_cmpxchg_case_8, pcp, o, n)
 #define this_cpu_cmpxchg_2(pcp, o, n)	\
-	_pcp_protect_return(cmpxchg_relaxed, pcp, o, n)
+	_pcp_wrap_cmpxchg(__percpu_cmpxchg_case_16, pcp, o, n)
 #define this_cpu_cmpxchg_4(pcp, o, n)	\
-	_pcp_protect_return(cmpxchg_relaxed, pcp, o, n)
+	_pcp_wrap_cmpxchg(__percpu_cmpxchg_case_32, pcp, o, n)
 #define this_cpu_cmpxchg_8(pcp, o, n)	\
-	_pcp_protect_return(cmpxchg_relaxed, pcp, o, n)
+	_pcp_wrap_cmpxchg(__percpu_cmpxchg_case_64, pcp, o, n)
 
 #define this_cpu_cmpxchg64(pcp, o, n)	this_cpu_cmpxchg_8(pcp, o, n)
 
-- 
2.30.2
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.