Re: [PATCH slab/for-next v4 5/8] mm/slab: allow kfree_rcu_sheaf() on PREEMPT_RT

"Vlastimil Babka (SUSE)" <[email protected]>
Newsgroups dev.linux.lists.linux-rt-devel,org.kernel.vger.bpf,org.kernel.vger.linux-kernel,org.kernel.vger.rcu,org.kvack.linux-mm
Message-ID <[email protected]>
On 7/20/26 14:44, Harry Yoo (Oracle) wrote:
> As suggested by Vlastimil Babka [1], kfree_rcu_sheaf() can be used
> on PREEMPT_RT if we always assume spinning is not allowed on PREEMPT_RT.
> This is because local_trylock and spinlock_t are safe to use with
> trylock and unlock as long as the kernel does not spin and the context
> is not NMI and not hardirq.
> 
> Now that __kfree_rcu_sheaf() knows how to handle SLAB_FREE_NOLOCK,
> relax the limitation and try the sheaves path on PREEMPT_RT as well.
> 
> Keep the lockdep map on non RT kernels. However, do not use the lockdep
> map on PREEMPT_RT to avoid suppressing valid lockdep warnings.
> 
> As pointed by Vlastimil Babka [2], on PREEMPT_RT it is unnecessary to
> defer call_rcu() under IRQ-disabled section or raw spinlock. However,
> let us avoid adding more complexity as the scenario is not supposed
> to be common on PREEMPT_RT, with a hope that call_rcu_nolock() will be
> soon supported in RCU.
> 
> Link: https://lore.kernel.org/linux-mm/[email protected] [1]
> Link: https://lore.kernel.org/linux-mm/[email protected] [2]
> Suggested-by: Vlastimil Babka (SUSE) <[email protected]>
> Signed-off-by: Harry Yoo (Oracle) <[email protected]>

Reviewed-by: Vlastimil Babka (SUSE) <[email protected]>

Nit:

> ---
>  mm/slab_common.c | 12 ++++++++++--
>  mm/slub.c        | 17 ++++++++++-------
>  2 files changed, 20 insertions(+), 9 deletions(-)
> 
> diff --git a/mm/slab_common.c b/mm/slab_common.c
> index 9c9ae1384f47..8c631bf97cd5 100644
> --- a/mm/slab_common.c
> +++ b/mm/slab_common.c
> @@ -1595,6 +1595,14 @@ static bool kfree_rcu_sheaf(void *obj)
>  {
>  	struct kmem_cache *s;
>  	struct slab *slab;
> +	unsigned int free_flags = SLAB_FREE_DEFAULT;
> +
> +	/*
> +	 * It is not safe to spin on PREEMPT_RT because the kernel might be
> +	 * holding a raw spinlock and slab acquires sleeping locks.
> +	 */
> +	if (IS_ENABLED(CONFIG_PREEMPT_RT))
> +		free_flags = SLAB_FREE_NOLOCK;
>  
>  	if (is_vmalloc_addr(obj))
>  		return false;
> @@ -1605,7 +1613,7 @@ static bool kfree_rcu_sheaf(void *obj)
>  
>  	s = slab->slab_cache;
>  	if (likely(!IS_ENABLED(CONFIG_NUMA) || slab_nid(slab) == numa_mem_id()))
> -		return __kfree_rcu_sheaf(s, obj, SLAB_FREE_DEFAULT);
> +		return __kfree_rcu_sheaf(s, obj, free_flags);
>  
>  	return false;
>  }
> @@ -1954,7 +1962,7 @@ void kvfree_call_rcu(struct rcu_head *head, void *ptr)
>  	if (!head)
>  		might_sleep();
>  
> -	if (!IS_ENABLED(CONFIG_PREEMPT_RT) && kfree_rcu_sheaf(ptr))
> +	if (kfree_rcu_sheaf(ptr))
>  		return;
>  
>  	// Queue the object but don't yet schedule the batch.
> diff --git a/mm/slub.c b/mm/slub.c
> index 8afa6b47b1f2..deac315d0f23 100644
> --- a/mm/slub.c
> +++ b/mm/slub.c
> @@ -6065,12 +6065,13 @@ static void rcu_free_sheaf(struct rcu_head *head)
>   * kvfree_call_rcu() can be called while holding a raw_spinlock_t. Since
>   * __kfree_rcu_sheaf() may acquire a spinlock_t (sleeping lock on PREEMPT_RT),
>   * this would violate lock nesting rules. Therefore, kvfree_call_rcu() avoids
> - * this problem by bypassing the sheaves layer entirely on PREEMPT_RT.
> + * this problem by passing allow_spin = false on PREEMPT_RT.

by passing SLAB_FREE_NOLOCK ?

>   *
>   * However, lockdep still complains that it is invalid to acquire spinlock_t
>   * while holding raw_spinlock_t, even on !PREEMPT_RT where spinlock_t is a
>   * spinning lock. Tell lockdep that acquiring spinlock_t is valid here
> - * by temporarily raising the wait-type to LD_WAIT_CONFIG.
> + * by temporarily raising the wait-type to LD_WAIT_CONFIG. Skip the lockdep map
> + * on PREEMPT_RT to avoid suppressing valid lockdep warnings.
>   */
>  static DEFINE_WAIT_OVERRIDE_MAP(kfree_rcu_sheaf_map, LD_WAIT_CONFIG);
>  
> @@ -6080,10 +6081,10 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags)
>  	struct slab_sheaf *rcu_sheaf;
>  	bool allow_spin = free_flags_allow_spinning(free_flags);
>  
> -	if (WARN_ON_ONCE(IS_ENABLED(CONFIG_PREEMPT_RT)))
> -		return false;
> +	VM_WARN_ON_ONCE(IS_ENABLED(CONFIG_PREEMPT_RT) && allow_spin);
>  
> -	lock_map_acquire_try(&kfree_rcu_sheaf_map);
> +	if (!IS_ENABLED(CONFIG_PREEMPT_RT))
> +		lock_map_acquire_try(&kfree_rcu_sheaf_map);
>  
>  	if (!local_trylock(&s->cpu_sheaves->lock))
>  		goto fail;
> @@ -6181,12 +6182,14 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags)
>  	local_unlock(&s->cpu_sheaves->lock);
>  
>  	stat(s, FREE_RCU_SHEAF);
> -	lock_map_release(&kfree_rcu_sheaf_map);
> +	if (!IS_ENABLED(CONFIG_PREEMPT_RT))
> +		lock_map_release(&kfree_rcu_sheaf_map);
>  	return true;
>  
>  fail:
>  	stat(s, FREE_RCU_SHEAF_FAIL);
> -	lock_map_release(&kfree_rcu_sheaf_map);
> +	if (!IS_ENABLED(CONFIG_PREEMPT_RT))
> +		lock_map_release(&kfree_rcu_sheaf_map);
>  	return false;
>  }
>  
>
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.