[PATCH] srcu: Keep a spare node array so srcu_gp_end() need not block in reclaim

David Woodhouse <[email protected]>
Newsgroups org.kernel.vger.rcu,org.kernel.vger.kvm,org.kernel.vger.linux-kernel
Message-ID <[email protected]>
From: David Woodhouse <[email protected]>

The one-time transition of an srcu_struct to SRCU_SIZE_BIG allocates
the srcu_node tree with GFP_KERNEL from srcu_gp_end(), which runs on
the same workqueue that processes grace periods for every srcu_struct
in the system — including those awaited from OOM/reclaim contexts such
as the OOM reaper via an mmu_notifier. If the allocation blocks in
reclaim, it can be waiting on the very OOM reaper whose grace period
is queued behind it: a deadlock.

The size of the array depends only on rcu_num_nodes, fixed once
rcu_init_geometry() has run, so one preallocated spare fits every
srcu_struct. Have srcu_gp_end() allocate with GFP_NOWAIT, falling back
to the spare, which is replenished from system_wq where blocking is
harmless. If both fail, nothing is lost: the srcu_struct simply
remains un-upgraded — fully functional, just contended — and the
upgrade is retried on a later grace period.

Signed-off-by: David Woodhouse <[email protected]>
Assisted-by: Claude:claude-mythos-5
---
Based on the discussion at 
https://lore.kernel.org/all/[email protected]/

 kernel/rcu/srcutree.c | 85 +++++++++++++++++++++++++++++++++++++++++--
 1 file changed, 82 insertions(+), 3 deletions(-)

diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c
index 7c2f7cc131f7..2601566c254a 100644
--- a/kernel/rcu/srcutree.c
+++ b/kernel/rcu/srcutree.c
@@ -123,6 +123,72 @@ static inline bool srcu_invl_snp_seq(unsigned long s)
 	return s == SRCU_SNP_INIT_SEQ;
 }
 
+/*
+ * A standing spare srcu_node array. The size of the allocation depends
+ * only on rcu_num_nodes, which is fixed once rcu_init_geometry() has run,
+ * so one preallocated array fits every srcu_struct in the system.
+ *
+ * This exists because srcu_gp_end() may need to allocate the array when
+ * a size transition is triggered by contention, and srcu_gp_end() runs
+ * on the same workqueue for every srcu_struct — including grace periods
+ * awaited from OOM/reclaim contexts (e.g. the OOM reaper via an
+ * mmu_notifier). Blocking there in GFP_KERNEL reclaim can deadlock: the
+ * reclaim may be waiting on the very OOM reaper whose grace period is
+ * queued behind this allocation.
+ *
+ * The allocation is therefore attempted with the caller's own flags
+ * (GFP_NOWAIT on the grace-period path), and the spare is raided only
+ * when that fails — i.e. under the memory pressure the spare exists
+ * for. The spare is replenished from a clean context on system_wq.
+ * Nothing on the grace-period path ever blocks in reclaim.
+ */
+static struct srcu_node *srcu_spare_nodes;
+
+static void srcu_spare_replenish_wq(struct work_struct *work)
+{
+	struct srcu_node *spare, *expect = NULL;
+
+	if (READ_ONCE(srcu_spare_nodes))
+		return;		/* Already refilled. */
+
+	spare = kzalloc_objs(*spare, rcu_num_nodes, GFP_KERNEL);
+	if (!spare)
+		return;
+	if (!try_cmpxchg(&srcu_spare_nodes, &expect, spare))
+		kfree(spare);	/* Someone else refilled it first. */
+}
+static DECLARE_WORK(srcu_spare_replenish_work, srcu_spare_replenish_wq);
+
+static struct srcu_node *srcu_alloc_nodes(gfp_t gfp_flags)
+{
+	struct srcu_node *node;
+
+	/*
+	 * Try the caller's own flags first, raiding the spare only if that
+	 * fails. For init_srcu_struct() the flags are GFP_KERNEL in the
+	 * caller's own task, where blocking is permitted: such a caller
+	 * only ever reaches the spare under genuine OOM, rather than
+	 * consuming it on any transient pressure. From srcu_gp_end() the
+	 * flags are GFP_NOWAIT, so nothing on the grace-period workqueue
+	 * ever blocks in reclaim, and the spare is the fallback it exists
+	 * to provide. If the spare is also gone (already raided, not yet
+	 * replenished), fail: srcu_gp_end() retries the size transition
+	 * on a later grace period.
+	 */
+	node = kzalloc_objs(*node, rcu_num_nodes, gfp_flags);
+	if (node)
+		return node;
+
+	/*
+	 * Kick the replenisher whether or not the raid succeeds: the
+	 * replenisher does not retry a failed allocation itself, so this
+	 * is what retries the refill on each pressure event.
+	 */
+	node = xchg(&srcu_spare_nodes, NULL);
+	schedule_work(&srcu_spare_replenish_work);
+	return node;
+}
+
 /*
  * Allocated and initialize SRCU combining tree.  Returns @true if
  * allocation succeeded and @false otherwise.
@@ -139,8 +205,7 @@ static bool init_srcu_struct_nodes(struct srcu_struct *ssp, gfp_t gfp_flags)
 
 	/* Initialize geometry if it has not already been initialized. */
 	rcu_init_geometry();
-	ssp->srcu_sup->node = kzalloc_objs(*ssp->srcu_sup->node, rcu_num_nodes,
-					   gfp_flags);
+	ssp->srcu_sup->node = srcu_alloc_nodes(gfp_flags);
 	if (!ssp->srcu_sup->node)
 		return false;
 
@@ -1004,7 +1069,7 @@ static void srcu_gp_end(struct srcu_struct *ssp)
 	/* Transition to big if needed. */
 	if (ss_state != SRCU_SIZE_SMALL && ss_state != SRCU_SIZE_BIG) {
 		if (ss_state == SRCU_SIZE_ALLOC)
-			init_srcu_struct_nodes(ssp, GFP_KERNEL);
+			init_srcu_struct_nodes(ssp, GFP_NOWAIT);
 		else
 			smp_store_release(&sup->srcu_size_state, ss_state + 1);
 	}
@@ -2111,6 +2176,20 @@ void __init srcu_init(void)
 		}
 	}
 
+	/*
+	 * Prime the spare node array if lazy (contention-triggered) size
+	 * transitions are possible, so that srcu_gp_end() never needs to
+	 * allocate. Early-boot GFP_KERNEL is implicitly non-blocking
+	 * (gfp_allowed_mask strips __GFP_RECLAIM until much later), and
+	 * failure here is harmless: the GFP_NOWAIT fallback and replenish
+	 * worker remain.
+	 */
+	if (SRCU_SIZING_IS_CONTEND() || SRCU_SIZING_IS_TORTURE()) {
+		rcu_init_geometry();
+		srcu_spare_nodes = kzalloc_objs(*srcu_spare_nodes,
+						rcu_num_nodes, GFP_KERNEL);
+	}
+
 	/*
 	 * Once that is set, call_srcu() can follow the normal path and
 	 * queue delayed work. This must follow RCU workqueues creation
-- 
2.43.0
smime.p7s (application/pkcs7-signature, 6 KB) - not displayed
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.