[PATCH] mm/swap: publish cluster tables after full initialization

Longlong Xia <[email protected]>
Newsgroups org.kernel.vger.stable,org.kvack.linux-mm
Message-ID <[email protected]>
From: Longlong Xia <[email protected]>

swap_cluster_populate() drops the local, global, and cluster locks
before its sleeping allocation.  The allocation helper publishes ci->table
before allocating the memcg table and, on some 32-bit configurations, the
zero bitmap.

A stale per-CPU or global cluster cursor can reach the isolated cluster in
that window.  Since CLUSTER_FLAG_NONE and a non-NULL table make the cluster
appear usable, it can allocate a slot without the auxiliary state.  An
auxiliary allocation failure can then tear down a table which is already in
use.

Allocate the complete set of tables into a private carrier.  Install the
auxiliary pointers and publish the main table only while holding ci->lock;
the slow path does this after reacquiring all allocator locks.  Allocation
failures now free only unpublished resources.

Fixes: b197d41462c2 ("mm/memcg, swap: store cgroup id in cluster table directly")
Cc: [email protected]
Assisted-by: Codex:gpt-5.6-sol
Signed-off-by: Longlong Xia <[email protected]>
---
 mm/swapfile.c | 167 +++++++++++++++++++++++++++++++++++---------------
 1 file changed, 117 insertions(+), 50 deletions(-)

diff --git a/mm/swapfile.c b/mm/swapfile.c
index 78b49b0658ad..2ca947c540e9 100644
--- a/mm/swapfile.c
+++ b/mm/swapfile.c
@@ -410,6 +410,99 @@ static inline unsigned int cluster_offset(struct swap_info_struct *si,
 	return cluster_index(si, ci) * SWAPFILE_CLUSTER;
 }
 
+struct swap_cluster_tables {
+	struct swap_table *table;
+#ifdef CONFIG_MEMCG
+	struct swap_memcg_table *memcg_table;
+#endif
+#if !SWAP_TABLE_HAS_ZEROFLAG
+	unsigned long *zero_bitmap;
+#endif
+};
+
+static void swap_cluster_tables_free(struct swap_cluster_tables *tables)
+{
+#ifdef CONFIG_MEMCG
+	kfree(tables->memcg_table);
+	tables->memcg_table = NULL;
+#endif
+
+#if !SWAP_TABLE_HAS_ZEROFLAG
+	kfree(tables->zero_bitmap);
+	tables->zero_bitmap = NULL;
+#endif
+
+	if (!tables->table)
+		return;
+
+	if (SWP_TABLE_USE_PAGE)
+		folio_put(virt_to_folio(tables->table));
+	else
+		kmem_cache_free(swap_table_cachep, tables->table);
+	tables->table = NULL;
+}
+
+static int swap_cluster_tables_alloc(struct swap_cluster_tables *tables,
+				     gfp_t gfp)
+{
+	struct folio *folio;
+
+	if (SWP_TABLE_USE_PAGE) {
+		folio = folio_alloc(gfp | __GFP_ZERO, 0);
+		if (folio)
+			tables->table = folio_address(folio);
+	} else {
+		tables->table = kmem_cache_zalloc(swap_table_cachep, gfp);
+	}
+	if (!tables->table)
+		return -ENOMEM;
+
+#ifdef CONFIG_MEMCG
+	if (!mem_cgroup_disabled()) {
+		tables->memcg_table = kzalloc_obj(*tables->memcg_table, gfp);
+		if (!tables->memcg_table)
+			goto free_tables;
+	}
+#endif
+
+#if !SWAP_TABLE_HAS_ZEROFLAG
+	tables->zero_bitmap = bitmap_zalloc(SWAPFILE_CLUSTER, gfp);
+	if (!tables->zero_bitmap)
+		goto free_tables;
+#endif
+
+	return 0;
+
+#if defined(CONFIG_MEMCG) || !SWAP_TABLE_HAS_ZEROFLAG
+free_tables:
+	swap_cluster_tables_free(tables);
+	return -ENOMEM;
+#endif
+}
+
+static void swap_cluster_tables_install(struct swap_cluster_info *ci,
+					struct swap_cluster_tables *tables)
+{
+	lockdep_assert_held(&ci->lock);
+	VM_WARN_ON_ONCE(ci->flags || !cluster_is_empty(ci));
+	VM_WARN_ON_ONCE(rcu_access_pointer(ci->table));
+
+#ifdef CONFIG_MEMCG
+	VM_WARN_ON_ONCE(ci->memcg_table);
+	ci->memcg_table = tables->memcg_table;
+	tables->memcg_table = NULL;
+#endif
+
+#if !SWAP_TABLE_HAS_ZEROFLAG
+	VM_WARN_ON_ONCE(ci->zero_bitmap);
+	ci->zero_bitmap = tables->zero_bitmap;
+	tables->zero_bitmap = NULL;
+#endif
+
+	rcu_assign_pointer(ci->table, tables->table);
+	tables->table = NULL;
+}
+
 static void swap_cluster_free_table_folio_rcu_cb(struct rcu_head *head)
 {
 	struct folio *folio;
@@ -446,50 +539,6 @@ static void swap_cluster_free_table(struct swap_cluster_info *ci)
 		 swap_cluster_free_table_folio_rcu_cb);
 }
 
-static int swap_cluster_alloc_table(struct swap_cluster_info *ci, gfp_t gfp)
-{
-	struct swap_table *table = NULL;
-	struct folio *folio;
-
-	/* The cluster must be empty and not on any list during allocation. */
-	VM_WARN_ON_ONCE(ci->flags || !cluster_is_empty(ci));
-	if (rcu_access_pointer(ci->table))
-		return 0;
-
-	if (SWP_TABLE_USE_PAGE) {
-		folio = folio_alloc(gfp | __GFP_ZERO, 0);
-		if (folio)
-			table = folio_address(folio);
-	} else {
-		table = kmem_cache_zalloc(swap_table_cachep, gfp);
-	}
-	if (!table)
-		return -ENOMEM;
-
-	rcu_assign_pointer(ci->table, table);
-
-#ifdef CONFIG_MEMCG
-	if (!mem_cgroup_disabled()) {
-		VM_WARN_ON_ONCE(ci->memcg_table);
-		ci->memcg_table = kzalloc_obj(*ci->memcg_table, gfp);
-		if (!ci->memcg_table) {
-			swap_cluster_free_table(ci);
-			return -ENOMEM;
-		}
-	}
-#endif
-
-#if !SWAP_TABLE_HAS_ZEROFLAG
-	VM_WARN_ON_ONCE(ci->zero_bitmap);
-	ci->zero_bitmap = bitmap_zalloc(SWAPFILE_CLUSTER, gfp);
-	if (!ci->zero_bitmap) {
-		swap_cluster_free_table(ci);
-		return -ENOMEM;
-	}
-#endif
-	return 0;
-}
-
 /*
  * Sanity check to ensure nothing leaked, and the specified range is empty.
  * One special case is that bad slots can't be freed, so check the number of
@@ -527,6 +576,7 @@ static struct swap_cluster_info *
 swap_cluster_populate(struct swap_info_struct *si,
 			 struct swap_cluster_info *ci)
 {
+	struct swap_cluster_tables tables = {};
 	int ret;
 
 	/*
@@ -538,9 +588,12 @@ swap_cluster_populate(struct swap_info_struct *si,
 		lockdep_assert_held(&si->global_cluster_lock);
 	lockdep_assert_held(&ci->lock);
 
-	if (!swap_cluster_alloc_table(ci, __GFP_HIGH | __GFP_NOMEMALLOC |
-					  __GFP_NOWARN))
+	ret = swap_cluster_tables_alloc(&tables, __GFP_HIGH | __GFP_NOMEMALLOC |
+					       __GFP_NOWARN);
+	if (!ret) {
+		swap_cluster_tables_install(ci, &tables);
 		return ci;
+	}
 
 	/*
 	 * Try a sleep allocation. Each isolated free cluster may cause
@@ -552,8 +605,8 @@ swap_cluster_populate(struct swap_info_struct *si,
 		spin_unlock(&si->global_cluster_lock);
 	local_unlock(&percpu_swap_cluster.lock);
 
-	ret = swap_cluster_alloc_table(ci, __GFP_HIGH | __GFP_NOMEMALLOC |
-					   GFP_KERNEL);
+	ret = swap_cluster_tables_alloc(&tables, __GFP_HIGH | __GFP_NOMEMALLOC |
+						GFP_KERNEL);
 
 	/*
 	 * Back to atomic context. We might have migrated to a new CPU with a
@@ -568,11 +621,19 @@ swap_cluster_populate(struct swap_info_struct *si,
 		spin_lock(&si->global_cluster_lock);
 	spin_lock(&ci->lock);
 
+	/* Nothing except this helper should populate an isolated cluster. */
+	if (WARN_ON_ONCE(cluster_table_is_alloced(ci))) {
+		swap_cluster_tables_free(&tables);
+		return ci;
+	}
+
 	if (ret) {
 		move_cluster(si, ci, &si->free_clusters, CLUSTER_FLAG_FREE);
 		spin_unlock(&ci->lock);
 		return NULL;
 	}
+
+	swap_cluster_tables_install(ci, &tables);
 	return ci;
 }
 
@@ -788,6 +849,7 @@ static int swap_cluster_setup_bad_slot(struct swap_info_struct *si,
 				       struct swap_cluster_info *cluster_info,
 				       unsigned int offset, bool mask)
 {
+	struct swap_cluster_tables tables = {};
 	unsigned int ci_off = offset % SWAPFILE_CLUSTER;
 	unsigned long idx = offset / SWAPFILE_CLUSTER;
 	struct swap_cluster_info *ci;
@@ -812,9 +874,14 @@ static int swap_cluster_setup_bad_slot(struct swap_info_struct *si,
 
 	ci = cluster_info + idx;
 	/* Need to allocate swap table first for initial bad slot marking. */
-	if (!ci->count && swap_cluster_alloc_table(ci, GFP_KERNEL))
-		return -ENOMEM;
+	if (!ci->count) {
+		ret = swap_cluster_tables_alloc(&tables, GFP_KERNEL);
+		if (ret)
+			return ret;
+	}
 	spin_lock(&ci->lock);
+	if (tables.table)
+		swap_cluster_tables_install(ci, &tables);
 	/* Check for duplicated bad swap slots. */
 	if (__swap_table_xchg(ci, ci_off, SWP_TB_BAD) != SWP_TB_NULL) {
 		pr_warn("Duplicated bad slot offset %d\n", offset);
-- 
2.43.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.