[PATCH v3 05/11] mm, swap: enable THP swapin for vswap entries

Nhat Pham <[email protected]>
Newsgroups gmane.linux.kernel,gmane.linux.kernel.mm,gmane.linux.documentation,gmane.linux.kernel.cgroups
Message-ID <[email protected]>
Swap a large folio back in as a unit when its vswap entries share a
THP-amenable backing (a contiguous physical run, or all zero-filled),
instead of always falling back to order-0 faults.

A zswap-backed or mixed-backing batch is still refused, and the fault
retries at a smaller order.

Signed-off-by: Nhat Pham <[email protected]>
---
 mm/memory.c     | 12 ++++++++----
 mm/swap_state.c | 17 +++++++++++++----
 mm/vswap.h      |  7 +++++++
 mm/zswap.c      | 18 ++++++++++++------
 4 files changed, 40 insertions(+), 14 deletions(-)

diff --git a/mm/memory.c b/mm/memory.c
index ba84565605a1..a1e106a5c5e6 100644
--- a/mm/memory.c
+++ b/mm/memory.c
@@ -4815,11 +4815,15 @@ static unsigned long thp_swapin_suitable_orders(struct vm_fault *vmf)
 	entry = softleaf_from_pte(vmf->orig_pte);
 
 	/*
-	 * THP swapin for vswap is not supported yet. Also, a large swapped
-	 * out folio could be partially or fully in zswap, which we lack
-	 * handling for. In both cases, fall back to order-0 swapin.
+	 * A large swapped out folio could be partially or fully in zswap.
+	 * For vswap entries the THP-amenability of the backing is checked
+	 * later under the cluster lock in __swap_cache_add_check, which
+	 * rejects ZSWAP and mixed batches via -EBUSY and triggers
+	 * order-fallback. For non-vswap entries we still need the
+	 * zswap_never_enabled() bail: zswap_load rejects large folios with
+	 * -EINVAL, which would SIGBUS the fault.
 	 */
-	if (is_vswap_entry(entry) || !zswap_never_enabled())
+	if (!is_vswap_entry(entry) && !zswap_never_enabled())
 		return 0;
 
 	/*
diff --git a/mm/swap_state.c b/mm/swap_state.c
index c61bb3eef62a..479814d19f50 100644
--- a/mm/swap_state.c
+++ b/mm/swap_state.c
@@ -173,6 +173,9 @@ static int __swap_cache_add_check(struct swap_cluster_info *ci,
 	unsigned int ci_off, ci_end;
 	unsigned long old_tb;
 	bool is_zero;
+	struct swap_cluster_info_dynamic *ci_dyn;
+	enum vswap_backing_type type;
+	int ret;
 
 	lockdep_assert_held(&ci->lock);
 
@@ -201,11 +204,17 @@ static int __swap_cache_add_check(struct swap_cluster_info *ci,
 		return 0;
 
 	/*
-	 * THP swapin for vswap is not supported yet; reject the batch so
-	 * swap_cache_alloc_folio falls back to order 0.
+	 * For a vswap entry batch, reject if the backing is not THP-amenable
+	 * (e.g. uniformly ZSWAP, or mixed). The order-fallback loop in
+	 * swap_cache_alloc_folio will retry with a smaller order on -EBUSY.
 	 */
-	if (is_vswap_entry(targ_entry))
-		return -EBUSY;
+	if (is_vswap_entry(targ_entry)) {
+		ci_dyn = container_of(ci, struct swap_cluster_info_dynamic, ci);
+		ret = __vswap_check_backing(ci_dyn, round_down(ci_off, nr),
+					    nr, &type);
+		if (ret != nr || type == VSWAP_ZSWAP)
+			return -EBUSY;
+	}
 
 	is_zero = __swap_table_test_zero(ci, ci_off);
 	ci_off = round_down(ci_off, nr);
diff --git a/mm/vswap.h b/mm/vswap.h
index 239b47b577d5..a921620f08be 100644
--- a/mm/vswap.h
+++ b/mm/vswap.h
@@ -378,6 +378,13 @@ static inline struct zswap_entry *vswap_zswap_load(swp_entry_t entry)
 static inline void folio_release_vswap_backing(struct folio *folio) {}
 static inline void folio_release_non_phys_swap_backing(struct folio *folio) {}
 
+static inline int __vswap_check_backing(struct swap_cluster_info_dynamic *ci_dyn,
+					unsigned int voff, int nr,
+					enum vswap_backing_type *typep)
+{
+	return 0;
+}
+
 static inline int vswap_cluster_alloc_vtable(struct swap_cluster_info_dynamic *ci_dyn)
 {
 	return 0;
diff --git a/mm/zswap.c b/mm/zswap.c
index d0c6ce2aa092..5dc338188a29 100644
--- a/mm/zswap.c
+++ b/mm/zswap.c
@@ -1630,13 +1630,19 @@ int zswap_load(struct folio *folio)
 		return -ENOENT;
 
 	/*
-	 * Large folios should not be swapped in while zswap is being used, as
-	 * they are not properly handled. Zswap does not properly load large
-	 * folios, and a large folio may only be partially in zswap.
+	 * zswap_load() does not support large folios. For non-vswap
+	 * entries this is unexpected on the swapin path: WARN and
+	 * sigbus. For vswap entries __swap_cache_add_check() has already
+	 * filtered out ZSWAP-backed THPs under the cluster lock, so the
+	 * large folio here is zero- or phys-backed; return -ENOENT so the
+	 * phys/zero IO path handles it.
 	 */
-	if (WARN_ON_ONCE(folio_test_large(folio))) {
-		folio_unlock(folio);
-		return -EINVAL;
+	if (folio_test_large(folio)) {
+		if (WARN_ON_ONCE(!swap_is_vswap(si))) {
+			folio_unlock(folio);
+			return -EINVAL;
+		}
+		return -ENOENT;
 	}
 
 	entry = zswap_entry_load(swp);
-- 
2.53.0-Meta
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.