[PATCH v3 1/2] mm, swap: distinguish a malformed swap entry from a dying device

Breno Leitao <[email protected]>
Newsgroups org.kvack.linux-mm,org.kernel.vger.linux-kernel
Message-ID <[email protected]>
get_swap_device() returns NULL for two different things: an entry whose
type names no swap device or whose offset is past the end of one, and a
device that swapoff is taking away. The first never becomes valid, the
second does, and callers cannot tell them apart.

Return ERR_PTR(-EIO) for the two malformed cases and keep NULL for
swapoff. copy_nonpresent_pte() already reports -EIO for an entry whose
type names no device.

Callers bail out on failure either way, so switch them to
IS_ERR_OR_NULL(), and let the two paths that drop the reference skip an
error pointer. No functional change.

Reviewed-by: Barry Song <[email protected]>
Acked-by: Kairui Song <[email protected]>
Signed-off-by: Breno Leitao <[email protected]>
---
 mm/memory.c      |  6 +++---
 mm/mincore.c     |  2 +-
 mm/shmem.c       |  2 +-
 mm/swap_state.c  |  4 ++--
 mm/swapfile.c    | 14 +++++++++-----
 mm/userfaultfd.c |  4 ++--
 mm/zswap.c       |  2 +-
 7 files changed, 19 insertions(+), 15 deletions(-)

diff --git a/mm/memory.c b/mm/memory.c
index d9cf941967cf0..03d8cf111d0be 100644
--- a/mm/memory.c
+++ b/mm/memory.c
@@ -4954,9 +4954,9 @@ vm_fault_t do_swap_page(struct vm_fault *vmf)
 		goto out;
 	}
 
-	/* Prevent swapoff from happening to us. */
+	/* Prevent swapoff from happening to us, and reject a bad entry. */
 	si = get_swap_device(entry);
-	if (unlikely(!si))
+	if (IS_ERR_OR_NULL(si))
 		goto out;
 
 	folio = swap_cache_get_folio(entry);
@@ -5266,7 +5266,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf)
 	if (vmf->pte)
 		pte_unmap_unlock(vmf->pte, vmf->ptl);
 out:
-	if (si)
+	if (!IS_ERR_OR_NULL(si))
 		put_swap_device(si);
 	return ret;
 out_nomap:
diff --git a/mm/mincore.c b/mm/mincore.c
index ff4ac82817683..c086836bc4bcc 100644
--- a/mm/mincore.c
+++ b/mm/mincore.c
@@ -71,7 +71,7 @@ static unsigned char mincore_swap(swp_entry_t entry, bool shmem)
 	 */
 	if (shmem) {
 		si = get_swap_device(entry);
-		if (!si)
+		if (IS_ERR_OR_NULL(si))
 			return 0;
 	}
 	folio = swap_cache_get_folio(entry);
diff --git a/mm/shmem.c b/mm/shmem.c
index 65572cbf1bd3c..d0a9f52bfed71 100644
--- a/mm/shmem.c
+++ b/mm/shmem.c
@@ -2276,7 +2276,7 @@ static int shmem_swapin_folio(struct inode *inode, pgoff_t index,
 
 	si = get_swap_device(index_entry);
 	order = shmem_confirm_swap(mapping, index, index_entry);
-	if (unlikely(!si)) {
+	if (IS_ERR_OR_NULL(si)) {
 		if (order < 0)
 			return -EEXIST;
 		else
diff --git a/mm/swap_state.c b/mm/swap_state.c
index 4b7a3303c463b..f2e86d6626ecc 100644
--- a/mm/swap_state.c
+++ b/mm/swap_state.c
@@ -715,7 +715,7 @@ struct folio *read_swap_cache_async(struct swap_io_ctx *ctx, swp_entry_t entry,
 	struct folio *folio;
 
 	si = get_swap_device(entry);
-	if (!si)
+	if (IS_ERR_OR_NULL(si))
 		return NULL;
 
 	mpol = get_vma_policy(vma, addr, 0, &ilx);
@@ -951,7 +951,7 @@ static struct folio *swap_vma_readahead(swp_entry_t targ_entry, gfp_t gfp_mask,
 		 */
 		if (swp_type(entry) != swp_type(targ_entry)) {
 			si = get_swap_device(entry);
-			if (!si)
+			if (IS_ERR_OR_NULL(si))
 				continue;
 		}
 		folio = swap_cache_read_folio(&ctx, entry, gfp_mask, mpol, ilx,
diff --git a/mm/swapfile.c b/mm/swapfile.c
index 4d4e3e3059f6b..3b1883930e943 100644
--- a/mm/swapfile.c
+++ b/mm/swapfile.c
@@ -1504,7 +1504,7 @@ int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp)
 	unsigned long offset = swp_offset(entry);
 
 	si = get_swap_device(entry);
-	if (!si)
+	if (IS_ERR_OR_NULL(si))
 		return 0;
 
 	ci = __swap_offset_to_cluster(si, offset);
@@ -1859,7 +1859,10 @@ void folio_put_swap(struct folio *folio, struct page *page)
  * Check whether swap entry is valid in the swap device.  If so,
  * return pointer to swap_info_struct, and keep the swap entry valid
  * via preventing the swap device from being swapoff, until
- * put_swap_device() is called.  Otherwise return NULL.
+ * put_swap_device() is called.  Return NULL for an empty entry or a
+ * device that is going away, and ERR_PTR(-EIO) if the entry's type
+ * names no swap device or its offset is past the end of one. These EIOs
+ * are preceded by pr_err().
  *
  * Notice that swapoff or swapoff+swapon can still happen before the
  * percpu_ref_tryget_live() in get_swap_device() or after the
@@ -1900,12 +1903,13 @@ struct swap_info_struct *get_swap_device(swp_entry_t entry)
 	return si;
 bad_nofile:
 	pr_err("%s: %s%08lx\n", __func__, Bad_file, entry.val);
+	return ERR_PTR(-EIO);
 out:
 	return NULL;
 put_out:
 	pr_err("%s: %s%08lx\n", __func__, Bad_offset, entry.val);
 	percpu_ref_put(&si->users);
-	return NULL;
+	return ERR_PTR(-EIO);
 }
 
 /*
@@ -2001,7 +2005,7 @@ int swp_swapcount(swp_entry_t entry)
 	int count;
 
 	si = get_swap_device(entry);
-	if (!si)
+	if (IS_ERR_OR_NULL(si))
 		return 0;
 
 	ci = swap_cluster_lock(si, swp_offset(entry));
@@ -2127,7 +2131,7 @@ void swap_put_entries_direct(swp_entry_t entry, int nr)
 	struct swap_info_struct *si;
 
 	si = get_swap_device(entry);
-	if (WARN_ON_ONCE(!si))
+	if (WARN_ON_ONCE(IS_ERR_OR_NULL(si)))
 		return;
 	if (WARN_ON_ONCE(end_offset > si->max))
 		goto out;
diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c
index 24a4d92ffa3c2..cba5e20a641ed 100644
--- a/mm/userfaultfd.c
+++ b/mm/userfaultfd.c
@@ -1700,7 +1700,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd
 		}
 
 		si = get_swap_device(entry);
-		if (unlikely(!si)) {
+		if (IS_ERR_OR_NULL(si)) {
 			ret = -EAGAIN;
 			goto out;
 		}
@@ -1757,7 +1757,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd
 	if (dst_pte)
 		pte_unmap(dst_pte);
 	mmu_notifier_invalidate_range_end(&range);
-	if (si)
+	if (!IS_ERR_OR_NULL(si))
 		put_swap_device(si);
 
 	return ret;
diff --git a/mm/zswap.c b/mm/zswap.c
index f7c9c89f6449c..bc9b931d6f447 100644
--- a/mm/zswap.c
+++ b/mm/zswap.c
@@ -997,7 +997,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry,
 
 	/* try to allocate swap cache folio */
 	si = get_swap_device(swpentry);
-	if (!si)
+	if (IS_ERR_OR_NULL(si))
 		return -EEXIST;
 
 	mpol = get_task_policy(current);

-- 
2.53.0-Meta
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.