[RFC PATCH 15/57] mm/collapse: check what a candidate would freeze

Kiryl Shutsemau <[email protected]>
Newsgroups org.kernel.vger.linux-trace-kernel,org.kernel.vger.bpf,org.kernel.vger.linux-kernel,org.kernel.vger.linux-kselftest,org.kvack.linux-mm
Message-ID <[email protected]>
From: "Kiryl Shutsemau (Meta)" <[email protected]>

The freeze takes folio locks, rewrites PTEs and flushes the TLB, and any
of that has to be undone slot by slot if the candidate turns out unfit --
while faulters on those sources wait.  So it decides first and acts
second.

This is the deciding half: walk every slot a candidate covers, under the
table's ptl, and answer whether all of it can be frozen.  It touches
nothing, so a refusal costs the round only the walk.

The walk goes in source spans, a span being consecutive PTEs mapping
consecutive pages of one folio.  No layout is refused for its shape:
where a span ends, the next slot starts one of its own, which is what
lets partially mapped and compound sources collapse.  A slot may also be
a hole or the zeropage, both of which the destination just zero-fills.

What a span has to satisfy, beyond being present, anonymous and not
uffd-armed:

 - Every live mapping of its folio is this span.  The freeze is
   whole-folio, so a live PTE anywhere else would race a zap whose
   folio_put() underflows the frozen count.  Under the ptl this is exact,
   since fork -- the only way an exclusive anon folio gains mappings --
   takes mmap_write.

 - Every page of it is PageAnonExclusive().  A shared folio has no
   refcount the freeze can pin down without the other mappers' ptls.

 - It is not MADV_FREE'd, unless the caller asked for the collapse.
   Copying a lazyfree page into a folio that is not lazyfree would quietly
   make memory the user offered up undroppable again, which is why the
   policy carries that choice.

Sub-PMD candidates also refuse folios already at or above their own order,
there being nothing to gain; a PMD candidate takes them, that being the
PTE-mapped-THP re-collapse case.

SCAN_PAGE_NOT_EXCLUSIVE joins enum scan_result and the trace symbol list.

Assisted-by: Claude-Code:claude-opus-5
Signed-off-by: Kiryl Shutsemau (Meta) <[email protected]>
---
 include/trace/events/huge_memory.h |   1 +
 mm/collapse.c                      | 177 +++++++++++++++++++++++++++++
 mm/collapse.h                      |   1 +
 3 files changed, 179 insertions(+)

diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h
index 68693eba82ef..ff938ac9c43c 100644
--- a/include/trace/events/huge_memory.h
+++ b/include/trace/events/huge_memory.h
@@ -41,6 +41,7 @@
 	EM( SCAN_COPY_MC,		"copy_poisoned_page")		\
 	EM( SCAN_PAGE_FILLED,		"page_filled")			\
 	EM( SCAN_PAGE_DIRTY_OR_WRITEBACK, "page_dirty_or_writeback")	\
+	EM( SCAN_PAGE_NOT_EXCLUSIVE,	"page_not_exclusive")		\
 	EMe(SCAN_ALLOC_LIGHT_MISS,	"alloc_light_miss")
 
 #undef EM
diff --git a/mm/collapse.c b/mm/collapse.c
index 4ec02071f588..c75d91cb9d48 100644
--- a/mm/collapse.c
+++ b/mm/collapse.c
@@ -386,6 +386,145 @@ static enum scan_result collapse_faultin(struct vm_area_struct *vma,
 	return result;
 }
 
+/*
+ * How many slots a source span starting at @first may cover: the pages left in
+ * its folio, capped at @max.  Every freeze-side walker bounds spans with this,
+ * so per-span batching of clears, locks and freezes cannot reach a slot the span
+ * does not cover.
+ */
+static unsigned int collapse_span_max(pte_t first, unsigned int max)
+{
+	struct page *page = pte_page(first);
+	struct folio *folio = page_folio(page);
+	unsigned int left = folio_nr_pages(folio) - folio_page_idx(folio, page);
+
+	return min(max, left);
+}
+
+/*
+ * Can this candidate's sources be frozen?  Every slot is checked and nothing is
+ * touched, so a refusal costs the round nothing but the walk.
+ *
+ * The walk is in source spans: a span is consecutive PTEs mapping consecutive
+ * pages of one folio, and it ends wherever the next PTE stops being the folio's
+ * next page.  No layout is refused for its shape -- the next slot simply starts
+ * its own span -- so partially mapped and compound sources collapse too.
+ *
+ * Caller holds mmap_read and the table's ptl.
+ */
+static enum scan_result collapse_check_candidate(struct vm_area_struct *vma,
+						 struct collapse_control *cc,
+						 struct collapse_candidate *cand,
+						 pte_t *pte)
+{
+	const unsigned int nr_pages = candidate_nr_pages(cand);
+	unsigned long addr;
+	unsigned int i;
+
+	for (i = 0, addr = cand->addr; i < nr_pages;) {
+		pte_t ptent = ptep_get(pte + i);
+		unsigned int nr, nr_max, k;
+		struct folio *folio;
+		struct page *page;
+
+		if (!pte_present(ptent)) {
+			/* Holes are population; swap and markers are not */
+			if (pte_none(ptent)) {
+				i++;
+				addr += PAGE_SIZE;
+				continue;
+			}
+			return SCAN_PTE_NON_PRESENT;
+		}
+		if (pte_uffd(ptent))
+			return SCAN_PTE_UFFD;
+
+		/* The zeropage zero-fills like a hole, and has no normal page */
+		if (is_zero_pfn(pte_pfn(ptent))) {
+			i++;
+			addr += PAGE_SIZE;
+			continue;
+		}
+		page = vm_normal_page(vma, addr, ptent);
+		if (!page || unlikely(is_zone_device_page(page)))
+			return SCAN_PAGE_NULL;
+
+		folio = page_folio(page);
+		if (!folio_test_anon(folio))
+			return SCAN_PAGE_ANON;
+
+		/*
+		 * Collapsing a MADV_FREE'd page would copy it into a folio that
+		 * is not lazyfree, quietly making memory the user offered up
+		 * undroppable again.
+		 */
+		if (cc->policy.skip_lazyfree &&
+		    !(vma->vm_flags & VM_DROPPABLE) &&
+		    folio_test_lazyfree(folio) && !pte_dirty(ptent))
+			return SCAN_PAGE_LAZYFREE;
+
+		/*
+		 * A sub-PMD candidate refuses folios of its own order and above:
+		 * collapsing those would gain nothing.  A PMD candidate accepts
+		 * every order up to its own -- the PTE-mapped-THP re-collapse
+		 * class.
+		 */
+		if (folio_order(folio) >= cand->order &&
+		    !is_pmd_order(cand->order))
+			return SCAN_PTE_MAPPED_HUGEPAGE;
+
+		/*
+		 * Exclusive anon only: the expected refcount of a shared folio
+		 * cannot be pinned down without its other mappers' ptls.
+		 * Swapcache membership is fine -- folio_expected_ref_count()
+		 * accounts those references.
+		 */
+		if (folio_maybe_mapped_shared(folio))
+			return SCAN_PAGE_NOT_EXCLUSIVE;
+
+		nr_max = collapse_span_max(ptent, nr_pages - i);
+		for (nr = 1; nr < nr_max; nr++) {
+			pte_t tail = ptep_get(pte + i + nr);
+
+			if (!pte_present(tail) ||
+			    pte_pfn(tail) != pte_pfn(ptent) + nr)
+				break;
+			if (pte_uffd(tail))
+				return SCAN_PTE_UFFD;
+		}
+
+		/*
+		 * Every live mapping of the folio must be this span: the freeze
+		 * is whole-folio, and a live PTE left anywhere else loses to a
+		 * racing zap -- its rmap drop is paired with a folio_put() that
+		 * would underflow the frozen count.  The check is race-free
+		 * under our ptl: in-window PTEs are ours, fork (the only way
+		 * exclusive anon gains mappings) takes mmap_write, and a folio
+		 * whose mappings all sit under this ptl cannot lose one either.
+		 * This also refuses a folio scattered across several spans of
+		 * the window, whose mapcount exceeds any single span.
+		 */
+		if (folio_mapcount(folio) != nr)
+			return SCAN_PAGE_COUNT;
+
+		/*
+		 * Every page of the span must be exclusive: the freeze accounts
+		 * only references it can see, and a non-exclusive page may be
+		 * unshared under us.  collapse_faultin() should have arranged
+		 * this; enforce it here, where it is depended on.
+		 */
+		for (k = 0; k < nr; k++) {
+			if (!PageAnonExclusive(pte_page(ptep_get(pte + i + k))))
+				return SCAN_PAGE_NOT_EXCLUSIVE;
+		}
+
+		i += nr;
+		addr += nr * PAGE_SIZE;
+	}
+
+	return SCAN_SUCCEED;
+}
+
 /*
  * Raise the two barriers on the sources of every candidate: migration entries in
  * their PTEs, then a frozen refcount.  Takes the table's ptl once for the whole
@@ -395,6 +534,44 @@ static enum scan_result collapse_faultin(struct vm_area_struct *vma,
 static void collapse_freeze(struct vm_area_struct *vma,
 			    struct collapse_control *cc, pmd_t *pmd)
 {
+	struct mm_struct *mm = vma->vm_mm;
+	pte_t *pte, *table;
+	spinlock_t *ptl;
+	unsigned int i;
+
+	pte = pte_offset_map_lock(mm, pmd, cc->candidates[0].addr, &ptl);
+	if (!pte) {
+		for (i = 0; i < cc->nr_candidates; i++) {
+			struct collapse_candidate *cand = &cc->candidates[i];
+
+			if (cand->state != CAND_SELECTED)
+				continue;
+			cand->state = CAND_SKIPPED;
+			cand->result = SCAN_NO_PTE_TABLE;
+		}
+		return;
+	}
+
+	/*
+	 * Index each candidate from the table base, not relative to
+	 * candidates[0]: a round is not necessarily address-ordered, so
+	 * candidates[0] need not be the lowest.  They all share one table.
+	 */
+	table = pte - pte_index(cc->candidates[0].addr);
+
+	for (i = 0; i < cc->nr_candidates; i++) {
+		struct collapse_candidate *cand = &cc->candidates[i];
+		pte_t *cand_pte = table + pte_index(cand->addr);
+
+		if (cand->state != CAND_SELECTED)
+			continue;
+
+		cand->result = collapse_check_candidate(vma, cc, cand, cand_pte);
+		if (cand->result != SCAN_SUCCEED)
+			cand->state = CAND_SKIPPED;
+	}
+
+	pte_unmap_unlock(pte, ptl);
 }
 
 /*
diff --git a/mm/collapse.h b/mm/collapse.h
index 0d6f77a7233b..747168104a72 100644
--- a/mm/collapse.h
+++ b/mm/collapse.h
@@ -46,6 +46,7 @@ enum scan_result {
 	SCAN_COPY_MC,
 	SCAN_PAGE_FILLED,
 	SCAN_PAGE_DIRTY_OR_WRITEBACK,
+	SCAN_PAGE_NOT_EXCLUSIVE,
 	SCAN_ALLOC_LIGHT_MISS,
 };
 
-- 
2.54.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.