[RFC PATCH 12/57] mm/collapse: allocate a destination per candidate

Kiryl Shutsemau <[email protected]>
Newsgroups org.kvack.linux-mm,org.kernel.vger.bpf,org.kernel.vger.linux-kernel,org.kernel.vger.linux-kselftest,org.kernel.vger.linux-trace-kernel
Message-ID <[email protected]>
From: "Kiryl Shutsemau (Meta)" <[email protected]>

Fill in the allocation, which happens on both sides of the freeze.

A destination is a folio of the candidate's order, charged to the memcg,
with the memcg's deferred-split list entry taken up front while sleeping
is still allowed: the PMD-order install would otherwise need one under
the pmd lock.

collapse_alloc() does all of that for one candidate with the gfp it is
handed, and counts nothing when it fails: what a miss means is up to the
caller.

collapse_provision() is the caller inside the window.  The sources are
frozen by then and a faulter on any of them is waiting, so it asks
without __GFP_DIRECT_RECLAIM: reclaim entered there would be paid for by
that faulter.

A candidate the allocator cannot spare one for is declined rather than
failed.  It keeps its freeze and records SCAN_ALLOC_LIGHT_MISS, which
asks for the reclaiming gfp so a later round can allocate for it before
freezing anything.  Where the policy forbids reclaim there is nothing
better to retry with, so the miss is the verdict: the real result is
recorded and the failure counters fire.

collapse_reserve() honours those requests, before the round takes any
lock.  This is where reclaim belongs: nothing is held or frozen, so a
long compaction costs only khugepaged's own progress, which is why the
mechanism being replaced allocated here too.  Having asked the allocator
to try hard, a miss there is a failure.

Nothing sets cand->reclaim yet, so collapse_reserve() has nothing to do.
The request comes from the selection side, which queues a region refused
at one order for another attempt.

The page table a PMD-order candidate deposits cannot be deferred the same
way.  pte_alloc_one() allocates with GFP_PGTABLE_USER and takes no gfp to
strip, so it may reclaim and sleep whatever the order asked for.
collapse_deposit() secures it ahead of the freeze and refuses the
candidate when it cannot.  A PMD-order window is a whole table, so there
is at most one such candidate and it is the first.

Assisted-by: Claude-Code:claude-opus-5
Signed-off-by: Kiryl Shutsemau (Meta) <[email protected]>
---
 include/trace/events/huge_memory.h |   3 +-
 mm/collapse.c                      | 126 +++++++++++++++++++++++++++++
 mm/collapse.h                      |   2 +
 mm/khugepaged.c                    |   4 +-
 4 files changed, 132 insertions(+), 3 deletions(-)

diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h
index 778f5a56956c..68693eba82ef 100644
--- a/include/trace/events/huge_memory.h
+++ b/include/trace/events/huge_memory.h
@@ -40,7 +40,8 @@
 	EM( SCAN_STORE_FAILED,		"store_failed")			\
 	EM( SCAN_COPY_MC,		"copy_poisoned_page")		\
 	EM( SCAN_PAGE_FILLED,		"page_filled")			\
-	EMe(SCAN_PAGE_DIRTY_OR_WRITEBACK, "page_dirty_or_writeback")
+	EM( SCAN_PAGE_DIRTY_OR_WRITEBACK, "page_dirty_or_writeback")	\
+	EMe(SCAN_ALLOC_LIGHT_MISS,	"alloc_light_miss")
 
 #undef EM
 #undef EMe
diff --git a/mm/collapse.c b/mm/collapse.c
index 25c0f72a9a68..58c8d83f3468 100644
--- a/mm/collapse.c
+++ b/mm/collapse.c
@@ -17,6 +17,7 @@
 #include <linux/slab.h>
 #include <linux/swap.h>
 #include <linux/userfaultfd_k.h>
+#include <linux/vmstat.h>
 
 #include <asm/tlb.h>
 #include "collapse.h"
@@ -114,6 +115,12 @@
 	min(COLLAPSE_BATCH_BYTES >> (PAGE_SHIFT + COLLAPSE_MIN_MTHP_ORDER), \
 	    COLLAPSE_TABLE_WINDOWS)
 
+/* How far a candidate got, and so what a failure has to undo for it */
+enum collapse_candidate_state {
+	CAND_SELECTED,		/* collected; nothing held on its behalf yet */
+	CAND_SKIPPED,		/* refused; nothing of it left to undo */
+};
+
 /*
  * A candidate is an (addr, order) window selected for collapse.  Selection
  * counts in PTE offsets -- the bitmap it reads and the alignment it honours are
@@ -123,7 +130,12 @@
 struct collapse_candidate {
 	unsigned long addr;
 	unsigned int order;
+	/* The light allocation missed last round: this one may reclaim for it */
+	bool reclaim;
+	enum collapse_candidate_state state;
 	enum scan_result result;
+	struct folio *new_folio;
+	pgtable_t deposit;		/* PMD order: fresh table to deposit */
 };
 
 static unsigned long candidate_start(const struct collapse_candidate *cand)
@@ -218,6 +230,43 @@ static void collapse_freeze(struct vm_area_struct *vma,
 {
 }
 
+/*
+ * Allocate one candidate's destination with @gfp: a folio of its order, charged,
+ * with the memcg's deferred-split list heads in place so the install cannot need
+ * to allocate under the pmd lock.  Those heads cost only the first collapse in a
+ * memcg.
+ *
+ * A failure counts nothing and changes nothing: what a miss means is the caller's
+ * policy.
+ */
+static enum scan_result collapse_alloc(struct mm_struct *mm,
+				       struct collapse_control *cc,
+				       struct collapse_candidate *cand,
+				       gfp_t gfp)
+{
+	struct folio *folio;
+
+	folio = __folio_alloc(gfp, cand->order, collapse_find_target_node(cc),
+			      &cc->alloc_nmask);
+	if (!folio)
+		return SCAN_ALLOC_HUGE_PAGE_FAIL;
+
+	if (unlikely(mem_cgroup_charge(folio, mm, gfp)) ||
+	    folio_memcg_alloc_deferred(folio)) {
+		folio_put(folio);
+		return SCAN_CGROUP_CHARGE_FAIL;
+	}
+
+	if (is_pmd_order(cand->order)) {
+		count_vm_event(THP_COLLAPSE_ALLOC);
+		count_memcg_folio_events(folio, THP_COLLAPSE_ALLOC, 1);
+	}
+	count_mthp_stat(cand->order, MTHP_STAT_COLLAPSE_ALLOC);
+	cand->new_folio = folio;
+
+	return SCAN_SUCCEED;
+}
+
 /*
  * Allocate ahead of the freeze for the candidates whose light allocation missed
  * last round.  This is where reclaim belongs: nothing is held or frozen, so a
@@ -227,6 +276,31 @@ static void collapse_freeze(struct vm_area_struct *vma,
  */
 static void collapse_reserve(struct mm_struct *mm, struct collapse_control *cc)
 {
+	unsigned int i;
+
+	for (i = 0; i < cc->nr_candidates; i++) {
+		struct collapse_candidate *cand = &cc->candidates[i];
+		enum scan_result result;
+
+		if (!cand->reclaim)
+			continue;
+		cand->reclaim = false;
+
+		result = collapse_alloc(mm, cc, cand, cc->policy.gfp);
+		if (result == SCAN_SUCCEED)
+			continue;
+
+		if (result == SCAN_ALLOC_HUGE_PAGE_FAIL) {
+			/* Asked the allocator to try hard and it still missed */
+			if (is_pmd_order(cand->order))
+				count_vm_event(THP_COLLAPSE_ALLOC_FAILED);
+			count_mthp_stat(cand->order,
+					MTHP_STAT_COLLAPSE_ALLOC_FAILED);
+		}
+
+		cand->state = CAND_SKIPPED;
+		cand->result = result;
+	}
 }
 
 /*
@@ -235,9 +309,28 @@ static void collapse_reserve(struct mm_struct *mm, struct collapse_control *cc)
  * to strip: order-0 or not, it may reclaim and sleep, which is what the window
  * exists to keep out.  The destination folio has a light gfp to fall back on and
  * so can be deferred; this has none.
+ *
+ * A round is one table and a PMD-order window is the whole of it, so such a
+ * candidate cannot share a round: if there is one it is the only one, and it is
+ * candidates[0].  This secures one page table, never a batch of them.
  */
 static void collapse_deposit(struct mm_struct *mm, struct collapse_control *cc)
 {
+	struct collapse_candidate *cand = &cc->candidates[0];
+
+	if (!is_pmd_order(cand->order))
+		return;
+
+	VM_WARN_ON_ONCE(cc->nr_candidates != 1);
+
+	if (cand->state != CAND_SELECTED)
+		return;
+
+	cand->deposit = pte_alloc_one(mm);
+	if (!cand->deposit) {
+		cand->state = CAND_SKIPPED;
+		cand->result = SCAN_ALLOC_HUGE_PAGE_FAIL;
+	}
 }
 
 /*
@@ -253,6 +346,35 @@ static void collapse_deposit(struct mm_struct *mm, struct collapse_control *cc)
 static void collapse_provision(struct mm_struct *mm,
 			       struct collapse_control *cc)
 {
+	const gfp_t gfp = cc->policy.gfp & ~__GFP_DIRECT_RECLAIM;
+	const bool may_retry = gfp != cc->policy.gfp;
+	unsigned int i;
+
+	for (i = 0; i < cc->nr_candidates; i++) {
+		struct collapse_candidate *cand = &cc->candidates[i];
+		enum scan_result result;
+
+		if (cand->state != CAND_SELECTED || cand->new_folio)
+			continue;
+
+		result = collapse_alloc(mm, cc, cand, gfp);
+		if (result == SCAN_SUCCEED)
+			continue;
+
+		if (may_retry) {
+			/* A charge miss too: charging may reclaim when allowed */
+			cand->result = SCAN_ALLOC_LIGHT_MISS;
+		} else {
+			/* The gfp a retry would use, so this is the answer */
+			if (result == SCAN_ALLOC_HUGE_PAGE_FAIL) {
+				if (is_pmd_order(cand->order))
+					count_vm_event(THP_COLLAPSE_ALLOC_FAILED);
+				count_mthp_stat(cand->order,
+						MTHP_STAT_COLLAPSE_ALLOC_FAILED);
+			}
+			cand->result = result;
+		}
+	}
 }
 
 /*
@@ -761,7 +883,11 @@ static void collapse_add_candidate(struct collapse_control *cc,
 	cc->nr_candidates++;
 	cand->addr = addr;
 	cand->order = order;
+	cand->reclaim = false;
+	cand->state = CAND_SELECTED;
 	cand->result = SCAN_FAIL;
+	cand->new_folio = NULL;
+	cand->deposit = NULL;
 }
 
 /*
diff --git a/mm/collapse.h b/mm/collapse.h
index c61db86dc6c2..feb2e0d57339 100644
--- a/mm/collapse.h
+++ b/mm/collapse.h
@@ -46,6 +46,7 @@ enum scan_result {
 	SCAN_COPY_MC,
 	SCAN_PAGE_FILLED,
 	SCAN_PAGE_DIRTY_OR_WRITEBACK,
+	SCAN_ALLOC_LIGHT_MISS,
 };
 
 /*
@@ -148,6 +149,7 @@ unsigned long collapse_possible_orders(struct vm_area_struct *vma,
 		vm_flags_t vm_flags, enum tva_type tva_flags);
 enum scan_result find_pmd_or_thp_or_none(struct mm_struct *mm,
 		unsigned long address, pmd_t **pmd);
+int collapse_find_target_node(struct collapse_control *cc);
 bool collapse_scan_abort(int nid, struct collapse_control *cc);
 unsigned int collapse_max_ptes_none(struct collapse_control *cc,
 		struct vm_area_struct *vma, unsigned int order);
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index 43f6107c953a..50b520961b9b 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -999,7 +999,7 @@ static void collapse_policy_forced(struct collapse_policy *p)
 }
 
 #ifdef CONFIG_NUMA
-static int collapse_find_target_node(struct collapse_control *cc)
+int collapse_find_target_node(struct collapse_control *cc)
 {
 	int nid, target_node = 0, max_value = 0;
 
@@ -1018,7 +1018,7 @@ static int collapse_find_target_node(struct collapse_control *cc)
 	return target_node;
 }
 #else
-static int collapse_find_target_node(struct collapse_control *cc)
+int collapse_find_target_node(struct collapse_control *cc)
 {
 	return 0;
 }
-- 
2.54.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.