[RFC PATCH 28/57] mm/collapse: remove the mechanism the engine replaces

Kiryl Shutsemau <[email protected]>
Newsgroups org.kernel.vger.linux-kselftest,org.kernel.vger.bpf,org.kernel.vger.linux-kernel,org.kernel.vger.linux-trace-kernel,org.kvack.linux-mm
Message-ID <[email protected]>
From: "Kiryl Shutsemau (Meta)" <[email protected]>

Nothing reaches the old anonymous collapse any more: the entry point was
rewired to the engine, and every function below it lost its last caller.

Delete the chain: the scan, the mTHP order walk, the collapse itself,
isolation, swap-in, the copy with its success and failure paths, the PTE
release helpers, folio_pte_referenced() and the pmd-still-valid check.

What stays is what the file paths and MADV_COLLAPSE still call:
alloc_charge_folio() for a file collapse's destination,
hugepage_vma_revalidate() for the VMA check after MADV_COLLAPSE drops
mmap_lock, and count_collapse_event() and collapse_control_init_scan()
for the file scan.

Four tracepoints lose their only emitter here: mm_khugepaged_scan_pmd,
mm_collapse_huge_page, mm_collapse_huge_page_isolate and
mm_collapse_huge_page_swapin.  Their definitions stay, now without an
emitter, and the engine reports through mm_collapse_candidate.

Assisted-by: Claude-Code:claude-opus-5
Signed-off-by: Kiryl Shutsemau (Meta) <[email protected]>
---
 mm/khugepaged.c | 984 +-----------------------------------------------
 1 file changed, 9 insertions(+), 975 deletions(-)

diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index 895183d92fb8..6203473f4953 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -545,370 +545,6 @@ void __khugepaged_exit(struct mm_struct *mm)
 	}
 }
 
-static void collapse_control_init_scan(struct collapse_control *cc)
-{
-	memset(cc->node_load, 0, sizeof(cc->node_load));
-	nodes_clear(cc->alloc_nmask);
-	bitmap_zero(cc->eligible_ptes, MAX_PTRS_PER_PTE);
-}
-
-static void release_pte_folio(struct folio *folio)
-{
-	node_stat_mod_folio(folio,
-			NR_ISOLATED_ANON + folio_is_file_lru(folio),
-			-folio_nr_pages(folio));
-	folio_unlock(folio);
-	folio_putback_lru(folio);
-}
-
-static void release_pte_pages(pte_t *pte, pte_t *_pte,
-		struct list_head *compound_pagelist)
-{
-	struct folio *folio, *tmp;
-
-	while (--_pte >= pte) {
-		pte_t pteval = ptep_get(_pte);
-		unsigned long pfn;
-
-		if (pte_none(pteval))
-			continue;
-		VM_WARN_ON_ONCE(!pte_present(pteval));
-		pfn = pte_pfn(pteval);
-		if (is_zero_pfn(pfn))
-			continue;
-		folio = pfn_folio(pfn);
-		if (folio_test_large(folio))
-			continue;
-		release_pte_folio(folio);
-	}
-
-	list_for_each_entry_safe(folio, tmp, compound_pagelist, lru) {
-		list_del(&folio->lru);
-		release_pte_folio(folio);
-	}
-}
-
-/*
- * folio_pte_referenced() - Check if a folio or its PTE mapping was recently used
- *
- * Return: true if recent access was observed through either the folio state
- * or the current PTE mapping.
- */
-static inline bool folio_pte_referenced(struct folio *folio,
-		struct vm_area_struct *vma, unsigned long addr, pte_t pteval)
-{
-	/* The folio was referenced previously ... */
-	if (folio_test_young(folio) || folio_test_referenced(folio))
-		return true;
-	/* ... or the PTE mapping was recently used */
-	return pte_young(pteval) || mmu_notifier_test_young(vma->vm_mm, addr);
-}
-
-static void count_collapse_event(unsigned int order, enum vm_event_item vm_event,
-		enum mthp_stat_item mthp_event)
-{
-	if (is_pmd_order(order))
-		count_vm_event(vm_event);
-	count_mthp_stat(order, mthp_event);
-}
-
-static enum scan_result __collapse_huge_page_isolate(struct vm_area_struct *vma,
-		unsigned long start_addr, pte_t *pte, struct collapse_control *cc,
-		unsigned int order, struct list_head *compound_pagelist)
-{
-	const unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, order);
-	const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, order);
-	const unsigned long nr_pages = 1UL << order;
-	struct page *page = NULL;
-	struct folio *folio = NULL;
-	unsigned long addr = start_addr;
-	pte_t *_pte;
-	int none_or_zero = 0, shared = 0, referenced = 0;
-	enum scan_result result = SCAN_FAIL;
-
-	for (_pte = pte; _pte < pte + nr_pages;
-	     _pte++, addr += PAGE_SIZE) {
-		pte_t pteval = ptep_get(_pte);
-		if (pte_none_or_zero(pteval)) {
-			if (++none_or_zero > max_ptes_none) {
-				result = SCAN_EXCEED_NONE_PTE;
-				count_collapse_event(order, THP_SCAN_EXCEED_NONE_PTE,
-						     MTHP_STAT_COLLAPSE_EXCEED_NONE);
-				goto out;
-			}
-			continue;
-		}
-		if (!pte_present(pteval)) {
-			result = SCAN_PTE_NON_PRESENT;
-			goto out;
-		}
-		if (pte_uffd(pteval)) {
-			result = SCAN_PTE_UFFD;
-			goto out;
-		}
-		page = vm_normal_page(vma, addr, pteval);
-		if (unlikely(!page) || unlikely(is_zone_device_page(page))) {
-			result = SCAN_PAGE_NULL;
-			goto out;
-		}
-
-		folio = page_folio(page);
-		VM_BUG_ON_FOLIO(!folio_test_anon(folio), folio);
-
-		/*
-		 * If the vma has the VM_DROPPABLE flag, the collapse will
-		 * preserve the lazyfree property without needing to skip.
-		 */
-		if (cc->policy.skip_lazyfree && !(vma->vm_flags & VM_DROPPABLE) &&
-		    folio_test_lazyfree(folio) && !pte_dirty(pteval)) {
-			result = SCAN_PAGE_LAZYFREE;
-			goto out;
-		}
-
-		/* See collapse_scan_pmd(). */
-		if (folio_maybe_mapped_shared(folio)) {
-			/*
-			 * TODO: Support shared pages without leading to further
-			 * mTHP collapses. Currently bringing in new pages via
-			 * shared may cause a future higher order collapse on a
-			 * rescan of the same range.
-			 */
-			if (++shared > max_ptes_shared) {
-				result = SCAN_EXCEED_SHARED_PTE;
-				count_collapse_event(order, THP_SCAN_EXCEED_SHARED_PTE,
-						     MTHP_STAT_COLLAPSE_EXCEED_SHARED);
-				goto out;
-			}
-		}
-		/*
-		 * TODO: In some cases of partially-mapped folios, we'd actually
-		 * want to collapse.
-		 */
-		if (!is_pmd_order(order) && folio_order(folio) >= order) {
-			result = SCAN_PTE_MAPPED_HUGEPAGE;
-			goto out;
-		}
-
-		if (folio_test_large(folio)) {
-			struct folio *f;
-
-			/*
-			 * Check if we have dealt with the compound page
-			 * already
-			 */
-			list_for_each_entry(f, compound_pagelist, lru) {
-				if (folio == f)
-					goto next;
-			}
-		}
-
-		/*
-		 * We can do it before folio_isolate_lru because the
-		 * folio can't be freed from under us. NOTE: folio lock
-		 * is needed to serialize against split_huge_page()
-		 * when invoked from the VM.
-		 */
-		if (!folio_trylock(folio)) {
-			result = SCAN_PAGE_LOCK;
-			goto out;
-		}
-
-		/*
-		 * Check if the page has any GUP (or other external) pins.
-		 *
-		 * The page table that maps the page has been already unlinked
-		 * from the page table tree and this process cannot get
-		 * an additional pin on the page.
-		 *
-		 * New pins can come later if the page is shared across fork,
-		 * but not from this process. The other process cannot write to
-		 * the page, only trigger CoW.
-		 */
-		if (folio_expected_ref_count(folio) != folio_ref_count(folio)) {
-			folio_unlock(folio);
-			result = SCAN_PAGE_COUNT;
-			goto out;
-		}
-
-		/*
-		 * Isolate the folio to avoid collapsing a hugepage
-		 * currently in use by the VM.
-		 */
-		if (!folio_isolate_lru(folio)) {
-			folio_unlock(folio);
-			result = SCAN_DEL_PAGE_LRU;
-			goto out;
-		}
-		node_stat_mod_folio(folio,
-				NR_ISOLATED_ANON + folio_is_file_lru(folio),
-				folio_nr_pages(folio));
-		VM_BUG_ON_FOLIO(!folio_test_locked(folio), folio);
-		VM_BUG_ON_FOLIO(folio_test_lru(folio), folio);
-
-		if (folio_test_large(folio))
-			list_add_tail(&folio->lru, compound_pagelist);
-next:
-		if (cc->policy.require_referenced &&
-		    folio_pte_referenced(folio, vma, addr, pteval))
-			referenced++;
-	}
-
-	if (unlikely(cc->policy.require_referenced && !referenced)) {
-		result = SCAN_LACK_REFERENCED_PAGE;
-	} else {
-		result = SCAN_SUCCEED;
-		trace_mm_collapse_huge_page_isolate(folio, none_or_zero,
-						    referenced, result, order);
-		return result;
-	}
-out:
-	release_pte_pages(pte, _pte, compound_pagelist);
-	trace_mm_collapse_huge_page_isolate(folio, none_or_zero,
-					    referenced, result, order);
-	return result;
-}
-
-static void __collapse_huge_page_copy_succeeded(pte_t *pte,
-		struct vm_area_struct *vma, unsigned long address,
-		spinlock_t *ptl, unsigned int order,
-		struct list_head *compound_pagelist)
-{
-	const unsigned long nr_pages = 1UL << order;
-	unsigned long end = address + (PAGE_SIZE * nr_pages);
-	struct folio *src, *tmp;
-	pte_t pteval;
-	pte_t *_pte;
-	unsigned int nr_ptes;
-
-	for (_pte = pte; _pte < pte + nr_pages; _pte += nr_ptes,
-	     address += nr_ptes * PAGE_SIZE) {
-		nr_ptes = 1;
-		pteval = ptep_get(_pte);
-		if (pte_none_or_zero(pteval)) {
-			add_mm_counter(vma->vm_mm, MM_ANONPAGES, 1);
-			if (pte_none(pteval))
-				continue;
-			/*
-			 * ptl mostly unnecessary.
-			 */
-			spin_lock(ptl);
-			ptep_clear(vma->vm_mm, address, _pte);
-			spin_unlock(ptl);
-			ksm_might_unmap_zero_page(vma->vm_mm, pteval);
-		} else {
-			struct page *src_page = pte_page(pteval);
-
-			src = page_folio(src_page);
-
-			if (folio_test_large(src)) {
-				unsigned int max_nr_ptes = (end - address) >> PAGE_SHIFT;
-
-				nr_ptes = folio_pte_batch(src, _pte, pteval, max_nr_ptes);
-			} else {
-				release_pte_folio(src);
-			}
-
-			/*
-			 * ptl mostly unnecessary, but preempt has to
-			 * be disabled to update the per-cpu stats
-			 * inside folio_remove_rmap_pte().
-			 */
-			spin_lock(ptl);
-			clear_ptes(vma->vm_mm, address, _pte, nr_ptes);
-			folio_remove_rmap_ptes(src, src_page, nr_ptes, vma);
-			spin_unlock(ptl);
-			free_swap_cache(src);
-			folio_put_refs(src, nr_ptes);
-		}
-	}
-
-	list_for_each_entry_safe(src, tmp, compound_pagelist, lru) {
-		list_del(&src->lru);
-		node_stat_sub_folio(src, NR_ISOLATED_ANON +
-				folio_is_file_lru(src));
-		folio_unlock(src);
-		free_swap_cache(src);
-		folio_putback_lru(src);
-	}
-}
-
-static void __collapse_huge_page_copy_failed(pte_t *pte,
-		pmd_t *pmd, pmd_t orig_pmd, struct vm_area_struct *vma,
-		unsigned int order, struct list_head *compound_pagelist)
-{
-	const unsigned long nr_pages = 1UL << order;
-	spinlock_t *pmd_ptl;
-
-	/*
-	 * Re-establish the PMD to point to the original page table
-	 * entry. Restoring PMD needs to be done prior to releasing
-	 * pages. Since pages are still isolated and locked here,
-	 * acquiring anon_vma_lock_write() is unnecessary.
-	 */
-	pmd_ptl = pmd_lock(vma->vm_mm, pmd);
-	pmd_populate(vma->vm_mm, pmd, pmd_pgtable(orig_pmd));
-	spin_unlock(pmd_ptl);
-	/*
-	 * Release both raw and compound pages isolated
-	 * in __collapse_huge_page_isolate.
-	 */
-	release_pte_pages(pte, pte + nr_pages, compound_pagelist);
-}
-
-/*
- * __collapse_huge_page_copy - attempts to copy memory contents from raw
- * pages to a hugepage. Cleans up the raw pages if copying succeeds;
- * otherwise restores the original page table and releases isolated raw pages.
- * Returns SCAN_SUCCEED if copying succeeds, otherwise returns SCAN_COPY_MC.
- *
- * @pte: starting of the PTEs to copy from
- * @folio: the new hugepage to copy contents to
- * @pmd: pointer to the new hugepage's PMD
- * @orig_pmd: the original raw pages' PMD
- * @vma: the original raw pages' virtual memory area
- * @address: starting address to copy
- * @ptl: lock on raw pages' PTEs
- * @compound_pagelist: list that stores compound pages
- */
-static enum scan_result __collapse_huge_page_copy(pte_t *pte, struct folio *folio,
-		pmd_t *pmd, pmd_t orig_pmd, struct vm_area_struct *vma,
-		unsigned long address, spinlock_t *ptl, unsigned int order,
-		struct list_head *compound_pagelist)
-{
-	const unsigned long nr_pages = 1UL << order;
-	unsigned int i;
-	enum scan_result result = SCAN_SUCCEED;
-
-	/*
-	 * Copying pages' contents is subject to memory poison at any iteration.
-	 */
-	for (i = 0; i < nr_pages; i++) {
-		pte_t pteval = ptep_get(pte + i);
-		struct page *page = folio_page(folio, i);
-		unsigned long src_addr = address + i * PAGE_SIZE;
-		struct page *src_page;
-
-		if (pte_none_or_zero(pteval)) {
-			clear_user_highpage(page, src_addr);
-			continue;
-		}
-		src_page = pte_page(pteval);
-		if (copy_mc_user_highpage(page, src_page, src_addr, vma) > 0) {
-			result = SCAN_COPY_MC;
-			break;
-		}
-	}
-
-	if (likely(result == SCAN_SUCCEED))
-		__collapse_huge_page_copy_succeeded(pte, vma, address, ptl,
-						    order, compound_pagelist);
-	else
-		__collapse_huge_page_copy_failed(pte, pmd, orig_pmd, vma,
-						 order, compound_pagelist);
-
-	return result;
-}
-
 static void khugepaged_alloc_sleep(void)
 {
 	DEFINE_WAIT(wait);
@@ -1089,119 +725,19 @@ enum scan_result find_pmd_or_thp_or_none(struct mm_struct *mm,
 	return check_pmd_state(*pmd);
 }
 
-static enum scan_result check_pmd_still_valid(struct mm_struct *mm,
-		unsigned long address, pmd_t *pmd)
+static void count_collapse_event(unsigned int order, enum vm_event_item vm_event,
+		enum mthp_stat_item mthp_event)
 {
-	pmd_t *new_pmd;
-	enum scan_result result = find_pmd_or_thp_or_none(mm, address, &new_pmd);
-
-	if (result != SCAN_SUCCEED)
-		return result;
-	if (new_pmd != pmd)
-		return SCAN_FAIL;
-	return SCAN_SUCCEED;
+	if (is_pmd_order(order))
+		count_vm_event(vm_event);
+	count_mthp_stat(order, mthp_event);
 }
 
-/*
- * Bring missing pages in from swap, to complete THP collapse.
- * Only done if collapse_scan_pmd() believes it is worthwhile.
- *
- * For mTHP orders the function bails on the first swap entry, because
- * faulting pages back in during collapse could re-populate PTEs that
- * push a later scan over the threshold for a higher-order collapse.
- *
- * Called and returns without pte mapped or spinlocks held.
- * Returns result: if not SCAN_SUCCEED, mmap_lock has been released.
- */
-static enum scan_result __collapse_huge_page_swapin(struct mm_struct *mm,
-		struct vm_area_struct *vma, unsigned long start_addr,
-		pmd_t *pmd, int referenced, unsigned int order)
+static void collapse_control_init_scan(struct collapse_control *cc)
 {
-	int swapped_in = 0;
-	vm_fault_t ret = 0;
-	unsigned long addr, end = start_addr + (PAGE_SIZE << order);
-	enum scan_result result;
-	pte_t *pte = NULL;
-	spinlock_t *ptl;
-
-	for (addr = start_addr; addr < end; addr += PAGE_SIZE) {
-		struct vm_fault vmf = {
-			.vma = vma,
-			.address = addr,
-			.pgoff = linear_page_index(vma, addr),
-			.flags = FAULT_FLAG_ALLOW_RETRY,
-			.pmd = pmd,
-		};
-
-		if (!pte++) {
-			/*
-			 * Here the ptl is only used to check pte_same() in
-			 * do_swap_page(), so readonly version is enough.
-			 */
-			pte = pte_offset_map_ro_nolock(mm, pmd, addr, &ptl);
-			if (!pte) {
-				mmap_read_unlock(mm);
-				result = SCAN_NO_PTE_TABLE;
-				goto out;
-			}
-		}
-
-		vmf.orig_pte = ptep_get_lockless(pte);
-		if (pte_none(vmf.orig_pte) ||
-		    pte_present(vmf.orig_pte))
-			continue;
-
-		/*
-		 * TODO: Support swapin without leading to further mTHP
-		 * collapses. Currently bringing in new pages via swapin may
-		 * cause a future higher order collapse on a rescan of the same
-		 * range.
-		 */
-		if (!is_pmd_order(order)) {
-			count_mthp_stat(order, MTHP_STAT_COLLAPSE_EXCEED_SWAP);
-			pte_unmap(pte);
-			mmap_read_unlock(mm);
-			result = SCAN_EXCEED_SWAP_PTE;
-			goto out;
-		}
-
-		vmf.pte = pte;
-		vmf.ptl = ptl;
-		ret = do_swap_page(&vmf);
-		/* Which unmaps pte (after perhaps re-checking the entry) */
-		pte = NULL;
-
-		/*
-		 * do_swap_page() returns VM_FAULT_RETRY with released mmap_lock.
-		 * Note we treat VM_FAULT_RETRY as VM_FAULT_ERROR here because
-		 * we do not retry here and swap entry will remain in pagetable
-		 * resulting in later failure.
-		 */
-		if (ret & VM_FAULT_RETRY) {
-			/* Likely, but not guaranteed, that page lock failed */
-			result = SCAN_PAGE_LOCK;
-			goto out;
-		}
-		if (ret & VM_FAULT_ERROR) {
-			mmap_read_unlock(mm);
-			result = SCAN_FAIL;
-			goto out;
-		}
-		swapped_in++;
-	}
-
-	if (pte)
-		pte_unmap(pte);
-
-	/* Drain LRU cache to remove extra pin on the swapped in pages */
-	if (swapped_in)
-		lru_add_drain();
-
-	result = SCAN_SUCCEED;
-out:
-	trace_mm_collapse_huge_page_swapin(mm, swapped_in, referenced, result,
-					   order);
-	return result;
+	memset(cc->node_load, 0, sizeof(cc->node_load));
+	nodes_clear(cc->alloc_nmask);
+	bitmap_zero(cc->eligible_ptes, MAX_PTRS_PER_PTE);
 }
 
 static enum scan_result alloc_charge_folio(struct folio **foliop, struct mm_struct *mm,
@@ -1234,197 +770,6 @@ static enum scan_result alloc_charge_folio(struct folio **foliop, struct mm_stru
 	return SCAN_SUCCEED;
 }
 
-/*
- * collapse_huge_page() expects the mmap_lock to be unlocked before entering and
- * will always return with the lock unlocked, to avoid holding the mmap_lock
- * while allocating a THP, as that could trigger direct reclaim/compaction.
- * Note that the VMA must be rechecked after grabbing the mmap_lock again.
- */
-static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long start_addr,
-		int referenced, int unmapped, struct collapse_control *cc,
-		unsigned int order)
-{
-	const unsigned long pmd_addr = start_addr & HPAGE_PMD_MASK;
-	const unsigned long end_addr = start_addr + (PAGE_SIZE << order);
-	LIST_HEAD(compound_pagelist);
-	pmd_t *pmd, _pmd;
-	pte_t *pte = NULL;
-	pgtable_t pgtable;
-	struct folio *folio;
-	spinlock_t *pmd_ptl, *pte_ptl;
-	enum scan_result result = SCAN_FAIL;
-	struct vm_area_struct *vma;
-	struct mmu_notifier_range range;
-	bool anon_vma_locked = false;
-
-	result = alloc_charge_folio(&folio, mm, cc, order);
-	if (result != SCAN_SUCCEED)
-		goto out_nolock;
-
-	if (folio_memcg_alloc_deferred(folio)) {
-		result = SCAN_ALLOC_HUGE_PAGE_FAIL;
-		goto out_nolock;
-	}
-
-	mmap_read_lock(mm);
-	result = hugepage_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true,
-					 &vma, cc, order);
-	if (result != SCAN_SUCCEED) {
-		mmap_read_unlock(mm);
-		goto out_nolock;
-	}
-
-	result = find_pmd_or_thp_or_none(mm, pmd_addr, &pmd);
-	if (result != SCAN_SUCCEED) {
-		mmap_read_unlock(mm);
-		goto out_nolock;
-	}
-
-	if (unmapped) {
-		/*
-		 * __collapse_huge_page_swapin() will return with mmap_lock
-		 * released when it fails. So we jump out_nolock directly in
-		 * that case.  Continuing to collapse causes inconsistency.
-		 */
-		result = __collapse_huge_page_swapin(mm, vma, start_addr, pmd,
-						     referenced, order);
-		if (result != SCAN_SUCCEED)
-			goto out_nolock;
-	}
-
-	mmap_read_unlock(mm);
-	/*
-	 * Prevent all access to pagetables with the exception of
-	 * gup_fast later handled by the pmdp_collapse_flush() and the VM
-	 * handled by the anon_vma lock + folio lock.
-	 *
-	 * UFFDIO_MOVE is prevented to race as well thanks to the
-	 * mmap_lock.
-	 */
-	mmap_write_lock(mm);
-	result = hugepage_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true,
-					 &vma, cc, order);
-	if (result != SCAN_SUCCEED)
-		goto out_up_write;
-	/* check if the pmd is still valid */
-	vma_start_write(vma);
-	result = check_pmd_still_valid(mm, pmd_addr, pmd);
-	if (result != SCAN_SUCCEED)
-		goto out_up_write;
-
-	anon_vma_lock_write(vma->anon_vma);
-	anon_vma_locked = true;
-
-	/*
-	 * Only notify about the PTE range we will actually modify. While we
-	 * temporary unmap the whole PTE table for mTHP collapse, we'll remap
-	 * it later, leaving other PTEs effectively unmodified. The locks we
-	 * hold prevent anybody from stumbling over such temporarily unmapped
-	 * PTE tables.
-	 */
-	mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, mm, start_addr,
-				end_addr);
-	mmu_notifier_invalidate_range_start(&range);
-
-	pmd_ptl = pmd_lock(mm, pmd); /* probably unnecessary */
-	/*
-	 * This removes any huge TLB entry from the CPU so we won't allow
-	 * huge and small TLB entries for the same virtual address to
-	 * avoid the risk of CPU bugs in that area.
-	 *
-	 * Parallel GUP-fast is fine since GUP-fast will back off when
-	 * it detects PMD is changed.
-	 */
-	_pmd = pmdp_collapse_flush(vma, pmd_addr, pmd);
-	spin_unlock(pmd_ptl);
-	mmu_notifier_invalidate_range_end(&range);
-	tlb_remove_table_sync_one();
-
-	pte = pte_offset_map_lock(mm, &_pmd, start_addr, &pte_ptl);
-	if (pte) {
-		result = __collapse_huge_page_isolate(vma, start_addr, pte, cc,
-						      order, &compound_pagelist);
-		spin_unlock(pte_ptl);
-	} else {
-		result = SCAN_NO_PTE_TABLE;
-	}
-
-	if (unlikely(result != SCAN_SUCCEED)) {
-		spin_lock(pmd_ptl);
-		VM_WARN_ON_ONCE(!pmd_none(*pmd));
-		/*
-		 * We can only use set_pmd_at() when establishing
-		 * hugepmds and never for establishing regular pmds that
-		 * points to regular pagetables. Use pmd_populate() for that
-		 */
-		pmd_populate(mm, pmd, pmd_pgtable(_pmd));
-		spin_unlock(pmd_ptl);
-		goto out_up_write;
-	}
-
-	/*
-	 * For PMD collapse all pages are isolated and locked so anon_vma
-	 * rmap can't run anymore. For mTHP collapse the PMD entry has been
-	 * removed and not all pages are isolated and locked, so we must hold
-	 * the lock to prevent neighboring folios from attempting to access
-	 * this PMD until its reinstalled.
-	 */
-	if (is_pmd_order(order)) {
-		anon_vma_unlock_write(vma->anon_vma);
-		anon_vma_locked = false;
-	}
-
-	result = __collapse_huge_page_copy(pte, folio, pmd, _pmd,
-					   vma, start_addr, pte_ptl,
-					   order, &compound_pagelist);
-	if (unlikely(result != SCAN_SUCCEED))
-		goto out_up_write;
-
-	/*
-	 * The smp_wmb() inside __folio_mark_uptodate() ensures the
-	 * copy_huge_page writes become visible before the set_pmd_at()
-	 * write.
-	 */
-	__folio_mark_uptodate(folio);
-	spin_lock(pmd_ptl);
-	VM_WARN_ON_ONCE(!pmd_none(*pmd));
-	if (is_pmd_order(order)) {
-		pgtable = pmd_pgtable(_pmd);
-		pgtable_trans_huge_deposit(mm, pmd, pgtable);
-		map_anon_folio_pmd_nopf(folio, pmd, vma, pmd_addr);
-	} else {
-		/*
-		 * Some architectures (e.g. MIPS) walk the live page table in
-		 * their implementation. update_mmu_cache_range() must be called
-		 * with a valid page table hierarchy and the PTE lock held.
-		 * Acquire it nested inside pmd_ptl when they are distinct locks.
-		 */
-		if (pte_ptl != pmd_ptl)
-			spin_lock_nested(pte_ptl, SINGLE_DEPTH_NESTING);
-		pmd_populate(mm, pmd, pmd_pgtable(_pmd));
-		map_anon_folio_pte_nopf(folio, pte, vma, start_addr,
-					  /*uffd_wp=*/ false);
-		if (pte_ptl != pmd_ptl)
-			spin_unlock(pte_ptl);
-	}
-	spin_unlock(pmd_ptl);
-
-	folio = NULL;
-
-	result = SCAN_SUCCEED;
-out_up_write:
-	if (pte)
-		pte_unmap(pte);
-	if (anon_vma_locked)
-		anon_vma_unlock_write(vma->anon_vma);
-	mmap_write_unlock(mm);
-out_nolock:
-	if (folio)
-		folio_put(folio);
-	trace_mm_collapse_huge_page(mm, result == SCAN_SUCCEED, result, order);
-	return result;
-}
-
 /* Return the highest naturally aligned order that fits at @offset within a PMD. */
 unsigned int max_order_from_offset(unsigned int offset)
 {
@@ -1434,317 +779,6 @@ unsigned int max_order_from_offset(unsigned int offset)
 	return min_t(unsigned int, __ffs(offset), HPAGE_PMD_ORDER);
 }
 
-/*
- * mthp_collapse() consumes the bitmap that is generated during
- * collapse_scan_pmd() to determine what regions and mTHP orders fit best.
- *
- * Each bit in cc->eligible_ptes marks a PTE the scan accepted as a collapse
- * source. We start at the PMD order and check if it is eligible for collapse;
- * if not, we check the left and right halves of the PTE page table we are
- * examining at a lower order.
- *
- * For each of these, we determine how many PTE entries are occupied in the
- * range of PTE entries we propose to collapse, then we compare this to a
- * threshold number of PTE entries which would need to be occupied for a
- * collapse to be permitted at that order (accounting for max_ptes_none).
- *
- * If a collapse is permitted, we attempt to collapse the PTE range into a
- * mTHP.
- */
-static enum scan_result mthp_collapse(struct mm_struct *mm,
-		unsigned long address, int referenced, int unmapped,
-		struct collapse_control *cc, unsigned long enabled_orders)
-{
-	unsigned int nr_occupied_ptes, nr_ptes, max_ptes_none;
-	enum scan_result last_result = SCAN_FAIL;
-	int collapsed = 0;
-	bool alloc_failed = false;
-	unsigned long collapse_address;
-	unsigned int offset = 0;
-	unsigned int order = HPAGE_PMD_ORDER;
-
-	while (offset < HPAGE_PMD_NR) {
-		nr_ptes = 1UL << order;
-
-		if (!test_bit(order, &enabled_orders))
-			goto next_order;
-
-		max_ptes_none = collapse_max_ptes_none(cc, NULL, order);
-		nr_occupied_ptes = bitmap_weight_from(cc->eligible_ptes, offset,
-						      offset + nr_ptes);
-
-		/*
-		 * Swap PTEs accepted during the scan are counted in @unmapped,
-		 * not in the eligible bitmap. Account them for the PMD-order
-		 * candidate.
-		 */
-		if (is_pmd_order(order))
-			nr_occupied_ptes += unmapped;
-
-		if (nr_occupied_ptes >= nr_ptes - max_ptes_none) {
-			enum scan_result ret;
-
-			collapse_address = address + offset * PAGE_SIZE;
-			ret = collapse_huge_page(mm, collapse_address, referenced,
-						 unmapped, cc, order);
-
-			switch (ret) {
-			/* Cases where we continue to next collapse candidate */
-			case SCAN_SUCCEED:
-				collapsed += nr_ptes;
-				fallthrough;
-			case SCAN_PTE_MAPPED_HUGEPAGE:
-				goto next_offset;
-			/* Cases where lower orders might still succeed */
-			case SCAN_ALLOC_HUGE_PAGE_FAIL:
-				alloc_failed = true;
-				fallthrough;
-			case SCAN_LACK_REFERENCED_PAGE:
-			case SCAN_EXCEED_NONE_PTE:
-			case SCAN_EXCEED_SWAP_PTE:
-			case SCAN_EXCEED_SHARED_PTE:
-			case SCAN_PAGE_LOCK:
-			case SCAN_PAGE_COUNT:
-			case SCAN_PAGE_NULL:
-			case SCAN_DEL_PAGE_LRU:
-			case SCAN_PTE_NON_PRESENT:
-			case SCAN_PTE_UFFD:
-			case SCAN_PAGE_LAZYFREE:
-				last_result = ret;
-				goto next_order;
-			/* Cases where no further collapse is possible */
-			case SCAN_PMD_MAPPED:
-				fallthrough;
-			default:
-				last_result = ret;
-				goto done;
-			}
-		}
-
-next_order:
-		/*
-		 * Continue with the next smaller order if there is still
-		 * any smaller order enabled. When at the smallest order
-		 * we must always move to the next offset.
-		 */
-		if (order > COLLAPSE_MIN_MTHP_ORDER &&
-		    (enabled_orders & GENMASK(order - 1, 0))) {
-			order--;
-			continue;
-		}
-next_offset:
-		/*
-		 * Advance past the region we just processed and determine the
-		 * highest order we can attempt next. Since huge pages must be
-		 * naturally aligned, the max order we can attempt next is
-		 * limited by the alignment of the new offset.
-		 * E.g. if we collapsed a order-2 mTHP at offset 0, offset
-		 * becomes 4 and __ffs(4) == 2, so the next attempt starts at
-		 * order 2.
-		 */
-		offset += nr_ptes;
-		order = max_order_from_offset(offset);
-	}
-done:
-	if (collapsed)
-		return SCAN_SUCCEED;
-	if (alloc_failed)
-		return SCAN_ALLOC_HUGE_PAGE_FAIL;
-	return last_result;
-}
-
-static enum scan_result __maybe_unused
-collapse_scan_pmd(struct mm_struct *mm,
-		struct vm_area_struct *vma, unsigned long start_addr,
-		bool *lock_dropped, struct collapse_control *cc)
-{
-	const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER);
-	const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER);
-	unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER);
-	enum tva_type tva_flags = cc->policy.tva_type;
-	pmd_t *pmd;
-	pte_t *pte, *_pte, pteval;
-	int i;
-	int none_or_zero = 0, shared = 0, referenced = 0;
-	enum scan_result result = SCAN_FAIL;
-	struct page *page = NULL;
-	struct folio *folio = NULL;
-	unsigned long addr;
-	unsigned long enabled_orders;
-	spinlock_t *ptl;
-	int node = NUMA_NO_NODE, unmapped = 0;
-
-	VM_BUG_ON(start_addr & ~HPAGE_PMD_MASK);
-
-	result = find_pmd_or_thp_or_none(mm, start_addr, &pmd);
-	if (result != SCAN_SUCCEED) {
-		cc->progress++;
-		goto out;
-	}
-
-	collapse_control_init_scan(cc);
-
-	enabled_orders = collapse_possible_orders(vma, vma->vm_flags, tva_flags);
-
-	/*
-	 * If PMD is the only enabled order, enforce max_ptes_none, otherwise
-	 * scan all pages to populate the bitmap for mTHP collapse. The bitmap
-	 * is then checked again in mthp_collapse() for each attempted order.
-	 */
-	if (enabled_orders != BIT(HPAGE_PMD_ORDER))
-		max_ptes_none = KHUGEPAGED_MAX_PTES_LIMIT;
-
-	pte = pte_offset_map_lock(mm, pmd, start_addr, &ptl);
-	if (!pte) {
-		cc->progress++;
-		result = SCAN_NO_PTE_TABLE;
-		goto out;
-	}
-
-	for (i = 0; i < HPAGE_PMD_NR; i++) {
-		_pte = pte + i;
-		addr = start_addr + i * PAGE_SIZE;
-		pteval = ptep_get(_pte);
-
-		cc->progress++;
-
-		if (pte_none_or_zero(pteval)) {
-			if (++none_or_zero > max_ptes_none) {
-				result = SCAN_EXCEED_NONE_PTE;
-				count_collapse_event(HPAGE_PMD_ORDER, THP_SCAN_EXCEED_NONE_PTE,
-						     MTHP_STAT_COLLAPSE_EXCEED_NONE);
-				goto out_unmap;
-			}
-			continue;
-		}
-		if (!pte_present(pteval)) {
-			if (++unmapped > max_ptes_swap) {
-				result = SCAN_EXCEED_SWAP_PTE;
-				count_collapse_event(HPAGE_PMD_ORDER, THP_SCAN_EXCEED_SWAP_PTE,
-						     MTHP_STAT_COLLAPSE_EXCEED_SWAP);
-				goto out_unmap;
-			}
-			/*
-			 * Always be strict with uffd-wp
-			 * enabled swap entries.  Please see
-			 * comment below for pte_uffd().
-			 */
-			if (pte_swp_uffd_any(pteval)) {
-				result = SCAN_PTE_UFFD;
-				goto out_unmap;
-			}
-			continue;
-		}
-		if (pte_uffd(pteval)) {
-			/*
-			 * Don't collapse the page if any of the small
-			 * PTEs are armed with uffd write protection.
-			 * Here we can also mark the new huge pmd as
-			 * write protected if any of the small ones is
-			 * marked but that could bring unknown
-			 * userfault messages that falls outside of
-			 * the registered range.  So, just be simple.
-			 */
-			result = SCAN_PTE_UFFD;
-			goto out_unmap;
-		}
-
-		page = vm_normal_page(vma, addr, pteval);
-		if (unlikely(!page) || unlikely(is_zone_device_page(page))) {
-			result = SCAN_PAGE_NULL;
-			goto out_unmap;
-		}
-		folio = page_folio(page);
-
-		/*
-		 * If the vma has the VM_DROPPABLE flag, the collapse will
-		 * preserve the lazyfree property without needing to skip.
-		 */
-		if (cc->policy.skip_lazyfree && !(vma->vm_flags & VM_DROPPABLE) &&
-		    folio_test_lazyfree(folio) && !pte_dirty(pteval)) {
-			result = SCAN_PAGE_LAZYFREE;
-			goto out_unmap;
-		}
-
-		if (!folio_test_anon(folio)) {
-			result = SCAN_PAGE_ANON;
-			goto out_unmap;
-		}
-
-		/*
-		 * We treat a single page as shared if any part of the THP
-		 * is shared.
-		 */
-		if (folio_maybe_mapped_shared(folio)) {
-			if (++shared > max_ptes_shared) {
-				result = SCAN_EXCEED_SHARED_PTE;
-				count_collapse_event(HPAGE_PMD_ORDER, THP_SCAN_EXCEED_SHARED_PTE,
-						     MTHP_STAT_COLLAPSE_EXCEED_SHARED);
-				goto out_unmap;
-			}
-		}
-
-		__set_bit(i, cc->eligible_ptes);
-		/*
-		 * Record which node the original page is from and save this
-		 * information to cc->node_load[].
-		 * Khugepaged will allocate hugepage from the node has the max
-		 * hit record.
-		 */
-		node = folio_nid(folio);
-		if (collapse_scan_abort(node, cc)) {
-			result = SCAN_SCAN_ABORT;
-			goto out_unmap;
-		}
-		cc->node_load[node]++;
-		if (!folio_test_lru(folio)) {
-			result = SCAN_PAGE_LRU;
-			goto out_unmap;
-		}
-		if (folio_test_locked(folio)) {
-			result = SCAN_PAGE_LOCK;
-			goto out_unmap;
-		}
-
-		/*
-		 * Check if the page has any GUP (or other external) pins.
-		 *
-		 * Here the check is racy, but such case is ephemeral and
-		 * we could always retry collapse later. Anyway the same
-		 * check will be done again later the risk seems low.
-		 */
-		if (folio_expected_ref_count(folio) != folio_ref_count(folio)) {
-			result = SCAN_PAGE_COUNT;
-			goto out_unmap;
-		}
-
-		if (cc->policy.require_referenced &&
-		    folio_pte_referenced(folio, vma, addr, pteval))
-			referenced++;
-	}
-	if (cc->policy.require_referenced &&
-		   (!referenced ||
-		    (unmapped && referenced < HPAGE_PMD_NR / 2))) {
-		result = SCAN_LACK_REFERENCED_PAGE;
-	} else {
-		result = SCAN_SUCCEED;
-	}
-out_unmap:
-	pte_unmap_unlock(pte, ptl);
-	if (result == SCAN_SUCCEED) {
-		/* collapse_huge_page() expects the lock to be dropped before calling */
-		mmap_read_unlock(mm);
-		result = mthp_collapse(mm, start_addr, referenced,
-				       unmapped, cc, enabled_orders);
-		/* mmap_lock was released above, set lock_dropped */
-		*lock_dropped = true;
-	}
-out:
-	trace_mm_khugepaged_scan_pmd(mm, folio, referenced,
-				     none_or_zero, result, unmapped);
-	return result;
-}
-
 static void collect_mm_slot(struct mm_slot *slot)
 {
 	struct mm_struct *mm = slot->mm;
-- 
2.54.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.