Re: [RFC v3 05/15] mm, swap: add xswap cluster grow via VM_SPARSE vmalloc

Klara Modin <[email protected]>
Newsgroups gmane.linux.kernel,gmane.linux.kernel.mm
Message-ID <[email protected]>
Hi,

On 2026-08-13 18:48:44 +0800, Baoquan He wrote:
> Implement dynamic cluster_info array growth for xswap devices using a
> VM_SPARSE vmalloc area:
> 
> 1. xswap_map_clusters(): Allocate physical pages and map them into
>    the pre-reserved VM_SPARSE KVA region via vm_area_map_pages().
> 
> 2. xswap_unmap_clusters(): Unmap pages from the VM_SPARSE area via
>    vm_area_unmap_pages() (used by the error/teardown paths, shrink
>    comes later).
> 
> 3. setup_swap_clusters_info() xswap path: Use get_vm_area(VM_SPARSE)
>    for the cluster_info array, lazily mapping only the initial chunk.
> 
> 4. free_swap_cluster_info() xswap path: Unmap all clusters and
>    free_vm_area().  Built on the refactoring in the previous patch.
> 
> 5. wait_for_allocation() xswap guard: Skip shrinker-unmapped clusters
>    beyond nr_clusters_mapped.
> 
> The grow path avoids emergency reserves via __GFP_HIGH|__GFP_NOMEMALLOC
> and wraps allocations with memalloc_noreclaim_save().  A per-device
> mutex (xswap_lock) serializes concurrent map/unmap page table
> modifications.
> 
> Signed-off-by: Baoquan He <[email protected]>
> ---
>  include/linux/swap.h |   1 +
>  mm/swapfile.c        | 257 ++++++++++++++++++++++++++++++++++++++++++-
>  2 files changed, 256 insertions(+), 2 deletions(-)
> 
> diff --git a/include/linux/swap.h b/include/linux/swap.h
> index 7ffc62a3b2d7..c824848c6cfc 100644
> --- a/include/linux/swap.h
> +++ b/include/linux/swap.h
> @@ -252,6 +252,7 @@ struct swap_info_struct {
>  	struct vm_struct	*cluster_vm;	/* VM_SPARSE area for xswap dynamic cluster_info */
>  	unsigned long		nr_clusters;	/* total cluster count for xswap */
>  	unsigned long		nr_clusters_mapped; /* currently mapped cluster count */
> +	struct mutex		xswap_lock;	/* serialize map/unmap operations */
>  #endif
>  	struct list_head free_clusters; /* free clusters list */
>  	struct list_head full_clusters; /* full clusters list */
> diff --git a/mm/swapfile.c b/mm/swapfile.c
> index 4ce30e9ecdf6..178c3b798f8e 100644
> --- a/mm/swapfile.c
> +++ b/mm/swapfile.c
> @@ -49,6 +49,25 @@
>  #include "internal.h"
>  #include "swap.h"
>  
> +#ifdef CONFIG_XSWAP
> +/*
> + * xswap: dynamically grow the cluster_info array via a VM_SPARSE area.
> + *
> + * XSWAP_GROW_CLUSTERS is the number of clusters to map in one grow
> + * operation.  It is set to the number of cluster_info structs that
> + * fit in a single page (at least 16), so that the vmalloc page table
> + * overhead is proportional to the number of clusters mapped.
> + */
> +#define XSWAP_GROW_CLUSTERS \
> +	max_t(unsigned long, PAGE_SIZE / sizeof(struct swap_cluster_info), 16)
> +
> +static int xswap_map_clusters(struct swap_info_struct *si,
> +			      unsigned long start_idx, unsigned long nr);
> +static void xswap_unmap_clusters(struct swap_info_struct *si,
> +				 unsigned long start_idx, unsigned long nr);
> +static int xswap_check_mapped(pte_t *pte, unsigned long addr, void *data);
> +#endif
> +
>  static void swap_range_alloc(struct swap_info_struct *si,
>  			     unsigned int nr_entries);
>  static bool folio_swapcache_freeable(struct folio *folio);
> @@ -2708,15 +2727,27 @@ static unsigned int find_next_to_unuse(struct swap_info_struct *si,
>  					unsigned int prev)
>  {
>  	unsigned int i;
> +	unsigned int end = si->max;
>  	unsigned long swp_tb;
>  
> +#ifdef CONFIG_XSWAP
> +	/* xswap may have shrunk and unmapped the cluster_info tail. */
> +	if (si->flags & SWP_XSWAP) {
> +		unsigned long mapped_end;
> +
> +		mapped_end = READ_ONCE(si->nr_clusters_mapped) * SWAPFILE_CLUSTER;
> +		if (mapped_end < end)
> +			end = mapped_end;
> +	}
> +#endif
> +
>  	/*
>  	 * No need for swap_lock here: we're just looking
>  	 * for whether an entry is in use, not modifying it; false
>  	 * hits are okay, and sys_swapoff() has already prevented new
>  	 * allocations from this area (while holding swap_lock).
>  	 */
> -	for (i = prev + 1; i < si->max; i++) {
> +	for (i = prev + 1; i < end; i++) {
>  		swp_tb = swap_table_get(__swap_offset_to_cluster(si, i),
>  					i % SWAPFILE_CLUSTER);
>  		if (!swp_tb_is_null(swp_tb) && !swp_tb_is_bad(swp_tb))
> @@ -2725,7 +2756,7 @@ static unsigned int find_next_to_unuse(struct swap_info_struct *si,
>  			cond_resched();
>  	}
>  
> -	if (i == si->max)
> +	if (i == end)
>  		i = 0;
>  
>  	return i;
> @@ -3041,6 +3072,13 @@ static void wait_for_allocation(struct swap_info_struct *si)
>  
>  	BUG_ON(si->flags & SWP_WRITEOK);
>  
> +#ifdef CONFIG_XSWAP
> +	/* Skip shrinker-unmapped cluster tail. */
> +	if (si->flags & SWP_XSWAP)
> +		end = min(end, READ_ONCE(si->nr_clusters_mapped) *
> +			  SWAPFILE_CLUSTER);
> +#endif
> +
>  	for (offset = 0; offset < end; offset += SWAPFILE_CLUSTER) {
>  		ci = swap_cluster_lock(si, offset);
>  		swap_cluster_unlock(ci);
> @@ -3057,6 +3095,19 @@ static void free_swap_cluster_info(struct swap_info_struct *si)
>  	if (!cluster_info)
>  		return;
>  
> +#ifdef CONFIG_XSWAP
> +	if (si->flags & SWP_XSWAP) {
> +		/* Unmap all mapped clusters and free the VM_SPARSE area */
> +		if (si->nr_clusters_mapped > 0)
> +			xswap_unmap_clusters(si, 0, si->nr_clusters_mapped);
> +		free_vm_area(si->cluster_vm);
> +		si->cluster_vm = NULL;
> +		si->nr_clusters = 0;
> +		si->nr_clusters_mapped = 0;
> +		return;
> +	}
> +#endif
> +
>  	nr_clusters = DIV_ROUND_UP(maxpages, SWAPFILE_CLUSTER);
>  	for (i = 0; i < nr_clusters; i++) {
>  		ci = cluster_info + i;
> @@ -3553,6 +3604,150 @@ static unsigned long read_swap_header(struct swap_info_struct *si,
>  	return maxpages;
>  }
>  
> +#ifdef CONFIG_XSWAP
> +static int xswap_map_clusters(struct swap_info_struct *si,
> +			      unsigned long start_idx, unsigned long nr)
> +{
> +	unsigned long start_addr = (unsigned long)si->cluster_info +
> +				   (size_t)start_idx * sizeof(struct swap_cluster_info);
> +	unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info);
> +	/* Round to page boundaries for vm_area_map_pages(). */
> +	unsigned long vm_start = PAGE_ALIGN(start_addr);
> +	unsigned long vm_end = PAGE_ALIGN(end_addr);
> +	unsigned int noreclaim_flags;
> +	unsigned long npages;
> +	struct page **pages;
> +	unsigned long i;
> +	int err;
> +
> +	mutex_lock(&si->xswap_lock);
> +
> +	if (vm_start >= vm_end) {
> +		/* All requested clusters fall within already-mapped pages. */
> +		for (i = start_idx; i < start_idx + nr; i++)
> +			spin_lock_init(&si->cluster_info[i].lock);
> +		WRITE_ONCE(si->nr_clusters_mapped, start_idx + nr);
> +		mutex_unlock(&si->xswap_lock);
> +		return 0;
> +	}
> +
> +	npages = (vm_end - vm_start) >> PAGE_SHIFT;
> +
> +	/* Prevent recursive reclaim during vmap page table allocation. */
> +	noreclaim_flags = memalloc_noreclaim_save();
> +
> +	pages = kmalloc_array(npages, sizeof(*pages),
> +			      __GFP_HIGH | __GFP_NOMEMALLOC | GFP_KERNEL);
> +	if (!pages) {
> +		memalloc_noreclaim_restore(noreclaim_flags);
> +		mutex_unlock(&si->xswap_lock);
> +		return -ENOMEM;
> +	}
> +
> +	for (i = 0; i < npages; i++) {
> +		/* __GFP_ZERO: cluster_info pointer fields must start NULL. */
> +		pages[i] = alloc_page(__GFP_HIGH | __GFP_NOMEMALLOC |
> +				      GFP_KERNEL | __GFP_ZERO);
> +		if (!pages[i])
> +			goto fail;
> +	}
> +
> +	/* Detect racing grower that already mapped these pages. */
> +	if (apply_to_existing_page_range(&init_mm, vm_start,
> +					 vm_end - vm_start,
> +					 xswap_check_mapped, NULL)) {
> +		i = npages;
> +		goto fail_nounmap;
> +	}
> +
> +	err = vm_area_map_pages(si->cluster_vm, vm_start, vm_end, pages);
> +	if (err) {
> +		/* -EBUSY: defensive, the page was already mapped. */
> +		if (err == -EBUSY) {
> +			i = npages;
> +			goto fail_nounmap;
> +		}
> +		i = npages;
> +		goto fail;
> +	}
> +
> +	kfree(pages);
> +	memalloc_noreclaim_restore(noreclaim_flags);
> +
> +	/* Initialize spinlocks for newly mapped clusters */
> +	for (i = start_idx; i < start_idx + nr; i++)
> +		spin_lock_init(&si->cluster_info[i].lock);
> +
> +	/*
> +	 * Pairs with READ_ONCE() in shrink/grow paths.
> +	 */
> +	WRITE_ONCE(si->nr_clusters_mapped, start_idx + nr);
> +	mutex_unlock(&si->xswap_lock);
> +	return 0;
> +
> +fail_nounmap:
> +	/*
> +	 * The concurrent grower already mapped the range, initialized the
> +	 * cluster spinlocks and advanced nr_clusters_mapped.  It may still
> +	 * be holding those locks while adding clusters to the free list, so
> +	 * do not touch them here; just free our unused pages.
> +	 */
> +	while (i > 0) {
> +		i--;
> +		if (pages[i])
> +			__free_page(pages[i]);
> +	}
> +	kfree(pages);
> +	memalloc_noreclaim_restore(noreclaim_flags);
> +	mutex_unlock(&si->xswap_lock);
> +	return 0;
> +
> +fail:
> +	while (i > 0) {
> +		i--;
> +		if (pages[i])
> +			__free_page(pages[i]);
> +	}
> +	memalloc_noreclaim_restore(noreclaim_flags);
> +	kfree(pages);
> +	mutex_unlock(&si->xswap_lock);
> +	return -ENOMEM;
> +}
> +
> +static void xswap_unmap_clusters(struct swap_info_struct *si,
> +				 unsigned long start_idx, unsigned long nr)
> +{
> +	unsigned long start_addr = (unsigned long)si->cluster_info +
> +				   (size_t)start_idx * sizeof(struct swap_cluster_info);
> +	unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info);
> +	/* Round to page boundaries for vm_area_unmap_pages(). */
> +	unsigned long vm_start = PAGE_ALIGN(start_addr);
> +	unsigned long vm_end = PAGE_ALIGN(end_addr);
> +
> +	mutex_lock(&si->xswap_lock);
> +
> +	if (vm_start >= vm_end) {
> +		WRITE_ONCE(si->nr_clusters_mapped, start_idx);
> +		mutex_unlock(&si->xswap_lock);
> +		return;
> +	}
> +
> +	vm_area_unmap_pages(si->cluster_vm, vm_start, vm_end);
> +	/* vm_area_unmap_pages() clears PTEs but does not free pages. */
> +	/* TODO: free backing pages via page table walk or tracking bitmap */
> +
> +	/* Pairs with READ_ONCE() in shrink/grow paths. */
> +	WRITE_ONCE(si->nr_clusters_mapped, start_idx);
> +	mutex_unlock(&si->xswap_lock);
> +}
> +
> +/* Return 1 at first present PTE to signal range is already mapped. */
> +static int xswap_check_mapped(pte_t *pte, unsigned long addr, void *data)
> +{
> +	return 1;
> +}
> +#endif /* CONFIG_XSWAP */
> +
>  static int setup_swap_clusters_info(struct swap_info_struct *si,
>  				    union swap_header *swap_header,
>  				    unsigned long maxpages)
> @@ -3562,6 +3757,64 @@ static int setup_swap_clusters_info(struct swap_info_struct *si,
>  	int err = -ENOMEM;
>  	unsigned long i;
>  
> +#ifdef CONFIG_XSWAP
> +	if (si->flags & SWP_XSWAP) {
> +		unsigned long size = PAGE_ALIGN(nr_clusters * sizeof(*cluster_info));
> +		struct vm_struct *vm;
> +
> +		vm = get_vm_area(size, VM_SPARSE);
> +		if (!vm)
> +			goto err;
> +
> +		cluster_info = vm->addr;
> +		si->cluster_vm = vm;
> +		si->nr_clusters = nr_clusters;
> +		si->cluster_info = cluster_info;

Should probably initialise the mutex here instead since
xswap_map_clusters() uses it?

> +
> +		/* Map the initial chunk (at least cluster 0) */
> +		if (xswap_map_clusters(si, 0, min_t(unsigned long,
> +					XSWAP_GROW_CLUSTERS, nr_clusters)))
> +			goto err_free_vm;

> +
> +		/* xswap: only cluster 0 slot 0 is bad */
> +		err = swap_cluster_setup_bad_slot(si, cluster_info, 0, false);
> +		if (err)
> +			goto err_unmap;
> +
> +		INIT_LIST_HEAD(&si->free_clusters);
> +		INIT_LIST_HEAD(&si->full_clusters);
> +		INIT_LIST_HEAD(&si->discard_clusters);
> +		for (i = 0; i < SWAP_NR_ORDERS; i++) {
> +			INIT_LIST_HEAD(&si->nonfull_clusters[i]);
> +			INIT_LIST_HEAD(&si->frag_clusters[i]);
> +		}
> +
> +		/* Mark mapped clusters: cluster 0 has 1 bad slot, rest free */
> +		for (i = 0; i < si->nr_clusters_mapped; i++) {
> +			struct swap_cluster_info *ci = &cluster_info[i];
> +
> +			if (i == 0) {
> +				ci->flags = CLUSTER_FLAG_NONFULL;
> +				list_add_tail(&ci->list, &si->nonfull_clusters[0]);
> +			} else {
> +				ci->flags = CLUSTER_FLAG_FREE;
> +				list_add_tail(&ci->list, &si->free_clusters);
> +			}
> +		}
> +
> +		mutex_init(&si->xswap_lock);
> +		return 0;
> +
> +err_unmap:
> +		xswap_unmap_clusters(si, 0, si->nr_clusters_mapped);
> +err_free_vm:
> +		free_vm_area(si->cluster_vm);
> +		si->cluster_vm = NULL;
> +		si->cluster_info = NULL;
> +		return err;
> +	}
> +#endif /* CONFIG_XSWAP */
> +
>  	cluster_info = kvzalloc_objs(*cluster_info, nr_clusters);
>  	if (!cluster_info)
>  		goto err;
> -- 
> 2.54.0
> 

Regards,
Klara Modin
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.