Re: [RFC v3 05/15] mm, swap: add xswap cluster grow via VM_SPARSE vmalloc
Klara Modin <[email protected]>
| Newsgroups | gmane.linux.kernel,gmane.linux.kernel.mm |
|---|---|
| Message-ID | <[email protected]> |
Hi, On 2026-08-13 18:48:44 +0800, Baoquan He wrote: > Implement dynamic cluster_info array growth for xswap devices using a > VM_SPARSE vmalloc area: > > 1. xswap_map_clusters(): Allocate physical pages and map them into > the pre-reserved VM_SPARSE KVA region via vm_area_map_pages(). > > 2. xswap_unmap_clusters(): Unmap pages from the VM_SPARSE area via > vm_area_unmap_pages() (used by the error/teardown paths, shrink > comes later). > > 3. setup_swap_clusters_info() xswap path: Use get_vm_area(VM_SPARSE) > for the cluster_info array, lazily mapping only the initial chunk. > > 4. free_swap_cluster_info() xswap path: Unmap all clusters and > free_vm_area(). Built on the refactoring in the previous patch. > > 5. wait_for_allocation() xswap guard: Skip shrinker-unmapped clusters > beyond nr_clusters_mapped. > > The grow path avoids emergency reserves via __GFP_HIGH|__GFP_NOMEMALLOC > and wraps allocations with memalloc_noreclaim_save(). A per-device > mutex (xswap_lock) serializes concurrent map/unmap page table > modifications. > > Signed-off-by: Baoquan He <[email protected]> > --- > include/linux/swap.h | 1 + > mm/swapfile.c | 257 ++++++++++++++++++++++++++++++++++++++++++- > 2 files changed, 256 insertions(+), 2 deletions(-) > > diff --git a/include/linux/swap.h b/include/linux/swap.h > index 7ffc62a3b2d7..c824848c6cfc 100644 > --- a/include/linux/swap.h > +++ b/include/linux/swap.h > @@ -252,6 +252,7 @@ struct swap_info_struct { > struct vm_struct *cluster_vm; /* VM_SPARSE area for xswap dynamic cluster_info */ > unsigned long nr_clusters; /* total cluster count for xswap */ > unsigned long nr_clusters_mapped; /* currently mapped cluster count */ > + struct mutex xswap_lock; /* serialize map/unmap operations */ > #endif > struct list_head free_clusters; /* free clusters list */ > struct list_head full_clusters; /* full clusters list */ > diff --git a/mm/swapfile.c b/mm/swapfile.c > index 4ce30e9ecdf6..178c3b798f8e 100644 > --- a/mm/swapfile.c > +++ b/mm/swapfile.c > @@ -49,6 +49,25 @@ > #include "internal.h" > #include "swap.h" > > +#ifdef CONFIG_XSWAP > +/* > + * xswap: dynamically grow the cluster_info array via a VM_SPARSE area. > + * > + * XSWAP_GROW_CLUSTERS is the number of clusters to map in one grow > + * operation. It is set to the number of cluster_info structs that > + * fit in a single page (at least 16), so that the vmalloc page table > + * overhead is proportional to the number of clusters mapped. > + */ > +#define XSWAP_GROW_CLUSTERS \ > + max_t(unsigned long, PAGE_SIZE / sizeof(struct swap_cluster_info), 16) > + > +static int xswap_map_clusters(struct swap_info_struct *si, > + unsigned long start_idx, unsigned long nr); > +static void xswap_unmap_clusters(struct swap_info_struct *si, > + unsigned long start_idx, unsigned long nr); > +static int xswap_check_mapped(pte_t *pte, unsigned long addr, void *data); > +#endif > + > static void swap_range_alloc(struct swap_info_struct *si, > unsigned int nr_entries); > static bool folio_swapcache_freeable(struct folio *folio); > @@ -2708,15 +2727,27 @@ static unsigned int find_next_to_unuse(struct swap_info_struct *si, > unsigned int prev) > { > unsigned int i; > + unsigned int end = si->max; > unsigned long swp_tb; > > +#ifdef CONFIG_XSWAP > + /* xswap may have shrunk and unmapped the cluster_info tail. */ > + if (si->flags & SWP_XSWAP) { > + unsigned long mapped_end; > + > + mapped_end = READ_ONCE(si->nr_clusters_mapped) * SWAPFILE_CLUSTER; > + if (mapped_end < end) > + end = mapped_end; > + } > +#endif > + > /* > * No need for swap_lock here: we're just looking > * for whether an entry is in use, not modifying it; false > * hits are okay, and sys_swapoff() has already prevented new > * allocations from this area (while holding swap_lock). > */ > - for (i = prev + 1; i < si->max; i++) { > + for (i = prev + 1; i < end; i++) { > swp_tb = swap_table_get(__swap_offset_to_cluster(si, i), > i % SWAPFILE_CLUSTER); > if (!swp_tb_is_null(swp_tb) && !swp_tb_is_bad(swp_tb)) > @@ -2725,7 +2756,7 @@ static unsigned int find_next_to_unuse(struct swap_info_struct *si, > cond_resched(); > } > > - if (i == si->max) > + if (i == end) > i = 0; > > return i; > @@ -3041,6 +3072,13 @@ static void wait_for_allocation(struct swap_info_struct *si) > > BUG_ON(si->flags & SWP_WRITEOK); > > +#ifdef CONFIG_XSWAP > + /* Skip shrinker-unmapped cluster tail. */ > + if (si->flags & SWP_XSWAP) > + end = min(end, READ_ONCE(si->nr_clusters_mapped) * > + SWAPFILE_CLUSTER); > +#endif > + > for (offset = 0; offset < end; offset += SWAPFILE_CLUSTER) { > ci = swap_cluster_lock(si, offset); > swap_cluster_unlock(ci); > @@ -3057,6 +3095,19 @@ static void free_swap_cluster_info(struct swap_info_struct *si) > if (!cluster_info) > return; > > +#ifdef CONFIG_XSWAP > + if (si->flags & SWP_XSWAP) { > + /* Unmap all mapped clusters and free the VM_SPARSE area */ > + if (si->nr_clusters_mapped > 0) > + xswap_unmap_clusters(si, 0, si->nr_clusters_mapped); > + free_vm_area(si->cluster_vm); > + si->cluster_vm = NULL; > + si->nr_clusters = 0; > + si->nr_clusters_mapped = 0; > + return; > + } > +#endif > + > nr_clusters = DIV_ROUND_UP(maxpages, SWAPFILE_CLUSTER); > for (i = 0; i < nr_clusters; i++) { > ci = cluster_info + i; > @@ -3553,6 +3604,150 @@ static unsigned long read_swap_header(struct swap_info_struct *si, > return maxpages; > } > > +#ifdef CONFIG_XSWAP > +static int xswap_map_clusters(struct swap_info_struct *si, > + unsigned long start_idx, unsigned long nr) > +{ > + unsigned long start_addr = (unsigned long)si->cluster_info + > + (size_t)start_idx * sizeof(struct swap_cluster_info); > + unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info); > + /* Round to page boundaries for vm_area_map_pages(). */ > + unsigned long vm_start = PAGE_ALIGN(start_addr); > + unsigned long vm_end = PAGE_ALIGN(end_addr); > + unsigned int noreclaim_flags; > + unsigned long npages; > + struct page **pages; > + unsigned long i; > + int err; > + > + mutex_lock(&si->xswap_lock); > + > + if (vm_start >= vm_end) { > + /* All requested clusters fall within already-mapped pages. */ > + for (i = start_idx; i < start_idx + nr; i++) > + spin_lock_init(&si->cluster_info[i].lock); > + WRITE_ONCE(si->nr_clusters_mapped, start_idx + nr); > + mutex_unlock(&si->xswap_lock); > + return 0; > + } > + > + npages = (vm_end - vm_start) >> PAGE_SHIFT; > + > + /* Prevent recursive reclaim during vmap page table allocation. */ > + noreclaim_flags = memalloc_noreclaim_save(); > + > + pages = kmalloc_array(npages, sizeof(*pages), > + __GFP_HIGH | __GFP_NOMEMALLOC | GFP_KERNEL); > + if (!pages) { > + memalloc_noreclaim_restore(noreclaim_flags); > + mutex_unlock(&si->xswap_lock); > + return -ENOMEM; > + } > + > + for (i = 0; i < npages; i++) { > + /* __GFP_ZERO: cluster_info pointer fields must start NULL. */ > + pages[i] = alloc_page(__GFP_HIGH | __GFP_NOMEMALLOC | > + GFP_KERNEL | __GFP_ZERO); > + if (!pages[i]) > + goto fail; > + } > + > + /* Detect racing grower that already mapped these pages. */ > + if (apply_to_existing_page_range(&init_mm, vm_start, > + vm_end - vm_start, > + xswap_check_mapped, NULL)) { > + i = npages; > + goto fail_nounmap; > + } > + > + err = vm_area_map_pages(si->cluster_vm, vm_start, vm_end, pages); > + if (err) { > + /* -EBUSY: defensive, the page was already mapped. */ > + if (err == -EBUSY) { > + i = npages; > + goto fail_nounmap; > + } > + i = npages; > + goto fail; > + } > + > + kfree(pages); > + memalloc_noreclaim_restore(noreclaim_flags); > + > + /* Initialize spinlocks for newly mapped clusters */ > + for (i = start_idx; i < start_idx + nr; i++) > + spin_lock_init(&si->cluster_info[i].lock); > + > + /* > + * Pairs with READ_ONCE() in shrink/grow paths. > + */ > + WRITE_ONCE(si->nr_clusters_mapped, start_idx + nr); > + mutex_unlock(&si->xswap_lock); > + return 0; > + > +fail_nounmap: > + /* > + * The concurrent grower already mapped the range, initialized the > + * cluster spinlocks and advanced nr_clusters_mapped. It may still > + * be holding those locks while adding clusters to the free list, so > + * do not touch them here; just free our unused pages. > + */ > + while (i > 0) { > + i--; > + if (pages[i]) > + __free_page(pages[i]); > + } > + kfree(pages); > + memalloc_noreclaim_restore(noreclaim_flags); > + mutex_unlock(&si->xswap_lock); > + return 0; > + > +fail: > + while (i > 0) { > + i--; > + if (pages[i]) > + __free_page(pages[i]); > + } > + memalloc_noreclaim_restore(noreclaim_flags); > + kfree(pages); > + mutex_unlock(&si->xswap_lock); > + return -ENOMEM; > +} > + > +static void xswap_unmap_clusters(struct swap_info_struct *si, > + unsigned long start_idx, unsigned long nr) > +{ > + unsigned long start_addr = (unsigned long)si->cluster_info + > + (size_t)start_idx * sizeof(struct swap_cluster_info); > + unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info); > + /* Round to page boundaries for vm_area_unmap_pages(). */ > + unsigned long vm_start = PAGE_ALIGN(start_addr); > + unsigned long vm_end = PAGE_ALIGN(end_addr); > + > + mutex_lock(&si->xswap_lock); > + > + if (vm_start >= vm_end) { > + WRITE_ONCE(si->nr_clusters_mapped, start_idx); > + mutex_unlock(&si->xswap_lock); > + return; > + } > + > + vm_area_unmap_pages(si->cluster_vm, vm_start, vm_end); > + /* vm_area_unmap_pages() clears PTEs but does not free pages. */ > + /* TODO: free backing pages via page table walk or tracking bitmap */ > + > + /* Pairs with READ_ONCE() in shrink/grow paths. */ > + WRITE_ONCE(si->nr_clusters_mapped, start_idx); > + mutex_unlock(&si->xswap_lock); > +} > + > +/* Return 1 at first present PTE to signal range is already mapped. */ > +static int xswap_check_mapped(pte_t *pte, unsigned long addr, void *data) > +{ > + return 1; > +} > +#endif /* CONFIG_XSWAP */ > + > static int setup_swap_clusters_info(struct swap_info_struct *si, > union swap_header *swap_header, > unsigned long maxpages) > @@ -3562,6 +3757,64 @@ static int setup_swap_clusters_info(struct swap_info_struct *si, > int err = -ENOMEM; > unsigned long i; > > +#ifdef CONFIG_XSWAP > + if (si->flags & SWP_XSWAP) { > + unsigned long size = PAGE_ALIGN(nr_clusters * sizeof(*cluster_info)); > + struct vm_struct *vm; > + > + vm = get_vm_area(size, VM_SPARSE); > + if (!vm) > + goto err; > + > + cluster_info = vm->addr; > + si->cluster_vm = vm; > + si->nr_clusters = nr_clusters; > + si->cluster_info = cluster_info; Should probably initialise the mutex here instead since xswap_map_clusters() uses it? > + > + /* Map the initial chunk (at least cluster 0) */ > + if (xswap_map_clusters(si, 0, min_t(unsigned long, > + XSWAP_GROW_CLUSTERS, nr_clusters))) > + goto err_free_vm; > + > + /* xswap: only cluster 0 slot 0 is bad */ > + err = swap_cluster_setup_bad_slot(si, cluster_info, 0, false); > + if (err) > + goto err_unmap; > + > + INIT_LIST_HEAD(&si->free_clusters); > + INIT_LIST_HEAD(&si->full_clusters); > + INIT_LIST_HEAD(&si->discard_clusters); > + for (i = 0; i < SWAP_NR_ORDERS; i++) { > + INIT_LIST_HEAD(&si->nonfull_clusters[i]); > + INIT_LIST_HEAD(&si->frag_clusters[i]); > + } > + > + /* Mark mapped clusters: cluster 0 has 1 bad slot, rest free */ > + for (i = 0; i < si->nr_clusters_mapped; i++) { > + struct swap_cluster_info *ci = &cluster_info[i]; > + > + if (i == 0) { > + ci->flags = CLUSTER_FLAG_NONFULL; > + list_add_tail(&ci->list, &si->nonfull_clusters[0]); > + } else { > + ci->flags = CLUSTER_FLAG_FREE; > + list_add_tail(&ci->list, &si->free_clusters); > + } > + } > + > + mutex_init(&si->xswap_lock); > + return 0; > + > +err_unmap: > + xswap_unmap_clusters(si, 0, si->nr_clusters_mapped); > +err_free_vm: > + free_vm_area(si->cluster_vm); > + si->cluster_vm = NULL; > + si->cluster_info = NULL; > + return err; > + } > +#endif /* CONFIG_XSWAP */ > + > cluster_info = kvzalloc_objs(*cluster_info, nr_clusters); > if (!cluster_info) > goto err; > -- > 2.54.0 > Regards, Klara Modin