[RFC v3 14/15] mm, swap: refactor swapoff + add xswap_destroy
Baoquan He <[email protected]>
| Newsgroups | org.kvack.linux-mm,org.kernel.vger.linux-kernel |
|---|---|
| Message-ID | <[email protected]> |
1. Extract __swapoff() from sys_swapoff(): the core teardown logic now lives in __swapoff(), shared by sys_swapoff() and the new xswap_destroy(). swap_file operations are guarded with NULL check so __swapoff() works for file-less devices too. sys_swapoff() retains file-matching; a NULL guard on p->swap_file ensures xswap devices are never matched by the file path. 2. Add xswap_destroy(int type): tears down a file-less xswap device by its swap type. Validates SWP_XSWAP | SWP_WRITEOK, removes from lists, delegates to __swapoff(). 3. Add /sys/kernel/mm/xswap/destroy: write a swap type to tear down that xswap device. Requires CAP_SYS_ADMIN. Signed-off-by: Baoquan He <[email protected]> --- mm/swapfile.c | 197 ++++++++++++++++++++++++++++++++++---------------- 1 file changed, 136 insertions(+), 61 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index aa83a76773fe..61c307b6a432 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -160,6 +160,7 @@ static void xswap_debugfs_del(struct swap_info_struct *si) #ifdef CONFIG_SYSFS static int xswap_create(int percent); +static int xswap_destroy(int type); /* /sys/kernel/mm/xswap/: create. * Per-device runtime size is tuned via debugfs type<N>_cluster_limit. */ @@ -192,8 +193,33 @@ static ssize_t xswap_create_store(struct kobject *kobj, static struct kobj_attribute xswap_create_attr = __ATTR(create, 0200, NULL, xswap_create_store); +static ssize_t xswap_destroy_store(struct kobject *kobj, + struct kobj_attribute *attr, + const char *buf, size_t count) +{ + unsigned long type; + int err; + + if (!capable(CAP_SYS_ADMIN)) + return -EPERM; + + err = kstrtoul(buf, 0, &type); + if (err) + return err; + + err = xswap_destroy(type); + if (err) + return err; + + return count; +} + +static struct kobj_attribute xswap_destroy_attr = __ATTR(destroy, 0200, NULL, + xswap_destroy_store); + static struct attribute *xswap_attrs[] = { &xswap_create_attr.attr, + &xswap_destroy_attr.attr, NULL, }; @@ -3354,61 +3380,13 @@ static void flush_percpu_swap_cluster(struct swap_info_struct *si) } -SYSCALL_DEFINE1(swapoff, const char __user *, specialfile) +/* Common swap teardown after list removal; shared by sys_swapoff() and + * xswap_destroy(). + */ +static int __swapoff(struct swap_info_struct *p) { - struct swap_info_struct *p = NULL; - struct file *swap_file, *victim; - struct address_space *mapping; - struct inode *inode; - int err, found = 0; - - if (!capable(CAP_SYS_ADMIN)) - return -EPERM; - - BUG_ON(!current->mm); - - CLASS(filename, pathname)(specialfile); - victim = file_open_name(pathname, O_RDWR|O_LARGEFILE, 0); - if (IS_ERR(victim)) - return PTR_ERR(victim); - - mapping = victim->f_mapping; - spin_lock(&swap_lock); - plist_for_each_entry(p, &swap_active_head, list) { - if (p->flags & SWP_WRITEOK) { - if (p->swap_file->f_mapping == mapping) { - found = 1; - break; - } - } - } - if (!found) { - err = -EINVAL; - spin_unlock(&swap_lock); - goto out_dput; - } - - /* Refuse swapoff while the device is pinned for hibernation */ - if (p->flags & SWP_HIBERNATION) { - err = -EBUSY; - spin_unlock(&swap_lock); - goto out_dput; - } - - if (!security_vm_enough_memory_mm(current->mm, p->pages)) - vm_unacct_memory(p->pages); - else { - err = -ENOMEM; - spin_unlock(&swap_lock); - goto out_dput; - } - spin_lock(&p->lock); - del_from_avail_list(p, true); - plist_del(&p->list, &swap_active_head); - atomic_long_sub(p->pages, &nr_swap_pages); - total_swap_pages -= p->pages; - spin_unlock(&p->lock); - spin_unlock(&swap_lock); + struct file *swap_file = NULL; + int err; wait_for_allocation(p); @@ -3419,7 +3397,7 @@ SYSCALL_DEFINE1(swapoff, const char __user *, specialfile) if (err) { /* re-insert swap space back into swap_list */ reinsert_swap_info(p); - goto out_dput; + return err; } /* @@ -3462,12 +3440,14 @@ SYSCALL_DEFINE1(swapoff, const char __user *, specialfile) p->max = 0; p->cluster_info = NULL; - inode = mapping->host; + if (swap_file) { + struct inode *inode = swap_file->f_mapping->host; - inode_lock(inode); - inode->i_flags &= ~S_SWAPFILE; - inode_unlock(inode); - filp_close(swap_file, NULL); + inode_lock(inode); + inode->i_flags &= ~S_SWAPFILE; + inode_unlock(inode); + filp_close(swap_file, NULL); + } /* * Clear the SWP_USED flag after all resources are freed so that swapon @@ -3478,10 +3458,69 @@ SYSCALL_DEFINE1(swapoff, const char __user *, specialfile) p->flags = 0; spin_unlock(&swap_lock); - err = 0; atomic_inc(&proc_poll_event); wake_up_interruptible(&proc_poll_wait); + return 0; +} + +SYSCALL_DEFINE1(swapoff, const char __user *, specialfile) +{ + struct swap_info_struct *p = NULL; + struct file *victim; + struct address_space *mapping; + int err, found = 0; + + if (!capable(CAP_SYS_ADMIN)) + return -EPERM; + + BUG_ON(!current->mm); + + CLASS(filename, pathname)(specialfile); + victim = file_open_name(pathname, O_RDWR|O_LARGEFILE, 0); + if (IS_ERR(victim)) + return PTR_ERR(victim); + + mapping = victim->f_mapping; + spin_lock(&swap_lock); + plist_for_each_entry(p, &swap_active_head, list) { + if (p->flags & SWP_WRITEOK) { + if (p->swap_file && p->swap_file->f_mapping == mapping) { + found = 1; + break; + } + } + } + if (!found) { + err = -EINVAL; + spin_unlock(&swap_lock); + goto out_dput; + } + + /* Refuse swapoff while the device is pinned for hibernation */ + if (p->flags & SWP_HIBERNATION) { + err = -EBUSY; + spin_unlock(&swap_lock); + goto out_dput; + } + + if (!security_vm_enough_memory_mm(current->mm, p->pages)) + vm_unacct_memory(p->pages); + else { + err = -ENOMEM; + spin_unlock(&swap_lock); + goto out_dput; + } + spin_lock(&p->lock); + del_from_avail_list(p, true); + plist_del(&p->list, &swap_active_head); + atomic_long_sub(p->pages, &nr_swap_pages); + total_swap_pages -= p->pages; + spin_unlock(&p->lock); + spin_unlock(&swap_lock); + + err = __swapoff(p); + out_dput: filp_close(victim, NULL); return err; @@ -4364,6 +4403,42 @@ static int xswap_create(int percent) spin_unlock(&swap_lock); return error; } + +/* Tear down a file-less xswap device by its swap type. */ +static int xswap_destroy(int type) +{ + struct swap_info_struct *p; + + p = swap_type_to_info(type); + if (!p) + return -EINVAL; + + spin_lock(&swap_lock); + if (!(p->flags & SWP_WRITEOK) || !(p->flags & SWP_XSWAP)) { + spin_unlock(&swap_lock); + return -EINVAL; + } + /* Refuse swapoff while the device is pinned for hibernation */ + if (p->flags & SWP_HIBERNATION) { + spin_unlock(&swap_lock); + return -EBUSY; + } + if (!security_vm_enough_memory_mm(current->mm, p->pages)) + vm_unacct_memory(p->pages); + else { + spin_unlock(&swap_lock); + return -ENOMEM; + } + spin_lock(&p->lock); + del_from_avail_list(p, true); + plist_del(&p->list, &swap_active_head); + atomic_long_sub(p->pages, &nr_swap_pages); + total_swap_pages -= p->pages; + spin_unlock(&p->lock); + spin_unlock(&swap_lock); + + return __swapoff(p); +} #endif /* CONFIG_SYSFS */ #endif /* CONFIG_XSWAP */ -- 2.54.0