[RFC PATCH 5/7] mm: add runtime toggle and sysctls for async teardown

Aditya Sharma <[email protected]>
Newsgroups org.kernel.vger.linux-trace-kernel,org.kernel.vger.linux-doc,org.kernel.vger.linux-kernel,org.kvack.linux-mm
Message-ID <[email protected]>
Add the userspace interface that arms and tunes off-CPU exit teardown,
and the static key that gates it.

Introduce async_mm_teardown_key (static key, default false) and gate the
deferral path in mmput_exit() on it via static_branch_unlikely(); when
off, mmput_exit() runs __mmput() inline. The get_mm_rss() read moves
inside the static_branch_unlikely() block, alongside eligibility and
exit_aio(), so the default (key off) path costs nothing beyond the
branch itself -- no RSS scan for every exiting process on kernels or
boots that never enable the feature. Under CONFIG_SYSCTL, register
three knobs under vm. with register_sysctl_init():

  - async_mm_teardown: write 1/0 to enable/disable. A custom handler
    flips the static key, serialized by a mutex so the key state always
    matches the last value written.
  - async_mm_teardown_thresh_pages: RSS threshold (default ~64MB).
  - async_mm_teardown_max_pending_pages: backpressure cap.

The threshold defaults are set in the initcall outside any CONFIG_SYSCTL
guard, since eligible()/reserve() read them whenever the key is on; only
the toggle variable, its lock, the handler and the table depend on
CONFIG_SYSCTL. With CONFIG_SYSCTL=n there is no path to enable the key,
so the feature stays off and the thresholds are inert.

This patch is still dormant: nothing calls mmput_exit() yet, so flipping
async_mm_teardown on has no observable effect. exit_mm() is routed
through mmput_exit() later in the series; only then does enabling the key
defer teardown. The toggle and gate are introduced here so that, once
routing lands, the feature defaults off and an explicit opt-in is
required.

Signed-off-by: Aditya Sharma <[email protected]>
---
 Documentation/admin-guide/sysctl/vm.rst | 38 +++++++++++
 kernel/fork.c                           | 90 ++++++++++++++++++++-----
 mm/Kconfig                              |  3 +-
 3 files changed, 114 insertions(+), 17 deletions(-)

diff --git a/Documentation/admin-guide/sysctl/vm.rst b/Documentation/admin-guide/sysctl/vm.rst
index 5b318d17a..ccda6c75f 100644
--- a/Documentation/admin-guide/sysctl/vm.rst
+++ b/Documentation/admin-guide/sysctl/vm.rst
@@ -22,6 +22,9 @@ the writeout of dirty data to disk.
 Currently, these files are in /proc/sys/vm:
 
 - admin_reserve_kbytes
+- async_mm_teardown
+- async_mm_teardown_max_pending_pages
+- async_mm_teardown_thresh_pages
 - compact_memory
 - compaction_proactiveness
 - compact_unevictable_allowed
@@ -108,6 +111,41 @@ On x86_64 this is about 128MB.
 Changing this takes effect whenever an application requests memory.
 
 
+async_mm_teardown
+=================
+
+Available only when CONFIG_ASYNC_MM_TEARDOWN is set. Controls whether the
+address-space teardown (exit_mmap()) of large exiting processes is deferred
+to the mm_reaper kernel thread instead of running on the exiting CPU.
+
+Writing 1 enables deferral, 0 disables it. Disabling makes mmput_exit() tear
+down inline again; it does not drain mms already queued. Defaults to 0.
+
+Deferred teardown consumes CPU in the mm_reaper kernel thread, which runs in
+the root cgroup: the teardown work of a task in a CPU-limited cgroup is not
+charged to that cgroup, similar to kswapd and the oom_reaper. Consider this
+before enabling the feature on systems that rely on strict per-cgroup CPU
+accounting.
+
+
+async_mm_teardown_max_pending_pages
+===================================
+
+Available only when CONFIG_ASYNC_MM_TEARDOWN is set. Backpressure cap, in
+pages, on the total RSS of mms queued for the mm_reaper but not yet torn
+down. An exit that would push the total over this cap tears down
+synchronously instead of queuing. Defaults to totalram_pages() / 4.
+
+
+async_mm_teardown_thresh_pages
+==============================
+
+Available only when CONFIG_ASYNC_MM_TEARDOWN is set. Minimum RSS, in pages,
+for an exiting mm to be eligible for deferral to the mm_reaper. Exiting mms
+below this threshold are always torn down synchronously. Defaults to 64MB
+worth of pages.
+
+
 compact_memory
 ==============
 
diff --git a/kernel/fork.c b/kernel/fork.c
index d45418e66..0976770db 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -3414,8 +3414,13 @@ subsys_initcall(init_fork_sysctl);
 static LLIST_HEAD(mm_reaper_list);
 static DECLARE_WAIT_QUEUE_HEAD(mm_reaper_wait);
 static atomic_long_t mm_reaper_pending_pages;
+static DEFINE_STATIC_KEY_FALSE(async_mm_teardown_key);
 static unsigned long sysctl_async_mm_teardown_thresh_pages;
 static unsigned long sysctl_async_mm_teardown_max_pending_pages;
+#ifdef CONFIG_SYSCTL
+static u8 sysctl_async_mm_teardown_enabled;
+static DEFINE_MUTEX(async_mm_teardown_lock);
+#endif
 
 static bool async_mm_teardown_eligible(struct mm_struct *mm, unsigned long rss)
 {
@@ -3485,8 +3490,6 @@ static int mm_reaper(void *unused)
 
 void mmput_exit(struct mm_struct *mm)
 {
-	unsigned long rss;
-
 	might_sleep();
 	/*
 	 * Fully ordered RMW: combined with exit_mm() (or exec_mmap(), the
@@ -3498,27 +3501,80 @@ void mmput_exit(struct mm_struct *mm)
 	if (!atomic_dec_and_test(&mm->mm_users))
 		return;
 
-	rss = get_mm_rss(mm);
+	if (static_branch_unlikely(&async_mm_teardown_key)) {
+		unsigned long rss = get_mm_rss(mm);
 
-	if (async_mm_teardown_eligible(mm, rss)) {
-		/*
-		 * exit_aio() can block indefinitely. Run it here so a stuck
-		 * AIO only hangs this task (same as how it would happen in
-		 * case of synchronous mmput()) instead of stranding every
-		 * teardown queued behind this mm
-		 */
-		exit_aio(mm);
+		if (async_mm_teardown_eligible(mm, rss)) {
+			/*
+			 * exit_aio() can block indefinitely. Run it here so a stuck
+			 * AIO only hangs this task (same as how it would happen in
+			 * case of synchronous mmput()) instead of stranding every
+			 * teardown queued behind this mm
+			 */
+			exit_aio(mm);
 
-		if (async_mm_teardown_reserve(rss)) {
-			mm->async_reap_rss = rss;
-			async_mm_teardown_queue(mm);
-			return;
+			if (async_mm_teardown_reserve(rss)) {
+				mm->async_reap_rss = rss;
+				async_mm_teardown_queue(mm);
+				return;
+			}
 		}
 	}
 
 	__mmput(mm);
 }
 
+#ifdef CONFIG_SYSCTL
+static int async_mm_teardown_enabled_handler(const struct ctl_table *table,
+					     int write, void *buffer,
+					     size_t *lenp, loff_t *ppos)
+{
+	int ret;
+
+	/*
+	 * Serialize concurrent writers so the static key state always matches
+	 * the last value written to the variable.
+	 */
+	guard(mutex)(&async_mm_teardown_lock);
+
+	ret = proc_dou8vec_minmax(table, write, buffer, lenp, ppos);
+	if (ret || !write)
+		return ret;
+
+	if (sysctl_async_mm_teardown_enabled)
+		static_branch_enable(&async_mm_teardown_key);
+	else
+		static_branch_disable(&async_mm_teardown_key);
+	return 0;
+}
+
+static const struct ctl_table async_mm_teardown_table[] = {
+	{
+		.procname	= "async_mm_teardown",
+		.data		= &sysctl_async_mm_teardown_enabled,
+		.maxlen		= sizeof(u8),
+		.mode		= 0644,
+		.proc_handler	= async_mm_teardown_enabled_handler,
+		.extra1		= SYSCTL_ZERO,
+		.extra2		= SYSCTL_ONE,
+	},
+	{
+		.procname	= "async_mm_teardown_thresh_pages",
+		.data		= &sysctl_async_mm_teardown_thresh_pages,
+		.maxlen		= sizeof(unsigned long),
+		.mode		= 0644,
+		.proc_handler	= proc_doulongvec_minmax,
+	},
+	{
+		.procname	= "async_mm_teardown_max_pending_pages",
+		.data		= &sysctl_async_mm_teardown_max_pending_pages,
+		.maxlen		= sizeof(unsigned long),
+		.mode		= 0644,
+		.proc_handler	= proc_doulongvec_minmax,
+	},
+};
+#endif /* CONFIG_SYSCTL */
+
 static int __init mm_reaper_init(void)
 {
 	struct task_struct *th;
@@ -3533,7 +3589,9 @@ static int __init mm_reaper_init(void)
 	sysctl_async_mm_teardown_thresh_pages      = SZ_64M >> PAGE_SHIFT;
 	sysctl_async_mm_teardown_max_pending_pages = totalram_pages() / 4;  /* TODO: placeholder */
 	wake_up_process(th);
-
+#ifdef CONFIG_SYSCTL
+	register_sysctl_init("vm", async_mm_teardown_table);
+#endif
 	return 0;
 }
 subsys_initcall(mm_reaper_init);
diff --git a/mm/Kconfig b/mm/Kconfig
index 296b16ad4..a9d925406 100644
--- a/mm/Kconfig
+++ b/mm/Kconfig
@@ -1507,7 +1507,8 @@ config ASYNC_MM_TEARDOWN
 	help
 	  Defer the address space teardown (exit_mmap()) of large exiting processes
 	  to a kernel thread instead of running it on the exiting CPU. The feature
-	  is off by default.
+	  is off by default and is enabled at runtime via the vm.async_mm_teardown
+	  sysctl.
 
 	  If unsure, say N.
 
-- 
2.34.1
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.