[PATCH v5 27/36] mm: add NODE_PRIVATE_CAP_USER_NUMA for userland numa controls

Gregory Price <[email protected]> Mon, 20 Jul 2026 15:34:21 -0400
Newsgroups org.kernel.vger.linux-debuggers,dev.linux.lists.damon,dev.linux.lists.driver-core,dev.linux.lists.nvdimm,org.kernel.vger.cgroups,org.kernel.vger.kvm,org.kernel.vger.linux-cxl,org.kernel.vger.linux-doc,org.kernel.vger.linux-fsdevel,org.kernel.vger.linux-kernel,org.kernel.vger.linux-kselftest,org.kvack.linux-mm
Message-ID <[email protected]>
Provide a mechanism to opt private nodes into userland numa management.

Add node_allows_user_numa() to encapsulate whether a node supports
userland NUMA controls (always true for normal nodes).

Placement - setting mempolicy via:
  - mbind()
  - set_mempolicy()
  - set_mempolicy_home_node()

  For mempolicy, enforcement lives in one place: mpol_set_nodemask()

  Private nodes are not N_MEMORY, so they are trimmed from a nodemask
  like a cpuset-trimmed node.  All-private nodemasks without a valid
  node collapse to empty and the mempolicy fails.

  home_node is not special-cased - it is only a preferred-nid hint, and
  placement is governed by the bind nodemask, so a home node pointed at
  an invalid node simply falls back via the normal fallback zonelists.

  (All the same behavior as a node/mask not intersecting cpuset.mems).

Migration - relocation of pages to/from a private node:
  - mbind(MPOL_MF_MOVE)
  - move_pages()
  - migrate_pages()

  mbind(MPOL_MF_MOVE) is a migration, so mempolicy and migration share
  a single opt-in control.

  The migration interfaces all check node eligibility and use
  ALLOC_ZONELIST_PRIVATE to allow eligible migration requests to
  move a folio to a private node..

  alloc_migration_target() carries mtc->zlsel into the allocator
  via __folio_alloc_zonelist().

Signed-off-by: Gregory Price <[email protected]>
---
 include/linux/node_private.h | 33 +++++++++++++++++++++++++++++++++
 mm/internal.h                |  1 +
 mm/mempolicy.c               | 27 ++++++++++++++++++++-------
 mm/migrate.c                 | 19 ++++++++++++++-----
 4 files changed, 68 insertions(+), 12 deletions(-)

diff --git a/include/linux/node_private.h b/include/linux/node_private.h
index f7cbae1309904..655fe9ec5cb61 100644
--- a/include/linux/node_private.h
+++ b/include/linux/node_private.h
@@ -13,6 +13,7 @@ struct page;
  * to let specific services operate on its node.
  */
 #define NODE_PRIVATE_CAP_RECLAIM	(1UL << 0)	/* allow mm reclaim */
+#define NODE_PRIVATE_CAP_USER_NUMA	(1UL << 1)	/* allow mempolicy */
 
 /**
  * struct node_private - Per-node container for N_MEMORY_PRIVATE nodes
@@ -69,6 +70,33 @@ static inline bool node_allows_reclaim(int nid)
 	return ret;
 }
 
+/**
+ * node_allows_user_numa - may userspace place or migrate memory here?
+ * @nid: the node to test
+ *
+ * Gate all userspace-directed memory operations on a private node.
+ *   - mbind()/set_mempolicy()
+ *   - move_pages()/migrate_pages()
+ *
+ * return: true for N_MEMORY and N_MEMORY_PRIVATE with CAP_USER_NUMA.
+ *         false for memoryless or opted-out private node.
+ */
+static inline bool node_allows_user_numa(int nid)
+{
+	struct node_private *np;
+	bool ret;
+
+	if (node_state(nid, N_MEMORY))
+		return true;
+	if (!node_state(nid, N_MEMORY_PRIVATE))
+		return false;
+	rcu_read_lock();
+	np = rcu_dereference(NODE_DATA(nid)->node_private);
+	ret = np && (np->caps & NODE_PRIVATE_CAP_USER_NUMA);
+	rcu_read_unlock();
+	return ret;
+}
+
 #else /* !CONFIG_NUMA */
 
 static inline bool folio_is_private_node(struct folio *folio)
@@ -91,6 +119,11 @@ static inline bool node_allows_reclaim(int nid)
 	return true;
 }
 
+static inline bool node_allows_user_numa(int nid)
+{
+	return true;
+}
+
 #endif /* CONFIG_NUMA */
 
 #if defined(CONFIG_NUMA) && defined(CONFIG_MEMORY_HOTPLUG)
diff --git a/mm/internal.h b/mm/internal.h
index 8329034ae561f..62e68acae08a3 100644
--- a/mm/internal.h
+++ b/mm/internal.h
@@ -1250,6 +1250,7 @@ struct migration_target_control {
 	nodemask_t *nmask;
 	gfp_t gfp_mask;
 	enum migrate_reason reason;
+	unsigned int alloc_flags;
 };
 
 /*
diff --git a/mm/mempolicy.c b/mm/mempolicy.c
index a3ffb09897489..fe42a510590a2 100644
--- a/mm/mempolicy.c
+++ b/mm/mempolicy.c
@@ -434,13 +434,14 @@ static int mpol_set_nodemask(struct mempolicy *pol,
 
 	/*
 	 * Private nodes are not in cpuset.mems, so they're always stripped.
-	 * Driver-allocated policies will already have MPOL_F_PRIVATE set,
-	 * if that's the case, add back in the requested set of private nodes.
+	 * Driver-allocated policies (MPOL_F_PRIVATE) and CAP_USER_NUMA private
+	 * nodes should be added back into the nodemask.
 	 */
 	for_each_node_mask(nid, *nodes) {
 		if (!node_is_private(nid))
 			continue;
-		if (pol->flags & MPOL_F_PRIVATE)
+		if ((pol->flags & MPOL_F_PRIVATE) ||
+		    node_allows_user_numa(nid))
 			node_set(nid, nsc->mask2);
 	}
 
@@ -696,7 +697,7 @@ static void queue_folios_pmd(pmd_t *pmd, struct mm_walk *walk)
 	}
 	if (!queue_folio_required(folio, qp))
 		return;
-	if (folio_is_private_node(folio))
+	if (!node_allows_user_numa(folio_nid(folio)))
 		return;
 	if (!(qp->flags & (MPOL_MF_MOVE | MPOL_MF_MOVE_ALL)) ||
 	    !vma_migratable(walk->vma) ||
@@ -752,7 +753,8 @@ static int queue_folios_pte_range(pmd_t *pmd, unsigned long addr,
 			continue;
 		}
 		folio = vm_normal_folio(vma, addr, ptent);
-		if (!folio || folio_is_private_managed(folio))
+		if (!folio || folio_is_zone_device(folio) ||
+		    !node_allows_user_numa(folio_nid(folio)))
 			continue;
 		if (folio_test_large(folio) && max_nr != 1)
 			nr = folio_pte_batch(folio, pte, ptent, max_nr);
@@ -827,7 +829,7 @@ static int queue_folios_hugetlb(pte_t *pte, unsigned long hmask,
 	folio = pfn_folio(pte_pfn(ptep));
 	if (!queue_folio_required(folio, qp))
 		goto unlock;
-	if (folio_is_private_node(folio))
+	if (!node_allows_user_numa(folio_nid(folio)))
 		goto unlock;
 	if (!(flags & (MPOL_MF_MOVE | MPOL_MF_MOVE_ALL)) ||
 	    !vma_migratable(walk->vma)) {
@@ -1412,6 +1414,8 @@ static long migrate_to_node(struct mm_struct *mm, int source, int dest,
 		.nid = dest,
 		.gfp_mask = GFP_HIGHUSER_MOVABLE | __GFP_THISNODE,
 		.reason = MR_SYSCALL,
+		.alloc_flags = node_is_private(dest) ?
+			 ALLOC_ZONELIST_PRIVATE : ALLOC_DEFAULT,
 	};
 
 	nodes_clear(nmask);
@@ -1985,9 +1989,10 @@ static int kernel_migrate_pages(pid_t pid, unsigned long maxnode,
 	struct mm_struct *mm = NULL;
 	struct task_struct *task;
 	nodemask_t task_nodes;
-	int err;
+	nodemask_t priv_ok;
 	nodemask_t *old;
 	nodemask_t *new;
+	int err, nid;
 	NODEMASK_SCRATCH(scratch);
 
 	if (!scratch)
@@ -2027,7 +2032,14 @@ static int kernel_migrate_pages(pid_t pid, unsigned long maxnode,
 	}
 	rcu_read_unlock();
 
+	/* Private nodes are stripped by cpuset checks. Allow eligible ones. */
+	nodes_clear(priv_ok);
+	for_each_node_mask(nid, *new)
+		if (node_is_private(nid) && node_allows_user_numa(nid))
+			node_set(nid, priv_ok);
+
 	task_nodes = cpuset_mems_allowed(task);
+	nodes_or(task_nodes, task_nodes, priv_ok);
 	/* Is the user allowed to access the target nodes? */
 	if (!nodes_subset(*new, task_nodes) && !capable(CAP_SYS_NICE)) {
 		err = -EPERM;
@@ -2035,6 +2047,7 @@ static int kernel_migrate_pages(pid_t pid, unsigned long maxnode,
 	}
 
 	task_nodes = cpuset_mems_allowed(current);
+	nodes_or(task_nodes, task_nodes, priv_ok);
 	if (!nodes_and(*new, *new, task_nodes))
 		goto out_put;
 
diff --git a/mm/migrate.c b/mm/migrate.c
index d20674c07b947..b548d79352a38 100644
--- a/mm/migrate.c
+++ b/mm/migrate.c
@@ -2231,7 +2231,8 @@ struct folio *alloc_migration_target(struct folio *src, unsigned long private)
 	if (is_highmem_idx(zidx) || zidx == ZONE_MOVABLE)
 		gfp_mask |= __GFP_HIGHMEM;
 
-	return __folio_alloc(gfp_mask, order, nid, mtc->nmask, ALLOC_DEFAULT);
+	return __folio_alloc(gfp_mask, order, nid, mtc->nmask,
+			     mtc->alloc_flags);
 }
 
 #ifdef CONFIG_NUMA_MIGRATION
@@ -2253,6 +2254,8 @@ static int do_move_pages_to_node(struct list_head *pagelist, int node)
 		.nid = node,
 		.gfp_mask = GFP_HIGHUSER_MOVABLE | __GFP_THISNODE,
 		.reason = MR_SYSCALL,
+		.alloc_flags = node_is_private(node) ?
+			 ALLOC_ZONELIST_PRIVATE : ALLOC_DEFAULT,
 	};
 
 	err = migrate_pages(pagelist, alloc_migration_target, NULL,
@@ -2268,7 +2271,8 @@ static int __add_folio_for_migration(struct folio *folio, int node,
 	if (is_zero_folio(folio) || is_huge_zero_folio(folio))
 		return -EFAULT;
 
-	if (folio_is_private_managed(folio))
+	if (folio_is_zone_device(folio) ||
+	    !node_allows_user_numa(folio_nid(folio)))
 		return -ENOENT;
 
 	if (folio_nid(folio) == node)
@@ -2392,11 +2396,14 @@ static int do_pages_move(struct mm_struct *mm, nodemask_t task_nodes,
 		err = -ENODEV;
 		if (node < 0 || node >= MAX_NUMNODES)
 			goto out_flush;
-		if (!node_state(node, N_MEMORY))
+
+		if (!node_allows_user_numa(node))
 			goto out_flush;
 
 		err = -EACCES;
-		if (!node_isset(node, task_nodes))
+		/* Private nodes are not partitioned by cpuset.mem */
+		if (!node_is_private(node) &&
+		    !node_isset(node, task_nodes))
 			goto out_flush;
 
 		if (current_node == NUMA_NO_NODE) {
@@ -2477,7 +2484,9 @@ static void do_pages_stat_array(struct mm_struct *mm, unsigned long nr_pages,
 		if (folio) {
 			if (is_zero_folio(folio) || is_huge_zero_folio(folio))
 				err = -EFAULT;
-			else if (folio_is_private_managed(folio))
+			else if (folio_is_zone_device(folio) ||
+				 (folio_is_private_node(folio) &&
+				  !node_allows_user_numa(folio_nid(folio))))
 				err = -ENOENT;
 			else
 				err = folio_nid(folio);
-- 
2.53.0-Meta