Re: [PATCH 01/10] memcg: add support for GPU page counters. (v5)

Thomas Hellström <[email protected]> Wed, 22 Jul 2026 16:38:58 +0200
Newsgroups org.kernel.vger.cgroups,org.freedesktop.lists.dri-devel,org.freedesktop.lists.intel-xe
Organization Intel Sweden AB, Registration Number: 556189-6027
Message-ID <[email protected]>
On Mon, 2026-07-06 at 15:22 +1000, Dave Airlie wrote:
> From: Dave Airlie <[email protected]>
> 
> This introduces 2 new statistics and 3 new memcontrol APIs for
> dealing
> with GPU system memory allocations.
> 
> The stats corresponds to the same stats in the global vmstat,
> for number of active GPU pages, and number of pages in pools that
> can be reclaimed.
> 
> The first API charges a order of pages to a objcg, and sets
> the objcg on the pages like kmem does, and updates the active/reclaim
> statistic.
> 
> The second API uncharges a page from the obj cgroup it is currently
> charged
> to.
> 
> The third API allows moving a page to/from reclaim and between obj
> cgroups.
> When pages are added to the pool lru, this just updates accounting.
> When pages are being removed from a pool lru, they can be taken from
> the parent objcg so this allows them to be uncharged from there and
> transferred
> to a new child objcg.
> 
> Signed-off-by: Dave Airlie <[email protected]>
> ---
> v2: use memcg_node_stat_items
> v3: fix null ptr dereference in uncharge
> v4: AI review: fix parameter names, fix problem with reclaim moving
> doing wrong thing
> v5: fix the build with CONFIG_MEMCG=n
> ---
>  Documentation/admin-guide/cgroup-v2.rst |   6 ++
>  include/linux/memcontrol.h              |  34 ++++++++
>  mm/memcontrol.c                         | 104
> ++++++++++++++++++++++++
>  3 files changed, 144 insertions(+)

Looks like there are a couple of sashiko issues remaining for this
patch:

https://sashiko.dev/#/message/20260706052330.1110909-2-airlied%40gmail.com

> 
> diff --git a/Documentation/admin-guide/cgroup-v2.rst
> b/Documentation/admin-guide/cgroup-v2.rst
> index 993446ab66d0..aa4f503770c5 100644
> --- a/Documentation/admin-guide/cgroup-v2.rst
> +++ b/Documentation/admin-guide/cgroup-v2.rst
> @@ -1573,6 +1573,12 @@ The following nested keys are defined.
>  	  vmalloc (npn)
>  		Amount of memory used for vmap backed memory.
>  
> +	  gpu_active (npn)
> +		Amount of system memory used for GPU devices.
> +
> +	  gpu_reclaim (npn)
> +		Amount of system memory cached for GPU devices.
> +
>  	  shmem
>  		Amount of cached filesystem data that is swap-
> backed,
>  		such as tmpfs, shm segments, shared anonymous
> mmap()s
> diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h
> index e1f46a0016fc..9d057f58c9d0 100644
> --- a/include/linux/memcontrol.h
> +++ b/include/linux/memcontrol.h
> @@ -1583,6 +1583,7 @@ static inline void
> mem_cgroup_flush_foreign(struct bdi_writeback *wb)
>  #endif	/* CONFIG_CGROUP_WRITEBACK */
>  
>  struct sock;
> +

Unrelated


>  #ifdef CONFIG_MEMCG
>  extern struct static_key_false memcg_sockets_enabled_key;
>  #define mem_cgroup_sockets_enabled
> static_branch_unlikely(&memcg_sockets_enabled_key)
> @@ -1629,6 +1630,17 @@ static inline u64
> mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg)
>  }
>  #endif
>  
> +bool mem_cgroup_charge_gpu_page(struct obj_cgroup *objcg, struct
> page *page,
> +			   unsigned int order,
> +			   gfp_t gfp_mask, bool reclaim);
> +void mem_cgroup_uncharge_gpu_page(struct page *page,
> +				  unsigned int order,
> +				  bool reclaim);
> +bool mem_cgroup_move_gpu_page_reclaim(struct obj_cgroup *objcg,
> +				      struct page *page,
> +				      unsigned int order,
> +				      bool to_reclaim);
> +
>  int alloc_shrinker_info(struct mem_cgroup *memcg);
>  void free_shrinker_info(struct mem_cgroup *memcg);
>  void set_shrinker_bit(struct mem_cgroup *memcg, int nid, int
> shrinker_id);
> @@ -1665,6 +1677,28 @@ static inline void
> mem_cgroup_sk_uncharge(const struct sock *sk,
>  {
>  }
>  
> +static inline bool mem_cgroup_charge_gpu_page(struct obj_cgroup
> *objcg,
> +					      struct page *page,
> +					      unsigned int order,
> +					      gfp_t gfp_mask, bool
> reclaim)
> +{
> +	return true;
> +}
> +
> +static inline void mem_cgroup_uncharge_gpu_page(struct page *page,
> +						unsigned int order,
> +						bool reclaim)
> +{
> +}
> +
> +static inline bool mem_cgroup_move_gpu_page_reclaim(struct
> obj_cgroup *objcg,
> +						    struct page
> *page,
> +						    unsigned int
> order,
> +						    bool to_reclaim)
> +{
> +	return true;
> +}
> +
>  static inline void set_shrinker_bit(struct mem_cgroup *memcg,
>  				    int nid, int shrinker_id)
>  {
> diff --git a/mm/memcontrol.c b/mm/memcontrol.c
> index 6dc4888a90f3..4c682b91cbbe 100644
> --- a/mm/memcontrol.c
> +++ b/mm/memcontrol.c
> @@ -423,6 +423,8 @@ static const unsigned int memcg_node_stat_items[]
> = {
>  #ifdef CONFIG_HUGETLB_PAGE
>  	NR_HUGETLB,
>  #endif
> +	NR_GPU_ACTIVE,
> +	NR_GPU_RECLAIM,
>  };
>  
>  static const unsigned int memcg_stat_items[] = {
> @@ -1553,6 +1555,8 @@ static const struct memory_stat memory_stats[]
> = {
>  	{
> "percpu",			MEMCG_PERCPU_B			},
>  	{
> "sock",			MEMCG_SOCK			},
>  	{
> "vmalloc",			NR_VMALLOC			},
> +	{
> "gpu_active",			NR_GPU_ACTIVE			},
> +	{
> "gpu_reclaim",		NR_GPU_RECLAIM	                },
>  	{
> "shmem",			NR_SHMEM			},
>  #ifdef CONFIG_ZSWAP
>  	{
> "zswap",			MEMCG_ZSWAP_B			},
> @@ -5508,6 +5512,106 @@ void mem_cgroup_flush_workqueue(void)
>  	flush_workqueue(memcg_wq);
>  }
>  
> +/**
> + * mem_cgroup_charge_gpu_page - charge a page to GPU memory tracking
Nit: Patch guidelines call for ()

> + * @objcg: objcg to charge, NULL charges root memcg
> + * @page: page to charge
> + * @order: page allocation order
> + * @gfp_mask: gfp mode
> + * @reclaim: charge the reclaim counter instead of the active one.

Return:

> 
> + *
> + * Charge the order sized @page to the objcg. Returns %true if the
> charge fit within
> + * @objcg's configured limit, %false if it doesn't.
> + */
> +bool mem_cgroup_charge_gpu_page(struct obj_cgroup *objcg, struct
> page *page,
> +				unsigned int order, gfp_t gfp_mask,
> bool reclaim)
> +{
> +	unsigned int nr_pages = 1 << order;
> +	struct mem_cgroup *memcg = NULL;
> +	struct lruvec *lruvec;
> +	int ret;
> +
> +	if (objcg) {
> +		memcg = get_mem_cgroup_from_objcg(objcg);
> +
> +		ret = try_charge_memcg(memcg, gfp_mask, nr_pages);
> +		if (ret) {
> +			mem_cgroup_put(memcg);
> +			return false;
> +		}
> +
> +		obj_cgroup_get(objcg);
> +		page_set_objcg(page, objcg);
> +	}
> +
> +	lruvec = mem_cgroup_lruvec(memcg, page_pgdat(page));
> +	mod_lruvec_state(lruvec, reclaim ? NR_GPU_RECLAIM :
> NR_GPU_ACTIVE, nr_pages);
> +
> +	mem_cgroup_put(memcg);
> +	return true;
> +}
> +EXPORT_SYMBOL_GPL(mem_cgroup_charge_gpu_page);
> +
> +/**
> + * mem_cgroup_uncharge_gpu_page - uncharge a page from GPU memory
> tracking
> + * @page: page to uncharge
> + * @order: order of the page allocation
> + * @reclaim: uncharge the reclaim counter instead of the active.
> + */
> +void mem_cgroup_uncharge_gpu_page(struct page *page,
> +				  unsigned int order, bool reclaim)
> +{
> +	struct obj_cgroup *objcg = page_objcg(page);
> +	struct mem_cgroup *memcg;
> +	struct lruvec *lruvec;
> +	int nr_pages = 1 << order;
> +
> +	memcg = objcg ? get_mem_cgroup_from_objcg(objcg) : NULL;
> +
> +	lruvec = mem_cgroup_lruvec(memcg, page_pgdat(page));
> +	mod_lruvec_state(lruvec, reclaim ? NR_GPU_RECLAIM :
> NR_GPU_ACTIVE, -nr_pages);
> +
> +	if (memcg && !mem_cgroup_is_root(memcg))
> +		refill_stock(memcg, nr_pages);
> +	page->memcg_data = 0;
> +	obj_cgroup_put(objcg);
> +	mem_cgroup_put(memcg);
> +}
> +EXPORT_SYMBOL_GPL(mem_cgroup_uncharge_gpu_page);
> +
> +/**
> + * mem_cgroup_move_gpu_reclaim - move pages from gpu to gpu reclaim
> and back
> + * @new_objcg: objcg to move page to, NULL if just stats update.
> + * @nr_pages: number of pages to move
> + * @to_reclaim: true moves pages into reclaim, false moves them back

Return:

> + */
> +bool mem_cgroup_move_gpu_page_reclaim(struct obj_cgroup *new_objcg,
> +				      struct page *page,
> +				      unsigned int order,
> +				      bool to_reclaim)
> +{
> +	struct obj_cgroup *objcg = page_objcg(page);
> +
> +	if (!objcg || !new_objcg || objcg == new_objcg) {
> +		struct mem_cgroup *memcg = objcg ?
> get_mem_cgroup_from_objcg(objcg) : NULL;
> +		struct lruvec *lruvec;
> +		unsigned long flags;
> +		int nr_pages = 1 << order;
> +
> +		lruvec = mem_cgroup_lruvec(memcg, page_pgdat(page));
> +		local_irq_save(flags);
> +		mod_lruvec_state(lruvec, to_reclaim ? NR_GPU_RECLAIM
> : NR_GPU_ACTIVE, nr_pages);
> +		mod_lruvec_state(lruvec, to_reclaim ? NR_GPU_ACTIVE
> : NR_GPU_RECLAIM, -nr_pages);
> +		local_irq_restore(flags);
> +		mem_cgroup_put(memcg);
> +		return true;
> +	} else {
> +		mem_cgroup_uncharge_gpu_page(page, order, true);
> +		return mem_cgroup_charge_gpu_page(new_objcg, page,
> order, 0, false);
> +	}
> +}
> +EXPORT_SYMBOL_GPL(mem_cgroup_move_gpu_page_reclaim);
> +
>  static int __init cgroup_memory(char *s)
>  {
>  	char *token;