Re: [PATCH 01/10] memcg: add support for GPU page counters. (v5)
Thomas Hellström <[email protected]> Wed, 22 Jul 2026 16:38:58 +0200
| Newsgroups | org.kernel.vger.cgroups,org.freedesktop.lists.dri-devel,org.freedesktop.lists.intel-xe |
|---|---|
| Organization | Intel Sweden AB, Registration Number: 556189-6027 |
| Message-ID | <[email protected]> |
On Mon, 2026-07-06 at 15:22 +1000, Dave Airlie wrote: > From: Dave Airlie <[email protected]> > > This introduces 2 new statistics and 3 new memcontrol APIs for > dealing > with GPU system memory allocations. > > The stats corresponds to the same stats in the global vmstat, > for number of active GPU pages, and number of pages in pools that > can be reclaimed. > > The first API charges a order of pages to a objcg, and sets > the objcg on the pages like kmem does, and updates the active/reclaim > statistic. > > The second API uncharges a page from the obj cgroup it is currently > charged > to. > > The third API allows moving a page to/from reclaim and between obj > cgroups. > When pages are added to the pool lru, this just updates accounting. > When pages are being removed from a pool lru, they can be taken from > the parent objcg so this allows them to be uncharged from there and > transferred > to a new child objcg. > > Signed-off-by: Dave Airlie <[email protected]> > --- > v2: use memcg_node_stat_items > v3: fix null ptr dereference in uncharge > v4: AI review: fix parameter names, fix problem with reclaim moving > doing wrong thing > v5: fix the build with CONFIG_MEMCG=n > --- > Documentation/admin-guide/cgroup-v2.rst | 6 ++ > include/linux/memcontrol.h | 34 ++++++++ > mm/memcontrol.c | 104 > ++++++++++++++++++++++++ > 3 files changed, 144 insertions(+) Looks like there are a couple of sashiko issues remaining for this patch: https://sashiko.dev/#/message/20260706052330.1110909-2-airlied%40gmail.com > > diff --git a/Documentation/admin-guide/cgroup-v2.rst > b/Documentation/admin-guide/cgroup-v2.rst > index 993446ab66d0..aa4f503770c5 100644 > --- a/Documentation/admin-guide/cgroup-v2.rst > +++ b/Documentation/admin-guide/cgroup-v2.rst > @@ -1573,6 +1573,12 @@ The following nested keys are defined. > vmalloc (npn) > Amount of memory used for vmap backed memory. > > + gpu_active (npn) > + Amount of system memory used for GPU devices. > + > + gpu_reclaim (npn) > + Amount of system memory cached for GPU devices. > + > shmem > Amount of cached filesystem data that is swap- > backed, > such as tmpfs, shm segments, shared anonymous > mmap()s > diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h > index e1f46a0016fc..9d057f58c9d0 100644 > --- a/include/linux/memcontrol.h > +++ b/include/linux/memcontrol.h > @@ -1583,6 +1583,7 @@ static inline void > mem_cgroup_flush_foreign(struct bdi_writeback *wb) > #endif /* CONFIG_CGROUP_WRITEBACK */ > > struct sock; > + Unrelated > #ifdef CONFIG_MEMCG > extern struct static_key_false memcg_sockets_enabled_key; > #define mem_cgroup_sockets_enabled > static_branch_unlikely(&memcg_sockets_enabled_key) > @@ -1629,6 +1630,17 @@ static inline u64 > mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg) > } > #endif > > +bool mem_cgroup_charge_gpu_page(struct obj_cgroup *objcg, struct > page *page, > + unsigned int order, > + gfp_t gfp_mask, bool reclaim); > +void mem_cgroup_uncharge_gpu_page(struct page *page, > + unsigned int order, > + bool reclaim); > +bool mem_cgroup_move_gpu_page_reclaim(struct obj_cgroup *objcg, > + struct page *page, > + unsigned int order, > + bool to_reclaim); > + > int alloc_shrinker_info(struct mem_cgroup *memcg); > void free_shrinker_info(struct mem_cgroup *memcg); > void set_shrinker_bit(struct mem_cgroup *memcg, int nid, int > shrinker_id); > @@ -1665,6 +1677,28 @@ static inline void > mem_cgroup_sk_uncharge(const struct sock *sk, > { > } > > +static inline bool mem_cgroup_charge_gpu_page(struct obj_cgroup > *objcg, > + struct page *page, > + unsigned int order, > + gfp_t gfp_mask, bool > reclaim) > +{ > + return true; > +} > + > +static inline void mem_cgroup_uncharge_gpu_page(struct page *page, > + unsigned int order, > + bool reclaim) > +{ > +} > + > +static inline bool mem_cgroup_move_gpu_page_reclaim(struct > obj_cgroup *objcg, > + struct page > *page, > + unsigned int > order, > + bool to_reclaim) > +{ > + return true; > +} > + > static inline void set_shrinker_bit(struct mem_cgroup *memcg, > int nid, int shrinker_id) > { > diff --git a/mm/memcontrol.c b/mm/memcontrol.c > index 6dc4888a90f3..4c682b91cbbe 100644 > --- a/mm/memcontrol.c > +++ b/mm/memcontrol.c > @@ -423,6 +423,8 @@ static const unsigned int memcg_node_stat_items[] > = { > #ifdef CONFIG_HUGETLB_PAGE > NR_HUGETLB, > #endif > + NR_GPU_ACTIVE, > + NR_GPU_RECLAIM, > }; > > static const unsigned int memcg_stat_items[] = { > @@ -1553,6 +1555,8 @@ static const struct memory_stat memory_stats[] > = { > { > "percpu", MEMCG_PERCPU_B }, > { > "sock", MEMCG_SOCK }, > { > "vmalloc", NR_VMALLOC }, > + { > "gpu_active", NR_GPU_ACTIVE }, > + { > "gpu_reclaim", NR_GPU_RECLAIM }, > { > "shmem", NR_SHMEM }, > #ifdef CONFIG_ZSWAP > { > "zswap", MEMCG_ZSWAP_B }, > @@ -5508,6 +5512,106 @@ void mem_cgroup_flush_workqueue(void) > flush_workqueue(memcg_wq); > } > > +/** > + * mem_cgroup_charge_gpu_page - charge a page to GPU memory tracking Nit: Patch guidelines call for () > + * @objcg: objcg to charge, NULL charges root memcg > + * @page: page to charge > + * @order: page allocation order > + * @gfp_mask: gfp mode > + * @reclaim: charge the reclaim counter instead of the active one. Return: > > + * > + * Charge the order sized @page to the objcg. Returns %true if the > charge fit within > + * @objcg's configured limit, %false if it doesn't. > + */ > +bool mem_cgroup_charge_gpu_page(struct obj_cgroup *objcg, struct > page *page, > + unsigned int order, gfp_t gfp_mask, > bool reclaim) > +{ > + unsigned int nr_pages = 1 << order; > + struct mem_cgroup *memcg = NULL; > + struct lruvec *lruvec; > + int ret; > + > + if (objcg) { > + memcg = get_mem_cgroup_from_objcg(objcg); > + > + ret = try_charge_memcg(memcg, gfp_mask, nr_pages); > + if (ret) { > + mem_cgroup_put(memcg); > + return false; > + } > + > + obj_cgroup_get(objcg); > + page_set_objcg(page, objcg); > + } > + > + lruvec = mem_cgroup_lruvec(memcg, page_pgdat(page)); > + mod_lruvec_state(lruvec, reclaim ? NR_GPU_RECLAIM : > NR_GPU_ACTIVE, nr_pages); > + > + mem_cgroup_put(memcg); > + return true; > +} > +EXPORT_SYMBOL_GPL(mem_cgroup_charge_gpu_page); > + > +/** > + * mem_cgroup_uncharge_gpu_page - uncharge a page from GPU memory > tracking > + * @page: page to uncharge > + * @order: order of the page allocation > + * @reclaim: uncharge the reclaim counter instead of the active. > + */ > +void mem_cgroup_uncharge_gpu_page(struct page *page, > + unsigned int order, bool reclaim) > +{ > + struct obj_cgroup *objcg = page_objcg(page); > + struct mem_cgroup *memcg; > + struct lruvec *lruvec; > + int nr_pages = 1 << order; > + > + memcg = objcg ? get_mem_cgroup_from_objcg(objcg) : NULL; > + > + lruvec = mem_cgroup_lruvec(memcg, page_pgdat(page)); > + mod_lruvec_state(lruvec, reclaim ? NR_GPU_RECLAIM : > NR_GPU_ACTIVE, -nr_pages); > + > + if (memcg && !mem_cgroup_is_root(memcg)) > + refill_stock(memcg, nr_pages); > + page->memcg_data = 0; > + obj_cgroup_put(objcg); > + mem_cgroup_put(memcg); > +} > +EXPORT_SYMBOL_GPL(mem_cgroup_uncharge_gpu_page); > + > +/** > + * mem_cgroup_move_gpu_reclaim - move pages from gpu to gpu reclaim > and back > + * @new_objcg: objcg to move page to, NULL if just stats update. > + * @nr_pages: number of pages to move > + * @to_reclaim: true moves pages into reclaim, false moves them back Return: > + */ > +bool mem_cgroup_move_gpu_page_reclaim(struct obj_cgroup *new_objcg, > + struct page *page, > + unsigned int order, > + bool to_reclaim) > +{ > + struct obj_cgroup *objcg = page_objcg(page); > + > + if (!objcg || !new_objcg || objcg == new_objcg) { > + struct mem_cgroup *memcg = objcg ? > get_mem_cgroup_from_objcg(objcg) : NULL; > + struct lruvec *lruvec; > + unsigned long flags; > + int nr_pages = 1 << order; > + > + lruvec = mem_cgroup_lruvec(memcg, page_pgdat(page)); > + local_irq_save(flags); > + mod_lruvec_state(lruvec, to_reclaim ? NR_GPU_RECLAIM > : NR_GPU_ACTIVE, nr_pages); > + mod_lruvec_state(lruvec, to_reclaim ? NR_GPU_ACTIVE > : NR_GPU_RECLAIM, -nr_pages); > + local_irq_restore(flags); > + mem_cgroup_put(memcg); > + return true; > + } else { > + mem_cgroup_uncharge_gpu_page(page, order, true); > + return mem_cgroup_charge_gpu_page(new_objcg, page, > order, 0, false); > + } > +} > +EXPORT_SYMBOL_GPL(mem_cgroup_move_gpu_page_reclaim); > + > static int __init cgroup_memory(char *s) > { > char *token;