[PATCH 2/2] mm/memcontrol: defer final objcg release from no-lock frees

From: Karl Mehltretter

Date: Thu Oct 01 2026 - 00:44:13 EST


kfree_nolock() and free_pages_nolock() can drop the last reference to a
killed object cgroup. percpu_ref then invokes obj_cgroup_release() in
the caller's context. The callback can uncharge pages. It then takes
objcg_lock, exits the percpu reference, and schedules an RCU free. A
no-lock free can therefore enter regular locking. On PREEMPT_RT this
can take a sleeping lock while the caller holds a raw scheduler lock.

Make the release callback add the object cgroup to an NMI-safe lockless
list and queue normal irq_work. Drain the list outside the no-lock
caller's context. On PREEMPT_RT normal irq work runs in irq_workd task
context. On non-RT the work remains safe to run from hard interrupt
context, as required by the existing release path.

Fixes: af92793e52c3 ("slab: Introduce kmalloc_nolock() and kfree_nolock().")
Assisted-by: LLM
Signed-off-by: Karl Mehltretter <kmehltretter@xxxxxxxxx>
---
include/linux/memcontrol.h | 1 +
mm/memcontrol.c | 29 ++++++++++++++++++++++++++---
2 files changed, 27 insertions(+), 3 deletions(-)

diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h
index 7d1c0ce189a88..c755d946430b6 100644
--- a/include/linux/memcontrol.h
+++ b/include/linux/memcontrol.h
@@ -186,6 +186,7 @@ struct obj_cgroup {
struct percpu_ref refcnt;
struct mem_cgroup *memcg;
atomic_t nr_charged_bytes;
+ struct llist_node release_node;
union {
struct list_head list; /* protected by objcg_lock */
struct rcu_head rcu;
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
index 856a7d07586cc..63b18c7f1965f 100644
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -33,6 +33,8 @@
#include <linux/sched/mm.h>
#include <linux/shmem_fs.h>
#include <linux/hugetlb.h>
+#include <linux/irq_work.h>
+#include <linux/llist.h>
#include <linux/pagemap.h>
#include <linux/folio_batch.h>
#include <linux/vm_event_item.h>
@@ -146,9 +148,12 @@ static void memcg_uncharge_kmem(struct mem_cgroup *memcg, unsigned int nr_pages)
memcg_uncharge(memcg, nr_pages);
}

-static void obj_cgroup_release(struct percpu_ref *ref)
+static LLIST_HEAD(objcg_release_list);
+static void obj_cgroup_release_workfn(struct irq_work *work);
+static DEFINE_IRQ_WORK(objcg_release_work, obj_cgroup_release_workfn);
+
+static void obj_cgroup_release_one(struct obj_cgroup *objcg)
{
- struct obj_cgroup *objcg = container_of(ref, struct obj_cgroup, refcnt);
unsigned int nr_bytes;
unsigned int nr_pages;
unsigned long flags;
@@ -189,10 +194,28 @@ static void obj_cgroup_release(struct percpu_ref *ref)
list_del(&objcg->list);
spin_unlock_irqrestore(&objcg_lock, flags);

- percpu_ref_exit(ref);
+ percpu_ref_exit(&objcg->refcnt);
kfree_rcu(objcg, rcu);
}

+static void obj_cgroup_release_workfn(struct irq_work *work)
+{
+ struct llist_node *node;
+ struct obj_cgroup *objcg, *next;
+
+ node = llist_del_all(&objcg_release_list);
+ llist_for_each_entry_safe(objcg, next, node, release_node)
+ obj_cgroup_release_one(objcg);
+}
+
+static void obj_cgroup_release(struct percpu_ref *ref)
+{
+ struct obj_cgroup *objcg = container_of(ref, struct obj_cgroup, refcnt);
+
+ llist_add(&objcg->release_node, &objcg_release_list);
+ irq_work_queue(&objcg_release_work);
+}
+
static struct obj_cgroup *obj_cgroup_alloc(void)
{
struct obj_cgroup *objcg;
--
2.53.0