[RFC PATCH 5/7] cgroup: take set_active_cgroup() time out of the cgroup's cpu.max
From: Shakeel Butt
Date: Thu Sep 24 2026 - 15:11:23 EST
Kernel work done under set_active_cgroup() shows up in the cgroup's
cpu.stat, but the cgroup's tasks still get their full cpu.max quota.
Remember the task's run time at each set_active_cgroup(). When the
active cgroup changes, charge the time used under the old one to its
cpu.max quota with cfs_bandwidth_charge(). The caller is in the root
cgroup, which has no cpu.max, so no quota is charged twice.
Signed-off-by: Shakeel Butt <shakeel.butt@xxxxxxxxx>
---
include/linux/sched.h | 2 ++
kernel/sched/core.c | 12 ++++++++++--
2 files changed, 12 insertions(+), 2 deletions(-)
diff --git a/include/linux/sched.h b/include/linux/sched.h
index 002941f60e88..7ecea9cfa702 100644
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -1358,6 +1358,8 @@ struct task_struct {
struct list_head cg_list;
/* If set, CPU time is charged here; see set_active_cgroup(): */
struct cgroup *active_cgroup;
+ /* se.sum_exec_runtime at the last set_active_cgroup(): */
+ u64 active_cgroup_start;
#ifdef CONFIG_PREEMPT_RT
struct llist_node cg_dead_lnode;
#endif /* CONFIG_PREEMPT_RT */
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index a487da494795..91fec6461e4f 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -5738,8 +5738,9 @@ unsigned long long task_sched_runtime(struct task_struct *p)
* @cgrp: the cgroup to charge, or NULL for current's own cgroup
*
* For kernel code that does work for a cgroup. The time it uses is charged
- * to @cgrp as kernel time. Returns the old value, which the caller restores
- * when done. @cgrp must stay alive until then.
+ * to @cgrp as kernel time, and taken out of @cgrp's cpu.max quota. Returns
+ * the old value, which the caller restores when done. @cgrp must stay alive
+ * until then.
*
* Only for callers in the root cgroup, like kworkers. The scheduler still
* runs the caller in its own cgroup, so from any other cgroup the time would
@@ -5753,6 +5754,7 @@ struct cgroup *set_active_cgroup(struct cgroup *cgrp)
struct cgroup *old;
struct rq_flags rf;
struct rq *rq;
+ u64 used;
WARN_ON_ONCE(!in_task());
WARN_ON_ONCE(cgrp && cgrp->root != &cgrp_dfl_root);
@@ -5768,9 +5770,15 @@ struct cgroup *set_active_cgroup(struct cgroup *cgrp)
rq->donor->sched_class->update_curr(rq);
old = p->active_cgroup;
+ used = p->se.sum_exec_runtime - p->active_cgroup_start;
+ p->active_cgroup_start = p->se.sum_exec_runtime;
psi_set_active_cgroup(p, cgrp);
task_rq_unlock(rq, p, &rf);
+ /* Take that time out of the old cgroup's cpu.max quota as well. */
+ if (old)
+ cfs_bandwidth_charge(old, used);
+
return old;
}
#endif
--
2.53.0-Meta