[RFC PATCH 4/7] sched/fair: add cfs_bandwidth_charge() for kernel work done for a cgroup
From: Shakeel Butt
Date: Thu Sep 24 2026 - 15:23:09 EST
set_active_cgroup() charges a kernel thread's CPU time to the cgroup it
works for, but only in cpu.stat. The work still does not count against
that cgroup's cpu.max.
Add cfs_bandwidth_charge(). It takes the time out of the cpu.max pool of
the cgroup's task group and of each limited ancestor, the same way the
group's own run time is taken. The work itself is never throttled: it
has already run. Instead the group's own tasks get less time afterwards.
What the pool cannot cover now becomes debt, paid out of the next
refills. The debt is capped at one period's quota, so a burst of work
cannot starve the group for long. Changing cpu.max clears it.
Signed-off-by: Shakeel Butt <shakeel.butt@xxxxxxxxx>
---
kernel/sched/core.c | 1 +
kernel/sched/fair.c | 43 +++++++++++++++++++++++++++++++++++++++++++
kernel/sched/sched.h | 8 ++++++++
3 files changed, 52 insertions(+)
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index b9e288b76da9..a487da494795 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -9809,6 +9809,7 @@ static int tg_set_cfs_bandwidth(struct task_group *tg,
cfs_b->period = ns_to_ktime(period);
cfs_b->quota = quota;
cfs_b->burst = burst;
+ cfs_b->debt = 0;
__refill_cfs_bandwidth_runtime(cfs_b);
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 69145dda0df5..84250bd5caa2 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -6626,6 +6626,7 @@ static inline u64 sched_cfs_bandwidth_slice(void)
void __refill_cfs_bandwidth_runtime(struct cfs_bandwidth *cfs_b)
{
s64 runtime;
+ u64 pay;
if (unlikely(cfs_b->quota == RUNTIME_INF))
return;
@@ -6638,9 +6639,51 @@ void __refill_cfs_bandwidth_runtime(struct cfs_bandwidth *cfs_b)
}
cfs_b->runtime = min(cfs_b->runtime, cfs_b->quota + cfs_b->burst);
+
+ /* Pay back the kernel work charged by cfs_bandwidth_charge(). */
+ pay = min(cfs_b->runtime, cfs_b->debt);
+ cfs_b->runtime -= pay;
+ cfs_b->debt -= pay;
+
cfs_b->runtime_snap = cfs_b->runtime;
}
+/*
+ * Kernel work used @delta of CPU time for @cgrp, see set_active_cgroup().
+ * Take it out of the quota of @cgrp's task group and of each ancestor with
+ * a limit, the same way the group's own run time is taken. What the pool
+ * cannot cover now becomes debt, paid out of the next refills. The debt is
+ * capped at one period's quota, so the work cannot starve the group for
+ * long.
+ */
+void cfs_bandwidth_charge(struct cgroup *cgrp, u64 delta)
+{
+ struct task_group *tg;
+
+ if (!cfs_bandwidth_used())
+ return;
+
+ guard(rcu)();
+ tg = css_tg(cgroup_e_css(cgrp, &cpu_cgrp_subsys));
+
+ /* No limit here or above. */
+ if (READ_ONCE(tg->cfs_bandwidth.hierarchical_quota) == RUNTIME_INF)
+ return;
+
+ for (; tg; tg = tg->parent) {
+ struct cfs_bandwidth *cfs_b = &tg->cfs_bandwidth;
+ u64 take;
+
+ guard(raw_spinlock_irqsave)(&cfs_b->lock);
+ if (cfs_b->quota == RUNTIME_INF)
+ continue;
+
+ take = min(delta, cfs_b->runtime);
+ cfs_b->runtime -= take;
+ cfs_b->debt = min(cfs_b->debt + delta - take, cfs_b->quota);
+ }
+}
+
static inline struct cfs_bandwidth *tg_cfs_bandwidth(struct task_group *tg)
{
return &tg->cfs_bandwidth;
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 6c3ad70e58b8..2d1adfd7ad37 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -456,6 +456,8 @@ struct cfs_bandwidth {
u64 runtime;
u64 burst;
u64 runtime_snap;
+ /* Kernel work charged by cfs_bandwidth_charge(), not paid yet: */
+ u64 debt;
s64 hierarchical_quota;
u8 idle;
@@ -618,6 +620,12 @@ static inline bool cfs_task_bw_constrained(struct task_struct *p) { return false
#endif /* !CONFIG_CGROUP_SCHED */
+#ifdef CONFIG_CFS_BANDWIDTH
+void cfs_bandwidth_charge(struct cgroup *cgrp, u64 delta);
+#else
+static inline void cfs_bandwidth_charge(struct cgroup *cgrp, u64 delta) { }
+#endif
+
/*
* A weight of 0 or 1 can cause arithmetics problems.
* A weight of a cfs_rq is the sum of weights of which entities
--
2.53.0-Meta