Kernel work done under set_active_cgroup() shows up in the cgroup's cpu.stat, but the cgroup's tasks still get their full cpu.max quota.
Remember the task's run time at each set_active_cgroup(). When the active cgroup changes, charge the time used under the old one to its cpu.max quota with cfs_bandwidth_charge(). The caller is in the root cgroup, which has no cpu.max, so no quota is charged twice. Signed-off-by: Shakeel Butt <[email protected]> --- include/linux/sched.h | 2 ++ kernel/sched/core.c | 12 ++++++++++-- 2 files changed, 12 insertions(+), 2 deletions(-) diff --git a/include/linux/sched.h b/include/linux/sched.h index 002941f60e88..7ecea9cfa702 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1358,6 +1358,8 @@ struct task_struct { struct list_head cg_list; /* If set, CPU time is charged here; see set_active_cgroup(): */ struct cgroup *active_cgroup; + /* se.sum_exec_runtime at the last set_active_cgroup(): */ + u64 active_cgroup_start; #ifdef CONFIG_PREEMPT_RT struct llist_node cg_dead_lnode; #endif /* CONFIG_PREEMPT_RT */ diff --git a/kernel/sched/core.c b/kernel/sched/core.c index a487da494795..91fec6461e4f 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -5738,8 +5738,9 @@ unsigned long long task_sched_runtime(struct task_struct *p) * @cgrp: the cgroup to charge, or NULL for current's own cgroup * * For kernel code that does work for a cgroup. The time it uses is charged - * to @cgrp as kernel time. Returns the old value, which the caller restores - * when done. @cgrp must stay alive until then. + * to @cgrp as kernel time, and taken out of @cgrp's cpu.max quota. Returns + * the old value, which the caller restores when done. @cgrp must stay alive + * until then. * * Only for callers in the root cgroup, like kworkers. The scheduler still * runs the caller in its own cgroup, so from any other cgroup the time would @@ -5753,6 +5754,7 @@ struct cgroup *set_active_cgroup(struct cgroup *cgrp) struct cgroup *old; struct rq_flags rf; struct rq *rq; + u64 used; WARN_ON_ONCE(!in_task()); WARN_ON_ONCE(cgrp && cgrp->root != &cgrp_dfl_root); @@ -5768,9 +5770,15 @@ struct cgroup *set_active_cgroup(struct cgroup *cgrp) rq->donor->sched_class->update_curr(rq); old = p->active_cgroup; + used = p->se.sum_exec_runtime - p->active_cgroup_start; + p->active_cgroup_start = p->se.sum_exec_runtime; psi_set_active_cgroup(p, cgrp); task_rq_unlock(rq, p, &rf); + /* Take that time out of the old cgroup's cpu.max quota as well. */ + if (old) + cfs_bandwidth_charge(old, used); + return old; } #endif -- 2.53.0-Meta

