Re: [RFC 06/10] Reclaim memory from blocked kernel stacks
From: Peter Zijlstra <peterz@infradead.org>
Date: 2026-09-01 12:32:52
Also in:
linux-mm, linux-rt-devel, lkml
Subsystem:
scheduler, the rest · Maintainers:
Ingo Molnar, Peter Zijlstra, Juri Lelli, Vincent Guittot, Linus Torvalds
On Fri, Aug 28, 2026 at 03:17:47PM -0400, Steven Rostedt wrote:
1) 8072 104 update_group_capacity+0x94/0x960 2) 7968 528 update_sd_lb_stats.constprop.0+0x426/0x39b0 3) 7440 424 sched_balance_find_src_group+0x8f/0x1150 4) 7016 552 sched_balance_rq+0x934/0x4130
Bah, yeah, those on-stack statistics just keep growing. This should probably help. Very lightly tested. Also we can probably relax the assertion to bh-disabled and avoid the extra irq-disable around sched_balance_rq(). Anybody got time to play around with this? ---
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 8dff37059faf..0c83b0856a95 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c@@ -11418,6 +11418,39 @@ struct sd_lb_stats { struct sg_lb_stats local_stat; /* Statistics of the local group */ }; +struct pcpu_lb_stats { + struct sd_lb_stats sds; + struct sg_lb_stats sgs; + struct sg_lb_stats local_sgs; + struct sg_lb_stats idlest_sgs; +}; + +static DEFINE_PER_CPU(struct pcpu_lb_stats, pcpu_lb_stats); + +static inline struct sd_lb_stats *this_sds(void) +{ + lockdep_assert_irqs_disabled(); + return this_cpu_ptr(&pcpu_lb_stats.sds); +} + +static inline struct sg_lb_stats *this_sgs(void) +{ + lockdep_assert_irqs_disabled(); + return this_cpu_ptr(&pcpu_lb_stats.sgs); +} + +static inline struct sg_lb_stats *this_local_sgs(void) +{ + lockdep_assert_irqs_disabled(); + return this_cpu_ptr(&pcpu_lb_stats.local_sgs); +} + +static inline struct sg_lb_stats *this_idlest_sgs(void) +{ + lockdep_assert_irqs_disabled(); + return this_cpu_ptr(&pcpu_lb_stats.idlest_sgs); +} + static inline void init_sd_lb_stats(struct sd_lb_stats *sds) { /*
@@ -12406,12 +12439,13 @@ static struct sched_group * sched_balance_find_dst_group(struct sched_domain *sd, struct task_struct *p, int this_cpu) { struct sched_group *idlest = NULL, *local = NULL, *group = sd->groups; - struct sg_lb_stats local_sgs, tmp_sgs; + struct sg_lb_stats *local_sgs = this_local_sgs(); struct sg_lb_stats *sgs; unsigned long imbalance; - struct sg_lb_stats idlest_sgs = { - .avg_load = UINT_MAX, - .group_type = group_overloaded, + struct sg_lb_stats *idlest_sgs = this_idlest_sgs(); + *idlest_sgs = (struct sg_lb_stats){ + .avg_load = UINT_MAX, + .group_type = group_overloaded, }; do {
@@ -12430,17 +12464,17 @@ sched_balance_find_dst_group(struct sched_domain *sd, struct task_struct *p, int sched_group_span(group)); if (local_group) { - sgs = &local_sgs; + sgs = local_sgs; local = group; } else { - sgs = &tmp_sgs; + sgs = this_sgs(); } update_sg_wakeup_stats(sd, group, sgs, p); - if (!local_group && update_pick_idlest(idlest, &idlest_sgs, group, sgs)) { + if (!local_group && update_pick_idlest(idlest, idlest_sgs, group, sgs)) { idlest = group; - idlest_sgs = *sgs; + *idlest_sgs = *sgs; } } while (group = group->next, group != sd->groups);
@@ -12458,17 +12492,17 @@ sched_balance_find_dst_group(struct sched_domain *sd, struct task_struct *p, int * If the local group is idler than the selected idlest group * don't try and push the task. */ - if (local_sgs.group_type < idlest_sgs.group_type) + if (local_sgs->group_type < idlest_sgs->group_type) return NULL; /* * If the local group is busier than the selected idlest group * try and push the task. */ - if (local_sgs.group_type > idlest_sgs.group_type) + if (local_sgs->group_type > idlest_sgs->group_type) return idlest; - switch (local_sgs.group_type) { + switch (local_sgs->group_type) { case group_overloaded: case group_fully_busy:
@@ -12486,17 +12520,17 @@ sched_balance_find_dst_group(struct sched_domain *sd, struct task_struct *p, int */ if ((sd->flags & SD_NUMA) && - ((idlest_sgs.avg_load + imbalance) >= local_sgs.avg_load)) + ((idlest_sgs->avg_load + imbalance) >= local_sgs->avg_load)) return NULL; /* * If the local group is less loaded than the selected * idlest group don't try and push any tasks. */ - if (idlest_sgs.avg_load >= (local_sgs.avg_load + imbalance)) + if (idlest_sgs->avg_load >= (local_sgs->avg_load + imbalance)) return NULL; - if (100 * local_sgs.avg_load <= sd->imbalance_pct * idlest_sgs.avg_load) + if (100 * local_sgs->avg_load <= sd->imbalance_pct * idlest_sgs->avg_load) return NULL; break;
@@ -12545,9 +12579,9 @@ sched_balance_find_dst_group(struct sched_domain *sd, struct task_struct *p, int imb_numa_nr = min(w, sd->imb_numa_nr); } - imbalance = abs(local_sgs.idle_cpus - idlest_sgs.idle_cpus); + imbalance = abs(local_sgs->idle_cpus - idlest_sgs->idle_cpus); if (!adjust_numa_imbalance(imbalance, - local_sgs.sum_nr_running + 1, + local_sgs->sum_nr_running + 1, imb_numa_nr)) { return NULL; }
@@ -12560,7 +12594,7 @@ sched_balance_find_dst_group(struct sched_domain *sd, struct task_struct *p, int * up that the group has less spare capacity but finally more * idle CPUs which means more opportunity to run task. */ - if (local_sgs.idle_cpus >= idlest_sgs.idle_cpus) + if (local_sgs->idle_cpus >= idlest_sgs->idle_cpus) return NULL; break; }
@@ -12647,14 +12681,13 @@ static inline void update_sd_lb_stats(struct lb_env *env, struct sd_lb_stats *sd { struct sched_group *sg = env->sd->groups; struct sg_lb_stats *local = &sds->local_stat; - struct sg_lb_stats tmp_sgs; unsigned long sum_util = 0; bool sg_overloaded = 0, sg_overutilized = 0; env->dst_core_idle = !sched_smt_active() || is_core_idle(env->dst_cpu); do { - struct sg_lb_stats *sgs = &tmp_sgs; + struct sg_lb_stats *sgs = this_sgs(); int local_group; local_group = cpumask_test_cpu(env->dst_cpu, sched_group_span(sg));
@@ -12929,21 +12962,21 @@ static inline void calculate_imbalance(struct lb_env *env, struct sd_lb_stats *s static struct sched_group *sched_balance_find_src_group(struct lb_env *env) { struct sg_lb_stats *local, *busiest; - struct sd_lb_stats sds; + struct sd_lb_stats *sds = this_sds(); - init_sd_lb_stats(&sds); + init_sd_lb_stats(sds); /* * Compute the various statistics relevant for load balancing at * this level. */ - update_sd_lb_stats(env, &sds); + update_sd_lb_stats(env, sds); /* There is no busy sibling group to pull tasks from */ - if (!sds.busiest) + if (!sds->busiest) goto out_balanced; - busiest = &sds.busiest_stat; + busiest = &sds->busiest_stat; /* Misfit tasks should be dealt with regardless of the avg load */ if (busiest->group_type == group_misfit_task)
@@ -12965,7 +12998,7 @@ static struct sched_group *sched_balance_find_src_group(struct lb_env *env) if (busiest->group_type == group_imbalanced) goto force_balance; - local = &sds.local_stat; + local = &sds->local_stat; /* * If the local group is busier than the selected busiest group * don't try and pull any tasks.
@@ -12986,14 +13019,14 @@ static struct sched_group *sched_balance_find_src_group(struct lb_env *env) goto out_balanced; /* XXX broken for overlapping NUMA groups */ - sds.avg_load = (sds.total_load * SCHED_CAPACITY_SCALE) / - sds.total_capacity; + sds->avg_load = (sds->total_load * SCHED_CAPACITY_SCALE) / + sds->total_capacity; /* * Don't pull any tasks if this group is already above the * domain average load. */ - if (local->avg_load >= sds.avg_load) + if (local->avg_load >= sds->avg_load) goto out_balanced; /*
@@ -13009,9 +13042,9 @@ static struct sched_group *sched_balance_find_src_group(struct lb_env *env) * Try to move all excess tasks to a sibling domain of the busiest * group's child domain. */ - if (sds.prefer_sibling && local->group_type == group_has_spare && + if (sds->prefer_sibling && local->group_type == group_has_spare && (busiest->group_type == group_llc_balance || - sibling_imbalance(env, &sds, busiest, local) > 1)) + sibling_imbalance(env, sds, busiest, local) > 1)) goto force_balance; if (busiest->group_type != group_overloaded) {
@@ -13025,7 +13058,7 @@ static struct sched_group *sched_balance_find_src_group(struct lb_env *env) } if (busiest->group_type == group_smt_balance && - smt_vs_nonsmt_groups(sds.local, sds.busiest)) { + smt_vs_nonsmt_groups(sds->local, sds->busiest)) { /* Let non SMT CPU pull from SMT CPU sharing with sibling */ goto force_balance; }
@@ -13054,8 +13087,8 @@ static struct sched_group *sched_balance_find_src_group(struct lb_env *env) force_balance: /* Looks like there is an imbalance. Compute it */ - calculate_imbalance(env, &sds); - return env->imbalance ? sds.busiest : NULL; + calculate_imbalance(env, sds); + return env->imbalance ? sds->busiest : NULL; out_balanced: env->imbalance = 0;
@@ -13958,6 +13991,7 @@ static void sched_balance_domains(struct rq *rq, enum cpu_idle_type idle) interval = get_sd_balance_interval(sd, busy); if (time_after_eq(jiffies, sd->last_balance + interval)) { + guard(irqsave)(); if (sched_balance_rq(cpu, rq, sd, idle, &continue_balancing)) { /* * The LBF_DST_PINNED logic could have changed