From: K Prateek Nayak <kprateek.nayak@amd.com>
To: Ingo Molnar <mingo@redhat.com>,
Peter Zijlstra <peterz@infradead.org>,
Juri Lelli <juri.lelli@redhat.com>,
Vincent Guittot <vincent.guittot@linaro.org>,
Anna-Maria Behnsen <anna-maria@linutronix.de>,
Frederic Weisbecker <frederic@kernel.org>,
Thomas Gleixner <tglx@linutronix.de>,
<linux-kernel@vger.kernel.org>
Cc: Dietmar Eggemann <dietmar.eggemann@arm.com>,
Steven Rostedt <rostedt@goodmis.org>,
Ben Segall <bsegall@google.com>, Mel Gorman <mgorman@suse.de>,
Valentin Schneider <vschneid@redhat.com>,
K Prateek Nayak <kprateek.nayak@amd.com>,
"Gautham R. Shenoy" <gautham.shenoy@amd.com>,
Swapnil Sapkal <swapnil.sapkal@amd.com>
Subject: [RFC PATCH 18/19] sched/fair: Optimize global "nohz.nr_cpus" tracking
Date: Thu, 4 Sep 2025 04:15:14 +0000 [thread overview]
Message-ID: <20250904041516.3046-19-kprateek.nayak@amd.com> (raw)
In-Reply-To: <20250904041516.3046-1-kprateek.nayak@amd.com>
Optimize "nohz.nr_cpus" by tracking number of "sd_nohz->shared" with
non-zero "nr_idle_cpus" count via "nohz.nr_doms" and only updating at
the boundary of "sd_nohz->shared->nr_idle_cpus" going from 0 -> 1 and
back from 1 -> 0.
This also introduces a chance of double accounting when a nohz idle
entry or the tick races with hotplug or cpuset as described in
__nohz_exit_idle_tracking().
__nohz_exit_idle_tracking() called when the sched_domain_shared nodes
tracking idle CPUs are freed is used to correct any potential double
accounting which can unnecessarily trigger nohz idle balances even when
all the CPUs have tick enabled.
Signed-off-by: K Prateek Nayak <kprateek.nayak@amd.com>
---
kernel/sched/fair.c | 63 ++++++++++++++++++++++++++++++++++++-----
kernel/sched/sched.h | 1 +
kernel/sched/topology.c | 1 +
3 files changed, 58 insertions(+), 7 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 5b693bd0fab4..d65acf7ea12e 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -7169,7 +7169,7 @@ static DEFINE_PER_CPU(cpumask_var_t, should_we_balance_tmpmask);
#ifdef CONFIG_NO_HZ_COMMON
static struct {
- atomic_t nr_cpus;
+ atomic_t nr_doms;
int has_blocked; /* Idle CPUS has blocked load */
int needs_update; /* Newly idle CPUs need their next_balance collated */
unsigned long next_balance; /* in jiffy units */
@@ -12408,7 +12408,7 @@ static void nohz_balancer_kick(struct rq *rq)
* None are in tickless mode and hence no need for NOHZ idle load
* balancing:
*/
- if (likely(!atomic_read(&nohz.nr_cpus)))
+ if (likely(!atomic_read(&nohz.nr_doms)))
return;
if (READ_ONCE(nohz.has_blocked) &&
@@ -12505,7 +12505,8 @@ static void set_cpu_sd_state_busy(int cpu)
return;
cpumask_clear_cpu(cpu, sd->shared->idle_cpus_mask);
- atomic_dec(&sd->shared->nr_idle_cpus);
+ if (!atomic_dec_return(&sd->shared->nr_idle_cpus))
+ atomic_dec(&nohz.nr_doms);
}
void nohz_balance_exit_idle(struct rq *rq)
@@ -12516,7 +12517,6 @@ void nohz_balance_exit_idle(struct rq *rq)
return;
WRITE_ONCE(rq->nohz_tick_stopped, 0);
- atomic_dec(&nohz.nr_cpus);
set_cpu_sd_state_busy(rq->cpu);
}
@@ -12535,7 +12535,58 @@ static void set_cpu_sd_state_idle(int cpu)
return;
cpumask_set_cpu(cpu, sd->shared->idle_cpus_mask);
- atomic_inc(&sd->shared->nr_idle_cpus);
+ if (!atomic_fetch_inc(&sd->shared->nr_idle_cpus))
+ atomic_inc(&nohz.nr_doms);
+}
+
+/*
+ * Correct nohz.nr_doms if sd_nohz->shared was found to have non-zero
+ * nr_idle_cpus when freeing. No local references to sds remain at
+ * this point and the only reference possible via "nohz_shared_list"
+ * will be dropped after the grace period.
+ */
+void __nohz_exit_idle_tracking(struct sched_domain_shared *sds)
+{
+
+ /*
+ * It is possible for a idle entry to race with sched domain rebuild like:
+ *
+ * CPU0 (hotplug) CPU1 (nohz idle)
+ *
+ * rq->offline(CPU1)
+ * set_cpu_sd_state_busy()
+ * rq->sd = sdd; # Processes IPI, re-enters nohz idle
+ * ... # For old sd_nohz
+ * ... atomic_fetch_inc(&sd_nohz->shared->nr_idle_cpus);
+ * ... atomic_inc(&nohz.nr_doms); # XXX: Accounted once
+ * update_top_cache_domains()
+ * rq->online(CPU1)
+ * # rq->nohz_tick_stopped is true
+ * set_cpu_sd_state_idle()
+ * # For new sd_nohz
+ * atomic_fetch_inc(&sd_nohz->shared->nr_idle_cpus);
+ * atomic_inc(&nohz.nr_doms); # XXX: Accounted twice
+ * ...
+ *
+ * "nohz.nr_doms" is used as an entry criteria in nohz_balancer_kick()
+ * and this double accounting can lead to wasted idle balancing
+ * triggers. Use this path to correct the accounting:
+ *
+ * # In sds_delayed_free()
+ * __nohz_exit_idle_tracking(sds)
+ * # sd->shared->nr_idle_cpus is != 0
+ * atomic_dec(&nohz.nr_doms); # XXX: Fixes nohz.nr_doms
+ */
+ if (atomic_read(&sds->nr_idle_cpus)) {
+ /*
+ * Reset the "nr_idle_cpus" indicator to prevent
+ * existing readers from traversing the idle mask
+ * to reduce chances of traversing the same CPU
+ * twice.
+ */
+ atomic_set(&sds->nr_idle_cpus, 0);
+ atomic_dec(&nohz.nr_doms);
+ }
}
static void cpu_sd_exit_nohz_balance(struct rq *rq)
@@ -12587,8 +12638,6 @@ void nohz_balance_enter_idle(int cpu)
WRITE_ONCE(rq->nohz_tick_stopped, 1);
- atomic_inc(&nohz.nr_cpus);
-
set_cpu_sd_state_idle(cpu);
/*
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 9cffcfbef1ae..fcf4503caada 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -3100,6 +3100,7 @@ extern void cfs_bandwidth_usage_dec(void);
DECLARE_PER_CPU(struct sched_domain __rcu *, sd_nohz);
extern struct list_head nohz_shared_list;
+extern void __nohz_exit_idle_tracking(struct sched_domain_shared *sds);
extern void nohz_balance_exit_idle(struct rq *rq);
#else /* !CONFIG_NO_HZ_COMMON: */
static inline void nohz_balance_exit_idle(struct rq *rq) { }
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
index 86e33ed07254..ee9eed8470ba 100644
--- a/kernel/sched/topology.c
+++ b/kernel/sched/topology.c
@@ -615,6 +615,7 @@ static int sds_delayed_free(struct sched_domain_shared *sds)
scoped_guard(raw_spinlock_irqsave, &nohz_shared_list_lock)
list_del_rcu(&sds->nohz_list_node);
+ __nohz_exit_idle_tracking(sds);
call_rcu(&sds->rcu, destroy_sched_domain_shared_rcu);
return 1;
}
--
2.34.1
next prev parent reply other threads:[~2025-09-04 4:21 UTC|newest]
Thread overview: 26+ messages / expand[flat|nested] mbox.gz Atom feed top
2025-09-04 4:14 [RFC PATCH 00/19] sched/fair: Distributed nohz idle CPU tracking for idle load balancing K Prateek Nayak
2025-09-04 4:14 ` [RFC PATCH 01/19] sched/fair: Simplify set_cpu_sd_state_*() with guards K Prateek Nayak
2025-09-24 20:26 ` Shrikanth Hegde
2025-09-25 2:11 ` K Prateek Nayak
2025-09-04 4:14 ` [RFC PATCH 02/19] sched/topology: Optimize sd->shared allocation and assignment K Prateek Nayak
2025-09-04 4:14 ` [RFC PATCH 03/19] sched/fair: Use rq->nohz_tick_stopped in update_nohz_stats() K Prateek Nayak
2025-09-24 20:17 ` Shrikanth Hegde
2025-09-25 1:48 ` K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 04/19] sched/fair: Use xchg() to set sd->nohz_idle state K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 05/19] sched/topology: Attach new hierarchy in rq_attach_root() K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 06/19] sched/fair: Fixup sd->nohz_idle state during hotplug / cpuset K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 07/19] sched/fair: Account idle cpus instead of busy cpus in sd->shared K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 08/19] sched/topology: Introduce fallback sd->shared assignment K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 09/19] sched/topology: Introduce percpu sd_nohz for nohz state tracking K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 10/19] sched/topology: Introduce "idle_cpus_mask" in sd->shared K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 11/19] sched/topology: Introduce "nohz_shared_list" to keep track of sd->shared K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 12/19] sched/fair: Reorder the barrier in nohz_balance_enter_idle() K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 13/19] sched/fair: Extract the main _nohz_idle_balance() loop into a helper K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 14/19] sched/fair: Convert find_new_ilb() to use nohz_shared_list K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 15/19] sched/fair: Introduce sched_asym_prefer_idle() for ILB kick K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 16/19] sched/fair: Convert sched_balance_nohz_idle() to use nohz_shared_list K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 17/19] sched/fair: Remove "nohz.idle_cpus_mask" K Prateek Nayak
2025-09-04 4:15 ` K Prateek Nayak [this message]
2025-09-24 20:02 ` [RFC PATCH 18/19] sched/fair: Optimize global "nohz.nr_cpus" tracking Shrikanth Hegde
2025-09-25 2:37 ` K Prateek Nayak
2025-09-04 4:15 ` [RFC PATCH 19/19] sched/topology: Add basic debug information for "nohz_shared_list" K Prateek Nayak
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20250904041516.3046-19-kprateek.nayak@amd.com \
--to=kprateek.nayak@amd.com \
--cc=anna-maria@linutronix.de \
--cc=bsegall@google.com \
--cc=dietmar.eggemann@arm.com \
--cc=frederic@kernel.org \
--cc=gautham.shenoy@amd.com \
--cc=juri.lelli@redhat.com \
--cc=linux-kernel@vger.kernel.org \
--cc=mgorman@suse.de \
--cc=mingo@redhat.com \
--cc=peterz@infradead.org \
--cc=rostedt@goodmis.org \
--cc=swapnil.sapkal@amd.com \
--cc=tglx@linutronix.de \
--cc=vincent.guittot@linaro.org \
--cc=vschneid@redhat.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®