From: Tim Chen <tim.c.chen@linux.intel.com>
To: Peter Zijlstra <peterz@infradead.org>,
Ingo Molnar <mingo@redhat.com>,
K Prateek Nayak <kprateek.nayak@amd.com>,
"Gautham R . Shenoy" <gautham.shenoy@amd.com>
Cc: Tim Chen <tim.c.chen@linux.intel.com>,
Vincent Guittot <vincent.guittot@linaro.org>,
Juri Lelli <juri.lelli@redhat.com>,
Dietmar Eggemann <dietmar.eggemann@arm.com>,
Steven Rostedt <rostedt@goodmis.org>,
Ben Segall <bsegall@google.com>, Mel Gorman <mgorman@suse.de>,
Valentin Schneider <vschneid@redhat.com>,
Madadi Vineeth Reddy <vineethr@linux.ibm.com>,
Hillf Danton <hdanton@sina.com>,
Shrikanth Hegde <sshegde@linux.ibm.com>,
Jianyong Wu <jianyong.wu@outlook.com>,
Yangyu Chen <cyy@cyyself.name>,
Tingyin Duan <tingyin.duan@gmail.com>,
Vern Hao <vernhao@tencent.com>, Len Brown <len.brown@intel.com>,
Aubrey Li <aubrey.li@intel.com>, Zhao Liu <zhao1.liu@intel.com>,
Chen Yu <yu.chen.surf@gmail.com>, Chen Yu <yu.c.chen@intel.com>,
Libo Chen <libo.chen@oracle.com>,
Adam Li <adamli@os.amperecomputing.com>,
Tim Chen <tim.c.chen@intel.com>,
linux-kernel@vger.kernel.org
Subject: [PATCH 05/19] sched/fair: Add LLC index mapping for CPUs
Date: Sat, 11 Oct 2025 11:24:42 -0700 [thread overview]
Message-ID: <7d75af576986cf447a171ce11f5e8a15a692e780.1760206683.git.tim.c.chen@linux.intel.com> (raw)
In-Reply-To: <cover.1760206683.git.tim.c.chen@linux.intel.com>
Introduce an index mapping between CPUs and their LLCs. This provides
a continuous per LLC index needed for cache-aware load balancing in
later patches.
The existing per_cpu llc_id usually points to the first CPU of the
LLC domain, which is sparse and unsuitable as an array index. Using
llc_id directly would waste memory.
With the new mapping, CPUs in the same LLC share a continuous index:
per_cpu(llc_idx, CPU=0...15) = 0
per_cpu(llc_idx, CPU=16...31) = 1
per_cpu(llc_idx, CPU=32...47) = 2
...
The maximum number of LLCs is limited by CONFIG_NR_LLCS. If the number
of LLCs available exceeds CONFIG_NR_LLCS, the cache aware load balance
is disabled. To further save memory, this array could be converted to
dynamic allocation in the future, or the LLC index could be made NUMA
node-wide.
As mentioned by Adam, if there is no domain with SD_SHARE_LLC, the
function update_llc_idx() should not be invoked to update the index;
otherwise, it will generate an invalid index.
Co-developed-by: Chen Yu <yu.c.chen@intel.com>
Signed-off-by: Chen Yu <yu.c.chen@intel.com>
Signed-off-by: Tim Chen <tim.c.chen@linux.intel.com>
---
include/linux/threads.h | 10 +++++++++
init/Kconfig | 9 ++++++++
kernel/sched/fair.c | 11 ++++++++++
kernel/sched/sched.h | 2 ++
kernel/sched/topology.c | 47 +++++++++++++++++++++++++++++++++++++++++
5 files changed, 79 insertions(+)
diff --git a/include/linux/threads.h b/include/linux/threads.h
index 1674a471b0b4..2c9b1adfe024 100644
--- a/include/linux/threads.h
+++ b/include/linux/threads.h
@@ -20,6 +20,16 @@
/* Places which use this should consider cpumask_var_t. */
#define NR_CPUS CONFIG_NR_CPUS
+#ifndef CONFIG_NR_LLCS
+#define CONFIG_NR_LLCS 1
+#endif
+
+#if CONFIG_NR_LLCS > NR_CPUS
+#define NR_LLCS NR_CPUS
+#else
+#define NR_LLCS CONFIG_NR_LLCS
+#endif
+
#define MIN_THREADS_LEFT_FOR_ROOT 4
/*
diff --git a/init/Kconfig b/init/Kconfig
index 4e625db7920a..6e4c96ccdda0 100644
--- a/init/Kconfig
+++ b/init/Kconfig
@@ -981,6 +981,15 @@ config SCHED_CACHE
resources within the same cache domain, reducing cache misses and
lowering data access latency.
+config NR_LLCS
+ int "Maximum number of Last Level Caches"
+ range 2 1024
+ depends on SMP && SCHED_CACHE
+ default 64
+ help
+ This allows you to specify the maximum number of last level caches
+ this kernel will support for cache aware scheduling.
+
config NUMA_BALANCING_DEFAULT_ENABLED
bool "Automatically enable NUMA aware memory/task placement"
default y
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 3d643449c48c..61c129bde8b6 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -1224,6 +1224,17 @@ static int llc_id(int cpu)
return per_cpu(sd_llc_id, cpu);
}
+/*
+ * continuous LLC index, starting from 0.
+ */
+static inline int llc_idx(int cpu)
+{
+ if (cpu < 0)
+ return -1;
+
+ return per_cpu(sd_llc_idx, cpu);
+}
+
void mm_init_sched(struct mm_struct *mm, struct mm_sched __percpu *_pcpu_sched)
{
unsigned long epoch;
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 60f1e51685ec..b448ad6dc51d 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -2039,6 +2039,7 @@ static inline struct sched_domain *lowest_flag_domain(int cpu, int flag)
DECLARE_PER_CPU(struct sched_domain __rcu *, sd_llc);
DECLARE_PER_CPU(int, sd_llc_size);
DECLARE_PER_CPU(int, sd_llc_id);
+DECLARE_PER_CPU(int, sd_llc_idx);
DECLARE_PER_CPU(int, sd_share_id);
DECLARE_PER_CPU(struct sched_domain_shared __rcu *, sd_llc_shared);
DECLARE_PER_CPU(struct sched_domain __rcu *, sd_numa);
@@ -2047,6 +2048,7 @@ DECLARE_PER_CPU(struct sched_domain __rcu *, sd_asym_cpucapacity);
extern struct static_key_false sched_asym_cpucapacity;
extern struct static_key_false sched_cluster_active;
+extern int max_llcs;
static __always_inline bool sched_asym_cpucap_active(void)
{
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
index 2675db980f70..4bd033060f1d 100644
--- a/kernel/sched/topology.c
+++ b/kernel/sched/topology.c
@@ -659,6 +659,7 @@ static void destroy_sched_domains(struct sched_domain *sd)
DEFINE_PER_CPU(struct sched_domain __rcu *, sd_llc);
DEFINE_PER_CPU(int, sd_llc_size);
DEFINE_PER_CPU(int, sd_llc_id);
+DEFINE_PER_CPU(int, sd_llc_idx);
DEFINE_PER_CPU(int, sd_share_id);
DEFINE_PER_CPU(struct sched_domain_shared __rcu *, sd_llc_shared);
DEFINE_PER_CPU(struct sched_domain __rcu *, sd_numa);
@@ -668,6 +669,40 @@ DEFINE_PER_CPU(struct sched_domain __rcu *, sd_asym_cpucapacity);
DEFINE_STATIC_KEY_FALSE(sched_asym_cpucapacity);
DEFINE_STATIC_KEY_FALSE(sched_cluster_active);
+int max_llcs = -1;
+
+static void update_llc_idx(int cpu)
+{
+#ifdef CONFIG_SCHED_CACHE
+ int idx = -1, llc_id = -1;
+
+ if (max_llcs > NR_LLCS)
+ return;
+
+ llc_id = per_cpu(sd_llc_id, cpu);
+ idx = per_cpu(sd_llc_idx, llc_id);
+
+ /*
+ * A new LLC is detected, increase the index
+ * by 1.
+ */
+ if (idx < 0) {
+ idx = max_llcs++;
+
+ if (max_llcs > NR_LLCS) {
+ if (static_branch_unlikely(&sched_cache_allowed))
+ static_branch_disable_cpuslocked(&sched_cache_allowed);
+
+ pr_warn_once("CONFIG_NR_LLCS is too small, disable cache aware load balance\n");
+ return;
+ }
+
+ per_cpu(sd_llc_idx, llc_id) = idx;
+ }
+ per_cpu(sd_llc_idx, cpu) = idx;
+#endif
+}
+
static void update_top_cache_domain(int cpu)
{
struct sched_domain_shared *sds = NULL;
@@ -687,6 +722,10 @@ static void update_top_cache_domain(int cpu)
per_cpu(sd_llc_id, cpu) = id;
rcu_assign_pointer(per_cpu(sd_llc_shared, cpu), sds);
+ /* only update the llc index for domain with SD_SHARE_LLC */
+ if (sd)
+ update_llc_idx(cpu);
+
sd = lowest_flag_domain(cpu, SD_CLUSTER);
if (sd)
id = cpumask_first(sched_domain_span(sd));
@@ -2452,6 +2491,14 @@ build_sched_domains(const struct cpumask *cpu_map, struct sched_domain_attr *att
bool has_asym = false;
bool has_cluster = false;
+#ifdef CONFIG_SCHED_CACHE
+ if (max_llcs < 0) {
+ for_each_possible_cpu(i)
+ per_cpu(sd_llc_idx, i) = -1;
+ max_llcs = 0;
+ }
+#endif
+
if (WARN_ON(cpumask_empty(cpu_map)))
goto error;
--
2.32.0
next prev parent reply other threads:[~2025-10-11 18:18 UTC|newest]
Thread overview: 116+ messages / expand[flat|nested] mbox.gz Atom feed top
2025-10-11 18:24 [PATCH 00/19] Cache Aware Scheduling Tim Chen
2025-10-11 18:24 ` [PATCH 01/19] sched/fair: Add infrastructure for cache-aware load balancing Tim Chen
2025-10-14 19:12 ` Madadi Vineeth Reddy
2025-10-15 4:54 ` Chen, Yu C
2025-10-15 19:32 ` Tim Chen
2025-10-16 3:11 ` Chen, Yu C
2025-10-15 11:54 ` Peter Zijlstra
2025-10-15 16:07 ` Chen, Yu C
2025-10-23 7:26 ` kernel test robot
2025-10-27 4:47 ` K Prateek Nayak
2025-10-27 13:35 ` Chen, Yu C
2025-10-11 18:24 ` [PATCH 02/19] sched/fair: Record per-LLC utilization to guide cache-aware scheduling decisions Tim Chen
2025-10-15 10:15 ` Peter Zijlstra
2025-10-15 16:27 ` Chen, Yu C
2025-10-27 5:01 ` K Prateek Nayak
2025-10-27 14:07 ` Chen, Yu C
2025-10-28 2:50 ` K Prateek Nayak
2025-10-11 18:24 ` [PATCH 03/19] sched/fair: Introduce helper functions to enforce LLC migration policy Tim Chen
2025-10-11 18:24 ` [PATCH 04/19] sched/fair: Introduce a static key to enable cache aware only for multi LLCs Tim Chen
2025-10-15 11:04 ` Peter Zijlstra
2025-10-15 16:25 ` Chen, Yu C
2025-10-15 16:36 ` Shrikanth Hegde
2025-10-15 17:01 ` Chen, Yu C
2025-10-16 7:42 ` Peter Zijlstra
2025-10-17 2:08 ` Chen, Yu C
2025-10-16 7:40 ` Peter Zijlstra
2025-10-27 5:42 ` K Prateek Nayak
2025-10-27 12:56 ` Chen, Yu C
2025-10-27 23:36 ` Tim Chen
2025-10-29 12:36 ` Chen, Yu C
2025-10-28 2:46 ` K Prateek Nayak
2025-10-11 18:24 ` Tim Chen [this message]
2025-10-15 11:08 ` [PATCH 05/19] sched/fair: Add LLC index mapping for CPUs Peter Zijlstra
2025-10-15 11:58 ` Peter Zijlstra
2025-10-15 20:12 ` Tim Chen
2025-10-11 18:24 ` [PATCH 06/19] sched/fair: Assign preferred LLC ID to processes Tim Chen
2025-10-14 5:16 ` Chen, Yu C
2025-10-15 11:15 ` Peter Zijlstra
2025-10-16 3:13 ` Chen, Yu C
2025-10-17 4:50 ` Chen, Yu C
2025-10-20 9:41 ` Vern Hao
2025-10-11 18:24 ` [PATCH 07/19] sched/fair: Track LLC-preferred tasks per runqueue Tim Chen
2025-10-15 12:05 ` Peter Zijlstra
2025-10-15 20:03 ` Tim Chen
2025-10-16 7:44 ` Peter Zijlstra
2025-10-16 20:06 ` Tim Chen
2025-10-27 6:04 ` K Prateek Nayak
2025-10-28 15:15 ` Chen, Yu C
2025-10-28 15:46 ` Tim Chen
2025-10-29 4:32 ` K Prateek Nayak
2025-10-29 12:48 ` Chen, Yu C
2025-10-29 4:00 ` K Prateek Nayak
2025-10-28 17:06 ` Tim Chen
2025-10-11 18:24 ` [PATCH 08/19] sched/fair: Introduce per runqueue task LLC preference counter Tim Chen
2025-10-15 12:21 ` Peter Zijlstra
2025-10-15 20:41 ` Tim Chen
2025-10-16 7:49 ` Peter Zijlstra
2025-10-21 8:28 ` Madadi Vineeth Reddy
2025-10-23 6:07 ` Chen, Yu C
2025-10-11 18:24 ` [PATCH 09/19] sched/fair: Count tasks prefering each LLC in a sched group Tim Chen
2025-10-15 12:22 ` Peter Zijlstra
2025-10-15 20:42 ` Tim Chen
2025-10-15 12:25 ` Peter Zijlstra
2025-10-15 20:43 ` Tim Chen
2025-10-27 8:33 ` K Prateek Nayak
2025-10-27 23:19 ` Tim Chen
2025-10-11 18:24 ` [PATCH 10/19] sched/fair: Prioritize tasks preferring destination LLC during balancing Tim Chen
2025-10-15 7:23 ` kernel test robot
2025-10-15 15:08 ` Peter Zijlstra
2025-10-15 21:28 ` Tim Chen
2025-10-15 15:10 ` Peter Zijlstra
2025-10-15 16:03 ` Chen, Yu C
2025-10-24 9:32 ` Aaron Lu
2025-10-27 2:00 ` Chen, Yu C
2025-10-29 9:51 ` Aaron Lu
2025-10-29 13:19 ` Chen, Yu C
2025-10-27 6:29 ` K Prateek Nayak
2025-10-28 12:11 ` Chen, Yu C
2025-10-11 18:24 ` [PATCH 11/19] sched/fair: Identify busiest sched_group for LLC-aware load balancing Tim Chen
2025-10-15 15:24 ` Peter Zijlstra
2025-10-15 21:18 ` Tim Chen
2025-10-11 18:24 ` [PATCH 12/19] sched/fair: Add migrate_llc_task migration type for cache-aware balancing Tim Chen
2025-10-27 9:04 ` K Prateek Nayak
2025-10-27 22:59 ` Tim Chen
2025-10-11 18:24 ` [PATCH 13/19] sched/fair: Handle moving single tasks to/from their preferred LLC Tim Chen
2025-10-11 18:24 ` [PATCH 14/19] sched/fair: Consider LLC preference when selecting tasks for load balancing Tim Chen
2025-10-11 18:24 ` [PATCH 15/19] sched/fair: Respect LLC preference in task migration and detach Tim Chen
2025-10-28 6:02 ` K Prateek Nayak
2025-10-28 11:58 ` Chen, Yu C
2025-10-28 15:30 ` Tim Chen
2025-10-29 4:15 ` K Prateek Nayak
2025-10-29 3:54 ` K Prateek Nayak
2025-10-29 14:23 ` Chen, Yu C
2025-10-29 21:09 ` Tim Chen
2025-10-30 4:19 ` K Prateek Nayak
2025-10-30 20:07 ` Tim Chen
2025-10-31 3:32 ` K Prateek Nayak
2025-10-31 15:17 ` Chen, Yu C
2025-11-03 21:41 ` Tim Chen
2025-11-03 22:07 ` Tim Chen
2025-10-11 18:24 ` [PATCH 16/19] sched/fair: Exclude processes with many threads from cache-aware scheduling Tim Chen
2025-10-23 7:22 ` kernel test robot
2025-10-11 18:24 ` [PATCH 17/19] sched/fair: Disable cache aware scheduling for processes with high thread counts Tim Chen
2025-10-22 17:21 ` Madadi Vineeth Reddy
2025-10-23 6:55 ` Chen, Yu C
2025-10-11 18:24 ` [PATCH 18/19] sched/fair: Avoid cache-aware scheduling for memory-heavy processes Tim Chen
2025-10-15 6:57 ` kernel test robot
2025-10-16 4:44 ` Chen, Yu C
2025-10-11 18:24 ` [PATCH 19/19] sched/fair: Add user control to adjust the tolerance of cache-aware scheduling Tim Chen
2025-10-29 8:07 ` Aaron Lu
2025-10-29 12:54 ` Chen, Yu C
2025-10-14 12:13 ` [PATCH 00/19] Cache Aware Scheduling Madadi Vineeth Reddy
2025-10-14 21:48 ` Tim Chen
2025-10-15 5:38 ` Chen, Yu C
2025-10-15 18:26 ` Madadi Vineeth Reddy
2025-10-16 4:57 ` Chen, Yu C
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=7d75af576986cf447a171ce11f5e8a15a692e780.1760206683.git.tim.c.chen@linux.intel.com \
--to=tim.c.chen@linux.intel.com \
--cc=adamli@os.amperecomputing.com \
--cc=aubrey.li@intel.com \
--cc=bsegall@google.com \
--cc=cyy@cyyself.name \
--cc=dietmar.eggemann@arm.com \
--cc=gautham.shenoy@amd.com \
--cc=hdanton@sina.com \
--cc=jianyong.wu@outlook.com \
--cc=juri.lelli@redhat.com \
--cc=kprateek.nayak@amd.com \
--cc=len.brown@intel.com \
--cc=libo.chen@oracle.com \
--cc=linux-kernel@vger.kernel.org \
--cc=mgorman@suse.de \
--cc=mingo@redhat.com \
--cc=peterz@infradead.org \
--cc=rostedt@goodmis.org \
--cc=sshegde@linux.ibm.com \
--cc=tim.c.chen@intel.com \
--cc=tingyin.duan@gmail.com \
--cc=vernhao@tencent.com \
--cc=vincent.guittot@linaro.org \
--cc=vineethr@linux.ibm.com \
--cc=vschneid@redhat.com \
--cc=yu.c.chen@intel.com \
--cc=yu.chen.surf@gmail.com \
--cc=zhao1.liu@intel.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®