mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [PATCH v3 sched_ext/for-7.4] sched_ext: cid: Represent clusters explicitly
@ 2026-09-27 13:37 Andrea Righi
  0 siblings, 0 replies; only message in thread
From: Andrea Righi @ 2026-09-27 13:37 UTC (permalink / raw)
  To: Tejun Heo, David Vernet, Changwoo Min
  Cc: Emil Tsalapatis, sched-ext, linux-kernel

The sched_ext CID topology gives each core, LLC and NUMA node a
contiguous CID range. Without cluster entries, schedulers cannot obtain
their CID range or global index.

Walk cores by cache-sharing cluster and add cluster_cid and cluster_idx
to scx_cid_topo. Treat the level inclusively: a core without a wider
cluster forms one of its own, while an LLC-wide cache-sharing group
forms one cluster. This gives every cluster a contiguous CID range and a
dense index.

The walk now takes cores in cluster order rather than LLC order. Shards
are still cut on core boundaries, so a shard doesn't need to contain a
whole cluster; only core, cluster, LLC and node are nested, in that
order.

The level follows whatever the arch reports through
topology_cluster_cpumask(), which on x86 is the L2 sharing mask and is
populated independently of CONFIG_SCHED_CLUSTER.

Tested on a 13th Gen Intel Core i7-13800H: six SMT P-cores occupy CPUs
0-11, eight E-cores occupy CPUs 12-19 in two four-core L2 modules, and
all 20 CPUs share one LLC; the P-core pairs form two-CID clusters, the
E-core modules form four-CID clusters, and cluster_idx is dense across
all eight clusters.

Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
Changes in v3:
 - Use an inclusive mask so an LLC-wide L2 remains one cluster (Tejun Heo).
 - Update the override kerneldoc and simplify the topology comments (Tejun Heo).
 - Document that shards may split clusters and that cluster_cid alone does not
   identify whether a wider cluster level exists.
 - Link to v2: https://lore.kernel.org/r/20260926212017.3351797-1-arighi@nvidia.com

Changes in v2:
 - Give each core without a wider cluster level its own cluster, with
   cluster_cid = core_cid and a dense cluster_idx (Tejun Heo).
 - Link to v1: https://lore.kernel.org/r/20260926152306.3190774-1-arighi@nvidia.com

 kernel/sched/ext/cid.c   | 43 +++++++++++++++++++++++++++++++---------
 kernel/sched/ext/cid.h   | 11 +++++-----
 kernel/sched/ext/types.h | 28 +++++++++++++++++++++-----
 3 files changed, 63 insertions(+), 19 deletions(-)

diff --git a/kernel/sched/ext/cid.c b/kernel/sched/ext/cid.c
index dc670975c5bfd..9fc2192570047 100644
--- a/kernel/sched/ext/cid.c
+++ b/kernel/sched/ext/cid.c
@@ -31,6 +31,7 @@ static struct scx_cid_tables *scx_cid_tables;	/* used only during alloc/free */
 #define SCX_CID_TOPO_NEG	(struct scx_cid_topo) {				\
 	.core_cid = -1, .core_idx = -1, .llc_cid = -1, .llc_idx = -1,		\
 	.node_cid = -1, .node_idx = -1, .shard_cid = -1, .shard_idx = -1,	\
+	.cluster_cid = -1, .cluster_idx = -1,					\
 }
 
 /*
@@ -182,12 +183,14 @@ s32 scx_cid_init(struct scx_sched *sch)
 	cpumask_var_t to_walk __free(free_cpumask_var) = CPUMASK_VAR_NULL;
 	cpumask_var_t node_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
 	cpumask_var_t llc_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
+	cpumask_var_t cluster_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
 	cpumask_var_t core_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
 	cpumask_var_t llc_fallback __free(free_cpumask_var) = CPUMASK_VAR_NULL;
 	cpumask_var_t online_no_topo __free(free_cpumask_var) = CPUMASK_VAR_NULL;
 	struct scx_cid_tables *tbls;
 	u32 next_cid = 0;
 	s32 next_node_idx = 0, next_llc_idx = 0, next_core_idx = 0;
+	s32 next_cluster_idx = 0;
 	s32 next_shard_idx = 0;
 	u32 shard_size, max_cids;
 	u32 notopo_in_shard;
@@ -215,6 +218,7 @@ s32 scx_cid_init(struct scx_sched *sch)
 	if (!zalloc_cpumask_var(&to_walk, GFP_KERNEL) ||
 	    !zalloc_cpumask_var(&node_scratch, GFP_KERNEL) ||
 	    !zalloc_cpumask_var(&llc_scratch, GFP_KERNEL) ||
+	    !zalloc_cpumask_var(&cluster_scratch, GFP_KERNEL) ||
 	    !zalloc_cpumask_var(&core_scratch, GFP_KERNEL) ||
 	    !zalloc_cpumask_var(&llc_fallback, GFP_KERNEL) ||
 	    !zalloc_cpumask_var(&online_no_topo, GFP_KERNEL))
@@ -256,6 +260,7 @@ s32 scx_cid_init(struct scx_sched *sch)
 			u32 cores_per_shard, nr_large;
 			u32 shard_local = 0, cores_in_shard = 0, cids_in_shard = 0;
 			s32 shard_cid, shard_idx;
+			s32 cluster_cid = -1, cluster_idx = -1;
 
 			/* llc_scratch = node_scratch & this llc */
 			cpumask_and(llc_scratch, node_scratch, llc_mask);
@@ -268,15 +273,32 @@ s32 scx_cid_init(struct scx_sched *sch)
 			tbls->shard_node[shard_idx] = nid;
 
 			while (!cpumask_empty(llc_scratch)) {
-				s32 lcpu = cpumask_first(llc_scratch);
-				const struct cpumask *sib = topology_sibling_cpumask(lcpu);
-				s32 core_cid = next_cid;
-				s32 core_idx = next_core_idx++;
-				s32 ccpu;
+				const struct cpumask *sib;
+				s32 core_cid, core_idx, lcpu, ccpu;
 				u32 max_cores, cids_in_core;
 
-				/* core_scratch = llc_scratch & this core */
-				cpumask_and(core_scratch, llc_scratch, sib);
+				/*
+				 * Take the cores of one cluster before moving
+				 * on, so that a cluster is a contiguous cid
+				 * range like the core, LLC and node levels.
+				 */
+				if (cpumask_empty(cluster_scratch)) {
+					s32 xcpu = cpumask_first(llc_scratch);
+
+					cpumask_or(cluster_scratch, topology_cluster_cpumask(xcpu),
+						   topology_sibling_cpumask(xcpu));
+					cpumask_and(cluster_scratch, cluster_scratch, llc_scratch);
+					cluster_cid = next_cid;
+					cluster_idx = next_cluster_idx++;
+				}
+
+				lcpu = cpumask_first(cluster_scratch);
+				sib = topology_sibling_cpumask(lcpu);
+				core_cid = next_cid;
+				core_idx = next_core_idx++;
+
+				/* core_scratch = cluster_scratch & this core */
+				cpumask_and(core_scratch, cluster_scratch, sib);
 				if (WARN_ON_ONCE(!cpumask_test_cpu(lcpu, core_scratch)))
 					return -EINVAL;
 
@@ -316,8 +338,11 @@ s32 scx_cid_init(struct scx_sched *sch)
 						.node_idx = node_idx,
 						.shard_cid = shard_cid,
 						.shard_idx = shard_idx,
+						.cluster_cid = cluster_cid,
+						.cluster_idx = cluster_idx,
 					};
 
+					cpumask_clear_cpu(ccpu, cluster_scratch);
 					cpumask_clear_cpu(ccpu, llc_scratch);
 					cpumask_clear_cpu(ccpu, node_scratch);
 					cpumask_clear_cpu(ccpu, to_walk);
@@ -461,8 +486,8 @@ __bpf_kfunc_start_defs();
  * starts must be strictly increasing with the first entry 0 and all values <
  * num_possible_cpus(). The last shard extends to num_possible_cpus() and no
  * shard may span more than SCX_CID_SHARD_MAX_CPUS cids. Topo info
- * (core/LLC/node) is cleared and the shard layout is set from the input. On
- * invalid input, abort the scheduler.
+ * (core/cluster/LLC/node) is cleared and the shard layout is set from the
+ * input. On invalid input, abort the scheduler.
  */
 __bpf_kfunc void scx_bpf_cid_override(const s32 *cpu_to_cid__arena, u32 cpu_to_cid_cnt,
 				      const s32 *shard_start__arena, u32 shard_start_cnt,
diff --git a/kernel/sched/ext/cid.h b/kernel/sched/ext/cid.h
index 2fe2311a0f995..709afbfb97c6f 100644
--- a/kernel/sched/ext/cid.h
+++ b/kernel/sched/ext/cid.h
@@ -13,11 +13,12 @@
  * kernel type sized for the maximum NR_CPUS (4k), with verbose helper sequences
  * for every op.
  *
- * cids give every cpu a dense, topology-ordered id. CPUs sharing a core, LLC or
- * NUMA node get contiguous cid ranges, so a topology unit becomes a (start,
- * length) slice of cid space. Communication can pass a slice instead of a
- * cpumask, and BPF code can process, for example, a u64 word's worth of cids at
- * a time.
+ * cids give every cpu a dense, topology-ordered id. CPUs in each core,
+ * cluster, LLC or NUMA node get contiguous cid ranges, so a topology unit
+ * becomes a (start, length) slice of cid space. A core without a wider cluster
+ * level forms a cluster of its own. Communication can pass a slice
+ * instead of a cpumask, and BPF code can process, for example, a u64 word's
+ * worth of cids at a time.
  *
  * The mapping is built once at root scheduler enable time by walking the
  * topology of online cpus only. Going by online cpus is out of necessity:
diff --git a/kernel/sched/ext/types.h b/kernel/sched/ext/types.h
index 139176cf9fc6e..38514d6bf1495 100644
--- a/kernel/sched/ext/types.h
+++ b/kernel/sched/ext/types.h
@@ -59,17 +59,31 @@ enum scx_consts {
 };
 
 /*
- * Per-cid topology info. For each topology level (core, LLC, node) and shard,
- * records the first cid in the unit and its global index. Global indices are
- * consecutive integers assigned in cid-walk order, so e.g. core_idx ranges over
- * [0, nr_cores_at_init) with no gaps. No-topo cids have core/LLC/node fields
- * set to -1 but always have valid shard assignments.
+ * Per-cid topology info. For each topology level (core, cluster, LLC, node) and
+ * shard, records the first cid in the unit and its global index. Global indices
+ * are consecutive integers assigned in cid-walk order, so e.g. core_idx ranges
+ * over [0, nr_cores_at_init) with no gaps. No-topo cids have core/cluster/LLC/
+ * node fields set to -1 but always have valid shard assignments.
+ *
+ * A cluster is the cache-sharing level between the core and the LLC. Where
+ * this level is absent, each core forms a cluster of its own. Where the cache
+ * level spans the LLC, the whole LLC forms one cluster. Each cluster has a
+ * unique cluster_cid and a dense cluster_idx, and its cids form a contiguous
+ * range.
+ *
+ * Every cid therefore has a cluster, so cluster_cid alone does not say whether
+ * the machine has the level: the first cluster of an LLC starts at the LLC's
+ * own base cid and a core-wide one at the core's.
  *
  * Shards are contiguous CID ranges used as scalable locking/work domains for
  * sub-scheduler operations. By default each LLC becomes one shard, split into
  * smaller shards if the LLC exceeds the target size. No-topo cids are packed
  * into their own max-sized shards.
  *
+ * Shards are cut on core boundaries only, so a shard doesn't need to contain a
+ * whole cluster: a cluster may straddle two shards. Only core, cluster, LLC
+ * and node are nested, in that order.
+ *
  * New fields are appended, never inserted: scx_bpf_cid_topo() copies this
  * struct out sized by the program's own layout, and an older program's copy
  * must stay a prefix of the kernel's.
@@ -82,6 +96,8 @@ enum scx_consts {
  * @node_idx: global index of that node, in [0, nr_nodes_at_init)
  * @shard_cid: first cid of this cid's shard
  * @shard_idx: global index of that shard, in [0, scx_nr_cid_shards)
+ * @cluster_cid: first cid of this cid's cluster
+ * @cluster_idx: global index of that cluster, in [0, nr_clusters_at_init)
  */
 struct scx_cid_topo {
 	s32 core_cid;
@@ -92,6 +108,8 @@ struct scx_cid_topo {
 	s32 node_idx;
 	s32 shard_cid;
 	s32 shard_idx;
+	s32 cluster_cid;
+	s32 cluster_idx;
 };
 
 enum scx_cid_consts {
-- 
2.55.0

^ permalink raw reply	[flat|nested] only message in thread

only message in thread, other threads:[~2026-09-27 13:37 UTC | newest]

Thread overview: (only message) (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-09-27 13:37 [PATCH v3 sched_ext/for-7.4] sched_ext: cid: Represent clusters explicitly Andrea Righi

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®