* [PATCH v3 sched_ext/for-7.4] sched_ext: cid: Represent clusters explicitly
@ 2026-09-27 13:37 Andrea Righi
0 siblings, 0 replies; only message in thread
From: Andrea Righi @ 2026-09-27 13:37 UTC (permalink / raw)
To: Tejun Heo, David Vernet, Changwoo Min
Cc: Emil Tsalapatis, sched-ext, linux-kernel
The sched_ext CID topology gives each core, LLC and NUMA node a
contiguous CID range. Without cluster entries, schedulers cannot obtain
their CID range or global index.
Walk cores by cache-sharing cluster and add cluster_cid and cluster_idx
to scx_cid_topo. Treat the level inclusively: a core without a wider
cluster forms one of its own, while an LLC-wide cache-sharing group
forms one cluster. This gives every cluster a contiguous CID range and a
dense index.
The walk now takes cores in cluster order rather than LLC order. Shards
are still cut on core boundaries, so a shard doesn't need to contain a
whole cluster; only core, cluster, LLC and node are nested, in that
order.
The level follows whatever the arch reports through
topology_cluster_cpumask(), which on x86 is the L2 sharing mask and is
populated independently of CONFIG_SCHED_CLUSTER.
Tested on a 13th Gen Intel Core i7-13800H: six SMT P-cores occupy CPUs
0-11, eight E-cores occupy CPUs 12-19 in two four-core L2 modules, and
all 20 CPUs share one LLC; the P-core pairs form two-CID clusters, the
E-core modules form four-CID clusters, and cluster_idx is dense across
all eight clusters.
Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
Changes in v3:
- Use an inclusive mask so an LLC-wide L2 remains one cluster (Tejun Heo).
- Update the override kerneldoc and simplify the topology comments (Tejun Heo).
- Document that shards may split clusters and that cluster_cid alone does not
identify whether a wider cluster level exists.
- Link to v2: https://lore.kernel.org/r/20260926212017.3351797-1-arighi@nvidia.com
Changes in v2:
- Give each core without a wider cluster level its own cluster, with
cluster_cid = core_cid and a dense cluster_idx (Tejun Heo).
- Link to v1: https://lore.kernel.org/r/20260926152306.3190774-1-arighi@nvidia.com
kernel/sched/ext/cid.c | 43 +++++++++++++++++++++++++++++++---------
kernel/sched/ext/cid.h | 11 +++++-----
kernel/sched/ext/types.h | 28 +++++++++++++++++++++-----
3 files changed, 63 insertions(+), 19 deletions(-)
diff --git a/kernel/sched/ext/cid.c b/kernel/sched/ext/cid.c
index dc670975c5bfd..9fc2192570047 100644
--- a/kernel/sched/ext/cid.c
+++ b/kernel/sched/ext/cid.c
@@ -31,6 +31,7 @@ static struct scx_cid_tables *scx_cid_tables; /* used only during alloc/free */
#define SCX_CID_TOPO_NEG (struct scx_cid_topo) { \
.core_cid = -1, .core_idx = -1, .llc_cid = -1, .llc_idx = -1, \
.node_cid = -1, .node_idx = -1, .shard_cid = -1, .shard_idx = -1, \
+ .cluster_cid = -1, .cluster_idx = -1, \
}
/*
@@ -182,12 +183,14 @@ s32 scx_cid_init(struct scx_sched *sch)
cpumask_var_t to_walk __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t node_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t llc_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
+ cpumask_var_t cluster_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t core_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t llc_fallback __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t online_no_topo __free(free_cpumask_var) = CPUMASK_VAR_NULL;
struct scx_cid_tables *tbls;
u32 next_cid = 0;
s32 next_node_idx = 0, next_llc_idx = 0, next_core_idx = 0;
+ s32 next_cluster_idx = 0;
s32 next_shard_idx = 0;
u32 shard_size, max_cids;
u32 notopo_in_shard;
@@ -215,6 +218,7 @@ s32 scx_cid_init(struct scx_sched *sch)
if (!zalloc_cpumask_var(&to_walk, GFP_KERNEL) ||
!zalloc_cpumask_var(&node_scratch, GFP_KERNEL) ||
!zalloc_cpumask_var(&llc_scratch, GFP_KERNEL) ||
+ !zalloc_cpumask_var(&cluster_scratch, GFP_KERNEL) ||
!zalloc_cpumask_var(&core_scratch, GFP_KERNEL) ||
!zalloc_cpumask_var(&llc_fallback, GFP_KERNEL) ||
!zalloc_cpumask_var(&online_no_topo, GFP_KERNEL))
@@ -256,6 +260,7 @@ s32 scx_cid_init(struct scx_sched *sch)
u32 cores_per_shard, nr_large;
u32 shard_local = 0, cores_in_shard = 0, cids_in_shard = 0;
s32 shard_cid, shard_idx;
+ s32 cluster_cid = -1, cluster_idx = -1;
/* llc_scratch = node_scratch & this llc */
cpumask_and(llc_scratch, node_scratch, llc_mask);
@@ -268,15 +273,32 @@ s32 scx_cid_init(struct scx_sched *sch)
tbls->shard_node[shard_idx] = nid;
while (!cpumask_empty(llc_scratch)) {
- s32 lcpu = cpumask_first(llc_scratch);
- const struct cpumask *sib = topology_sibling_cpumask(lcpu);
- s32 core_cid = next_cid;
- s32 core_idx = next_core_idx++;
- s32 ccpu;
+ const struct cpumask *sib;
+ s32 core_cid, core_idx, lcpu, ccpu;
u32 max_cores, cids_in_core;
- /* core_scratch = llc_scratch & this core */
- cpumask_and(core_scratch, llc_scratch, sib);
+ /*
+ * Take the cores of one cluster before moving
+ * on, so that a cluster is a contiguous cid
+ * range like the core, LLC and node levels.
+ */
+ if (cpumask_empty(cluster_scratch)) {
+ s32 xcpu = cpumask_first(llc_scratch);
+
+ cpumask_or(cluster_scratch, topology_cluster_cpumask(xcpu),
+ topology_sibling_cpumask(xcpu));
+ cpumask_and(cluster_scratch, cluster_scratch, llc_scratch);
+ cluster_cid = next_cid;
+ cluster_idx = next_cluster_idx++;
+ }
+
+ lcpu = cpumask_first(cluster_scratch);
+ sib = topology_sibling_cpumask(lcpu);
+ core_cid = next_cid;
+ core_idx = next_core_idx++;
+
+ /* core_scratch = cluster_scratch & this core */
+ cpumask_and(core_scratch, cluster_scratch, sib);
if (WARN_ON_ONCE(!cpumask_test_cpu(lcpu, core_scratch)))
return -EINVAL;
@@ -316,8 +338,11 @@ s32 scx_cid_init(struct scx_sched *sch)
.node_idx = node_idx,
.shard_cid = shard_cid,
.shard_idx = shard_idx,
+ .cluster_cid = cluster_cid,
+ .cluster_idx = cluster_idx,
};
+ cpumask_clear_cpu(ccpu, cluster_scratch);
cpumask_clear_cpu(ccpu, llc_scratch);
cpumask_clear_cpu(ccpu, node_scratch);
cpumask_clear_cpu(ccpu, to_walk);
@@ -461,8 +486,8 @@ __bpf_kfunc_start_defs();
* starts must be strictly increasing with the first entry 0 and all values <
* num_possible_cpus(). The last shard extends to num_possible_cpus() and no
* shard may span more than SCX_CID_SHARD_MAX_CPUS cids. Topo info
- * (core/LLC/node) is cleared and the shard layout is set from the input. On
- * invalid input, abort the scheduler.
+ * (core/cluster/LLC/node) is cleared and the shard layout is set from the
+ * input. On invalid input, abort the scheduler.
*/
__bpf_kfunc void scx_bpf_cid_override(const s32 *cpu_to_cid__arena, u32 cpu_to_cid_cnt,
const s32 *shard_start__arena, u32 shard_start_cnt,
diff --git a/kernel/sched/ext/cid.h b/kernel/sched/ext/cid.h
index 2fe2311a0f995..709afbfb97c6f 100644
--- a/kernel/sched/ext/cid.h
+++ b/kernel/sched/ext/cid.h
@@ -13,11 +13,12 @@
* kernel type sized for the maximum NR_CPUS (4k), with verbose helper sequences
* for every op.
*
- * cids give every cpu a dense, topology-ordered id. CPUs sharing a core, LLC or
- * NUMA node get contiguous cid ranges, so a topology unit becomes a (start,
- * length) slice of cid space. Communication can pass a slice instead of a
- * cpumask, and BPF code can process, for example, a u64 word's worth of cids at
- * a time.
+ * cids give every cpu a dense, topology-ordered id. CPUs in each core,
+ * cluster, LLC or NUMA node get contiguous cid ranges, so a topology unit
+ * becomes a (start, length) slice of cid space. A core without a wider cluster
+ * level forms a cluster of its own. Communication can pass a slice
+ * instead of a cpumask, and BPF code can process, for example, a u64 word's
+ * worth of cids at a time.
*
* The mapping is built once at root scheduler enable time by walking the
* topology of online cpus only. Going by online cpus is out of necessity:
diff --git a/kernel/sched/ext/types.h b/kernel/sched/ext/types.h
index 139176cf9fc6e..38514d6bf1495 100644
--- a/kernel/sched/ext/types.h
+++ b/kernel/sched/ext/types.h
@@ -59,17 +59,31 @@ enum scx_consts {
};
/*
- * Per-cid topology info. For each topology level (core, LLC, node) and shard,
- * records the first cid in the unit and its global index. Global indices are
- * consecutive integers assigned in cid-walk order, so e.g. core_idx ranges over
- * [0, nr_cores_at_init) with no gaps. No-topo cids have core/LLC/node fields
- * set to -1 but always have valid shard assignments.
+ * Per-cid topology info. For each topology level (core, cluster, LLC, node) and
+ * shard, records the first cid in the unit and its global index. Global indices
+ * are consecutive integers assigned in cid-walk order, so e.g. core_idx ranges
+ * over [0, nr_cores_at_init) with no gaps. No-topo cids have core/cluster/LLC/
+ * node fields set to -1 but always have valid shard assignments.
+ *
+ * A cluster is the cache-sharing level between the core and the LLC. Where
+ * this level is absent, each core forms a cluster of its own. Where the cache
+ * level spans the LLC, the whole LLC forms one cluster. Each cluster has a
+ * unique cluster_cid and a dense cluster_idx, and its cids form a contiguous
+ * range.
+ *
+ * Every cid therefore has a cluster, so cluster_cid alone does not say whether
+ * the machine has the level: the first cluster of an LLC starts at the LLC's
+ * own base cid and a core-wide one at the core's.
*
* Shards are contiguous CID ranges used as scalable locking/work domains for
* sub-scheduler operations. By default each LLC becomes one shard, split into
* smaller shards if the LLC exceeds the target size. No-topo cids are packed
* into their own max-sized shards.
*
+ * Shards are cut on core boundaries only, so a shard doesn't need to contain a
+ * whole cluster: a cluster may straddle two shards. Only core, cluster, LLC
+ * and node are nested, in that order.
+ *
* New fields are appended, never inserted: scx_bpf_cid_topo() copies this
* struct out sized by the program's own layout, and an older program's copy
* must stay a prefix of the kernel's.
@@ -82,6 +96,8 @@ enum scx_consts {
* @node_idx: global index of that node, in [0, nr_nodes_at_init)
* @shard_cid: first cid of this cid's shard
* @shard_idx: global index of that shard, in [0, scx_nr_cid_shards)
+ * @cluster_cid: first cid of this cid's cluster
+ * @cluster_idx: global index of that cluster, in [0, nr_clusters_at_init)
*/
struct scx_cid_topo {
s32 core_cid;
@@ -92,6 +108,8 @@ struct scx_cid_topo {
s32 node_idx;
s32 shard_cid;
s32 shard_idx;
+ s32 cluster_cid;
+ s32 cluster_idx;
};
enum scx_cid_consts {
--
2.55.0
^ permalink raw reply [flat|nested] only message in thread
only message in thread, other threads:[~2026-09-27 13:37 UTC | newest]
Thread overview: (only message) (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-09-27 13:37 [PATCH v3 sched_ext/for-7.4] sched_ext: cid: Represent clusters explicitly Andrea Righi
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®