From: Tejun Heo <tj@kernel.org>
To: sched-ext@lists.linux.dev
Cc: David Vernet <void@manifault.com>,
Andrea Righi <arighi@nvidia.com>,
Changwoo Min <changwoo@igalia.com>,
Emil Tsalapatis <emil@etsalapatis.com>,
David Dai <david.dai@linux.dev>,
linux-kernel@vger.kernel.org, Tejun Heo <tj@kernel.org>
Subject: [PATCH 1/3] sched_ext: Add ops.sub_cid_sched_updated() to report the sched running on a cid
Date: Mon, 5 Oct 2026 07:55:18 -1000 [thread overview]
Message-ID: <20261005175520.2756986-2-tj@kernel.org> (raw)
In-Reply-To: <20261005175520.2756986-1-tj@kernel.org>
A parent sched that delegates cpus to sub-scheds has no way of telling how
much each delegatee is using them. A parent that sizes its delegations by
demand cannot measure it, and one that wants to hand a cid from one sub to
another cannot see which sub it would take the cid from.
Add ops.sub_cid_sched_updated(), invoked on a sched when what it sees
running on a cid changes: NONE when no task of its subtree runs there, SELF
for its own task, or a direct child's cgroup id. Changes inside a child's
subtree are not reported.
The op costs one pointer compare per context switch when the sched does not
change, a notification reaches only the scheds that see a change, and the
whole mechanism sits behind a static key that is on only while a loaded
sched implements the op. After a sched's tasks are gone, its disable closes
every rq session still pointing at it, so nothing refers to it once it is
freed.
The op is cid-form only like the other sub ops and runs with the rq lock
held on the context switch path.
Signed-off-by: Tejun Heo <tj@kernel.org>
---
kernel/sched/ext/ext.c | 20 ++++-
kernel/sched/ext/internal.h | 29 +++++++
kernel/sched/ext/sub.c | 166 ++++++++++++++++++++++++++++++++++++
kernel/sched/ext/sub.h | 6 ++
kernel/sched/sched.h | 2 +
5 files changed, 222 insertions(+), 1 deletion(-)
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 96c904b39601..248b39d09ce3 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -3333,6 +3333,8 @@ static void scx_start_task_running(struct rq *rq, struct task_struct *p)
if (p->scx.flags & SCX_TASK_RUN_TRACKED)
return;
+ scx_cid_sched_update(rq, sch);
+
if (SCX_HAS_OP(sch, running))
SCX_CALL_OP_TASK(sch, running, rq, p);
@@ -3593,8 +3595,10 @@ static void put_prev_task_scx(struct rq *rq, struct task_struct *p,
}
switch_class:
- if (next && next->sched_class != &ext_sched_class)
+ if (next && next->sched_class != &ext_sched_class) {
+ scx_cid_sched_update(rq, NULL);
switch_class(rq, next);
+ }
}
static void kick_sync_wait_bal_cb(struct rq *rq)
@@ -4724,6 +4728,13 @@ static void switching_to_scx(struct rq *rq, struct task_struct *p)
static void switched_from_scx(struct rq *rq, struct task_struct *p)
{
+ /*
+ * A class change of the running task: sched_change_begin() put @p with
+ * no successor, which leaves rq->scx.sched set.
+ */
+ if (task_current_donor(rq, p))
+ scx_cid_sched_update(rq, NULL);
+
if (task_dead_and_done(p))
return;
@@ -7047,6 +7058,8 @@ static void scx_root_disable(struct scx_sched *sch)
}
}
+ scx_ops_cid_sched_updated_disable(sch);
+
/* no task is on scx, turn off all the switches and flush in-progress calls */
static_branch_disable(&__scx_enabled);
static_branch_disable(&__scx_is_cid_type);
@@ -8255,6 +8268,8 @@ static void scx_root_enable_workfn(struct kthread_work *work)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);
+ scx_ops_cid_sched_updated_enable(sch);
+
if (sch->ops.cpu_acquire || sch->ops.cpu_release)
sch->ops.flags |= SCX_OPS_HAS_CPU_PREEMPT;
@@ -8905,6 +8920,7 @@ static void sched_ext_ops_cid__set_cmask(struct task_struct *p, const struct scx
static void sched_ext_ops_cid__enable(struct task_struct *p, struct scx_enable_args *args) {}
static void sched_ext_ops__sub_caps_updated(const struct scx_cmask *cmask__arena, u64 caps) {}
static void sched_ext_ops__sub_ecaps_updated(s32 cid, u64 before, u64 after) {}
+static void sched_ext_ops__sub_cid_sched_updated(s32 cid, u64 sched) {}
static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
.select_cid = sched_ext_ops__select_cpu,
@@ -8939,6 +8955,7 @@ static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
.sub_detach = sched_ext_ops__sub_detach,
.sub_caps_updated = sched_ext_ops__sub_caps_updated,
.sub_ecaps_updated = sched_ext_ops__sub_ecaps_updated,
+ .sub_cid_sched_updated = sched_ext_ops__sub_cid_sched_updated,
.cid_online = sched_ext_ops__cpu_online,
.cid_offline = sched_ext_ops__cpu_offline,
.init_cids = sched_ext_ops__init_cids,
@@ -11678,6 +11695,7 @@ static int __init scx_init(void)
CID_OFFSET_MATCH(sub_detach, sub_detach);
CID_OFFSET_MATCH(sub_caps_updated, sub_caps_updated);
CID_OFFSET_MATCH(sub_ecaps_updated, sub_ecaps_updated);
+ CID_OFFSET_MATCH(sub_cid_sched_updated, sub_cid_sched_updated);
CID_OFFSET_MATCH(init_cids, init_cids);
CID_OFFSET_MATCH(init, init);
CID_OFFSET_MATCH(exit, exit);
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index d65fec631bdf..23cd34bef684 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -366,6 +366,18 @@ struct scx_sub_detach_args {
char *cgroup_path;
};
+/*
+ * What a sched sees running on a cid, as reported by
+ * ops.sub_cid_sched_updated(): NONE when the cid is idle or runs a task outside
+ * the sched's subtree, SELF for the sched itself, or a direct child by its
+ * cgroup id. The sentinels lie outside the cgroup id space, whose low 32 bits
+ * are an idr value of at most INT_MAX.
+ */
+enum scx_cid_sched_consts {
+ SCX_CID_SCHED_NONE = (u64)-2,
+ SCX_CID_SCHED_SELF = (u64)-1,
+};
+
/**
* struct sched_ext_ops - Operation table for BPF scheduler implementation
*
@@ -900,6 +912,22 @@ struct sched_ext_ops {
*/
void (*sub_ecaps_updated)(s32 cid, u64 before, u64 after);
+ /**
+ * @sub_cid_sched_updated: The sched running on a cid changed
+ * @cid: cid whose running sched changed
+ * @sched: what this sched now sees running there
+ *
+ * Invoked when the sched owning the task running on @cid changes.
+ * @sched is %SCX_CID_SCHED_NONE when no task of this sched's subtree
+ * runs there, %SCX_CID_SCHED_SELF for its own task, or the cgroup id of
+ * the direct child whose subtree the task belongs to. Changes inside a
+ * child's subtree, whether across tasks or nested sub-scheds, are not
+ * reported. Every sched for which @sched changes is notified.
+ *
+ * Runs with the rq lock held on the context switch path.
+ */
+ void (*sub_cid_sched_updated)(s32 cid, u64 sched);
+
/*
* All online ops must come before ops.cpu_online().
*/
@@ -1154,6 +1182,7 @@ struct sched_ext_ops_cid {
void (*sub_detach)(struct scx_sub_detach_args *args);
void (*sub_caps_updated)(const struct scx_cmask *cmask__arena, u64 caps);
void (*sub_ecaps_updated)(s32 cid, u64 before, u64 after);
+ void (*sub_cid_sched_updated)(s32 cid, u64 sched);
void (*cid_online)(s32 cid);
void (*cid_offline)(s32 cid);
s32 (*init_cids)(void);
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 48e17aeb1cb6..ce45bc2bb0e0 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -28,6 +28,13 @@
*/
DEFINE_STATIC_KEY_FALSE(__scx_has_subs);
+/*
+ * On while a loaded sched implements ops.sub_cid_sched_updated(), see
+ * scx_ops_cid_sched_updated_enable().
+ */
+static DEFINE_STATIC_KEY_FALSE(__scx_ops_cid_sched_updated_enabled);
+static s32 scx_nr_ops_cid_sched_updated; /* such scheds, under scx_enable_mutex */
+
/* latched at root enable before any rescue runs */
static s32 scx_rescue_bw_1024;
static s64 scx_rescue_quantum_ns;
@@ -92,6 +99,161 @@ struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sche
return scx_skip_subtree_pre(pos, root);
}
+/**
+ * scx_cid_sched_id_from - Resolve what @from sees when @sch runs on a cid
+ * @sch: sched whose task runs there, %NULL for none
+ * @from: sched asking
+ *
+ * Return %SCX_CID_SCHED_NONE when @sch is %NULL or outside @from's subtree,
+ * %SCX_CID_SCHED_SELF for @from itself, or else the cgroup id of the direct
+ * child of @from on the way down to @sch.
+ */
+static u64 scx_cid_sched_id_from(struct scx_sched *sch, struct scx_sched *from)
+{
+ if (!sch || !scx_is_descendant(sch, from))
+ return SCX_CID_SCHED_NONE;
+ if (sch == from)
+ return SCX_CID_SCHED_SELF;
+ return sch->ancestors[from->level + 1]->ops.sub_cgroup_id;
+}
+
+/* notify @from of what it now sees running on @rq's cid */
+static void scx_cid_notify_sched_updated(struct rq *rq, struct scx_sched *sch,
+ struct scx_sched *from)
+{
+ if (SCX_HAS_OP(from, sub_cid_sched_updated))
+ SCX_CALL_OP(from, sub_cid_sched_updated, rq, __scx_cpu_to_cid(cpu_of(rq)),
+ scx_cid_sched_id_from(sch, from));
+}
+
+/**
+ * scx_cid_sched_update - Update rq->scx.sched and notify the scheds affected
+ * @rq: rq whose running session opens or closes
+ * @sch: sched starting a session, %NULL to close the open one
+ *
+ * Every sched whose scx_cid_sched_id_from() value changes is notified of the
+ * new one. Above the deepest common ancestor of the two scheds, both map to the
+ * same child, so the walk stops there. With root R, children A and B, and A's
+ * child A1, a switch from an A1 task to a B task reports:
+ *
+ * R R: A -> B
+ * / \
+ * A B A: A1 -> NONE B: NONE -> SELF
+ * |
+ * A1 A1: SELF -> NONE
+ *
+ * When the task on one side is not an ext task, there is no common ancestor and
+ * only the other side is walked.
+ */
+void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch)
+{
+ struct scx_sched *prev = rq->scx.sched;
+ s32 level, common = -1;
+
+ lockdep_assert_rq_held(rq);
+
+ if (!static_branch_unlikely(&__scx_ops_cid_sched_updated_enabled) || prev == sch)
+ return;
+
+ rq->scx.sched = sch;
+
+ if (prev && sch) {
+ for (level = min(prev->level, sch->level); level >= 0; level--)
+ if (prev->ancestors[level] == sch->ancestors[level])
+ break;
+ common = level;
+ scx_cid_notify_sched_updated(rq, sch, prev->ancestors[common]);
+ }
+ if (prev)
+ for (level = common + 1; level <= prev->level; level++)
+ scx_cid_notify_sched_updated(rq, sch, prev->ancestors[level]);
+ if (sch)
+ for (level = common + 1; level <= sch->level; level++)
+ scx_cid_notify_sched_updated(rq, sch, sch->ancestors[level]);
+}
+
+/**
+ * scx_ops_cid_sched_updated_enable - Maintain rq->scx.sched if @sch has the op
+ * @sch: sched being enabled, its has_op filled in
+ *
+ * rq->scx.sched is maintained only while a loaded sched implements
+ * ops.sub_cid_sched_updated(), so that everyone else pays one static branch per
+ * session open or close. The first such sched turns the key on and then
+ * initializes every rq from the task running there, in that order so that no
+ * session change in between goes unrecorded. While the key is off every
+ * rq->scx.sched is NULL, see scx_ops_cid_sched_updated_disable().
+ */
+void scx_ops_cid_sched_updated_enable(struct scx_sched *sch)
+{
+ s32 cpu;
+
+ lockdep_assert_held(&scx_enable_mutex);
+
+ if (!SCX_HAS_OP(sch, sub_cid_sched_updated) || scx_nr_ops_cid_sched_updated++)
+ return;
+
+ static_branch_enable(&__scx_ops_cid_sched_updated_enabled);
+
+ for_each_possible_cpu(cpu) {
+ struct rq *rq = cpu_rq(cpu);
+ struct task_struct *p;
+
+ guard(rq_lock_irqsave)(rq);
+ p = rq->donor;
+ if (p->sched_class == &ext_sched_class &&
+ (p->scx.flags & SCX_TASK_RUN_TRACKED))
+ rq->scx.sched = scx_task_sched(p);
+ }
+}
+
+/**
+ * scx_ops_cid_sched_updated_disable - Stop maintaining rq->scx.sched for @sch
+ * @sch: sched being disabled, its tasks all gone
+ *
+ * Close the session on every rq that still points at @sch, before ops.exit(),
+ * so that no sched is notified of @sch once it is gone. The last sched with the
+ * op turns the key off first and clears every rq, so while the key is off every
+ * rq->scx.sched is NULL. A sched whose enable failed before its has_op fill was
+ * never counted, so has_op gates the decrement.
+ */
+void scx_ops_cid_sched_updated_disable(struct scx_sched *sch)
+{
+ bool last;
+ s32 cpu;
+
+ lockdep_assert_held(&scx_enable_mutex);
+
+ last = SCX_HAS_OP(sch, sub_cid_sched_updated) && !--scx_nr_ops_cid_sched_updated;
+ if (last)
+ static_branch_disable(&__scx_ops_cid_sched_updated_enabled);
+
+ /*
+ * Re-homing restarts a running task's session under its new sched. A
+ * task that blocked while its cpu's dispatch had the rq lock dropped is
+ * put and set next again without a restart, so rq->scx.sched keeps @sch
+ * until the pick completes:
+ *
+ * cpu 0, __schedule() cpu 1, scx_sub_disable(@sch)
+ * T of @sch blocks, stops running
+ * dispatch drops the rq lock
+ * re-home T, nothing to restart
+ * this sweep
+ * pick N, N's session opens
+ *
+ * Closing the session here reports NONE to @sch and its ancestors now
+ * and lets the pick report N's sched from NONE instead of from @sch.
+ */
+ for_each_possible_cpu(cpu) {
+ struct rq *rq = cpu_rq(cpu);
+
+ guard(rq_lock_irqsave)(rq);
+ if (rq->scx.sched == sch)
+ scx_cid_sched_update(rq, NULL);
+ if (last)
+ rq->scx.sched = NULL;
+ }
+}
+
static struct scx_sched *scx_find_sub_sched(u64 cgroup_id)
{
return rhashtable_lookup(&scx_sched_hash, &cgroup_id,
@@ -1571,6 +1733,8 @@ void scx_sub_disable(struct scx_sched *sch)
scx_cgroup_unlock();
percpu_up_write(&scx_fork_rwsem);
+ scx_ops_cid_sched_updated_disable(sch);
+
/*
* All tasks are moved off of @sch but there may still be on-going
* operations (e.g. ops.select_cpu()). Drain them by flushing RCU. Use
@@ -1793,6 +1957,8 @@ void scx_sub_enable_workfn(struct kthread_work *work)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);
+ scx_ops_cid_sched_updated_enable(sch);
+
percpu_down_write(&scx_fork_rwsem);
scx_cgroup_lock();
diff --git a/kernel/sched/ext/sub.h b/kernel/sched/ext/sub.h
index c0174e523f2a..3cd31b24e649 100644
--- a/kernel/sched/ext/sub.h
+++ b/kernel/sched/ext/sub.h
@@ -16,6 +16,9 @@
struct scx_sched *scx_skip_subtree_pre(struct scx_sched *pos, struct scx_sched *root);
struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root);
+void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch);
+void scx_ops_cid_sched_updated_enable(struct scx_sched *sch);
+void scx_ops_cid_sched_updated_disable(struct scx_sched *sch);
void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch);
struct cgroup *sch_cgroup(struct scx_sched *sch);
void set_cgroup_sched(struct cgroup *cgrp, struct scx_sched *sch);
@@ -94,6 +97,9 @@ static inline struct scx_dispatch_q *scx_resolve_local_dsq(struct scx_sched *sch
static inline struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root) { return pos ? NULL : root; }
static inline struct scx_sched *scx_skip_subtree_pre(struct scx_sched *pos, struct scx_sched *root) { return NULL; }
+static inline void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch) {}
+static inline void scx_ops_cid_sched_updated_enable(struct scx_sched *sch) {}
+static inline void scx_ops_cid_sched_updated_disable(struct scx_sched *sch) {}
static inline void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch) {}
static inline struct cgroup *sch_cgroup(struct scx_sched *sch) { return NULL; }
static inline const char *sch_cgrp_path(struct scx_sched *sch) { return "/"; }
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 0a34f0da1ec3..fc65296f5c74 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -833,6 +833,8 @@ struct scx_rq {
#endif
u64 clock; /* current per-rq clock -- see scx_bpf_now() */
#ifdef CONFIG_EXT_SUB_SCHED
+ /* sched of the open running session, see scx_cid_sched_update() */
+ struct scx_sched *sched;
struct llist_head ecaps_to_sync; /* pending ecaps syncs */
struct task_struct *sub_dispatch_prev;
#endif
--
2.55.0
next prev parent reply other threads:[~2026-10-05 17:55 UTC|newest]
Thread overview: 4+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-05 17:55 [PATCHSET sched_ext/for-7.4] sched_ext: Add ops.sub_cid_sched_updated() Tejun Heo
2026-10-05 17:55 ` Tejun Heo [this message]
2026-10-05 17:55 ` [PATCH 2/3] sched_ext: scx_qmap: Size the cid range buffer for large machines Tejun Heo
2026-10-05 17:55 ` [PATCH 3/3] sched_ext: scx_qmap: Show actual cid use per participant Tejun Heo
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261005175520.2756986-2-tj@kernel.org \
--to=tj@kernel.org \
--cc=arighi@nvidia.com \
--cc=changwoo@igalia.com \
--cc=david.dai@linux.dev \
--cc=emil@etsalapatis.com \
--cc=linux-kernel@vger.kernel.org \
--cc=sched-ext@lists.linux.dev \
--cc=void@manifault.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®