mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Tejun Heo <tj@kernel.org>
To: sched-ext@lists.linux.dev
Cc: David Vernet <void@manifault.com>,
	Andrea Righi <arighi@nvidia.com>,
	Changwoo Min <changwoo@igalia.com>,
	Emil Tsalapatis <emil@etsalapatis.com>,
	David Dai <david.dai@linux.dev>,
	linux-kernel@vger.kernel.org, Tejun Heo <tj@kernel.org>
Subject: [PATCH 1/3] sched_ext: Add ops.sub_cid_sched_updated() to report the sched running on a cid
Date: Mon,  5 Oct 2026 07:55:18 -1000	[thread overview]
Message-ID: <20261005175520.2756986-2-tj@kernel.org> (raw)
In-Reply-To: <20261005175520.2756986-1-tj@kernel.org>

A parent sched that delegates cpus to sub-scheds has no way of telling how
much each delegatee is using them. A parent that sizes its delegations by
demand cannot measure it, and one that wants to hand a cid from one sub to
another cannot see which sub it would take the cid from.

Add ops.sub_cid_sched_updated(), invoked on a sched when what it sees
running on a cid changes: NONE when no task of its subtree runs there, SELF
for its own task, or a direct child's cgroup id. Changes inside a child's
subtree are not reported.

The op costs one pointer compare per context switch when the sched does not
change, a notification reaches only the scheds that see a change, and the
whole mechanism sits behind a static key that is on only while a loaded
sched implements the op. After a sched's tasks are gone, its disable closes
every rq session still pointing at it, so nothing refers to it once it is
freed.

The op is cid-form only like the other sub ops and runs with the rq lock
held on the context switch path.

Signed-off-by: Tejun Heo <tj@kernel.org>
---
 kernel/sched/ext/ext.c      |  20 ++++-
 kernel/sched/ext/internal.h |  29 +++++++
 kernel/sched/ext/sub.c      | 166 ++++++++++++++++++++++++++++++++++++
 kernel/sched/ext/sub.h      |   6 ++
 kernel/sched/sched.h        |   2 +
 5 files changed, 222 insertions(+), 1 deletion(-)

diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 96c904b39601..248b39d09ce3 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -3333,6 +3333,8 @@ static void scx_start_task_running(struct rq *rq, struct task_struct *p)
 	if (p->scx.flags & SCX_TASK_RUN_TRACKED)
 		return;
 
+	scx_cid_sched_update(rq, sch);
+
 	if (SCX_HAS_OP(sch, running))
 		SCX_CALL_OP_TASK(sch, running, rq, p);
 
@@ -3593,8 +3595,10 @@ static void put_prev_task_scx(struct rq *rq, struct task_struct *p,
 	}
 
 switch_class:
-	if (next && next->sched_class != &ext_sched_class)
+	if (next && next->sched_class != &ext_sched_class) {
+		scx_cid_sched_update(rq, NULL);
 		switch_class(rq, next);
+	}
 }
 
 static void kick_sync_wait_bal_cb(struct rq *rq)
@@ -4724,6 +4728,13 @@ static void switching_to_scx(struct rq *rq, struct task_struct *p)
 
 static void switched_from_scx(struct rq *rq, struct task_struct *p)
 {
+	/*
+	 * A class change of the running task: sched_change_begin() put @p with
+	 * no successor, which leaves rq->scx.sched set.
+	 */
+	if (task_current_donor(rq, p))
+		scx_cid_sched_update(rq, NULL);
+
 	if (task_dead_and_done(p))
 		return;
 
@@ -7047,6 +7058,8 @@ static void scx_root_disable(struct scx_sched *sch)
 		}
 	}
 
+	scx_ops_cid_sched_updated_disable(sch);
+
 	/* no task is on scx, turn off all the switches and flush in-progress calls */
 	static_branch_disable(&__scx_enabled);
 	static_branch_disable(&__scx_is_cid_type);
@@ -8255,6 +8268,8 @@ static void scx_root_enable_workfn(struct kthread_work *work)
 		if (((void (**)(void))ops)[i])
 			set_bit(i, sch->has_op);
 
+	scx_ops_cid_sched_updated_enable(sch);
+
 	if (sch->ops.cpu_acquire || sch->ops.cpu_release)
 		sch->ops.flags |= SCX_OPS_HAS_CPU_PREEMPT;
 
@@ -8905,6 +8920,7 @@ static void sched_ext_ops_cid__set_cmask(struct task_struct *p, const struct scx
 static void sched_ext_ops_cid__enable(struct task_struct *p, struct scx_enable_args *args) {}
 static void sched_ext_ops__sub_caps_updated(const struct scx_cmask *cmask__arena, u64 caps) {}
 static void sched_ext_ops__sub_ecaps_updated(s32 cid, u64 before, u64 after) {}
+static void sched_ext_ops__sub_cid_sched_updated(s32 cid, u64 sched) {}
 
 static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
 	.select_cid		= sched_ext_ops__select_cpu,
@@ -8939,6 +8955,7 @@ static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
 	.sub_detach		= sched_ext_ops__sub_detach,
 	.sub_caps_updated	= sched_ext_ops__sub_caps_updated,
 	.sub_ecaps_updated	= sched_ext_ops__sub_ecaps_updated,
+	.sub_cid_sched_updated	= sched_ext_ops__sub_cid_sched_updated,
 	.cid_online		= sched_ext_ops__cpu_online,
 	.cid_offline		= sched_ext_ops__cpu_offline,
 	.init_cids		= sched_ext_ops__init_cids,
@@ -11678,6 +11695,7 @@ static int __init scx_init(void)
 	CID_OFFSET_MATCH(sub_detach, sub_detach);
 	CID_OFFSET_MATCH(sub_caps_updated, sub_caps_updated);
 	CID_OFFSET_MATCH(sub_ecaps_updated, sub_ecaps_updated);
+	CID_OFFSET_MATCH(sub_cid_sched_updated, sub_cid_sched_updated);
 	CID_OFFSET_MATCH(init_cids, init_cids);
 	CID_OFFSET_MATCH(init, init);
 	CID_OFFSET_MATCH(exit, exit);
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index d65fec631bdf..23cd34bef684 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -366,6 +366,18 @@ struct scx_sub_detach_args {
 	char			*cgroup_path;
 };
 
+/*
+ * What a sched sees running on a cid, as reported by
+ * ops.sub_cid_sched_updated(): NONE when the cid is idle or runs a task outside
+ * the sched's subtree, SELF for the sched itself, or a direct child by its
+ * cgroup id. The sentinels lie outside the cgroup id space, whose low 32 bits
+ * are an idr value of at most INT_MAX.
+ */
+enum scx_cid_sched_consts {
+	SCX_CID_SCHED_NONE = (u64)-2,
+	SCX_CID_SCHED_SELF = (u64)-1,
+};
+
 /**
  * struct sched_ext_ops - Operation table for BPF scheduler implementation
  *
@@ -900,6 +912,22 @@ struct sched_ext_ops {
 	 */
 	void (*sub_ecaps_updated)(s32 cid, u64 before, u64 after);
 
+	/**
+	 * @sub_cid_sched_updated: The sched running on a cid changed
+	 * @cid: cid whose running sched changed
+	 * @sched: what this sched now sees running there
+	 *
+	 * Invoked when the sched owning the task running on @cid changes.
+	 * @sched is %SCX_CID_SCHED_NONE when no task of this sched's subtree
+	 * runs there, %SCX_CID_SCHED_SELF for its own task, or the cgroup id of
+	 * the direct child whose subtree the task belongs to. Changes inside a
+	 * child's subtree, whether across tasks or nested sub-scheds, are not
+	 * reported. Every sched for which @sched changes is notified.
+	 *
+	 * Runs with the rq lock held on the context switch path.
+	 */
+	void (*sub_cid_sched_updated)(s32 cid, u64 sched);
+
 	/*
 	 * All online ops must come before ops.cpu_online().
 	 */
@@ -1154,6 +1182,7 @@ struct sched_ext_ops_cid {
 	void (*sub_detach)(struct scx_sub_detach_args *args);
 	void (*sub_caps_updated)(const struct scx_cmask *cmask__arena, u64 caps);
 	void (*sub_ecaps_updated)(s32 cid, u64 before, u64 after);
+	void (*sub_cid_sched_updated)(s32 cid, u64 sched);
 	void (*cid_online)(s32 cid);
 	void (*cid_offline)(s32 cid);
 	s32 (*init_cids)(void);
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 48e17aeb1cb6..ce45bc2bb0e0 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -28,6 +28,13 @@
  */
 DEFINE_STATIC_KEY_FALSE(__scx_has_subs);
 
+/*
+ * On while a loaded sched implements ops.sub_cid_sched_updated(), see
+ * scx_ops_cid_sched_updated_enable().
+ */
+static DEFINE_STATIC_KEY_FALSE(__scx_ops_cid_sched_updated_enabled);
+static s32 scx_nr_ops_cid_sched_updated;	/* such scheds, under scx_enable_mutex */
+
 /* latched at root enable before any rescue runs */
 static s32 scx_rescue_bw_1024;
 static s64 scx_rescue_quantum_ns;
@@ -92,6 +99,161 @@ struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sche
 	return scx_skip_subtree_pre(pos, root);
 }
 
+/**
+ * scx_cid_sched_id_from - Resolve what @from sees when @sch runs on a cid
+ * @sch: sched whose task runs there, %NULL for none
+ * @from: sched asking
+ *
+ * Return %SCX_CID_SCHED_NONE when @sch is %NULL or outside @from's subtree,
+ * %SCX_CID_SCHED_SELF for @from itself, or else the cgroup id of the direct
+ * child of @from on the way down to @sch.
+ */
+static u64 scx_cid_sched_id_from(struct scx_sched *sch, struct scx_sched *from)
+{
+	if (!sch || !scx_is_descendant(sch, from))
+		return SCX_CID_SCHED_NONE;
+	if (sch == from)
+		return SCX_CID_SCHED_SELF;
+	return sch->ancestors[from->level + 1]->ops.sub_cgroup_id;
+}
+
+/* notify @from of what it now sees running on @rq's cid */
+static void scx_cid_notify_sched_updated(struct rq *rq, struct scx_sched *sch,
+					 struct scx_sched *from)
+{
+	if (SCX_HAS_OP(from, sub_cid_sched_updated))
+		SCX_CALL_OP(from, sub_cid_sched_updated, rq, __scx_cpu_to_cid(cpu_of(rq)),
+			    scx_cid_sched_id_from(sch, from));
+}
+
+/**
+ * scx_cid_sched_update - Update rq->scx.sched and notify the scheds affected
+ * @rq: rq whose running session opens or closes
+ * @sch: sched starting a session, %NULL to close the open one
+ *
+ * Every sched whose scx_cid_sched_id_from() value changes is notified of the
+ * new one. Above the deepest common ancestor of the two scheds, both map to the
+ * same child, so the walk stops there. With root R, children A and B, and A's
+ * child A1, a switch from an A1 task to a B task reports:
+ *
+ *         R        R:  A -> B
+ *        / \
+ *       A   B      A:  A1 -> NONE      B: NONE -> SELF
+ *       |
+ *       A1         A1: SELF -> NONE
+ *
+ * When the task on one side is not an ext task, there is no common ancestor and
+ * only the other side is walked.
+ */
+void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch)
+{
+	struct scx_sched *prev = rq->scx.sched;
+	s32 level, common = -1;
+
+	lockdep_assert_rq_held(rq);
+
+	if (!static_branch_unlikely(&__scx_ops_cid_sched_updated_enabled) || prev == sch)
+		return;
+
+	rq->scx.sched = sch;
+
+	if (prev && sch) {
+		for (level = min(prev->level, sch->level); level >= 0; level--)
+			if (prev->ancestors[level] == sch->ancestors[level])
+				break;
+		common = level;
+		scx_cid_notify_sched_updated(rq, sch, prev->ancestors[common]);
+	}
+	if (prev)
+		for (level = common + 1; level <= prev->level; level++)
+			scx_cid_notify_sched_updated(rq, sch, prev->ancestors[level]);
+	if (sch)
+		for (level = common + 1; level <= sch->level; level++)
+			scx_cid_notify_sched_updated(rq, sch, sch->ancestors[level]);
+}
+
+/**
+ * scx_ops_cid_sched_updated_enable - Maintain rq->scx.sched if @sch has the op
+ * @sch: sched being enabled, its has_op filled in
+ *
+ * rq->scx.sched is maintained only while a loaded sched implements
+ * ops.sub_cid_sched_updated(), so that everyone else pays one static branch per
+ * session open or close. The first such sched turns the key on and then
+ * initializes every rq from the task running there, in that order so that no
+ * session change in between goes unrecorded. While the key is off every
+ * rq->scx.sched is NULL, see scx_ops_cid_sched_updated_disable().
+ */
+void scx_ops_cid_sched_updated_enable(struct scx_sched *sch)
+{
+	s32 cpu;
+
+	lockdep_assert_held(&scx_enable_mutex);
+
+	if (!SCX_HAS_OP(sch, sub_cid_sched_updated) || scx_nr_ops_cid_sched_updated++)
+		return;
+
+	static_branch_enable(&__scx_ops_cid_sched_updated_enabled);
+
+	for_each_possible_cpu(cpu) {
+		struct rq *rq = cpu_rq(cpu);
+		struct task_struct *p;
+
+		guard(rq_lock_irqsave)(rq);
+		p = rq->donor;
+		if (p->sched_class == &ext_sched_class &&
+		    (p->scx.flags & SCX_TASK_RUN_TRACKED))
+			rq->scx.sched = scx_task_sched(p);
+	}
+}
+
+/**
+ * scx_ops_cid_sched_updated_disable - Stop maintaining rq->scx.sched for @sch
+ * @sch: sched being disabled, its tasks all gone
+ *
+ * Close the session on every rq that still points at @sch, before ops.exit(),
+ * so that no sched is notified of @sch once it is gone. The last sched with the
+ * op turns the key off first and clears every rq, so while the key is off every
+ * rq->scx.sched is NULL. A sched whose enable failed before its has_op fill was
+ * never counted, so has_op gates the decrement.
+ */
+void scx_ops_cid_sched_updated_disable(struct scx_sched *sch)
+{
+	bool last;
+	s32 cpu;
+
+	lockdep_assert_held(&scx_enable_mutex);
+
+	last = SCX_HAS_OP(sch, sub_cid_sched_updated) && !--scx_nr_ops_cid_sched_updated;
+	if (last)
+		static_branch_disable(&__scx_ops_cid_sched_updated_enabled);
+
+	/*
+	 * Re-homing restarts a running task's session under its new sched. A
+	 * task that blocked while its cpu's dispatch had the rq lock dropped is
+	 * put and set next again without a restart, so rq->scx.sched keeps @sch
+	 * until the pick completes:
+	 *
+	 *   cpu 0, __schedule()                 cpu 1, scx_sub_disable(@sch)
+	 *   T of @sch blocks, stops running
+	 *   dispatch drops the rq lock
+	 *                                       re-home T, nothing to restart
+	 *                                       this sweep
+	 *   pick N, N's session opens
+	 *
+	 * Closing the session here reports NONE to @sch and its ancestors now
+	 * and lets the pick report N's sched from NONE instead of from @sch.
+	 */
+	for_each_possible_cpu(cpu) {
+		struct rq *rq = cpu_rq(cpu);
+
+		guard(rq_lock_irqsave)(rq);
+		if (rq->scx.sched == sch)
+			scx_cid_sched_update(rq, NULL);
+		if (last)
+			rq->scx.sched = NULL;
+	}
+}
+
 static struct scx_sched *scx_find_sub_sched(u64 cgroup_id)
 {
 	return rhashtable_lookup(&scx_sched_hash, &cgroup_id,
@@ -1571,6 +1733,8 @@ void scx_sub_disable(struct scx_sched *sch)
 	scx_cgroup_unlock();
 	percpu_up_write(&scx_fork_rwsem);
 
+	scx_ops_cid_sched_updated_disable(sch);
+
 	/*
 	 * All tasks are moved off of @sch but there may still be on-going
 	 * operations (e.g. ops.select_cpu()). Drain them by flushing RCU. Use
@@ -1793,6 +1957,8 @@ void scx_sub_enable_workfn(struct kthread_work *work)
 		if (((void (**)(void))ops)[i])
 			set_bit(i, sch->has_op);
 
+	scx_ops_cid_sched_updated_enable(sch);
+
 	percpu_down_write(&scx_fork_rwsem);
 	scx_cgroup_lock();
 
diff --git a/kernel/sched/ext/sub.h b/kernel/sched/ext/sub.h
index c0174e523f2a..3cd31b24e649 100644
--- a/kernel/sched/ext/sub.h
+++ b/kernel/sched/ext/sub.h
@@ -16,6 +16,9 @@
 
 struct scx_sched *scx_skip_subtree_pre(struct scx_sched *pos, struct scx_sched *root);
 struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root);
+void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch);
+void scx_ops_cid_sched_updated_enable(struct scx_sched *sch);
+void scx_ops_cid_sched_updated_disable(struct scx_sched *sch);
 void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch);
 struct cgroup *sch_cgroup(struct scx_sched *sch);
 void set_cgroup_sched(struct cgroup *cgrp, struct scx_sched *sch);
@@ -94,6 +97,9 @@ static inline struct scx_dispatch_q *scx_resolve_local_dsq(struct scx_sched *sch
 
 static inline struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root) { return pos ? NULL : root; }
 static inline struct scx_sched *scx_skip_subtree_pre(struct scx_sched *pos, struct scx_sched *root) { return NULL; }
+static inline void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch) {}
+static inline void scx_ops_cid_sched_updated_enable(struct scx_sched *sch) {}
+static inline void scx_ops_cid_sched_updated_disable(struct scx_sched *sch) {}
 static inline void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch) {}
 static inline struct cgroup *sch_cgroup(struct scx_sched *sch) { return NULL; }
 static inline const char *sch_cgrp_path(struct scx_sched *sch) { return "/"; }
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 0a34f0da1ec3..fc65296f5c74 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -833,6 +833,8 @@ struct scx_rq {
 #endif
 	u64			clock;			/* current per-rq clock -- see scx_bpf_now() */
 #ifdef CONFIG_EXT_SUB_SCHED
+	/* sched of the open running session, see scx_cid_sched_update() */
+	struct scx_sched	*sched;
 	struct llist_head	ecaps_to_sync;		/* pending ecaps syncs */
 	struct task_struct	*sub_dispatch_prev;
 #endif
-- 
2.55.0


  reply	other threads:[~2026-10-05 17:55 UTC|newest]

Thread overview: 4+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-05 17:55 [PATCHSET sched_ext/for-7.4] sched_ext: Add ops.sub_cid_sched_updated() Tejun Heo
2026-10-05 17:55 ` Tejun Heo [this message]
2026-10-05 17:55 ` [PATCH 2/3] sched_ext: scx_qmap: Size the cid range buffer for large machines Tejun Heo
2026-10-05 17:55 ` [PATCH 3/3] sched_ext: scx_qmap: Show actual cid use per participant Tejun Heo

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261005175520.2756986-2-tj@kernel.org \
    --to=tj@kernel.org \
    --cc=arighi@nvidia.com \
    --cc=changwoo@igalia.com \
    --cc=david.dai@linux.dev \
    --cc=emil@etsalapatis.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=sched-ext@lists.linux.dev \
    --cc=void@manifault.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®