* [PATCH 1/2] sched_ext: Factor out sync_pcpu_ecaps()
2026-10-09 9:43 [PATCHSET sched_ext/for-7.4] sched_ext: Report a sub-scheduler's attach-time ecaps before lifting its enable bypass Tejun Heo
@ 2026-10-09 9:43 ` Tejun Heo
2026-10-09 9:43 ` [PATCH 2/2] sched_ext: Report a sub-scheduler's attach-time ecaps before lifting its enable bypass Tejun Heo
1 sibling, 0 replies; 3+ messages in thread
From: Tejun Heo @ 2026-10-09 9:43 UTC (permalink / raw)
To: sched-ext
Cc: David Vernet, Andrea Righi, Changwoo Min, Emil Tsalapatis,
David Dai, linux-kernel, Tejun Heo
Factor sync_pcpu_ecaps() out of __scx_process_sync_ecaps() so that a sched's
ecaps can be synced and reported outside the drain. No functional change.
Signed-off-by: Tejun Heo <tj@kernel.org>
---
kernel/sched/ext/sub.c | 127 +++++++++++++++++++++++------------------
1 file changed, 72 insertions(+), 55 deletions(-)
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 51a53467cf98..313e69841bc9 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -1036,9 +1036,9 @@ static void queue_sync_ecaps(struct scx_sched *sch, s32 cid)
struct scx_sched_pcpu *pcpu = per_cpu_ptr(sch->pcpu, cpu);
/*
- * Pairs with smp_mb() in __scx_process_sync_ecaps(). Either the check
- * below sees the node off the list and queues it, or the in-flight sync
- * sees the caps[] update made before this call.
+ * Pairs with smp_mb() in sync_pcpu_ecaps(). Either the check below sees
+ * the node off the list and queues it, or the in-flight sync sees the
+ * caps[] update made before this call.
*/
smp_mb();
@@ -1061,9 +1061,13 @@ static void discard_queued_syncs(struct rq *rq)
}
/**
- * __scx_process_sync_ecaps - Sync this cpu's ecaps to pshard->caps[]
- * @rq: the cid's cpu rq
+ * sync_pcpu_ecaps - Sync @pcpu's ecaps to its pshard caps on @rq's cid
+ * @rq: the cid's cpu rq, locked and active
+ * @pcpu: the sched's per-cpu state on @rq's cpu
+ * @cid: @rq's cid
+ * @shard: @cid's pshard index
* @prev: @rq's previous task from the in-progress dispatch
+ * @report: whether a change is reported to the sched
*
* pshard->caps[] is the target configuration. pcpu->ecaps is the effective
* transposed copy owned by the cid's cpu and written only here under @rq's
@@ -1072,6 +1076,67 @@ static void discard_queued_syncs(struct rq *rq)
* A sched that newly gains baseline access here is owed an update_idle() so it
* learns the cid's idle state. Such a gain arms the per-rq
* %SCX_RQ_SUB_IDLE_RENOTIFY gate so the next idle pick delivers it.
+ *
+ * Return the caps lost.
+ */
+static u64 sync_pcpu_ecaps(struct rq *rq, struct scx_sched_pcpu *pcpu, s32 cid, s32 shard,
+ struct task_struct *prev, bool report)
+{
+ struct scx_pshard *ps = pcpu->sch->pshard[shard];
+ u64 old, ecaps, lost, gained;
+
+ /* pairs with smp_mb() in queue_sync_ecaps(), see there */
+ smp_mb();
+
+ old = READ_ONCE(pcpu->ecaps);
+ ecaps = calc_effective_caps(ps, cid);
+ WRITE_ONCE(pcpu->ecaps, ecaps);
+
+ lost = old & ~ecaps;
+ gained = ecaps & ~old;
+
+ /*
+ * Tell the sched its effective caps on this cid changed. The invocation
+ * is equivalent to the dispatch path and may drop and re-acquire the rq
+ * lock temporarily while the caller still holds other queued syncs
+ * privately, see scx_discard_ecaps_to_sync(). The dispatch kfuncs
+ * resolve their context on the executing cpu, which under core
+ * scheduling can differ from @rq's cpu, so the context is set up there.
+ * The rq recorded in it keeps the dispatches targeting @rq.
+ */
+ if (ecaps != pcpu->reported_ecaps &&
+ SCX_HAS_OP(pcpu->sch, sub_ecaps_updated) && report) {
+ struct scx_dsp_ctx *dspc = &this_cpu_ptr(pcpu->sch->pcpu)->dsp_ctx;
+
+ dspc->rq = rq;
+ /* stash @prev so nested dispatches can access it */
+ rq->scx.sub_dispatch_prev = prev;
+ SCX_CALL_OP(pcpu->sch, sub_ecaps_updated, rq, scx_cpu_arg(cpu_of(rq)),
+ pcpu->reported_ecaps, ecaps);
+ rq->scx.sub_dispatch_prev = NULL;
+ scx_flush_dispatch_buf(pcpu->sch, rq);
+ pcpu->reported_ecaps = ecaps;
+ }
+
+ /*
+ * Gaining baseline access owes an update_idle() so the sched learns the
+ * cpu's idle state. Arm the per-rq gate so the next idle pick flushes
+ * it. Losing access drops any pending notify.
+ */
+ if (gained & SCX_CAP_BASE) {
+ pcpu->idle_renotify = true;
+ rq->scx.flags |= SCX_RQ_SUB_IDLE_RENOTIFY;
+ } else if (lost & SCX_CAP_BASE) {
+ pcpu->idle_renotify = false;
+ }
+
+ return lost;
+}
+
+/**
+ * __scx_process_sync_ecaps - Sync this cpu's ecaps to pshard->caps[]
+ * @rq: the cid's cpu rq
+ * @prev: @rq's previous task from the in-progress dispatch
*/
void __scx_process_sync_ecaps(struct rq *rq, struct task_struct *prev)
{
@@ -1104,58 +1169,10 @@ void __scx_process_sync_ecaps(struct rq *rq, struct task_struct *prev)
llist_for_each_safe(pos, tmp, batch) {
struct scx_sched_pcpu *pcpu =
container_of(pos, struct scx_sched_pcpu, ecaps_to_sync_node);
- struct scx_pshard *ps = pcpu->sch->pshard[shard];
- u64 old, ecaps, lost, gained;
init_llist_node(pos);
-
- /* pairs with smp_mb() in queue_sync_ecaps(), see there */
- smp_mb();
-
- old = READ_ONCE(pcpu->ecaps);
- ecaps = calc_effective_caps(ps, cid);
- WRITE_ONCE(pcpu->ecaps, ecaps);
-
- lost = old & ~ecaps;
- gained = ecaps & ~old;
- lost_all |= lost;
-
- /*
- * Tell the sched its effective caps on this cid changed. The
- * invocation is equivalent to the dispatch path and may drop
- * and re-acquire the rq lock temporarily while the rest of
- * @batch is held privately, see scx_discard_ecaps_to_sync().
- * The dispatch kfuncs resolve their context on the executing
- * cpu, which under core scheduling can differ from @rq's cpu,
- * so the context is set up there. The rq recorded in it keeps
- * the dispatches targeting @rq.
- */
- if (ecaps != pcpu->reported_ecaps &&
- SCX_HAS_OP(pcpu->sch, sub_ecaps_updated) &&
- !scx_bypassing(pcpu->sch, cpu)) {
- struct scx_dsp_ctx *dspc = &this_cpu_ptr(pcpu->sch->pcpu)->dsp_ctx;
-
- dspc->rq = rq;
- /* stash @prev so nested dispatches can access it */
- rq->scx.sub_dispatch_prev = prev;
- SCX_CALL_OP(pcpu->sch, sub_ecaps_updated, rq, scx_cpu_arg(cpu),
- pcpu->reported_ecaps, ecaps);
- rq->scx.sub_dispatch_prev = NULL;
- scx_flush_dispatch_buf(pcpu->sch, rq);
- pcpu->reported_ecaps = ecaps;
- }
-
- /*
- * Gaining baseline access owes an update_idle() so the sched
- * learns the cpu's idle state. Arm the per-rq gate so the next
- * idle pick flushes it. Losing access drops any pending notify.
- */
- if (gained & SCX_CAP_BASE) {
- pcpu->idle_renotify = true;
- rq->scx.flags |= SCX_RQ_SUB_IDLE_RENOTIFY;
- } else if (lost & SCX_CAP_BASE) {
- pcpu->idle_renotify = false;
- }
+ lost_all |= sync_pcpu_ecaps(rq, pcpu, cid, shard, prev,
+ !scx_bypassing(pcpu->sch, cpu));
}
/*
--
2.55.0
^ permalink raw reply [flat|nested] 3+ messages in thread* [PATCH 2/2] sched_ext: Report a sub-scheduler's attach-time ecaps before lifting its enable bypass
2026-10-09 9:43 [PATCHSET sched_ext/for-7.4] sched_ext: Report a sub-scheduler's attach-time ecaps before lifting its enable bypass Tejun Heo
2026-10-09 9:43 ` [PATCH 1/2] sched_ext: Factor out sync_pcpu_ecaps() Tejun Heo
@ 2026-10-09 9:43 ` Tejun Heo
1 sibling, 0 replies; 3+ messages in thread
From: Tejun Heo @ 2026-10-09 9:43 UTC (permalink / raw)
To: sched-ext
Cc: David Vernet, Andrea Righi, Changwoo Min, Emil Tsalapatis,
David Dai, linux-kernel, Tejun Heo
A sub-scheduler learns which cids it holds through ops.sub_ecaps_updated().
Today, the grants its parent makes when it attaches are reported only after
its enable bypass is lifted, which is also when its tasks are handed to it.
So the sub-scheduler gets its first ops.enqueue() calls before it has been
notified of a single cid and has to hold the tasks somewhere until the
grants arrive.
Report the attach-time grants before lifting bypass instead. The
sub-scheduler then has its complete cid view when the first task arrives and
can place it right away. Nothing else reaches its ops in the meantime, as it
is still bypassing.
Signed-off-by: Tejun Heo <tj@kernel.org>
---
kernel/sched/ext/sub.c | 73 +++++++++++++++++++++++++++++++++++++-----
1 file changed, 65 insertions(+), 8 deletions(-)
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 313e69841bc9..889fe322198e 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -1069,6 +1069,9 @@ static void discard_queued_syncs(struct rq *rq)
* @prev: @rq's previous task from the in-progress dispatch
* @report: whether a change is reported to the sched
*
+ * @prev may be NULL when seeding a bypassing sched. Only a nested
+ * scx_bpf_sub_dispatch() reads it, and a sched being enabled has no children.
+ *
* pshard->caps[] is the target configuration. pcpu->ecaps is the effective
* transposed copy owned by the cid's cpu and written only here under @rq's
* lock.
@@ -1107,13 +1110,21 @@ static u64 sync_pcpu_ecaps(struct rq *rq, struct scx_sched_pcpu *pcpu, s32 cid,
if (ecaps != pcpu->reported_ecaps &&
SCX_HAS_OP(pcpu->sch, sub_ecaps_updated) && report) {
struct scx_dsp_ctx *dspc = &this_cpu_ptr(pcpu->sch->pcpu)->dsp_ctx;
+ struct task_struct *prev_stash;
dspc->rq = rq;
- /* stash @prev so nested dispatches can access it */
+ /*
+ * Stash @prev for nested dispatches. The enable path calls this
+ * while the cpu's own dispatch may be inside an op with the rq
+ * lock dropped. That dispatch needs its stash back when it
+ * resumes, so save and restore it. A bypassing sched's op never
+ * drops the lock, so nothing else writes the stash in between.
+ */
+ prev_stash = rq->scx.sub_dispatch_prev;
rq->scx.sub_dispatch_prev = prev;
SCX_CALL_OP(pcpu->sch, sub_ecaps_updated, rq, scx_cpu_arg(cpu_of(rq)),
pcpu->reported_ecaps, ecaps);
- rq->scx.sub_dispatch_prev = NULL;
+ rq->scx.sub_dispatch_prev = prev_stash;
scx_flush_dispatch_buf(pcpu->sch, rq);
pcpu->reported_ecaps = ecaps;
}
@@ -1184,6 +1195,48 @@ void __scx_process_sync_ecaps(struct rq *rq, struct task_struct *prev)
scx_schedule_reenq_local(rq, SCX_REENQ_CAP_REVOKE);
}
+/**
+ * scx_sub_seed_ecaps - Report @sch's attach-time ecaps before lifting bypass
+ * @sch: sub-scheduler being enabled, still bypassing
+ *
+ * The grants made during ops.sub_attach() were applied without calling
+ * ops.sub_ecaps_updated(), as @sch had no ops registered yet or was already
+ * bypassing. Report them before lifting bypass, so that @sch is notified of its
+ * cids before any task reaches its ops. A sync still queued on a cpu finds
+ * nothing new afterwards. Kicks and inserts from the op are dropped while
+ * bypassing. Lifting bypass reschedules all cpus.
+ */
+static void scx_sub_seed_ecaps(struct scx_sched *sch)
+{
+ s32 cpu;
+
+ for_each_possible_cpu(cpu) {
+ struct rq *rq = cpu_rq(cpu);
+ s32 cid, shard;
+
+ /* unpinned, the lock state the op is called in from dispatch */
+ guard(raw_spin_rq_lock_irqsave)(rq);
+
+ /*
+ * When a cpu goes offline, its ecaps are cleared and must stay
+ * zero until it comes back online. Don't write non-zero ecaps
+ * on an inactive cpu.
+ */
+ if (!cpu_active(cpu))
+ continue;
+ cid = __scx_cpu_to_cid(cpu);
+ shard = rcu_dereference_all(scx_cid_to_shard)[cid];
+
+ /*
+ * An aborting sched gets no op calls.
+ * __scx_process_sync_ecaps() gets that from its bypass test,
+ * but @sch is bypassing here, so test aborting directly.
+ */
+ sync_pcpu_ecaps(rq, per_cpu_ptr(sch->pcpu, cpu), cid, shard, NULL,
+ !READ_ONCE(sch->aborting));
+ }
+}
+
/**
* scx_unbypass_replay_ecaps - Replay a bypass-suppressed ecaps notification
* @rq: rq of the cpu leaving bypass
@@ -1192,9 +1245,8 @@ void __scx_process_sync_ecaps(struct rq *rq, struct task_struct *prev)
* scx_process_sync_ecaps() consumes syncs while bypassing without delivering
* ops.sub_ecaps_updated(), leaving reported_ecaps stale. Nothing re-queues a
* sync when bypass lifts, so without a replay a cid that never changes again
- * would never be notified. The attach-time initial grants are the acute case
- * as they are consumed during the enable bypass window. Re-queue a sync for
- * any undelivered delta so the next dispatch delivers it.
+ * would never be notified. Re-queue a sync for any undelivered delta so the
+ * next dispatch delivers it.
*/
void scx_unbypass_replay_ecaps(struct rq *rq, struct scx_sched *sch)
{
@@ -2145,10 +2197,15 @@ void scx_sub_enable_workfn(struct kthread_work *work)
scx_cgroup_unlock();
percpu_up_write(&scx_fork_rwsem);
- scx_bypass(sch, false);
-
- /* @sch is enabled; deliver any caps owed since its sub_attach() */
+ /*
+ * @sch is enabled but still bypassing. Report the caps and ecaps
+ * granted during ops.sub_attach() before lifting bypass and letting its
+ * tasks reach ops.enqueue().
+ */
scx_sub_seed_caps(sch);
+ scx_sub_seed_ecaps(sch);
+
+ scx_bypass(sch, false);
pr_info("sched_ext: BPF sub-scheduler \"%s\" enabled\n", sch->ops.name);
kobject_uevent(&sch->kobj, KOBJ_ADD);
--
2.55.0
^ permalink raw reply [flat|nested] 3+ messages in thread