* [PATCH 1/2] sched_ext: Add ops.sub_child_ecaps_updated() to report a child's effective cap changes
2026-10-08 9:32 [PATCHSET v2 sched_ext/for-7.4] sched_ext: Add ops.sub_child_ecaps_updated() Tejun Heo
@ 2026-10-08 9:32 ` Tejun Heo
2026-10-08 9:32 ` [PATCH 2/2] sched_ext: scx_qmap: Reset a child's cpuperf targets once its PERF revoke takes effect Tejun Heo
1 sibling, 0 replies; 4+ messages in thread
From: Tejun Heo @ 2026-10-08 9:32 UTC (permalink / raw)
To: sched-ext
Cc: David Vernet, Andrea Righi, Changwoo Min, Emil Tsalapatis,
David Dai, linux-kernel, Tejun Heo
A grant or revoke only records the target caps. They take effect on a cid at
its next dispatch, and nothing tells the parent when. Without that the
parent cannot act on a revoke: a child that ran a cid at a low cpuperf
target leaves the target there when PERF is revoked, as the kernel resets
targets only at root enable. Nor can the parent tell when it may schedule on
the cid again.
Add ops.sub_child_ecaps_updated(), delivered to the direct parent right
after the child's own ops.sub_ecaps_updated() with the child's cgroup id and
the same before and after caps, in the same dispatch context. Both
deliveries are suppressed while the child is bypassing and replayed together
afterwards. A disabled child reports nothing: ops.sub_detach() is where the
parent restores what it had delegated.
A sub is now bypassed before it is linked, so that the grants queued for it
from ops.sub_attach() are consumed while it is bypassed and replayed to it
and the parent once it is enabled. Otherwise a sync consumed before the
sub's ops are registered would reach the parent but never the sub.
v2: Drop the disable report with its sleep lock and nested-dispatch gate,
the parent restores from ops.sub_detach() instead.
Signed-off-by: Tejun Heo <tj@kernel.org>
---
kernel/sched/ext/ext.c | 4 +++
kernel/sched/ext/internal.h | 30 +++++++++++++++++--
kernel/sched/ext/sub.c | 59 ++++++++++++++++++++++++++-----------
3 files changed, 72 insertions(+), 21 deletions(-)
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 248b39d09ce3..2095b78b6bf6 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -8921,6 +8921,7 @@ static void sched_ext_ops_cid__enable(struct task_struct *p, struct scx_enable_a
static void sched_ext_ops__sub_caps_updated(const struct scx_cmask *cmask__arena, u64 caps) {}
static void sched_ext_ops__sub_ecaps_updated(s32 cid, u64 before, u64 after) {}
static void sched_ext_ops__sub_cid_sched_updated(s32 cid, u64 sched) {}
+static void sched_ext_ops__sub_child_ecaps_updated(u64 cgroup_id, s32 cid, u64 before, u64 after) {}
static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
.select_cid = sched_ext_ops__select_cpu,
@@ -8956,6 +8957,7 @@ static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
.sub_caps_updated = sched_ext_ops__sub_caps_updated,
.sub_ecaps_updated = sched_ext_ops__sub_ecaps_updated,
.sub_cid_sched_updated = sched_ext_ops__sub_cid_sched_updated,
+ .sub_child_ecaps_updated = sched_ext_ops__sub_child_ecaps_updated,
.cid_online = sched_ext_ops__cpu_online,
.cid_offline = sched_ext_ops__cpu_offline,
.init_cids = sched_ext_ops__init_cids,
@@ -11552,6 +11554,7 @@ static const u32 scx_kf_allow_flags[] = {
[SCX_OP_IDX(sub_attach)] = SCX_KF_ALLOW_UNLOCKED,
[SCX_OP_IDX(sub_detach)] = SCX_KF_ALLOW_UNLOCKED,
[SCX_OP_IDX(sub_ecaps_updated)] = SCX_KF_ALLOW_ENQUEUE | SCX_KF_ALLOW_DISPATCH,
+ [SCX_OP_IDX(sub_child_ecaps_updated)] = SCX_KF_ALLOW_ENQUEUE | SCX_KF_ALLOW_DISPATCH,
[SCX_OP_IDX(cpu_online)] = SCX_KF_ALLOW_UNLOCKED,
[SCX_OP_IDX(cpu_offline)] = SCX_KF_ALLOW_UNLOCKED,
[SCX_OP_IDX(init_cids)] = SCX_KF_ALLOW_UNLOCKED | SCX_KF_ALLOW_INIT_CIDS,
@@ -11696,6 +11699,7 @@ static int __init scx_init(void)
CID_OFFSET_MATCH(sub_caps_updated, sub_caps_updated);
CID_OFFSET_MATCH(sub_ecaps_updated, sub_ecaps_updated);
CID_OFFSET_MATCH(sub_cid_sched_updated, sub_cid_sched_updated);
+ CID_OFFSET_MATCH(sub_child_ecaps_updated, sub_child_ecaps_updated);
CID_OFFSET_MATCH(init_cids, init_cids);
CID_OFFSET_MATCH(init, init);
CID_OFFSET_MATCH(exit, exit);
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index 8dcab02a38ab..2b637973244c 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -878,8 +878,9 @@ struct sched_ext_ops {
* @args: argument container, see the struct definition
*
* The sub-scheduler holds no caps by this point and can no longer
- * affect any cid. Whatever was delegated to it, e.g. cpuperf targets,
- * is the parent's to restore here.
+ * affect any cid. Its caps went away without a
+ * sub_child_ecaps_updated() report, so whatever was delegated to it,
+ * e.g. cpuperf targets, is the parent's to restore here.
*/
void (*sub_detach)(struct scx_sub_detach_args *args);
@@ -932,6 +933,25 @@ struct sched_ext_ops {
*/
void (*sub_cid_sched_updated)(s32 cid, u64 sched);
+ /**
+ * @sub_child_ecaps_updated: A child's effective caps on a cid changed
+ * @cgroup_id: cgroup id of the direct child
+ * @cid: the cid whose effective caps changed
+ * @before: the child's effective caps as of the last delivery
+ * @after: the child's effective caps now
+ *
+ * Invoked right after the child's sub_ecaps_updated(), once a grant or
+ * revoke is in effect on the cpu. From here on the parent can act on a
+ * cap the child lost, e.g. reset the cpuperf target or schedule on the
+ * cid. A detaching child's caps go away without a report, see
+ * sub_detach(). A cpu going offline drops caps without a report.
+ *
+ * Runs in dispatch context with rq lock held, and can perform all
+ * operations allowed in ops.dispatch() including inserting/moving
+ * tasks.
+ */
+ void (*sub_child_ecaps_updated)(u64 cgroup_id, s32 cid, u64 before, u64 after);
+
/*
* All online ops must come before ops.cpu_online().
*/
@@ -1190,6 +1210,7 @@ struct sched_ext_ops_cid {
void (*sub_caps_updated)(const struct scx_cmask *cmask__arena, u64 caps);
void (*sub_ecaps_updated)(s32 cid, u64 before, u64 after);
void (*sub_cid_sched_updated)(s32 cid, u64 sched);
+ void (*sub_child_ecaps_updated)(u64 cgroup_id, s32 cid, u64 before, u64 after);
void (*cid_online)(s32 cid);
void (*cid_offline)(s32 cid);
s32 (*init_cids)(void);
@@ -1453,7 +1474,10 @@ struct scx_sched_pcpu {
struct llist_node ecaps_to_sync_node;
/* owed a forced update_idle() re-notify on this cpu */
bool idle_renotify;
- /* effective caps as of the last sub_ecaps_updated() delivery */
+ /*
+ * ecaps as of the last sub_ecaps_updated() and
+ * sub_child_ecaps_updated() delivery. See __scx_process_sync_ecaps().
+ */
u64 reported_ecaps;
/*
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 51a53467cf98..c7c95661280b 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -1121,27 +1121,43 @@ void __scx_process_sync_ecaps(struct rq *rq, struct task_struct *prev)
lost_all |= lost;
/*
- * Tell the sched its effective caps on this cid changed. The
- * invocation is equivalent to the dispatch path and may drop
- * and re-acquire the rq lock temporarily while the rest of
- * @batch is held privately, see scx_discard_ecaps_to_sync().
- * The dispatch kfuncs resolve their context on the executing
- * cpu, which under core scheduling can differ from @rq's cpu,
- * so the context is set up there. The rq recorded in it keeps
- * the dispatches targeting @rq.
+ * Tell the sched and its parent that the sched's effective caps
+ * on this cid changed. The invocations are equivalent to the
+ * dispatch path and may drop and re-acquire the rq lock
+ * temporarily while the rest of @batch is held privately, see
+ * scx_discard_ecaps_to_sync(). The dispatch kfuncs resolve
+ * their context on the executing cpu, which under core
+ * scheduling can differ from @rq's cpu, so the context is set
+ * up there. The rq recorded in it keeps the dispatches
+ * targeting @rq.
+ *
+ * Bypass propagates down the hierarchy, so a sched that isn't
+ * bypassing has no bypassing parent. The child's bypass state
+ * gates both deliveries. Both report the same before value, so
+ * one reported_ecaps covers them.
*/
- if (ecaps != pcpu->reported_ecaps &&
- SCX_HAS_OP(pcpu->sch, sub_ecaps_updated) &&
- !scx_bypassing(pcpu->sch, cpu)) {
- struct scx_dsp_ctx *dspc = &this_cpu_ptr(pcpu->sch->pcpu)->dsp_ctx;
+ if (ecaps != pcpu->reported_ecaps && !scx_bypassing(pcpu->sch, cpu)) {
+ struct scx_sched *parent = scx_parent(pcpu->sch);
+ struct scx_dsp_ctx *dspc;
- dspc->rq = rq;
/* stash @prev so nested dispatches can access it */
rq->scx.sub_dispatch_prev = prev;
- SCX_CALL_OP(pcpu->sch, sub_ecaps_updated, rq, scx_cpu_arg(cpu),
- pcpu->reported_ecaps, ecaps);
+ if (SCX_HAS_OP(pcpu->sch, sub_ecaps_updated)) {
+ dspc = &this_cpu_ptr(pcpu->sch->pcpu)->dsp_ctx;
+ dspc->rq = rq;
+ SCX_CALL_OP(pcpu->sch, sub_ecaps_updated, rq,
+ scx_cpu_arg(cpu), pcpu->reported_ecaps, ecaps);
+ scx_flush_dispatch_buf(pcpu->sch, rq);
+ }
+ if (SCX_HAS_OP(parent, sub_child_ecaps_updated)) {
+ dspc = &this_cpu_ptr(parent->pcpu)->dsp_ctx;
+ dspc->rq = rq;
+ SCX_CALL_OP(parent, sub_child_ecaps_updated, rq,
+ pcpu->sch->ops.sub_cgroup_id, scx_cpu_arg(cpu),
+ pcpu->reported_ecaps, ecaps);
+ scx_flush_dispatch_buf(parent, rq);
+ }
rq->scx.sub_dispatch_prev = NULL;
- scx_flush_dispatch_buf(pcpu->sch, rq);
pcpu->reported_ecaps = ecaps;
}
@@ -1940,6 +1956,15 @@ void scx_sub_enable_workfn(struct kthread_work *work)
if (ret)
goto err_disable;
+ /*
+ * Bypass before @sch is linked and grants can reach it. The parent's
+ * delivery advances reported_ecaps, so a sync consumed before @sch's
+ * ops are registered would reach the parent and never be replayed to
+ * @sch. While bypassed, the syncs are consumed without a delivery and
+ * replayed to both at unbypass.
+ */
+ scx_bypass(sch, true);
+
ret = scx_link_sched(sch);
if (ret)
goto err_disable;
@@ -1985,8 +2010,6 @@ void scx_sub_enable_workfn(struct kthread_work *work)
}
sch->sub_attached = true;
- scx_bypass(sch, true);
-
for (i = SCX_OPI_BEGIN; i < SCX_OPI_END; i++)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);
--
2.55.0
^ permalink raw reply [flat|nested] 4+ messages in thread* [PATCH 2/2] sched_ext: scx_qmap: Reset a child's cpuperf targets once its PERF revoke takes effect
2026-10-08 9:32 [PATCHSET v2 sched_ext/for-7.4] sched_ext: Add ops.sub_child_ecaps_updated() Tejun Heo
2026-10-08 9:32 ` [PATCH 1/2] sched_ext: Add ops.sub_child_ecaps_updated() to report a child's effective cap changes Tejun Heo
@ 2026-10-08 9:32 ` Tejun Heo
1 sibling, 0 replies; 4+ messages in thread
From: Tejun Heo @ 2026-10-08 9:32 UTC (permalink / raw)
To: sched-ext
Cc: David Vernet, Andrea Righi, Changwoo Min, Emil Tsalapatis,
David Dai, linux-kernel, Tejun Heo
A child that ran a cid at a low cpuperf target leaves the target there when
PERF is revoked, since the kernel resets targets only at root enable. Reset
the target from ops.sub_child_ecaps_updated() once the revoke is in effect.
A detach reports nothing, so ops.sub_detach() resets the child's cids, the
pool included. Count the deliveries in the hier stats.
v2: Reset from ops.sub_detach() for a detaching child's cids.
Signed-off-by: Tejun Heo <tj@kernel.org>
---
tools/sched_ext/scx_qmap.bpf.c | 33 +++++++++++++++++++++++++++++++--
tools/sched_ext/scx_qmap.c | 7 +++++--
tools/sched_ext/scx_qmap.h | 1 +
3 files changed, 37 insertions(+), 4 deletions(-)
diff --git a/tools/sched_ext/scx_qmap.bpf.c b/tools/sched_ext/scx_qmap.bpf.c
index 7f69394556b6..4c152b44bc91 100644
--- a/tools/sched_ext/scx_qmap.bpf.c
+++ b/tools/sched_ext/scx_qmap.bpf.c
@@ -2167,12 +2167,30 @@ s32 BPF_STRUCT_OPS(qmap_sub_attach, struct scx_sub_attach_args *args)
void BPF_STRUCT_OPS(qmap_sub_detach, struct scx_sub_detach_args *args)
{
- s32 i;
+ u64 cgid = args->ops->sub_cgroup_id;
+ s32 nr_cids = qa.nr_cids;
+ s32 i, cid;
+
+ if (nr_cids < 0 || nr_cids > SCX_QMAP_MAX_CPUS) {
+ scx_bpf_error("-ERANGE");
+ return;
+ }
for (i = 0; i < MAX_SUB_SCHEDS; i++) {
- if (qa.sub_sched_ctxs[i].cgroup_id != args->ops->sub_cgroup_id)
+ if (qa.sub_sched_ctxs[i].cgroup_id != cgid)
continue;
+ /*
+ * The child's caps are gone without a report, see
+ * ops.sub_detach(). cpuperf targets persist until the next
+ * write, reset the child's excl cids and the pool it may have
+ * held.
+ */
+ bpf_arena_for(cid, 0, nr_cids)
+ if (cmask_test(cid, &qa.sub_sched_ctxs[i].granted_cids.mask) ||
+ cmask_test(cid, &qa.rr_cids.mask))
+ scx_bpf_cidperf_set(cid, SCX_CPUPERF_ONE);
+
qa.sub_sched_ctxs[i].cgroup_id = 0;
qa.sub_sched_ctxs[i].weight = 100;
cmask_init(&qa.sub_sched_ctxs[i].granted_cids.mask, 0, qa.nr_cids);
@@ -2208,6 +2226,16 @@ void BPF_STRUCT_OPS(qmap_sub_ecaps_updated, s32 cid, u64 before, u64 after)
execute_partition();
}
+void BPF_STRUCT_OPS(qmap_sub_child_ecaps_updated, u64 cgroup_id, s32 cid, u64 before,
+ u64 after)
+{
+ __sync_fetch_and_add(&qa.nr_child_ecaps, 1);
+
+ /* a child's last target must not outlive its PERF cap */
+ if ((before & ~after) & SCX_CAP_PERF)
+ scx_bpf_cidperf_set(cid, SCX_CPUPERF_ONE);
+}
+
SCX_OPS_CID_DEFINE(qmap_ops,
.flags = SCX_OPS_ENQ_EXITING | SCX_OPS_TID_TO_TASK,
.select_cid = (void *)qmap_select_cid,
@@ -2232,6 +2260,7 @@ SCX_OPS_CID_DEFINE(qmap_ops,
.sub_caps_updated = (void *)qmap_sub_caps_updated,
.sub_ecaps_updated = (void *)qmap_sub_ecaps_updated,
.sub_cid_sched_updated = (void *)qmap_sub_cid_sched_updated,
+ .sub_child_ecaps_updated = (void *)qmap_sub_child_ecaps_updated,
.init_cids = (void *)qmap_init_cids,
.init = (void *)qmap_init,
.exit = (void *)qmap_exit,
diff --git a/tools/sched_ext/scx_qmap.c b/tools/sched_ext/scx_qmap.c
index e83d5e2147dd..17cf584a2a3d 100644
--- a/tools/sched_ext/scx_qmap.c
+++ b/tools/sched_ext/scx_qmap.c
@@ -113,6 +113,7 @@ struct hier_prev {
u64 nr_enq_blocked;
u64 nr_inject_attempts;
u64 nr_rescue_dsp;
+ u64 nr_child_ecaps;
};
/* current wall-clock time as "HH:MM:SS" for the startup and interval headers */
@@ -208,18 +209,20 @@ static void print_hier(struct qmap_arena *qa, struct hier_prev *prev, u64 own_cg
}
format_cid_ranges(qa, CID_SHARED, ranges, sizeof(ranges));
- printf("hier : nsub=%llu excl=%u shared=%s rr=%s reenq cap/immed +%llu/+%llu blocked=+%llu inj=+%llu rescue=+%llu\n",
+ printf("hier : nsub=%llu excl=%u shared=%s rr=%s reenq cap/immed +%llu/+%llu blocked=+%llu inj=+%llu rescue=+%llu child_ecaps=+%llu\n",
(unsigned long long)qa->nr_sub_scheds, qa->part.nr_excl, ranges, rr,
(unsigned long long)(qa->nr_reenq_cap - prev->nr_reenq_cap),
(unsigned long long)(qa->nr_reenq_immed - prev->nr_reenq_immed),
(unsigned long long)(qa->nr_enq_blocked - prev->nr_enq_blocked),
(unsigned long long)(qa->nr_inject_attempts - prev->nr_inject_attempts),
- (unsigned long long)(qa->nr_rescue_dsp - prev->nr_rescue_dsp));
+ (unsigned long long)(qa->nr_rescue_dsp - prev->nr_rescue_dsp),
+ (unsigned long long)(qa->nr_child_ecaps - prev->nr_child_ecaps));
prev->nr_reenq_cap = qa->nr_reenq_cap;
prev->nr_reenq_immed = qa->nr_reenq_immed;
prev->nr_enq_blocked = qa->nr_enq_blocked;
prev->nr_inject_attempts = qa->nr_inject_attempts;
prev->nr_rescue_dsp = qa->nr_rescue_dsp;
+ prev->nr_child_ecaps = qa->nr_child_ecaps;
/*
* alloc is the cid-time the partition handed each participant, and used
diff --git a/tools/sched_ext/scx_qmap.h b/tools/sched_ext/scx_qmap.h
index 949459d06a18..9e6eefb39735 100644
--- a/tools/sched_ext/scx_qmap.h
+++ b/tools/sched_ext/scx_qmap.h
@@ -195,6 +195,7 @@ struct qmap_arena {
u64 nr_enq_blocked; /* SCX_ENQ_BLOCKED dispatches */
u64 nr_inject_attempts; /* fault-injection: dispatches to an unheld cid */
u64 nr_rescue_dsp; /* SCX_ENQ_RESCUE dispatch attempts */
+ u64 nr_child_ecaps; /* ops.sub_child_ecaps_updated() deliveries */
u32 inject_mode; /* fault-injection mode (QMAP_INJ_*) */
};
--
2.55.0
^ permalink raw reply [flat|nested] 4+ messages in thread