* [PATCH 1/3] sched_ext: Add ops.sub_cid_sched_updated() to report the sched running on a cid
2026-10-05 17:55 [PATCHSET sched_ext/for-7.4] sched_ext: Add ops.sub_cid_sched_updated() Tejun Heo
@ 2026-10-05 17:55 ` Tejun Heo
2026-10-05 17:55 ` [PATCH 2/3] sched_ext: scx_qmap: Size the cid range buffer for large machines Tejun Heo
2026-10-05 17:55 ` [PATCH 3/3] sched_ext: scx_qmap: Show actual cid use per participant Tejun Heo
2 siblings, 0 replies; 4+ messages in thread
From: Tejun Heo @ 2026-10-05 17:55 UTC (permalink / raw)
To: sched-ext
Cc: David Vernet, Andrea Righi, Changwoo Min, Emil Tsalapatis,
David Dai, linux-kernel, Tejun Heo
A parent sched that delegates cpus to sub-scheds has no way of telling how
much each delegatee is using them. A parent that sizes its delegations by
demand cannot measure it, and one that wants to hand a cid from one sub to
another cannot see which sub it would take the cid from.
Add ops.sub_cid_sched_updated(), invoked on a sched when what it sees
running on a cid changes: NONE when no task of its subtree runs there, SELF
for its own task, or a direct child's cgroup id. Changes inside a child's
subtree are not reported.
The op costs one pointer compare per context switch when the sched does not
change, a notification reaches only the scheds that see a change, and the
whole mechanism sits behind a static key that is on only while a loaded
sched implements the op. After a sched's tasks are gone, its disable closes
every rq session still pointing at it, so nothing refers to it once it is
freed.
The op is cid-form only like the other sub ops and runs with the rq lock
held on the context switch path.
Signed-off-by: Tejun Heo <tj@kernel.org>
---
kernel/sched/ext/ext.c | 20 ++++-
kernel/sched/ext/internal.h | 29 +++++++
kernel/sched/ext/sub.c | 166 ++++++++++++++++++++++++++++++++++++
kernel/sched/ext/sub.h | 6 ++
kernel/sched/sched.h | 2 +
5 files changed, 222 insertions(+), 1 deletion(-)
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 96c904b39601..248b39d09ce3 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -3333,6 +3333,8 @@ static void scx_start_task_running(struct rq *rq, struct task_struct *p)
if (p->scx.flags & SCX_TASK_RUN_TRACKED)
return;
+ scx_cid_sched_update(rq, sch);
+
if (SCX_HAS_OP(sch, running))
SCX_CALL_OP_TASK(sch, running, rq, p);
@@ -3593,8 +3595,10 @@ static void put_prev_task_scx(struct rq *rq, struct task_struct *p,
}
switch_class:
- if (next && next->sched_class != &ext_sched_class)
+ if (next && next->sched_class != &ext_sched_class) {
+ scx_cid_sched_update(rq, NULL);
switch_class(rq, next);
+ }
}
static void kick_sync_wait_bal_cb(struct rq *rq)
@@ -4724,6 +4728,13 @@ static void switching_to_scx(struct rq *rq, struct task_struct *p)
static void switched_from_scx(struct rq *rq, struct task_struct *p)
{
+ /*
+ * A class change of the running task: sched_change_begin() put @p with
+ * no successor, which leaves rq->scx.sched set.
+ */
+ if (task_current_donor(rq, p))
+ scx_cid_sched_update(rq, NULL);
+
if (task_dead_and_done(p))
return;
@@ -7047,6 +7058,8 @@ static void scx_root_disable(struct scx_sched *sch)
}
}
+ scx_ops_cid_sched_updated_disable(sch);
+
/* no task is on scx, turn off all the switches and flush in-progress calls */
static_branch_disable(&__scx_enabled);
static_branch_disable(&__scx_is_cid_type);
@@ -8255,6 +8268,8 @@ static void scx_root_enable_workfn(struct kthread_work *work)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);
+ scx_ops_cid_sched_updated_enable(sch);
+
if (sch->ops.cpu_acquire || sch->ops.cpu_release)
sch->ops.flags |= SCX_OPS_HAS_CPU_PREEMPT;
@@ -8905,6 +8920,7 @@ static void sched_ext_ops_cid__set_cmask(struct task_struct *p, const struct scx
static void sched_ext_ops_cid__enable(struct task_struct *p, struct scx_enable_args *args) {}
static void sched_ext_ops__sub_caps_updated(const struct scx_cmask *cmask__arena, u64 caps) {}
static void sched_ext_ops__sub_ecaps_updated(s32 cid, u64 before, u64 after) {}
+static void sched_ext_ops__sub_cid_sched_updated(s32 cid, u64 sched) {}
static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
.select_cid = sched_ext_ops__select_cpu,
@@ -8939,6 +8955,7 @@ static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
.sub_detach = sched_ext_ops__sub_detach,
.sub_caps_updated = sched_ext_ops__sub_caps_updated,
.sub_ecaps_updated = sched_ext_ops__sub_ecaps_updated,
+ .sub_cid_sched_updated = sched_ext_ops__sub_cid_sched_updated,
.cid_online = sched_ext_ops__cpu_online,
.cid_offline = sched_ext_ops__cpu_offline,
.init_cids = sched_ext_ops__init_cids,
@@ -11678,6 +11695,7 @@ static int __init scx_init(void)
CID_OFFSET_MATCH(sub_detach, sub_detach);
CID_OFFSET_MATCH(sub_caps_updated, sub_caps_updated);
CID_OFFSET_MATCH(sub_ecaps_updated, sub_ecaps_updated);
+ CID_OFFSET_MATCH(sub_cid_sched_updated, sub_cid_sched_updated);
CID_OFFSET_MATCH(init_cids, init_cids);
CID_OFFSET_MATCH(init, init);
CID_OFFSET_MATCH(exit, exit);
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index d65fec631bdf..23cd34bef684 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -366,6 +366,18 @@ struct scx_sub_detach_args {
char *cgroup_path;
};
+/*
+ * What a sched sees running on a cid, as reported by
+ * ops.sub_cid_sched_updated(): NONE when the cid is idle or runs a task outside
+ * the sched's subtree, SELF for the sched itself, or a direct child by its
+ * cgroup id. The sentinels lie outside the cgroup id space, whose low 32 bits
+ * are an idr value of at most INT_MAX.
+ */
+enum scx_cid_sched_consts {
+ SCX_CID_SCHED_NONE = (u64)-2,
+ SCX_CID_SCHED_SELF = (u64)-1,
+};
+
/**
* struct sched_ext_ops - Operation table for BPF scheduler implementation
*
@@ -900,6 +912,22 @@ struct sched_ext_ops {
*/
void (*sub_ecaps_updated)(s32 cid, u64 before, u64 after);
+ /**
+ * @sub_cid_sched_updated: The sched running on a cid changed
+ * @cid: cid whose running sched changed
+ * @sched: what this sched now sees running there
+ *
+ * Invoked when the sched owning the task running on @cid changes.
+ * @sched is %SCX_CID_SCHED_NONE when no task of this sched's subtree
+ * runs there, %SCX_CID_SCHED_SELF for its own task, or the cgroup id of
+ * the direct child whose subtree the task belongs to. Changes inside a
+ * child's subtree, whether across tasks or nested sub-scheds, are not
+ * reported. Every sched for which @sched changes is notified.
+ *
+ * Runs with the rq lock held on the context switch path.
+ */
+ void (*sub_cid_sched_updated)(s32 cid, u64 sched);
+
/*
* All online ops must come before ops.cpu_online().
*/
@@ -1154,6 +1182,7 @@ struct sched_ext_ops_cid {
void (*sub_detach)(struct scx_sub_detach_args *args);
void (*sub_caps_updated)(const struct scx_cmask *cmask__arena, u64 caps);
void (*sub_ecaps_updated)(s32 cid, u64 before, u64 after);
+ void (*sub_cid_sched_updated)(s32 cid, u64 sched);
void (*cid_online)(s32 cid);
void (*cid_offline)(s32 cid);
s32 (*init_cids)(void);
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 48e17aeb1cb6..ce45bc2bb0e0 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -28,6 +28,13 @@
*/
DEFINE_STATIC_KEY_FALSE(__scx_has_subs);
+/*
+ * On while a loaded sched implements ops.sub_cid_sched_updated(), see
+ * scx_ops_cid_sched_updated_enable().
+ */
+static DEFINE_STATIC_KEY_FALSE(__scx_ops_cid_sched_updated_enabled);
+static s32 scx_nr_ops_cid_sched_updated; /* such scheds, under scx_enable_mutex */
+
/* latched at root enable before any rescue runs */
static s32 scx_rescue_bw_1024;
static s64 scx_rescue_quantum_ns;
@@ -92,6 +99,161 @@ struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sche
return scx_skip_subtree_pre(pos, root);
}
+/**
+ * scx_cid_sched_id_from - Resolve what @from sees when @sch runs on a cid
+ * @sch: sched whose task runs there, %NULL for none
+ * @from: sched asking
+ *
+ * Return %SCX_CID_SCHED_NONE when @sch is %NULL or outside @from's subtree,
+ * %SCX_CID_SCHED_SELF for @from itself, or else the cgroup id of the direct
+ * child of @from on the way down to @sch.
+ */
+static u64 scx_cid_sched_id_from(struct scx_sched *sch, struct scx_sched *from)
+{
+ if (!sch || !scx_is_descendant(sch, from))
+ return SCX_CID_SCHED_NONE;
+ if (sch == from)
+ return SCX_CID_SCHED_SELF;
+ return sch->ancestors[from->level + 1]->ops.sub_cgroup_id;
+}
+
+/* notify @from of what it now sees running on @rq's cid */
+static void scx_cid_notify_sched_updated(struct rq *rq, struct scx_sched *sch,
+ struct scx_sched *from)
+{
+ if (SCX_HAS_OP(from, sub_cid_sched_updated))
+ SCX_CALL_OP(from, sub_cid_sched_updated, rq, __scx_cpu_to_cid(cpu_of(rq)),
+ scx_cid_sched_id_from(sch, from));
+}
+
+/**
+ * scx_cid_sched_update - Update rq->scx.sched and notify the scheds affected
+ * @rq: rq whose running session opens or closes
+ * @sch: sched starting a session, %NULL to close the open one
+ *
+ * Every sched whose scx_cid_sched_id_from() value changes is notified of the
+ * new one. Above the deepest common ancestor of the two scheds, both map to the
+ * same child, so the walk stops there. With root R, children A and B, and A's
+ * child A1, a switch from an A1 task to a B task reports:
+ *
+ * R R: A -> B
+ * / \
+ * A B A: A1 -> NONE B: NONE -> SELF
+ * |
+ * A1 A1: SELF -> NONE
+ *
+ * When the task on one side is not an ext task, there is no common ancestor and
+ * only the other side is walked.
+ */
+void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch)
+{
+ struct scx_sched *prev = rq->scx.sched;
+ s32 level, common = -1;
+
+ lockdep_assert_rq_held(rq);
+
+ if (!static_branch_unlikely(&__scx_ops_cid_sched_updated_enabled) || prev == sch)
+ return;
+
+ rq->scx.sched = sch;
+
+ if (prev && sch) {
+ for (level = min(prev->level, sch->level); level >= 0; level--)
+ if (prev->ancestors[level] == sch->ancestors[level])
+ break;
+ common = level;
+ scx_cid_notify_sched_updated(rq, sch, prev->ancestors[common]);
+ }
+ if (prev)
+ for (level = common + 1; level <= prev->level; level++)
+ scx_cid_notify_sched_updated(rq, sch, prev->ancestors[level]);
+ if (sch)
+ for (level = common + 1; level <= sch->level; level++)
+ scx_cid_notify_sched_updated(rq, sch, sch->ancestors[level]);
+}
+
+/**
+ * scx_ops_cid_sched_updated_enable - Maintain rq->scx.sched if @sch has the op
+ * @sch: sched being enabled, its has_op filled in
+ *
+ * rq->scx.sched is maintained only while a loaded sched implements
+ * ops.sub_cid_sched_updated(), so that everyone else pays one static branch per
+ * session open or close. The first such sched turns the key on and then
+ * initializes every rq from the task running there, in that order so that no
+ * session change in between goes unrecorded. While the key is off every
+ * rq->scx.sched is NULL, see scx_ops_cid_sched_updated_disable().
+ */
+void scx_ops_cid_sched_updated_enable(struct scx_sched *sch)
+{
+ s32 cpu;
+
+ lockdep_assert_held(&scx_enable_mutex);
+
+ if (!SCX_HAS_OP(sch, sub_cid_sched_updated) || scx_nr_ops_cid_sched_updated++)
+ return;
+
+ static_branch_enable(&__scx_ops_cid_sched_updated_enabled);
+
+ for_each_possible_cpu(cpu) {
+ struct rq *rq = cpu_rq(cpu);
+ struct task_struct *p;
+
+ guard(rq_lock_irqsave)(rq);
+ p = rq->donor;
+ if (p->sched_class == &ext_sched_class &&
+ (p->scx.flags & SCX_TASK_RUN_TRACKED))
+ rq->scx.sched = scx_task_sched(p);
+ }
+}
+
+/**
+ * scx_ops_cid_sched_updated_disable - Stop maintaining rq->scx.sched for @sch
+ * @sch: sched being disabled, its tasks all gone
+ *
+ * Close the session on every rq that still points at @sch, before ops.exit(),
+ * so that no sched is notified of @sch once it is gone. The last sched with the
+ * op turns the key off first and clears every rq, so while the key is off every
+ * rq->scx.sched is NULL. A sched whose enable failed before its has_op fill was
+ * never counted, so has_op gates the decrement.
+ */
+void scx_ops_cid_sched_updated_disable(struct scx_sched *sch)
+{
+ bool last;
+ s32 cpu;
+
+ lockdep_assert_held(&scx_enable_mutex);
+
+ last = SCX_HAS_OP(sch, sub_cid_sched_updated) && !--scx_nr_ops_cid_sched_updated;
+ if (last)
+ static_branch_disable(&__scx_ops_cid_sched_updated_enabled);
+
+ /*
+ * Re-homing restarts a running task's session under its new sched. A
+ * task that blocked while its cpu's dispatch had the rq lock dropped is
+ * put and set next again without a restart, so rq->scx.sched keeps @sch
+ * until the pick completes:
+ *
+ * cpu 0, __schedule() cpu 1, scx_sub_disable(@sch)
+ * T of @sch blocks, stops running
+ * dispatch drops the rq lock
+ * re-home T, nothing to restart
+ * this sweep
+ * pick N, N's session opens
+ *
+ * Closing the session here reports NONE to @sch and its ancestors now
+ * and lets the pick report N's sched from NONE instead of from @sch.
+ */
+ for_each_possible_cpu(cpu) {
+ struct rq *rq = cpu_rq(cpu);
+
+ guard(rq_lock_irqsave)(rq);
+ if (rq->scx.sched == sch)
+ scx_cid_sched_update(rq, NULL);
+ if (last)
+ rq->scx.sched = NULL;
+ }
+}
+
static struct scx_sched *scx_find_sub_sched(u64 cgroup_id)
{
return rhashtable_lookup(&scx_sched_hash, &cgroup_id,
@@ -1571,6 +1733,8 @@ void scx_sub_disable(struct scx_sched *sch)
scx_cgroup_unlock();
percpu_up_write(&scx_fork_rwsem);
+ scx_ops_cid_sched_updated_disable(sch);
+
/*
* All tasks are moved off of @sch but there may still be on-going
* operations (e.g. ops.select_cpu()). Drain them by flushing RCU. Use
@@ -1793,6 +1957,8 @@ void scx_sub_enable_workfn(struct kthread_work *work)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);
+ scx_ops_cid_sched_updated_enable(sch);
+
percpu_down_write(&scx_fork_rwsem);
scx_cgroup_lock();
diff --git a/kernel/sched/ext/sub.h b/kernel/sched/ext/sub.h
index c0174e523f2a..3cd31b24e649 100644
--- a/kernel/sched/ext/sub.h
+++ b/kernel/sched/ext/sub.h
@@ -16,6 +16,9 @@
struct scx_sched *scx_skip_subtree_pre(struct scx_sched *pos, struct scx_sched *root);
struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root);
+void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch);
+void scx_ops_cid_sched_updated_enable(struct scx_sched *sch);
+void scx_ops_cid_sched_updated_disable(struct scx_sched *sch);
void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch);
struct cgroup *sch_cgroup(struct scx_sched *sch);
void set_cgroup_sched(struct cgroup *cgrp, struct scx_sched *sch);
@@ -94,6 +97,9 @@ static inline struct scx_dispatch_q *scx_resolve_local_dsq(struct scx_sched *sch
static inline struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root) { return pos ? NULL : root; }
static inline struct scx_sched *scx_skip_subtree_pre(struct scx_sched *pos, struct scx_sched *root) { return NULL; }
+static inline void scx_cid_sched_update(struct rq *rq, struct scx_sched *sch) {}
+static inline void scx_ops_cid_sched_updated_enable(struct scx_sched *sch) {}
+static inline void scx_ops_cid_sched_updated_disable(struct scx_sched *sch) {}
static inline void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch) {}
static inline struct cgroup *sch_cgroup(struct scx_sched *sch) { return NULL; }
static inline const char *sch_cgrp_path(struct scx_sched *sch) { return "/"; }
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 0a34f0da1ec3..fc65296f5c74 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -833,6 +833,8 @@ struct scx_rq {
#endif
u64 clock; /* current per-rq clock -- see scx_bpf_now() */
#ifdef CONFIG_EXT_SUB_SCHED
+ /* sched of the open running session, see scx_cid_sched_update() */
+ struct scx_sched *sched;
struct llist_head ecaps_to_sync; /* pending ecaps syncs */
struct task_struct *sub_dispatch_prev;
#endif
--
2.55.0
^ permalink raw reply [flat|nested] 4+ messages in thread* [PATCH 3/3] sched_ext: scx_qmap: Show actual cid use per participant
2026-10-05 17:55 [PATCHSET sched_ext/for-7.4] sched_ext: Add ops.sub_cid_sched_updated() Tejun Heo
2026-10-05 17:55 ` [PATCH 1/3] sched_ext: Add ops.sub_cid_sched_updated() to report the sched running on a cid Tejun Heo
2026-10-05 17:55 ` [PATCH 2/3] sched_ext: scx_qmap: Size the cid range buffer for large machines Tejun Heo
@ 2026-10-05 17:55 ` Tejun Heo
2 siblings, 0 replies; 4+ messages in thread
From: Tejun Heo @ 2026-10-05 17:55 UTC (permalink / raw)
To: sched-ext
Cc: David Vernet, Andrea Righi, Changwoo Min, Emil Tsalapatis,
David Dai, linux-kernel, Tejun Heo
qmap's hier stats show how much cid-time the partition handed each
participant, not how much of it was used. Track the sched running on each
cid from ops.sub_cid_sched_updated() and charge the intervals to the
participant, then print the use next to the allocation.
Signed-off-by: Tejun Heo <tj@kernel.org>
---
tools/sched_ext/scx_qmap.bpf.c | 87 ++++++++++++++++++++++++++++++++--
tools/sched_ext/scx_qmap.c | 38 ++++++++++++---
tools/sched_ext/scx_qmap.h | 11 +++++
3 files changed, 127 insertions(+), 9 deletions(-)
diff --git a/tools/sched_ext/scx_qmap.bpf.c b/tools/sched_ext/scx_qmap.bpf.c
index 3566e3e02e21..7f69394556b6 100644
--- a/tools/sched_ext/scx_qmap.bpf.c
+++ b/tools/sched_ext/scx_qmap.bpf.c
@@ -1782,13 +1782,90 @@ static void redistribute(void)
}
/*
- * Userspace pokes this (PROG_RUN) to bring alloc_ns[] current before reading
- * it for the stats display. Skipping when the partition guard is held is
- * fine - alloc_ts is untouched, so the elapsed time is charged next time.
+ * Owner id for a @sched value of ops.sub_cid_sched_updated(). A child is
+ * attached before its first task runs and its tasks are re-homed before it
+ * detaches, so a child's cgroup id always has its slot.
+ */
+static s32 cid_sched_owner(u64 sched)
+{
+ s32 i;
+
+ if (sched == SCX_CID_SCHED_NONE)
+ return CID_NONE;
+ if (sched == SCX_CID_SCHED_SELF)
+ return CID_SELF;
+ bpf_for(i, 0, MAX_SUB_SCHEDS)
+ if (qa.sub_sched_ctxs[i].cgroup_id == sched)
+ return i;
+ return CID_NONE;
+}
+
+/* used_ns[] is summed from every cpu, hence the atomic adds */
+static void cid_sched_charge(s32 cid, u64 now)
+{
+ s32 owner = qa.cid_sched[cid];
+ u64 delta = now - qa.cid_sched_since[cid];
+
+ if (owner >= 0 && owner < MAX_SUB_SCHEDS)
+ __sync_fetch_and_add(&qa.used_ns[owner], delta);
+ else if (owner == CID_SELF)
+ __sync_fetch_and_add(&qa.self_used_ns, delta);
+ qa.cid_sched_since[cid] = now;
+}
+
+void BPF_STRUCT_OPS(qmap_sub_cid_sched_updated, s32 cid, u64 sched)
+{
+ if (cid < 0 || cid >= SCX_QMAP_MAX_CPUS)
+ return;
+
+ cid_sched_charge(cid, bpf_ktime_get_ns());
+ qa.cid_sched[cid] = cid_sched_owner(sched);
+}
+
+/*
+ * Snapshot the used time for the stats display: the closed intervals plus the
+ * ones still open. The reads race the notifications on other cpus, so an
+ * interval closing in between can be missing from one snapshot or counted in
+ * two. The next snapshot evens it out and the display floors a negative
+ * difference at zero.
+ */
+static void snapshot_used(void)
+{
+ u64 now = bpf_ktime_get_ns();
+ s32 nr_cids = qa.nr_cids;
+ s32 cid, i;
+
+ if (nr_cids < 0 || nr_cids > SCX_QMAP_MAX_CPUS)
+ return;
+
+ bpf_for(i, 0, MAX_SUB_SCHEDS)
+ qa.used_snap_ns[i] = qa.used_ns[i];
+ qa.self_used_snap_ns = qa.self_used_ns;
+
+ bpf_for(cid, 0, nr_cids) {
+ s32 owner = qa.cid_sched[cid];
+ u64 since = qa.cid_sched_since[cid];
+
+ /* restarted after @now by a notification on another cpu */
+ if (since > now)
+ continue;
+ if (owner >= 0 && owner < MAX_SUB_SCHEDS)
+ qa.used_snap_ns[owner] += now - since;
+ else if (owner == CID_SELF)
+ qa.self_used_snap_ns += now - since;
+ }
+}
+
+/*
+ * Userspace pokes this (PROG_RUN) to bring alloc_ns[] and the used snapshot
+ * current before reading them for the stats display. Skipping the alloc part
+ * when the partition guard is held is fine - alloc_ts is untouched, so the
+ * elapsed time is charged next time.
*/
SEC("syscall")
int flush_alloc(void *ctx)
{
+ snapshot_used();
if (part_try_start()) {
account_alloc();
part_end();
@@ -1940,6 +2017,9 @@ s32 BPF_STRUCT_OPS_SLEEPABLE(qmap_init)
/* cache the cid count, trusted to be <= SCX_QMAP_MAX_CPUS hereafter */
qa.nr_cids = nr_cids;
+ bpf_for(i, 0, nr_cids)
+ qa.cid_sched[i] = CID_NONE;
+
/* cmasks are embedded in qa, so they only need initializing */
cmask_init(&qa.idle_cids.mask, 0, nr_cids);
cmask_init(&qa.rr_cids.mask, 0, nr_cids);
@@ -2151,6 +2231,7 @@ SCX_OPS_CID_DEFINE(qmap_ops,
.sub_detach = (void *)qmap_sub_detach,
.sub_caps_updated = (void *)qmap_sub_caps_updated,
.sub_ecaps_updated = (void *)qmap_sub_ecaps_updated,
+ .sub_cid_sched_updated = (void *)qmap_sub_cid_sched_updated,
.init_cids = (void *)qmap_init_cids,
.init = (void *)qmap_init,
.exit = (void *)qmap_exit,
diff --git a/tools/sched_ext/scx_qmap.c b/tools/sched_ext/scx_qmap.c
index 96e485c9e877..e83d5e2147dd 100644
--- a/tools/sched_ext/scx_qmap.c
+++ b/tools/sched_ext/scx_qmap.c
@@ -105,6 +105,8 @@ struct hier_prev {
u64 alloc_ns[MAX_SUB_SCHEDS];
u64 self_alloc_ns;
u64 alloc_window_ns;
+ u64 used_snap_ns[MAX_SUB_SCHEDS];
+ u64 self_used_snap_ns;
u64 nr_dsps[MAX_SUB_SCHEDS];
u64 nr_reenq_cap;
u64 nr_reenq_immed;
@@ -159,6 +161,18 @@ static void format_cid_ranges(struct qmap_arena *qa, s32 owner, char *buf, size_
strcpy(buf, "-");
}
+/*
+ * Delta of a cumulative ns counter over the interval, as a fraction of the
+ * interval. The used snapshot can briefly run behind the previous one (see
+ * snapshot_used()), hence the floor.
+ */
+static double delta_ratio(u64 cur, u64 prev, double secs)
+{
+ s64 delta = cur - prev;
+
+ return secs > 0 && delta > 0 ? delta / (secs * 1e9) : 0.0;
+}
+
/* partition summary + one row per sched: weight, cpus, dispatch rate, cids */
static void print_hier(struct qmap_arena *qa, struct hier_prev *prev, u64 own_cgid)
{
@@ -207,15 +221,24 @@ static void print_hier(struct qmap_arena *qa, struct hier_prev *prev, u64 own_cg
prev->nr_inject_attempts = qa->nr_inject_attempts;
prev->nr_rescue_dsp = qa->nr_rescue_dsp;
- printf("hier : %-4s %10s %4s %6s %8s %s\n",
- "", "cgroup", "w", "alloc", "disp/s", "cids");
+ /*
+ * alloc is the cid-time the partition handed each participant, and used
+ * is the cid-time its tasks actually ran, per
+ * ops.sub_cid_sched_updated(). Both are in cpus over the window.
+ */
+ printf("hier : %-4s %10s %4s %6s %6s %8s %s\n",
+ "", "cgroup", "w", "alloc", "used", "disp/s", "cids");
format_cid_ranges(qa, CID_SELF, ranges, sizeof(ranges));
- printf("hier : %-4s %10llu %4u %6.2f %8s %s\n", "self",
+ printf("hier : %-4s %10llu %4u %6.2f %6.2f %8s %s\n", "self",
(unsigned long long)own_cgid, 100,
- secs > 0 ? (qa->self_alloc_ns - prev->self_alloc_ns) / (secs * 1e9) : 0.0,
+ delta_ratio(qa->self_alloc_ns, prev->self_alloc_ns, secs),
+ delta_ratio(qa->self_used_snap_ns, prev->self_used_snap_ns, secs),
"-", ranges);
prev->self_alloc_ns = qa->self_alloc_ns;
+ /* used accrues without a window, so prev moves only with the window */
+ if (secs > 0)
+ prev->self_used_snap_ns = qa->self_used_snap_ns;
for (i = 0; i < MAX_SUB_SCHEDS; i++) {
struct sub_sched_ctx *sc = &qa->sub_sched_ctxs[i];
@@ -225,12 +248,15 @@ static void print_hier(struct qmap_arena *qa, struct hier_prev *prev, u64 own_cg
snprintf(who, sizeof(who), "sub%u", i);
format_cid_ranges(qa, i, ranges, sizeof(ranges));
- printf("hier : %-4s %10llu %4u %6.2f %8.1f %s\n", who,
+ printf("hier : %-4s %10llu %4u %6.2f %6.2f %8.1f %s\n", who,
(unsigned long long)sc->cgroup_id, sc->weight,
- secs > 0 ? (qa->alloc_ns[i] - prev->alloc_ns[i]) / (secs * 1e9) : 0.0,
+ delta_ratio(qa->alloc_ns[i], prev->alloc_ns[i], secs),
+ delta_ratio(qa->used_snap_ns[i], prev->used_snap_ns[i], secs),
secs > 0 ? (sc->nr_dsps - prev->nr_dsps[i]) / secs : 0.0,
ranges);
prev->alloc_ns[i] = qa->alloc_ns[i];
+ if (secs > 0)
+ prev->used_snap_ns[i] = qa->used_snap_ns[i];
prev->nr_dsps[i] = sc->nr_dsps;
}
}
diff --git a/tools/sched_ext/scx_qmap.h b/tools/sched_ext/scx_qmap.h
index 089c5176c4ec..949459d06a18 100644
--- a/tools/sched_ext/scx_qmap.h
+++ b/tools/sched_ext/scx_qmap.h
@@ -163,6 +163,17 @@ struct qmap_arena {
u64 alloc_ts; /* last accounting timestamp */
u64 alloc_window_ns; /* total accounted time, the alloc denominator */
+ /*
+ * The per-cid fields are written only by that cid's notifications and
+ * read by flush_alloc() for the snapshot userspace displays.
+ */
+ s32 cid_sched[SCX_QMAP_MAX_CPUS]; /* per cid: owner id of the sched running there */
+ u64 cid_sched_since[SCX_QMAP_MAX_CPUS]; /* when that last changed */
+ u64 used_ns[MAX_SUB_SCHEDS]; /* per child slot, closed intervals */
+ u64 self_used_ns;
+ u64 used_snap_ns[MAX_SUB_SCHEDS]; /* used_ns[] plus the intervals open at the flush */
+ u64 self_used_snap_ns;
+
/* bpf-internal cmasks (embedded, see struct qmap_cmask) */
struct qmap_cmask self_cids; /* cids this node runs its own tasks on */
struct qmap_cmask avail_cids; /* cids with caps in effect on the cpu */
--
2.55.0
^ permalink raw reply [flat|nested] 4+ messages in thread