* [PATCH sched_ext/for-7.3-fixes] sched_ext: Don't leave balance callbacks queued across a pick retry
@ 2026-10-10 9:52 Andrea Righi
0 siblings, 0 replies; only message in thread
From: Andrea Righi @ 2026-10-10 9:52 UTC (permalink / raw)
To: Tejun Heo, David Vernet, Changwoo Min; +Cc: sched-ext, linux-kernel
The dispatch path can request deferred work to be run from a balance
callback and pick_task_scx() queues rq->scx.deferred_bal_cb right after
dispatching. However, a balance callback must be consumed before the rq
lock is released and pick_task_scx() is not necessarily the end of the
pick:
- it returns RETRY_TASK when a higher priority class task has been
enqueued while the dispatch had the rq lock released;
- the dl_server pick can return NULL, in which case the lower classes
are picked next.
In both cases the callback stays queued while the core keeps picking,
and the rq lock can be released again, either by prev_balance() or by
another dispatch. For example, with an RT task as prev:
__schedule()
__pick_next_task()
pick_task_scx()
dispatch_pick()
maybe_queue_balance_callback() <- deferred_bal_cb queued
return RETRY_TASK
prev_balance()
balance_rt()
pull_rt_task()
double_lock_balance() <- rq lock released
Any CPU acquiring the rq lock in that window finds a pending balance
callback and triggers the following warning in rq_pin_lock():
WARNING: kernel/sched/sched.h:1919 at rq_lock+0x7f/0xe0, CPU#10: stress-ng-sleep/533
CPU: 10 UID: 0 PID: 533 Comm: stress-ng-sleep Not tainted 7.2.0-rc6-virtme #61 PREEMPT(full)
Sched_ext: qmap (enabled+all)
Call Trace:
<IRQ>
try_to_wake_up+0x2c1/0x5c0
hrtimer_wakeup+0x22/0x30
__hrtimer_run_queues+0x1d5/0x350
hrtimer_interrupt+0x106/0x1e0
__sysvec_apic_timer_interrupt+0x6f/0x210
sysvec_apic_timer_interrupt+0x71/0x90
</IRQ>
This can be reproduced running scx_qmap -I together with a SCHED_FIFO
workload that frequently sleeps, e.g.:
$ vng -v --cpus 16 --memory 4G -- \
'tools/sched_ext/build/bin/scx_qmap -I >/dev/null & \
stress-ng --sleep 4 --sched fifo --sched-prio 1 -t 30 -q'
Fix this by queueing the balance callbacks only when the pick is final.
Otherwise, defer the pending work to the irq_work path, which doesn't
depend on the current rq lock section and leave the kick sync wait
pending for the next pick.
Fixes: 4c95380701f5 ("sched/ext: Fold balance_scx() into pick_task_scx()")
Cc: stable@vger.kernel.org # v6.19+
Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
kernel/sched/ext/ext.c | 74 +++++++++++++++++++++++++++++-------------
1 file changed, 51 insertions(+), 23 deletions(-)
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index ac8fe70a9fb6f..10b3b95f70a1b 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -3311,18 +3311,6 @@ static enum scx_dsp_verdict dispatch_pick(struct rq *rq, struct rq_flags *rf,
rq_unpin_lock(rq, rf);
verdict = dispatch_one(rq, prev);
rq_repin_lock(rq, rf);
- maybe_queue_balance_callback(rq);
-
- /*
- * Defer to a balance callback which can drop rq lock and enable IRQs.
- * Waiting directly in the pick path would deadlock against CPUs sending
- * us IPIs (e.g. TLB flushes) while we wait for them.
- */
- if (unlikely(rq->scx.kick_sync_pending)) {
- rq->scx.kick_sync_pending = false;
- queue_balance_callback(rq, &rq->scx.kick_sync_bal_cb,
- kick_sync_wait_bal_cb);
- }
return verdict;
}
@@ -3349,16 +3337,8 @@ static enum scx_dsp_verdict dispatch_core_pick(struct rq *rq, struct rq_flags *r
verdict = dispatch_one(rq, prev);
- if (cpu_of(rq) == smp_processor_id()) {
- maybe_queue_balance_callback(rq);
-
- /* see dispatch_pick() */
- if (unlikely(rq->scx.kick_sync_pending)) {
- rq->scx.kick_sync_pending = false;
- queue_balance_callback(rq, &rq->scx.kick_sync_bal_cb,
- kick_sync_wait_bal_cb);
- }
- } else if (unlikely(rq->scx.flags & SCX_RQ_BAL_CB_PENDING)) {
+ if (cpu_of(rq) != smp_processor_id() &&
+ unlikely(rq->scx.flags & SCX_RQ_BAL_CB_PENDING)) {
/*
* Balance callbacks must run in the context that queued them,
* so they can't be queued on another CPU's rq. Run the deferred
@@ -3384,8 +3364,45 @@ static enum scx_dsp_verdict dispatch_core_pick(struct rq *rq, struct rq_flags *r
}
#endif /* CONFIG_SCHED_CORE */
+/*
+ * Queue the balance callbacks requested by the dispatch path. Balance callbacks
+ * must stay queued only until the end of the current rq lock section, so they
+ * can be used only when the pick is @final. Otherwise, the core is going to
+ * retry or continue the pick, which may release the rq lock (e.g.,
+ * prev_balance() or another dispatch) and expose the queued callbacks to
+ * rq_pin_lock() on other CPUs. Punt the deferred work to irq_work in that case.
+ */
+static void pick_queue_balance_callbacks(struct rq *rq, bool final)
+{
+ lockdep_assert_rq_held(rq);
+
+ if (cpu_of(rq) != smp_processor_id())
+ return;
+
+ if (!final) {
+ if (rq->scx.flags & SCX_RQ_BAL_CB_PENDING) {
+ rq->scx.flags &= ~SCX_RQ_BAL_CB_PENDING;
+ schedule_deferred(rq);
+ }
+ return;
+ }
+
+ maybe_queue_balance_callback(rq);
+
+ /*
+ * Defer to a balance callback which can drop rq lock and enable IRQs.
+ * Waiting directly in the pick path would deadlock against CPUs sending
+ * us IPIs (e.g., TLB flushes) while we wait for them.
+ */
+ if (unlikely(rq->scx.kick_sync_pending)) {
+ rq->scx.kick_sync_pending = false;
+ queue_balance_callback(rq, &rq->scx.kick_sync_bal_cb,
+ kick_sync_wait_bal_cb);
+ }
+}
+
static struct task_struct *
-do_pick_task_scx(struct rq *rq, struct rq_flags *rf, bool force_scx)
+__do_pick_task_scx(struct rq *rq, struct rq_flags *rf, bool force_scx)
{
struct task_struct *prev = rq->curr;
enum scx_dsp_verdict verdict;
@@ -3447,6 +3464,17 @@ do_pick_task_scx(struct rq *rq, struct rq_flags *rf, bool force_scx)
return p;
}
+static struct task_struct *
+do_pick_task_scx(struct rq *rq, struct rq_flags *rf, bool force_scx)
+{
+ struct task_struct *p = __do_pick_task_scx(rq, rf, force_scx);
+
+ /* after a NULL pick from the dl_server the lower classes still pick */
+ pick_queue_balance_callbacks(rq, p != RETRY_TASK && (p || !force_scx));
+
+ return p;
+}
+
static struct task_struct *pick_task_scx(struct rq *rq, struct rq_flags *rf)
{
return do_pick_task_scx(rq, rf, false);
--
2.56.0
^ permalink raw reply [flat|nested] only message in thread
only message in thread, other threads:[~2026-10-10 9:52 UTC | newest]
Thread overview: (only message) (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-10-10 9:52 [PATCH sched_ext/for-7.3-fixes] sched_ext: Don't leave balance callbacks queued across a pick retry Andrea Righi
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®