mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [PATCH] sched_ext: Hold DSQ refs for deferred reenqueues
@ 2026-09-30 14:34 Hui Su
  2026-09-30 15:20 ` Hui Su
  2026-09-30 15:26 ` Andrea Righi
  0 siblings, 2 replies; 5+ messages in thread
From: Hui Su @ 2026-09-30 14:34 UTC (permalink / raw)
  To: sched-ext
  Cc: tj, void, arighi, changwoo, mingo, peterz, juri.lelli,
	vincent.guittot, dietmar.eggemann, rostedt, bsegall, mgorman,
	vschneid, kprateek.nayak, linux-kernel, Hui Su, stable

A deferred user-DSQ node can be detached by
process_deferred_reenq_users() before the DSQ RCU callback reaches
exit_dsq(). Once detached, exit_dsq() can no longer find the node,
while the deferred path still uses the raw DSQ pointer after dropping
deferred_reenq_lock. The callback can therefore free the DSQ before
the deferred path checks its ID or calls reenq_user().

An RCU grace period only delays reclamation past pre-existing RCU
read-side critical sections. It doesn't protect a deferred reenqueue
which has detached its node and keeps using the raw DSQ pointer
afterwards.

Take a reference under deferred_reenq_lock before detaching the node.
The RCU callback drops the base reference after exit_dsq(), and the
deferred path drops its reference after its final DSQ access. This
keeps the object alive until all detached reenqueues finish while
preserving invalidated-DSQ behavior.

A KASAN regression test of the pre-fix kernel reported the
use-after-free while processing the deferred reenqueue:

    BUG: KASAN: slab-use-after-free in run_deferred+0x1312/0x1710
    Read of size 8 at addr ffff8880087009b0 by task swapper/3/0
    Call Trace:
     <IRQ>
     run_deferred+0x1312/0x1710
     ttwu_do_activate+0x29a/0x600
     try_to_wake_up+0x815/0x1700

The patched kernel completed the same regression test without a KASAN
report.

Fixes: 84b1a0ea0b7c ("sched_ext: Implement scx_bpf_dsq_reenq() for user DSQs")
Cc: stable@vger.kernel.org # v7.1+
Signed-off-by: Hui Su <sh_def@163.com>

diff --git a/include/linux/sched/ext.h b/include/linux/sched/ext.h
index 23f9e178bc5a..3344cf33d324 100644
--- a/include/linux/sched/ext.h
+++ b/include/linux/sched/ext.h
@@ -13,6 +13,7 @@
 
 #include <linux/llist.h>
 #include <linux/rhashtable-types.h>
+#include <linux/refcount.h>
 
 enum scx_public_consts {
 	SCX_OPS_NAME_LEN	= 128,
@@ -92,6 +93,8 @@ struct scx_dispatch_q {
 	struct llist_node	free_node;
 	struct scx_sched	*sched;
 	struct scx_dsq_pcpu __percpu *pcpu_user;
+	/* one base ref held until deferred reclamation, plus detached consumers */
+	refcount_t		deferred_reenq_refs;
 	struct rcu_head		rcu;
 };
 
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 405d0d1038f8..1df0ff7e3b72 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -5057,6 +5057,7 @@ static void process_deferred_reenq_users(struct rq *rq)
 			dsq_pcpu = container_of(dru, struct scx_dsq_pcpu,
 						deferred_reenq_user);
 			dsq = dsq_pcpu->dsq;
+			refcount_inc(&dsq->deferred_reenq_refs);
 			reenq_flags = dru->flags;
 			WRITE_ONCE(dru->flags, 0);
 			list_del_init(&dru->node);
@@ -5068,10 +5069,13 @@ static void process_deferred_reenq_users(struct rq *rq)
 		/* destroy_dsq() may have raced and invalidated @dsq, nothing to reenq */
 		dsq_id = READ_ONCE(dsq->id);
 		if (unlikely(dsq_id == SCX_DSQ_INVALID))
-			continue;
+			goto put_dsq;
 
 		BUG_ON(dsq_id & SCX_DSQ_FLAG_BUILTIN);
 		reenq_user(rq, dsq, reenq_flags);
+
+put_dsq:
+		refcount_dec(&dsq->deferred_reenq_refs);
 	}
 }
 
@@ -5565,6 +5569,7 @@ s32 scx_init_dsq(struct scx_dispatch_q *dsq, u64 dsq_id, struct scx_sched *sch)
 	if (dsq_id & SCX_DSQ_FLAG_BUILTIN)
 		return 0;
 
+	refcount_set(&dsq->deferred_reenq_refs, 1);
 	dsq->pcpu_user = alloc_percpu(struct scx_dsq_pcpu);
 	if (!dsq->pcpu_user)
 		return -ENOMEM;
@@ -5591,25 +5596,33 @@ static void exit_dsq(struct scx_dispatch_q *dsq)
 		struct scx_deferred_reenq_user *dru = &pcpu->deferred_reenq_user;
 		struct rq *rq = cpu_rq(cpu);
 
-		/*
-		 * There must have been a RCU grace period since the last
-		 * insertion and @dsq should be off the deferred list by now.
-		 */
-		if (WARN_ON_ONCE(!list_empty(&dru->node))) {
-			guard(raw_spinlock_irqsave)(&rq->scx.deferred_reenq_lock);
+		guard(raw_spinlock_irqsave)(&rq->scx.deferred_reenq_lock);
+
+		if (WARN_ON_ONCE(!list_empty(&dru->node)))
 			list_del_init(&dru->node);
-		}
 	}
 
 	free_percpu(dsq->pcpu_user);
 }
 
+static void free_dsq_finish_rcufn(struct rcu_head *rcu)
+{
+	struct scx_dispatch_q *dsq = container_of(rcu, struct scx_dispatch_q, rcu);
+
+	if (!refcount_dec_if_one(&dsq->deferred_reenq_refs)) {
+		call_rcu(&dsq->rcu, free_dsq_finish_rcufn);
+		return;
+	}
+
+	kfree(dsq);
+}
+
 static void free_dsq_rcufn(struct rcu_head *rcu)
 {
 	struct scx_dispatch_q *dsq = container_of(rcu, struct scx_dispatch_q, rcu);
 
 	exit_dsq(dsq);
-	kfree(dsq);
+	free_dsq_finish_rcufn(rcu);
 }
 
 static void free_dsq_irq_workfn(struct irq_work *irq_work)


^ permalink raw reply	[flat|nested] 5+ messages in thread
* [PATCH] sched_ext: Hold DSQ refs for deferred reenqueues
@ 2026-09-30 10:17 Hui Su
  0 siblings, 0 replies; 5+ messages in thread
From: Hui Su @ 2026-09-30 10:17 UTC (permalink / raw)
  To: sched-ext
  Cc: tj, void, arighi, changwoo, mingo, peterz, juri.lelli,
	vincent.guittot, dietmar.eggemann, rostedt, bsegall, mgorman,
	vschneid, kprateek.nayak, linux-kernel, Hui Su, stable

A deferred user-DSQ node can be detached by
process_deferred_reenq_users() before the DSQ RCU callback reaches
exit_dsq(). Once detached, exit_dsq() can no longer find the node,
while the deferred path still uses the raw DSQ pointer after dropping
deferred_reenq_lock. The callback can therefore free the DSQ before
the deferred path checks its ID or calls reenq_user().

An RCU grace period only delays reclamation past pre-existing RCU
read-side critical sections. It doesn't protect a deferred reenqueue
which has detached its node and keeps using the raw DSQ pointer
afterwards.

Take a reference under deferred_reenq_lock before detaching the node.
The RCU callback drops the base reference after exit_dsq(), and the
deferred path drops its reference after its final DSQ access. This
keeps the object alive until all detached reenqueues finish while
preserving invalidated-DSQ behavior.

A KASAN regression test of the pre-fix kernel reported the
use-after-free while processing the deferred reenqueue:

    BUG: KASAN: slab-use-after-free in run_deferred+0x1312/0x1710
    Read of size 8 at addr ffff8880087009b0 by task swapper/3/0
    Call Trace:
     <IRQ>
     run_deferred+0x1312/0x1710
     ttwu_do_activate+0x29a/0x600
     try_to_wake_up+0x815/0x1700

The patched kernel completed the same regression test without a KASAN
report.

Fixes: 84b1a0ea0b7c ("sched_ext: Implement scx_bpf_dsq_reenq() for user DSQs")
Cc: stable@vger.kernel.org # v7.1+
Signed-off-by: Hui Su <sh_def@163.com>

diff --git a/include/linux/sched/ext.h b/include/linux/sched/ext.h
index 23f9e178bc5a..1d36196b2238 100644
--- a/include/linux/sched/ext.h
+++ b/include/linux/sched/ext.h
@@ -13,6 +13,7 @@
 
 #include <linux/llist.h>
 #include <linux/rhashtable-types.h>
+#include <linux/refcount.h>
 
 enum scx_public_consts {
 	SCX_OPS_NAME_LEN	= 128,
@@ -92,6 +93,8 @@ struct scx_dispatch_q {
 	struct llist_node	free_node;
 	struct scx_sched	*sched;
 	struct scx_dsq_pcpu __percpu *pcpu_user;
+	/* one base ref held until deferred reclamation, plus detached workers */
+	refcount_t		deferred_reenq_refs;
 	struct rcu_head		rcu;
 };
 
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 405d0d1038f8..9fd18fa5725b 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -5057,6 +5057,7 @@ static void process_deferred_reenq_users(struct rq *rq)
 			dsq_pcpu = container_of(dru, struct scx_dsq_pcpu,
 						deferred_reenq_user);
 			dsq = dsq_pcpu->dsq;
+			refcount_inc(&dsq->deferred_reenq_refs);
 			reenq_flags = dru->flags;
 			WRITE_ONCE(dru->flags, 0);
 			list_del_init(&dru->node);
@@ -5068,10 +5069,14 @@ static void process_deferred_reenq_users(struct rq *rq)
 		/* destroy_dsq() may have raced and invalidated @dsq, nothing to reenq */
 		dsq_id = READ_ONCE(dsq->id);
 		if (unlikely(dsq_id == SCX_DSQ_INVALID))
-			continue;
+			goto put_dsq;
 
 		BUG_ON(dsq_id & SCX_DSQ_FLAG_BUILTIN);
 		reenq_user(rq, dsq, reenq_flags);
+
+put_dsq:
+		if (refcount_dec_and_test(&dsq->deferred_reenq_refs))
+			kfree(dsq);
 	}
 }
 
@@ -5565,6 +5570,7 @@ s32 scx_init_dsq(struct scx_dispatch_q *dsq, u64 dsq_id, struct scx_sched *sch)
 	if (dsq_id & SCX_DSQ_FLAG_BUILTIN)
 		return 0;
 
+	refcount_set(&dsq->deferred_reenq_refs, 1);
 	dsq->pcpu_user = alloc_percpu(struct scx_dsq_pcpu);
 	if (!dsq->pcpu_user)
 		return -ENOMEM;
@@ -5609,7 +5615,8 @@ static void free_dsq_rcufn(struct rcu_head *rcu)
 	struct scx_dispatch_q *dsq = container_of(rcu, struct scx_dispatch_q, rcu);
 
 	exit_dsq(dsq);
-	kfree(dsq);
+	if (refcount_dec_and_test(&dsq->deferred_reenq_refs))
+		kfree(dsq);
 }
 
 static void free_dsq_irq_workfn(struct irq_work *irq_work)


^ permalink raw reply	[flat|nested] 5+ messages in thread

end of thread, other threads:[~2026-09-30 17:15 UTC | newest]

Thread overview: 5+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-09-30 14:34 [PATCH] sched_ext: Hold DSQ refs for deferred reenqueues Hui Su
2026-09-30 15:20 ` Hui Su
2026-09-30 17:15   ` Tejun Heo
2026-09-30 15:26 ` Andrea Righi
  -- strict thread matches above, loose matches on Subject: below --
2026-09-30 10:17 Hui Su

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®