mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [PATCH v3] sched: Lift cgroup update locking to core to prevent CFS/SCX divergence
@ 2026-08-24  7:49 Michal Blaszczyk
  2026-08-24 11:24 ` Andrea Righi
  0 siblings, 1 reply; 2+ messages in thread
From: Michal Blaszczyk @ 2026-08-24  7:49 UTC (permalink / raw)
  To: Peter Zijlstra, Tejun Heo, David Vernet, Andrea Righi, Changwoo Min
  Cc: Michal Blaszczyk, Kuba Piecuch, sched-ext, linux-kernel

Concurrent writes to cgroup control files (such as cpu.shares or
cpu.weight) can lead to state divergence between CFS and SCX.

For instance, in cpu_shares_write_u64(), the CFS update is serialized
by shares_mutex (internal to fair.c), but this lock is dropped before
scx_group_set_weight() is called. The latter only acquires a read
semaphore (scx_cgroup_ops_rwsem), allowing multiple threads to evaluate
and act on the sched_ext update concurrently.

This serialization gap allows concurrent writes to interleave.
As a result, the recorded state in CFS, the SCX internal bookkeeping
(e.g., tg->scx.weight), and the BPF scheduler itself can end up operating
on completely distinct parameters (pairwise distinct values).

Similar races are present in tg_set_bandwidth(), cpu_idle_write_s64(),
cpu_weight_write_u64(), and cpu_weight_nice_write_s64().

Fix this by moving the CFS locking up into the core layer in
`kernel/sched/core.c`. By acquiring these locks directly in the core
write handlers, both the CFS and SCX callbacks are executed atomically
under the same lock.

Fixes: 819513666966 ("sched_ext: Add cgroup support")
Signed-off-by: Michal Blaszczyk <michalblk@google.com>
---
v3:
- Renamed the shares and cfs_constraints mutexes.

 kernel/sched/core.c  | 35 +++++++++++++++++++++++------------
 kernel/sched/fair.c  | 25 +++++++++++++------------
 kernel/sched/sched.h |  7 +++++++
 3 files changed, 43 insertions(+), 24 deletions(-)

diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index f5f7ff8c680a..4673a78cb9e8 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -9779,6 +9779,8 @@ static int cpu_uclamp_max_show(struct seq_file *sf, void *v)
 }
 #endif /* CONFIG_UCLAMP_TASK_GROUP */
 
+DEFINE_MUTEX(cpu_weight_mutex);
+
 #ifdef CONFIG_GROUP_SCHED_WEIGHT
 static unsigned long tg_weight(struct task_group *tg)
 {
@@ -9796,7 +9798,10 @@ static int cpu_shares_write_u64(struct cgroup_subsys_state *css,
 
 	if (shareval > scale_load_down(ULONG_MAX))
 		shareval = MAX_SHARES;
-	ret = sched_group_set_shares(css_tg(css), scale_load(shareval));
+
+	guard(mutex)(&cpu_weight_mutex);
+
+	ret = sched_group_set_shares_locked(css_tg(css), scale_load(shareval));
 	if (!ret)
 		scx_group_set_weight(css_tg(css),
 				     sched_weight_to_cgroup(shareval));
@@ -9811,8 +9816,6 @@ static u64 cpu_shares_read_u64(struct cgroup_subsys_state *css,
 #endif /* CONFIG_GROUP_SCHED_WEIGHT */
 
 #ifdef CONFIG_CFS_BANDWIDTH
-static DEFINE_MUTEX(cfs_constraints_mutex);
-
 static int __cfs_schedulable(struct task_group *tg, u64 period, u64 runtime);
 
 static int tg_set_cfs_bandwidth(struct task_group *tg,
@@ -9831,13 +9834,6 @@ static int tg_set_cfs_bandwidth(struct task_group *tg,
 
 	burst = (u64)burst_us * NSEC_PER_USEC;
 
-	/*
-	 * Prevent race between setting of cfs_rq->runtime_enabled and
-	 * unthrottle_offline_cfs_rqs().
-	 */
-	guard(cpus_read_lock)();
-	guard(mutex)(&cfs_constraints_mutex);
-
 	ret = __cfs_schedulable(tg, period, quota);
 	if (ret)
 		return ret;
@@ -10089,6 +10085,8 @@ static u64 cpu_period_read_u64(struct cgroup_subsys_state *css,
 	return period_us;
 }
 
+static DEFINE_MUTEX(cpu_max_mutex);
+
 static int tg_set_bandwidth(struct task_group *tg,
 			    u64 period_us, u64 quota_us, u64 burst_us)
 {
@@ -10131,6 +10129,13 @@ static int tg_set_bandwidth(struct task_group *tg,
 					burst_us + quota_us > max_bw_runtime_us))
 		return -EINVAL;
 
+	/*
+	 * Prevent race between setting of cfs_rq->runtime_enabled and
+	 * unthrottle_offline_cfs_rqs().
+	 */
+	guard(cpus_read_lock)();
+	guard(mutex)(&cpu_max_mutex);
+
 #ifdef CONFIG_CFS_BANDWIDTH
 	ret = tg_set_cfs_bandwidth(tg, period_us, quota_us, burst_us);
 #endif /* CONFIG_CFS_BANDWIDTH */
@@ -10229,6 +10234,8 @@ static int cpu_idle_write_s64(struct cgroup_subsys_state *css,
 {
 	int ret;
 
+	guard(mutex)(&cpu_weight_mutex);
+
 	ret = sched_group_set_idle(css_tg(css), idle);
 	if (!ret)
 		scx_group_set_idle(css_tg(css), idle);
@@ -10405,7 +10412,9 @@ static int cpu_weight_write_u64(struct cgroup_subsys_state *css,
 
 	weight = sched_weight_from_cgroup(cgrp_weight);
 
-	ret = sched_group_set_shares(css_tg(css), scale_load(weight));
+	guard(mutex)(&cpu_weight_mutex);
+
+	ret = sched_group_set_shares_locked(css_tg(css), scale_load(weight));
 	if (!ret)
 		scx_group_set_weight(css_tg(css), cgrp_weight);
 	return ret;
@@ -10442,7 +10451,9 @@ static int cpu_weight_nice_write_s64(struct cgroup_subsys_state *css,
 	idx = array_index_nospec(idx, 40);
 	weight = sched_prio_to_weight[idx];
 
-	ret = sched_group_set_shares(css_tg(css), scale_load(weight));
+	guard(mutex)(&cpu_weight_mutex);
+
+	ret = sched_group_set_shares_locked(css_tg(css), scale_load(weight));
 	if (!ret)
 		scx_group_set_weight(css_tg(css),
 				     sched_weight_to_cgroup(weight));
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 001140132a7d..4e0a38b0cb3c 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -15392,13 +15392,11 @@ void init_tg_cfs_entry(struct task_group *tg, struct cfs_rq *cfs_rq,
 	se->parent = parent;
 }
 
-static DEFINE_MUTEX(shares_mutex);
-
 static int __sched_group_set_shares(struct task_group *tg, unsigned long shares)
 {
 	int i;
 
-	lockdep_assert_held(&shares_mutex);
+	lockdep_assert_held(&cpu_weight_mutex);
 
 	/*
 	 * We can't change the weight of the root cgroup.
@@ -15430,36 +15428,40 @@ static int __sched_group_set_shares(struct task_group *tg, unsigned long shares)
 	return 0;
 }
 
-int sched_group_set_shares(struct task_group *tg, unsigned long shares)
+int sched_group_set_shares_locked(struct task_group *tg, unsigned long shares)
 {
 	int ret;
 
-	mutex_lock(&shares_mutex);
+	lockdep_assert_held(&cpu_weight_mutex);
+
 	if (tg_is_idle(tg))
 		ret = -EINVAL;
 	else
 		ret = __sched_group_set_shares(tg, shares);
-	mutex_unlock(&shares_mutex);
 
 	return ret;
 }
 
+int sched_group_set_shares(struct task_group *tg, unsigned long shares)
+{
+	guard(mutex)(&cpu_weight_mutex);
+	return sched_group_set_shares_locked(tg, shares);
+}
+
 int sched_group_set_idle(struct task_group *tg, long idle)
 {
 	int i;
 
+	lockdep_assert_held(&cpu_weight_mutex);
+
 	if (tg == &root_task_group)
 		return -EINVAL;
 
 	if (idle < 0 || idle > 1)
 		return -EINVAL;
 
-	mutex_lock(&shares_mutex);
-
-	if (tg->idle == idle) {
-		mutex_unlock(&shares_mutex);
+	if (tg->idle == idle)
 		return 0;
-	}
 
 	tg->idle = idle;
 
@@ -15505,7 +15507,6 @@ int sched_group_set_idle(struct task_group *tg, long idle)
 	else
 		__sched_group_set_shares(tg, NICE_0_LOAD);
 
-	mutex_unlock(&shares_mutex);
 	return 0;
 }
 
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 26ae13c86b69..a989b54f7017 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -599,7 +599,10 @@ extern void sched_release_group(struct task_group *tg);
 extern void sched_move_task(struct task_struct *tsk, bool for_autogroup);
 
 #ifdef CONFIG_FAIR_GROUP_SCHED
+extern struct mutex cpu_weight_mutex;
+
 extern int sched_group_set_shares(struct task_group *tg, unsigned long shares);
+extern int sched_group_set_shares_locked(struct task_group *tg, unsigned long shares);
 
 extern int sched_group_set_idle(struct task_group *tg, long idle);
 
@@ -607,6 +610,10 @@ extern void set_task_rq_fair(struct sched_entity *se,
 			     struct cfs_rq *prev, struct cfs_rq *next);
 #else /* !CONFIG_FAIR_GROUP_SCHED: */
 static inline int sched_group_set_shares(struct task_group *tg, unsigned long shares) { return 0; }
+static inline int sched_group_set_shares_locked(struct task_group *tg, unsigned long shares)
+{
+	return 0;
+}
 static inline int sched_group_set_idle(struct task_group *tg, long idle) { return 0; }
 #endif /* !CONFIG_FAIR_GROUP_SCHED */
 
-- 
2.55.0.860.g4b6b3295ed-goog


^ permalink raw reply	[flat|nested] 2+ messages in thread

* Re: [PATCH v3] sched: Lift cgroup update locking to core to prevent CFS/SCX divergence
  2026-08-24  7:49 [PATCH v3] sched: Lift cgroup update locking to core to prevent CFS/SCX divergence Michal Blaszczyk
@ 2026-08-24 11:24 ` Andrea Righi
  0 siblings, 0 replies; 2+ messages in thread
From: Andrea Righi @ 2026-08-24 11:24 UTC (permalink / raw)
  To: Michal Blaszczyk
  Cc: Peter Zijlstra, Tejun Heo, David Vernet, Changwoo Min,
	Kuba Piecuch, sched-ext, linux-kernel

Hi Michal,

On Mon, Aug 24, 2026 at 07:49:13AM +0000, Michal Blaszczyk wrote:
> Concurrent writes to cgroup control files (such as cpu.shares or
> cpu.weight) can lead to state divergence between CFS and SCX.
> 
> For instance, in cpu_shares_write_u64(), the CFS update is serialized
> by shares_mutex (internal to fair.c), but this lock is dropped before
> scx_group_set_weight() is called. The latter only acquires a read
> semaphore (scx_cgroup_ops_rwsem), allowing multiple threads to evaluate
> and act on the sched_ext update concurrently.
> 
> This serialization gap allows concurrent writes to interleave.
> As a result, the recorded state in CFS, the SCX internal bookkeeping
> (e.g., tg->scx.weight), and the BPF scheduler itself can end up operating
> on completely distinct parameters (pairwise distinct values).
> 
> Similar races are present in tg_set_bandwidth(), cpu_idle_write_s64(),
> cpu_weight_write_u64(), and cpu_weight_nice_write_s64().
> 
> Fix this by moving the CFS locking up into the core layer in
> `kernel/sched/core.c`. By acquiring these locks directly in the core
> write handlers, both the CFS and SCX callbacks are executed atomically
> under the same lock.
> 
> Fixes: 819513666966 ("sched_ext: Add cgroup support")
> Signed-off-by: Michal Blaszczyk <michalblk@google.com>

IIUC the bandwidth read-modify-write paths still snapshot the unchanged
parameters before taking cpu_max_mutex, so concurrent writes to different
bandwidth knobs can overwrite each other.

However, this shouldn't be a blocker for this patch. The serialization aspect
could be addressed in a separate patch, maybe extending the lock over both the
snapshot and update?

For now:

Reviewed-by: Andrea Righi <arighi@nvidia.com>

Thanks,
-Andrea

> ---
> v3:
> - Renamed the shares and cfs_constraints mutexes.
> 
>  kernel/sched/core.c  | 35 +++++++++++++++++++++++------------
>  kernel/sched/fair.c  | 25 +++++++++++++------------
>  kernel/sched/sched.h |  7 +++++++
>  3 files changed, 43 insertions(+), 24 deletions(-)
> 
> diff --git a/kernel/sched/core.c b/kernel/sched/core.c
> index f5f7ff8c680a..4673a78cb9e8 100644
> --- a/kernel/sched/core.c
> +++ b/kernel/sched/core.c
> @@ -9779,6 +9779,8 @@ static int cpu_uclamp_max_show(struct seq_file *sf, void *v)
>  }
>  #endif /* CONFIG_UCLAMP_TASK_GROUP */
>  
> +DEFINE_MUTEX(cpu_weight_mutex);
> +
>  #ifdef CONFIG_GROUP_SCHED_WEIGHT
>  static unsigned long tg_weight(struct task_group *tg)
>  {
> @@ -9796,7 +9798,10 @@ static int cpu_shares_write_u64(struct cgroup_subsys_state *css,
>  
>  	if (shareval > scale_load_down(ULONG_MAX))
>  		shareval = MAX_SHARES;
> -	ret = sched_group_set_shares(css_tg(css), scale_load(shareval));
> +
> +	guard(mutex)(&cpu_weight_mutex);
> +
> +	ret = sched_group_set_shares_locked(css_tg(css), scale_load(shareval));
>  	if (!ret)
>  		scx_group_set_weight(css_tg(css),
>  				     sched_weight_to_cgroup(shareval));
> @@ -9811,8 +9816,6 @@ static u64 cpu_shares_read_u64(struct cgroup_subsys_state *css,
>  #endif /* CONFIG_GROUP_SCHED_WEIGHT */
>  
>  #ifdef CONFIG_CFS_BANDWIDTH
> -static DEFINE_MUTEX(cfs_constraints_mutex);
> -
>  static int __cfs_schedulable(struct task_group *tg, u64 period, u64 runtime);
>  
>  static int tg_set_cfs_bandwidth(struct task_group *tg,
> @@ -9831,13 +9834,6 @@ static int tg_set_cfs_bandwidth(struct task_group *tg,
>  
>  	burst = (u64)burst_us * NSEC_PER_USEC;
>  
> -	/*
> -	 * Prevent race between setting of cfs_rq->runtime_enabled and
> -	 * unthrottle_offline_cfs_rqs().
> -	 */
> -	guard(cpus_read_lock)();
> -	guard(mutex)(&cfs_constraints_mutex);
> -
>  	ret = __cfs_schedulable(tg, period, quota);
>  	if (ret)
>  		return ret;
> @@ -10089,6 +10085,8 @@ static u64 cpu_period_read_u64(struct cgroup_subsys_state *css,
>  	return period_us;
>  }
>  
> +static DEFINE_MUTEX(cpu_max_mutex);
> +
>  static int tg_set_bandwidth(struct task_group *tg,
>  			    u64 period_us, u64 quota_us, u64 burst_us)
>  {
> @@ -10131,6 +10129,13 @@ static int tg_set_bandwidth(struct task_group *tg,
>  					burst_us + quota_us > max_bw_runtime_us))
>  		return -EINVAL;
>  
> +	/*
> +	 * Prevent race between setting of cfs_rq->runtime_enabled and
> +	 * unthrottle_offline_cfs_rqs().
> +	 */
> +	guard(cpus_read_lock)();
> +	guard(mutex)(&cpu_max_mutex);
> +
>  #ifdef CONFIG_CFS_BANDWIDTH
>  	ret = tg_set_cfs_bandwidth(tg, period_us, quota_us, burst_us);
>  #endif /* CONFIG_CFS_BANDWIDTH */
> @@ -10229,6 +10234,8 @@ static int cpu_idle_write_s64(struct cgroup_subsys_state *css,
>  {
>  	int ret;
>  
> +	guard(mutex)(&cpu_weight_mutex);
> +
>  	ret = sched_group_set_idle(css_tg(css), idle);
>  	if (!ret)
>  		scx_group_set_idle(css_tg(css), idle);
> @@ -10405,7 +10412,9 @@ static int cpu_weight_write_u64(struct cgroup_subsys_state *css,
>  
>  	weight = sched_weight_from_cgroup(cgrp_weight);
>  
> -	ret = sched_group_set_shares(css_tg(css), scale_load(weight));
> +	guard(mutex)(&cpu_weight_mutex);
> +
> +	ret = sched_group_set_shares_locked(css_tg(css), scale_load(weight));
>  	if (!ret)
>  		scx_group_set_weight(css_tg(css), cgrp_weight);
>  	return ret;
> @@ -10442,7 +10451,9 @@ static int cpu_weight_nice_write_s64(struct cgroup_subsys_state *css,
>  	idx = array_index_nospec(idx, 40);
>  	weight = sched_prio_to_weight[idx];
>  
> -	ret = sched_group_set_shares(css_tg(css), scale_load(weight));
> +	guard(mutex)(&cpu_weight_mutex);
> +
> +	ret = sched_group_set_shares_locked(css_tg(css), scale_load(weight));
>  	if (!ret)
>  		scx_group_set_weight(css_tg(css),
>  				     sched_weight_to_cgroup(weight));
> diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> index 001140132a7d..4e0a38b0cb3c 100644
> --- a/kernel/sched/fair.c
> +++ b/kernel/sched/fair.c
> @@ -15392,13 +15392,11 @@ void init_tg_cfs_entry(struct task_group *tg, struct cfs_rq *cfs_rq,
>  	se->parent = parent;
>  }
>  
> -static DEFINE_MUTEX(shares_mutex);
> -
>  static int __sched_group_set_shares(struct task_group *tg, unsigned long shares)
>  {
>  	int i;
>  
> -	lockdep_assert_held(&shares_mutex);
> +	lockdep_assert_held(&cpu_weight_mutex);
>  
>  	/*
>  	 * We can't change the weight of the root cgroup.
> @@ -15430,36 +15428,40 @@ static int __sched_group_set_shares(struct task_group *tg, unsigned long shares)
>  	return 0;
>  }
>  
> -int sched_group_set_shares(struct task_group *tg, unsigned long shares)
> +int sched_group_set_shares_locked(struct task_group *tg, unsigned long shares)
>  {
>  	int ret;
>  
> -	mutex_lock(&shares_mutex);
> +	lockdep_assert_held(&cpu_weight_mutex);
> +
>  	if (tg_is_idle(tg))
>  		ret = -EINVAL;
>  	else
>  		ret = __sched_group_set_shares(tg, shares);
> -	mutex_unlock(&shares_mutex);
>  
>  	return ret;
>  }
>  
> +int sched_group_set_shares(struct task_group *tg, unsigned long shares)
> +{
> +	guard(mutex)(&cpu_weight_mutex);
> +	return sched_group_set_shares_locked(tg, shares);
> +}
> +
>  int sched_group_set_idle(struct task_group *tg, long idle)
>  {
>  	int i;
>  
> +	lockdep_assert_held(&cpu_weight_mutex);
> +
>  	if (tg == &root_task_group)
>  		return -EINVAL;
>  
>  	if (idle < 0 || idle > 1)
>  		return -EINVAL;
>  
> -	mutex_lock(&shares_mutex);
> -
> -	if (tg->idle == idle) {
> -		mutex_unlock(&shares_mutex);
> +	if (tg->idle == idle)
>  		return 0;
> -	}
>  
>  	tg->idle = idle;
>  
> @@ -15505,7 +15507,6 @@ int sched_group_set_idle(struct task_group *tg, long idle)
>  	else
>  		__sched_group_set_shares(tg, NICE_0_LOAD);
>  
> -	mutex_unlock(&shares_mutex);
>  	return 0;
>  }
>  
> diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
> index 26ae13c86b69..a989b54f7017 100644
> --- a/kernel/sched/sched.h
> +++ b/kernel/sched/sched.h
> @@ -599,7 +599,10 @@ extern void sched_release_group(struct task_group *tg);
>  extern void sched_move_task(struct task_struct *tsk, bool for_autogroup);
>  
>  #ifdef CONFIG_FAIR_GROUP_SCHED
> +extern struct mutex cpu_weight_mutex;
> +
>  extern int sched_group_set_shares(struct task_group *tg, unsigned long shares);
> +extern int sched_group_set_shares_locked(struct task_group *tg, unsigned long shares);
>  
>  extern int sched_group_set_idle(struct task_group *tg, long idle);
>  
> @@ -607,6 +610,10 @@ extern void set_task_rq_fair(struct sched_entity *se,
>  			     struct cfs_rq *prev, struct cfs_rq *next);
>  #else /* !CONFIG_FAIR_GROUP_SCHED: */
>  static inline int sched_group_set_shares(struct task_group *tg, unsigned long shares) { return 0; }
> +static inline int sched_group_set_shares_locked(struct task_group *tg, unsigned long shares)
> +{
> +	return 0;
> +}
>  static inline int sched_group_set_idle(struct task_group *tg, long idle) { return 0; }
>  #endif /* !CONFIG_FAIR_GROUP_SCHED */
>  
> -- 
> 2.55.0.860.g4b6b3295ed-goog
> 

^ permalink raw reply	[flat|nested] 2+ messages in thread

end of thread, other threads:[~2026-08-24 11:24 UTC | newest]

Thread overview: 2+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-08-24  7:49 [PATCH v3] sched: Lift cgroup update locking to core to prevent CFS/SCX divergence Michal Blaszczyk
2026-08-24 11:24 ` Andrea Righi

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®