mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [RFC PATCH 0/2] sched/fair: Remove tunable scaling
@ 2026-10-02 12:45 Tor Vic
  2026-10-02 12:45 ` [RFC PATCH 1/2] sched/fair: Remove the 'none' and 'linear' scaling of tunables Tor Vic
  2026-10-02 12:45 ` [RFC PATCH 2/2] sched/fair: Remove sched_base_slice scaling altogether Tor Vic
  0 siblings, 2 replies; 3+ messages in thread
From: Tor Vic @ 2026-10-02 12:45 UTC (permalink / raw)
  To: peterz, mingo, juri.lelli, vincent.guittot, dietmar.eggemann,
	kprateek.nayak
  Cc: linux-kernel

This series removes the scaling of tunables, which was introduced around
Linux 2.6.33.

I split this into two separate patches:

The first removes the "none" and "linear" scaling which are likely not used much
and have limited usefulness, as one can always set sched_base_slice through
debugfs.
It keeps the logarithmic scaling which is already the default.

The second removes the entire scaling mecanism and sets a fixed sched_base_slice
of 2 milliseconds, a value which has been chosen somewhat arbitrarily.

Currently, the base_slice multiplied by the log2 factor can take the values of
0.7 (1 CPU), 1.4 (2-3 CPUs), 2.1 (4-7 CPUs) and 2.8 msecs (8+ CPUs), so it seems
that the new value of 2 msecs is a sane default, which I also use on my consumer
x86_64 machines.
However I'll leave that open for discussion. Note that the base_slice value has
changed quite often over the years, lastly in 2025 [1].

The series applies to 7.3-rc5, and is build- and boot tested on a measly 2C/4T
Skylake machine. I marked it as RFC because I don't know whether these changes
are actually wanted by the sched people - and because I'm not a developer.
It does however remove quite a lot of code.

[1] commit 2ae891b826958b60919ea21c727f77bcd6ffcc2c

Tor Vic (2):
  sched/fair: Remove the 'none' and 'linear' scaling of tunables
  sched/fair: Remove sched_base_slice scaling altogether

 include/linux/sched/sysctl.h |  7 ----
 kernel/sched/core.c          |  1 -
 kernel/sched/debug.c         | 52 ------------------------
 kernel/sched/fair.c          | 79 ++----------------------------------
 kernel/sched/sched.h         |  5 ---
 5 files changed, 3 insertions(+), 141 deletions(-)

-- 
2.55.0


^ permalink raw reply	[flat|nested] 3+ messages in thread

* [RFC PATCH 1/2] sched/fair: Remove the 'none' and 'linear' scaling of tunables
  2026-10-02 12:45 [RFC PATCH 0/2] sched/fair: Remove tunable scaling Tor Vic
@ 2026-10-02 12:45 ` Tor Vic
  2026-10-02 12:45 ` [RFC PATCH 2/2] sched/fair: Remove sched_base_slice scaling altogether Tor Vic
  1 sibling, 0 replies; 3+ messages in thread
From: Tor Vic @ 2026-10-02 12:45 UTC (permalink / raw)
  To: peterz, mingo, juri.lelli, vincent.guittot, dietmar.eggemann,
	kprateek.nayak
  Cc: linux-kernel, Tor Vic

Get rid of the tunable scalings 'none' and 'linear' as these do not
appear to be very useful.

Only keep the 'log' scaling, which is already the default.

Signed-off-by: Tor Vic <torvic9@mailbox.org>
---
 include/linux/sched/sysctl.h |  7 -----
 kernel/sched/debug.c         | 52 ------------------------------------
 kernel/sched/fair.c          | 28 +------------------
 kernel/sched/sched.h         |  2 --
 4 files changed, 1 insertion(+), 88 deletions(-)

diff --git a/include/linux/sched/sysctl.h b/include/linux/sched/sysctl.h
index 5a64582b086b..0b923cc771e0 100644
--- a/include/linux/sched/sysctl.h
+++ b/include/linux/sched/sysctl.h
@@ -12,13 +12,6 @@ extern unsigned long sysctl_hung_task_timeout_secs;
 enum { sysctl_hung_task_timeout_secs = 0 };
 #endif
 
-enum sched_tunable_scaling {
-	SCHED_TUNABLESCALING_NONE,
-	SCHED_TUNABLESCALING_LOG,
-	SCHED_TUNABLESCALING_LINEAR,
-	SCHED_TUNABLESCALING_END,
-};
-
 #define NUMA_BALANCING_DISABLED		0x0
 #define NUMA_BALANCING_NORMAL		0x1
 #define NUMA_BALANCING_MEMORY_TIERING	0x2
diff --git a/kernel/sched/debug.c b/kernel/sched/debug.c
index 72236db67983..b159d7a1cfa2 100644
--- a/kernel/sched/debug.c
+++ b/kernel/sched/debug.c
@@ -170,46 +170,6 @@ static const struct file_operations sched_feat_fops = {
 	.release	= single_release,
 };
 
-static ssize_t sched_scaling_write(struct file *filp, const char __user *ubuf,
-				   size_t cnt, loff_t *ppos)
-{
-	unsigned int scaling;
-	int ret;
-
-	ret = kstrtouint_from_user(ubuf, cnt, 10, &scaling);
-	if (ret)
-		return ret;
-
-	if (scaling >= SCHED_TUNABLESCALING_END)
-		return -EINVAL;
-
-	sysctl_sched_tunable_scaling = scaling;
-	if (sched_update_scaling())
-		return -EINVAL;
-
-	*ppos += cnt;
-	return cnt;
-}
-
-static int sched_scaling_show(struct seq_file *m, void *v)
-{
-	seq_printf(m, "%d\n", sysctl_sched_tunable_scaling);
-	return 0;
-}
-
-static int sched_scaling_open(struct inode *inode, struct file *filp)
-{
-	return single_open(filp, sched_scaling_show, NULL);
-}
-
-static const struct file_operations sched_scaling_fops = {
-	.open		= sched_scaling_open,
-	.write		= sched_scaling_write,
-	.read		= seq_read,
-	.llseek		= seq_lseek,
-	.release	= single_release,
-};
-
 #ifdef CONFIG_SCHED_CACHE
 static ssize_t
 sched_cache_enable_write(struct file *filp, const char __user *ubuf,
@@ -726,7 +686,6 @@ static __init int sched_init_debug(void)
 	debugfs_create_u32("latency_warn_ms", 0644, debugfs_sched, &sysctl_resched_latency_warn_ms);
 	debugfs_create_u32("latency_warn_once", 0644, debugfs_sched, &sysctl_resched_latency_warn_once);
 
-	debugfs_create_file("tunable_scaling", 0644, debugfs_sched, NULL, &sched_scaling_fops);
 	debugfs_create_u32("migration_cost_ns", 0644, debugfs_sched, &sysctl_sched_migration_cost);
 	debugfs_create_u32("nr_migrate", 0644, debugfs_sched, &sysctl_sched_nr_migrate);
 
@@ -1238,12 +1197,6 @@ do {									\
 	SEQ_printf(m, "\n");
 }
 
-static const char *sched_tunable_scaling_names[] = {
-	"none",
-	"logarithmic",
-	"linear"
-};
-
 static void sched_debug_header(struct seq_file *m)
 {
 	u64 ktime, sched_clk, cpu_clk;
@@ -1286,11 +1239,6 @@ static void sched_debug_header(struct seq_file *m)
 #undef PN
 #undef P
 
-	SEQ_printf(m, "  .%-40s: %d (%s)\n",
-		"sysctl_sched_tunable_scaling",
-		sysctl_sched_tunable_scaling,
-		sched_tunable_scaling_names[sysctl_sched_tunable_scaling]);
-	SEQ_printf(m, "\n");
 }
 
 static int sched_debug_show(struct seq_file *m, void *v)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 57360f5cdde4..2f9b7d2fe813 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -61,19 +61,6 @@
 #include "stats.h"
 #include "autogroup.h"
 
-/*
- * The initial- and re-scaling of tunables is configurable
- *
- * Options are:
- *
- *   SCHED_TUNABLESCALING_NONE - unscaled, always *1
- *   SCHED_TUNABLESCALING_LOG - scaled logarithmically, *1+ilog(ncpus)
- *   SCHED_TUNABLESCALING_LINEAR - scaled linear, *ncpus
- *
- * (default SCHED_TUNABLESCALING_LOG = *(1+ilog(ncpus))
- */
-unsigned int sysctl_sched_tunable_scaling = SCHED_TUNABLESCALING_LOG;
-
 /*
  * Default base time slice (request size r_i) for SCHED_NORMAL/SCHED_BATCH:
  *
@@ -198,20 +185,7 @@ static inline void update_load_set(struct load_weight *lw, unsigned long w)
 static unsigned int get_update_sysctl_factor(void)
 {
 	unsigned int cpus = min_t(unsigned int, num_online_cpus(), 8);
-	unsigned int factor;
-
-	switch (sysctl_sched_tunable_scaling) {
-	case SCHED_TUNABLESCALING_NONE:
-		factor = 1;
-		break;
-	case SCHED_TUNABLESCALING_LINEAR:
-		factor = cpus;
-		break;
-	case SCHED_TUNABLESCALING_LOG:
-	default:
-		factor = 1 + ilog2(cpus);
-		break;
-	}
+	unsigned int factor = 1 + ilog2(cpus);
 
 	return factor;
 }
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index e656c7059bf8..7a36b3210757 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -3153,8 +3153,6 @@ extern unsigned int sysctl_sched_base_slice;
 extern int sysctl_resched_latency_warn_ms;
 extern int sysctl_resched_latency_warn_once;
 
-extern unsigned int sysctl_sched_tunable_scaling;
-
 extern unsigned int sysctl_numa_balancing_scan_delay;
 extern unsigned int sysctl_numa_balancing_scan_period_min;
 extern unsigned int sysctl_numa_balancing_scan_period_max;
-- 
2.55.0


^ permalink raw reply	[flat|nested] 3+ messages in thread

* [RFC PATCH 2/2] sched/fair: Remove sched_base_slice scaling altogether
  2026-10-02 12:45 [RFC PATCH 0/2] sched/fair: Remove tunable scaling Tor Vic
  2026-10-02 12:45 ` [RFC PATCH 1/2] sched/fair: Remove the 'none' and 'linear' scaling of tunables Tor Vic
@ 2026-10-02 12:45 ` Tor Vic
  1 sibling, 0 replies; 3+ messages in thread
From: Tor Vic @ 2026-10-02 12:45 UTC (permalink / raw)
  To: peterz, mingo, juri.lelli, vincent.guittot, dietmar.eggemann,
	kprateek.nayak
  Cc: linux-kernel, Tor Vic

After having removed the 'none' and 'linear' scalings, remove the
entire tunable scaling mechanism altogether, and just set a fixed
default sched_base_slice of 2 msec.

This can still be changed through debugfs.

Signed-off-by: Tor Vic <torvic9@mailbox.org>
---
 kernel/sched/core.c  |  1 -
 kernel/sched/fair.c  | 53 +++-----------------------------------------
 kernel/sched/sched.h |  3 ---
 3 files changed, 3 insertions(+), 54 deletions(-)

diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 1fe40de6ebe3..2a8a242e20e5 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -8942,7 +8942,6 @@ void __init sched_init_smp(void)
 	if (set_cpus_allowed_ptr(current, housekeeping_cpumask(HK_TYPE_DOMAIN)) < 0)
 		BUG();
 	current->flags &= ~PF_NO_SETAFFINITY;
-	sched_init_granularity();
 
 	init_sched_rt_class();
 	init_sched_dl_class();
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 2f9b7d2fe813..3d59af44abdf 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -67,10 +67,10 @@
  * Under EEVDF this is the request size used to compute the virtual
  * deadline; see update_deadline().
  *
- * (default: 0.70 msec * (1 + ilog(ncpus)), units: nanoseconds)
+ * (default: 2 msec)
  */
-unsigned int sysctl_sched_base_slice			= 700000ULL;
-static unsigned int normalized_sysctl_sched_base_slice	= 700000ULL;
+unsigned int sysctl_sched_base_slice			= 2000000ULL;
+static unsigned int normalized_sysctl_sched_base_slice	= 2000000ULL;
 
 __read_mostly unsigned int sysctl_sched_migration_cost	= 500000UL;
 
@@ -173,38 +173,6 @@ static inline void update_load_set(struct load_weight *lw, unsigned long w)
 	lw->inv_weight = 0;
 }
 
-/*
- * Increase the granularity value when there are more CPUs,
- * because with more CPUs the 'effective latency' as visible
- * to users decreases. But the relationship is not linear,
- * so pick a second-best guess by going with the log2 of the
- * number of CPUs.
- *
- * This idea comes from the SD scheduler of Con Kolivas:
- */
-static unsigned int get_update_sysctl_factor(void)
-{
-	unsigned int cpus = min_t(unsigned int, num_online_cpus(), 8);
-	unsigned int factor = 1 + ilog2(cpus);
-
-	return factor;
-}
-
-static void update_sysctl(void)
-{
-	unsigned int factor = get_update_sysctl_factor();
-
-#define SET_SYSCTL(name) \
-	(sysctl_##name = (factor) * normalized_sysctl_##name)
-	SET_SYSCTL(sched_base_slice);
-#undef SET_SYSCTL
-}
-
-void __init sched_init_granularity(void)
-{
-	update_sysctl();
-}
-
 #ifndef CONFIG_64BIT
 #define WMULT_CONST	(~0U)
 #define WMULT_SHIFT	32
@@ -1242,17 +1210,6 @@ struct sched_entity *__pick_last_entity(struct cfs_rq *cfs_rq)
 /**************************************************************
  * Scheduling class statistics methods:
  */
-int sched_update_scaling(void)
-{
-	unsigned int factor = get_update_sysctl_factor();
-
-#define WRT_SYSCTL(name) \
-	(normalized_sysctl_##name = sysctl_##name / (factor))
-	WRT_SYSCTL(sched_base_slice);
-#undef WRT_SYSCTL
-
-	return 0;
-}
 
 static void clear_buddies(struct cfs_rq *cfs_rq, struct sched_entity *se);
 
@@ -15011,15 +14968,11 @@ void sched_balance_trigger(struct rq *rq)
 
 static void rq_online_fair(struct rq *rq)
 {
-	update_sysctl();
-
 	update_runtime_enabled(rq);
 }
 
 static void rq_offline_fair(struct rq *rq)
 {
-	update_sysctl();
-
 	/* Ensure any throttled groups are reachable by pick_next_task */
 	unthrottle_offline_cfs_rqs(rq);
 
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 7a36b3210757..a323fdeb7125 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -2302,8 +2302,6 @@ extern int group_balance_cpu(struct sched_group *sg);
 extern void update_sched_domain_debugfs(void);
 extern void dirty_sched_domain_sysctl(int cpu);
 
-extern int sched_update_scaling(void);
-
 static inline const struct cpumask *task_user_cpus(struct task_struct *p)
 {
 	if (!p->user_cpus_ptr)
@@ -2987,7 +2985,6 @@ extern void schedule_idle(void);
 asmlinkage void schedule_user(void);
 
 extern void sysrq_sched_debug_show(void);
-extern void sched_init_granularity(void);
 extern void update_max_interval(void);
 
 extern void init_sched_dl_class(void);
-- 
2.55.0


^ permalink raw reply	[flat|nested] 3+ messages in thread

end of thread, other threads:[~2026-10-02 12:45 UTC | newest]

Thread overview: 3+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-10-02 12:45 [RFC PATCH 0/2] sched/fair: Remove tunable scaling Tor Vic
2026-10-02 12:45 ` [RFC PATCH 1/2] sched/fair: Remove the 'none' and 'linear' scaling of tunables Tor Vic
2026-10-02 12:45 ` [RFC PATCH 2/2] sched/fair: Remove sched_base_slice scaling altogether Tor Vic

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®