* [PATCH 1/2] sched/fair: Honor asymmetric SMT priority in idle selection
2026-09-29 16:54 [PATCH v7 0/2] sched: Enable preferred SMT siblings on NVIDIA Olympus Andrea Righi
@ 2026-09-29 16:54 ` Andrea Righi
2026-09-29 16:54 ` [PATCH 2/2] sched/topology: Add asymmetric SMT packing override Andrea Righi
1 sibling, 0 replies; 3+ messages in thread
From: Andrea Righi @ 2026-09-29 16:54 UTC (permalink / raw)
To: Ingo Molnar, Peter Zijlstra, Juri Lelli, Vincent Guittot, Will Deacon
Cc: Dietmar Eggemann, Steven Rostedt, Ben Segall, Mel Gorman,
Valentin Schneider, K Prateek Nayak, Christian Loehle,
Srikar Dronamraju, Shrikanth Hegde, Phil Auld, Breno Leitao,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Lee Trager,
Vikram Sethi, Kayra Cizmeci, linux-doc, linux-kernel
POWER7 uses SD_ASYM_PACKING at the shared-capacity SMT level to order
hardware threads, and NVIDIA Olympus benefits from the same policy. Idle
CPU selection does not consult that order, so a task can wake on an
arbitrary sibling and remain there until load balancing corrects the
placement. On these systems, that initial choice can prevent the core
from entering its preferred lower-thread resource mode and cause a large
and persistent performance loss.
When idle selection finds an available CPU in an SMT core, choose the
highest-priority available sibling. On SMT2 Olympus this only changes
selection on fully idle cores. A partially idle core has only one
available CPU. On wider SMT systems such as POWER7, it also fills
available siblings in priority order while the core is partially busy.
Apply the preference to idle-core and idle-CPU scans,
asymmetric-capacity scans, target, previous, recently-used CPU fast
paths and the slow path. Inspect the lowest scheduling domain directly,
but require both CPUs to share its span because isolcpus can split
hardware siblings across scheduling domains.
Keep physical-core capacity selection independent from SMT sibling
ordering. SD_ASYM_CPUCAPACITY first selects among cores with different
maximum capacities, then SD_ASYM_PACKING selects the preferred available
sibling inside the chosen core, whose siblings continue to share equal
capacity.
Reviewed-by: Srikar Dronamraju <srikar@linux.ibm.com>
Reviewed-by: K Prateek Nayak <kprateek.nayak@amd.com>
Reviewed-by: Vincent Guittot <vincent.guittot@linaro.org>
Reviewed-by: Kayra Cizmeci <kayracizmeci@gmail.com>
Tested-by: K Prateek Nayak <kprateek.nayak@amd.com>
Tested-by: Kayra Cizmeci <kayracizmeci@gmail.com>
Tested-by: Breno Leitao <leitao@debian.org>
Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
kernel/sched/fair.c | 85 ++++++++++++++++++++++++++++++++++++---------
1 file changed, 68 insertions(+), 17 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index e707da7177dfe..5c98d8dfce5d8 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -8799,6 +8799,35 @@ static inline bool test_idle_cores(int cpu)
return false;
}
+/*
+ * Redirect a CPU to a higher-priority available sibling in its SMT domain,
+ * subject to task affinity.
+ */
+static inline int select_idle_smt_cpu(struct task_struct *p, int cpu)
+{
+ struct sched_domain *sd;
+ int best = cpu;
+ int sibling;
+
+ if (!sched_smt_active())
+ return cpu;
+
+ sd = rcu_dereference_all(cpu_rq(cpu)->sd);
+ if (!sd || !(sd->flags & SD_SHARE_CPUCAPACITY) ||
+ !(sd->flags & SD_ASYM_PACKING))
+ return cpu;
+
+ for_each_cpu_and(sibling, sched_domain_span(sd), p->cpus_ptr) {
+ if (sibling == best || !choose_idle_cpu(sibling, p))
+ continue;
+
+ if (sched_asym_prefer(sibling, best))
+ best = sibling;
+ }
+
+ return best;
+}
+
/*
* Scans the local SMT mask to see if the entire core is idle, and records this
* information in sd_balance_shared->has_idle_cores.
@@ -9183,7 +9212,7 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
if (choose_idle_cpu(target, p) &&
asym_fits_cpu(task_util, util_min, util_max, target))
- return target;
+ goto select_smt_priority;
/*
* If the previous CPU is cache affine and idle, don't be stupid:
@@ -9193,8 +9222,10 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
asym_fits_cpu(task_util, util_min, util_max, prev)) {
if (!static_branch_unlikely(&sched_cluster_active) ||
- cpus_share_resources(prev, target))
- return prev;
+ cpus_share_resources(prev, target)) {
+ target = prev;
+ goto select_smt_priority;
+ }
prev_aff = prev;
}
@@ -9212,7 +9243,8 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
prev == smp_processor_id() &&
this_rq()->nr_running <= 1 &&
asym_fits_cpu(task_util, util_min, util_max, prev)) {
- return prev;
+ target = prev;
+ goto select_smt_priority;
}
/* Check a recently used CPU as a potential idle candidate: */
@@ -9226,8 +9258,10 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
asym_fits_cpu(task_util, util_min, util_max, recent_used_cpu)) {
if (!static_branch_unlikely(&sched_cluster_active) ||
- cpus_share_resources(recent_used_cpu, target))
- return recent_used_cpu;
+ cpus_share_resources(recent_used_cpu, target)) {
+ target = recent_used_cpu;
+ goto select_smt_priority;
+ }
} else {
recent_used_cpu = -1;
@@ -9249,7 +9283,11 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
*/
if (sd) {
i = select_idle_capacity(p, sd, target);
- return ((unsigned)i < nr_cpumask_bits) ? i : target;
+ if ((unsigned int)i < nr_cpumask_bits) {
+ target = i;
+ goto select_smt_priority;
+ }
+ return target;
}
}
@@ -9262,14 +9300,18 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
if (!has_idle_core && cpus_share_cache(prev, target)) {
i = select_idle_smt(p, sd, prev);
- if ((unsigned int)i < nr_cpumask_bits)
- return i;
+ if ((unsigned int)i < nr_cpumask_bits) {
+ target = i;
+ goto select_smt_priority;
+ }
}
}
i = select_idle_cpu(p, sd, has_idle_core, target);
- if ((unsigned)i < nr_cpumask_bits)
- return i;
+ if ((unsigned int)i < nr_cpumask_bits) {
+ target = i;
+ goto select_smt_priority;
+ }
/*
* For cluster machines which have lower sharing cache like L2 or
@@ -9277,12 +9319,19 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
* first. But prev_cpu or recent_used_cpu may also be a good candidate,
* use them if possible when no idle CPU found in select_idle_cpu().
*/
- if ((unsigned int)prev_aff < nr_cpumask_bits)
- return prev_aff;
- if ((unsigned int)recent_used_cpu < nr_cpumask_bits)
- return recent_used_cpu;
+ if ((unsigned int)prev_aff < nr_cpumask_bits) {
+ target = prev_aff;
+ goto select_smt_priority;
+ }
+ if ((unsigned int)recent_used_cpu < nr_cpumask_bits) {
+ target = recent_used_cpu;
+ goto select_smt_priority;
+ }
return target;
+
+select_smt_priority:
+ return select_idle_smt_cpu(p, target);
}
/**
@@ -9959,8 +10008,10 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
}
/* Slow path */
- if (unlikely(sd))
- return sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
+ if (unlikely(sd)) {
+ new_cpu = sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
+ return select_idle_smt_cpu(p, new_cpu);
+ }
/* Fast path */
if (wake_flags & WF_TTWU)
--
2.55.0
^ permalink raw reply [flat|nested] 3+ messages in thread* [PATCH 2/2] sched/topology: Add asymmetric SMT packing override
2026-09-29 16:54 [PATCH v7 0/2] sched: Enable preferred SMT siblings on NVIDIA Olympus Andrea Righi
2026-09-29 16:54 ` [PATCH 1/2] sched/fair: Honor asymmetric SMT priority in idle selection Andrea Righi
@ 2026-09-29 16:54 ` Andrea Righi
1 sibling, 0 replies; 3+ messages in thread
From: Andrea Righi @ 2026-09-29 16:54 UTC (permalink / raw)
To: Ingo Molnar, Peter Zijlstra, Juri Lelli, Vincent Guittot, Will Deacon
Cc: Dietmar Eggemann, Steven Rostedt, Ben Segall, Mel Gorman,
Valentin Schneider, K Prateek Nayak, Christian Loehle,
Srikar Dronamraju, Shrikanth Hegde, Phil Auld, Breno Leitao,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Lee Trager,
Vikram Sethi, Kayra Cizmeci, linux-doc, linux-kernel
Architectures can use SD_ASYM_PACKING to describe preferred CPU ordering
at the SMT scheduling domain. Some systems benefit from this policy, but
their firmware cannot describe the preference and inferring it from the
CPU model would embed a platform-specific policy in the kernel.
Add sched_smt_asym_packing=on boot option to force SD_ASYM_PACKING at
the SMT level. Omitting the option preserves the topology provided by
the architecture or firmware.
Apply the override centrally to domains with SD_SHARE_CPUCAPACITY so it
also covers architectures with custom SMT topology callbacks, including
powerpc.
When enabled, priority remains defined by arch_asym_cpu_priority(). The
weak default prefers lower-numbered logical CPUs, while architecture
overrides remain authoritative. Siblings with equal priorities remain
unordered. In particular, x86 with CONFIG_SCHED_MC_PRIO normally gives
both SMT siblings the same core priority, so enabling the option there
does not prioritize a sibling.
Tested-by: Breno Leitao <leitao@debian.org>
Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
.../admin-guide/kernel-parameters.txt | 12 ++++++++++++
kernel/sched/topology.c | 18 ++++++++++++++++++
2 files changed, 30 insertions(+)
diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt
index e75344f4e0cde..8157ff434cb5e 100644
--- a/Documentation/admin-guide/kernel-parameters.txt
+++ b/Documentation/admin-guide/kernel-parameters.txt
@@ -6800,6 +6800,18 @@ Kernel parameters
solution to mutex-based priority inversion.
Format: <bool>
+ sched_smt_asym_packing= [KNL,SMP]
+ Format: on
+ Force asymmetric packing at the SMT scheduling domain.
+ Idle CPU selection prefers siblings with a higher
+ arch_asym_cpu_priority(). The default implementation
+ prefers lower-numbered logical CPUs, but architecture
+ overrides remain authoritative. Equal priorities do
+ not establish a sibling preference. For example, x86
+ with CONFIG_SCHED_MC_PRIO normally assigns the same
+ core priority to both SMT siblings, so this option
+ does not favor either sibling.
+
sched_verbose [KNL,EARLY] Enables verbose scheduler debug messages.
schedstats= [KNL,X86] Enable or disable scheduled statistics.
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
index 3dab0253976fb..919d0fb00bd9f 100644
--- a/kernel/sched/topology.c
+++ b/kernel/sched/topology.c
@@ -32,6 +32,20 @@ static int __init sched_debug_setup(char *str)
}
early_param("sched_verbose", sched_debug_setup);
+#ifdef CONFIG_SCHED_SMT
+static bool sched_smt_asym_packing __read_mostly;
+
+static int __init setup_sched_smt_asym_packing(char *str)
+{
+ if (strcmp(str, "on"))
+ return 0;
+
+ sched_smt_asym_packing = true;
+ return 1;
+}
+__setup("sched_smt_asym_packing=", setup_sched_smt_asym_packing);
+#endif
+
static inline bool sched_debug(void)
{
return sched_debug_verbose;
@@ -1954,6 +1968,10 @@ sd_init(struct sched_domain_topology_level *tl,
if (WARN_ONCE(sd_flags & ~TOPOLOGY_SD_FLAGS,
"wrong sd_flags in topology description\n"))
sd_flags &= TOPOLOGY_SD_FLAGS;
+#ifdef CONFIG_SCHED_SMT
+ if (sched_smt_asym_packing && (sd_flags & SD_SHARE_CPUCAPACITY))
+ sd_flags |= SD_ASYM_PACKING;
+#endif
sd_flags |= asym_cpu_capacity_classify(sd_span, cpu_map);
*sd = (struct sched_domain){
--
2.55.0
^ permalink raw reply [flat|nested] 3+ messages in thread