mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: K Prateek Nayak <kprateek.nayak@amd.com>
To: Peter Zijlstra <peterz@infradead.org>,
	Chen Yu <yu.c.chen@intel.com>,
	"Tim Chen" <tim.c.chen@linux.intel.com>,
	Ingo Molnar <mingo@redhat.com>,
	Juri Lelli <juri.lelli@redhat.com>,
	Vincent Guittot <vincent.guittot@linaro.org>,
	"Andrew Morton" <akpm@linux-foundation.org>,
	Arnd Bergmann <arnd@arndb.de>, <linux-kernel@vger.kernel.org>,
	<linux-arch@vger.kernel.org>, <linux-s390@vger.kernel.org>,
	<linuxppc-dev@lists.ozlabs.org>, <linux-mips@vger.kernel.org>,
	<loongarch@lists.linux.dev>, <driver-core@lists.linux.dev>,
	Sudeep Holla <sudeep.holla@kernel.org>,
	"Greg Kroah-Hartman" <gregkh@linuxfoundation.org>,
	"Rafael J. Wysocki" <rafael@kernel.org>,
	Danilo Krummrich <dakr@kernel.org>,
	Huacai Chen <chenhuacai@kernel.org>,
	Thomas Bogendoerfer <tsbogend@alpha.franken.de>,
	Jiaxun Yang <jiaxun.yang@flygoat.com>,
	Madhavan Srinivasan <maddy@linux.ibm.com>,
	Heiko Carstens <hca@linux.ibm.com>,
	Vasily Gorbik <gor@linux.ibm.com>,
	Alexander Gordeev <agordeev@linux.ibm.com>,
	"David S. Miller" <davem@davemloft.net>,
	Andreas Larsson <andreas@gaisler.com>,
	"Thomas Gleixner" <tglx@kernel.org>,
	Borislav Petkov <bp@alien8.de>,
	Dave Hansen <dave.hansen@linux.intel.com>, <x86@kernel.org>
Cc: Dietmar Eggemann <dietmar.eggemann@arm.com>,
	Steven Rostedt <rostedt@goodmis.org>,
	Ben Segall <bsegall@google.com>, Mel Gorman <mgorman@suse.de>,
	Valentin Schneider <vschneid@redhat.com>,
	Shrikanth Hegde <sshegde@linux.ibm.com>,
	K Prateek Nayak <kprateek.nayak@amd.com>,
	"WANG Xuerui" <kernel@xen0n.name>,
	Michael Ellerman <mpe@ellerman.id.au>,
	"Nicholas Piggin" <npiggin@gmail.com>,
	Christophe Leroy <chleroy@kernel.org>,
	"Christian Borntraeger" <borntraeger@linux.ibm.com>,
	Sven Schnelle <svens@linux.ibm.com>,
	"H. Peter Anvin" <hpa@zytor.com>
Subject: [RFC PATCH v3 10/13] lib/sbm: Dynamically allocate sbm index when CPU is activated
Date: Thu, 1 Oct 2026 19:28:46 +0000	[thread overview]
Message-ID: <20261001192849.74788-11-kprateek.nayak@amd.com> (raw)
In-Reply-To: <20261001192849.74788-1-kprateek.nayak@amd.com>

Add infrastructure to establish CPU to sparsebitmap (sbm) index relation
before CPU is turned active. CPU coming online looks for a free slot
based on its instance ID and acquires a free slot.

If slots are exhausted, a new leaf is allocated for an instance ID. New
CPUs activating with same instance ID can claim free slots on the same
leaf but must never exceed max_threads_per_instance.

For architectures that have not initialized the sbm topology, sbm core
overrides the arch_sbm_cpu_instance_id() in sbm_cpu_instance_id()
wrapper to always return 0 keeping all CPUs on same instance.

Data structures that require fast access have been runtime constified to
enable faster access in kernel hot paths.

Signed-off-by: K Prateek Nayak <kprateek.nayak@amd.com>
---
 include/asm-generic/vmlinux.lds.h |   6 +-
 include/linux/sbm.h               |  14 +++
 init/main.c                       |   6 +
 kernel/sched/core.c               |  17 +++
 lib/sbm.c                         | 175 +++++++++++++++++++++++++++++-
 5 files changed, 215 insertions(+), 3 deletions(-)

diff --git a/include/asm-generic/vmlinux.lds.h b/include/asm-generic/vmlinux.lds.h
index b2988aa12f66..b0346d382cea 100644
--- a/include/asm-generic/vmlinux.lds.h
+++ b/include/asm-generic/vmlinux.lds.h
@@ -981,7 +981,11 @@
 		RUNTIME_CONST(ptr, __bfilp_cache)			\
 		RUNTIME_CONST(shift, __futex_shift)			\
 		RUNTIME_CONST(mask,  __futex_mask)			\
-		RUNTIME_CONST(ptr,   __futex_queues)
+		RUNTIME_CONST(ptr,   __futex_queues)			\
+		RUNTIME_CONST(shift, __sbm_shift)			\
+		RUNTIME_CONST(mask,  __sbm_mask)			\
+		RUNTIME_CONST(ptr,   __sbm_cpu_to_idx)			\
+		RUNTIME_CONST(ptr,   __sbm_idx_to_cpu)
 
 /* Alignment must be consistent with (kunit_suite *) in include/kunit/test.h */
 #define KUNIT_TABLE()							\
diff --git a/include/linux/sbm.h b/include/linux/sbm.h
index adac12ed233a..232b0076bb3f 100644
--- a/include/linux/sbm.h
+++ b/include/linux/sbm.h
@@ -2,7 +2,21 @@
 #ifndef _LINUX_SBM_H
 #define _LINUX_SBM_H
 
+/*
+ * Masks and shifts for sbm index to translate
+ * a sbm leaf to CPU.
+ */
+extern int __sbm_shift;
+extern int __sbm_mask;
+
 int arch_sbm_cpu_instance_id(int cpu);
 void sbm_set_topology(int num_instances, int max_threads_per_instance);
 
+int sbm_cpu_to_idx(int cpu);
+int sbm_idx_to_cpu(int idx);
+
+int alloc_sbm_index(int cpu);
+void free_sbm_index(int cpu);
+int sbm_init(void);
+
 #endif /* _LINUX_SBM_H */
diff --git a/init/main.c b/init/main.c
index 2613d3f9b3ce..b7406bd3acc8 100644
--- a/init/main.c
+++ b/init/main.c
@@ -72,6 +72,7 @@
 #include <linux/pid_namespace.h>
 #include <linux/device/driver.h>
 #include <linux/kthread.h>
+#include <linux/sbm.h>
 #include <linux/sched.h>
 #include <linux/sched/init.h>
 #include <linux/signal.h>
@@ -1652,6 +1653,11 @@ static noinline void __init kernel_init_freeable(void)
 
 	smp_prepare_cpus(setup_max_cpus);
 
+	sbm_init();
+
+	/* Finish initializing boot CPU since it is already active. */
+	alloc_sbm_index(smp_processor_id());
+
 	workqueue_init();
 
 	init_mm_internals();
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 0bb86a43a592..977f579da410 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -61,6 +61,7 @@
 #include <linux/rcuwait_api.h>
 #include <linux/rseq.h>
 #include <linux/sched/wake_q.h>
+#include <linux/sbm.h>
 #include <linux/scs.h>
 #include <linux/slab.h>
 #include <linux/syscalls.h>
@@ -8636,6 +8637,14 @@ int sched_cpu_activate(unsigned int cpu)
 	 */
 	balance_push_set(cpu, false);
 
+	alloc_sbm_index(cpu);
+
+	/*
+	 * Make sure sbm mappings are visible
+	 * before CPU is toggled active.
+	 */
+	smp_mb();
+
 	/*
 	 * When going up, increment the number of cores with SMT present.
 	 */
@@ -8688,6 +8697,14 @@ int sched_cpu_deactivate(unsigned int cpu)
 
 	set_cpu_active(cpu, false);
 
+	/*
+	 * Make sure CPU is inactive before
+	 * sbm indices are reclaimed.
+	 */
+	smp_mb();
+
+	free_sbm_index(cpu);
+
 	/*
 	 * From this point forward, this CPU will refuse to run any task that
 	 * is not: migrate_disable() or KTHREAD_IS_PER_CPU, and will actively
diff --git a/lib/sbm.c b/lib/sbm.c
index 82280e3306de..e5b0508b6825 100644
--- a/lib/sbm.c
+++ b/lib/sbm.c
@@ -1,10 +1,58 @@
 /* SPDX-License-Identifier: GPL-2.0 */
 #include <linux/sbm.h>
 #include <linux/init.h>
+#include <linux/log2.h>
+#include <linux/slab.h>
+#include <linux/cache.h>
 #include <linux/printk.h>
+#include <linux/cpumask.h>
+#include <linux/jump_label.h>
 
-static int sbm_max_threads_per_instance = -1;
-static int sbm_num_instance = -1;
+#include <asm/runtime-const.h>
+
+static int sbm_max_threads_per_instance __ro_after_init = -1;
+static int sbm_num_instance __ro_after_init = -1;
+
+int __sbm_shift __ro_after_init;
+int __sbm_mask __ro_after_init;
+
+static struct {
+	int		instance_id;	/* Instance ID linked to the leaf. */
+	unsigned long	allocated_mask;	/* Set of IDs that have been allocated. */
+} *__sbm_idx_metadata __ro_after_init;
+
+/* Translations between cpu <-> sbm leaf */
+static int *__sbm_cpu_to_idx __ro_after_init;
+static int *__sbm_idx_to_cpu __ro_after_init;
+
+static __always_inline int *_sbm_cpu_to_idx(void)
+{
+	return runtime_const_ptr(__sbm_cpu_to_idx);
+}
+
+static __always_inline int *_sbm_idx_to_cpu(void)
+{
+	return runtime_const_ptr(__sbm_idx_to_cpu);
+}
+
+int sbm_cpu_to_idx(int cpu)
+{
+	return _sbm_cpu_to_idx()[cpu];
+}
+
+int sbm_idx_to_cpu(int idx)
+{
+	return _sbm_idx_to_cpu()[idx];
+}
+
+/*
+ * Certain architectures may skip initializing sbm propoerties
+ * while having an arch_sbm_cpu_instance_id() definition.
+ *
+ * In such cases, don't trust the arch/ side redefine and use
+ * the default single instance mapping.
+ */
+static DEFINE_STATIC_KEY_FALSE(sbm_arch_initialized);
 
 /*
  * In absence of an arch definition, consider all CPUs to
@@ -15,6 +63,72 @@ int __weak arch_sbm_cpu_instance_id(int cpu)
 	return 0;
 }
 
+static int sbm_cpu_to_instance(int cpu)
+{
+	if (static_branch_likely(&sbm_arch_initialized))
+		return arch_sbm_cpu_instance_id(cpu);
+
+	return 0;
+}
+
+int alloc_sbm_index(int cpu)
+{
+	int cpu_instance = sbm_cpu_to_instance(cpu);
+	int i, idx = BITS_PER_LONG, free_index = -1;
+
+	for (i = 0; i < sbm_num_instance; ++i) {
+		if (__sbm_idx_metadata[i].instance_id == cpu_instance) {
+			idx = find_first_zero_bit(&__sbm_idx_metadata[i].allocated_mask,
+						  BITS_PER_LONG);
+
+			if (idx < BITS_PER_LONG)
+				break;
+		}
+		if (free_index == -1 && __sbm_idx_metadata[i].instance_id == -1)
+			free_index = i;
+	}
+
+	if (i == sbm_num_instance && free_index == -1)
+		return -ENOENT;
+
+	if (i == sbm_num_instance) {
+		__sbm_idx_metadata[free_index].instance_id = cpu_instance;
+		i = free_index;
+		idx = 0;
+	}
+
+	WARN_ON_ONCE(idx >= sbm_max_threads_per_instance);
+
+	__set_bit(idx, &__sbm_idx_metadata[i].allocated_mask);
+
+	idx = (i << __sbm_shift) + idx;
+	_sbm_idx_to_cpu()[idx] = cpu;
+	_sbm_cpu_to_idx()[cpu] = idx;
+
+	return 0;
+}
+
+void free_sbm_index(int cpu)
+{
+	int idx = sbm_cpu_to_idx(cpu);
+	u32 leaf;
+
+	if (idx < 0)
+		return;
+
+	_sbm_idx_to_cpu()[idx] = -1;
+	_sbm_cpu_to_idx()[cpu] = -1;
+
+	leaf = runtime_const_shift_right_32(idx, __sbm_shift);
+	idx = runtime_const_mask_32(idx, __sbm_mask);
+
+	__clear_bit(idx, &__sbm_idx_metadata[leaf].allocated_mask);
+
+	if (find_first_bit(&__sbm_idx_metadata[leaf].allocated_mask, BITS_PER_LONG) ==
+	    BITS_PER_LONG)
+		__sbm_idx_metadata[leaf].instance_id = -1;
+}
+
 void __init sbm_set_topology(int num_instances, int max_threads_per_instance)
 {
 	sbm_max_threads_per_instance = max_threads_per_instance;
@@ -25,3 +139,60 @@ void __init sbm_set_topology(int num_instances, int max_threads_per_instance)
 		sbm_max_threads_per_instance);
 }
 
+int __init sbm_init(void)
+{
+	int i;
+
+	if (sbm_max_threads_per_instance > 0 && sbm_num_instance > 0) {
+		static_branch_enable(&sbm_arch_initialized);
+		goto init_properties;
+	}
+
+	sbm_max_threads_per_instance = BITS_PER_LONG;
+	sbm_num_instance = (nr_cpumask_bits / BITS_PER_LONG) + 1;
+
+init_properties:
+	/*
+	 * If the number of CPUs per instance cross bitmask word boundary,
+	 * split the instances into samller chunks on BITS_PER_LONG and
+	 * increase the order of leaves.
+	 */
+	if (sbm_max_threads_per_instance > BITS_PER_LONG) {
+		int split = (sbm_max_threads_per_instance + BITS_PER_LONG - 1) / BITS_PER_LONG;
+
+		sbm_max_threads_per_instance = BITS_PER_LONG;
+		sbm_num_instance *= split;
+	}
+
+	sbm_max_threads_per_instance = roundup_pow_of_two(sbm_max_threads_per_instance);
+
+	__sbm_shift = ilog2(sbm_max_threads_per_instance);
+	__sbm_mask = sbm_max_threads_per_instance - 1;
+
+	__sbm_idx_metadata = kzalloc_objs(*__sbm_idx_metadata, sbm_num_instance);
+	__sbm_cpu_to_idx = kzalloc_objs(*__sbm_cpu_to_idx, nr_cpumask_bits);
+	__sbm_idx_to_cpu = kzalloc_objs(*__sbm_idx_to_cpu,
+					sbm_max_threads_per_instance * sbm_num_instance);
+
+	BUG_ON(!__sbm_cpu_to_idx || !__sbm_idx_to_cpu || !__sbm_idx_metadata);
+
+	for (i = 0; i < nr_cpumask_bits; ++i)
+		__sbm_cpu_to_idx[i] = -1;
+
+	/* Set all instance id to -1 to allow for future allocations to claim them. */
+	for (i = 0; i < sbm_num_instance; ++i)
+		__sbm_idx_metadata[i].instance_id = -1;
+
+	runtime_const_init(shift, __sbm_shift);
+	runtime_const_init(mask,  __sbm_mask);
+	runtime_const_init(ptr,   __sbm_cpu_to_idx);
+	runtime_const_init(ptr,   __sbm_idx_to_cpu);
+
+	barrier();
+
+	pr_info("sbm instance count: %d (maximum threads per instance: %d)\n",
+		sbm_num_instance,
+		sbm_max_threads_per_instance);
+
+	return 0;
+}
-- 
2.34.1


  parent reply	other threads:[~2026-10-01 19:33 UTC|newest]

Thread overview: 18+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-01 19:28 [RFC PATCH v3 00/13] lib, sched: Introduce sparsebitmap (sbm) K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 01/13] lib/sbm: Introduce helpers for architectures to configure LLC properties K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 02/13] drivers/base/arch_topology: Add support for initializing sbm topology K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 03/13] LoongArch: Initialize CPU _PXM relation for disabled CPUs from SRAT K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 04/13] LoongArch: Configure sbm topology during SMP preparation K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 05/13] MIPS: Initialize sbm topology on multi-node systems K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 06/13] powerpc/setup: Initialize sbm topology based on coregroup / NUMA topology K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 07/13] s390/topology: Initialize sbm topology during topology_init_early() K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 08/13] sparc64: Initialize sbm topology on multi-LLC system K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 09/13] x86/cpu/topology: Initialize sbm topology after topology parsing K Prateek Nayak
2026-10-03  8:27   ` Chen Yu
2026-10-04  6:17     ` K Prateek Nayak
2026-10-01 19:28 ` K Prateek Nayak [this message]
2026-10-01 19:28 ` [RFC PATCH v3 11/13] lib/sbm: Add helpers to allocate, set, clear, and traverse the bits on sbm K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 12/13] sched/fair: Allocate nohz.idle_cpus_mask during sched_init_smp() K Prateek Nayak
2026-10-01 19:28 ` [RFC PATCH v3 13/13] sched/fair: Switch nohz.idle_cpus to use sbm K Prateek Nayak
2026-10-03  9:10 ` [RFC PATCH v3 00/13] lib, sched: Introduce sparsebitmap (sbm) Chen Yu
2026-10-04  6:13   ` K Prateek Nayak

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261001192849.74788-11-kprateek.nayak@amd.com \
    --to=kprateek.nayak@amd.com \
    --cc=agordeev@linux.ibm.com \
    --cc=akpm@linux-foundation.org \
    --cc=andreas@gaisler.com \
    --cc=arnd@arndb.de \
    --cc=borntraeger@linux.ibm.com \
    --cc=bp@alien8.de \
    --cc=bsegall@google.com \
    --cc=chenhuacai@kernel.org \
    --cc=chleroy@kernel.org \
    --cc=dakr@kernel.org \
    --cc=dave.hansen@linux.intel.com \
    --cc=davem@davemloft.net \
    --cc=dietmar.eggemann@arm.com \
    --cc=driver-core@lists.linux.dev \
    --cc=gor@linux.ibm.com \
    --cc=gregkh@linuxfoundation.org \
    --cc=hca@linux.ibm.com \
    --cc=hpa@zytor.com \
    --cc=jiaxun.yang@flygoat.com \
    --cc=juri.lelli@redhat.com \
    --cc=kernel@xen0n.name \
    --cc=linux-arch@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mips@vger.kernel.org \
    --cc=linux-s390@vger.kernel.org \
    --cc=linuxppc-dev@lists.ozlabs.org \
    --cc=loongarch@lists.linux.dev \
    --cc=maddy@linux.ibm.com \
    --cc=mgorman@suse.de \
    --cc=mingo@redhat.com \
    --cc=mpe@ellerman.id.au \
    --cc=npiggin@gmail.com \
    --cc=peterz@infradead.org \
    --cc=rafael@kernel.org \
    --cc=rostedt@goodmis.org \
    --cc=sshegde@linux.ibm.com \
    --cc=sudeep.holla@kernel.org \
    --cc=svens@linux.ibm.com \
    --cc=tglx@kernel.org \
    --cc=tim.c.chen@linux.intel.com \
    --cc=tsbogend@alpha.franken.de \
    --cc=vincent.guittot@linaro.org \
    --cc=vschneid@redhat.com \
    --cc=x86@kernel.org \
    --cc=yu.c.chen@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®