[RFC PATCH v3 10/13] lib/sbm: Dynamically allocate sbm index when CPU is activated
From: K Prateek Nayak
Date: Thu Oct 01 2026 - 15:33:35 EST
Add infrastructure to establish CPU to sparsebitmap (sbm) index relation
before CPU is turned active. CPU coming online looks for a free slot
based on its instance ID and acquires a free slot.
If slots are exhausted, a new leaf is allocated for an instance ID. New
CPUs activating with same instance ID can claim free slots on the same
leaf but must never exceed max_threads_per_instance.
For architectures that have not initialized the sbm topology, sbm core
overrides the arch_sbm_cpu_instance_id() in sbm_cpu_instance_id()
wrapper to always return 0 keeping all CPUs on same instance.
Data structures that require fast access have been runtime constified to
enable faster access in kernel hot paths.
Signed-off-by: K Prateek Nayak <kprateek.nayak@xxxxxxx>
---
include/asm-generic/vmlinux.lds.h | 6 +-
include/linux/sbm.h | 14 +++
init/main.c | 6 +
kernel/sched/core.c | 17 +++
lib/sbm.c | 175 +++++++++++++++++++++++++++++-
5 files changed, 215 insertions(+), 3 deletions(-)
diff --git a/include/asm-generic/vmlinux.lds.h b/include/asm-generic/vmlinux.lds.h
index b2988aa12f66..b0346d382cea 100644
--- a/include/asm-generic/vmlinux.lds.h
+++ b/include/asm-generic/vmlinux.lds.h
@@ -981,7 +981,11 @@
RUNTIME_CONST(ptr, __bfilp_cache) \
RUNTIME_CONST(shift, __futex_shift) \
RUNTIME_CONST(mask, __futex_mask) \
- RUNTIME_CONST(ptr, __futex_queues)
+ RUNTIME_CONST(ptr, __futex_queues) \
+ RUNTIME_CONST(shift, __sbm_shift) \
+ RUNTIME_CONST(mask, __sbm_mask) \
+ RUNTIME_CONST(ptr, __sbm_cpu_to_idx) \
+ RUNTIME_CONST(ptr, __sbm_idx_to_cpu)
/* Alignment must be consistent with (kunit_suite *) in include/kunit/test.h */
#define KUNIT_TABLE() \
diff --git a/include/linux/sbm.h b/include/linux/sbm.h
index adac12ed233a..232b0076bb3f 100644
--- a/include/linux/sbm.h
+++ b/include/linux/sbm.h
@@ -2,7 +2,21 @@
#ifndef _LINUX_SBM_H
#define _LINUX_SBM_H
+/*
+ * Masks and shifts for sbm index to translate
+ * a sbm leaf to CPU.
+ */
+extern int __sbm_shift;
+extern int __sbm_mask;
+
int arch_sbm_cpu_instance_id(int cpu);
void sbm_set_topology(int num_instances, int max_threads_per_instance);
+int sbm_cpu_to_idx(int cpu);
+int sbm_idx_to_cpu(int idx);
+
+int alloc_sbm_index(int cpu);
+void free_sbm_index(int cpu);
+int sbm_init(void);
+
#endif /* _LINUX_SBM_H */
diff --git a/init/main.c b/init/main.c
index 2613d3f9b3ce..b7406bd3acc8 100644
--- a/init/main.c
+++ b/init/main.c
@@ -72,6 +72,7 @@
#include <linux/pid_namespace.h>
#include <linux/device/driver.h>
#include <linux/kthread.h>
+#include <linux/sbm.h>
#include <linux/sched.h>
#include <linux/sched/init.h>
#include <linux/signal.h>
@@ -1652,6 +1653,11 @@ static noinline void __init kernel_init_freeable(void)
smp_prepare_cpus(setup_max_cpus);
+ sbm_init();
+
+ /* Finish initializing boot CPU since it is already active. */
+ alloc_sbm_index(smp_processor_id());
+
workqueue_init();
init_mm_internals();
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 0bb86a43a592..977f579da410 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -61,6 +61,7 @@
#include <linux/rcuwait_api.h>
#include <linux/rseq.h>
#include <linux/sched/wake_q.h>
+#include <linux/sbm.h>
#include <linux/scs.h>
#include <linux/slab.h>
#include <linux/syscalls.h>
@@ -8636,6 +8637,14 @@ int sched_cpu_activate(unsigned int cpu)
*/
balance_push_set(cpu, false);
+ alloc_sbm_index(cpu);
+
+ /*
+ * Make sure sbm mappings are visible
+ * before CPU is toggled active.
+ */
+ smp_mb();
+
/*
* When going up, increment the number of cores with SMT present.
*/
@@ -8688,6 +8697,14 @@ int sched_cpu_deactivate(unsigned int cpu)
set_cpu_active(cpu, false);
+ /*
+ * Make sure CPU is inactive before
+ * sbm indices are reclaimed.
+ */
+ smp_mb();
+
+ free_sbm_index(cpu);
+
/*
* From this point forward, this CPU will refuse to run any task that
* is not: migrate_disable() or KTHREAD_IS_PER_CPU, and will actively
diff --git a/lib/sbm.c b/lib/sbm.c
index 82280e3306de..e5b0508b6825 100644
--- a/lib/sbm.c
+++ b/lib/sbm.c
@@ -1,10 +1,58 @@
/* SPDX-License-Identifier: GPL-2.0 */
#include <linux/sbm.h>
#include <linux/init.h>
+#include <linux/log2.h>
+#include <linux/slab.h>
+#include <linux/cache.h>
#include <linux/printk.h>
+#include <linux/cpumask.h>
+#include <linux/jump_label.h>
-static int sbm_max_threads_per_instance = -1;
-static int sbm_num_instance = -1;
+#include <asm/runtime-const.h>
+
+static int sbm_max_threads_per_instance __ro_after_init = -1;
+static int sbm_num_instance __ro_after_init = -1;
+
+int __sbm_shift __ro_after_init;
+int __sbm_mask __ro_after_init;
+
+static struct {
+ int instance_id; /* Instance ID linked to the leaf. */
+ unsigned long allocated_mask; /* Set of IDs that have been allocated. */
+} *__sbm_idx_metadata __ro_after_init;
+
+/* Translations between cpu <-> sbm leaf */
+static int *__sbm_cpu_to_idx __ro_after_init;
+static int *__sbm_idx_to_cpu __ro_after_init;
+
+static __always_inline int *_sbm_cpu_to_idx(void)
+{
+ return runtime_const_ptr(__sbm_cpu_to_idx);
+}
+
+static __always_inline int *_sbm_idx_to_cpu(void)
+{
+ return runtime_const_ptr(__sbm_idx_to_cpu);
+}
+
+int sbm_cpu_to_idx(int cpu)
+{
+ return _sbm_cpu_to_idx()[cpu];
+}
+
+int sbm_idx_to_cpu(int idx)
+{
+ return _sbm_idx_to_cpu()[idx];
+}
+
+/*
+ * Certain architectures may skip initializing sbm propoerties
+ * while having an arch_sbm_cpu_instance_id() definition.
+ *
+ * In such cases, don't trust the arch/ side redefine and use
+ * the default single instance mapping.
+ */
+static DEFINE_STATIC_KEY_FALSE(sbm_arch_initialized);
/*
* In absence of an arch definition, consider all CPUs to
@@ -15,6 +63,72 @@ int __weak arch_sbm_cpu_instance_id(int cpu)
return 0;
}
+static int sbm_cpu_to_instance(int cpu)
+{
+ if (static_branch_likely(&sbm_arch_initialized))
+ return arch_sbm_cpu_instance_id(cpu);
+
+ return 0;
+}
+
+int alloc_sbm_index(int cpu)
+{
+ int cpu_instance = sbm_cpu_to_instance(cpu);
+ int i, idx = BITS_PER_LONG, free_index = -1;
+
+ for (i = 0; i < sbm_num_instance; ++i) {
+ if (__sbm_idx_metadata[i].instance_id == cpu_instance) {
+ idx = find_first_zero_bit(&__sbm_idx_metadata[i].allocated_mask,
+ BITS_PER_LONG);
+
+ if (idx < BITS_PER_LONG)
+ break;
+ }
+ if (free_index == -1 && __sbm_idx_metadata[i].instance_id == -1)
+ free_index = i;
+ }
+
+ if (i == sbm_num_instance && free_index == -1)
+ return -ENOENT;
+
+ if (i == sbm_num_instance) {
+ __sbm_idx_metadata[free_index].instance_id = cpu_instance;
+ i = free_index;
+ idx = 0;
+ }
+
+ WARN_ON_ONCE(idx >= sbm_max_threads_per_instance);
+
+ __set_bit(idx, &__sbm_idx_metadata[i].allocated_mask);
+
+ idx = (i << __sbm_shift) + idx;
+ _sbm_idx_to_cpu()[idx] = cpu;
+ _sbm_cpu_to_idx()[cpu] = idx;
+
+ return 0;
+}
+
+void free_sbm_index(int cpu)
+{
+ int idx = sbm_cpu_to_idx(cpu);
+ u32 leaf;
+
+ if (idx < 0)
+ return;
+
+ _sbm_idx_to_cpu()[idx] = -1;
+ _sbm_cpu_to_idx()[cpu] = -1;
+
+ leaf = runtime_const_shift_right_32(idx, __sbm_shift);
+ idx = runtime_const_mask_32(idx, __sbm_mask);
+
+ __clear_bit(idx, &__sbm_idx_metadata[leaf].allocated_mask);
+
+ if (find_first_bit(&__sbm_idx_metadata[leaf].allocated_mask, BITS_PER_LONG) ==
+ BITS_PER_LONG)
+ __sbm_idx_metadata[leaf].instance_id = -1;
+}
+
void __init sbm_set_topology(int num_instances, int max_threads_per_instance)
{
sbm_max_threads_per_instance = max_threads_per_instance;
@@ -25,3 +139,60 @@ void __init sbm_set_topology(int num_instances, int max_threads_per_instance)
sbm_max_threads_per_instance);
}
+int __init sbm_init(void)
+{
+ int i;
+
+ if (sbm_max_threads_per_instance > 0 && sbm_num_instance > 0) {
+ static_branch_enable(&sbm_arch_initialized);
+ goto init_properties;
+ }
+
+ sbm_max_threads_per_instance = BITS_PER_LONG;
+ sbm_num_instance = (nr_cpumask_bits / BITS_PER_LONG) + 1;
+
+init_properties:
+ /*
+ * If the number of CPUs per instance cross bitmask word boundary,
+ * split the instances into samller chunks on BITS_PER_LONG and
+ * increase the order of leaves.
+ */
+ if (sbm_max_threads_per_instance > BITS_PER_LONG) {
+ int split = (sbm_max_threads_per_instance + BITS_PER_LONG - 1) / BITS_PER_LONG;
+
+ sbm_max_threads_per_instance = BITS_PER_LONG;
+ sbm_num_instance *= split;
+ }
+
+ sbm_max_threads_per_instance = roundup_pow_of_two(sbm_max_threads_per_instance);
+
+ __sbm_shift = ilog2(sbm_max_threads_per_instance);
+ __sbm_mask = sbm_max_threads_per_instance - 1;
+
+ __sbm_idx_metadata = kzalloc_objs(*__sbm_idx_metadata, sbm_num_instance);
+ __sbm_cpu_to_idx = kzalloc_objs(*__sbm_cpu_to_idx, nr_cpumask_bits);
+ __sbm_idx_to_cpu = kzalloc_objs(*__sbm_idx_to_cpu,
+ sbm_max_threads_per_instance * sbm_num_instance);
+
+ BUG_ON(!__sbm_cpu_to_idx || !__sbm_idx_to_cpu || !__sbm_idx_metadata);
+
+ for (i = 0; i < nr_cpumask_bits; ++i)
+ __sbm_cpu_to_idx[i] = -1;
+
+ /* Set all instance id to -1 to allow for future allocations to claim them. */
+ for (i = 0; i < sbm_num_instance; ++i)
+ __sbm_idx_metadata[i].instance_id = -1;
+
+ runtime_const_init(shift, __sbm_shift);
+ runtime_const_init(mask, __sbm_mask);
+ runtime_const_init(ptr, __sbm_cpu_to_idx);
+ runtime_const_init(ptr, __sbm_idx_to_cpu);
+
+ barrier();
+
+ pr_info("sbm instance count: %d (maximum threads per instance: %d)\n",
+ sbm_num_instance,
+ sbm_max_threads_per_instance);
+
+ return 0;
+}
--
2.34.1