[RFC PATCH v3 06/13] powerpc/setup: Initialize sbm topology based on coregroup / NUMA topology
From: K Prateek Nayak
Date: Thu Oct 01 2026 - 15:32:45 EST
Initialize sparsebitmap (sbm) topology based on the coregroup
information. Each coregorup gets its own sparsemask leaf.
pSeries and memory hotplug are interesting since a hotplug can place a
newly added CPU on any online node. This requires special care to allow
estimating bitmask size considering the worst case scenarios - each CPU
onlined is on a separate node, and this is a new N_CPU node.
Platform may enforce a stricter standards for the CPUs being online and
what nodes they can be mapped to but the current implementations makes
no assumptions and considers each CPU can be onlined on a unique node.
pSeries systems that can hotplug CPUs (detected using smp_ops) use the
NUMA topology instead for sbm initialization. arch_sbm_cpu_instance_id()
on these systems use cpu_to_node() mappings to match CPUs to sbm
instances.
XXX: This requires further optimizations to shorten sparsemask
traversals by keeping the number of leaf nodes to a minimum. If there
are nuances I'm not aware of, please reach out.
Signed-off-by: K Prateek Nayak <kprateek.nayak@xxxxxxx>
---
Tested on ppc64le_defconfig with:
qemu-system-ppc64 \
-M pseries \
-cpu power10 \
-smp sockets=2,cores=2,threads=4 \
-m 10G -nographic \
-kernel vmlinux \
-append "root=/dev/ram sched_debug"
and also on ppce500 VM based on instructions in
https://www.qemu.org/docs/master/system/ppc/ppce500.html
---
arch/powerpc/kernel/setup-common.c | 88 ++++++++++++++++++++
arch/powerpc/platforms/pseries/hotplug-cpu.c | 10 +++
2 files changed, 98 insertions(+)
diff --git a/arch/powerpc/kernel/setup-common.c b/arch/powerpc/kernel/setup-common.c
index 4afaba19b586..4b57ad553172 100644
--- a/arch/powerpc/kernel/setup-common.c
+++ b/arch/powerpc/kernel/setup-common.c
@@ -11,6 +11,7 @@
#include <linux/export.h>
#include <linux/panic_notifier.h>
#include <linux/string.h>
+#include <linux/sbm.h>
#include <linux/sched.h>
#include <linux/init.h>
#include <linux/kernel.h>
@@ -602,6 +603,87 @@ static __init int add_pcspkr(void)
device_initcall(add_pcspkr);
#endif /* CONFIG_PCSPKR_PLATFORM */
+int arch_sbm_cpu_instance_id(int cpu)
+{
+ /*
+ * In case of pSeries processors, sbm masks are
+ * grouped by nodes where the cpuhotplug
+ * operations can remove and re-add same logical
+ * CPUs on different nodes.
+ *
+ * See comment in pseries_cpu_hotplug_init().
+ */
+ if (smp_ops->cpu_disable)
+ return cpu_to_node(cpu);
+
+ return cpu_to_coregroup_id(cpu);
+}
+
+static void __init setup_sbm_topology(void)
+{
+ int num_sbm_instances, max_threads_per_instance = 1;
+ struct cpumask *cpu_sbm_setup_map;
+ int i, *__node_thread_count;
+ int disabled_cpus = 0;
+
+ cpu_sbm_setup_map = memblock_alloc_or_panic(cpumask_size(), __alignof__(long));
+ __node_thread_count = memblock_alloc_or_panic(nr_cpu_ids * sizeof(int),
+ __alignof__(int));
+
+ memset(__node_thread_count, 0, nr_cpu_ids * sizeof(int));
+ memset(cpu_sbm_setup_map, 0, cpumask_size());
+
+ for_each_possible_cpu(i) {
+ bool found = false;
+ int j;
+
+ if (!cpu_present(i)) {
+ disabled_cpus += 1;
+ continue;
+ }
+
+ for_each_cpu(j, cpu_sbm_setup_map) {
+ if (cpu_to_coregroup_id(i) == cpu_to_coregroup_id(j)) {
+ found = true;
+ break;
+ }
+ }
+
+ if (!found) {
+ cpumask_set_cpu(i, cpu_sbm_setup_map);
+ __node_thread_count[i] = 1;
+ continue;
+ }
+
+ __node_thread_count[j] += 1;
+ max_threads_per_instance = max(max_threads_per_instance,
+ __node_thread_count[j]);
+ }
+
+ /*
+ * If CPUs are disabled, they may pop up on any online node.
+ *
+ * XXX: Any implementation nuances that can help this?
+ * pSeries says only online nodes can be extended.
+ */
+ if (disabled_cpus) {
+ num_sbm_instances = num_sbm_instances + disabled_cpus;
+ } else {
+ num_sbm_instances = cpumask_weight(cpu_sbm_setup_map);
+ }
+
+ /*
+ * If disabled threads exists, assume the maximum threads per
+ * instance can extend by the number of disabled threads if they
+ * are all added to the same node.
+ */
+ sbm_set_topology(num_sbm_instances,
+ max_threads_per_instance + disabled_cpus);
+
+ memblock_free(__node_thread_count, nr_cpu_ids * sizeof(int));
+ memblock_free(cpu_sbm_setup_map, cpumask_size());
+}
+
static char ppc_hw_desc_buf[128] __initdata;
struct seq_buf ppc_hw_desc __initdata = {
@@ -1006,6 +1088,12 @@ void __init setup_arch(char **cmdline_p)
early_memtest(min_low_pfn << PAGE_SHIFT, max_low_pfn << PAGE_SHIFT);
+ /*
+ * setup_arch() below can override topology for
+ * pSeries platforms as a result of hotplug nuances.
+ */
+ setup_sbm_topology();
+
if (ppc_md.setup_arch)
ppc_md.setup_arch();
diff --git a/arch/powerpc/platforms/pseries/hotplug-cpu.c b/arch/powerpc/platforms/pseries/hotplug-cpu.c
index bc6926dbf148..7c1c1ac3efde 100644
--- a/arch/powerpc/platforms/pseries/hotplug-cpu.c
+++ b/arch/powerpc/platforms/pseries/hotplug-cpu.c
@@ -19,6 +19,7 @@
#include <linux/kernel.h>
#include <linux/interrupt.h>
#include <linux/delay.h>
+#include <linux/sbm.h>
#include <linux/sched.h> /* for idle_task_exit */
#include <linux/sched/hotplug.h>
#include <linux/cpu.h>
@@ -870,6 +871,15 @@ void __init pseries_cpu_hotplug_init(void)
return;
}
+ /*
+ * find_cpu_id_range() only looks at online nodes.
+ *
+ * XXX: Is it possible for a CPU attached memory node to come
+ * online after this point? May need num_possbile_nodes() then
+ * unless there are platform nuances that can help optimize.
+ */
+ sbm_set_topology(num_online_nodes(), num_possible_cpus());
+
smp_ops->cpu_offline_self = pseries_cpu_offline_self;
smp_ops->cpu_disable = pseries_cpu_disable;
smp_ops->cpu_die = pseries_cpu_die;
--
2.34.1