diff mbox series

[v3,3/3] powerpc/numa: Support coregroup on PowerNV

Message ID 20260902123458.340456-8-srikar@linux.ibm.com (mailing list archive)
State New
Headers show
Series powerpc/numa: Enable coregroup support on PowerNV | expand

Checks

Context Check Description
snowpatch_ozlabs/github-powerpc_ppctests success Successfully ran 10 jobs.

Commit Message

Srikar Dronamraju Sept. 2, 2026, 12:35 p.m. UTC
Coregroup support on powerpc has so far been limited to PowerVM LPARs.
However, PowerNV can also support coregroups when firmware exposes the
required coregroup information through the associativity hierarchy.

Detect coregroup support by checking whether primary_domain_index is the
penultimate domain in the CPU node's ibm,associativity property. On
PowerNV, a non-penultimate primary_domain_index indicates that firmware
provides an additional level for coregroup information.

This keeps the logic compatible with PowerVM systems, where
primary_domain_index is likewise not the penultimate associativity
domain.

Signed-off-by: Srikar Dronamraju <srikar@linux.ibm.com>
---
Changelog from v2:
- Handle comments from Ritesh (one hunk needed to be moved from patch 2 to
  patch 3 to build correctly)

Changelog from v1: https://lkml.kernel.org/r/20260524010017.140408-1-srikar@linux.ibm.com
- Handle comments from Christophe Leroy; make code more flat

 arch/powerpc/mm/numa.c | 58 ++++++++++++++++++++++++++++++++++--------
 1 file changed, 48 insertions(+), 10 deletions(-)

Comments

Shrikanth Hegde Sept. 4, 2026, 10:04 a.m. UTC | #1
On 9/2/26 6:05 PM, Srikar Dronamraju wrote:
> Coregroup support on powerpc has so far been limited to PowerVM LPARs.
> However, PowerNV can also support coregroups when firmware exposes the
> required coregroup information through the associativity hierarchy.
> 

Existing firmware does expose this info already?

> Detect coregroup support by checking whether primary_domain_index is the
> penultimate domain in the CPU node's ibm,associativity property. On
> PowerNV, a non-penultimate primary_domain_index indicates that firmware
> provides an additional level for coregroup information.
> 
> This keeps the logic compatible with PowerVM systems, where
> primary_domain_index is likewise not the penultimate associativity
> domain.
> 

Could you please put the ibm,associativity on this powernv? as well
PowerVM's so that one understands it better?

> Signed-off-by: Srikar Dronamraju <srikar@linux.ibm.com>
> ---
> Changelog from v2:
> - Handle comments from Ritesh (one hunk needed to be moved from patch 2 to
>    patch 3 to build correctly)
> 
> Changelog from v1: https://lkml.kernel.org/r/20260524010017.140408-1-srikar@linux.ibm.com
> - Handle comments from Christophe Leroy; make code more flat
> 
>   arch/powerpc/mm/numa.c | 58 ++++++++++++++++++++++++++++++++++--------
>   1 file changed, 48 insertions(+), 10 deletions(-)
> 
> diff --git a/arch/powerpc/mm/numa.c b/arch/powerpc/mm/numa.c
> index 5f326b005a2a..e97b624203ea 100644
> --- a/arch/powerpc/mm/numa.c
> +++ b/arch/powerpc/mm/numa.c
> @@ -889,12 +889,32 @@ static int __init numa_setup_drmem_lmb(struct drmem_lmb *lmb,
>   	return 0;
>   }
>   
> +/*
> + * If hierarchy extends beyond primary_domain_index + 1, then next
> + * level corresponds to coregroup.
> + */
> +static int detect_and_enable_coregroup(const __be32 *associativity, int index)
> +{
> +	if (!associativity || index == -1)
> +		goto out;
> +
> +	index = of_read_number(associativity, 1);
> +
> +	if (index > primary_domain_index + 1) {
> +		coregroup_enabled = 1;
> +		return index;
> +	}
> +out:
> +	coregroup_enabled = 0;
> +	return -1;
> +}
> +
>   static int __init parse_numa_properties(void)
>   {
>   	struct device_node *memory, *pci;
> -	int default_nid = 0;
> -	unsigned long i;
> +	int default_nid = 0, index = 0;
>   	const __be32 *associativity;
> +	unsigned long i;
>   
>   	if (numa_enabled == 0) {
>   		pr_warn("disabled by user\n");
> @@ -927,7 +947,6 @@ static int __init parse_numa_properties(void)
>   	 */
>   	for_each_present_cpu(i) {
>   		__be32 vphn_assoc[VPHN_ASSOC_BUFSIZE];
> -		struct device_node *cpu;
>   		int nid = NUMA_NO_NODE;
>   
>   		memset(vphn_assoc, 0, VPHN_ASSOC_BUFSIZE * sizeof(__be32));
> @@ -935,7 +954,9 @@ static int __init parse_numa_properties(void)
>   		if (__vphn_get_associativity(i, vphn_assoc) == 0) {
>   			nid = associativity_to_nid(vphn_assoc);
>   			initialize_form1_numa_distance(vphn_assoc);
> +			index = detect_and_enable_coregroup(vphn_assoc, index);

nit: I don't see return value being used.

Other than that, rest looks good to me.
Reviewed-by: Shrikanth Hegde <sshegde@linux.ibm.com>

>   		} else {
> +			struct device_node *cpu;
>   
>   			/*
>   			 * Don't fall back to default_nid yet -- we will plug
> @@ -948,6 +969,7 @@ static int __init parse_numa_properties(void)
>   			associativity = of_get_associativity(cpu);
>   			if (associativity) {
>   				nid = associativity_to_nid(associativity);
> +				index = detect_and_enable_coregroup(associativity, index);
>   				initialize_form1_numa_distance(associativity);
>   			}
>   			of_node_put(cpu);
> @@ -1445,7 +1467,9 @@ static long vphn_get_associativity(unsigned long cpu,
>   
>   int cpu_to_coregroup_id(int cpu)
>   {
> -	__be32 associativity[VPHN_ASSOC_BUFSIZE] = {0};
> +	int coregroup_id = cpu_to_core_id(cpu);
> +	struct device_node *cpunode = NULL;
> +	const __be32 *associativity;
>   	int index;
>   
>   	if (cpu < 0 || cpu > nr_cpu_ids)
> @@ -1454,17 +1478,31 @@ int cpu_to_coregroup_id(int cpu)
>   	if (!coregroup_enabled)
>   		goto out;
>   
> -	if (!firmware_has_feature(FW_FEATURE_VPHN))
> -		goto out;
> +	if (firmware_has_feature(FW_FEATURE_VPHN)) {
> +		__be32 tmp[VPHN_ASSOC_BUFSIZE] = {0};
>   
> -	if (vphn_get_associativity(cpu, associativity))
> +		if (vphn_get_associativity(cpu, tmp))
> +			goto out;
> +
> +		associativity = tmp;
> +
> +	} else {
> +		cpunode = of_get_cpu_node(cpu, NULL);
> +		if (!cpunode)
> +			goto out;
> +
> +		associativity = of_get_associativity(cpunode);
> +	}
> +	if (!associativity)
>   		goto out;
>   
>   	index = of_read_number(associativity, 1);
>   	if (index > primary_domain_index + 1)
> -		return of_read_number(&associativity[index - 1], 1);
> +		coregroup_id = of_read_number(&associativity[index - 1], 1);
>   
>   out:
> -	return cpu_to_core_id(cpu);
> -}
> +	if (cpunode)
> +		of_node_put(cpunode);
>   
> +	return coregroup_id;
> +}
Srikar Dronamraju Sept. 4, 2026, 11:34 a.m. UTC | #2
* Shrikanth Hegde <sshegde@linux.ibm.com> [2026-09-04 15:34:50]:

Thanks Shrikanth for taking a look.

> 
> 
> On 9/2/26 6:05 PM, Srikar Dronamraju wrote:
> > Coregroup support on powerpc has so far been limited to PowerVM LPARs.
> > However, PowerNV can also support coregroups when firmware exposes the
> > required coregroup information through the associativity hierarchy.
> > 
> 
> Existing firmware does expose this info already?

The corresponding skiboot changes were sent to the skiboot mailing list.

> 
> > Detect coregroup support by checking whether primary_domain_index is the
> > penultimate domain in the CPU node's ibm,associativity property. On
> > PowerNV, a non-penultimate primary_domain_index indicates that firmware
> > provides an additional level for coregroup information.
> > 
> > This keeps the logic compatible with PowerVM systems, where
> > primary_domain_index is likewise not the penultimate associativity
> > domain.
> > 
> 
> Could you please put the ibm,associativity on this powernv? as well
> PowerVM's so that one understands it better?

With this patch and the skiboot change, the ibm,associativity will look
similar.

For example on a Power10 Baremetal box with skiboot changes.

$ lsprop /proc/device-tree/cpus/PowerPC,POWER10@*/ibm,associativity |& head -n 20
PowerPC,POWER10@0/ibm,associativity
		 00000005 00000000 00000000 00000000 00000000 00000000
PowerPC,POWER10@100/ibm,associativity
		 00000005 00000000 00000001 00000001 00000002 00000000
PowerPC,POWER10@108/ibm,associativity
		 00000005 00000000 00000001 00000001 00000002 00000002
PowerPC,POWER10@10/ibm,associativity
		 00000005 00000000 00000000 00000000 00000001 00000004
PowerPC,POWER10@110/ibm,associativity
		 00000005 00000000 00000001 00000001 00000003 00000004
PowerPC,POWER10@118/ibm,associativity
		 00000005 00000000 00000001 00000001 00000003 00000006
PowerPC,POWER10@120/ibm,associativity
		 00000005 00000000 00000001 00000001 00000002 00000008
PowerPC,POWER10@128/ibm,associativity
		 00000005 00000000 00000001 00000001 00000002 0000000a
PowerPC,POWER10@130/ibm,associativity
		 00000005 00000000 00000001 00000001 00000003 0000000c
PowerPC,POWER10@138/ibm,associativity
		 00000005 00000000 00000001 00000001 00000003 0000000e

> 
> > Signed-off-by: Srikar Dronamraju <srikar@linux.ibm.com>
> > ---
> > Changelog from v2:
> > - Handle comments from Ritesh (one hunk needed to be moved from patch 2 to
> >    patch 3 to build correctly)
> > 
> > Changelog from v1: https://lkml.kernel.org/r/20260524010017.140408-1-srikar@linux.ibm.com
> > - Handle comments from Christophe Leroy; make code more flat
> > 
> >   arch/powerpc/mm/numa.c | 58 ++++++++++++++++++++++++++++++++++--------
> >   1 file changed, 48 insertions(+), 10 deletions(-)
diff mbox series

Patch

diff --git a/arch/powerpc/mm/numa.c b/arch/powerpc/mm/numa.c
index 5f326b005a2a..e97b624203ea 100644
--- a/arch/powerpc/mm/numa.c
+++ b/arch/powerpc/mm/numa.c
@@ -889,12 +889,32 @@  static int __init numa_setup_drmem_lmb(struct drmem_lmb *lmb,
 	return 0;
 }
 
+/*
+ * If hierarchy extends beyond primary_domain_index + 1, then next
+ * level corresponds to coregroup.
+ */
+static int detect_and_enable_coregroup(const __be32 *associativity, int index)
+{
+	if (!associativity || index == -1)
+		goto out;
+
+	index = of_read_number(associativity, 1);
+
+	if (index > primary_domain_index + 1) {
+		coregroup_enabled = 1;
+		return index;
+	}
+out:
+	coregroup_enabled = 0;
+	return -1;
+}
+
 static int __init parse_numa_properties(void)
 {
 	struct device_node *memory, *pci;
-	int default_nid = 0;
-	unsigned long i;
+	int default_nid = 0, index = 0;
 	const __be32 *associativity;
+	unsigned long i;
 
 	if (numa_enabled == 0) {
 		pr_warn("disabled by user\n");
@@ -927,7 +947,6 @@  static int __init parse_numa_properties(void)
 	 */
 	for_each_present_cpu(i) {
 		__be32 vphn_assoc[VPHN_ASSOC_BUFSIZE];
-		struct device_node *cpu;
 		int nid = NUMA_NO_NODE;
 
 		memset(vphn_assoc, 0, VPHN_ASSOC_BUFSIZE * sizeof(__be32));
@@ -935,7 +954,9 @@  static int __init parse_numa_properties(void)
 		if (__vphn_get_associativity(i, vphn_assoc) == 0) {
 			nid = associativity_to_nid(vphn_assoc);
 			initialize_form1_numa_distance(vphn_assoc);
+			index = detect_and_enable_coregroup(vphn_assoc, index);
 		} else {
+			struct device_node *cpu;
 
 			/*
 			 * Don't fall back to default_nid yet -- we will plug
@@ -948,6 +969,7 @@  static int __init parse_numa_properties(void)
 			associativity = of_get_associativity(cpu);
 			if (associativity) {
 				nid = associativity_to_nid(associativity);
+				index = detect_and_enable_coregroup(associativity, index);
 				initialize_form1_numa_distance(associativity);
 			}
 			of_node_put(cpu);
@@ -1445,7 +1467,9 @@  static long vphn_get_associativity(unsigned long cpu,
 
 int cpu_to_coregroup_id(int cpu)
 {
-	__be32 associativity[VPHN_ASSOC_BUFSIZE] = {0};
+	int coregroup_id = cpu_to_core_id(cpu);
+	struct device_node *cpunode = NULL;
+	const __be32 *associativity;
 	int index;
 
 	if (cpu < 0 || cpu > nr_cpu_ids)
@@ -1454,17 +1478,31 @@  int cpu_to_coregroup_id(int cpu)
 	if (!coregroup_enabled)
 		goto out;
 
-	if (!firmware_has_feature(FW_FEATURE_VPHN))
-		goto out;
+	if (firmware_has_feature(FW_FEATURE_VPHN)) {
+		__be32 tmp[VPHN_ASSOC_BUFSIZE] = {0};
 
-	if (vphn_get_associativity(cpu, associativity))
+		if (vphn_get_associativity(cpu, tmp))
+			goto out;
+
+		associativity = tmp;
+
+	} else {
+		cpunode = of_get_cpu_node(cpu, NULL);
+		if (!cpunode)
+			goto out;
+
+		associativity = of_get_associativity(cpunode);
+	}
+	if (!associativity)
 		goto out;
 
 	index = of_read_number(associativity, 1);
 	if (index > primary_domain_index + 1)
-		return of_read_number(&associativity[index - 1], 1);
+		coregroup_id = of_read_number(&associativity[index - 1], 1);
 
 out:
-	return cpu_to_core_id(cpu);
-}
+	if (cpunode)
+		of_node_put(cpunode);
 
+	return coregroup_id;
+}