- Resolves: RHEL-184814, Dynamically removing cores from an LPAR is not taking care of NUMA topology

- Resolves: RHEL-180740, DLPAR Memory remove/add operations fail with SAP HANA DB loaded
- Resolves: RHEL-212027, Continuous migration failure error messages observed while running memhotplug
This commit is contained in:
Than Ngo 2026-08-03 20:00:27 +02:00
parent 95a2148f78
commit ffdbafc5d8
12 changed files with 1883 additions and 3 deletions

View File

@ -0,0 +1,163 @@
commit 54cf30c7d274c8aab2a7ae589ab056f52dfffc62
Author: Aboorva Devarajan <aboorvad@linux.ibm.com>
Date: Sat Dec 7 21:54:44 2024 -0500
cpu_info_helpers: Add helper function to retrieve present CPU core list
Introduce get_present_core_list helper function to accurately parse
and retrieve the list of present CPU cores, addressing gaps in core
numbering caused by dynamic addition or removal of CPUs (via CPU DLPAR
operation)
Utilizes the present CPU list from `sys/devices/system/cpu/present`
to handle non-contiguous CPU IDs. Accurately maps core IDs to CPUs
considering specified number of threads per CPU, addressing gaps in
core numbering.
Signed-off-by: Aboorva Devarajan <aboorvad@linux.ibm.com>
diff --git a/src/common/cpu_info_helpers.c b/src/common/cpu_info_helpers.c
index 8c57db8..756e792 100644
--- a/src/common/cpu_info_helpers.c
+++ b/src/common/cpu_info_helpers.c
@@ -203,6 +203,113 @@ int __get_one_smt_state(int core, int threads_per_cpu)
return smt_state;
}
+int get_present_cpu_count(void)
+{
+ int start, end, total_cpus = 0;
+ size_t len = 0;
+ char *line = NULL;
+ FILE *fp;
+ char *token;
+
+ fp = fopen(CPU_PRESENT_PATH, "r");
+ if (!fp) {
+ perror("Error opening CPU_PRESENT_PATH");
+ return -1;
+ }
+
+ if (getline(&line, &len, fp) == -1) {
+ perror("Error reading CPU_PRESENT_PATH");
+ fclose(fp);
+ free(line);
+ return -1;
+ }
+ fclose(fp);
+
+ token = strtok(line, ",");
+ while (token) {
+ if (sscanf(token, "%d-%d", &start, &end) == 2) {
+ total_cpus += (end - start + 1);
+ } else if (sscanf(token, "%d", &start) == 1) {
+ total_cpus++;
+ }
+ token = strtok(NULL, ",");
+ }
+
+ free(line);
+ return total_cpus;
+}
+
+int get_present_core_list(int **present_cores, int *num_present_cores, int threads_per_cpu)
+{
+ FILE *fp = NULL;
+ char *line = NULL;
+ char *token = NULL;
+ size_t len = 0;
+ ssize_t read;
+ int core_count = 0;
+ int core_list_size;
+ int *cores = NULL;
+ int start, end, i;
+
+ if (threads_per_cpu <= 0) {
+ fprintf(stderr, "Invalid threads_per_cpu value, got %d expected >= 1\n", threads_per_cpu);
+ return -1;
+ }
+
+ core_list_size = get_present_cpu_count() / threads_per_cpu;
+ if (core_list_size <= 0) {
+ fprintf(stderr, "Error while calculating core list size\n");
+ return -1;
+ }
+
+ cores = malloc(core_list_size * sizeof(int));
+ if (!cores) {
+ perror("Memory allocation failed");
+ goto cleanup;
+ }
+
+ fp = fopen(CPU_PRESENT_PATH, "r");
+ if (!fp) {
+ perror("Error opening file");
+ goto cleanup;
+ }
+
+ read = getline(&line, &len, fp);
+ if (read == -1) {
+ perror("Error reading file");
+ goto cleanup;
+ }
+
+ token = strtok(line, ",");
+ while (token) {
+ if (sscanf(token, "%d-%d", &start, &end) == 2) {
+ for (i = start; i <= end; i++) {
+ if (i % threads_per_cpu == 0) {
+ cores[core_count++] = i / threads_per_cpu;
+ }
+ }
+ } else if (sscanf(token, "%d", &start) == 1) {
+ if (start % threads_per_cpu == 0) {
+ cores[core_count++] = start / threads_per_cpu;
+ }
+ }
+ token = strtok(NULL, ",");
+ }
+
+ *present_cores = cores;
+ *num_present_cores = core_count;
+ free(line);
+ return 0;
+
+cleanup:
+ if (fp) {
+ fclose(fp);
+ }
+ free(line);
+ free(cores);
+ return -1;
+}
+
static void print_cpu_list(const cpu_set_t *cpuset, int cpuset_size,
int cpus_in_system)
{
diff --git a/src/common/cpu_info_helpers.h b/src/common/cpu_info_helpers.h
index c063fff..77e6ad7 100644
--- a/src/common/cpu_info_helpers.h
+++ b/src/common/cpu_info_helpers.h
@@ -24,9 +24,10 @@
#ifndef _CPU_INFO_HELPERS_H
#define _CPU_INFO_HELPERS_H
-#define SYSFS_CPUDIR "/sys/devices/system/cpu/cpu%d"
-#define SYSFS_SUBCORES "/sys/devices/system/cpu/subcores_per_core"
-#define INTSERV_PATH "/proc/device-tree/cpus/%s/ibm,ppc-interrupt-server#s"
+#define SYSFS_CPUDIR "/sys/devices/system/cpu/cpu%d"
+#define SYSFS_SUBCORES "/sys/devices/system/cpu/subcores_per_core"
+#define INTSERV_PATH "/proc/device-tree/cpus/%s/ibm,ppc-interrupt-server#s"
+#define CPU_PRESENT_PATH "/sys/devices/system/cpu/present"
#define SYSFS_PATH_MAX 128
@@ -39,6 +40,8 @@ extern int num_subcores(void);
extern int get_attribute(char *path, const char *fmt, int *value);
extern int get_cpu_info(int *threads_per_cpu, int *cpus_in_system,
int *threads_in_system);
+extern int get_present_core_list(int **present_cores, int *num_present_cores,
+ int threads_per_cpu);
extern int __is_smt_capable(int threads_in_system);
extern int __get_one_smt_state(int core, int threads_per_cpu);
extern int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu,

View File

@ -0,0 +1,195 @@
commit 908986e43b6c67288aa084fe13155964caa807a8
Author: Haren Myneni <haren@linux.ibm.com>
Date: Sat May 16 11:34:22 2026 -0700
drmgr: Add NUMA based CPU removal
The current CPU removal process reads CPU list from last CPU and
remove CPUs based on the userspace request. This process can
result in CPU less NUMA nodes even though these nodes have more
memory which can affect on the system performance.
This patch adds NUMA aware CPU removal process to remove CPUs from
specific NUMA nodes and maintains NUMA balance. The selection of
node from which the CPU to be removed is based on the available
memory per CPU in that node called node ratio. So CPU is selected
from the node which has lower ratio.
If the NUMA topology can't be read, fallback using the current
process.
The node selection process is as follows:
- For each CPU removal request, update node ratios and sort the list.
- Select the next removable CPU from the dr_info CPU list and it
should belong to the first node.
- CPU associated to memory less nodes is considered first and then
the first node that has memory in the list.
- Repeat all CPUs in dr_info list until the next removable CPU is
matched with node CPU bitmap.
- The total number of CPU threads in the selected node is
decremented and cleared in the node CPU bitmap.
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/drslot_chrp_cpu.c b/src/drmgr/drslot_chrp_cpu.c
index e367634..60956a0 100644
--- a/src/drmgr/drslot_chrp_cpu.c
+++ b/src/drmgr/drslot_chrp_cpu.c
@@ -164,6 +164,141 @@ static struct dr_node *get_available_cpu_by_index(struct dr_info *dr_info)
return cpu;
}
+/*
+ * Return node if CPU ID matches in node CPU bitmap.
+ */
+static struct ppcnuma_node *match_cpu_node(struct ppcnuma_node *node,
+ struct dr_node *cpu)
+{
+ int nid;
+
+ if (cpu->cpu_threads) {
+ nid = numa_node_of_cpu(cpu->cpu_threads->id);
+ if (nid == node->node_id) {
+ if (numa_bitmask_isbitset(node->cpus,
+ cpu->cpu_threads->id))
+ return node;
+ }
+ }
+
+ return NULL;
+}
+
+/*
+ * Return node if CPU belongs to any memoryless NUMA node.
+ */
+static struct ppcnuma_node *find_cpu_memless_node(struct dr_node *cpu)
+{
+ struct ppcnuma_node *node = NULL;
+ int nid;
+
+ ppcnuma_foreach_node(&numa, nid, node) {
+ if (node->n_lmbs)
+ continue;
+
+ if (match_cpu_node(node, cpu))
+ return node;
+ }
+
+ return NULL;
+}
+
+/*
+ * The node list is sorted by node ratio (less memory per CPU).
+ * So consider the first node
+ * Return node if CPU belongs to the first NUMA node which
+ * has memory.
+ */
+static struct ppcnuma_node *find_cpu_numa_node(struct dr_node *cpu)
+{
+ struct ppcnuma_node *node = NULL;
+ int found = 0;
+
+ ppcnuma_foreach_node_by_ratio(&numa, node) {
+ if (node->n_cpus && node->n_lmbs) {
+ found = 1;
+ break;
+ }
+ }
+
+ if (found && match_cpu_node(node, cpu))
+ return node;
+
+ return NULL;
+}
+
+/*
+ * Calculate node ratio based on amount of memory per CPU and sort
+ * the node ratio list.
+ */
+static void cpu_update_node_ratio(void)
+{
+ struct ppcnuma_node *node;
+ int nid;
+
+ ppcnuma_foreach_node(&numa, nid, node) {
+ if (!node->n_lmbs || !node->n_cpus)
+ continue;
+
+ /*
+ * Node ratio = n_lmbs per CPU
+ */
+ node->ratio = (node->n_lmbs * 100) / node->n_cpus;
+ }
+
+ order_numa_node_ratio_list();
+}
+
+/*
+ * Scan CPUs from the last one in the list and select the first CPU
+ * based on:
+ * - CPU from memory less node
+ * - If no CPUs are available in memory less nodes, CPU belongs to
+ * the first node from node ratio list.
+ */
+static struct dr_node *numa_get_next_cpu(struct dr_info *dr_info)
+{
+ struct ppcnuma_node *node;
+ struct dr_node *cpu = NULL;
+ struct thread *t;
+ int i, found = 0;
+
+ /*
+ * Update node ratio for each CPU removal request
+ */
+ cpu_update_node_ratio();
+
+ /* Find the first cpu with an online thread */
+ for (cpu = dr_info->all_cpus; cpu; cpu = cpu->next) {
+ if (cpu->unusable)
+ continue;
+
+ if (numa.memless_cpu_count)
+ node = find_cpu_memless_node(cpu);
+ else
+ node = find_cpu_numa_node(cpu);
+
+ if (!node)
+ continue;
+
+ t = cpu->cpu_threads;
+ for (i = 0; i < cpu->cpu_nthreads && t; i++, t = t->next) {
+ if (get_thread_state(t) == ONLINE)
+ found = 1;
+ numa_bitmask_clearbit(node->cpus, t->id);
+ }
+ if (found) {
+ node->n_cpus -= cpu->cpu_nthreads;
+ numa.cpu_count -= cpu->cpu_nthreads;
+ if (!node->n_lmbs)
+ numa.memless_cpu_count -= cpu->cpu_nthreads;
+ return cpu;
+ }
+ }
+
+ return NULL;
+}
+
/*
* Scan all CPUs from the last one for the next available CPU.
* Used only for non-NUMA based CPU removal.
@@ -202,8 +337,12 @@ static struct dr_node *get_next_available_cpu(struct dr_info *dr_info)
cpu = survivor;
} else if (usr_action == REMOVE) {
- /* Find the first cpu with an online thread */
- cpu = get_next_cpu(dr_info);
+ if (numa_enabled)
+ /* Find the first CPU from NUMA nodes */
+ cpu = numa_get_next_cpu(dr_info);
+ else
+ /* Find the first cpu with an online thread */
+ cpu = get_next_cpu(dr_info);
}
if (!cpu)

View File

@ -0,0 +1,117 @@
commit 4cf04b9e75db0da2c199fa12003433b3d0427fd0
Author: Haren Myneni <haren@linux.ibm.com>
Date: Sat May 16 11:34:21 2026 -0700
drmgr: Add NUMA configuration update for CPU remove
This patch adds NUMA node config update for CPU removal. Updates
number of LMBs and CPUs for each node and also calculates total
number of CPUs from memory less nodes. This node configuration is
used to identify the node based on the node ratio from which the
CPU is selected to remove
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Reviewed-by: Dave Marquardt <davemarq@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/common_numa.h b/src/drmgr/common_numa.h
index 7aea026..edf4349 100644
--- a/src/drmgr/common_numa.h
+++ b/src/drmgr/common_numa.h
@@ -38,6 +38,7 @@ struct ppcnuma_topology {
unsigned int lmb_count;
unsigned int cpuless_node_count;
unsigned int cpuless_lmb_count;
+ unsigned int memless_cpu_count;
unsigned int node_count, node_min, node_max;
struct ppcnuma_node *nodes[MAX_NUMNODES];
struct ppcnuma_node *ratio;
diff --git a/src/drmgr/drslot_chrp_cpu.c b/src/drmgr/drslot_chrp_cpu.c
index 6a21663..e367634 100644
--- a/src/drmgr/drslot_chrp_cpu.c
+++ b/src/drmgr/drslot_chrp_cpu.c
@@ -26,10 +26,14 @@
#include <sys/types.h>
#include <dirent.h>
#include <librtas.h>
+#include <numa.h>
#include "dr.h"
#include "drcpu.h"
#include "drpci.h"
#include "ofdt.h"
+#include "common_numa.h"
+
+#define DEFAULT_LMB_SIZE 0x10000000 /* 256MB */
struct cpu_operation;
typedef int (cpu_op_func_t) (void);
@@ -395,6 +399,43 @@ static int smt_threads_func(struct dr_info *dr_info)
return rc;
}
+/*
+ * Per node CPUs are defined as part of build_numa_topology().
+ * This function calculates number of LMBs per node based on
+ * node mememory / lmb-size.
+ * n_cpus and n_lmbs are used to determine node ratio.
+ */
+static int cpu_update_numa_config(void)
+{
+ struct ppcnuma_node *node;
+ unsigned long long node_size;
+ int rc, nid;
+ uint64_t lmb_sz;
+
+ rc = get_dynamic_lmb_size(&lmb_sz);
+ /*
+ * Use the default value if lmb-size property is not available.
+ * For CPU removal, node ratio will be calculated based on
+ * total n_lmbs per CPU.
+ */
+ if (rc)
+ lmb_sz = DEFAULT_LMB_SIZE;
+
+ ppcnuma_foreach_node(&numa, nid, node) {
+ node_size = numa_node_size(nid, 0);
+ /*
+ * Node has memory
+ * n_lmbs = Total memory / lmb-size
+ */
+ if (node_size) {
+ node->n_lmbs = node_size / lmb_sz;
+ } else
+ numa.memless_cpu_count += node->n_cpus;
+ }
+
+ return 0;
+}
+
int valid_cpu_options(void)
{
/* default to a quantity of 1 */
@@ -442,6 +483,15 @@ int drslot_chrp_cpu(void)
return -1;
}
+ /*
+ * Maintain NUMA aware hotplug only for remove and with count request.
+ */
+ if (usr_drc_count && (usr_action == REMOVE)) {
+ build_numa_topology();
+ if (numa_enabled)
+ cpu_update_numa_config();
+ }
+
/* If a user specifies a drc name, the quantity to add/remove is
* one. Enforce that here so the loops in add/remove code behave
* accordingly.
@@ -473,6 +523,9 @@ int drslot_chrp_cpu(void)
if (usr_action == ADD || usr_action == REMOVE)
run_hooks(DRC_TYPE_CPU, usr_action, HOOK_POST, count);
+ if ((usr_action == REMOVE) && numa_enabled)
+ free_numa_topology();
+
free_cpu_drc_info(&dr_info);
return rc;
}

View File

@ -0,0 +1,70 @@
commit fc252eaf3f1e2b6dcf2a646459ac7399db0ab18d
Author: Haren Myneni <haren@linux.ibm.com>
Date: Sat May 16 11:34:19 2026 -0700
drmgr: Add get_next_cpu() to identify the removable CPU
Move code which identifies the removable CPU to get_next_cpu().
This function is used only for the current non-numa based CPU
removal but helps to add for NUMA based CPU removal code in later
patch.
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/drslot_chrp_cpu.c b/src/drmgr/drslot_chrp_cpu.c
index 3ef24f4..6a21663 100644
--- a/src/drmgr/drslot_chrp_cpu.c
+++ b/src/drmgr/drslot_chrp_cpu.c
@@ -160,11 +160,33 @@ static struct dr_node *get_available_cpu_by_index(struct dr_info *dr_info)
return cpu;
}
+/*
+ * Scan all CPUs from the last one for the next available CPU.
+ * Used only for non-NUMA based CPU removal.
+ */
+static struct dr_node *get_next_cpu(struct dr_info *dr_info)
+{
+ struct dr_node *cpu = NULL;
+ struct thread *t;
+
+ /* Find the first cpu with an online thread */
+ for (cpu = dr_info->all_cpus; cpu; cpu = cpu->next) {
+ if (cpu->unusable)
+ continue;
+
+ for (t = cpu->cpu_threads; t; t = t->next) {
+ if (get_thread_state(t) == ONLINE)
+ return cpu;
+ }
+ }
+
+ return NULL;
+}
+
static struct dr_node *get_next_available_cpu(struct dr_info *dr_info)
{
struct dr_node *cpu = NULL;
struct dr_node *survivor = NULL;
- struct thread *t;
if (usr_action == ADD) {
for (cpu = dr_info->all_cpus; cpu; cpu = cpu->next) {
@@ -177,15 +199,7 @@ static struct dr_node *get_next_available_cpu(struct dr_info *dr_info)
cpu = survivor;
} else if (usr_action == REMOVE) {
/* Find the first cpu with an online thread */
- for (cpu = dr_info->all_cpus; cpu; cpu = cpu->next) {
- if (cpu->unusable)
- continue;
-
- for (t = cpu->cpu_threads; t; t = t->next) {
- if (get_thread_state(t) == ONLINE)
- return cpu;
- }
- }
+ cpu = get_next_cpu(dr_info);
}
if (!cpu)

View File

@ -0,0 +1,106 @@
commit 1dee26c2a60dd9a5264331cc20143a5590289187
Author: Haren Myneni <haren@linux.ibm.com>
Date: Mon May 18 22:37:53 2026 -0700
drmgr: Add timeout signal handling for NUMA memory REMOVE
RMC expects drmgr timeout based on value passed with -w option and
drmgr can check fequently for each LMB removal request and exits
once reached the timeout value. But this check happens only in the
user space and does not consider when executes in the kernel to
remove memory. In the case of LMB removal, the kernel expects to
run longer until all pages in LMB are isolated and can return to
the user space for any pending signals.
This patch enables SIGALRM signal based on the user defined
value which generates signal when the timer expires. It allows
the kernel interface returns in case [age isolation is taking
longer and drmgr timeout.
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c
index fe04ad1..5a08839 100644
--- a/src/drmgr/drslot_chrp_mem.c
+++ b/src/drmgr/drslot_chrp_mem.c
@@ -26,6 +26,9 @@
#include <dirent.h>
#include <inttypes.h>
#include <time.h>
+#include <signal.h>
+#include <unistd.h>
+#include <stdbool.h>
#include <sys/wait.h>
#include <sys/stat.h>
#include "dr.h"
@@ -35,6 +38,7 @@
uint64_t block_sz_bytes = 0;
static char *state_strs[] = {"offline", "online"};
+sig_atomic_t numa_mem_timeout = 0;
static char *usagestr = "-c mem {-a | -r} {-q <quantity> -p {variable_weight | ent_capacity} | {-q <quantity> | -s [<drc_name> | <drc_index>]}}";
@@ -50,6 +54,14 @@ mem_usage(char **pusage)
*pusage = usagestr;
}
+/*
+ * SIGALRM handler for drmgr timeout
+ */
+void mem_timeout_handler(int sig)
+{
+ numa_mem_timeout = 1;
+}
+
/**
* report_resource_count
* @brief Report the number of LMBs that were added or removed.
@@ -1681,13 +1693,45 @@ static void clear_numa_lmb_links(void)
node->lmbs = NULL;
}
+/*
+ * Setup SIGALRM signal based on timeout value passed by the user
+ * (with -w option). In the case of LMB removal, the kernel
+ * interface can run longer until all pages in LMB are isolated
+ * and can return to the user space if any pending signals.
+ * This SIGALRM signal can exit the kernel in case LMB removal
+ * is taking longer than timeout.
+ */
+static int drmem_timer_setup(void)
+{
+ struct sigaction sigact;
+
+ if (!usr_timeout)
+ return 0;
+
+ sigact.sa_handler = mem_timeout_handler;
+ sigemptyset(&sigact.sa_mask);
+ sigact.sa_flags = 0;
+ if (sigaction(SIGALRM, &sigact, NULL))
+ return -1;
+
+ alarm(usr_timeout);
+ return 0;
+}
+
static int numa_based_remove(uint32_t count)
{
struct lmb_list_head *lmb_list;
struct ppcnuma_node *node;
- int nid;
+ int nid, rc = 0;
uint32_t done = 0;
+ /*
+ * Enable alarm signal handler for usr_timeout
+ */
+ rc = drmem_timer_setup();
+ if (rc)
+ return rc;
+
/*
* Read the LMBs
* Link the LMBs to their node

View File

@ -0,0 +1,116 @@
commit 1c4812fb35510c7e02ff27fc39badbe5eef24611
Author: Haren Myneni <haren@linux.ibm.com>
Date: Sat May 16 11:34:20 2026 -0700
drmgr: Allocate CPU bitmap for each NUMA node
The current code allocates one bitmap and uses it to determine
number of valid CPUs for each NUMA node and then frees the bitmap.
The NUMA based CPU removal needs this bitmap per node to determine
the valid CPU in that node. So this patch retains bitmap per node
and frees it after DLPAR memory / CPU removal.
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Reviewed-by: Dave Marquardt <davemarq@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/common_numa.c b/src/drmgr/common_numa.c
index 6bc2ea8..eba6c3d 100644
--- a/src/drmgr/common_numa.c
+++ b/src/drmgr/common_numa.c
@@ -84,9 +84,6 @@ static int read_numa_topology(struct ppcnuma_topology *numa)
rc = 0;
- /* In case of allocation error, the libnuma is calling exit() */
- cpus = numa_allocate_cpumask();
-
for (nid = 0; nid <= max_node; nid++) {
if (!numa_bitmask_isbitset(numa_nodes_ptr, nid))
@@ -98,20 +95,24 @@ static int read_numa_topology(struct ppcnuma_topology *numa)
break;
}
+ /* In case of allocation error, the libnuma is calling exit() */
+ cpus = numa_allocate_cpumask();
+
rc = numa_node_to_cpus(nid, cpus);
- if (rc < 0)
+ if (rc < 0) {
+ numa_bitmask_free(cpus);
break;
+ }
/* Count the CPUs in that node */
for (i = 0; i < cpus->size; i++)
if (numa_bitmask_isbitset(cpus, i))
node->n_cpus++;
+ node->cpus = cpus;
numa->cpu_count += node->n_cpus;
}
- numa_bitmask_free(cpus);
-
if (rc) {
ppcnuma_foreach_node(numa, nid, node)
node->n_cpus = 0;
@@ -160,6 +161,20 @@ void build_numa_topology(void)
numa_enabled = 1;
}
+void free_numa_topology(void)
+{
+ struct ppcnuma_node *node;
+ int nid;
+
+ ppcnuma_foreach_node(&numa, nid, node) {
+ if (node) {
+ if (node->cpus)
+ numa_bitmask_free(node->cpus);
+ free(node);
+ }
+ }
+}
+
void order_numa_node_ratio_list(void)
{
int nid;
diff --git a/src/drmgr/common_numa.h b/src/drmgr/common_numa.h
index 2b0901e..7aea026 100644
--- a/src/drmgr/common_numa.h
+++ b/src/drmgr/common_numa.h
@@ -30,6 +30,7 @@ struct ppcnuma_node {
unsigned int ratio;
struct dr_node *lmbs; /* linked by lmb_numa_next */
struct ppcnuma_node *ratio_next;
+ struct bitmask *cpus;
};
struct ppcnuma_topology {
@@ -48,6 +49,7 @@ extern int numa_enabled;
extern struct ppcnuma_topology numa;
void build_numa_topology(void);
void order_numa_node_ratio_list(void);
+void free_numa_topology(void);
struct ppcnuma_node *ppcnuma_fetch_node(struct ppcnuma_topology *numa,
int node_id);
diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c
index 2d22bff..fe04ad1 100644
--- a/src/drmgr/drslot_chrp_mem.c
+++ b/src/drmgr/drslot_chrp_mem.c
@@ -1731,9 +1731,10 @@ int do_mem_kernel_dlpar(void)
if (usr_action == REMOVE && usr_drc_count && !usr_drc_index) {
build_numa_topology();
if (numa_enabled) {
- if (!numa_based_remove(usr_drc_count))
+ rc = numa_based_remove(usr_drc_count);
+ free_numa_topology();
+ if (!rc)
return 0;
-
/*
* If the NUMA based removal failed, lets try the legacy
* way.

View File

@ -0,0 +1,28 @@
commit 3a7d0b61e7a5f364660007a45cc9dcf1c3cb5ab4
Author: Haren Myneni <haren@linux.ibm.com>
Date: Mon May 18 22:37:52 2026 -0700
drmgr: Allow signals mentioned in new sigset_t
The current code uses sigdelset() to unblock some signals such as
SIGBUS, SIGALRM, and etc. But sigprocmask() with SIG_BLOCK sets
the union of this new set and the current sigset which ends up
blocking some of them (Ex: SIGALRM). Insted of using SIG_BLOCK,
SIG_SETMASK allows to use the complete new sigset_t.
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/common.c b/src/drmgr/common.c
index 70f4dfd..68041a9 100644
--- a/src/drmgr/common.c
+++ b/src/drmgr/common.c
@@ -939,7 +939,7 @@ sig_setup(void)
sigdelset(&sigset, SIGABRT);
/* Now block all remaining signals */
- rc = sigprocmask(SIG_BLOCK, &sigset, NULL);
+ rc = sigprocmask(SIG_SETMASK, &sigset, NULL);
if (rc)
return -1;

View File

@ -0,0 +1,87 @@
commit 7926241f576de24df8b342858f5278911e035658
Author: Haren Myneni <haren@linux.ibm.com>
Date: Mon May 18 22:37:54 2026 -0700
drmgr: Do not remove LMBs when the timer expires
The current code selects LMBs from all NUMA node based on node
ratio and removes them until reaches the the requested limit. This
patch stops removing LMBs when receives SIGALRM signal based on
the user defined timeout.
Also reports total numaber of LMBs removed with DR_TOTAL_RESOURCES
to RMC even for partial memory removal or with an error.
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c
index 5a08839..4a3fe3a 100644
--- a/src/drmgr/drslot_chrp_mem.c
+++ b/src/drmgr/drslot_chrp_mem.c
@@ -1478,6 +1478,9 @@ static int remove_lmb_from_node(struct ppcnuma_node *node, uint32_t count)
count, node->n_lmbs, node->node_id);
for (lmb = node->lmbs; lmb && done < count; lmb = lmb->lmb_numa_next) {
+ if (numa_mem_timeout)
+ break;
+
unlinked++;
err = remove_lmb_by_index(lmb->drc_index);
if (err)
@@ -1563,6 +1566,9 @@ static int remove_cpuless_lmbs(uint32_t count)
this_loop = 0;
ppcnuma_foreach_node(&numa, nid, node) {
+ if (numa_mem_timeout)
+ break;
+
if (!node->n_lmbs || node->n_cpus)
continue;
@@ -1656,6 +1662,9 @@ static int remove_cpu_lmbs(uint32_t count)
this_loop = 0;
ppcnuma_foreach_node_by_ratio(&numa, node) {
+ if (numa_mem_timeout)
+ break;
+
if (!node->n_lmbs || !node->n_cpus)
continue;
@@ -1739,14 +1748,13 @@ static int numa_based_remove(uint32_t count)
*/
lmb_list = get_lmbs(LMB_NORMAL_SORT);
if (lmb_list == NULL) {
- clear_numa_lmb_links();
- return -1;
+ rc = -1;
+ goto out_clear;
}
if (!numa.node_count) {
- clear_numa_lmb_links();
- free_lmbs(lmb_list);
- return -EINVAL;
+ rc = -EINVAL;
+ goto out_free;
}
ppcnuma_foreach_node(&numa, nid, node) {
@@ -1759,11 +1767,12 @@ static int numa_based_remove(uint32_t count)
done += remove_cpu_lmbs(count);
- report_resource_count(done);
-
- clear_numa_lmb_links();
+out_free:
free_lmbs(lmb_list);
- return 0;
+out_clear:
+ clear_numa_lmb_links();
+ report_resource_count(done);
+ return rc;
}
int do_mem_kernel_dlpar(void)

View File

@ -0,0 +1,150 @@
commit 27dd4cce01450d94ab58fa2e632eac34307a4a82
Author: Haren Myneni <haren@linux.ibm.com>
Date: Sat May 16 11:34:17 2026 -0700
drmgr: Move numa_topology code to common_numa.c
Move build_numa_topology and sort NUMA node ratio list code to
common_numa.c. These functions will also be used for NUMA aware
CPU removal in later patch.
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/common_numa.c b/src/drmgr/common_numa.c
index 898aab6..6bc2ea8 100644
--- a/src/drmgr/common_numa.c
+++ b/src/drmgr/common_numa.c
@@ -27,6 +27,9 @@
#include "drmem.h" /* for DYNAMIC_RECONFIG_MEM */
#include "common_numa.h"
+int numa_enabled = 0;
+struct ppcnuma_topology numa;
+
struct ppcnuma_node *ppcnuma_fetch_node(struct ppcnuma_topology *numa, int nid)
{
struct ppcnuma_node *node;
@@ -118,7 +121,7 @@ static int read_numa_topology(struct ppcnuma_topology *numa)
return rc;
}
-int ppcnuma_get_topology(struct ppcnuma_topology *numa)
+static int ppcnuma_get_topology(struct ppcnuma_topology *numa)
{
int rc;
@@ -145,3 +148,35 @@ int ppcnuma_get_topology(struct ppcnuma_topology *numa)
return 0;
}
+
+void build_numa_topology(void)
+{
+ int rc;
+
+ rc = ppcnuma_get_topology(&numa);
+ if (rc)
+ return;
+
+ numa_enabled = 1;
+}
+
+void order_numa_node_ratio_list(void)
+{
+ int nid;
+ struct ppcnuma_node *node, *n, **p;
+
+ numa.ratio = NULL;
+
+ /* Create an ordered link of the nodes */
+ ppcnuma_foreach_node(&numa, nid, node) {
+ if (!node->n_lmbs || !node->n_cpus)
+ continue;
+
+ p = &numa.ratio;
+ for (n = numa.ratio;
+ n && n->ratio < node->ratio; n = n->ratio_next)
+ p = &n->ratio_next;
+ *p = node;
+ node->ratio_next = n;
+ }
+}
diff --git a/src/drmgr/common_numa.h b/src/drmgr/common_numa.h
index c209a3e..2b0901e 100644
--- a/src/drmgr/common_numa.h
+++ b/src/drmgr/common_numa.h
@@ -44,7 +44,11 @@ struct ppcnuma_topology {
struct assoc_arrays aa;
};
-int ppcnuma_get_topology(struct ppcnuma_topology *numa);
+extern int numa_enabled;
+extern struct ppcnuma_topology numa;
+void build_numa_topology(void);
+void order_numa_node_ratio_list(void);
+
struct ppcnuma_node *ppcnuma_fetch_node(struct ppcnuma_topology *numa,
int node_id);
diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c
index 4a36c73..eb75ccf 100644
--- a/src/drmgr/drslot_chrp_mem.c
+++ b/src/drmgr/drslot_chrp_mem.c
@@ -38,9 +38,6 @@ static char *state_strs[] = {"offline", "online"};
static char *usagestr = "-c mem {-a | -r} {-q <quantity> -p {variable_weight | ent_capacity} | {-q <quantity> | -s [<drc_name> | <drc_index>]}}";
-static struct ppcnuma_topology numa;
-static int numa_enabled = 0;
-
/**
* mem_usage
* @brief return usage string
@@ -1605,7 +1602,7 @@ static int remove_cpuless_lmbs(uint32_t count)
static void update_node_ratio(void)
{
int nid;
- struct ppcnuma_node *node, *n, **p;
+ struct ppcnuma_node *node;
uint32_t cpu_ratio, mem_ratio;
/*
@@ -1626,18 +1623,7 @@ static void update_node_ratio(void)
node->ratio = (cpu_ratio * 9 + mem_ratio) / 10;
}
- /* Create an ordered link of the nodes */
- ppcnuma_foreach_node(&numa, nid, node) {
- if (!node->n_lmbs || !node->n_cpus)
- continue;
-
- p = &numa.ratio;
- for (n = numa.ratio;
- n && n->ratio < node->ratio; n = n->ratio_next)
- p = &n->ratio_next;
- *p = node;
- node->ratio_next = n;
- }
+ order_numa_node_ratio_list();
}
/*
@@ -1693,17 +1679,6 @@ static int remove_cpu_lmbs(uint32_t count)
return done;
}
-static void build_numa_topology(void)
-{
- int rc;
-
- rc = ppcnuma_get_topology(&numa);
- if (rc)
- return;
-
- numa_enabled = 1;
-}
-
static void clear_numa_lmb_links(void)
{
int nid;

View File

@ -0,0 +1,81 @@
commit cbdbefdb8c82acae997cf1a30816e69b323f9979
Author: Haren Myneni <haren@linux.ibm.com>
Date: Sat May 16 11:34:18 2026 -0700
drmgr: Move read lmb-size property code to common_ofdt.c
get_dynamic_lmb_size() is used to get lmb size from "ibm,lmb-size"
property. This lmb size is needed to determine the number of LMBs
and used to find NUMA node ratio for NUMA aware memory and CPU
removal code. So move this function to common_ofdt.c
Signed-off-by: Haren Myneni <haren@linux.ibm.com>
Signed-off-by: Tyrel Datwyler <tyreld@linux.ibm.com>
diff --git a/src/drmgr/common_ofdt.c b/src/drmgr/common_ofdt.c
index 1e5fe53..0655559 100644
--- a/src/drmgr/common_ofdt.c
+++ b/src/drmgr/common_ofdt.c
@@ -28,6 +28,7 @@
#include <errno.h>
#include "dr.h"
#include "ofdt.h"
+#include "drmem.h"
#define RTAS_DIRECTORY "/proc/device-tree/rtas"
#define CHOSEN_DIRECTORY "/proc/device-tree/chosen"
@@ -932,3 +933,19 @@ int of_associativity_to_node(const char *dir, int min_common_depth)
return be32toh(prop[min_common_depth]);
}
+int get_dynamic_lmb_size(uint64_t *lmb_sz)
+{
+ int rc = 0;
+ uint64_t sz;
+
+ rc = get_property(DYNAMIC_RECONFIG_MEM, "ibm,lmb-size",
+ &sz, sizeof(sz));
+ if (rc) {
+ say(DEBUG, "Could not retrieve drconf LMB size\n");
+ return rc;
+ }
+
+ /* convert for LE systems */
+ *lmb_sz = be64toh(sz);
+ return 0;
+}
diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c
index eb75ccf..2d22bff 100644
--- a/src/drmgr/drslot_chrp_mem.c
+++ b/src/drmgr/drslot_chrp_mem.c
@@ -506,16 +506,9 @@ get_dynamic_reconfig_lmbs(struct lmb_list_head *lmb_list)
uint64_t lmb_sz;
int rc = 0;
- rc = get_property(DYNAMIC_RECONFIG_MEM, "ibm,lmb-size",
- &lmb_sz, sizeof(lmb_sz));
-
- /* convert for LE systems */
- lmb_sz = be64toh(lmb_sz);
-
- if (rc) {
- say(DEBUG, "Could not retrieve drconf LMB size\n");
+ rc = get_dynamic_lmb_size(&lmb_sz);
+ if (rc)
return rc;
- }
if (stat(DYNAMIC_RECONFIG_MEM_V1, &sbuf) == 0) {
rc = get_dynamic_reconfig_lmbs_v1(lmb_sz, lmb_list);
diff --git a/src/drmgr/ofdt.h b/src/drmgr/ofdt.h
index e9ebd03..c79ed65 100644
--- a/src/drmgr/ofdt.h
+++ b/src/drmgr/ofdt.h
@@ -185,6 +185,7 @@ int get_assoc_arrays(const char *dir, struct assoc_arrays *aa,
int min_common_depth);
int of_associativity_to_node(const char *dir, int min_common_depth);
int init_node(struct dr_node *);
+int get_dynamic_lmb_size(uint64_t *lmb_sz);
static inline int aa_index_to_node(struct assoc_arrays *aa, uint32_t aa_index)
{

View File

@ -0,0 +1,749 @@
commit e5fd24a6e35c3be78c96d6887e3774852bbe4674
Author: Aboorva Devarajan <aboorvad@linux.ibm.com>
Date: Wed Jan 1 22:56:07 2025 -0500
ppc64_cpu: Fix handling of non-contiguous CPU IDs
In ppc64le environments, adding or removing CPUs dynamically through
DLPAR can create gaps in CPU IDs, such as `0-103,120-151`, in this
case CPUs 104-119 are missing.
ppc64_cpu doesn't handles this scenario and always considers CPU IDs
to be contiguous causing issues in core numbering, cpu info and SMT
mode reporting.
To illustrate the issues this patch fixes, consider the following
system configuration:
$ lscpu
Architecture: ppc64le
Byte Order: Little Endian
CPU(s): 136
On-line CPU(s) list: 0-103,120-151
**Note: CPU IDs are non-contiguous**
-----------------------------------------------------------------
Before Patch:
-----------------------------------------------------------------
$ ppc64_cpu --info
Core 0: 0* 1* 2* 3* 4* 5* 6* 7*
Core 1: 8* 9* 10* 11* 12* 13* 14* 15*
Core 2: 16* 17* 18* 19* 20* 21* 22* 23*
Core 3: 24* 25* 26* 27* 28* 29* 30* 31*
Core 4: 32* 33* 34* 35* 36* 37* 38* 39*
Core 5: 40* 41* 42* 43* 44* 45* 46* 47*
Core 6: 48* 49* 50* 51* 52* 53* 54* 55*
Core 7: 56* 57* 58* 59* 60* 61* 62* 63*
Core 8: 64* 65* 66* 67* 68* 69* 70* 71*
Core 9: 72* 73* 74* 75* 76* 77* 78* 79*
Core 10: 80* 81* 82* 83* 84* 85* 86* 87*
Core 11: 88* 89* 90* 91* 92* 93* 94* 95*
Core 12: 96* 97* 98* 99* 100* 101* 102* 103*
........................................................... *gap*
Core 13: 120* 121* 122* 123* 124* 125* 126* 127*
Core 14: 128* 129* 130* 131* 132* 133* 134* 135*
Core 15: 136* 137* 138* 139* 140* 141* 142* 143*
Core 16: 144* 145* 146* 147* 148* 149* 150* 151*
**Although the CPU IDs are non contiguous, associated core IDs are
represented in contiguous order, which makes it harder to interpret
this clearly.**
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
$ ppc64_cpu --cores-on
Number of cores online = 15
**Expected: Number of online cores = 17**
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
$ ppc64_cpu --offline-cores
Cores offline = 13, 14
**Even though no cores are actually offline, two cores (13, 14)
are displayed as offline.**
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
$ ppc64_cpu --online-cores
Cores online = 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 15, 16
**The list of online cores is missing two cores (13, 14).**
-----------------------------------------------------------------
To resolve this, use the present CPU list from sysfs to assign
numbers to CPUs and cores, which will make this accurate.
$ cat /sys/devices/system/cpu/present
0-103,120-151
With this patch, the command output correctly reflects the
current CPU configuration, providing a more precise representation
of the system state.
-----------------------------------------------------------------
After Patch:
-----------------------------------------------------------------
$ ppc64_cpu --info
Core 0: 0* 1* 2* 3* 4* 5* 6* 7*
Core 1: 8* 9* 10* 11* 12* 13* 14* 15*
Core 2: 16* 17* 18* 19* 20* 21* 22* 23*
Core 3: 24* 25* 26* 27* 28* 29* 30* 31*
Core 4: 32* 33* 34* 35* 36* 37* 38* 39*
Core 5: 40* 41* 42* 43* 44* 45* 46* 47*
Core 6: 48* 49* 50* 51* 52* 53* 54* 55*
Core 7: 56* 57* 58* 59* 60* 61* 62* 63*
Core 8: 64* 65* 66* 67* 68* 69* 70* 71*
Core 9: 72* 73* 74* 75* 76* 77* 78* 79*
Core 10: 80* 81* 82* 83* 84* 85* 86* 87*
Core 11: 88* 89* 90* 91* 92* 93* 94* 95*
Core 12: 96* 97* 98* 99* 100* 101* 102* 103*
........................................................... *gap*
Core 15: 120* 121* 122* 123* 124* 125* 126* 127*
Core 16: 128* 129* 130* 131* 132* 133* 134* 135*
Core 17: 136* 137* 138* 139* 140* 141* 142* 143*
Core 18: 144* 145* 146* 147* 148* 149* 150* 151*
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
$ ppc64_cpu --cores-on
Number of cores online = 17
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
$ ppc64_cpu --offline-cores
Cores offline =
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
$ ppc64_cpu --online-cores
Cores online = 0,1,2,3,4,5,6,7,8,9,10,11,12,15,16,17,18
-----------------------------------------------------------------
Signed-off-by: Aboorva Devarajan <aboorvad@linux.ibm.com>
diff --git a/src/common/cpu_info_helpers.c b/src/common/cpu_info_helpers.c
index 756e792..e75cf6e 100644
--- a/src/common/cpu_info_helpers.c
+++ b/src/common/cpu_info_helpers.c
@@ -311,67 +311,94 @@ cleanup:
}
static void print_cpu_list(const cpu_set_t *cpuset, int cpuset_size,
- int cpus_in_system)
+ int threads_per_cpu)
{
- int core;
+ int *present_cores = NULL;
+ int num_present_cores;
+ int start, end, i = 0;
const char *comma = "";
- for (core = 0; core < cpus_in_system; core++) {
- int begin = core;
- if (CPU_ISSET_S(core, cpuset_size, cpuset)) {
- while (CPU_ISSET_S(core+1, cpuset_size, cpuset))
- core++;
+ if (get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu) != 0) {
+ fprintf(stderr, "Failed to get present_cores list\n");
+ return;
+ }
- if (core > begin)
- printf("%s%d-%d", comma, begin, core);
- else
- printf("%s%d", comma, core);
+ while (i < num_present_cores) {
+ start = present_cores[i];
+ if (CPU_ISSET_S(start, cpuset_size, cpuset)) {
+ end = start;
+ while (i + 1 < num_present_cores &&
+ CPU_ISSET_S(present_cores[i + 1], cpuset_size, cpuset) &&
+ present_cores[i + 1] == end + 1) {
+ end = present_cores[++i];
+ }
+ if (start == end) {
+ printf("%s%d", comma, start);
+ } else {
+ printf("%s%d-%d", comma, start, end);
+ }
comma = ",";
}
+ i++;
}
+ free(present_cores);
}
-int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu,
- bool print_smt_state)
+int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu, bool print_smt_state)
{
- int thread, c, smt_state = 0;
cpu_set_t **cpu_states = NULL;
- int cpu_state_size = CPU_ALLOC_SIZE(cpus_in_system);
- int start_cpu = 0, stop_cpu = cpus_in_system;
+ int thread, smt_state = -1;
+ int cpu_state_size;
int rc = 0;
+ int i, core_id, threads_online;
+ int *present_cores = NULL;
+ int num_present_cores;
- cpu_states = (cpu_set_t **)calloc(threads_per_cpu, sizeof(cpu_set_t));
- if (!cpu_states)
+ if (get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu) != 0) {
+ fprintf(stderr, "Failed to get present core list\n");
return -ENOMEM;
+ }
+ cpu_state_size = CPU_ALLOC_SIZE(num_present_cores);
+ cpu_states = (cpu_set_t **)calloc(threads_per_cpu, sizeof(cpu_set_t *));
+ if (!cpu_states) {
+ rc = -ENOMEM;
+ goto cleanup_present_cores;
+ }
for (thread = 0; thread < threads_per_cpu; thread++) {
- cpu_states[thread] = CPU_ALLOC(cpus_in_system);
+ cpu_states[thread] = CPU_ALLOC(num_present_cores);
+ if (!cpu_states[thread]) {
+ rc = -ENOMEM;
+ goto cleanup_cpu_states;
+ }
CPU_ZERO_S(cpu_state_size, cpu_states[thread]);
}
- for (c = start_cpu; c < stop_cpu; c++) {
- int threads_online = __get_one_smt_state(c, threads_per_cpu);
-
+ for (i = 0; i < num_present_cores; i++) {
+ core_id = present_cores[i];
+ threads_online = __get_one_smt_state(core_id, threads_per_cpu);
if (threads_online < 0) {
rc = threads_online;
- goto cleanup_get_smt;
+ goto cleanup_cpu_states;
+ }
+ if (threads_online) {
+ CPU_SET_S(core_id, cpu_state_size, cpu_states[threads_online - 1]);
}
- if (threads_online)
- CPU_SET_S(c, cpu_state_size,
- cpu_states[threads_online - 1]);
}
for (thread = 0; thread < threads_per_cpu; thread++) {
if (CPU_COUNT_S(cpu_state_size, cpu_states[thread])) {
- if (smt_state == 0)
+ if (smt_state == -1)
smt_state = thread + 1;
else if (smt_state > 0)
smt_state = 0; /* mix of SMT modes */
}
}
- if (!print_smt_state)
- return smt_state;
+ if (!print_smt_state) {
+ rc = smt_state;
+ goto cleanup_cpu_states;
+ }
if (smt_state == 1) {
if (numeric)
@@ -380,11 +407,9 @@ int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu,
printf("SMT is off\n");
} else if (smt_state == 0) {
for (thread = 0; thread < threads_per_cpu; thread++) {
- if (CPU_COUNT_S(cpu_state_size,
- cpu_states[thread])) {
+ if (CPU_COUNT_S(cpu_state_size, cpu_states[thread])) {
printf("SMT=%d: ", thread + 1);
- print_cpu_list(cpu_states[thread],
- cpu_state_size, cpus_in_system);
+ print_cpu_list(cpu_states[thread], cpu_state_size, threads_per_cpu);
printf("\n");
}
}
@@ -392,9 +417,12 @@ int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu,
printf("SMT=%d\n", smt_state);
}
-cleanup_get_smt:
+cleanup_cpu_states:
for (thread = 0; thread < threads_per_cpu; thread++)
CPU_FREE(cpu_states[thread]);
+ free(cpu_states);
+cleanup_present_cores:
+ free(present_cores);
return rc;
}
diff --git a/src/ppc64_cpu.c b/src/ppc64_cpu.c
index 4017240..0233d29 100644
--- a/src/ppc64_cpu.c
+++ b/src/ppc64_cpu.c
@@ -52,7 +52,6 @@
#define DSCR_DEFAULT_PATH "/sys/devices/system/cpu/dscr_default"
-#define MAX_NR_CPUS 1024
#define DIAGNOSTICS_RUN_MODE 42
#define CPU_OFFLINE -1
@@ -266,21 +265,31 @@ static int get_one_smt_state(int core)
static int get_smt_state(void)
{
int smt_state = -1;
- int i;
+ int i, rc;
+ int *present_cores;
+ int num_present_cores;
+
+ rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu);
+ if (rc != 0) {
+ return -1;
+ }
+
+ for (i = 0; i < num_present_cores; i++) {
+ int cpu_state = get_one_smt_state(present_cores[i]);
- for (i = 0; i < cpus_in_system; i++) {
- int cpu_state = get_one_smt_state(i);
if (cpu_state == 0)
continue;
if (smt_state == -1)
smt_state = cpu_state;
+
if (smt_state != cpu_state) {
smt_state = -1;
break;
}
}
+ free(present_cores);
return smt_state;
}
@@ -313,20 +322,36 @@ static int set_smt_state(int smt_state)
{
int i, j, rc = 0;
int error = 0;
+ int cpu_base, cpu_id, core_id;
+ int *present_cores = NULL;
+ int num_present_cores;
if (!sysattr_is_writeable("online")) {
perror("Cannot set smt state");
return -1;
}
- for (i = 0; i < threads_in_system; i += threads_per_cpu) {
+ rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu);
+
+ if (rc != 0) {
+ fprintf(stderr, "Failed to retrieve present core list\n");
+ return rc;
+ }
+
+ for (i = 0; i < num_present_cores; i++) {
+
+ core_id = present_cores[i];
+ cpu_base = core_id * threads_per_cpu;
+
/* Online means any thread on this core running, so check all
* threads in the core, not just the first. */
for (j = 0; j < threads_per_cpu; j++) {
- if (!cpu_online(i + j))
+ cpu_id = cpu_base + j;
+
+ if (!cpu_online(cpu_id))
continue;
- rc = set_one_smt_state(i, smt_state);
+ rc = set_one_smt_state(cpu_base, smt_state);
/* Record an error, but do not check result: if we
* have failed to set this core, keep trying
* subsequent ones. */
@@ -336,10 +361,13 @@ static int set_smt_state(int smt_state)
}
}
+ free(present_cores);
+
if (error) {
- fprintf(stderr, "One or more cpus could not be on/offlined\n");
+ fprintf(stderr, "One or more CPUs could not be on/offlined\n");
return -1;
}
+
return rc;
}
@@ -459,8 +487,8 @@ static int do_subcores_per_core(char *state)
}
printf("Subcores per core: %d\n", subcore_state);
} else {
- /* Kernel decides what values are valid, so no need to
- * check here. */
+ /* Kernel decides what values are valid, so no need to
+ * check here. */
subcore_state = strtol(state, NULL, 0);
rc = set_attribute(SYSFS_SUBCORES, "%d", subcore_state);
if (rc) {
@@ -1038,7 +1066,7 @@ static int set_all_threads_off(int cpu, int smt_state)
snprintf(path, SYSFS_PATH_MAX, SYSFS_CPUDIR"/%s", i, "online");
rc = offline_thread(path);
if (rc == -1)
- printf("Unable to take cpu%d offline", i);
+ printf("Unable to take CPU %d offline\n", i);
}
return rc;
@@ -1065,11 +1093,13 @@ static int set_one_core(int smt_state, int core, int state)
static int do_online_cores(char *cores, int state)
{
int smt_state;
- int *core_state, *desired_core_state;
+ int *core_state = NULL, *desired_core_state = NULL;
int i, rc = 0;
- int core;
+ int core, valid = 0, core_idx = 0;
char *str, *token, *end_token;
bool first_core = true;
+ int *present_cores = NULL;
+ int num_present_cores;
if (cores) {
if (!sysattr_is_writeable("online")) {
@@ -1083,49 +1113,62 @@ static int do_online_cores(char *cores, int state)
}
}
+ rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu);
+ if (rc != 0) {
+ fprintf(stderr, "Failed to retrieve present core list\n");
+ return rc;
+ }
+
smt_state = get_smt_state();
- core_state = calloc(cpus_in_system, sizeof(int));
- if (!core_state)
+ core_state = calloc(num_present_cores, sizeof(int));
+ if (!core_state) {
+ free(present_cores);
return -ENOMEM;
+ }
- for (i = 0; i < cpus_in_system ; i++)
- core_state[i] = (get_one_smt_state(i) > 0);
+ for (i = 0; i < num_present_cores; i++) {
+ core_state[i] = (get_one_smt_state(present_cores[i]) > 0);
+ }
if (!cores) {
printf("Cores %s = ", state == 0 ? "offline" : "online");
- for (i = 0; i < cpus_in_system; i++) {
+ for (i = 0; i < num_present_cores; i++) {
if (core_state[i] == state) {
if (first_core)
first_core = false;
else
printf(",");
- printf("%d", i);
+ printf("%d", present_cores[i]);
}
}
printf("\n");
free(core_state);
+ free(present_cores);
return 0;
}
if (smt_state == -1) {
printf("Bad or inconsistent SMT state: use ppc64_cpu --smt=on|off to set all\n"
- "cores to have the same number of online threads to continue.\n");
+ "cores to have the same number of online threads to continue.\n");
do_info();
+ free(present_cores);
return -1;
}
- desired_core_state = calloc(cpus_in_system, sizeof(int));
+ desired_core_state = calloc(num_present_cores, sizeof(int));
if (!desired_core_state) {
free(core_state);
+ free(present_cores);
return -ENOMEM;
}
- for (i = 0; i < cpus_in_system; i++)
+ for (i = 0; i < num_present_cores; i++) {
/*
* Not specified on command-line
*/
desired_core_state[i] = -1;
+ }
str = cores;
while (1) {
@@ -1141,42 +1184,57 @@ static int do_online_cores(char *cores, int state)
rc = -1;
continue;
}
- if (core >= cpus_in_system || core < 0) {
+
+ for (i = 0; i < num_present_cores; i++) {
+ if (core == present_cores[i]) {
+ valid = 1;
+ core_idx = i;
+ break;
+ }
+ }
+
+ if (!valid) {
printf("Invalid core to %s: %d\n", state == 0 ? "offline" : "online", core);
rc = -1;
continue;
}
- desired_core_state[core] = state;
+
+ desired_core_state[core_idx] = state;
}
if (rc) {
- free(core_state);
- free(desired_core_state);
- return rc;
+ goto cleanup;
}
- for (i = 0; i < cpus_in_system; i++) {
+ for (i = 0; i < num_present_cores; i++) {
if (desired_core_state[i] != -1) {
- rc = set_one_core(smt_state, i, state);
- if (rc)
+ rc = set_one_core(smt_state, present_cores[i], state);
+ if (rc) {
+ fprintf(stderr, "Failed to set core %d to %s\n", present_cores[i], state == 0 ? "offline" : "online");
break;
+ }
}
}
+cleanup:
free(core_state);
free(desired_core_state);
+ free(present_cores);
+
return rc;
}
static int do_cores_on(char *state)
{
int smt_state;
- int *core_state;
- int cores_now_online = 0;
- int i, rc;
+ int cores_now_online = 0, core_id = 0;
+ int i, rc = 0;
int number_to_have, number_to_change = 0, number_changed = 0;
+ int *core_state = NULL;
int new_state;
char *end_state;
+ int *present_cores = NULL;
+ int num_present_cores;
if (state) {
if (!sysattr_is_writeable("online")) {
@@ -1194,24 +1252,33 @@ static int do_cores_on(char *state)
if (!core_state)
return -ENOMEM;
- for (i = 0; i < cpus_in_system ; i++) {
- core_state[i] = (get_one_smt_state(i) > 0);
- if (core_state[i])
+ rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu);
+ if (rc != 0) {
+ fprintf(stderr, "Failed to retrieve present core list\n");
+ free(core_state);
+ return rc;
+ }
+ for (i = 0; i < num_present_cores; i++) {
+ int core = present_cores[i];
+ core_state[i] = (get_one_smt_state(core) > 0);
+ if (core_state[i]) {
cores_now_online++;
+ }
}
if (!state) {
printf("Number of cores online = %d\n", cores_now_online);
- free(core_state);
- return 0;
+ rc = 0;
+ goto cleanup;
}
smt_state = get_smt_state();
if (smt_state == -1) {
printf("Bad or inconsistent SMT state: use ppc64_cpu --smt=on|off to set all\n"
- "cores to have the same number of online threads to continue.\n");
+ "cores to have the same number of online threads to continue.\n");
do_info();
- return -1;
+ rc = -1;
+ goto cleanup;
}
if (!strcmp(state, "all")) {
@@ -1227,15 +1294,16 @@ static int do_cores_on(char *state)
}
if (number_to_have == cores_now_online) {
- free(core_state);
- return 0;
+ rc = 0;
+ goto cleanup;
}
- if (number_to_have > cpus_in_system) {
- printf("Cannot online more cores than are present.\n");
+ if (number_to_have <= 0 || number_to_have > cpus_in_system) {
+ printf("Error: Invalid number of cores requested: %d, possible values \
+ should be in range: (1-%d)\n", number_to_have, cpus_in_system);
do_cores_present();
- free(core_state);
- return -1;
+ rc = -1;
+ goto cleanup;
}
if (number_to_have > cores_now_online) {
@@ -1248,41 +1316,50 @@ static int do_cores_on(char *state)
if (new_state) {
for (i = 0; i < cpus_in_system; i++) {
+ core_id = present_cores[i];
if (!core_state[i]) {
- rc = set_one_core(smt_state, i, new_state);
- if (!rc)
+ rc = set_one_core(smt_state, core_id, new_state);
+ if (!rc) {
number_changed++;
- if (number_changed >= number_to_change)
+ }
+ if (number_changed >= number_to_change) {
break;
+ }
}
}
} else {
- for (i = cpus_in_system - 1; i > 0; i--) {
+ for (i = cpus_in_system - 1; i >= 0; i--) {
+ core_id = present_cores[i];
if (core_state[i]) {
- rc = set_one_core(smt_state, i, new_state);
- if (!rc)
+ rc = set_one_core(smt_state, core_id, new_state);
+ if (!rc) {
number_changed++;
- if (number_changed >= number_to_change)
+ }
+ if (number_changed >= number_to_change) {
break;
+ }
}
}
}
if (number_changed != number_to_change) {
cores_now_online = 0;
- for (i = 0; i < cpus_in_system ; i++) {
- if (cpu_online(i * threads_per_cpu))
+ for (i = 0; i < cpus_in_system; i++) {
+ core_id = present_cores[i];
+ if (cpu_online(core_id * threads_per_cpu)) {
cores_now_online++;
+ }
}
printf("Failed to set requested number of cores online.\n"
- "Requested: %d cores, Onlined: %d cores\n",
- number_to_have, cores_now_online);
- free(core_state);
- return -1;
+ "Requested: %d cores, Onlined: %d cores\n",
+ number_to_have, cores_now_online);
+ rc = -1;
}
+cleanup:
free(core_state);
- return 0;
+ free(present_cores);
+ return rc;
}
static bool core_is_online(int core)
@@ -1294,35 +1371,45 @@ static int do_info(void)
{
int i, j, thread_num;
char online;
- int core, subcores = 0;
+ int subcores = 0, core_id = 0;
+ int *present_cores = NULL;
+ int num_present_cores;
- if (is_subcore_capable())
+ if (is_subcore_capable()) {
subcores = num_subcores();
+ }
- for (i = 0, core = 0; core < cpus_in_system; i++) {
+ int rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu);
+ if (rc != 0) {
+ fprintf(stderr, "Failed to retrieve present core list\n");
+ return rc;
+ }
- if (!core_is_online(i))
+ for (i = 0; i < num_present_cores; i++) {
+ core_id = present_cores[i];
+ if (!core_is_online(core_id)) {
continue;
+ }
if (subcores > 1) {
- if (core % subcores == 0)
- printf("Core %3d:\n", core/subcores);
- printf(" Subcore %3d: ", core);
+ if (core_id % subcores == 0) {
+ printf("Core %3d:\n", core_id / subcores);
+ }
+ printf(" Subcore %3d: ", core_id);
} else {
- printf("Core %3d: ", core);
+ printf("Core %3d: ", core_id);
}
- thread_num = i * threads_per_cpu;
- for (j = 0; j < threads_per_cpu; j++, thread_num++) {
+ for (j = 0; j < threads_per_cpu; j++) {
+ thread_num = core_id * threads_per_cpu + j;
online = cpu_online(thread_num) ? '*' : ' ';
printf("%4d%c ", thread_num, online);
}
printf("\n");
- core++;
}
+ free(present_cores);
return 0;
}
-
static void usage(void)
{
printf(

View File

@ -1,6 +1,6 @@
Name: powerpc-utils
Version: 1.3.13
Release: 2%{?dist}
Release: 3%{?dist}
Summary: PERL-based scripts for maintaining and servicing PowerPC systems
License: GPLv2
@ -10,8 +10,21 @@ Source1: nx-gzip.udev
Patch0: powerpc-utils-1.3.10-manpages.patch
# upstream patches
Patch10: powerpc-utils-fix-return-value-from-do_replace.patch
Patch11: powerpc-utils-lparstat-print-memory-mode-correctly.patch
Patch: powerpc-utils-1.3.13-cpu_info_helpers.patch
Patch: powerpc-utils-1.3.13-ppc64_cpu-Fix-handling-of-non-contiguous-CPU-IDs.patch
Patch: powerpc-utils-fix-return-value-from-do_replace.patch
Patch: powerpc-utils-lparstat-print-memory-mode-correctly.patch
# Dynamically removing cores from an LPAR is not taking care of NUMA topology (DLPAR)
Patch: powerpc-utils-1.3.13-drmgr_Move_numa_topology_code_to_common_numa.patch
Patch: powerpc-utils-1.3.13-drmgr_Move_read_lmb-size_property_code_to_common_ofdt.patch
Patch: powerpc-utils-1.3.13-drmgr_Add_get_next_cpu_to_identify_the_removable_CPU.patch
Patch: powerpc-utils-1.3.13-drmgr_Allocate_CPU_bitmap_for_each_NUMA_node.patch
Patch: powerpc-utils-1.3.13-drmgr_Add_NUMA_configuration_update_for_CPU_remove.patch
Patch: powerpc-utils-1.3.13-drmgr_Add_NUMA_based_CPU_removal.patch
# DLPAR Memory remove/add operations fail with SAP HANA DB loaded
Patch: powerpc-utils-1.3.13-drmgr_Allow_signals_mentioned_in_new_sigset_t.patch
Patch: powerpc-utils-1.3.13-drmgr_Add_timeout_signal_handling_for_NUMA_memory_REMOVE.patch
Patch: powerpc-utils-1.3.13-drmgr_Do_not_remove_LMBs_when_the_timer_expires.patch
ExclusiveArch: ppc %{power64}
@ -213,6 +226,11 @@ systemctl enable hcn-init.service >/dev/null 2>&1 || :
%changelog
* Mon Aug 03 2026 Than Ngo <than@redhat.com> - 1.3.13-3
- Resolves: RHEL-184814, Dynamically removing cores from an LPAR is not taking care of NUMA topology
- Resolves: RHEL-180740, DLPAR Memory remove/add operations fail with SAP HANA DB loaded
- Resolves: RHEL-212027, Continuous migration failure error messages observed while running memhotplug
* Tue Mar 11 2025 Than Ngo <than@redhat.com> - 1.3.13-2
- Resolves: RHEL-80729, drmgr: return correct value
- Resolves: RHEL-80277, lparstat: print memory mode correctly