From ffdbafc5d810ba3c26618dad3fe6403038ad263a Mon Sep 17 00:00:00 2001 From: Than Ngo Date: Mon, 3 Aug 2026 20:00:27 +0200 Subject: [PATCH] - Resolves: RHEL-184814, Dynamically removing cores from an LPAR is not taking care of NUMA topology - Resolves: RHEL-180740, DLPAR Memory remove/add operations fail with SAP HANA DB loaded - Resolves: RHEL-212027, Continuous migration failure error messages observed while running memhotplug --- powerpc-utils-1.3.13-cpu_info_helpers.patch | 163 ++++ ....13-drmgr_Add_NUMA_based_CPU_removal.patch | 195 +++++ ..._configuration_update_for_CPU_remove.patch | 117 +++ ...xt_cpu_to_identify_the_removable_CPU.patch | 70 ++ ...gnal_handling_for_NUMA_memory_REMOVE.patch | 106 +++ ...locate_CPU_bitmap_for_each_NUMA_node.patch | 116 +++ ...ow_signals_mentioned_in_new_sigset_t.patch | 28 + ...t_remove_LMBs_when_the_timer_expires.patch | 87 ++ ...ve_numa_topology_code_to_common_numa.patch | 150 ++++ ...mb-size_property_code_to_common_ofdt.patch | 81 ++ ...x-handling-of-non-contiguous-CPU-IDs.patch | 749 ++++++++++++++++++ powerpc-utils.spec | 24 +- 12 files changed, 1883 insertions(+), 3 deletions(-) create mode 100644 powerpc-utils-1.3.13-cpu_info_helpers.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Add_NUMA_based_CPU_removal.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Add_NUMA_configuration_update_for_CPU_remove.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Add_get_next_cpu_to_identify_the_removable_CPU.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Add_timeout_signal_handling_for_NUMA_memory_REMOVE.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Allocate_CPU_bitmap_for_each_NUMA_node.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Allow_signals_mentioned_in_new_sigset_t.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Do_not_remove_LMBs_when_the_timer_expires.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Move_numa_topology_code_to_common_numa.patch create mode 100644 powerpc-utils-1.3.13-drmgr_Move_read_lmb-size_property_code_to_common_ofdt.patch create mode 100644 powerpc-utils-1.3.13-ppc64_cpu-Fix-handling-of-non-contiguous-CPU-IDs.patch diff --git a/powerpc-utils-1.3.13-cpu_info_helpers.patch b/powerpc-utils-1.3.13-cpu_info_helpers.patch new file mode 100644 index 0000000..ce32681 --- /dev/null +++ b/powerpc-utils-1.3.13-cpu_info_helpers.patch @@ -0,0 +1,163 @@ +commit 54cf30c7d274c8aab2a7ae589ab056f52dfffc62 +Author: Aboorva Devarajan +Date: Sat Dec 7 21:54:44 2024 -0500 + + cpu_info_helpers: Add helper function to retrieve present CPU core list + + Introduce get_present_core_list helper function to accurately parse + and retrieve the list of present CPU cores, addressing gaps in core + numbering caused by dynamic addition or removal of CPUs (via CPU DLPAR + operation) + + Utilizes the present CPU list from `sys/devices/system/cpu/present` + to handle non-contiguous CPU IDs. Accurately maps core IDs to CPUs + considering specified number of threads per CPU, addressing gaps in + core numbering. + + Signed-off-by: Aboorva Devarajan + +diff --git a/src/common/cpu_info_helpers.c b/src/common/cpu_info_helpers.c +index 8c57db8..756e792 100644 +--- a/src/common/cpu_info_helpers.c ++++ b/src/common/cpu_info_helpers.c +@@ -203,6 +203,113 @@ int __get_one_smt_state(int core, int threads_per_cpu) + return smt_state; + } + ++int get_present_cpu_count(void) ++{ ++ int start, end, total_cpus = 0; ++ size_t len = 0; ++ char *line = NULL; ++ FILE *fp; ++ char *token; ++ ++ fp = fopen(CPU_PRESENT_PATH, "r"); ++ if (!fp) { ++ perror("Error opening CPU_PRESENT_PATH"); ++ return -1; ++ } ++ ++ if (getline(&line, &len, fp) == -1) { ++ perror("Error reading CPU_PRESENT_PATH"); ++ fclose(fp); ++ free(line); ++ return -1; ++ } ++ fclose(fp); ++ ++ token = strtok(line, ","); ++ while (token) { ++ if (sscanf(token, "%d-%d", &start, &end) == 2) { ++ total_cpus += (end - start + 1); ++ } else if (sscanf(token, "%d", &start) == 1) { ++ total_cpus++; ++ } ++ token = strtok(NULL, ","); ++ } ++ ++ free(line); ++ return total_cpus; ++} ++ ++int get_present_core_list(int **present_cores, int *num_present_cores, int threads_per_cpu) ++{ ++ FILE *fp = NULL; ++ char *line = NULL; ++ char *token = NULL; ++ size_t len = 0; ++ ssize_t read; ++ int core_count = 0; ++ int core_list_size; ++ int *cores = NULL; ++ int start, end, i; ++ ++ if (threads_per_cpu <= 0) { ++ fprintf(stderr, "Invalid threads_per_cpu value, got %d expected >= 1\n", threads_per_cpu); ++ return -1; ++ } ++ ++ core_list_size = get_present_cpu_count() / threads_per_cpu; ++ if (core_list_size <= 0) { ++ fprintf(stderr, "Error while calculating core list size\n"); ++ return -1; ++ } ++ ++ cores = malloc(core_list_size * sizeof(int)); ++ if (!cores) { ++ perror("Memory allocation failed"); ++ goto cleanup; ++ } ++ ++ fp = fopen(CPU_PRESENT_PATH, "r"); ++ if (!fp) { ++ perror("Error opening file"); ++ goto cleanup; ++ } ++ ++ read = getline(&line, &len, fp); ++ if (read == -1) { ++ perror("Error reading file"); ++ goto cleanup; ++ } ++ ++ token = strtok(line, ","); ++ while (token) { ++ if (sscanf(token, "%d-%d", &start, &end) == 2) { ++ for (i = start; i <= end; i++) { ++ if (i % threads_per_cpu == 0) { ++ cores[core_count++] = i / threads_per_cpu; ++ } ++ } ++ } else if (sscanf(token, "%d", &start) == 1) { ++ if (start % threads_per_cpu == 0) { ++ cores[core_count++] = start / threads_per_cpu; ++ } ++ } ++ token = strtok(NULL, ","); ++ } ++ ++ *present_cores = cores; ++ *num_present_cores = core_count; ++ free(line); ++ return 0; ++ ++cleanup: ++ if (fp) { ++ fclose(fp); ++ } ++ free(line); ++ free(cores); ++ return -1; ++} ++ + static void print_cpu_list(const cpu_set_t *cpuset, int cpuset_size, + int cpus_in_system) + { +diff --git a/src/common/cpu_info_helpers.h b/src/common/cpu_info_helpers.h +index c063fff..77e6ad7 100644 +--- a/src/common/cpu_info_helpers.h ++++ b/src/common/cpu_info_helpers.h +@@ -24,9 +24,10 @@ + #ifndef _CPU_INFO_HELPERS_H + #define _CPU_INFO_HELPERS_H + +-#define SYSFS_CPUDIR "/sys/devices/system/cpu/cpu%d" +-#define SYSFS_SUBCORES "/sys/devices/system/cpu/subcores_per_core" +-#define INTSERV_PATH "/proc/device-tree/cpus/%s/ibm,ppc-interrupt-server#s" ++#define SYSFS_CPUDIR "/sys/devices/system/cpu/cpu%d" ++#define SYSFS_SUBCORES "/sys/devices/system/cpu/subcores_per_core" ++#define INTSERV_PATH "/proc/device-tree/cpus/%s/ibm,ppc-interrupt-server#s" ++#define CPU_PRESENT_PATH "/sys/devices/system/cpu/present" + + #define SYSFS_PATH_MAX 128 + +@@ -39,6 +40,8 @@ extern int num_subcores(void); + extern int get_attribute(char *path, const char *fmt, int *value); + extern int get_cpu_info(int *threads_per_cpu, int *cpus_in_system, + int *threads_in_system); ++extern int get_present_core_list(int **present_cores, int *num_present_cores, ++ int threads_per_cpu); + extern int __is_smt_capable(int threads_in_system); + extern int __get_one_smt_state(int core, int threads_per_cpu); + extern int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu, diff --git a/powerpc-utils-1.3.13-drmgr_Add_NUMA_based_CPU_removal.patch b/powerpc-utils-1.3.13-drmgr_Add_NUMA_based_CPU_removal.patch new file mode 100644 index 0000000..fe22265 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Add_NUMA_based_CPU_removal.patch @@ -0,0 +1,195 @@ +commit 908986e43b6c67288aa084fe13155964caa807a8 +Author: Haren Myneni +Date: Sat May 16 11:34:22 2026 -0700 + + drmgr: Add NUMA based CPU removal + + The current CPU removal process reads CPU list from last CPU and + remove CPUs based on the userspace request. This process can + result in CPU less NUMA nodes even though these nodes have more + memory which can affect on the system performance. + + This patch adds NUMA aware CPU removal process to remove CPUs from + specific NUMA nodes and maintains NUMA balance. The selection of + node from which the CPU to be removed is based on the available + memory per CPU in that node called node ratio. So CPU is selected + from the node which has lower ratio. + + If the NUMA topology can't be read, fallback using the current + process. + + The node selection process is as follows: + - For each CPU removal request, update node ratios and sort the list. + - Select the next removable CPU from the dr_info CPU list and it + should belong to the first node. + - CPU associated to memory less nodes is considered first and then + the first node that has memory in the list. + - Repeat all CPUs in dr_info list until the next removable CPU is + matched with node CPU bitmap. + - The total number of CPU threads in the selected node is + decremented and cleared in the node CPU bitmap. + + Signed-off-by: Haren Myneni + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/drslot_chrp_cpu.c b/src/drmgr/drslot_chrp_cpu.c +index e367634..60956a0 100644 +--- a/src/drmgr/drslot_chrp_cpu.c ++++ b/src/drmgr/drslot_chrp_cpu.c +@@ -164,6 +164,141 @@ static struct dr_node *get_available_cpu_by_index(struct dr_info *dr_info) + return cpu; + } + ++/* ++ * Return node if CPU ID matches in node CPU bitmap. ++ */ ++static struct ppcnuma_node *match_cpu_node(struct ppcnuma_node *node, ++ struct dr_node *cpu) ++{ ++ int nid; ++ ++ if (cpu->cpu_threads) { ++ nid = numa_node_of_cpu(cpu->cpu_threads->id); ++ if (nid == node->node_id) { ++ if (numa_bitmask_isbitset(node->cpus, ++ cpu->cpu_threads->id)) ++ return node; ++ } ++ } ++ ++ return NULL; ++} ++ ++/* ++ * Return node if CPU belongs to any memoryless NUMA node. ++ */ ++static struct ppcnuma_node *find_cpu_memless_node(struct dr_node *cpu) ++{ ++ struct ppcnuma_node *node = NULL; ++ int nid; ++ ++ ppcnuma_foreach_node(&numa, nid, node) { ++ if (node->n_lmbs) ++ continue; ++ ++ if (match_cpu_node(node, cpu)) ++ return node; ++ } ++ ++ return NULL; ++} ++ ++/* ++ * The node list is sorted by node ratio (less memory per CPU). ++ * So consider the first node ++ * Return node if CPU belongs to the first NUMA node which ++ * has memory. ++ */ ++static struct ppcnuma_node *find_cpu_numa_node(struct dr_node *cpu) ++{ ++ struct ppcnuma_node *node = NULL; ++ int found = 0; ++ ++ ppcnuma_foreach_node_by_ratio(&numa, node) { ++ if (node->n_cpus && node->n_lmbs) { ++ found = 1; ++ break; ++ } ++ } ++ ++ if (found && match_cpu_node(node, cpu)) ++ return node; ++ ++ return NULL; ++} ++ ++/* ++ * Calculate node ratio based on amount of memory per CPU and sort ++ * the node ratio list. ++ */ ++static void cpu_update_node_ratio(void) ++{ ++ struct ppcnuma_node *node; ++ int nid; ++ ++ ppcnuma_foreach_node(&numa, nid, node) { ++ if (!node->n_lmbs || !node->n_cpus) ++ continue; ++ ++ /* ++ * Node ratio = n_lmbs per CPU ++ */ ++ node->ratio = (node->n_lmbs * 100) / node->n_cpus; ++ } ++ ++ order_numa_node_ratio_list(); ++} ++ ++/* ++ * Scan CPUs from the last one in the list and select the first CPU ++ * based on: ++ * - CPU from memory less node ++ * - If no CPUs are available in memory less nodes, CPU belongs to ++ * the first node from node ratio list. ++ */ ++static struct dr_node *numa_get_next_cpu(struct dr_info *dr_info) ++{ ++ struct ppcnuma_node *node; ++ struct dr_node *cpu = NULL; ++ struct thread *t; ++ int i, found = 0; ++ ++ /* ++ * Update node ratio for each CPU removal request ++ */ ++ cpu_update_node_ratio(); ++ ++ /* Find the first cpu with an online thread */ ++ for (cpu = dr_info->all_cpus; cpu; cpu = cpu->next) { ++ if (cpu->unusable) ++ continue; ++ ++ if (numa.memless_cpu_count) ++ node = find_cpu_memless_node(cpu); ++ else ++ node = find_cpu_numa_node(cpu); ++ ++ if (!node) ++ continue; ++ ++ t = cpu->cpu_threads; ++ for (i = 0; i < cpu->cpu_nthreads && t; i++, t = t->next) { ++ if (get_thread_state(t) == ONLINE) ++ found = 1; ++ numa_bitmask_clearbit(node->cpus, t->id); ++ } ++ if (found) { ++ node->n_cpus -= cpu->cpu_nthreads; ++ numa.cpu_count -= cpu->cpu_nthreads; ++ if (!node->n_lmbs) ++ numa.memless_cpu_count -= cpu->cpu_nthreads; ++ return cpu; ++ } ++ } ++ ++ return NULL; ++} ++ + /* + * Scan all CPUs from the last one for the next available CPU. + * Used only for non-NUMA based CPU removal. +@@ -202,8 +337,12 @@ static struct dr_node *get_next_available_cpu(struct dr_info *dr_info) + + cpu = survivor; + } else if (usr_action == REMOVE) { +- /* Find the first cpu with an online thread */ +- cpu = get_next_cpu(dr_info); ++ if (numa_enabled) ++ /* Find the first CPU from NUMA nodes */ ++ cpu = numa_get_next_cpu(dr_info); ++ else ++ /* Find the first cpu with an online thread */ ++ cpu = get_next_cpu(dr_info); + } + + if (!cpu) diff --git a/powerpc-utils-1.3.13-drmgr_Add_NUMA_configuration_update_for_CPU_remove.patch b/powerpc-utils-1.3.13-drmgr_Add_NUMA_configuration_update_for_CPU_remove.patch new file mode 100644 index 0000000..5c202b9 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Add_NUMA_configuration_update_for_CPU_remove.patch @@ -0,0 +1,117 @@ +commit 4cf04b9e75db0da2c199fa12003433b3d0427fd0 +Author: Haren Myneni +Date: Sat May 16 11:34:21 2026 -0700 + + drmgr: Add NUMA configuration update for CPU remove + + This patch adds NUMA node config update for CPU removal. Updates + number of LMBs and CPUs for each node and also calculates total + number of CPUs from memory less nodes. This node configuration is + used to identify the node based on the node ratio from which the + CPU is selected to remove + + Signed-off-by: Haren Myneni + Reviewed-by: Dave Marquardt + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/common_numa.h b/src/drmgr/common_numa.h +index 7aea026..edf4349 100644 +--- a/src/drmgr/common_numa.h ++++ b/src/drmgr/common_numa.h +@@ -38,6 +38,7 @@ struct ppcnuma_topology { + unsigned int lmb_count; + unsigned int cpuless_node_count; + unsigned int cpuless_lmb_count; ++ unsigned int memless_cpu_count; + unsigned int node_count, node_min, node_max; + struct ppcnuma_node *nodes[MAX_NUMNODES]; + struct ppcnuma_node *ratio; +diff --git a/src/drmgr/drslot_chrp_cpu.c b/src/drmgr/drslot_chrp_cpu.c +index 6a21663..e367634 100644 +--- a/src/drmgr/drslot_chrp_cpu.c ++++ b/src/drmgr/drslot_chrp_cpu.c +@@ -26,10 +26,14 @@ + #include + #include + #include ++#include + #include "dr.h" + #include "drcpu.h" + #include "drpci.h" + #include "ofdt.h" ++#include "common_numa.h" ++ ++#define DEFAULT_LMB_SIZE 0x10000000 /* 256MB */ + + struct cpu_operation; + typedef int (cpu_op_func_t) (void); +@@ -395,6 +399,43 @@ static int smt_threads_func(struct dr_info *dr_info) + return rc; + } + ++/* ++ * Per node CPUs are defined as part of build_numa_topology(). ++ * This function calculates number of LMBs per node based on ++ * node mememory / lmb-size. ++ * n_cpus and n_lmbs are used to determine node ratio. ++ */ ++static int cpu_update_numa_config(void) ++{ ++ struct ppcnuma_node *node; ++ unsigned long long node_size; ++ int rc, nid; ++ uint64_t lmb_sz; ++ ++ rc = get_dynamic_lmb_size(&lmb_sz); ++ /* ++ * Use the default value if lmb-size property is not available. ++ * For CPU removal, node ratio will be calculated based on ++ * total n_lmbs per CPU. ++ */ ++ if (rc) ++ lmb_sz = DEFAULT_LMB_SIZE; ++ ++ ppcnuma_foreach_node(&numa, nid, node) { ++ node_size = numa_node_size(nid, 0); ++ /* ++ * Node has memory ++ * n_lmbs = Total memory / lmb-size ++ */ ++ if (node_size) { ++ node->n_lmbs = node_size / lmb_sz; ++ } else ++ numa.memless_cpu_count += node->n_cpus; ++ } ++ ++ return 0; ++} ++ + int valid_cpu_options(void) + { + /* default to a quantity of 1 */ +@@ -442,6 +483,15 @@ int drslot_chrp_cpu(void) + return -1; + } + ++ /* ++ * Maintain NUMA aware hotplug only for remove and with count request. ++ */ ++ if (usr_drc_count && (usr_action == REMOVE)) { ++ build_numa_topology(); ++ if (numa_enabled) ++ cpu_update_numa_config(); ++ } ++ + /* If a user specifies a drc name, the quantity to add/remove is + * one. Enforce that here so the loops in add/remove code behave + * accordingly. +@@ -473,6 +523,9 @@ int drslot_chrp_cpu(void) + if (usr_action == ADD || usr_action == REMOVE) + run_hooks(DRC_TYPE_CPU, usr_action, HOOK_POST, count); + ++ if ((usr_action == REMOVE) && numa_enabled) ++ free_numa_topology(); ++ + free_cpu_drc_info(&dr_info); + return rc; + } diff --git a/powerpc-utils-1.3.13-drmgr_Add_get_next_cpu_to_identify_the_removable_CPU.patch b/powerpc-utils-1.3.13-drmgr_Add_get_next_cpu_to_identify_the_removable_CPU.patch new file mode 100644 index 0000000..74c0712 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Add_get_next_cpu_to_identify_the_removable_CPU.patch @@ -0,0 +1,70 @@ +commit fc252eaf3f1e2b6dcf2a646459ac7399db0ab18d +Author: Haren Myneni +Date: Sat May 16 11:34:19 2026 -0700 + + drmgr: Add get_next_cpu() to identify the removable CPU + + Move code which identifies the removable CPU to get_next_cpu(). + This function is used only for the current non-numa based CPU + removal but helps to add for NUMA based CPU removal code in later + patch. + + Signed-off-by: Haren Myneni + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/drslot_chrp_cpu.c b/src/drmgr/drslot_chrp_cpu.c +index 3ef24f4..6a21663 100644 +--- a/src/drmgr/drslot_chrp_cpu.c ++++ b/src/drmgr/drslot_chrp_cpu.c +@@ -160,11 +160,33 @@ static struct dr_node *get_available_cpu_by_index(struct dr_info *dr_info) + return cpu; + } + ++/* ++ * Scan all CPUs from the last one for the next available CPU. ++ * Used only for non-NUMA based CPU removal. ++ */ ++static struct dr_node *get_next_cpu(struct dr_info *dr_info) ++{ ++ struct dr_node *cpu = NULL; ++ struct thread *t; ++ ++ /* Find the first cpu with an online thread */ ++ for (cpu = dr_info->all_cpus; cpu; cpu = cpu->next) { ++ if (cpu->unusable) ++ continue; ++ ++ for (t = cpu->cpu_threads; t; t = t->next) { ++ if (get_thread_state(t) == ONLINE) ++ return cpu; ++ } ++ } ++ ++ return NULL; ++} ++ + static struct dr_node *get_next_available_cpu(struct dr_info *dr_info) + { + struct dr_node *cpu = NULL; + struct dr_node *survivor = NULL; +- struct thread *t; + + if (usr_action == ADD) { + for (cpu = dr_info->all_cpus; cpu; cpu = cpu->next) { +@@ -177,15 +199,7 @@ static struct dr_node *get_next_available_cpu(struct dr_info *dr_info) + cpu = survivor; + } else if (usr_action == REMOVE) { + /* Find the first cpu with an online thread */ +- for (cpu = dr_info->all_cpus; cpu; cpu = cpu->next) { +- if (cpu->unusable) +- continue; +- +- for (t = cpu->cpu_threads; t; t = t->next) { +- if (get_thread_state(t) == ONLINE) +- return cpu; +- } +- } ++ cpu = get_next_cpu(dr_info); + } + + if (!cpu) diff --git a/powerpc-utils-1.3.13-drmgr_Add_timeout_signal_handling_for_NUMA_memory_REMOVE.patch b/powerpc-utils-1.3.13-drmgr_Add_timeout_signal_handling_for_NUMA_memory_REMOVE.patch new file mode 100644 index 0000000..94f5d46 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Add_timeout_signal_handling_for_NUMA_memory_REMOVE.patch @@ -0,0 +1,106 @@ +commit 1dee26c2a60dd9a5264331cc20143a5590289187 +Author: Haren Myneni +Date: Mon May 18 22:37:53 2026 -0700 + + drmgr: Add timeout signal handling for NUMA memory REMOVE + + RMC expects drmgr timeout based on value passed with -w option and + drmgr can check fequently for each LMB removal request and exits + once reached the timeout value. But this check happens only in the + user space and does not consider when executes in the kernel to + remove memory. In the case of LMB removal, the kernel expects to + run longer until all pages in LMB are isolated and can return to + the user space for any pending signals. + + This patch enables SIGALRM signal based on the user defined + value which generates signal when the timer expires. It allows + the kernel interface returns in case [age isolation is taking + longer and drmgr timeout. + + Signed-off-by: Haren Myneni + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c +index fe04ad1..5a08839 100644 +--- a/src/drmgr/drslot_chrp_mem.c ++++ b/src/drmgr/drslot_chrp_mem.c +@@ -26,6 +26,9 @@ + #include + #include + #include ++#include ++#include ++#include + #include + #include + #include "dr.h" +@@ -35,6 +38,7 @@ + + uint64_t block_sz_bytes = 0; + static char *state_strs[] = {"offline", "online"}; ++sig_atomic_t numa_mem_timeout = 0; + + static char *usagestr = "-c mem {-a | -r} {-q -p {variable_weight | ent_capacity} | {-q | -s [ | ]}}"; + +@@ -50,6 +54,14 @@ mem_usage(char **pusage) + *pusage = usagestr; + } + ++/* ++ * SIGALRM handler for drmgr timeout ++ */ ++void mem_timeout_handler(int sig) ++{ ++ numa_mem_timeout = 1; ++} ++ + /** + * report_resource_count + * @brief Report the number of LMBs that were added or removed. +@@ -1681,13 +1693,45 @@ static void clear_numa_lmb_links(void) + node->lmbs = NULL; + } + ++/* ++ * Setup SIGALRM signal based on timeout value passed by the user ++ * (with -w option). In the case of LMB removal, the kernel ++ * interface can run longer until all pages in LMB are isolated ++ * and can return to the user space if any pending signals. ++ * This SIGALRM signal can exit the kernel in case LMB removal ++ * is taking longer than timeout. ++ */ ++static int drmem_timer_setup(void) ++{ ++ struct sigaction sigact; ++ ++ if (!usr_timeout) ++ return 0; ++ ++ sigact.sa_handler = mem_timeout_handler; ++ sigemptyset(&sigact.sa_mask); ++ sigact.sa_flags = 0; ++ if (sigaction(SIGALRM, &sigact, NULL)) ++ return -1; ++ ++ alarm(usr_timeout); ++ return 0; ++} ++ + static int numa_based_remove(uint32_t count) + { + struct lmb_list_head *lmb_list; + struct ppcnuma_node *node; +- int nid; ++ int nid, rc = 0; + uint32_t done = 0; + ++ /* ++ * Enable alarm signal handler for usr_timeout ++ */ ++ rc = drmem_timer_setup(); ++ if (rc) ++ return rc; ++ + /* + * Read the LMBs + * Link the LMBs to their node diff --git a/powerpc-utils-1.3.13-drmgr_Allocate_CPU_bitmap_for_each_NUMA_node.patch b/powerpc-utils-1.3.13-drmgr_Allocate_CPU_bitmap_for_each_NUMA_node.patch new file mode 100644 index 0000000..4a79b36 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Allocate_CPU_bitmap_for_each_NUMA_node.patch @@ -0,0 +1,116 @@ +commit 1c4812fb35510c7e02ff27fc39badbe5eef24611 +Author: Haren Myneni +Date: Sat May 16 11:34:20 2026 -0700 + + drmgr: Allocate CPU bitmap for each NUMA node + + The current code allocates one bitmap and uses it to determine + number of valid CPUs for each NUMA node and then frees the bitmap. + The NUMA based CPU removal needs this bitmap per node to determine + the valid CPU in that node. So this patch retains bitmap per node + and frees it after DLPAR memory / CPU removal. + + Signed-off-by: Haren Myneni + Reviewed-by: Dave Marquardt + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/common_numa.c b/src/drmgr/common_numa.c +index 6bc2ea8..eba6c3d 100644 +--- a/src/drmgr/common_numa.c ++++ b/src/drmgr/common_numa.c +@@ -84,9 +84,6 @@ static int read_numa_topology(struct ppcnuma_topology *numa) + + rc = 0; + +- /* In case of allocation error, the libnuma is calling exit() */ +- cpus = numa_allocate_cpumask(); +- + for (nid = 0; nid <= max_node; nid++) { + + if (!numa_bitmask_isbitset(numa_nodes_ptr, nid)) +@@ -98,20 +95,24 @@ static int read_numa_topology(struct ppcnuma_topology *numa) + break; + } + ++ /* In case of allocation error, the libnuma is calling exit() */ ++ cpus = numa_allocate_cpumask(); ++ + rc = numa_node_to_cpus(nid, cpus); +- if (rc < 0) ++ if (rc < 0) { ++ numa_bitmask_free(cpus); + break; ++ } + + /* Count the CPUs in that node */ + for (i = 0; i < cpus->size; i++) + if (numa_bitmask_isbitset(cpus, i)) + node->n_cpus++; + ++ node->cpus = cpus; + numa->cpu_count += node->n_cpus; + } + +- numa_bitmask_free(cpus); +- + if (rc) { + ppcnuma_foreach_node(numa, nid, node) + node->n_cpus = 0; +@@ -160,6 +161,20 @@ void build_numa_topology(void) + numa_enabled = 1; + } + ++void free_numa_topology(void) ++{ ++ struct ppcnuma_node *node; ++ int nid; ++ ++ ppcnuma_foreach_node(&numa, nid, node) { ++ if (node) { ++ if (node->cpus) ++ numa_bitmask_free(node->cpus); ++ free(node); ++ } ++ } ++} ++ + void order_numa_node_ratio_list(void) + { + int nid; +diff --git a/src/drmgr/common_numa.h b/src/drmgr/common_numa.h +index 2b0901e..7aea026 100644 +--- a/src/drmgr/common_numa.h ++++ b/src/drmgr/common_numa.h +@@ -30,6 +30,7 @@ struct ppcnuma_node { + unsigned int ratio; + struct dr_node *lmbs; /* linked by lmb_numa_next */ + struct ppcnuma_node *ratio_next; ++ struct bitmask *cpus; + }; + + struct ppcnuma_topology { +@@ -48,6 +49,7 @@ extern int numa_enabled; + extern struct ppcnuma_topology numa; + void build_numa_topology(void); + void order_numa_node_ratio_list(void); ++void free_numa_topology(void); + + struct ppcnuma_node *ppcnuma_fetch_node(struct ppcnuma_topology *numa, + int node_id); +diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c +index 2d22bff..fe04ad1 100644 +--- a/src/drmgr/drslot_chrp_mem.c ++++ b/src/drmgr/drslot_chrp_mem.c +@@ -1731,9 +1731,10 @@ int do_mem_kernel_dlpar(void) + if (usr_action == REMOVE && usr_drc_count && !usr_drc_index) { + build_numa_topology(); + if (numa_enabled) { +- if (!numa_based_remove(usr_drc_count)) ++ rc = numa_based_remove(usr_drc_count); ++ free_numa_topology(); ++ if (!rc) + return 0; +- + /* + * If the NUMA based removal failed, lets try the legacy + * way. diff --git a/powerpc-utils-1.3.13-drmgr_Allow_signals_mentioned_in_new_sigset_t.patch b/powerpc-utils-1.3.13-drmgr_Allow_signals_mentioned_in_new_sigset_t.patch new file mode 100644 index 0000000..74aec67 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Allow_signals_mentioned_in_new_sigset_t.patch @@ -0,0 +1,28 @@ +commit 3a7d0b61e7a5f364660007a45cc9dcf1c3cb5ab4 +Author: Haren Myneni +Date: Mon May 18 22:37:52 2026 -0700 + + drmgr: Allow signals mentioned in new sigset_t + + The current code uses sigdelset() to unblock some signals such as + SIGBUS, SIGALRM, and etc. But sigprocmask() with SIG_BLOCK sets + the union of this new set and the current sigset which ends up + blocking some of them (Ex: SIGALRM). Insted of using SIG_BLOCK, + SIG_SETMASK allows to use the complete new sigset_t. + + Signed-off-by: Haren Myneni + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/common.c b/src/drmgr/common.c +index 70f4dfd..68041a9 100644 +--- a/src/drmgr/common.c ++++ b/src/drmgr/common.c +@@ -939,7 +939,7 @@ sig_setup(void) + sigdelset(&sigset, SIGABRT); + + /* Now block all remaining signals */ +- rc = sigprocmask(SIG_BLOCK, &sigset, NULL); ++ rc = sigprocmask(SIG_SETMASK, &sigset, NULL); + if (rc) + return -1; + diff --git a/powerpc-utils-1.3.13-drmgr_Do_not_remove_LMBs_when_the_timer_expires.patch b/powerpc-utils-1.3.13-drmgr_Do_not_remove_LMBs_when_the_timer_expires.patch new file mode 100644 index 0000000..34b9645 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Do_not_remove_LMBs_when_the_timer_expires.patch @@ -0,0 +1,87 @@ +commit 7926241f576de24df8b342858f5278911e035658 +Author: Haren Myneni +Date: Mon May 18 22:37:54 2026 -0700 + + drmgr: Do not remove LMBs when the timer expires + + The current code selects LMBs from all NUMA node based on node + ratio and removes them until reaches the the requested limit. This + patch stops removing LMBs when receives SIGALRM signal based on + the user defined timeout. + + Also reports total numaber of LMBs removed with DR_TOTAL_RESOURCES + to RMC even for partial memory removal or with an error. + + Signed-off-by: Haren Myneni + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c +index 5a08839..4a3fe3a 100644 +--- a/src/drmgr/drslot_chrp_mem.c ++++ b/src/drmgr/drslot_chrp_mem.c +@@ -1478,6 +1478,9 @@ static int remove_lmb_from_node(struct ppcnuma_node *node, uint32_t count) + count, node->n_lmbs, node->node_id); + + for (lmb = node->lmbs; lmb && done < count; lmb = lmb->lmb_numa_next) { ++ if (numa_mem_timeout) ++ break; ++ + unlinked++; + err = remove_lmb_by_index(lmb->drc_index); + if (err) +@@ -1563,6 +1566,9 @@ static int remove_cpuless_lmbs(uint32_t count) + + this_loop = 0; + ppcnuma_foreach_node(&numa, nid, node) { ++ if (numa_mem_timeout) ++ break; ++ + if (!node->n_lmbs || node->n_cpus) + continue; + +@@ -1656,6 +1662,9 @@ static int remove_cpu_lmbs(uint32_t count) + + this_loop = 0; + ppcnuma_foreach_node_by_ratio(&numa, node) { ++ if (numa_mem_timeout) ++ break; ++ + if (!node->n_lmbs || !node->n_cpus) + continue; + +@@ -1739,14 +1748,13 @@ static int numa_based_remove(uint32_t count) + */ + lmb_list = get_lmbs(LMB_NORMAL_SORT); + if (lmb_list == NULL) { +- clear_numa_lmb_links(); +- return -1; ++ rc = -1; ++ goto out_clear; + } + + if (!numa.node_count) { +- clear_numa_lmb_links(); +- free_lmbs(lmb_list); +- return -EINVAL; ++ rc = -EINVAL; ++ goto out_free; + } + + ppcnuma_foreach_node(&numa, nid, node) { +@@ -1759,11 +1767,12 @@ static int numa_based_remove(uint32_t count) + + done += remove_cpu_lmbs(count); + +- report_resource_count(done); +- +- clear_numa_lmb_links(); ++out_free: + free_lmbs(lmb_list); +- return 0; ++out_clear: ++ clear_numa_lmb_links(); ++ report_resource_count(done); ++ return rc; + } + + int do_mem_kernel_dlpar(void) diff --git a/powerpc-utils-1.3.13-drmgr_Move_numa_topology_code_to_common_numa.patch b/powerpc-utils-1.3.13-drmgr_Move_numa_topology_code_to_common_numa.patch new file mode 100644 index 0000000..539ce90 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Move_numa_topology_code_to_common_numa.patch @@ -0,0 +1,150 @@ +commit 27dd4cce01450d94ab58fa2e632eac34307a4a82 +Author: Haren Myneni +Date: Sat May 16 11:34:17 2026 -0700 + + drmgr: Move numa_topology code to common_numa.c + + Move build_numa_topology and sort NUMA node ratio list code to + common_numa.c. These functions will also be used for NUMA aware + CPU removal in later patch. + + Signed-off-by: Haren Myneni + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/common_numa.c b/src/drmgr/common_numa.c +index 898aab6..6bc2ea8 100644 +--- a/src/drmgr/common_numa.c ++++ b/src/drmgr/common_numa.c +@@ -27,6 +27,9 @@ + #include "drmem.h" /* for DYNAMIC_RECONFIG_MEM */ + #include "common_numa.h" + ++int numa_enabled = 0; ++struct ppcnuma_topology numa; ++ + struct ppcnuma_node *ppcnuma_fetch_node(struct ppcnuma_topology *numa, int nid) + { + struct ppcnuma_node *node; +@@ -118,7 +121,7 @@ static int read_numa_topology(struct ppcnuma_topology *numa) + return rc; + } + +-int ppcnuma_get_topology(struct ppcnuma_topology *numa) ++static int ppcnuma_get_topology(struct ppcnuma_topology *numa) + { + int rc; + +@@ -145,3 +148,35 @@ int ppcnuma_get_topology(struct ppcnuma_topology *numa) + + return 0; + } ++ ++void build_numa_topology(void) ++{ ++ int rc; ++ ++ rc = ppcnuma_get_topology(&numa); ++ if (rc) ++ return; ++ ++ numa_enabled = 1; ++} ++ ++void order_numa_node_ratio_list(void) ++{ ++ int nid; ++ struct ppcnuma_node *node, *n, **p; ++ ++ numa.ratio = NULL; ++ ++ /* Create an ordered link of the nodes */ ++ ppcnuma_foreach_node(&numa, nid, node) { ++ if (!node->n_lmbs || !node->n_cpus) ++ continue; ++ ++ p = &numa.ratio; ++ for (n = numa.ratio; ++ n && n->ratio < node->ratio; n = n->ratio_next) ++ p = &n->ratio_next; ++ *p = node; ++ node->ratio_next = n; ++ } ++} +diff --git a/src/drmgr/common_numa.h b/src/drmgr/common_numa.h +index c209a3e..2b0901e 100644 +--- a/src/drmgr/common_numa.h ++++ b/src/drmgr/common_numa.h +@@ -44,7 +44,11 @@ struct ppcnuma_topology { + struct assoc_arrays aa; + }; + +-int ppcnuma_get_topology(struct ppcnuma_topology *numa); ++extern int numa_enabled; ++extern struct ppcnuma_topology numa; ++void build_numa_topology(void); ++void order_numa_node_ratio_list(void); ++ + struct ppcnuma_node *ppcnuma_fetch_node(struct ppcnuma_topology *numa, + int node_id); + +diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c +index 4a36c73..eb75ccf 100644 +--- a/src/drmgr/drslot_chrp_mem.c ++++ b/src/drmgr/drslot_chrp_mem.c +@@ -38,9 +38,6 @@ static char *state_strs[] = {"offline", "online"}; + + static char *usagestr = "-c mem {-a | -r} {-q -p {variable_weight | ent_capacity} | {-q | -s [ | ]}}"; + +-static struct ppcnuma_topology numa; +-static int numa_enabled = 0; +- + /** + * mem_usage + * @brief return usage string +@@ -1605,7 +1602,7 @@ static int remove_cpuless_lmbs(uint32_t count) + static void update_node_ratio(void) + { + int nid; +- struct ppcnuma_node *node, *n, **p; ++ struct ppcnuma_node *node; + uint32_t cpu_ratio, mem_ratio; + + /* +@@ -1626,18 +1623,7 @@ static void update_node_ratio(void) + node->ratio = (cpu_ratio * 9 + mem_ratio) / 10; + } + +- /* Create an ordered link of the nodes */ +- ppcnuma_foreach_node(&numa, nid, node) { +- if (!node->n_lmbs || !node->n_cpus) +- continue; +- +- p = &numa.ratio; +- for (n = numa.ratio; +- n && n->ratio < node->ratio; n = n->ratio_next) +- p = &n->ratio_next; +- *p = node; +- node->ratio_next = n; +- } ++ order_numa_node_ratio_list(); + } + + /* +@@ -1693,17 +1679,6 @@ static int remove_cpu_lmbs(uint32_t count) + return done; + } + +-static void build_numa_topology(void) +-{ +- int rc; +- +- rc = ppcnuma_get_topology(&numa); +- if (rc) +- return; +- +- numa_enabled = 1; +-} +- + static void clear_numa_lmb_links(void) + { + int nid; diff --git a/powerpc-utils-1.3.13-drmgr_Move_read_lmb-size_property_code_to_common_ofdt.patch b/powerpc-utils-1.3.13-drmgr_Move_read_lmb-size_property_code_to_common_ofdt.patch new file mode 100644 index 0000000..b1c5ff5 --- /dev/null +++ b/powerpc-utils-1.3.13-drmgr_Move_read_lmb-size_property_code_to_common_ofdt.patch @@ -0,0 +1,81 @@ +commit cbdbefdb8c82acae997cf1a30816e69b323f9979 +Author: Haren Myneni +Date: Sat May 16 11:34:18 2026 -0700 + + drmgr: Move read lmb-size property code to common_ofdt.c + + get_dynamic_lmb_size() is used to get lmb size from "ibm,lmb-size" + property. This lmb size is needed to determine the number of LMBs + and used to find NUMA node ratio for NUMA aware memory and CPU + removal code. So move this function to common_ofdt.c + + Signed-off-by: Haren Myneni + Signed-off-by: Tyrel Datwyler + +diff --git a/src/drmgr/common_ofdt.c b/src/drmgr/common_ofdt.c +index 1e5fe53..0655559 100644 +--- a/src/drmgr/common_ofdt.c ++++ b/src/drmgr/common_ofdt.c +@@ -28,6 +28,7 @@ + #include + #include "dr.h" + #include "ofdt.h" ++#include "drmem.h" + + #define RTAS_DIRECTORY "/proc/device-tree/rtas" + #define CHOSEN_DIRECTORY "/proc/device-tree/chosen" +@@ -932,3 +933,19 @@ int of_associativity_to_node(const char *dir, int min_common_depth) + return be32toh(prop[min_common_depth]); + } + ++int get_dynamic_lmb_size(uint64_t *lmb_sz) ++{ ++ int rc = 0; ++ uint64_t sz; ++ ++ rc = get_property(DYNAMIC_RECONFIG_MEM, "ibm,lmb-size", ++ &sz, sizeof(sz)); ++ if (rc) { ++ say(DEBUG, "Could not retrieve drconf LMB size\n"); ++ return rc; ++ } ++ ++ /* convert for LE systems */ ++ *lmb_sz = be64toh(sz); ++ return 0; ++} +diff --git a/src/drmgr/drslot_chrp_mem.c b/src/drmgr/drslot_chrp_mem.c +index eb75ccf..2d22bff 100644 +--- a/src/drmgr/drslot_chrp_mem.c ++++ b/src/drmgr/drslot_chrp_mem.c +@@ -506,16 +506,9 @@ get_dynamic_reconfig_lmbs(struct lmb_list_head *lmb_list) + uint64_t lmb_sz; + int rc = 0; + +- rc = get_property(DYNAMIC_RECONFIG_MEM, "ibm,lmb-size", +- &lmb_sz, sizeof(lmb_sz)); +- +- /* convert for LE systems */ +- lmb_sz = be64toh(lmb_sz); +- +- if (rc) { +- say(DEBUG, "Could not retrieve drconf LMB size\n"); ++ rc = get_dynamic_lmb_size(&lmb_sz); ++ if (rc) + return rc; +- } + + if (stat(DYNAMIC_RECONFIG_MEM_V1, &sbuf) == 0) { + rc = get_dynamic_reconfig_lmbs_v1(lmb_sz, lmb_list); +diff --git a/src/drmgr/ofdt.h b/src/drmgr/ofdt.h +index e9ebd03..c79ed65 100644 +--- a/src/drmgr/ofdt.h ++++ b/src/drmgr/ofdt.h +@@ -185,6 +185,7 @@ int get_assoc_arrays(const char *dir, struct assoc_arrays *aa, + int min_common_depth); + int of_associativity_to_node(const char *dir, int min_common_depth); + int init_node(struct dr_node *); ++int get_dynamic_lmb_size(uint64_t *lmb_sz); + + static inline int aa_index_to_node(struct assoc_arrays *aa, uint32_t aa_index) + { diff --git a/powerpc-utils-1.3.13-ppc64_cpu-Fix-handling-of-non-contiguous-CPU-IDs.patch b/powerpc-utils-1.3.13-ppc64_cpu-Fix-handling-of-non-contiguous-CPU-IDs.patch new file mode 100644 index 0000000..cd448fe --- /dev/null +++ b/powerpc-utils-1.3.13-ppc64_cpu-Fix-handling-of-non-contiguous-CPU-IDs.patch @@ -0,0 +1,749 @@ +commit e5fd24a6e35c3be78c96d6887e3774852bbe4674 +Author: Aboorva Devarajan +Date: Wed Jan 1 22:56:07 2025 -0500 + + ppc64_cpu: Fix handling of non-contiguous CPU IDs + + In ppc64le environments, adding or removing CPUs dynamically through + DLPAR can create gaps in CPU IDs, such as `0-103,120-151`, in this + case CPUs 104-119 are missing. + + ppc64_cpu doesn't handles this scenario and always considers CPU IDs + to be contiguous causing issues in core numbering, cpu info and SMT + mode reporting. + + To illustrate the issues this patch fixes, consider the following + system configuration: + + $ lscpu + Architecture: ppc64le + Byte Order: Little Endian + CPU(s): 136 + On-line CPU(s) list: 0-103,120-151 + + **Note: CPU IDs are non-contiguous** + + ----------------------------------------------------------------- + Before Patch: + ----------------------------------------------------------------- + + $ ppc64_cpu --info + Core 0: 0* 1* 2* 3* 4* 5* 6* 7* + Core 1: 8* 9* 10* 11* 12* 13* 14* 15* + Core 2: 16* 17* 18* 19* 20* 21* 22* 23* + Core 3: 24* 25* 26* 27* 28* 29* 30* 31* + Core 4: 32* 33* 34* 35* 36* 37* 38* 39* + Core 5: 40* 41* 42* 43* 44* 45* 46* 47* + Core 6: 48* 49* 50* 51* 52* 53* 54* 55* + Core 7: 56* 57* 58* 59* 60* 61* 62* 63* + Core 8: 64* 65* 66* 67* 68* 69* 70* 71* + Core 9: 72* 73* 74* 75* 76* 77* 78* 79* + Core 10: 80* 81* 82* 83* 84* 85* 86* 87* + Core 11: 88* 89* 90* 91* 92* 93* 94* 95* + Core 12: 96* 97* 98* 99* 100* 101* 102* 103* + ........................................................... *gap* + Core 13: 120* 121* 122* 123* 124* 125* 126* 127* + Core 14: 128* 129* 130* 131* 132* 133* 134* 135* + Core 15: 136* 137* 138* 139* 140* 141* 142* 143* + Core 16: 144* 145* 146* 147* 148* 149* 150* 151* + + **Although the CPU IDs are non contiguous, associated core IDs are + represented in contiguous order, which makes it harder to interpret + this clearly.** + + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + + $ ppc64_cpu --cores-on + Number of cores online = 15 + + **Expected: Number of online cores = 17** + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + + $ ppc64_cpu --offline-cores + Cores offline = 13, 14 + + **Even though no cores are actually offline, two cores (13, 14) + are displayed as offline.** + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + + $ ppc64_cpu --online-cores + Cores online = 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 15, 16 + + **The list of online cores is missing two cores (13, 14).** + ----------------------------------------------------------------- + + To resolve this, use the present CPU list from sysfs to assign + numbers to CPUs and cores, which will make this accurate. + + $ cat /sys/devices/system/cpu/present + 0-103,120-151 + + With this patch, the command output correctly reflects the + current CPU configuration, providing a more precise representation + of the system state. + + ----------------------------------------------------------------- + After Patch: + ----------------------------------------------------------------- + + $ ppc64_cpu --info + Core 0: 0* 1* 2* 3* 4* 5* 6* 7* + Core 1: 8* 9* 10* 11* 12* 13* 14* 15* + Core 2: 16* 17* 18* 19* 20* 21* 22* 23* + Core 3: 24* 25* 26* 27* 28* 29* 30* 31* + Core 4: 32* 33* 34* 35* 36* 37* 38* 39* + Core 5: 40* 41* 42* 43* 44* 45* 46* 47* + Core 6: 48* 49* 50* 51* 52* 53* 54* 55* + Core 7: 56* 57* 58* 59* 60* 61* 62* 63* + Core 8: 64* 65* 66* 67* 68* 69* 70* 71* + Core 9: 72* 73* 74* 75* 76* 77* 78* 79* + Core 10: 80* 81* 82* 83* 84* 85* 86* 87* + Core 11: 88* 89* 90* 91* 92* 93* 94* 95* + Core 12: 96* 97* 98* 99* 100* 101* 102* 103* + ........................................................... *gap* + Core 15: 120* 121* 122* 123* 124* 125* 126* 127* + Core 16: 128* 129* 130* 131* 132* 133* 134* 135* + Core 17: 136* 137* 138* 139* 140* 141* 142* 143* + Core 18: 144* 145* 146* 147* 148* 149* 150* 151* + + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + + $ ppc64_cpu --cores-on + Number of cores online = 17 + + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + + $ ppc64_cpu --offline-cores + Cores offline = + + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + + $ ppc64_cpu --online-cores + Cores online = 0,1,2,3,4,5,6,7,8,9,10,11,12,15,16,17,18 + + ----------------------------------------------------------------- + + Signed-off-by: Aboorva Devarajan + +diff --git a/src/common/cpu_info_helpers.c b/src/common/cpu_info_helpers.c +index 756e792..e75cf6e 100644 +--- a/src/common/cpu_info_helpers.c ++++ b/src/common/cpu_info_helpers.c +@@ -311,67 +311,94 @@ cleanup: + } + + static void print_cpu_list(const cpu_set_t *cpuset, int cpuset_size, +- int cpus_in_system) ++ int threads_per_cpu) + { +- int core; ++ int *present_cores = NULL; ++ int num_present_cores; ++ int start, end, i = 0; + const char *comma = ""; + +- for (core = 0; core < cpus_in_system; core++) { +- int begin = core; +- if (CPU_ISSET_S(core, cpuset_size, cpuset)) { +- while (CPU_ISSET_S(core+1, cpuset_size, cpuset)) +- core++; ++ if (get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu) != 0) { ++ fprintf(stderr, "Failed to get present_cores list\n"); ++ return; ++ } + +- if (core > begin) +- printf("%s%d-%d", comma, begin, core); +- else +- printf("%s%d", comma, core); ++ while (i < num_present_cores) { ++ start = present_cores[i]; ++ if (CPU_ISSET_S(start, cpuset_size, cpuset)) { ++ end = start; ++ while (i + 1 < num_present_cores && ++ CPU_ISSET_S(present_cores[i + 1], cpuset_size, cpuset) && ++ present_cores[i + 1] == end + 1) { ++ end = present_cores[++i]; ++ } ++ if (start == end) { ++ printf("%s%d", comma, start); ++ } else { ++ printf("%s%d-%d", comma, start, end); ++ } + comma = ","; + } ++ i++; + } ++ free(present_cores); + } + +-int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu, +- bool print_smt_state) ++int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu, bool print_smt_state) + { +- int thread, c, smt_state = 0; + cpu_set_t **cpu_states = NULL; +- int cpu_state_size = CPU_ALLOC_SIZE(cpus_in_system); +- int start_cpu = 0, stop_cpu = cpus_in_system; ++ int thread, smt_state = -1; ++ int cpu_state_size; + int rc = 0; ++ int i, core_id, threads_online; ++ int *present_cores = NULL; ++ int num_present_cores; + +- cpu_states = (cpu_set_t **)calloc(threads_per_cpu, sizeof(cpu_set_t)); +- if (!cpu_states) ++ if (get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu) != 0) { ++ fprintf(stderr, "Failed to get present core list\n"); + return -ENOMEM; ++ } ++ cpu_state_size = CPU_ALLOC_SIZE(num_present_cores); ++ cpu_states = (cpu_set_t **)calloc(threads_per_cpu, sizeof(cpu_set_t *)); ++ if (!cpu_states) { ++ rc = -ENOMEM; ++ goto cleanup_present_cores; ++ } + + for (thread = 0; thread < threads_per_cpu; thread++) { +- cpu_states[thread] = CPU_ALLOC(cpus_in_system); ++ cpu_states[thread] = CPU_ALLOC(num_present_cores); ++ if (!cpu_states[thread]) { ++ rc = -ENOMEM; ++ goto cleanup_cpu_states; ++ } + CPU_ZERO_S(cpu_state_size, cpu_states[thread]); + } + +- for (c = start_cpu; c < stop_cpu; c++) { +- int threads_online = __get_one_smt_state(c, threads_per_cpu); +- ++ for (i = 0; i < num_present_cores; i++) { ++ core_id = present_cores[i]; ++ threads_online = __get_one_smt_state(core_id, threads_per_cpu); + if (threads_online < 0) { + rc = threads_online; +- goto cleanup_get_smt; ++ goto cleanup_cpu_states; ++ } ++ if (threads_online) { ++ CPU_SET_S(core_id, cpu_state_size, cpu_states[threads_online - 1]); + } +- if (threads_online) +- CPU_SET_S(c, cpu_state_size, +- cpu_states[threads_online - 1]); + } + + for (thread = 0; thread < threads_per_cpu; thread++) { + if (CPU_COUNT_S(cpu_state_size, cpu_states[thread])) { +- if (smt_state == 0) ++ if (smt_state == -1) + smt_state = thread + 1; + else if (smt_state > 0) + smt_state = 0; /* mix of SMT modes */ + } + } + +- if (!print_smt_state) +- return smt_state; ++ if (!print_smt_state) { ++ rc = smt_state; ++ goto cleanup_cpu_states; ++ } + + if (smt_state == 1) { + if (numeric) +@@ -380,11 +407,9 @@ int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu, + printf("SMT is off\n"); + } else if (smt_state == 0) { + for (thread = 0; thread < threads_per_cpu; thread++) { +- if (CPU_COUNT_S(cpu_state_size, +- cpu_states[thread])) { ++ if (CPU_COUNT_S(cpu_state_size, cpu_states[thread])) { + printf("SMT=%d: ", thread + 1); +- print_cpu_list(cpu_states[thread], +- cpu_state_size, cpus_in_system); ++ print_cpu_list(cpu_states[thread], cpu_state_size, threads_per_cpu); + printf("\n"); + } + } +@@ -392,9 +417,12 @@ int __do_smt(bool numeric, int cpus_in_system, int threads_per_cpu, + printf("SMT=%d\n", smt_state); + } + +-cleanup_get_smt: ++cleanup_cpu_states: + for (thread = 0; thread < threads_per_cpu; thread++) + CPU_FREE(cpu_states[thread]); ++ free(cpu_states); ++cleanup_present_cores: ++ free(present_cores); + + return rc; + } +diff --git a/src/ppc64_cpu.c b/src/ppc64_cpu.c +index 4017240..0233d29 100644 +--- a/src/ppc64_cpu.c ++++ b/src/ppc64_cpu.c +@@ -52,7 +52,6 @@ + + #define DSCR_DEFAULT_PATH "/sys/devices/system/cpu/dscr_default" + +-#define MAX_NR_CPUS 1024 + #define DIAGNOSTICS_RUN_MODE 42 + #define CPU_OFFLINE -1 + +@@ -266,21 +265,31 @@ static int get_one_smt_state(int core) + static int get_smt_state(void) + { + int smt_state = -1; +- int i; ++ int i, rc; ++ int *present_cores; ++ int num_present_cores; ++ ++ rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu); ++ if (rc != 0) { ++ return -1; ++ } ++ ++ for (i = 0; i < num_present_cores; i++) { ++ int cpu_state = get_one_smt_state(present_cores[i]); + +- for (i = 0; i < cpus_in_system; i++) { +- int cpu_state = get_one_smt_state(i); + if (cpu_state == 0) + continue; + + if (smt_state == -1) + smt_state = cpu_state; ++ + if (smt_state != cpu_state) { + smt_state = -1; + break; + } + } + ++ free(present_cores); + return smt_state; + } + +@@ -313,20 +322,36 @@ static int set_smt_state(int smt_state) + { + int i, j, rc = 0; + int error = 0; ++ int cpu_base, cpu_id, core_id; ++ int *present_cores = NULL; ++ int num_present_cores; + + if (!sysattr_is_writeable("online")) { + perror("Cannot set smt state"); + return -1; + } + +- for (i = 0; i < threads_in_system; i += threads_per_cpu) { ++ rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu); ++ ++ if (rc != 0) { ++ fprintf(stderr, "Failed to retrieve present core list\n"); ++ return rc; ++ } ++ ++ for (i = 0; i < num_present_cores; i++) { ++ ++ core_id = present_cores[i]; ++ cpu_base = core_id * threads_per_cpu; ++ + /* Online means any thread on this core running, so check all + * threads in the core, not just the first. */ + for (j = 0; j < threads_per_cpu; j++) { +- if (!cpu_online(i + j)) ++ cpu_id = cpu_base + j; ++ ++ if (!cpu_online(cpu_id)) + continue; + +- rc = set_one_smt_state(i, smt_state); ++ rc = set_one_smt_state(cpu_base, smt_state); + /* Record an error, but do not check result: if we + * have failed to set this core, keep trying + * subsequent ones. */ +@@ -336,10 +361,13 @@ static int set_smt_state(int smt_state) + } + } + ++ free(present_cores); ++ + if (error) { +- fprintf(stderr, "One or more cpus could not be on/offlined\n"); ++ fprintf(stderr, "One or more CPUs could not be on/offlined\n"); + return -1; + } ++ + return rc; + } + +@@ -459,8 +487,8 @@ static int do_subcores_per_core(char *state) + } + printf("Subcores per core: %d\n", subcore_state); + } else { +- /* Kernel decides what values are valid, so no need to +- * check here. */ ++ /* Kernel decides what values are valid, so no need to ++ * check here. */ + subcore_state = strtol(state, NULL, 0); + rc = set_attribute(SYSFS_SUBCORES, "%d", subcore_state); + if (rc) { +@@ -1038,7 +1066,7 @@ static int set_all_threads_off(int cpu, int smt_state) + snprintf(path, SYSFS_PATH_MAX, SYSFS_CPUDIR"/%s", i, "online"); + rc = offline_thread(path); + if (rc == -1) +- printf("Unable to take cpu%d offline", i); ++ printf("Unable to take CPU %d offline\n", i); + } + + return rc; +@@ -1065,11 +1093,13 @@ static int set_one_core(int smt_state, int core, int state) + static int do_online_cores(char *cores, int state) + { + int smt_state; +- int *core_state, *desired_core_state; ++ int *core_state = NULL, *desired_core_state = NULL; + int i, rc = 0; +- int core; ++ int core, valid = 0, core_idx = 0; + char *str, *token, *end_token; + bool first_core = true; ++ int *present_cores = NULL; ++ int num_present_cores; + + if (cores) { + if (!sysattr_is_writeable("online")) { +@@ -1083,49 +1113,62 @@ static int do_online_cores(char *cores, int state) + } + } + ++ rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu); ++ if (rc != 0) { ++ fprintf(stderr, "Failed to retrieve present core list\n"); ++ return rc; ++ } ++ + smt_state = get_smt_state(); + +- core_state = calloc(cpus_in_system, sizeof(int)); +- if (!core_state) ++ core_state = calloc(num_present_cores, sizeof(int)); ++ if (!core_state) { ++ free(present_cores); + return -ENOMEM; ++ } + +- for (i = 0; i < cpus_in_system ; i++) +- core_state[i] = (get_one_smt_state(i) > 0); ++ for (i = 0; i < num_present_cores; i++) { ++ core_state[i] = (get_one_smt_state(present_cores[i]) > 0); ++ } + + if (!cores) { + printf("Cores %s = ", state == 0 ? "offline" : "online"); +- for (i = 0; i < cpus_in_system; i++) { ++ for (i = 0; i < num_present_cores; i++) { + if (core_state[i] == state) { + if (first_core) + first_core = false; + else + printf(","); +- printf("%d", i); ++ printf("%d", present_cores[i]); + } + } + printf("\n"); + free(core_state); ++ free(present_cores); + return 0; + } + + if (smt_state == -1) { + printf("Bad or inconsistent SMT state: use ppc64_cpu --smt=on|off to set all\n" +- "cores to have the same number of online threads to continue.\n"); ++ "cores to have the same number of online threads to continue.\n"); + do_info(); ++ free(present_cores); + return -1; + } + +- desired_core_state = calloc(cpus_in_system, sizeof(int)); ++ desired_core_state = calloc(num_present_cores, sizeof(int)); + if (!desired_core_state) { + free(core_state); ++ free(present_cores); + return -ENOMEM; + } + +- for (i = 0; i < cpus_in_system; i++) ++ for (i = 0; i < num_present_cores; i++) { + /* + * Not specified on command-line + */ + desired_core_state[i] = -1; ++ } + + str = cores; + while (1) { +@@ -1141,42 +1184,57 @@ static int do_online_cores(char *cores, int state) + rc = -1; + continue; + } +- if (core >= cpus_in_system || core < 0) { ++ ++ for (i = 0; i < num_present_cores; i++) { ++ if (core == present_cores[i]) { ++ valid = 1; ++ core_idx = i; ++ break; ++ } ++ } ++ ++ if (!valid) { + printf("Invalid core to %s: %d\n", state == 0 ? "offline" : "online", core); + rc = -1; + continue; + } +- desired_core_state[core] = state; ++ ++ desired_core_state[core_idx] = state; + } + + if (rc) { +- free(core_state); +- free(desired_core_state); +- return rc; ++ goto cleanup; + } + +- for (i = 0; i < cpus_in_system; i++) { ++ for (i = 0; i < num_present_cores; i++) { + if (desired_core_state[i] != -1) { +- rc = set_one_core(smt_state, i, state); +- if (rc) ++ rc = set_one_core(smt_state, present_cores[i], state); ++ if (rc) { ++ fprintf(stderr, "Failed to set core %d to %s\n", present_cores[i], state == 0 ? "offline" : "online"); + break; ++ } + } + } + ++cleanup: + free(core_state); + free(desired_core_state); ++ free(present_cores); ++ + return rc; + } + + static int do_cores_on(char *state) + { + int smt_state; +- int *core_state; +- int cores_now_online = 0; +- int i, rc; ++ int cores_now_online = 0, core_id = 0; ++ int i, rc = 0; + int number_to_have, number_to_change = 0, number_changed = 0; ++ int *core_state = NULL; + int new_state; + char *end_state; ++ int *present_cores = NULL; ++ int num_present_cores; + + if (state) { + if (!sysattr_is_writeable("online")) { +@@ -1194,24 +1252,33 @@ static int do_cores_on(char *state) + if (!core_state) + return -ENOMEM; + +- for (i = 0; i < cpus_in_system ; i++) { +- core_state[i] = (get_one_smt_state(i) > 0); +- if (core_state[i]) ++ rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu); ++ if (rc != 0) { ++ fprintf(stderr, "Failed to retrieve present core list\n"); ++ free(core_state); ++ return rc; ++ } ++ for (i = 0; i < num_present_cores; i++) { ++ int core = present_cores[i]; ++ core_state[i] = (get_one_smt_state(core) > 0); ++ if (core_state[i]) { + cores_now_online++; ++ } + } + + if (!state) { + printf("Number of cores online = %d\n", cores_now_online); +- free(core_state); +- return 0; ++ rc = 0; ++ goto cleanup; + } + + smt_state = get_smt_state(); + if (smt_state == -1) { + printf("Bad or inconsistent SMT state: use ppc64_cpu --smt=on|off to set all\n" +- "cores to have the same number of online threads to continue.\n"); ++ "cores to have the same number of online threads to continue.\n"); + do_info(); +- return -1; ++ rc = -1; ++ goto cleanup; + } + + if (!strcmp(state, "all")) { +@@ -1227,15 +1294,16 @@ static int do_cores_on(char *state) + } + + if (number_to_have == cores_now_online) { +- free(core_state); +- return 0; ++ rc = 0; ++ goto cleanup; + } + +- if (number_to_have > cpus_in_system) { +- printf("Cannot online more cores than are present.\n"); ++ if (number_to_have <= 0 || number_to_have > cpus_in_system) { ++ printf("Error: Invalid number of cores requested: %d, possible values \ ++ should be in range: (1-%d)\n", number_to_have, cpus_in_system); + do_cores_present(); +- free(core_state); +- return -1; ++ rc = -1; ++ goto cleanup; + } + + if (number_to_have > cores_now_online) { +@@ -1248,41 +1316,50 @@ static int do_cores_on(char *state) + + if (new_state) { + for (i = 0; i < cpus_in_system; i++) { ++ core_id = present_cores[i]; + if (!core_state[i]) { +- rc = set_one_core(smt_state, i, new_state); +- if (!rc) ++ rc = set_one_core(smt_state, core_id, new_state); ++ if (!rc) { + number_changed++; +- if (number_changed >= number_to_change) ++ } ++ if (number_changed >= number_to_change) { + break; ++ } + } + } + } else { +- for (i = cpus_in_system - 1; i > 0; i--) { ++ for (i = cpus_in_system - 1; i >= 0; i--) { ++ core_id = present_cores[i]; + if (core_state[i]) { +- rc = set_one_core(smt_state, i, new_state); +- if (!rc) ++ rc = set_one_core(smt_state, core_id, new_state); ++ if (!rc) { + number_changed++; +- if (number_changed >= number_to_change) ++ } ++ if (number_changed >= number_to_change) { + break; ++ } + } + } + } + + if (number_changed != number_to_change) { + cores_now_online = 0; +- for (i = 0; i < cpus_in_system ; i++) { +- if (cpu_online(i * threads_per_cpu)) ++ for (i = 0; i < cpus_in_system; i++) { ++ core_id = present_cores[i]; ++ if (cpu_online(core_id * threads_per_cpu)) { + cores_now_online++; ++ } + } + printf("Failed to set requested number of cores online.\n" +- "Requested: %d cores, Onlined: %d cores\n", +- number_to_have, cores_now_online); +- free(core_state); +- return -1; ++ "Requested: %d cores, Onlined: %d cores\n", ++ number_to_have, cores_now_online); ++ rc = -1; + } + ++cleanup: + free(core_state); +- return 0; ++ free(present_cores); ++ return rc; + } + + static bool core_is_online(int core) +@@ -1294,35 +1371,45 @@ static int do_info(void) + { + int i, j, thread_num; + char online; +- int core, subcores = 0; ++ int subcores = 0, core_id = 0; ++ int *present_cores = NULL; ++ int num_present_cores; + +- if (is_subcore_capable()) ++ if (is_subcore_capable()) { + subcores = num_subcores(); ++ } + +- for (i = 0, core = 0; core < cpus_in_system; i++) { ++ int rc = get_present_core_list(&present_cores, &num_present_cores, threads_per_cpu); ++ if (rc != 0) { ++ fprintf(stderr, "Failed to retrieve present core list\n"); ++ return rc; ++ } + +- if (!core_is_online(i)) ++ for (i = 0; i < num_present_cores; i++) { ++ core_id = present_cores[i]; ++ if (!core_is_online(core_id)) { + continue; ++ } + + if (subcores > 1) { +- if (core % subcores == 0) +- printf("Core %3d:\n", core/subcores); +- printf(" Subcore %3d: ", core); ++ if (core_id % subcores == 0) { ++ printf("Core %3d:\n", core_id / subcores); ++ } ++ printf(" Subcore %3d: ", core_id); + } else { +- printf("Core %3d: ", core); ++ printf("Core %3d: ", core_id); + } + +- thread_num = i * threads_per_cpu; +- for (j = 0; j < threads_per_cpu; j++, thread_num++) { ++ for (j = 0; j < threads_per_cpu; j++) { ++ thread_num = core_id * threads_per_cpu + j; + online = cpu_online(thread_num) ? '*' : ' '; + printf("%4d%c ", thread_num, online); + } + printf("\n"); +- core++; + } ++ free(present_cores); + return 0; + } +- + static void usage(void) + { + printf( diff --git a/powerpc-utils.spec b/powerpc-utils.spec index fccee54..bed50b3 100644 --- a/powerpc-utils.spec +++ b/powerpc-utils.spec @@ -1,6 +1,6 @@ Name: powerpc-utils Version: 1.3.13 -Release: 2%{?dist} +Release: 3%{?dist} Summary: PERL-based scripts for maintaining and servicing PowerPC systems License: GPLv2 @@ -10,8 +10,21 @@ Source1: nx-gzip.udev Patch0: powerpc-utils-1.3.10-manpages.patch # upstream patches -Patch10: powerpc-utils-fix-return-value-from-do_replace.patch -Patch11: powerpc-utils-lparstat-print-memory-mode-correctly.patch +Patch: powerpc-utils-1.3.13-cpu_info_helpers.patch +Patch: powerpc-utils-1.3.13-ppc64_cpu-Fix-handling-of-non-contiguous-CPU-IDs.patch +Patch: powerpc-utils-fix-return-value-from-do_replace.patch +Patch: powerpc-utils-lparstat-print-memory-mode-correctly.patch +# Dynamically removing cores from an LPAR is not taking care of NUMA topology (DLPAR) +Patch: powerpc-utils-1.3.13-drmgr_Move_numa_topology_code_to_common_numa.patch +Patch: powerpc-utils-1.3.13-drmgr_Move_read_lmb-size_property_code_to_common_ofdt.patch +Patch: powerpc-utils-1.3.13-drmgr_Add_get_next_cpu_to_identify_the_removable_CPU.patch +Patch: powerpc-utils-1.3.13-drmgr_Allocate_CPU_bitmap_for_each_NUMA_node.patch +Patch: powerpc-utils-1.3.13-drmgr_Add_NUMA_configuration_update_for_CPU_remove.patch +Patch: powerpc-utils-1.3.13-drmgr_Add_NUMA_based_CPU_removal.patch +# DLPAR Memory remove/add operations fail with SAP HANA DB loaded +Patch: powerpc-utils-1.3.13-drmgr_Allow_signals_mentioned_in_new_sigset_t.patch +Patch: powerpc-utils-1.3.13-drmgr_Add_timeout_signal_handling_for_NUMA_memory_REMOVE.patch +Patch: powerpc-utils-1.3.13-drmgr_Do_not_remove_LMBs_when_the_timer_expires.patch ExclusiveArch: ppc %{power64} @@ -213,6 +226,11 @@ systemctl enable hcn-init.service >/dev/null 2>&1 || : %changelog +* Mon Aug 03 2026 Than Ngo - 1.3.13-3 +- Resolves: RHEL-184814, Dynamically removing cores from an LPAR is not taking care of NUMA topology +- Resolves: RHEL-180740, DLPAR Memory remove/add operations fail with SAP HANA DB loaded +- Resolves: RHEL-212027, Continuous migration failure error messages observed while running memhotplug + * Tue Mar 11 2025 Than Ngo - 1.3.13-2 - Resolves: RHEL-80729, drmgr: return correct value - Resolves: RHEL-80277, lparstat: print memory mode correctly