|
[Date Prev][Date Next][Thread Prev][Thread Next][Date Index][Thread Index] [RFC PATCH 4/4] xenpm: Rework get-core-temp around CPU status map and topology
Current get-core-temp core selection logic has some corner cases,
in particular when CPUs are offlined.
When some CPUs are offline, we observe various issues :
- XENPF_resource_op fails with ENODEV on offline CPUs
- cores_per_socket can be incorrect if there are offline CPUs on
the first socket, leading to incorrect socket count calculation
Rather than trying to workaround the problem, rework the algorithm to
check if the CPU is online using XEN_SYSCTL_get_cpu_status_map and
use CPU topology to select CPU sockets in a more robust way.
Signed-off-by: Teddy Astie <teddy.astie@xxxxxxxxxx>
---
I'm not happy with the way sockets are handled. Using a bitmap
would be a better option, but it would make things more complicated
than it already is.
tools/misc/xenpm.c | 67 +++++++++++++++++++++++++++++++++++++++++-----
1 file changed, 60 insertions(+), 7 deletions(-)
diff --git a/tools/misc/xenpm.c b/tools/misc/xenpm.c
index ecb39c911d..f6d1241496 100644
--- a/tools/misc/xenpm.c
+++ b/tools/misc/xenpm.c
@@ -1419,9 +1419,11 @@ static int fetch_dts_temp(xc_interface *xch, uint32_t
cpu, bool package, int *te
static void get_core_temp(int argc, char *argv[])
{
- int temp = -1, cpu = -1;
- unsigned int socket;
+ int temp = -1, cpu = -1, socket = -1;
bool has_data = false;
+ xc_cputopo_t *cputopo = NULL;
+ unsigned int max_cpus = max_cpu_nr;
+ xc_cpumap_t cpu_online_map = NULL;
if ( argc > 0 )
parse_cpuid(argv[0], &cpu);
@@ -1439,25 +1441,72 @@ static void get_core_temp(int argc, char *argv[])
return;
}
+ cputopo = calloc(max_cpu_nr, sizeof(*cputopo));
+ if ( !cputopo )
+ {
+ fprintf(stderr, "Unable to allocate CPU topology list\n");
+ goto out;
+ }
+
+ cpu_online_map = xc_cpumap_alloc(xc_handle);
+ if ( !cpu_online_map )
+ {
+ fprintf(stderr, "Unable to allocate CPU status bitmap\n");
+ goto out;
+ }
+
+ if ( xc_cputopoinfo(xc_handle, &max_cpus, cputopo) )
+ {
+ fprintf(stderr,
+ "Unable to get CPU topology (%d - %s)\n",
+ errno, strerror(errno));
+ goto out;
+ }
+
+ if ( (xc_get_cpu_status_map(xc_handle, max_cpu_nr, cpu_online_map,
+ NULL, NULL)) )
+ {
+ fprintf(stderr, "Unable to get CPU online map (%d - %s)\n",
+ errno, strerror(errno));
+ /* Assume all populated */
+ memset(cpu_online_map, 0xff, xc_get_cpumap_size(xc_handle));
+ }
+
+ if ( max_cpus > max_cpu_nr )
+ {
+ fprintf(stderr, "xc_cputopoinfo returned more CPU than physinfo ?\n");
+ max_cpus = max_cpu_nr;
+ }
+
/* Per socket measurement */
- for ( socket = 0, cpu = 0; cpu < max_cpu_nr;
- socket++, cpu += physinfo.cores_per_socket *
physinfo.threads_per_core )
+ for ( cpu = 0; cpu < max_cpus; cpu++ )
{
+ if ( !xc_cpumap_testcpu(cpu, cpu_online_map) ||
+ cputopo->core == XEN_INVALID_CORE_ID ||
+ (int)(cputopo[cpu].socket) <= socket )
+ continue;
+
+ socket = cputopo[cpu].socket;
+
if ( fetch_dts_temp(xc_handle, cpu, true, &temp) )
{
fprintf(stderr,
"[Package%u] Unable to fetch temperature (%d - %s)\n",
- cpu, errno, strerror(errno));
+ socket, errno, strerror(errno));
/* CPU may not support package temperatures, but still support DTS
*/
break;
}
has_data = true;
- printf("Package%u: %d°C\n", socket, temp);
+ printf("Package%d: %d°C\n", socket, temp);
}
- for ( cpu = 0; cpu < max_cpu_nr; cpu += physinfo.threads_per_core )
+ for ( cpu = 0; cpu < max_cpus; cpu += physinfo.threads_per_core )
{
+ if ( !xc_cpumap_testcpu(cpu, cpu_online_map) ||
+ cputopo[cpu].core == XEN_INVALID_CORE_ID )
+ continue;
+
if ( fetch_dts_temp(xc_handle, cpu, false, &temp) )
{
fprintf(stderr, "[CPU%d] Unable to fetch temperature (%d - %s)\n",
@@ -1469,6 +1518,10 @@ static void get_core_temp(int argc, char *argv[])
printf("CPU%d: %d°C\n", cpu, temp);
}
+out:
+ free(cputopo);
+ free(cpu_online_map);
+
if ( !has_data )
exit(EXIT_FAILURE);
}
--
2.55.0
--
Teddy Astie | Vates XCP-ng Developer
XCP-ng & Xen Orchestra - Vates solutions
web: https://vates.tech
|
![]() |
Lists.xenproject.org is hosted with RackSpace, monitoring our |