GPU threads: block instead of spin, and stay on the GPU's NUMA node
- set_gpu_blocking_sync(): every device is put in cudaDeviceScheduleBlockingSync before its context exists, so a host thread waiting on the GPU sleeps instead of spinning on a core. On a 16M rotation run a fifth of all CPU time was that spinning; wall time unchanged within noise. Called first thing in rugnux. - enable_gpu_numa_binding(): from then on pin_gpu() (and the new pin_gpu(dev), used by the first-pass spot workers that take a card by index) also keeps the thread on the CPUs of the NUMA node the card hangs off. The node and its CPUs come from /sys (no libnuma), intersected with the process's own mask; Linux only, and nothing happens on a machine with a single node. rugnux turns it on; the broker does not. - A thread inherits its creator's affinity, so the shared ParallelFor pool would run every later pass on one socket if a pinned worker created it: its threads now reset to the mask the process started with (common/ThreadAffinity). Byte-identical output. The NUMA part is a no-op on the single-node test box and still has to be measured on a two-socket machine. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
This commit is contained in:
@@ -0,0 +1,104 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#include "ThreadAffinity.h"
|
||||
|
||||
#ifdef __linux__
|
||||
|
||||
#include <cstdio>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <sstream>
|
||||
|
||||
#include <pthread.h>
|
||||
#include <sched.h>
|
||||
|
||||
namespace {
|
||||
// The process's CPUs, taken before any thread has been pinned: static initialisation runs on the
|
||||
// main thread before main().
|
||||
cpu_set_t StartMask() {
|
||||
cpu_set_t mask;
|
||||
CPU_ZERO(&mask);
|
||||
if (sched_getaffinity(0, sizeof(mask), &mask) != 0)
|
||||
for (int c = 0; c < CPU_SETSIZE; c++)
|
||||
CPU_SET(c, &mask);
|
||||
return mask;
|
||||
}
|
||||
const cpu_set_t start_mask = StartMask();
|
||||
|
||||
int NumaNodeCount() {
|
||||
int n = 0;
|
||||
std::error_code ec;
|
||||
for (const auto &e : std::filesystem::directory_iterator("/sys/devices/system/node", ec)) {
|
||||
const std::string name = e.path().filename().string();
|
||||
if (name.rfind("node", 0) == 0 && name.size() > 4
|
||||
&& name.find_first_not_of("0123456789", 4) == std::string::npos)
|
||||
n++;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
// "0-23,48-71" -> the CPUs it names.
|
||||
bool ParseCpuList(const std::string &list, cpu_set_t &out) {
|
||||
CPU_ZERO(&out);
|
||||
std::stringstream ss(list);
|
||||
std::string range;
|
||||
bool any = false;
|
||||
while (std::getline(ss, range, ',')) {
|
||||
int lo, hi;
|
||||
if (std::sscanf(range.c_str(), "%d-%d", &lo, &hi) == 2) {
|
||||
} else if (std::sscanf(range.c_str(), "%d", &lo) == 1) {
|
||||
hi = lo;
|
||||
} else
|
||||
continue;
|
||||
for (int c = lo; c <= hi && c < CPU_SETSIZE; c++) {
|
||||
CPU_SET(c, &out);
|
||||
any = true;
|
||||
}
|
||||
}
|
||||
return any;
|
||||
}
|
||||
}
|
||||
|
||||
int NumaNodeOfPciDevice(const std::string &pci_bus_id) {
|
||||
if (NumaNodeCount() < 2)
|
||||
return -1;
|
||||
unsigned domain, bus, device, function;
|
||||
if (std::sscanf(pci_bus_id.c_str(), "%x:%x:%x.%x", &domain, &bus, &device, &function) != 4)
|
||||
return -1;
|
||||
char path[128];
|
||||
std::snprintf(path, sizeof(path), "/sys/bus/pci/devices/%04x:%02x:%02x.%x/numa_node",
|
||||
domain, bus, device, function);
|
||||
std::ifstream f(path);
|
||||
int node = -1;
|
||||
if (!(f >> node))
|
||||
return -1;
|
||||
return node;
|
||||
}
|
||||
|
||||
void PinThreadToNumaNode(int node) {
|
||||
if (node < 0)
|
||||
return;
|
||||
std::ifstream f("/sys/devices/system/node/node" + std::to_string(node) + "/cpulist");
|
||||
std::string list;
|
||||
cpu_set_t node_cpus;
|
||||
if (!std::getline(f, list) || !ParseCpuList(list, node_cpus))
|
||||
return;
|
||||
cpu_set_t mask;
|
||||
CPU_AND(&mask, &node_cpus, &start_mask);
|
||||
if (CPU_COUNT(&mask) == 0)
|
||||
return;
|
||||
pthread_setaffinity_np(pthread_self(), sizeof(mask), &mask);
|
||||
}
|
||||
|
||||
void RestoreThreadAffinity() {
|
||||
pthread_setaffinity_np(pthread_self(), sizeof(start_mask), &start_mask);
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
int NumaNodeOfPciDevice(const std::string &) { return -1; }
|
||||
void PinThreadToNumaNode(int) {}
|
||||
void RestoreThreadAffinity() {}
|
||||
|
||||
#endif
|
||||
Reference in New Issue
Block a user