Files
Jungfraujoch/common/ThreadAffinity.cpp
leonarski_fandClaude Opus 5.5 4b1d7feebd GPU threads: block instead of spin, and stay on the GPU's NUMA node
- set_gpu_blocking_sync(): every device is put in
  cudaDeviceScheduleBlockingSync before its context exists, so a host thread
  waiting on the GPU sleeps instead of spinning on a core. On a 16M rotation
  run a fifth of all CPU time was that spinning; wall time unchanged within
  noise. Called first thing in rugnux.
- enable_gpu_numa_binding(): from then on pin_gpu() (and the new
  pin_gpu(dev), used by the first-pass spot workers that take a card by
  index) also keeps the thread on the CPUs of the NUMA node the card hangs
  off. The node and its CPUs come from /sys (no libnuma), intersected with
  the process's own mask; Linux only, and nothing happens on a machine with a
  single node. rugnux turns it on; the broker does not.
- A thread inherits its creator's affinity, so the shared ParallelFor pool
  would run every later pass on one socket if a pinned worker created it:
  its threads now reset to the mask the process started with
  (common/ThreadAffinity).

Byte-identical output. The NUMA part is a no-op on the single-node test box
and still has to be measured on a two-socket machine.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
2026-09-26 19:30:01 +02:00

105 lines
3.0 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "ThreadAffinity.h"
#ifdef __linux__
#include <cstdio>
#include <filesystem>
#include <fstream>
#include <sstream>
#include <pthread.h>
#include <sched.h>
namespace {
// The process's CPUs, taken before any thread has been pinned: static initialisation runs on the
// main thread before main().
cpu_set_t StartMask() {
cpu_set_t mask;
CPU_ZERO(&mask);
if (sched_getaffinity(0, sizeof(mask), &mask) != 0)
for (int c = 0; c < CPU_SETSIZE; c++)
CPU_SET(c, &mask);
return mask;
}
const cpu_set_t start_mask = StartMask();
int NumaNodeCount() {
int n = 0;
std::error_code ec;
for (const auto &e : std::filesystem::directory_iterator("/sys/devices/system/node", ec)) {
const std::string name = e.path().filename().string();
if (name.rfind("node", 0) == 0 && name.size() > 4
&& name.find_first_not_of("0123456789", 4) == std::string::npos)
n++;
}
return n;
}
// "0-23,48-71" -> the CPUs it names.
bool ParseCpuList(const std::string &list, cpu_set_t &out) {
CPU_ZERO(&out);
std::stringstream ss(list);
std::string range;
bool any = false;
while (std::getline(ss, range, ',')) {
int lo, hi;
if (std::sscanf(range.c_str(), "%d-%d", &lo, &hi) == 2) {
} else if (std::sscanf(range.c_str(), "%d", &lo) == 1) {
hi = lo;
} else
continue;
for (int c = lo; c <= hi && c < CPU_SETSIZE; c++) {
CPU_SET(c, &out);
any = true;
}
}
return any;
}
}
int NumaNodeOfPciDevice(const std::string &pci_bus_id) {
if (NumaNodeCount() < 2)
return -1;
unsigned domain, bus, device, function;
if (std::sscanf(pci_bus_id.c_str(), "%x:%x:%x.%x", &domain, &bus, &device, &function) != 4)
return -1;
char path[128];
std::snprintf(path, sizeof(path), "/sys/bus/pci/devices/%04x:%02x:%02x.%x/numa_node",
domain, bus, device, function);
std::ifstream f(path);
int node = -1;
if (!(f >> node))
return -1;
return node;
}
void PinThreadToNumaNode(int node) {
if (node < 0)
return;
std::ifstream f("/sys/devices/system/node/node" + std::to_string(node) + "/cpulist");
std::string list;
cpu_set_t node_cpus;
if (!std::getline(f, list) || !ParseCpuList(list, node_cpus))
return;
cpu_set_t mask;
CPU_AND(&mask, &node_cpus, &start_mask);
if (CPU_COUNT(&mask) == 0)
return;
pthread_setaffinity_np(pthread_self(), sizeof(mask), &mask);
}
void RestoreThreadAffinity() {
pthread_setaffinity_np(pthread_self(), sizeof(start_mask), &start_mask);
}
#else
int NumaNodeOfPciDevice(const std::string &) { return -1; }
void PinThreadToNumaNode(int) {}
void RestoreThreadAffinity() {}
#endif