Files
leonarski_fandClaude Opus 5.5 3c5ff3f5df rugnux: on a machine of several sockets, keep to the sockets the GPUs are on
With -N left to choose, rugnux now confines the process (main thread, the CPUs
RestoreThreadAffinity hands back, and what AvailableCpus counts) to the CPUs of
the packages its cards hang off, and sizes its threads to them. Nothing changes
where the allowed CPUs sit in one package, there is no card or a card's NUMA
node is unknown, or the cards' packages already hold every allowed CPU. A mask
set from outside (taskset, numactl, a cpuset) is kept: the decision works
within it. An explicit -N keeps the old behaviour.

On a 48-CPU server with four cards on two of its four NUMA nodes, a rotation
run took 24.5-24.7 s confined to those nodes' 24 CPUs against 27.7-27.9 s on
all 48 (scaling and merging 8.1 -> 6.2 s), and two larger sets moved
95-99 -> 84-87 s and 131-135 -> 113-115 s. On a single-socket machine this is
a no-op.

The decision itself (CpusOfGpuPackages) is a pure function, tested on
synthetic 4-package, 2-package hyper-threaded, cards-on-both, no-card and
restricted-mask topologies.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01EizBKhTqrAYago9KJG3dA3
2026-10-11 09:27:32 +02:00

190 lines
6.1 KiB
Plaintext

// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <fstream>
#include <thread>
#include <atomic>
#include <mutex>
#include <vector>
#include <cuda.h>
#include "CUDAWrapper.h"
#include "ThreadAffinity.h"
#include "JFJochException.h"
inline void cuda_err(cudaError_t val) {
if (val != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
}
int32_t get_gpu_count() {
int device_count;
cudaError_t val = cudaGetDeviceCount(&device_count);
switch (val) {
case cudaSuccess:
return device_count;
case cudaErrorNoDevice:
case cudaErrorInsufficientDriver:
return 0;
default:
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
}
}
std::vector<std::string> get_gpu_names() {
std::vector<std::string> names;
const int32_t count = get_gpu_count();
names.reserve(count);
for (int32_t i = 0; i < count; i++) {
cudaDeviceProp prop{};
// A device that cannot be queried still exists and still gets work, so it is listed - just
// without a name. Losing the whole list over one unreadable device would be worse.
if (cudaGetDeviceProperties(&prop, i) == cudaSuccess)
names.emplace_back(prop.name);
else
names.emplace_back("unknown GPU");
}
return names;
}
size_t get_gpu_total_memory(int32_t dev_id) {
cudaDeviceProp prop{};
if (cudaGetDeviceProperties(&prop, dev_id) != cudaSuccess)
return 0;
return prop.totalGlobalMem;
}
void set_gpu(int32_t dev_id) {
auto dev_count = get_gpu_count();
// Ignore if no GPU present
if (dev_count > 0) {
if ((dev_id < 0) || (dev_id >= dev_count))
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Device ID cannot be negative");
cuda_err(cudaSetDevice(dev_id));
}
}
namespace {
std::atomic<bool> gpu_numa_binding{false};
// The NUMA node of each device, read once.
int GpuNumaNode(int32_t dev_id) {
static std::mutex m;
static std::vector<int> node;
std::lock_guard<std::mutex> lock(m);
if (node.empty()) {
const int32_t count = get_gpu_count();
node.assign(count, -1);
for (int32_t i = 0; i < count; i++) {
char bus_id[32] = {};
if (cudaDeviceGetPCIBusId(bus_id, sizeof(bus_id), i) == cudaSuccess)
node[i] = NumaNodeOfPciDevice(bus_id);
}
}
return dev_id < static_cast<int32_t>(node.size()) ? node[dev_id] : -1;
}
}
std::vector<int> get_gpu_numa_nodes() {
std::vector<int> nodes;
for (int32_t i = 0; i < get_gpu_count(); i++)
nodes.push_back(GpuNumaNode(i));
return nodes;
}
void enable_gpu_numa_binding() {
gpu_numa_binding = true;
}
void pin_gpu(int32_t dev_id) {
if (get_gpu_count() == 0)
return;
set_gpu(dev_id);
if (gpu_numa_binding)
PinThreadToNumaNode(GpuNumaNode(dev_id));
}
ScopedGpuPin::ScopedGpuPin(int32_t dev_id) {
if (get_gpu_count() == 0)
return;
cuda_err(cudaGetDevice(&prev_device));
const int node = gpu_numa_binding ? GpuNumaNode(dev_id) : -1;
if (node >= 0)
prev_cpus = GetThreadCpus();
set_gpu(dev_id);
PinThreadToNumaNode(node);
}
ScopedGpuPin::~ScopedGpuPin() {
SetThreadCpus(prev_cpus);
if (prev_device >= 0)
cudaSetDevice(prev_device); // not set_gpu: a destructor must not throw
}
void pin_gpu() {
static std::atomic<uint32_t> counter{0};
auto dev_count = get_gpu_count();
if (dev_count > 0)
pin_gpu(static_cast<int32_t>(counter.fetch_add(1) % dev_count));
}
void set_gpu_blocking_sync() {
// Through the driver's primary-context flags, which a context created later takes: the runtime's
// cudaSetDevice creates the device's context, serially, and on four cards that was most of a
// second before the run had started. The function is taken from the driver at run time, so the
// program does not link against it and still starts on a machine without one.
void *fn = nullptr;
cudaDriverEntryPointQueryResult found{};
if (get_gpu_count() == 0
|| cudaGetDriverEntryPointByVersion("cuDevicePrimaryCtxSetFlags", &fn, 12000, cudaEnableDefault,
&found) != cudaSuccess
|| found != cudaDriverEntryPointSuccess) {
cudaGetLastError();
return;
}
const auto set_flags = reinterpret_cast<CUresult (*)(CUdevice, unsigned int)>(fn);
for (int32_t i = 0; i < get_gpu_count(); i++)
set_flags(i, CU_CTX_SCHED_BLOCKING_SYNC);
}
void create_gpu_contexts(int32_t first_device) {
std::vector<std::thread> threads;
for (int32_t i = first_device; i < get_gpu_count(); i++)
threads.emplace_back([i] {
// A device that fails here fails again, with its error, where it is used.
if (cudaSetDevice(i) != cudaSuccess || cudaFree(nullptr) != cudaSuccess)
cudaGetLastError();
});
for (auto &t : threads)
t.join();
}
void release_gpu_contexts() {
std::vector<std::thread> threads;
for (int32_t i = 0; i < get_gpu_count(); i++)
threads.emplace_back([i] {
if (cudaSetDevice(i) != cudaSuccess || cudaDeviceReset() != cudaSuccess)
cudaGetLastError();
});
for (auto &t : threads)
t.join();
}
void cuda_clear_error() {
cudaGetLastError();
}
void cuda_throw_if_context_lost() {
// cudaFree(nullptr) frees nothing, but a lost context fails it. The last error cannot say the same:
// once cudaGetLastError() has returned a sticky error, it and cudaPeekAtLastError() report success.
const cudaError_t err = cudaFree(nullptr);
if (err != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
std::string("CUDA device unusable after an unrecoverable error: ")
+ cudaGetErrorString(err));
}