- set_gpu_blocking_sync(): every device is put in cudaDeviceScheduleBlockingSync before its context exists, so a host thread waiting on the GPU sleeps instead of spinning on a core. On a 16M rotation run a fifth of all CPU time was that spinning; wall time unchanged within noise. Called first thing in rugnux. - enable_gpu_numa_binding(): from then on pin_gpu() (and the new pin_gpu(dev), used by the first-pass spot workers that take a card by index) also keeps the thread on the CPUs of the NUMA node the card hangs off. The node and its CPUs come from /sys (no libnuma), intersected with the process's own mask; Linux only, and nothing happens on a machine with a single node. rugnux turns it on; the broker does not. - A thread inherits its creator's affinity, so the shared ParallelFor pool would run every later pass on one socket if a pinned worker created it: its threads now reset to the mask the process started with (common/ThreadAffinity). Byte-identical output. The NUMA part is a no-op on the single-node test box and still has to be measured on a two-socket machine. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
117 lines
3.4 KiB
Plaintext
117 lines
3.4 KiB
Plaintext
// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include <fstream>
|
|
#include <atomic>
|
|
#include <mutex>
|
|
#include <vector>
|
|
|
|
#include "CUDAWrapper.h"
|
|
#include "ThreadAffinity.h"
|
|
#include "JFJochException.h"
|
|
|
|
inline void cuda_err(cudaError_t val) {
|
|
if (val != cudaSuccess)
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
|
|
}
|
|
|
|
int32_t get_gpu_count() {
|
|
int device_count;
|
|
cudaError_t val = cudaGetDeviceCount(&device_count);
|
|
switch (val) {
|
|
case cudaSuccess:
|
|
return device_count;
|
|
case cudaErrorNoDevice:
|
|
case cudaErrorInsufficientDriver:
|
|
return 0;
|
|
default:
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
|
|
}
|
|
|
|
}
|
|
|
|
std::vector<std::string> get_gpu_names() {
|
|
std::vector<std::string> names;
|
|
const int32_t count = get_gpu_count();
|
|
names.reserve(count);
|
|
for (int32_t i = 0; i < count; i++) {
|
|
cudaDeviceProp prop{};
|
|
// A device that cannot be queried still exists and still gets work, so it is listed - just
|
|
// without a name. Losing the whole list over one unreadable device would be worse.
|
|
if (cudaGetDeviceProperties(&prop, i) == cudaSuccess)
|
|
names.emplace_back(prop.name);
|
|
else
|
|
names.emplace_back("unknown GPU");
|
|
}
|
|
return names;
|
|
}
|
|
|
|
void set_gpu(int32_t dev_id) {
|
|
auto dev_count = get_gpu_count();
|
|
|
|
// Ignore if no GPU present
|
|
if (dev_count > 0) {
|
|
if ((dev_id < 0) || (dev_id >= dev_count))
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Device ID cannot be negative");
|
|
|
|
cuda_err(cudaSetDevice(dev_id));
|
|
}
|
|
}
|
|
|
|
namespace {
|
|
std::atomic<bool> gpu_numa_binding{false};
|
|
|
|
// The NUMA node of each device, read once.
|
|
int GpuNumaNode(int32_t dev_id) {
|
|
static std::mutex m;
|
|
static std::vector<int> node;
|
|
std::lock_guard lock(m);
|
|
if (node.empty()) {
|
|
const int32_t count = get_gpu_count();
|
|
node.assign(count, -1);
|
|
for (int32_t i = 0; i < count; i++) {
|
|
char bus_id[32] = {};
|
|
if (cudaDeviceGetPCIBusId(bus_id, sizeof(bus_id), i) == cudaSuccess)
|
|
node[i] = NumaNodeOfPciDevice(bus_id);
|
|
}
|
|
}
|
|
return dev_id < static_cast<int32_t>(node.size()) ? node[dev_id] : -1;
|
|
}
|
|
}
|
|
|
|
void enable_gpu_numa_binding() {
|
|
gpu_numa_binding = true;
|
|
}
|
|
|
|
void pin_gpu(int32_t dev_id) {
|
|
if (get_gpu_count() == 0)
|
|
return;
|
|
set_gpu(dev_id);
|
|
if (gpu_numa_binding)
|
|
PinThreadToNumaNode(GpuNumaNode(dev_id));
|
|
}
|
|
|
|
void pin_gpu() {
|
|
static std::atomic<uint32_t> counter{0};
|
|
auto dev_count = get_gpu_count();
|
|
if (dev_count > 0)
|
|
pin_gpu(static_cast<int32_t>(counter.fetch_add(1) % dev_count));
|
|
}
|
|
|
|
void set_gpu_blocking_sync() {
|
|
const int32_t count = get_gpu_count();
|
|
for (int32_t i = 0; i < count; i++) {
|
|
if (cudaSetDevice(i) != cudaSuccess)
|
|
continue;
|
|
// cudaErrorSetOnActiveProcess where a context exists already: it keeps its flags.
|
|
if (cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync) != cudaSuccess)
|
|
cudaGetLastError();
|
|
}
|
|
if (count > 0)
|
|
cudaSetDevice(0);
|
|
}
|
|
|
|
void cuda_clear_error() {
|
|
cudaGetLastError();
|
|
}
|