With -N left to choose, rugnux now confines the process (main thread, the CPUs RestoreThreadAffinity hands back, and what AvailableCpus counts) to the CPUs of the packages its cards hang off, and sizes its threads to them. Nothing changes where the allowed CPUs sit in one package, there is no card or a card's NUMA node is unknown, or the cards' packages already hold every allowed CPU. A mask set from outside (taskset, numactl, a cpuset) is kept: the decision works within it. An explicit -N keeps the old behaviour. On a 48-CPU server with four cards on two of its four NUMA nodes, a rotation run took 24.5-24.7 s confined to those nodes' 24 CPUs against 27.7-27.9 s on all 48 (scaling and merging 8.1 -> 6.2 s), and two larger sets moved 95-99 -> 84-87 s and 131-135 -> 113-115 s. On a single-socket machine this is a no-op. The decision itself (CpusOfGpuPackages) is a pure function, tested on synthetic 4-package, 2-package hyper-threaded, cards-on-both, no-card and restricted-mask topologies. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EizBKhTqrAYago9KJG3dA3
190 lines
6.1 KiB
Plaintext
190 lines
6.1 KiB
Plaintext
// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include <fstream>
|
|
#include <thread>
|
|
#include <atomic>
|
|
#include <mutex>
|
|
#include <vector>
|
|
|
|
#include <cuda.h>
|
|
|
|
#include "CUDAWrapper.h"
|
|
#include "ThreadAffinity.h"
|
|
#include "JFJochException.h"
|
|
|
|
inline void cuda_err(cudaError_t val) {
|
|
if (val != cudaSuccess)
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
|
|
}
|
|
|
|
int32_t get_gpu_count() {
|
|
int device_count;
|
|
cudaError_t val = cudaGetDeviceCount(&device_count);
|
|
switch (val) {
|
|
case cudaSuccess:
|
|
return device_count;
|
|
case cudaErrorNoDevice:
|
|
case cudaErrorInsufficientDriver:
|
|
return 0;
|
|
default:
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
|
|
}
|
|
|
|
}
|
|
|
|
std::vector<std::string> get_gpu_names() {
|
|
std::vector<std::string> names;
|
|
const int32_t count = get_gpu_count();
|
|
names.reserve(count);
|
|
for (int32_t i = 0; i < count; i++) {
|
|
cudaDeviceProp prop{};
|
|
// A device that cannot be queried still exists and still gets work, so it is listed - just
|
|
// without a name. Losing the whole list over one unreadable device would be worse.
|
|
if (cudaGetDeviceProperties(&prop, i) == cudaSuccess)
|
|
names.emplace_back(prop.name);
|
|
else
|
|
names.emplace_back("unknown GPU");
|
|
}
|
|
return names;
|
|
}
|
|
|
|
size_t get_gpu_total_memory(int32_t dev_id) {
|
|
cudaDeviceProp prop{};
|
|
if (cudaGetDeviceProperties(&prop, dev_id) != cudaSuccess)
|
|
return 0;
|
|
return prop.totalGlobalMem;
|
|
}
|
|
|
|
void set_gpu(int32_t dev_id) {
|
|
auto dev_count = get_gpu_count();
|
|
|
|
// Ignore if no GPU present
|
|
if (dev_count > 0) {
|
|
if ((dev_id < 0) || (dev_id >= dev_count))
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Device ID cannot be negative");
|
|
|
|
cuda_err(cudaSetDevice(dev_id));
|
|
}
|
|
}
|
|
|
|
namespace {
|
|
std::atomic<bool> gpu_numa_binding{false};
|
|
|
|
// The NUMA node of each device, read once.
|
|
int GpuNumaNode(int32_t dev_id) {
|
|
static std::mutex m;
|
|
static std::vector<int> node;
|
|
std::lock_guard<std::mutex> lock(m);
|
|
if (node.empty()) {
|
|
const int32_t count = get_gpu_count();
|
|
node.assign(count, -1);
|
|
for (int32_t i = 0; i < count; i++) {
|
|
char bus_id[32] = {};
|
|
if (cudaDeviceGetPCIBusId(bus_id, sizeof(bus_id), i) == cudaSuccess)
|
|
node[i] = NumaNodeOfPciDevice(bus_id);
|
|
}
|
|
}
|
|
return dev_id < static_cast<int32_t>(node.size()) ? node[dev_id] : -1;
|
|
}
|
|
}
|
|
|
|
std::vector<int> get_gpu_numa_nodes() {
|
|
std::vector<int> nodes;
|
|
for (int32_t i = 0; i < get_gpu_count(); i++)
|
|
nodes.push_back(GpuNumaNode(i));
|
|
return nodes;
|
|
}
|
|
|
|
void enable_gpu_numa_binding() {
|
|
gpu_numa_binding = true;
|
|
}
|
|
|
|
void pin_gpu(int32_t dev_id) {
|
|
if (get_gpu_count() == 0)
|
|
return;
|
|
set_gpu(dev_id);
|
|
if (gpu_numa_binding)
|
|
PinThreadToNumaNode(GpuNumaNode(dev_id));
|
|
}
|
|
|
|
ScopedGpuPin::ScopedGpuPin(int32_t dev_id) {
|
|
if (get_gpu_count() == 0)
|
|
return;
|
|
cuda_err(cudaGetDevice(&prev_device));
|
|
const int node = gpu_numa_binding ? GpuNumaNode(dev_id) : -1;
|
|
if (node >= 0)
|
|
prev_cpus = GetThreadCpus();
|
|
set_gpu(dev_id);
|
|
PinThreadToNumaNode(node);
|
|
}
|
|
|
|
ScopedGpuPin::~ScopedGpuPin() {
|
|
SetThreadCpus(prev_cpus);
|
|
if (prev_device >= 0)
|
|
cudaSetDevice(prev_device); // not set_gpu: a destructor must not throw
|
|
}
|
|
|
|
void pin_gpu() {
|
|
static std::atomic<uint32_t> counter{0};
|
|
auto dev_count = get_gpu_count();
|
|
if (dev_count > 0)
|
|
pin_gpu(static_cast<int32_t>(counter.fetch_add(1) % dev_count));
|
|
}
|
|
|
|
void set_gpu_blocking_sync() {
|
|
// Through the driver's primary-context flags, which a context created later takes: the runtime's
|
|
// cudaSetDevice creates the device's context, serially, and on four cards that was most of a
|
|
// second before the run had started. The function is taken from the driver at run time, so the
|
|
// program does not link against it and still starts on a machine without one.
|
|
void *fn = nullptr;
|
|
cudaDriverEntryPointQueryResult found{};
|
|
if (get_gpu_count() == 0
|
|
|| cudaGetDriverEntryPointByVersion("cuDevicePrimaryCtxSetFlags", &fn, 12000, cudaEnableDefault,
|
|
&found) != cudaSuccess
|
|
|| found != cudaDriverEntryPointSuccess) {
|
|
cudaGetLastError();
|
|
return;
|
|
}
|
|
const auto set_flags = reinterpret_cast<CUresult (*)(CUdevice, unsigned int)>(fn);
|
|
for (int32_t i = 0; i < get_gpu_count(); i++)
|
|
set_flags(i, CU_CTX_SCHED_BLOCKING_SYNC);
|
|
}
|
|
|
|
void create_gpu_contexts(int32_t first_device) {
|
|
std::vector<std::thread> threads;
|
|
for (int32_t i = first_device; i < get_gpu_count(); i++)
|
|
threads.emplace_back([i] {
|
|
// A device that fails here fails again, with its error, where it is used.
|
|
if (cudaSetDevice(i) != cudaSuccess || cudaFree(nullptr) != cudaSuccess)
|
|
cudaGetLastError();
|
|
});
|
|
for (auto &t : threads)
|
|
t.join();
|
|
}
|
|
|
|
void release_gpu_contexts() {
|
|
std::vector<std::thread> threads;
|
|
for (int32_t i = 0; i < get_gpu_count(); i++)
|
|
threads.emplace_back([i] {
|
|
if (cudaSetDevice(i) != cudaSuccess || cudaDeviceReset() != cudaSuccess)
|
|
cudaGetLastError();
|
|
});
|
|
for (auto &t : threads)
|
|
t.join();
|
|
}
|
|
|
|
void cuda_clear_error() {
|
|
cudaGetLastError();
|
|
}
|
|
|
|
void cuda_throw_if_context_lost() {
|
|
// cudaFree(nullptr) frees nothing, but a lost context fails it. The last error cannot say the same:
|
|
// once cudaGetLastError() has returned a sticky error, it and cudaPeekAtLastError() report success.
|
|
const cudaError_t err = cudaFree(nullptr);
|
|
if (err != cudaSuccess)
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
|
|
std::string("CUDA device unusable after an unrecoverable error: ")
|
|
+ cudaGetErrorString(err));
|
|
}
|