// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #include #include #include #include #include "CUDAWrapper.h" #include "ThreadAffinity.h" #include "JFJochException.h" inline void cuda_err(cudaError_t val) { if (val != cudaSuccess) throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val)); } int32_t get_gpu_count() { int device_count; cudaError_t val = cudaGetDeviceCount(&device_count); switch (val) { case cudaSuccess: return device_count; case cudaErrorNoDevice: case cudaErrorInsufficientDriver: return 0; default: throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val)); } } std::vector get_gpu_names() { std::vector names; const int32_t count = get_gpu_count(); names.reserve(count); for (int32_t i = 0; i < count; i++) { cudaDeviceProp prop{}; // A device that cannot be queried still exists and still gets work, so it is listed - just // without a name. Losing the whole list over one unreadable device would be worse. if (cudaGetDeviceProperties(&prop, i) == cudaSuccess) names.emplace_back(prop.name); else names.emplace_back("unknown GPU"); } return names; } void set_gpu(int32_t dev_id) { auto dev_count = get_gpu_count(); // Ignore if no GPU present if (dev_count > 0) { if ((dev_id < 0) || (dev_id >= dev_count)) throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Device ID cannot be negative"); cuda_err(cudaSetDevice(dev_id)); } } namespace { std::atomic gpu_numa_binding{false}; // The NUMA node of each device, read once. int GpuNumaNode(int32_t dev_id) { static std::mutex m; static std::vector node; std::lock_guard lock(m); if (node.empty()) { const int32_t count = get_gpu_count(); node.assign(count, -1); for (int32_t i = 0; i < count; i++) { char bus_id[32] = {}; if (cudaDeviceGetPCIBusId(bus_id, sizeof(bus_id), i) == cudaSuccess) node[i] = NumaNodeOfPciDevice(bus_id); } } return dev_id < static_cast(node.size()) ? node[dev_id] : -1; } } void enable_gpu_numa_binding() { gpu_numa_binding = true; } void pin_gpu(int32_t dev_id) { if (get_gpu_count() == 0) return; set_gpu(dev_id); if (gpu_numa_binding) PinThreadToNumaNode(GpuNumaNode(dev_id)); } void pin_gpu() { static std::atomic counter{0}; auto dev_count = get_gpu_count(); if (dev_count > 0) pin_gpu(static_cast(counter.fetch_add(1) % dev_count)); } void set_gpu_blocking_sync() { const int32_t count = get_gpu_count(); for (int32_t i = 0; i < count; i++) { if (cudaSetDevice(i) != cudaSuccess) continue; // cudaErrorSetOnActiveProcess where a context exists already: it keeps its flags. if (cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync) != cudaSuccess) cudaGetLastError(); } if (count > 0) cudaSetDevice(0); } void cuda_clear_error() { cudaGetLastError(); } void cuda_throw_if_context_lost() { // cudaFree(nullptr) frees nothing, but a lost context fails it. The last error cannot say the same: // once cudaGetLastError() has returned a sticky error, it and cudaPeekAtLastError() report success. const cudaError_t err = cudaFree(nullptr); if (err != cudaSuccess) throw JFJochException(JFJochExceptionCategory::GPUCUDAError, std::string("CUDA device unusable after an unrecoverable error: ") + cudaGetErrorString(err)); }