// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #pragma once #include #include #include #include int32_t get_gpu_count(); // Names of the visible GPUs, in device order and one entry per device, so repeated cards repeat. // Empty without CUDA and on a machine with no device, which is also what get_gpu_count() == 0 says. std::vector get_gpu_names(); // The same list collapsed for a person: "4x NVIDIA A100-SXM4-80GB", or several such groups separated // by ", " on a mixed machine. Empty when no GPU is visible. std::string get_gpu_description(); void set_gpu(int32_t dev_id); // Pin the calling thread to the next GPU in round-robin order, using a process-wide counter // (counter++ % get_gpu_count()). Call once per thread; no thread id needed. No-op when no GPU // is visible. Honours CUDA_VISIBLE_DEVICES via get_gpu_count(). void pin_gpu(); // From here on, pin_gpu() also keeps the calling thread on the CPUs of the NUMA node its GPU is // attached to (ThreadAffinity.h) - on a machine with one node, or without CUDA, nothing changes. // Off unless a program asks for it. void enable_gpu_numa_binding(); // pin_gpu() onto a given device: set_gpu(dev_id), and the NUMA binding above where it is enabled. // For a worker that takes its card by index rather than round-robin. void pin_gpu(int32_t dev_id); // Have every GPU's host threads BLOCK in a synchronisation (cudaDeviceScheduleBlockingSync) instead // of spinning on a core until the device finishes. Must be called before anything creates a CUDA // context; a device that already has one keeps the flags it was created with. No-op without CUDA. void set_gpu_blocking_sync(); // Drop the error CUDA has recorded for the calling thread. Call it where a CUDA failure has been // HANDLED - a device route that fell back to the host, an indexing attempt whose failure was turned // into a result - because the error otherwise stays as the thread's last error and the next // cuda_err(cudaGetLastError()) after some later kernel launch reports it, over work that went fine. // A sticky error (an illegal access, say) is not cleared by this, and nothing here pretends it is: // the context is gone in that case and every later call fails on its own. No-op without CUDA. void cuda_clear_error(); // Throw if the CUDA context is lost - a sticky error (an illegal access, say) that every later call // on this device will fail with. For a device route about to fall back to the host: there is no // fallback from that, and falling back only moves the failure somewhere less clear. A handled, // non-sticky failure passes. No-op without CUDA. void cuda_throw_if_context_lost(); // GPU work that runs beside the main line of a process and gives all of its device memory back when it // ends: a probe pass run on a copy of the run while the run itself scales and merges. Hold one for as // long as that work runs. Build-independent - it only counts. class GPUWorkBeside { public: GPUWorkBeside(); ~GPUWorkBeside(); GPUWorkBeside(const GPUWorkBeside &) = delete; GPUWorkBeside &operator=(const GPUWorkBeside &) = delete; }; // Wait until no GPUWorkBeside is held, for at most `timeout`. True once none is - at once if none was. bool wait_for_gpu_work_beside(std::chrono::seconds timeout);