Files
Jungfraujoch/common/CUDAWrapper.cu
T
leonarski_f 84228bf8be
Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
v1.0.0-rc.173 (#83)
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports.
* jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls.
* Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results.
* Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable.
* Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate.
* Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do.
* Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence.
* Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags.
* Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check.
* Rugnux: Clear error messages when a data set needs more GPU or host memory than is available.

Reviewed-on: #83
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-09-29 15:57:32 +02:00

127 lines
3.9 KiB
Plaintext

// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <fstream>
#include <atomic>
#include <mutex>
#include <vector>
#include "CUDAWrapper.h"
#include "ThreadAffinity.h"
#include "JFJochException.h"
inline void cuda_err(cudaError_t val) {
if (val != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
}
int32_t get_gpu_count() {
int device_count;
cudaError_t val = cudaGetDeviceCount(&device_count);
switch (val) {
case cudaSuccess:
return device_count;
case cudaErrorNoDevice:
case cudaErrorInsufficientDriver:
return 0;
default:
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
}
}
std::vector<std::string> get_gpu_names() {
std::vector<std::string> names;
const int32_t count = get_gpu_count();
names.reserve(count);
for (int32_t i = 0; i < count; i++) {
cudaDeviceProp prop{};
// A device that cannot be queried still exists and still gets work, so it is listed - just
// without a name. Losing the whole list over one unreadable device would be worse.
if (cudaGetDeviceProperties(&prop, i) == cudaSuccess)
names.emplace_back(prop.name);
else
names.emplace_back("unknown GPU");
}
return names;
}
void set_gpu(int32_t dev_id) {
auto dev_count = get_gpu_count();
// Ignore if no GPU present
if (dev_count > 0) {
if ((dev_id < 0) || (dev_id >= dev_count))
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Device ID cannot be negative");
cuda_err(cudaSetDevice(dev_id));
}
}
namespace {
std::atomic<bool> gpu_numa_binding{false};
// The NUMA node of each device, read once.
int GpuNumaNode(int32_t dev_id) {
static std::mutex m;
static std::vector<int> node;
std::lock_guard<std::mutex> lock(m);
if (node.empty()) {
const int32_t count = get_gpu_count();
node.assign(count, -1);
for (int32_t i = 0; i < count; i++) {
char bus_id[32] = {};
if (cudaDeviceGetPCIBusId(bus_id, sizeof(bus_id), i) == cudaSuccess)
node[i] = NumaNodeOfPciDevice(bus_id);
}
}
return dev_id < static_cast<int32_t>(node.size()) ? node[dev_id] : -1;
}
}
void enable_gpu_numa_binding() {
gpu_numa_binding = true;
}
void pin_gpu(int32_t dev_id) {
if (get_gpu_count() == 0)
return;
set_gpu(dev_id);
if (gpu_numa_binding)
PinThreadToNumaNode(GpuNumaNode(dev_id));
}
void pin_gpu() {
static std::atomic<uint32_t> counter{0};
auto dev_count = get_gpu_count();
if (dev_count > 0)
pin_gpu(static_cast<int32_t>(counter.fetch_add(1) % dev_count));
}
void set_gpu_blocking_sync() {
const int32_t count = get_gpu_count();
for (int32_t i = 0; i < count; i++) {
if (cudaSetDevice(i) != cudaSuccess)
continue;
// cudaErrorSetOnActiveProcess where a context exists already: it keeps its flags.
if (cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync) != cudaSuccess)
cudaGetLastError();
}
if (count > 0)
cudaSetDevice(0);
}
void cuda_clear_error() {
cudaGetLastError();
}
void cuda_throw_if_context_lost() {
// cudaFree(nullptr) frees nothing, but a lost context fails it. The last error cannot say the same:
// once cudaGetLastError() has returned a sticky error, it and cudaPeekAtLastError() report success.
const cudaError_t err = cudaFree(nullptr);
if (err != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
std::string("CUDA device unusable after an unrecoverable error: ")
+ cudaGetErrorString(err));
}