A sticky error (an illegal address, say) is reported by whichever worker synchronises next - usually
the bslz4 device decode - and MXAnalysisWithoutFPGA::Analyze and Rugnux::MaskDefectivePixels then
logged "falling back to host" and carried on, although the context is gone and the run fails anyway,
later and less clearly. cuda_throw_if_context_lost() now throws a GPUCUDAError ("CUDA device
unusable after an unrecoverable error: ...") there first, which the image loops already treat as
fatal (IsFatalResourceError), like a CUDA out-of-memory.
The last error cannot tell: once cudaGetLastError() has returned a sticky error, it and
cudaPeekAtLastError() both report success. cudaFree(nullptr) frees nothing and does not
synchronise, but returns the sticky error (checked: after an illegal-address kernel it returns
cudaErrorIllegalAddress; with only a pending out-of-memory it returns success), so a handled,
non-sticky decode failure still falls back - MXAnalysis_HandledDeviceDecodeFailureLeavesNoError
passes.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
52 lines
1.2 KiB
C++
52 lines
1.2 KiB
C++
// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "CUDAWrapper.h"
|
|
|
|
// Build-independent: the CUDA build gets get_gpu_names() from CUDAWrapper.cu, the CPU-only build from
|
|
// the stub below, and this collapses whichever list came back. Four identical cards read better as
|
|
// "4x <name>" than as the same name four times, and a mixed machine keeps one group per model.
|
|
std::string get_gpu_description() {
|
|
const auto names = get_gpu_names();
|
|
|
|
std::string out;
|
|
for (size_t i = 0; i < names.size();) {
|
|
size_t n = 1;
|
|
while (i + n < names.size() && names[i + n] == names[i])
|
|
n++;
|
|
if (!out.empty())
|
|
out += ", ";
|
|
if (n > 1)
|
|
out += std::to_string(n) + "x ";
|
|
out += names[i];
|
|
i += n;
|
|
}
|
|
return out;
|
|
}
|
|
|
|
#ifndef JFJOCH_USE_CUDA
|
|
|
|
int32_t get_gpu_count() {
|
|
return 0;
|
|
}
|
|
|
|
std::vector<std::string> get_gpu_names() {
|
|
return {};
|
|
}
|
|
|
|
void set_gpu(int32_t dev_id) {}
|
|
|
|
void pin_gpu() {}
|
|
|
|
void pin_gpu(int32_t dev_id) {}
|
|
|
|
void enable_gpu_numa_binding() {}
|
|
|
|
void set_gpu_blocking_sync() {}
|
|
|
|
void cuda_clear_error() {}
|
|
|
|
void cuda_throw_if_context_lost() {}
|
|
|
|
#endif
|