Files
Jungfraujoch/tests/CudaHandledErrorTest.cpp
T
leonarski_f 6dfe065365
Build Packages / Create release (push) Successful in 16s
Build Packages / build:rugnux:aarch64 (cross) (push) Successful in 8m27s
Build Packages / build:rugnux-tgz (x86_64) (push) Successful in 9m15s
Build Packages / build:viewer-tgz:cpu (push) Successful in 10m11s
Build Packages / build:viewer-tgz:cuda (push) Successful in 12m6s
Build Packages / build:rpm (rocky8_nocuda) (push) Successful in 15m44s
Build Packages / build:rpm (rocky9_nocuda) (push) Successful in 16m1s
Build Packages / build:windows:nocuda (push) Successful in 17m29s
Build Packages / build:windows:cuda (push) Successful in 19m58s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 24m7s
Build Packages / build:rpm (ubuntu2404_nocuda) (push) Successful in 19m8s
Build Packages / build:rugnux:windows (push) Successful in 10m58s
Build Packages / build:rpm (ubuntu2204_nocuda) (push) Successful in 20m46s
Build Packages / Generate python client (push) Successful in 53s
Build Packages / build:rpm (rocky8_sls9) (push) Successful in 20m13s
Build Packages / Build documentation (push) Successful in 1m36s
Build Packages / build:rpm (rocky9_sls9) (push) Successful in 19m57s
Build Packages / build:rpm (rocky8) (push) Successful in 18m7s
Build Packages / build:rpm (rocky9) (push) Successful in 18m54s
Build Packages / build:rpm (ubuntu2204) (push) Successful in 19m32s
Build Packages / build:rpm (ubuntu2404) (push) Successful in 17m30s
Build Packages / Unit tests (push) Successful in 1h39m2s
v1.0.0-rc.172 (#82)
* Fixed `jfjoch_broker` cancelling every data collection with a CUDA "out of memory" error after long operation: GPU memory no longer leaks with each collection.
* Rugnux scales a rotation sweep until the per-frame scales settle instead of for a fixed three rounds, and says so when they did not - merged intensities, and the space group, resolution cut and frame rejection read off them, change accordingly; `--scaling-iterations` is now the cap on that loop (default 100).
* Rugnux places every frame of a marCCD, SMV or miniCBF series at the spindle angle its own header states, so a series with missing frames, or with angles written modulo 360, is no longer read at the wrong geometry or refused.
* Every rotation run writes two diagnostic files beside its reflections: `<prefix>_detector.jpg`, the detector projection with the pixel mask and the detected beam-stop shadow drawn on it, and `<prefix>_plot.txt`, one row per image.

Reviewed-on: #82
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-09-22 06:48:37 +02:00

86 lines
3.7 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute
// SPDX-License-Identifier: GPL-3.0-only
#include <catch2/catch_all.hpp>
#include "../common/CUDAWrapper.h"
#ifdef JFJOCH_USE_CUDA
#include <cuda_runtime.h>
#include <limits>
#include "../common/AzimuthalIntegrationMapping.h"
#include "../common/AzimuthalIntegrationProfile.h"
#include "../common/DiffractionExperiment.h"
#include "../common/PixelMask.h"
#include "../compression/JFJochCompressor.h"
#include "../image_analysis/IndexAndRefine.h"
#include "../image_analysis/MXAnalysisWithoutFPGA.h"
#include "../image_analysis/indexing/CUDAMemHelpers.h"
// What makes a handled CUDA failure dangerous: the failed call leaves the error behind as the
// thread's last error, and nothing but cudaGetLastError() takes it away - later successful calls
// neither clear it nor return it. So a route that failed, was caught and was replaced by another
// one hands its error to the next cudaGetLastError() in the code that followed, which is the check
// after a kernel launch. That is how a broker logged a raw "out of memory" over an image whose
// buffers were all allocated. cuda_clear_error() is what a handled failure discharges it with.
TEST_CASE("CudaDevicePtr_FailedAllocationLeavesErrorUntilCleared", "[CUDAMemHelpers]") {
if (get_gpu_count() == 0)
SKIP("No CUDA GPU present");
cuda_clear_error();
REQUIRE_THROWS(CudaDevicePtr<uint8_t>(std::numeric_limits<size_t>::max(), CudaAlloc::Synchronous));
// Work that goes fine in between changes nothing: the error is still waiting.
CudaDevicePtr<uint8_t> fine(1 << 16);
REQUIRE(cudaStreamSynchronize(cuda_allocation_stream()) == cudaSuccess);
cuda_clear_error();
CHECK(cudaGetLastError() == cudaSuccess);
}
// The device decoder is allowed to fail - the host decompresses instead - but the host route runs
// kernels of its own and checks cudaGetLastError() after them. An image must come out of that
// whole, and with nothing left behind for the next one.
TEST_CASE("MXAnalysis_HandledDeviceDecodeFailureLeavesNoError", "[CUDAMemHelpers]") {
if (get_gpu_count() == 0)
SKIP("No CUDA GPU present");
constexpr size_t width = 1030, height = 1064;
DiffractionExperiment experiment(DetDECTRIS(width, height, "Test", {}));
experiment.ImagesPerTrigger(1).NumTriggers(1).PixelSigned(false)
.DetectorDistance_mm(75).BeamX_pxl(515).BeamY_pxl(532).IncidentEnergy_keV(12.4);
PixelMask pixel_mask(experiment);
AzimuthalIntegrationMapping mapping(experiment, pixel_mask);
IndexAndRefine index_and_refine(experiment, nullptr);
MXAnalysisWithoutFPGA analysis(experiment, mapping, pixel_mask, index_and_refine);
std::vector<uint16_t> pixels(width * height, 0);
for (size_t i = 0; i < pixels.size(); i += 997)
pixels[i] = 500;
JFJochBitShuffleCompressor compressor(CompressionAlgorithm::BSHUF_LZ4);
const std::vector<uint8_t> compressed = compressor.Compress(pixels);
REQUIRE(!compressed.empty());
DataMessage message{};
message.number = 0;
message.image = CompressedImage(compressed.data(), compressed.size(), width, height,
CompressedImageMode::Uint16, CompressionAlgorithm::BSHUF_LZ4);
SpotFindingSettings settings = DiffractionExperiment::DefaultDataProcessingSettings();
settings.enable = false;
// Stage the device route into failing, by leaving behind the error a request the card cannot
// serve leaves.
void *unservable = nullptr;
REQUIRE(cudaMalloc(&unservable, std::numeric_limits<size_t>::max()) != cudaSuccess);
AzimuthalIntegrationProfile profile(mapping);
REQUIRE_NOTHROW(analysis.Analyze(message, profile, settings));
CHECK(cudaGetLastError() == cudaSuccess);
}
#endif