// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #pragma once // GPU adaptive spot finder that FUSES azimuthal integration and spot finding into one image pass. // // The CPU adaptive finder (AdaptiveSpotFinderCPU) and the azimuthal integrator both bin every pixel // into resolution rings and reduce (sum / sum^2 / count). Today azint runs on the GPU while the // adaptive finder re-does the identical per-ring reduction on the HOST - a wasted second pass over a // ~10 MP image. This engine does the ring reduction on the GPU and drives BOTH products from it: // - the azimuthal-integration profile (mean intensity per ring, in flat-field-corrected space), and // - the per-ring background (mean, sigma, peak-excluded via two sigma-clip passes) that sets the // self-calibrating spot-detection threshold (in raw photon counts). // It then flags strong pixels (value >= ring threshold) into a packed bit buffer and hands it to the // shared host connected-component extractor (ImageSpotFinder::ExtractSpots). // // Numerically it reproduces AdaptiveSpotFinderCPU: the same three-pass robust background, the same // per-ring threshold formula (shared via AdaptiveThreshold.h, computed on the host once per frame), // and the same raw-count detection test. The only differences from the CPU are those inherent to a // GPU reduction (float per-ring accumulation in atomic order vs the CPU's serial double sums), which // shift a handful of borderline pixels at most. The corrected sums for the azint profile are // accumulated in the SAME plain first pass, so one reduction feeds both products. #include #include #include "ImageSpotFinder.h" #include "SpotFindingSettings.h" #include "../../common/AzimuthalIntegrationProfile.h" #include "../../common/AzimuthalIntegrationMapping.h" #include "../indexing/CUDAMemHelpers.h" class AdaptiveSpotFinderGPU : public ImageSpotFinder { const AzimuthalIntegrationMapping &mapping; std::shared_ptr stream; const int nbins; const size_t npix; int reduce_threads = 128; int reduce_blocks = 0; int flag_threads = 256; int flag_blocks = 0; size_t shared_plain = 0; // per-block shared bytes for the plain pass (raw + corrected rings) size_t shared_clip = 0; // per-block shared bytes for a sigma-clip pass (raw rings only) bool use_shared = true; // false -> nbins too large for shared memory, use the global-atomics kernel // Static mapping inputs (uploaded once). CudaDevicePtr gpu_pixel_to_bin; CudaDevicePtr gpu_corrections; // Raw per-ring accumulators (re-zeroed each pass) + derived stats used to clip and threshold. // double, like the CPU engine's ring accumulators: the ring sigma is the cancelling difference // sum2/n - m^2, and the block atomics that fill these arrive in an arbitrary order. CudaDevicePtr gpu_sum; CudaDevicePtr gpu_sum2; CudaDevicePtr gpu_count; CudaDevicePtr gpu_mean; // per-ring raw mean (clip predicate) CudaDevicePtr gpu_sigma; // per-ring raw sigma (clip predicate) // Corrected per-ring accumulators (plain first pass only) -> azimuthal-integration profile. CudaDevicePtr gpu_sum_corr; CudaDevicePtr gpu_sum2_corr; // Per-ring detection threshold (host-computed, uploaded) and the strong-pixel bit buffer. CudaDevicePtr gpu_thr; CudaDevicePtr gpu_strong; // Host mirrors of the small per-ring transfers. std::vector host_sum; // clipped raw sum } input to the host threshold computation std::vector host_sum2; // clipped raw sum^2 } std::vector host_count; // clipped raw count } std::vector host_thr; // per-ring threshold (empty -> frame had no valid pixels) std::vector prof_sum; // plain corrected sum } azimuthal-integration profile std::vector prof_sum2; // plain corrected sum^2 } std::vector prof_count; // plain pixel count } CudaRegisteredVector output_buffer_reg; // pins the base-class bit buffer for fast D2H AzimuthalIntegrationProfile last_profile; // filled every Run(), retrievable via GetProfile() // One reduction pass over the image into the raw accumulators. clip_k <= 0 -> plain pass (all // valid pixels); clip_k > 0 -> keep only pixels within clip_k sigma of the current gpu_mean. // accumulate_corrected additionally fills gpu_sum_corr/gpu_sum2_corr for the profile (plain pass). void ReducePass(const ImagePreprocessorBuffer &image, float clip_k, bool accumulate_corrected); // Finalize gpu_mean/gpu_sigma from the current raw accumulators (per ring). void FinalizeStats(); // Host: per-ring threshold from the clipped raw stats and the single knob E (false pixels/frame). void ComputeThresholds(const SpotFindingSettings &settings); public: AdaptiveSpotFinderGPU(const AzimuthalIntegrationMapping &mapping, std::shared_ptr stream); ~AdaptiveSpotFinderGPU() override = default; AdaptiveSpotFinderGPU(const AdaptiveSpotFinderGPU &) = delete; AdaptiveSpotFinderGPU &operator=(const AdaptiveSpotFinderGPU &) = delete; void Detect(const ImagePreprocessorBuffer &image, const SpotFindingSettings &settings) override; // The azimuthal profile computed as a byproduct of the last Detect() - lets this engine replace the // separate azint pass in the analysis pipeline. [[nodiscard]] const AzimuthalIntegrationProfile &GetProfile() const { return last_profile; } };