The local-box SNR kernel (analyze_pixel) was 55% of all GPU kernel time on a rotation run and held the image loop GPU-bound. Three exact changes: - analyze_pixel is rewritten as one warp per 32 output columns, each lane holding the vertical sums of two input columns in registers and the 31-wide horizontal window taken from warp prefix scans. No shared memory and no block-wide synchronisation; the sums are modular 64-bit integers, so the result is the same bits as before. - The adaptive finder keeps only (local-test & ring threshold), and a pixel's second-pass result depends only on its own window, so the second pass is evaluated at the ring pixels alone (analyze_candidates) instead of densely followed by and_bits. - The first pass is then only read within NBX of a ring pixel, so a warp whose tile no ring pixel can reach skips it; it also no longer reads an all-zero previous-pass buffer. Checked bit for bit against the old kernel on every frame of a 16M rotation run, and p.hkl / p_unmerged.mtz md5-identical on two inhouse EIGER2 16M sets. Total kernel time 19.8 -> 12.1 s, image loop 2.69 -> 1.37 ms/image; wall 44.0 -> 37.2 s and 56.0 -> 48.7 s. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
51 lines
2.5 KiB
C++
51 lines
2.5 KiB
C++
// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <vector>
|
|
|
|
#include "SpotFindingSettings.h"
|
|
#include "ImageSpotFinder.h"
|
|
#include "SpotExtractorGPU.h"
|
|
#include "../indexing/CUDAMemHelpers.h"
|
|
|
|
class ImageSpotFinderGPU : public ImageSpotFinder {
|
|
protected:
|
|
// Protected rather than private because AdaptiveSpotFinderGPU derives from this engine: it is
|
|
// this same local-box detection with the fixed photon floor replaced by a per-resolution-ring
|
|
// one, so it reuses the stream, the bit buffers and the extractor rather than owning a second
|
|
// set of them.
|
|
std::shared_ptr<CudaStream> stream;
|
|
|
|
CudaDevicePtr<uint32_t> gpu_out_0;
|
|
CudaDevicePtr<uint32_t> gpu_out_1; // holds the strong-pixel bits after Detect()
|
|
SpotExtractorGPU extractor;
|
|
|
|
private:
|
|
const int numberOfCudaThreads = 128; // #threads per block of analyze_pixel (one warp per 32 columns)
|
|
const int numberOfWaves = 32; // #row bands of analyze_pixel
|
|
const int windowSizeLimit = 32; // limit on the window size (2nby+1, 2nbx+1): a warp holds 2 x 32 columns
|
|
int candidateBlocks = 0; // grid of analyze_candidates (256 threads per block)
|
|
|
|
void RunDetect(const ImagePreprocessorBuffer &image, const SpotFindingSettings &settings,
|
|
const uint32_t *gpu_candidates);
|
|
protected:
|
|
// Detect, but with the result wanted only at the candidate pixels (device bit buffer, the layout
|
|
// of the output): gpu_out_1 ends up holding (Detect's result & candidates), and the second pass is
|
|
// evaluated at the candidates alone. Asynchronous - it does not wait for the stream.
|
|
void DetectAt(const ImagePreprocessorBuffer &image, const SpotFindingSettings &settings,
|
|
const uint32_t *gpu_candidates);
|
|
public:
|
|
ImageSpotFinderGPU(int32_t width, int32_t height, std::shared_ptr<CudaStream> stream);
|
|
~ImageSpotFinderGPU() override = default;
|
|
|
|
void Detect(const ImagePreprocessorBuffer &image, const SpotFindingSettings &settings) override;
|
|
void SetResolutionMaskBits(const std::vector<uint32_t> &packed_mask) override;
|
|
[[nodiscard]] uint32_t StrongPixelCount() const override { return extractor.StrongPixelCount(); }
|
|
const std::vector<DiffractionSpot> &ExtractComponents(const ImagePreprocessorBuffer &image,
|
|
const SpotFindingSettings &settings) override;
|
|
};
|
|
|
|
|