DivideOutIncidentFlux was still the last fully serial pass in Ingest: a sweep over every observation to take each frame's mean background, and another to divide every rlp by its frame's flux. Ten gigabytes of traffic on one thread. The per-frame means go a frame at a time rather than an observation at a time, so each frame's running sum stays in one thread and in the order it had - splitting by observation would cut a frame across two threads and the partial sums would have to be recombined, which is a different sequence of roundings. The divide is per-element and splits anywhere. The adaptive spot finder synchronised after flagging strong pixels. The extractor that reads those pixels runs on the same stream, so the ordering already guaranteed the flagging had finished; the wait only idled the host, once per image. Measured on a crystal with 66 million partial observations: Ingest 8.5 s and 7.7 s -> 7.1 s and 6.6 s, whole crystal 1m24s -> 1m17s. Merged statistics unchanged. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
72 lines
3.8 KiB
C++
72 lines
3.8 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <cstdint>
|
|
#include <memory>
|
|
#include <vector>
|
|
|
|
#include "BraggIntegrationEngine.h"
|
|
#include "../indexing/CUDAMemHelpers.h"
|
|
|
|
// CUDA engine: reproduces BraggIntegrationEngineCPU up to floating-point precision. Each stage is a
|
|
// kernel with one CUDA block per reflection cooperating over the small window via shared-memory
|
|
// reductions (the natural mapping for thousands of independent, tiny per-spot integrations).
|
|
//
|
|
// Pipeline (profile modes): reset -> mark_mask -> boxsum -> learn_profile -> build_profiles -> fit
|
|
// (the resolution shell is computed inline, so there is no separate shell pass). BoxSum mode stops
|
|
// after boxsum (that pass is the BraggIntegrate2D box integrator and the seed of the profile fit).
|
|
// The preprocessed image already lives on the device (ImagePreprocessorBufferGPU::getGPUBuffer());
|
|
// only the per-frame predicted centres are uploaded.
|
|
class BraggIntegrationEngineGPU : public BraggIntegrationEngine {
|
|
std::shared_ptr<CudaStream> stream;
|
|
int threads;
|
|
size_t fit_shared_bytes;
|
|
int rad_w = 0; // radial-background window of boxsum, in bins of one pixel
|
|
size_t boxsum_shared_bytes = 0;
|
|
|
|
size_t capacity = 0; // per-reflection device/host arrays hold at least this many reflections
|
|
|
|
// --- per-reflection device arrays (grown by EnsureCapacity) ---
|
|
CudaDevicePtr<float> d_px_x, d_px_y, d_d;
|
|
CudaDevicePtr<int> d_cx, d_cy;
|
|
CudaDevicePtr<float> d_I, d_sigma, d_bkg, d_bkg_var, d_var_bkg, d_obs_x, d_obs_y;
|
|
CudaDevicePtr<float> d_isum; // box-sum raw sum, for the radial correction
|
|
CudaDevicePtr<int> d_ninner, d_rbin, d_kbin;
|
|
CudaDevicePtr<uint8_t> d_ok, d_strong, d_has_obs;
|
|
|
|
// --- radial background curvature correction (see BraggIntegrationEngine) ---
|
|
int n_rad = 0; // radial bins, 0 when the correction is off
|
|
CudaDevicePtr<float> d_rad_sum, d_k_diff;
|
|
CudaDevicePtr<int> d_rad_cnt;
|
|
|
|
// --- fixed-size device arrays ---
|
|
// The learning/fit math is single precision: FP64 is heavily throttled on consumer GPUs and the
|
|
// extraction is Poisson-noise limited, so float reproduces the double CPU path to ~1e-4.
|
|
CudaDevicePtr<uint8_t> d_mask; // per-pixel inner-stencil reflection mask
|
|
// Per-pixel (distance, reflection) key naming the nearest predicted centre; allocated only when
|
|
// an overlap treatment is on, so the default path costs no extra device memory.
|
|
CudaDevicePtr<uint32_t> d_owner;
|
|
CudaDevicePtr<float> d_shell_grid, d_global_grid; // learned profile accumulators (N_SHELL*GG, GG)
|
|
CudaDevicePtr<float> d_shell_P, d_global_P; // normalised profiles (empirical mode)
|
|
CudaDevicePtr<float> d_mom; // learned 2nd moments, 3 per shell + global
|
|
CudaDevicePtr<float> d_sigma2_r, d_sigma2_t; // radial/tangential widths, N_SHELL + global
|
|
CudaDevicePtr<int> d_shell_n, d_global_n;
|
|
CudaDevicePtr<unsigned long long> d_invd2; // [min,max] inv-d^2 as monotonic bit patterns
|
|
|
|
// --- host staging (copied back once per frame) ---
|
|
std::vector<float> h_px_x, h_px_y, h_d;
|
|
std::vector<float> h_I, h_sigma, h_bkg, h_var_bkg, h_obs_x, h_obs_y;
|
|
std::vector<uint8_t> h_ok, h_has_obs;
|
|
|
|
|
|
void EnsureCapacity(size_t n);
|
|
|
|
public:
|
|
BraggIntegrationEngineGPU(const DiffractionExperiment &experiment, std::shared_ptr<CudaStream> stream);
|
|
std::vector<Reflection> Run(const ImagePreprocessorBuffer &image,
|
|
const std::vector<Reflection> &predicted, size_t npredicted,
|
|
int64_t image_number) override;
|
|
};
|