Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports. * jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls. * Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results. * Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable. * Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate. * Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do. * Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence. * Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags. * Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check. * Rugnux: Clear error messages when a data set needs more GPU or host memory than is available. Reviewed-on: #83 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
93 lines
5.3 KiB
C++
93 lines
5.3 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <cstdint>
|
|
#include <memory>
|
|
#include <vector>
|
|
|
|
#include "BraggIntegrationEngine.h"
|
|
#include "../indexing/CUDAMemHelpers.h"
|
|
|
|
// CUDA engine: reproduces BraggIntegrationEngineCPU up to floating-point precision. Each stage is a
|
|
// kernel with one CUDA block per reflection cooperating over the small window via shared-memory
|
|
// reductions (the natural mapping for thousands of independent, tiny per-spot integrations).
|
|
//
|
|
// Pipeline (profile modes): reset -> mark_mask -> boxsum -> learn_profile -> build_profiles -> fit
|
|
// (the resolution shell is computed inline, so there is no separate shell pass). BoxSum mode stops
|
|
// after boxsum (that pass is the BraggIntegrate2D box integrator and the seed of the profile fit).
|
|
// The preprocessed image already lives on the device (ImagePreprocessorBufferGPU::getGPUBuffer());
|
|
// only the per-frame predicted centres are uploaded.
|
|
class BraggIntegrationEngineGPU : public BraggIntegrationEngine {
|
|
std::shared_ptr<CudaStream> stream;
|
|
int threads;
|
|
size_t fit_shared_bytes;
|
|
int rad_w = 0; // radial-background window of boxsum, in bins of one pixel
|
|
size_t boxsum_shared_bytes = 0;
|
|
|
|
size_t capacity = 0; // per-reflection device/host arrays hold at least this many reflections
|
|
|
|
// Whether d_mask / d_owner may still carry the marks of an earlier image. Run() clears what it
|
|
// marked before it returns when that is cheaper than clearing the frame, and this is then false in
|
|
// the steady state; it is true before the first call, after one that threw part-way through, and
|
|
// whenever the marks covered enough of the frame that clearing all of it was the cheaper choice.
|
|
bool dirty = true;
|
|
size_t mask_box_px = 0; // pixels one reflection's mark_mask box covers, at the widest aperture
|
|
|
|
// --- per-reflection device arrays (grown by EnsureCapacity) ---
|
|
CudaDevicePtr<float> d_px_x, d_px_y, d_d;
|
|
CudaDevicePtr<uint8_t> d_mark; // the reflection marks its signal region in d_mask
|
|
CudaDevicePtr<int> d_cx, d_cy;
|
|
CudaDevicePtr<float> d_I, d_sigma, d_bkg, d_bkg_var, d_var_bkg, d_obs_x, d_obs_y;
|
|
CudaDevicePtr<float> d_isum; // box-sum raw sum, for the radial correction
|
|
CudaDevicePtr<int> d_ninner, d_rbin, d_kbin;
|
|
CudaDevicePtr<uint8_t> d_ok, d_strong, d_has_obs;
|
|
|
|
// --- radial background curvature correction (see BraggIntegrationEngine) ---
|
|
int n_rad = 0; // radial bins, 0 when the correction is off
|
|
CudaDevicePtr<unsigned long long> d_rad_sum; // integer pixel sums, see boxsum
|
|
CudaDevicePtr<float> d_k_diff;
|
|
CudaDevicePtr<int> d_rad_cnt;
|
|
|
|
// --- fixed-size device arrays ---
|
|
// The learning/fit math is single precision: FP64 is heavily throttled on consumer GPUs and the
|
|
// extraction is Poisson-noise limited, so float reproduces the double CPU path to ~1e-4.
|
|
CudaDevicePtr<uint8_t> d_mask; // per-pixel inner-stencil reflection mask
|
|
// Per-pixel (distance, reflection) key naming the nearest predicted centre; allocated only when
|
|
// an overlap treatment is on, so the default path costs no extra device memory.
|
|
CudaDevicePtr<uint32_t> d_owner;
|
|
// Fixed-point (see PROFILE_FIXED): a float atomicAdd here made the profile depend on the order
|
|
// the blocks arrived in, and with it every intensity fitted through it.
|
|
CudaDevicePtr<unsigned long long> d_shell_grid, d_global_grid; // learned profile accumulators (N_SHELL*GG, GG)
|
|
CudaDevicePtr<float> d_shell_P, d_global_P; // normalised profiles (empirical mode)
|
|
CudaDevicePtr<unsigned long long> d_mom; // learned 2nd moments, 3 per shell + global
|
|
CudaDevicePtr<float> d_sigma2_r, d_sigma2_t; // radial/tangential widths, N_SHELL + global
|
|
CudaDevicePtr<int> d_shell_n, d_global_n;
|
|
CudaDevicePtr<unsigned long long> d_invd2; // [min,max] inv-d^2 as monotonic bit patterns
|
|
// BraggIntegrationCounts, accumulated on the device so an image costs no transfer; brought back
|
|
// only when Counts() is asked for.
|
|
CudaDevicePtr<unsigned long long> d_counts;
|
|
|
|
// --- host staging (copied back once per frame) ---
|
|
// Pinned, like every other engine's staging: a copy out of pageable memory does not return until the
|
|
// driver has staged it through a bounce buffer, so eight of them in a row are eight serialised
|
|
// round-trips rather than eight queued transfers.
|
|
CudaHostPtr<float> h_px_x, h_px_y, h_d;
|
|
CudaHostPtr<uint8_t> h_mark;
|
|
CudaHostPtr<float> h_I, h_sigma, h_bkg, h_var_bkg, h_obs_x, h_obs_y;
|
|
CudaHostPtr<uint8_t> h_ok, h_has_obs;
|
|
|
|
void EnsureCapacity(size_t n);
|
|
|
|
public:
|
|
BraggIntegrationEngineGPU(const DiffractionExperiment &experiment, std::shared_ptr<CudaStream> stream);
|
|
std::vector<Reflection> Run(const ImagePreprocessorBuffer &image,
|
|
const std::vector<Reflection> &predicted, size_t npredicted,
|
|
int64_t image_number) override;
|
|
|
|
// Brings the two device counters back before answering. Synchronises the stream, so ask once a
|
|
// pass rather than once an image.
|
|
[[nodiscard]] BraggIntegrationCounts Counts() const override;
|
|
};
|