Build Packages / Create release (push) Successful in 17s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m22s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m37s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 9m33s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 10m39s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 11m4s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 13m19s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 17m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 18m49s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 19m10s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m26s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m31s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 18m54s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m45s
Build Packages / Generate python client (push) Successful in 37s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 20m20s
Build Packages / Build documentation (push) Successful in 1m32s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m37s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m6s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 19m49s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 20m29s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 17m2s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 14m27s
Build Packages / Unit tests (push) Successful in 1h18m12s
* Rugnux: Performance improvements on GPU and CPU (more of the pre-scan and of scaling on the GPU, faster CPU spot finding and crystal refinement), with unchanged results. * Rugnux: More robust processing - patches of persistently hot pixels are masked, an inconsistent merge triggers a retry at the measured beam centre, and builds targeting different CPU levels give the same results. * Rugnux: Improved scaling and merging - reflections with an overloaded pixel are dropped, as in XDS, sparse rotation sweeps are scaled more reliably, and French-Wilson amplitudes use an anisotropic Wilson prior. * Rugnux: Improved space-group determination - glide planes in groups without a centre of symmetry, screw axes from short or weak axial rows kept when a higher group is adopted, and more reliable decisions on twinned and pseudo-symmetric crystals. * Rugnux: Improved small-molecule processing - spots that grow wider than the integration disk and split spots are integrated over their measured footprint, sparse lattices are integrated on every frame, and the `.hkl` file holds unmerged scaled reflections (SHELX HKLF 4). * Rugnux: Reads Rigaku d*TREK SMV images (Saturn CCD), including detector 2theta and encoded pixel overflows; home-source (rotating-anode) datasets were added to the validation battery. * jfjoch_viewer: Fixed processing failing at the end with "Wrong JPEG library version" on Linux; the merge window shows the space group with proper subscripts and a checklist of crystal pathologies. Reviewed-on: #84 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
98 lines
5.6 KiB
C++
98 lines
5.6 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <cstdint>
|
|
#include <memory>
|
|
#include <vector>
|
|
|
|
#include "BraggIntegrationEngine.h"
|
|
#include "../indexing/CUDAMemHelpers.h"
|
|
#include "../../common/PixelMask.h"
|
|
|
|
// CUDA engine: reproduces BraggIntegrationEngineCPU up to floating-point precision. Each stage is a
|
|
// kernel with one CUDA block per reflection cooperating over the small window via shared-memory
|
|
// reductions (the natural mapping for thousands of independent, tiny per-spot integrations).
|
|
//
|
|
// Pipeline (profile modes): reset -> mark_mask -> boxsum -> learn_profile -> build_profiles -> fit
|
|
// (the resolution shell is computed inline, so there is no separate shell pass). BoxSum mode stops
|
|
// after boxsum (that pass is the BraggIntegrate2D box integrator and the seed of the profile fit).
|
|
// The preprocessed image already lives on the device (ImagePreprocessorBufferGPU::getGPUBuffer());
|
|
// only the per-frame predicted centres are uploaded.
|
|
class BraggIntegrationEngineGPU : public BraggIntegrationEngine {
|
|
std::shared_ptr<CudaStream> stream;
|
|
int threads;
|
|
size_t fit_shared_bytes;
|
|
int rad_w = 0; // radial-background window of boxsum, in bins of one pixel
|
|
size_t boxsum_shared_bytes = 0;
|
|
|
|
size_t capacity = 0; // per-reflection device/host arrays hold at least this many reflections
|
|
|
|
// Whether d_mask / d_owner may still carry the marks of an earlier image. Run() clears what it
|
|
// marked before it returns when that is cheaper than clearing the frame, and this is then false in
|
|
// the steady state; it is true before the first call, after one that threw part-way through, and
|
|
// whenever the marks covered enough of the frame that clearing all of it was the cheaper choice.
|
|
bool dirty = true;
|
|
size_t mask_box_px = 0; // pixels one reflection's mark_mask box covers, at the widest aperture
|
|
|
|
// --- per-reflection device arrays (grown by EnsureCapacity) ---
|
|
CudaDevicePtr<float> d_px_x, d_px_y, d_d;
|
|
CudaDevicePtr<uint8_t> d_mark; // the reflection marks its signal region in d_mask
|
|
CudaDevicePtr<int> d_cx, d_cy;
|
|
CudaDevicePtr<float> d_I, d_sigma, d_bkg, d_bkg_var, d_var_bkg, d_obs_x, d_obs_y;
|
|
CudaDevicePtr<float> d_isum; // box-sum raw sum, for the radial correction
|
|
CudaDevicePtr<int> d_ninner, d_rbin, d_kbin;
|
|
// The run's pixel mask, one byte per pixel (PixelMask::GetBinaryMask), shared with the
|
|
// preprocessor's copy. See BraggIntegrationEngineCPU::static_mask.
|
|
std::shared_ptr<CudaDevicePtr<uint8_t>> d_static_mask;
|
|
CudaDevicePtr<uint8_t> d_ok, d_strong, d_has_obs, d_overloaded;
|
|
|
|
// --- radial background curvature correction (see BraggIntegrationEngine) ---
|
|
int n_rad = 0; // radial bins, 0 when the correction is off
|
|
CudaDevicePtr<unsigned long long> d_rad_sum; // integer pixel sums, see boxsum
|
|
CudaDevicePtr<float> d_k_diff;
|
|
CudaDevicePtr<int> d_rad_cnt;
|
|
|
|
// --- fixed-size device arrays ---
|
|
// The learning/fit math is single precision: FP64 is heavily throttled on consumer GPUs and the
|
|
// extraction is Poisson-noise limited, so float reproduces the double CPU path to ~1e-4.
|
|
CudaDevicePtr<uint8_t> d_mask; // per-pixel inner-stencil reflection mask
|
|
// Per-pixel (distance, reflection) key naming the nearest predicted centre; allocated only when
|
|
// an overlap treatment is on, so the default path costs no extra device memory.
|
|
CudaDevicePtr<uint32_t> d_owner;
|
|
// Fixed-point (see PROFILE_FIXED): a float atomicAdd here made the profile depend on the order
|
|
// the blocks arrived in, and with it every intensity fitted through it.
|
|
CudaDevicePtr<unsigned long long> d_shell_grid, d_global_grid; // learned profile accumulators (N_SHELL*GG, GG)
|
|
CudaDevicePtr<float> d_shell_P, d_global_P; // normalised profiles (empirical mode)
|
|
CudaDevicePtr<unsigned long long> d_mom; // learned 2nd moments, 3 per shell + global
|
|
CudaDevicePtr<float> d_sigma2_r, d_sigma2_t; // radial/tangential widths, N_SHELL + global
|
|
CudaDevicePtr<int> d_shell_n, d_global_n;
|
|
CudaDevicePtr<unsigned long long> d_invd2; // [min,max] inv-d^2 as monotonic bit patterns
|
|
// BraggIntegrationCounts, accumulated on the device so an image costs no transfer; brought back
|
|
// only when Counts() is asked for.
|
|
CudaDevicePtr<unsigned long long> d_counts;
|
|
|
|
// --- host staging (copied back once per frame) ---
|
|
// Pinned, like every other engine's staging: a copy out of pageable memory does not return until the
|
|
// driver has staged it through a bounce buffer, so eight of them in a row are eight serialised
|
|
// round-trips rather than eight queued transfers.
|
|
CudaHostPtr<float> h_px_x, h_px_y, h_d;
|
|
CudaHostPtr<uint8_t> h_mark;
|
|
CudaHostPtr<float> h_I, h_sigma, h_bkg, h_var_bkg, h_obs_x, h_obs_y;
|
|
CudaHostPtr<uint8_t> h_ok, h_has_obs, h_overloaded;
|
|
|
|
void EnsureCapacity(size_t n);
|
|
|
|
public:
|
|
BraggIntegrationEngineGPU(const DiffractionExperiment &experiment, std::shared_ptr<CudaStream> stream,
|
|
const PixelMask &mask);
|
|
std::vector<Reflection> Run(const ImagePreprocessorBuffer &image,
|
|
const std::vector<Reflection> &predicted, size_t npredicted,
|
|
int64_t image_number) override;
|
|
|
|
// Brings the two device counters back before answering. Synchronises the stream, so ask once a
|
|
// pass rather than once an image.
|
|
[[nodiscard]] BraggIntegrationCounts Counts() const override;
|
|
};
|