diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index 8dcb538ac..3122fff96 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -50,7 +50,7 @@ either way. Eigen is header-only: only its headers reach the binaries, and no Ei ## Vendored directly in the repository -These live in the source tree (see the path) rather than being fetched; traccc is the exception - code adapted into first-party files rather than a vendored directory, see the note at the end of this file. +These live in the source tree (see the path) rather than being fetched; traccc and the Ceres-derived minimiser are the exceptions - code adapted into first-party files rather than a vendored directory, see the notes at the end of this file. | Component | Path | Copyright | License (SPDX) | License text | |---|---|---|---|---| @@ -69,6 +69,7 @@ These live in the source tree (see the path) rather than being fetched; traccc i | [pocketfft](https://github.com/mreineck/pocketfft) | `gemmi_gph/gemmi/third_party/pocketfft_hdronly.h` | Max-Planck-Society; Peter Bell; MIT (FFTW-derived parts) | BSD-3-Clause | [pocketfft.txt](licenses/pocketfft.txt) | | [tinydir](https://github.com/cxong/tinydir) | `gemmi_gph/gemmi/third_party/tinydir.h` | Cong Xu, Lautis Sun, Baudouin Feildel, Andargor | BSD-2-Clause | [tinydir.txt](licenses/tinydir.txt) | | [traccc (ACTS)](https://github.com/acts-project/traccc) | `image_analysis/spot_finding/StrongPixelSet.cpp`, `SpotExtractorGPU.cu` | CERN, for the benefit of the ACTS project | MPL-2.0 | [traccc.txt](licenses/traccc.txt) | +| [Ceres Solver](https://github.com/ceres-solver/ceres-solver) (adapted) | `image_analysis/geom_refinement/LMSolver.h`, `LMSolver.cpp` | Google Inc. | BSD-3-Clause | [ceres-solver.txt](licenses/ceres-solver.txt) | | [xbflash.qspi](https://github.com/Xilinx/XRT) | `tools/xbflash.qspi/` | Xilinx / AMD | Apache-2.0 | [xbflash-qspi.txt](licenses/xbflash-qspi.txt) | | [wingetopt](https://github.com/alex85k/wingetopt) | `tools/wingetopt/` | Todd C. Miller; The NetBSD Foundation | ISC AND BSD-2-Clause | [wingetopt.txt](licenses/wingetopt.txt) | @@ -107,6 +108,10 @@ served frontend, so the shipped web UI carries its own attribution. and `SpotExtractorGPU.cu` follows the design of its GPU counterpart. MPL-2.0 is file-level, so both files name the origin at the top and are covered by `licenses/traccc.txt`. See [ACKNOWLEDGEMENT.md](docs/ACKNOWLEDGEMENT.md) for the citation. +* **Ceres Solver** is also fetched and linked (table above); separately, `LMSolver.h`/`.cpp` re-implement + its trust-region Levenberg-Marquardt minimiser, projected line search, polynomial step choice and + sphere manifold for the crystal refinement, following its source. Both files name the origin at the + top and are covered by `licenses/ceres-solver.txt`. * **FFTW** is GPL-2.0-or-later — compatible with, and absorbed by, this project's GPL-3.0 license. * **Apache-2.0** components: where upstream ships a `NOTICE` file, it is reproduced in the corresponding `licenses/` text. diff --git a/image_analysis/CMakeLists.txt b/image_analysis/CMakeLists.txt index 6abc665d9..d9a50ec73 100644 --- a/image_analysis/CMakeLists.txt +++ b/image_analysis/CMakeLists.txt @@ -99,7 +99,10 @@ ADD_LIBRARY(JFJochImageAnalysis STATIC beam_stop/ShadowFinder.cpp beam_stop/ShadowFinder.h $<$:beam_stop/ShadowAccumulatorGPU.cu> + $<$:beam_stop/ShadowMaskGPU.cu> beam_stop/ShadowAccumulatorGPU.h + beam_stop/ShadowFinderInternal.h + beam_stop/ShadowMaskGPU.h rotation_indexer/RotationIndexer.cpp rotation_indexer/RotationIndexer.h WriteReflections.cpp diff --git a/image_analysis/azint/AzIntEngineGPU.cu b/image_analysis/azint/AzIntEngineGPU.cu index b35adaa95..7de941260 100644 --- a/image_analysis/azint/AzIntEngineGPU.cu +++ b/image_analysis/azint/AzIntEngineGPU.cu @@ -8,6 +8,15 @@ inline void cuda_err(cudaError_t val) { throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val)); } +// Pushes one ring run's totals to the shared accumulators; nothing for an empty run. +__device__ __forceinline__ void flush_azim_run(float *s_sum, float *s_sum2, uint32_t *s_count, + int b, float r_sum, float r_sum2, uint32_t r_count) { + if (r_count == 0) return; // also covers the initial "no ring yet" + atomicAdd(&s_sum[b], r_sum); + atomicAdd(&s_sum2[b], r_sum2); + atomicAdd(&s_count[b], r_count); +} + __global__ void gpu_azim_shared( const uint16_t *__restrict__ pixel_to_bin, @@ -33,19 +42,49 @@ void gpu_azim_shared( __syncthreads(); - for (size_t idx = blockIdx.x * blockDim.x + threadIdx.x; - idx < num_pixels; - idx += blockDim.x * gridDim.x) { - uint16_t bin = pixel_to_bin[idx]; + // Four pixels per thread, read as vector loads, and a running total per ring pushed to shared + // memory only when the ring changes: consecutive pixels along a row mostly share a ring, and an + // atomic per pixel on the same few addresses is what this kernel was limited by. The same scheme + // as the adaptive spot finder's ring pass (reduce_rings_shared). The buffers come straight from + // cudaMalloc, aligned for int4/float4; the npix % 4 leftovers are done one at a time below. + const size_t stride = static_cast(blockDim.x) * gridDim.x; + const size_t nquad = num_pixels / 4; + for (size_t q = blockIdx.x * blockDim.x + threadIdx.x; q < nquad; q += stride) { + const int4 v4 = reinterpret_cast(input_buffer)[q]; + const ushort4 b4 = reinterpret_cast(pixel_to_bin)[q]; + const float4 c4 = reinterpret_cast(corrections)[q]; + const int32_t vq[4] = {v4.x, v4.y, v4.z, v4.w}; + const uint16_t bq[4] = {b4.x, b4.y, b4.z, b4.w}; + const float cq[4] = {c4.x, c4.y, c4.z, c4.w}; - int32_t v = input_buffer[idx]; - bool valid = (v != INT32_MIN) & (v != INT32_MAX); + int r_b = -1; + float r_sum = 0.0f, r_sum2 = 0.0f; + uint32_t r_count = 0; + #pragma unroll + for (int k = 0; k < 4; k++) { + const int32_t v = vq[k]; + const int b = bq[k]; + if (v == INT32_MIN || v == INT32_MAX || b >= azint_bins) continue; + if (b != r_b) { + flush_azim_run(s_sum, s_sum2, s_count, r_b, r_sum, r_sum2, r_count); + r_b = b; + r_sum = 0.0f; r_sum2 = 0.0f; r_count = 0; + } + const float val = static_cast(v) * cq[k]; + r_sum += val; + r_sum2 += val * val; + r_count += 1; + } + flush_azim_run(s_sum, s_sum2, s_count, r_b, r_sum, r_sum2, r_count); + } - if (bin < azint_bins && valid) { + for (size_t idx = 4 * nquad + blockIdx.x * blockDim.x + threadIdx.x; idx < num_pixels; idx += stride) { + const uint16_t bin = pixel_to_bin[idx]; + const int32_t v = input_buffer[idx]; + if (bin < azint_bins && v != INT32_MIN && v != INT32_MAX) { const float val = static_cast(v) * corrections[idx]; - const float val2 = val * val; atomicAdd(&s_sum[bin], val); - atomicAdd(&s_sum2[bin], val2); + atomicAdd(&s_sum2[bin], val * val); atomicAdd(&s_count[bin], 1); } } diff --git a/image_analysis/beam_stop/ShadowAccumulatorGPU.cu b/image_analysis/beam_stop/ShadowAccumulatorGPU.cu index 240c83d86..1499f1fec 100644 --- a/image_analysis/beam_stop/ShadowAccumulatorGPU.cu +++ b/image_analysis/beam_stop/ShadowAccumulatorGPU.cu @@ -190,3 +190,13 @@ void ShadowAccumulatorGPU::Download(std::vector &max_value, std::vector cudaMemcpyDeviceToHost, *stream)); cuda_err(cudaStreamSynchronize(*stream)); } + +std::vector ShadowAccumulatorGPU::Mask(const ShadowMaskSetup &setup, const std::vector &pixel_mask) { + FoldPending(); + return ShadowMaskOnDevice(setup, pixel_mask, gpu_max, gpu_sum, gpu_count, frames, *stream); +} + +std::vector ShadowAccumulatorGPU::MeanProjection(const std::vector &pixel_mask) { + FoldPending(); + return MeanProjectionOnDevice(pixel_mask, gpu_sum, gpu_count, npixels, *stream); +} diff --git a/image_analysis/beam_stop/ShadowAccumulatorGPU.h b/image_analysis/beam_stop/ShadowAccumulatorGPU.h index c45a424c7..71907a0be 100644 --- a/image_analysis/beam_stop/ShadowAccumulatorGPU.h +++ b/image_analysis/beam_stop/ShadowAccumulatorGPU.h @@ -10,6 +10,7 @@ #include "../../common/CompressedImage.h" #include "../image_preprocessing/BSLZ4DecoderGPU.h" #include "../indexing/CUDAMemHelpers.h" +#include "ShadowMaskGPU.h" // The beam-stop projection accumulated on the device: only the compressed chunk crosses PCIe, and // both the decode and the per-pixel maximum / sum / count run on the GPU. The projection comes back @@ -59,6 +60,11 @@ public: [[nodiscard]] uint32_t GetFrameCount() const { return frames; } + // The beam-stop mask and the mean projection, made where the projection is (ShadowMaskGPU.h), so + // that it does not have to come back at all. + std::vector Mask(const ShadowMaskSetup &setup, const std::vector &pixel_mask); + std::vector MeanProjection(const std::vector &pixel_mask); + // Bring the projection back to the host, folding in whatever the last batch still holds. Cheap // to call once; it moves 20 bytes per pixel. void Download(std::vector &max_value, std::vector &sum_value, diff --git a/image_analysis/beam_stop/ShadowFinder.cpp b/image_analysis/beam_stop/ShadowFinder.cpp index 715ad1198..3d62d6d6d 100644 --- a/image_analysis/beam_stop/ShadowFinder.cpp +++ b/image_analysis/beam_stop/ShadowFinder.cpp @@ -2,6 +2,7 @@ // SPDX-License-Identifier: GPL-3.0-only #include "ShadowFinder.h" +#include "ShadowFinderInternal.h" #include #include @@ -10,7 +11,6 @@ #include #include #include -#include #include #include @@ -19,67 +19,12 @@ #include "../../common/ParallelFor.h" #include "../../common/JFJochException.h" -// A pixel is shadow when its background is below this fraction of the background it is -// compared against. -constexpr float SHADOW_RATIO = 0.50f; +using namespace shadow_finder; -// The boundary grows outward into partially shadowed pixels down to this fraction, but no -// further than PENUMBRA_MAX_PX from the core. A pin or a loop casts a wide half-shadow, so the -// reach is a good deal more than the beam stop's own edge needs. -constexpr float PENUMBRA_RATIO = 0.75f; -constexpr int PENUMBRA_MAX_PX = 30; - -// Bridge small breaks along the holder arm. A module gap wider than this is bridged separately for -// the arm search (see bridge_gaps). -constexpr int BRIDGE_PX = 6; - -// A pixel whose maximum reaches this recorded a real reflection and is never masked - a -// beam stop cannot block a reflection that was measured. -constexpr int64_t MIN_REFLECTION = 25; - -// How far below the background it is compared against a pixel must sit before the dip is -// believed, in standard deviations of the counts that back it. The counts are photons, so their -// scatter is Poisson and the deficit is measured against it rather than against a fixed number: -// on a well-exposed sweep a third of the background missing is overwhelming, and on a handful of -// low-background frames the same third is noise. Without this a six-frame pre-scan of a -// low-background sweep masks three quarters of the detector. -constexpr double MIN_DEFICIT_SIGMA = 6.0; - -// Smallest region the per-pixel test may return. A shadow is cast by something physical and is -// correspondingly large; an isolated patch this small is the background wandering, not hardware. -// This is what keeps the test specific now that a shadow no longer has to touch the direct beam. -constexpr int MIN_SHADOW_PIXELS = 2000; - -// ... and of those, how many must be deep (below SHADOW_RATIO) for a region of merely DIM pixels - -// hardware that lets part of the beam through - to count. A thin holder arm is dim along most of -// its length and deep in places; the background drifting over a detector's edge is dim everywhere -// and deep nowhere. -constexpr int MIN_CORE_PIXELS = 200; - -// The arm search compares a pixel with its ring as the ring actually varies around the beam. A ring -// of background is not flat once divided by the polarization factor when that factor is not the -// beam's: the remainder is a second harmonic in azimuth, cos 2phi, which reaches tens of percent at -// high angle. It is measured over radial bands of HARMONIC_BAND_PX, from the median of each of -// HARMONIC_SECTORS sectors that holds at least MIN_SECTOR_PIXELS pixels. -constexpr int HARMONIC_BAND_PX = 64; -constexpr int HARMONIC_SECTORS = 24; -constexpr int MIN_SECTOR_PIXELS = 200; - -// Side of the box the background is pooled over before testing. Its area is how many pixels back -// a ring's countability test, which decides where an azimuthal comparison is possible at all. -constexpr int POOL_PX = 5; -constexpr double MEAN_POOLED_PIXELS = POOL_PX * POOL_PX; - -// A ring with fewer valid pixels than this says nothing about whether it was counted. -constexpr int MIN_RING_PIXELS = 32; - -// A ring lies wholly inside the stop when its background is below this fraction of the background -// further out. This asks about a whole ring rather than about a pixel, so it keeps a threshold of -// its own and does not follow SHADOW_RATIO. -constexpr float BLOCKED_RING_RATIO = 0.35f; +static_assert(ShadowFinder::SHADOW == MASK_SHADOW && ShadowFinder::TRANSMITTING == MASK_TRANSMITTING); // Binary-image helpers on a width*height frame stored row-major as char (0/1). All run once, -// at GetMask() time; the BFS forms keep them O(pixels) rather than O(pixels * radius). +// at GetMask() time, and all are O(pixels) rather than O(pixels * radius). namespace { // A per-pixel array of GetMask(). A std::vector zeroes what it allocates on the thread that makes it, @@ -186,10 +131,12 @@ Plane erode(const Plane &in, int W, int H, int r, size_t nthreads) { // continues on both sides of it is one shadow - but a gap can be wider than BRIDGE_PX reaches (17 px // between the rows of PILATUS modules), and an arm crossing one fell apart into pieces each too small // to be believed. -Plane bridge_gaps(const Plane ®ion, const Plane &valid, int W, int H) { +Plane bridge_gaps(const Plane ®ion, const Plane &valid, int W, int H, size_t nthreads) { Plane out = region; + // The lines of one direction are independent: each reads `region` and only ever sets its own pixels. auto walk = [&](int n_lines, int len, auto index) { - for (int line = 0; line < n_lines; line++) { + ParallelChunks(n_lines, nthreads, [&](int lo, int hi) { + for (int line = lo; line < hi; line++) { int k = 0; while (k < len) { if (valid[index(line, k)]) { k++; continue; } @@ -199,12 +146,99 @@ Plane bridge_gaps(const Plane ®ion, const Plane &valid, int for (int j = start; j < k; j++) out[index(line, j)] = 1; } } + }); }; walk(H, W, [W](int y, int x) { return static_cast(y) * W + x; }); walk(W, H, [W](int x, int y) { return static_cast(y) * W + x; }); return out; } +// The 8-connected components of `member`: each member pixel gets the index of its component, dense +// from 0 and in no particular order, and every other pixel -1. +// +// Labelled in parallel. Each band of rows is flooded on its own, then the pieces that touch across a +// band boundary are joined. Which pixels share a component is all a caller reads, and that does not +// depend on how the rows were split. +struct Components { + Plane id; + int count = 0; +}; + +Components label_components(const Plane &member, int W, int H, size_t nthreads) { + const int bands = std::min(64, H); // never more bands than rows, so none is empty + std::vector band_row(bands + 1); + for (int b = 0; b <= bands; b++) + band_row[b] = static_cast(static_cast(b) * H / bands); + + Components out; + out.id = Plane(member.size()); + std::vector band_pieces(bands, 0); + ParallelFor(bands, nthreads, [&](int b) { + const size_t lo = static_cast(band_row[b]) * W, hi = static_cast(band_row[b + 1]) * W; + std::fill(out.id.begin() + lo, out.id.begin() + hi, -1); + std::vector stack; + int pieces = 0; + for (size_t start = lo; start < hi; start++) { + if (!member[start] || out.id[start] >= 0) + continue; + out.id[start] = pieces; + stack.push_back(start); + while (!stack.empty()) { + const size_t i = stack.back(); stack.pop_back(); + const int y = static_cast(i / W), x = static_cast(i % W); + for (int dy = -1; dy <= 1; dy++) + for (int dx = -1; dx <= 1; dx++) { + const int yy = y + dy, xx = x + dx; + if (yy < band_row[b] || yy >= band_row[b + 1] || xx < 0 || xx >= W) + continue; + const size_t j = static_cast(yy) * W + xx; + if (member[j] && out.id[j] < 0) { out.id[j] = pieces; stack.push_back(j); } + } + } + pieces++; + } + band_pieces[b] = pieces; + }); + + // A piece is named by its band's first index plus its number in the band, and the pieces are + // joined across each boundary row by union-find. + std::vector first(bands + 1, 0); + for (int b = 0; b < bands; b++) + first[b + 1] = first[b] + band_pieces[b]; + std::vector parent(first[bands]); + for (size_t k = 0; k < parent.size(); k++) + parent[k] = static_cast(k); + const auto find = [&](int k) { + while (parent[k] != k) { parent[k] = parent[parent[k]]; k = parent[k]; } + return k; + }; + for (int b = 0; b + 1 < bands; b++) { + const size_t above = static_cast(band_row[b + 1] - 1) * W, below = above + W; + for (int x = 0; x < W; x++) { + if (!member[above + x]) + continue; + for (int xx = std::max(0, x - 1); xx <= std::min(W - 1, x + 1); xx++) + if (member[below + xx]) { + const int ra = find(first[b] + out.id[above + x]); + const int rb = find(first[b + 1] + out.id[below + xx]); + if (ra != rb) parent[std::max(ra, rb)] = std::min(ra, rb); + } + } + } + std::vector dense(parent.size(), -1), component(parent.size()); + for (size_t k = 0; k < parent.size(); k++) { + const int root = find(static_cast(k)); + if (dense[root] < 0) dense[root] = out.count++; + component[k] = dense[root]; + } + ParallelFor(bands, nthreads, [&](int b) { + const size_t lo = static_cast(band_row[b]) * W, hi = static_cast(band_row[b + 1]) * W; + for (size_t i = lo; i < hi; i++) + if (out.id[i] >= 0) out.id[i] = component[first[b] + out.id[i]]; + }); + return out; +} + // Fill holes: background not reachable from the image border becomes region. // // The flood is run over the bounding box of `region` grown by one, not the whole detector. Outside @@ -212,52 +246,53 @@ Plane bridge_gaps(const Plane ®ion, const Plane &valid, int // outside is one border-connected component: a background pixel inside the box is border-connected // exactly when it reaches the ring. The beam stop occupies a small part of a detector, so this is // the same answer over a fraction of the pixels. -Plane fill_holes(const Plane ®ion, int W, int H) { +Plane fill_holes(const Plane ®ion, int W, int H, size_t nthreads) { + std::vector row_x0(H, W), row_x1(H, -1); + ParallelChunks(H, nthreads, [&](int ylo, int yhi) { + for (int y = ylo; y < yhi; y++) + for (int x = 0; x < W; x++) + if (region[static_cast(y) * W + x]) { + row_x0[y] = std::min(row_x0[y], x); + row_x1[y] = x; + } + }); int x0 = W, x1 = -1, y0 = H, y1 = -1; for (int y = 0; y < H; y++) - for (int x = 0; x < W; x++) - if (region[static_cast(y) * W + x]) { - x0 = std::min(x0, x); x1 = std::max(x1, x); - y0 = std::min(y0, y); y1 = std::max(y1, y); - } + if (row_x1[y] >= 0) { + x0 = std::min(x0, row_x0[y]); x1 = std::max(x1, row_x1[y]); + y0 = std::min(y0, y); y1 = y; + } if (x1 < 0) return region; // nothing to enclose x0 = std::max(0, x0 - 1); x1 = std::min(W - 1, x1 + 1); y0 = std::max(0, y0 - 1); y1 = std::min(H - 1, y1 + 1); + // The background of the box, in components; one that reaches the box's edge is outside. const int BW = x1 - x0 + 1, BH = y1 - y0 + 1; - std::vector bg_visited(static_cast(BW) * BH, 0); - std::queue q; // indices into the box - auto push = [&](int bx, int by) { - const int j = by * BW + bx; - if (!region[static_cast(by + y0) * W + bx + x0] && !bg_visited[j]) { - bg_visited[j] = 1; q.push(j); - } + Plane background(static_cast(BW) * BH); + ParallelChunks(BH, nthreads, [&](int lo, int hi) { + for (int by = lo; by < hi; by++) + for (int bx = 0; bx < BW; bx++) + background[static_cast(by) * BW + bx] = !region[static_cast(by + y0) * W + bx + x0]; + }); + const auto pieces = label_components(background, BW, BH, nthreads); + std::vector outside(pieces.count, 0); + const auto edge = [&](int bx, int by) { + const int c = pieces.id[static_cast(by) * BW + bx]; + if (c >= 0) outside[c] = 1; }; - for (int bx = 0; bx < BW; bx++) { push(bx, 0); push(bx, BH - 1); } - for (int by = 0; by < BH; by++) { push(0, by); push(BW - 1, by); } - while (!q.empty()) { - const int i = q.front(); q.pop(); - const int by = i / BW, bx = i % BW; - for (int dy = -1; dy <= 1; dy++) - for (int dx = -1; dx <= 1; dx++) { - const int yy = by + dy, xx = bx + dx; - if (yy < 0 || yy >= BH || xx < 0 || xx >= BW) - continue; - const int j = yy * BW + xx; - if (!region[static_cast(yy + y0) * W + xx + x0] && !bg_visited[j]) { - bg_visited[j] = 1; q.push(j); - } - } - } + for (int bx = 0; bx < BW; bx++) { edge(bx, 0); edge(bx, BH - 1); } + for (int by = 0; by < BH; by++) { edge(0, by); edge(BW - 1, by); } Plane out = region; - for (int by = 0; by < BH; by++) - for (int bx = 0; bx < BW; bx++) { - const size_t i = static_cast(by + y0) * W + bx + x0; - if (!region[i] && !bg_visited[by * BW + bx]) - out[i] = 1; - } + ParallelChunks(BH, nthreads, [&](int lo, int hi) { + for (int by = lo; by < hi; by++) + for (int bx = 0; bx < BW; bx++) { + const int c = pieces.id[static_cast(by) * BW + bx]; + if (c >= 0 && !outside[c]) + out[static_cast(by + y0) * W + bx + x0] = 1; + } + }); return out; } @@ -321,19 +356,40 @@ struct RingValues { RingValues bin_by_ring(const Plane &values, const Plane &valid, const Plane &radius, int max_radius, size_t nthreads) { + // Counted and scattered by blocks of pixels in parallel: each block writes its values of a ring + // after those of the blocks before it, so every ring holds its values in pixel order, as a single + // pass would leave them - and they are sorted below in any case. + constexpr int BLOCKS = 64; + const size_t n = values.size(); + const auto block_begin = [n](int b) { return n * b / BLOCKS; }; + const size_t rings = static_cast(max_radius) + 1; + std::vector cursor(BLOCKS * rings, 0); + ParallelFor(BLOCKS, nthreads, [&](int b) { + int *count = cursor.data() + b * rings; + for (size_t i = block_begin(b); i < block_begin(b + 1); i++) + if (valid[i]) + count[radius[i]]++; + }); + RingValues rv; rv.offset.assign(max_radius + 2, 0); - for (size_t i = 0; i < values.size(); i++) - if (valid[i]) - rv.offset[radius[i] + 1]++; - for (int r = 0; r <= max_radius; r++) - rv.offset[r + 1] += rv.offset[r]; + for (size_t r = 0; r < rings; r++) { + int at = rv.offset[r]; + for (int b = 0; b < BLOCKS; b++) { + const int count = cursor[b * rings + r]; + cursor[b * rings + r] = at; + at += count; + } + rv.offset[r + 1] = at; + } rv.values.resize(rv.offset[max_radius + 1]); - std::vector cursor(rv.offset.begin(), rv.offset.end() - 1); - for (size_t i = 0; i < values.size(); i++) - if (valid[i]) - rv.values[cursor[radius[i]]++] = values[i]; + ParallelFor(BLOCKS, nthreads, [&](int b) { + int *next = cursor.data() + b * rings; + for (size_t i = block_begin(b); i < block_begin(b + 1); i++) + if (valid[i]) + rv.values[next[radius[i]]++] = values[i]; + }); // Sorted once; the three iterations then only pick a rank and count a prefix. ParallelFor(max_radius + 1, nthreads, [&](int r) { @@ -355,7 +411,6 @@ ShadowFinder::ShadowFinder(const DiffractionExperiment &experiment, const PixelM if (pixel_mask.size() != static_cast(width) * height) throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "ShadowFinder: pixel mask does not match the detector"); - SetShardCount(1); #ifdef JFJOCH_USE_CUDA if (get_gpu_count() > 0) { const size_t npixels = static_cast(width) * height; @@ -372,98 +427,37 @@ void ShadowFinder::BeamCenter(float x, float y) { beam_y = y; } -// A shard's accumulators are allocated when a frame is first added to it, not here: with a GPU they -// are never used at all, and on a 16 Mpx detector eight of them are 2.9 GB to allocate and clear - -// which measured 0.8 s of the pre-scan, all of it wasted. -void ShadowFinder::SetShardCount(size_t n) { - shards.clear(); - shards.resize(std::max(1, n)); -} - +#ifdef JFJOCH_USE_CUDA ShadowFinder::Projection ShadowFinder::Reduce() const { -#ifdef JFJOCH_USE_CUDA - // The device holds its own projection. Bring it back and let it take part in the fold below as - // one more shard; when every frame went to the GPU it is the whole answer. - Projection device; - if (Gpu() && gpu->GetFrameCount() > 0) { - gpu->Download(device.max_value, device.sum_value, device.valid_count); - device.frames = gpu->GetFrameCount(); - bool host_empty = true; - for (const auto &p : shards) - host_empty = host_empty && (p.frames == 0); - if (host_empty) - return device; - } -#endif - - // Only when that shard actually holds something: its accumulators are allocated on first use, so - // an unused shard is empty rather than zeroed, and returning it would hand the callers below a - // projection they index by pixel. - if (shards.size() == 1 && shards[0].frames > 0 -#ifdef JFJOCH_USE_CUDA - && !(gpu && gpu->GetFrameCount() > 0) -#endif - ) - return shards[0]; - + // The device holds its own projection. Bring it back; when every frame went to the GPU it is the + // whole answer. Projection out; - const size_t npixels = static_cast(width) * height; - out.max_value.assign(npixels, 0); - out.sum_value.assign(npixels, 0); - out.valid_count.assign(npixels, 0); - for (const auto &p : shards) - out.frames += p.frames; -#ifdef JFJOCH_USE_CUDA - out.frames += device.frames; -#endif - - // Each worker owns a slice of the pixels and folds every shard into it. The sums and counts are - // integers and a pixel is touched by one worker only, so the result is the same as folding them - // one shard at a time on one thread - this is several hundred megabytes per shard and is limited - // by memory rather than by arithmetic. - const size_t nthreads = std::max(1, std::min(std::thread::hardware_concurrency(), - shards.size() * 2)); - const size_t chunk = (npixels + nthreads - 1) / nthreads; - std::vector> futures; - futures.reserve(nthreads); - for (size_t t = 0; t < nthreads; t++) { - const size_t lo = t * chunk, hi = std::min(npixels, lo + chunk); - if (lo >= hi) break; - futures.emplace_back(std::async(std::launch::async, [&, lo, hi] { -#ifdef JFJOCH_USE_CUDA - const Projection *extra[1] = {&device}; - for (const auto *pp : extra) { - const auto &p = *pp; - if (p.frames == 0) continue; - for (size_t i = lo; i < hi; i++) { - if (p.valid_count[i] == 0) - continue; - if (out.valid_count[i] == 0 || p.max_value[i] > out.max_value[i]) - out.max_value[i] = p.max_value[i]; - out.sum_value[i] += p.sum_value[i]; - out.valid_count[i] += p.valid_count[i]; - } - } -#endif - for (const auto &p : shards) { - if (p.frames == 0) continue; // never used, and its accumulators were never allocated - for (size_t i = lo; i < hi; i++) { - if (p.valid_count[i] == 0) - continue; - if (out.valid_count[i] == 0 || p.max_value[i] > out.max_value[i]) - out.max_value[i] = p.max_value[i]; - out.sum_value[i] += p.sum_value[i]; - out.valid_count[i] += p.valid_count[i]; - } - } - })); + if (Gpu() && gpu->GetFrameCount() > 0) { + gpu->Download(out.max_value, out.sum_value, out.valid_count); + out.frames = gpu->GetFrameCount(); } - for (auto &f : futures) f.get(); + if (host.frames == 0) + return out; + + // A pixel is touched by one worker only and the sums and counts are integers, so the result is + // the same as folding on one thread. + out.frames += host.frames; + ParallelChunks(static_cast(out.max_value.size()), std::thread::hardware_concurrency(), [&](int lo, int hi) { + for (int i = lo; i < hi; i++) { + if (host.valid_count[i] == 0) + continue; + if (out.valid_count[i] == 0 || host.max_value[i] > out.max_value[i]) + out.max_value[i] = host.max_value[i]; + out.sum_value[i] += host.sum_value[i]; + out.valid_count[i] += host.valid_count[i]; + } + }); return out; } +#endif template -void ShadowFinder::Add(const T *ptr, Projection &p) { +void ShadowFinder::Add(const T *ptr, size_t begin, size_t end) { // The pixel type's sentinel extreme marks "no data" (module gap / masked): the // preprocessor/writer stores INT*_MIN for signed and UINT*_MAX for unsigned. For signed // types the opposite extreme is a genuine saturated value and is kept, so a saturated @@ -474,27 +468,23 @@ void ShadowFinder::Add(const T *ptr, Projection &p) { else masked = std::numeric_limits::max(); - for (size_t i = 0; i < p.max_value.size(); i++) { + for (size_t i = begin; i < end; i++) { const T v = ptr[i]; if (v == masked) continue; const int64_t vi = static_cast(v); - if (p.valid_count[i] == 0 || vi > p.max_value[i]) - p.max_value[i] = vi; - p.sum_value[i] += vi; - p.valid_count[i]++; + if (host.valid_count[i] == 0 || vi > host.max_value[i]) + host.max_value[i] = vi; + host.sum_value[i] += vi; + host.valid_count[i]++; } - p.frames++; } -void ShadowFinder::AddImage(const DataMessage &data, std::vector &buffer, size_t shard) { +void ShadowFinder::AddImage(const DataMessage &data, std::vector &buffer) { if (static_cast(data.image.GetWidth()) * data.image.GetHeight() != static_cast(width) * height) throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "ShadowFinder: image size does not match the detector"); - if (shard >= shards.size()) - throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, - "ShadowFinder: shard out of range"); #ifdef JFJOCH_USE_CUDA // One device, so the frames queue here - but each is only a chunk upload plus two kernels, and @@ -511,25 +501,37 @@ void ShadowFinder::AddImage(const DataMessage &data, std::vector &buffe } #endif - Projection &p = shards[shard]; - if (p.max_value.empty()) { - const size_t npixels = static_cast(width) * height; - p.max_value.assign(npixels, 0); - p.sum_value.assign(npixels, 0); - p.valid_count.assign(npixels, 0); + const size_t npixels = static_cast(width) * height; + { + std::unique_lock ul(host_mutex); + if (host.max_value.empty()) { + host.max_value.resize(npixels); + host.sum_value.resize(npixels); + host.valid_count.resize(npixels); + } } const auto ptr = data.image.GetUncompressedPtr(buffer); - switch (data.image.GetMode()) { - case CompressedImageMode::Int8: Add(reinterpret_cast(ptr), p); break; - case CompressedImageMode::Uint8: Add(reinterpret_cast(ptr), p); break; - case CompressedImageMode::Int16: Add(reinterpret_cast(ptr), p); break; - case CompressedImageMode::Uint16: Add(reinterpret_cast(ptr), p); break; - case CompressedImageMode::Int32: Add(reinterpret_cast(ptr), p); break; - case CompressedImageMode::Uint32: Add(reinterpret_cast(ptr), p); break; - default: - throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, - "ShadowFinder: unsupported image mode"); + const size_t rows_per_band = (static_cast(height) + BANDS - 1) / BANDS; + const size_t first = next_band.fetch_add(1); + for (size_t b = 0; b < BANDS; b++) { + const size_t band = (first + b) % BANDS; + const size_t begin = std::min(npixels, band * rows_per_band * width); + const size_t end = std::min(npixels, (band + 1) * rows_per_band * width); + std::lock_guard lock(band_mutex[band]); + switch (data.image.GetMode()) { + case CompressedImageMode::Int8: Add(reinterpret_cast(ptr), begin, end); break; + case CompressedImageMode::Uint8: Add(reinterpret_cast(ptr), begin, end); break; + case CompressedImageMode::Int16: Add(reinterpret_cast(ptr), begin, end); break; + case CompressedImageMode::Uint16: Add(reinterpret_cast(ptr), begin, end); break; + case CompressedImageMode::Int32: Add(reinterpret_cast(ptr), begin, end); break; + case CompressedImageMode::Uint32: Add(reinterpret_cast(ptr), begin, end); break; + default: + throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, + "ShadowFinder: unsupported image mode"); + } } + std::unique_lock ul(host_mutex); + host.frames++; } #ifdef JFJOCH_USE_CUDA @@ -552,8 +554,7 @@ ShadowAccumulatorGPU *ShadowFinder::Gpu() const { uint32_t ShadowFinder::GetFrameCount() const { std::unique_lock ul(m); - uint32_t frames = 0; - for (const auto &p : shards) frames += p.frames; + uint32_t frames = host.frames; #ifdef JFJOCH_USE_CUDA if (gpu) frames += gpu->GetFrameCount(); #endif @@ -561,26 +562,39 @@ uint32_t ShadowFinder::GetFrameCount() const { } const ShadowFinder::Projection &ShadowFinder::Reduced() const { - if (!reduced) - reduced = Reduce(); - return *reduced; +#ifdef JFJOCH_USE_CUDA + if (gpu && gpu->GetFrameCount() > 0) { + if (!reduced) + reduced = Reduce(); + return *reduced; + } +#endif + return host; } void ShadowFinder::ReleaseProjection() { +#ifdef JFJOCH_USE_CUDA std::unique_lock ul(m); reduced.reset(); +#endif } std::vector ShadowFinder::GetMeanProjection() const { std::unique_lock ul(m); +#ifdef JFJOCH_USE_CUDA + if (Gpu() && gpu->GetFrameCount() > 0 && host.frames == 0) + return gpu->MeanProjection(pixel_mask); +#endif const Projection &p = Reduced(); const auto &sum_value = p.sum_value; const auto &valid_count = p.valid_count; - std::vector mean(static_cast(width) * height, NAN); - for (size_t i = 0; i < mean.size(); i++) - if (valid_count[i] > 0 && pixel_mask[i] == 0) - mean[i] = static_cast(static_cast(sum_value[i]) / valid_count[i]); + std::vector mean(static_cast(width) * height); + ParallelChunks(static_cast(mean.size()), std::thread::hardware_concurrency(), [&](int lo, int hi) { + for (int i = lo; i < hi; i++) + mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0 + ? static_cast(static_cast(sum_value[i]) / valid_count[i]) : NAN; + }); return mean; } @@ -588,6 +602,28 @@ std::vector ShadowFinder::GetMask(size_t nthreads) const { std::unique_lock ul(m); if (nthreads == 0) nthreads = std::max(1u, std::thread::hardware_concurrency()); +#ifdef JFJOCH_USE_CUDA + // Where every frame went to the device the mask is made there too, from the projection as it lies. + if (Gpu() && gpu->GetFrameCount() > 0 && host.frames == 0) { + const float diag = std::hypot(static_cast(width), static_cast(height)); + if (!std::isfinite(beam_x) || !std::isfinite(beam_y) + || std::fabs(beam_x - width * 0.5f) > 4.0f * diag || std::fabs(beam_y - height * 0.5f) > 4.0f * diag) + return std::vector(static_cast(width) * height, 0); + ShadowMaskSetup setup; + setup.width = width; + setup.height = height; + setup.beam_x = beam_x; + setup.beam_y = beam_y; + const auto rot = geometry.GetDetectorMatrix().arr(); + for (int k = 0; k < 9; k++) + setup.det_matrix[k] = rot[k]; + setup.pixel_size_mm = geometry.GetPixelSize_mm(); + setup.distance_mm = geometry.GetDetectorDistance_mm(); + setup.has_polarization = polarization.has_value(); + setup.polarization = polarization.value_or(0.0f); + return gpu->Mask(setup, pixel_mask); + } +#endif const Projection &p = Reduced(); const auto &max_value = p.max_value; const auto &sum_value = p.sum_value; @@ -727,9 +763,8 @@ std::vector ShadowFinder::GetMask(size_t nthreads) const { // Innermost rings hold only a handful of pixels, too few to judge, so they are stepped over // rather than allowed to end the walk. std::vector ring_pixels(max_radius + 1, 0); - for (int i = 0; i < n_pixels; i++) - if (valid[i]) - ring_pixels[radius[i]]++; + for (int rad = 0; rad <= max_radius; rad++) + ring_pixels[rad] = rings.offset[rad + 1] - rings.offset[rad]; // A ring lies inside the stop when its background is a fraction of what this detector's // background typically is. Counting statistics cannot decide this: on a bright dataset the @@ -741,25 +776,7 @@ std::vector ShadowFinder::GetMask(size_t nthreads) const { // disk it masked. The median over the rings this walk is willing to judge is what "typically" // means, and it is robust from both sides - to a bright ring, and to a corner ring of a handful // of pixels whose median is one pixel's mean. - std::vector judgeable; - for (int rad = 0; rad <= max_radius; rad++) - if (ring_pixels[rad] >= MIN_RING_PIXELS) - judgeable.push_back(baseline[rad]); - float typical_background = 0.0f; - if (!judgeable.empty()) { - const auto middle = judgeable.begin() + judgeable.size() / 2; - std::nth_element(judgeable.begin(), middle, judgeable.end()); - typical_background = *middle; - } - - int blocked_out_to = -1; - for (int rad = 0; rad <= max_radius; rad++) { - if (ring_pixels[rad] < MIN_RING_PIXELS) - continue; - if (baseline[rad] >= BLOCKED_RING_RATIO * typical_background) - break; - blocked_out_to = rad; - } + const int blocked_out_to = BlockedOutTo(baseline, ring_pixels); // The counts a pixel's pooled background is made of, and the counts the ring says it should // have had. The test is on the deficit between them, in units of its own Poisson scatter. @@ -789,36 +806,20 @@ std::vector ShadowFinder::GetMask(size_t nthreads) const { // and the stop. What keeps the test specific instead is size, since the background wanders by a // pixel or two at a time and hardware does not. const Plane bridged = dilate(low, W, H, BRIDGE_PX, nthreads); - Plane region = filled_plane(n_pixels, 0, nthreads); + Plane region(n_pixels); { - Plane seen = filled_plane(n_pixels, 0, nthreads); - std::vector component; - std::queue q; - for (int start = 0; start < n_pixels; start++) { - if (!bridged[start] || seen[start]) - continue; - component.clear(); - int n_low = 0; - seen[start] = 1; - q.push(start); - while (!q.empty()) { - const int i = q.front(); q.pop(); - component.push_back(i); - n_low += low[i]; - const int y = i / W, x = i % W; - for (int dy = -1; dy <= 1; dy++) - for (int dx = -1; dx <= 1; dx++) { - const int yy = y + dy, xx = x + dx; - if (yy < 0 || yy >= H || xx < 0 || xx >= W) - continue; - const int j = yy * W + xx; - if (bridged[j] && !seen[j]) { seen[j] = 1; q.push(j); } - } - } - if (n_low >= MIN_SHADOW_PIXELS) - for (const int i : component) - region[i] = low[i]; - } + const auto pieces = label_components(bridged, W, H, nthreads); + std::vector> n_low(pieces.count); + ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) { + for (int i = lo; i < hi; i++) + if (pieces.id[i] >= 0 && low[i]) + n_low[pieces.id[i]].fetch_add(1, std::memory_order_relaxed); + }); + ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) { + for (int i = lo; i < hi; i++) + region[i] = pieces.id[i] >= 0 && n_low[pieces.id[i]].load(std::memory_order_relaxed) >= MIN_SHADOW_PIXELS + ? low[i] : 0; + }); } // The rings that lie wholly inside the stop are decided by the ring walk above rather than by @@ -866,7 +867,7 @@ std::vector ShadowFinder::GetMask(size_t nthreads) const { region = erode(dilate(region, W, H, 2, nthreads), W, H, 2, nthreads); - region = fill_holes(region, W, H); + region = fill_holes(region, W, H, nthreads); // Expose recorded reflections - done last, with no fill afterwards, so a spot the shadow @@ -903,60 +904,42 @@ std::vector ShadowFinder::GetMask(size_t nthreads) const { // allowed to explain a dim sector away - where it would ask for more than the median, the median // stands - so the step can only drop what it found before, never find something new. const int n_bands = max_radius / HARMONIC_BAND_PX + 1; - std::vector> sector_values(static_cast(n_bands) * HARMONIC_SECTORS); - for (int i = 0; i < n_pixels; i++) { - if (!valid[i] || region[i]) - continue; - const float dx = static_cast(i % W) - beam_x, dy = static_cast(i / W) - beam_y; - const double phi = std::atan2(dy, dx) + std::numbers::pi; - const int sector = std::min(HARMONIC_SECTORS - 1, static_cast(phi / (2.0 * std::numbers::pi) * HARMONIC_SECTORS)); - sector_values[static_cast(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector].push_back(ratio[i]); - } - std::vector harm_c(n_bands, 0.0f), harm_s(n_bands, 0.0f); - ParallelFor(n_bands, nthreads, [&](int band) { - std::vector med(HARMONIC_SECTORS, -1.0), c(HARMONIC_SECTORS), s(HARMONIC_SECTORS); - for (int k = 0; k < HARMONIC_SECTORS; k++) { - auto &v = sector_values[static_cast(band) * HARMONIC_SECTORS + k]; - if (v.size() >= MIN_SECTOR_PIXELS) { - std::nth_element(v.begin(), v.begin() + v.size() / 2, v.end()); - med[k] = v[v.size() / 2]; - } - const double phi = (k + 0.5) * 2.0 * std::numbers::pi / HARMONIC_SECTORS - std::numbers::pi; - c[k] = std::cos(2 * phi); - s[k] = std::sin(2 * phi); - } - // Least squares of med = m + p cos 2phi + q sin 2phi over the sectors that are not themselves - // dim against the fit, three times over as the baseline is; the model is relative, 1 + (p cos + q sin)/m. - std::vector use(HARMONIC_SECTORS); - for (int k = 0; k < HARMONIC_SECTORS; k++) - use[k] = med[k] >= 0; - for (int iter = 0; iter < 3; iter++) { - double n = 0, sc = 0, ss = 0, scc = 0, sss = 0, scs = 0, y = 0, yc = 0, ys = 0; - for (int k = 0; k < HARMONIC_SECTORS; k++) { - if (!use[k]) continue; - n++; sc += c[k]; ss += s[k]; scc += c[k] * c[k]; sss += s[k] * s[k]; scs += c[k] * s[k]; - y += med[k]; yc += med[k] * c[k]; ys += med[k] * s[k]; - } - // Sectors crowded into a narrow arc cannot tell a harmonic from a level; the determinant of - // the normal matrix, per sector cubed, is 1/4 on a full ring. - const double det = n * (scc * sss - scs * scs) - sc * (sc * sss - scs * ss) + ss * (sc * scs - scc * ss); - if (n < 6 || det < 0.01 * n * n * n) { - harm_c[band] = harm_s[band] = 0.0f; - return; - } - const double m = (y * (scc * sss - scs * scs) - sc * (yc * sss - scs * ys) + ss * (yc * scs - scc * ys)) / det; - const double p = (n * (yc * sss - ys * scs) - y * (sc * sss - scs * ss) + ss * (sc * ys - yc * ss)) / det; - const double q = (n * (scc * ys - scs * yc) - sc * (sc * ys - yc * ss) + y * (sc * scs - scc * ss)) / det; - if (m <= 0) { - harm_c[band] = harm_s[band] = 0.0f; - return; - } - harm_c[band] = static_cast(p / m); - harm_s[band] = static_cast(q / m); - for (int k = 0; k < HARMONIC_SECTORS; k++) - use[k] = med[k] >= 0 && med[k] >= PENUMBRA_RATIO * (1.0 + harm_c[band] * c[k] + harm_s[band] * s[k]); + const size_t n_sectors = static_cast(n_bands) * HARMONIC_SECTORS; + // Gathered by blocks of rows in parallel and joined in block order. Only the median of each + // sector is read, and that is the same whatever order its values were gathered in. + constexpr int SECTOR_BLOCKS = 64; + std::vector>> block_values(SECTOR_BLOCKS); + ParallelFor(SECTOR_BLOCKS, nthreads, [&](int b) { + auto &values = block_values[b]; + values.resize(n_sectors); + const int lo = static_cast(static_cast(n_pixels) * b / SECTOR_BLOCKS); + const int hi = static_cast(static_cast(n_pixels) * (b + 1) / SECTOR_BLOCKS); + for (int i = lo; i < hi; i++) { + if (!valid[i] || region[i]) + continue; + const float dx = static_cast(i % W) - beam_x, dy = static_cast(i / W) - beam_y; + const double phi = std::atan2(dy, dx) + std::numbers::pi; + const int sector = std::min(HARMONIC_SECTORS - 1, static_cast(phi / (2.0 * std::numbers::pi) * HARMONIC_SECTORS)); + values[static_cast(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector].push_back(ratio[i]); } }); + std::vector> sector_values(n_sectors); + ParallelFor(static_cast(n_sectors), nthreads, [&](int k) { + for (const auto &values : block_values) + sector_values[k].insert(sector_values[k].end(), values[k].begin(), values[k].end()); + }); + block_values.clear(); + // The median of each sector with enough pixels to have one. + std::vector sector_median(n_sectors, -1.0); + ParallelFor(static_cast(n_sectors), nthreads, [&](int k) { + auto &v = sector_values[k]; + if (v.size() >= MIN_SECTOR_PIXELS) { + std::nth_element(v.begin(), v.begin() + v.size() / 2, v.end()); + sector_median[k] = v[v.size() / 2]; + } + }); + std::vector harm_c, harm_s; + HarmonicFit(sector_median, n_bands, harm_c, harm_s); Plane dim = filled_plane(n_pixels, 0, nthreads); ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) { @@ -978,36 +961,108 @@ std::vector ShadowFinder::GetMask(size_t nthreads) const { && poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA; } }); - const Plane joined = bridge_gaps(dilate(dim, W, H, BRIDGE_PX, nthreads), valid, W, H); - Plane seen = filled_plane(n_pixels, 0, nthreads); - std::vector component; - std::queue q; - for (int start = 0; start < n_pixels; start++) { - if (!joined[start] || seen[start]) - continue; - component.clear(); - int n_dim = 0, n_low = 0; - seen[start] = 1; - q.push(start); - while (!q.empty()) { - const int i = q.front(); q.pop(); - component.push_back(i); - n_dim += dim[i]; - n_low += dim[i] && low[i]; - const int y = i / W, x = i % W; - for (int dy = -1; dy <= 1; dy++) - for (int dx = -1; dx <= 1; dx++) { - const int yy = y + dy, xx = x + dx; - if (yy < 0 || yy >= H || xx < 0 || xx >= W) - continue; - const int j = yy * W + xx; - if (joined[j] && !seen[j]) { seen[j] = 1; q.push(j); } - } + const Plane joined = bridge_gaps(dilate(dim, W, H, BRIDGE_PX, nthreads), valid, W, H, nthreads); + const auto pieces = label_components(joined, W, H, nthreads); + std::vector> n_dim(pieces.count), n_low(pieces.count); + ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) { + for (int i = lo; i < hi; i++) + if (pieces.id[i] >= 0 && dim[i]) { + n_dim[pieces.id[i]].fetch_add(1, std::memory_order_relaxed); + if (low[i]) + n_low[pieces.id[i]].fetch_add(1, std::memory_order_relaxed); + } + }); + ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) { + for (int i = lo; i < hi; i++) { + const int c = pieces.id[i]; + if (c >= 0 && dim[i] && n_dim[c].load(std::memory_order_relaxed) >= MIN_SHADOW_PIXELS + && n_low[c].load(std::memory_order_relaxed) >= MIN_CORE_PIXELS) + mask[i] = TRANSMITTING; } - if (n_dim >= MIN_SHADOW_PIXELS && n_low >= MIN_CORE_PIXELS) - for (const int i : component) - if (dim[i]) - mask[i] = TRANSMITTING; - } + }); return mask; } + +namespace shadow_finder { + +int BlockedOutTo(const std::vector &baseline, const std::vector &ring_pixels) { + const int max_radius = static_cast(baseline.size()) - 1; + // A ring lies inside the stop when its background is a fraction of what this detector's + // background typically is. Counting statistics cannot decide this: on a bright dataset the + // shadow is still well counted. The comparison used to be against the LARGEST background of any + // ring further out, and that reads a sample whose background peaks in a strong ring away from + // the beam - a powder standard, a strong solvent ring - as a beam stop the size of that ring: + // the ordinary background inside it is legitimately below a third of the peak. On one corpus + // dataset it declared 16 % of the detector to be stop, with diffraction rings visible inside the + // disk it masked. The median over the rings this walk is willing to judge is what "typically" + // means, and it is robust from both sides - to a bright ring, and to a corner ring of a handful + // of pixels whose median is one pixel's mean. + std::vector judgeable; + for (int rad = 0; rad <= max_radius; rad++) + if (ring_pixels[rad] >= MIN_RING_PIXELS) + judgeable.push_back(baseline[rad]); + float typical_background = 0.0f; + if (!judgeable.empty()) { + const auto middle = judgeable.begin() + judgeable.size() / 2; + std::nth_element(judgeable.begin(), middle, judgeable.end()); + typical_background = *middle; + } + + int blocked_out_to = -1; + for (int rad = 0; rad <= max_radius; rad++) { + if (ring_pixels[rad] < MIN_RING_PIXELS) + continue; + if (baseline[rad] >= BLOCKED_RING_RATIO * typical_background) + break; + blocked_out_to = rad; + } + return blocked_out_to; +} + +void HarmonicFit(const std::vector §or_median, int n_bands, + std::vector &harm_c, std::vector &harm_s) { + harm_c.assign(n_bands, 0.0f); + harm_s.assign(n_bands, 0.0f); + for (int band = 0; band < n_bands; band++) { + std::vector med(HARMONIC_SECTORS), c(HARMONIC_SECTORS), s(HARMONIC_SECTORS); + for (int k = 0; k < HARMONIC_SECTORS; k++) { + med[k] = sector_median[static_cast(band) * HARMONIC_SECTORS + k]; + const double phi = (k + 0.5) * 2.0 * std::numbers::pi / HARMONIC_SECTORS - std::numbers::pi; + c[k] = std::cos(2 * phi); + s[k] = std::sin(2 * phi); + } + // Least squares of med = m + p cos 2phi + q sin 2phi over the sectors that are not themselves + // dim against the fit, three times over as the baseline is; the model is relative, 1 + (p cos + q sin)/m. + std::vector use(HARMONIC_SECTORS); + for (int k = 0; k < HARMONIC_SECTORS; k++) + use[k] = med[k] >= 0; + for (int iter = 0; iter < 3; iter++) { + double n = 0, sc = 0, ss = 0, scc = 0, sss = 0, scs = 0, y = 0, yc = 0, ys = 0; + for (int k = 0; k < HARMONIC_SECTORS; k++) { + if (!use[k]) continue; + n++; sc += c[k]; ss += s[k]; scc += c[k] * c[k]; sss += s[k] * s[k]; scs += c[k] * s[k]; + y += med[k]; yc += med[k] * c[k]; ys += med[k] * s[k]; + } + // Sectors crowded into a narrow arc cannot tell a harmonic from a level; the determinant of + // the normal matrix, per sector cubed, is 1/4 on a full ring. + const double det = n * (scc * sss - scs * scs) - sc * (sc * sss - scs * ss) + ss * (sc * scs - scc * ss); + if (n < 6 || det < 0.01 * n * n * n) { + harm_c[band] = harm_s[band] = 0.0f; + break; + } + const double m = (y * (scc * sss - scs * scs) - sc * (yc * sss - scs * ys) + ss * (yc * scs - scc * ys)) / det; + const double p = (n * (yc * sss - ys * scs) - y * (sc * sss - scs * ss) + ss * (sc * ys - yc * ss)) / det; + const double q = (n * (scc * ys - scs * yc) - sc * (sc * ys - yc * ss) + y * (sc * scs - scc * ss)) / det; + if (m <= 0) { + harm_c[band] = harm_s[band] = 0.0f; + break; + } + harm_c[band] = static_cast(p / m); + harm_s[band] = static_cast(q / m); + for (int k = 0; k < HARMONIC_SECTORS; k++) + use[k] = med[k] >= 0 && med[k] >= PENUMBRA_RATIO * (1.0 + harm_c[band] * c[k] + harm_s[band] * s[k]); + } + } +} + +} // namespace shadow_finder diff --git a/image_analysis/beam_stop/ShadowFinder.h b/image_analysis/beam_stop/ShadowFinder.h index 1b7e66ff2..866e3df4d 100644 --- a/image_analysis/beam_stop/ShadowFinder.h +++ b/image_analysis/beam_stop/ShadowFinder.h @@ -4,6 +4,7 @@ #pragma once #include +#include #include #include #include @@ -36,9 +37,9 @@ // // Frames are chosen by the caller; the detection needs enough of them that the background // is counted rather than guessed (see MIN_EXPECTED_COUNTS in the .cpp). -// Thread-safe: workers call AddImage concurrently, each naming a shard of its own (see -// SetShardCount) - so no two threads touch the same accumulator and nothing is locked while -// an image is added. The shards are summed when the projection is read. +// Thread-safe: workers call AddImage concurrently. The projection is split into bands of rows, each +// with a lock of its own, and a worker adding a frame starts at a different band from the one before +// it, so workers meet only when they reach the same band. class ShadowFinder { mutable std::mutex m; @@ -62,22 +63,27 @@ class ShadowFinder { std::vector pixel_mask; // pixels already masked carry no background to test - // Per-pixel projection over the frames added so far (converted geometry). One set per shard: - // the sums and counts are integers, so summing the shards is exact and the result does not - // depend on how the frames were spread over them. + // Per-pixel projection over the frames added so far (converted geometry). The sums and counts are + // integers and the maximum is a maximum, so the result does not depend on the order the frames + // arrive in. struct Projection { std::vector max_value; std::vector sum_value; std::vector valid_count; uint32_t frames = 0; }; - std::vector shards; + // The frames added on the host. Allocated by the first of them: with a GPU there are usually none. + Projection host; + std::mutex host_mutex; // guards the allocation and the frame count, not the sums + static constexpr size_t BANDS = 64; + std::mutex band_mutex[BANDS]; + std::atomic next_band{0}; #ifdef JFJOCH_USE_CUDA // Present when a GPU is available. Frames it can decode are accumulated there instead of on the // host - only the compressed chunk crosses PCIe - and its projection is folded in with the - // shards when the mask is read. Frames it cannot take (anything but bitshuffle+LZ4) still go to - // a host shard, so a run mixing compressions is handled without a second code path. + // host projection when the mask is read. Frames it cannot take (anything but bitshuffle+LZ4) still + // go to the host, so a run mixing compressions is handled without a second code path. // Built on a thread of its own: it allocates and clears several hundred megabytes of device // memory, and cudaMalloc synchronises the whole device, so doing it in the constructor would // stall the caller before it has read its first frame. The first AddImage waits for it, by @@ -90,17 +96,22 @@ class ShadowFinder { [[nodiscard]] ShadowAccumulatorGPU *Gpu() const; #endif - template void Add(const T *ptr, Projection &p); + // Add the pixels [begin, end) of one frame to the host projection. + template void Add(const T *ptr, size_t begin, size_t end); - // Sum the shards into one projection. max_value is only taken from a shard that actually - // counted the pixel - a shard that never saw it holds 0, which would beat a genuinely +#ifdef JFJOCH_USE_CUDA + // The device's projection with the host's folded in. max_value is only taken from a projection + // that actually counted the pixel - one that never saw it holds 0, which would beat a genuinely // negative maximum. [[nodiscard]] Projection Reduce() const; - // The projection, reduced on its first read and kept. The ring-centre fit, the mask and the - // beam-centre capture all read the same one, and on a 16 Mpx detector each reduction is 360 MB - // brought back from the device into fresh memory. Called with `m` held. + // That projection, made on its first read and kept. The ring-centre fit, the mask and the + // beam-centre capture all read the same one, and on a 16 Mpx detector each is 360 MB brought back + // from the device into fresh memory. mutable std::optional reduced; +#endif + // The projection the frames added so far make: the host's, or the one above where the device + // took frames. Called with `m` held. [[nodiscard]] const Projection &Reduced() const; public: @@ -116,20 +127,16 @@ public: // hardware. The projection is not centred on anything, so this may be set after the frames. void BeamCenter(float x, float y); - // Give each worker a shard to accumulate into. Must be called before the first AddImage, - // and costs 20 bytes per pixel per shard. - void SetShardCount(size_t n); - - // Accumulate one full converted-geometry image into shard `shard`. Gap / masked pixels - // (the pixel type's sentinel extreme) are skipped. `buffer` is scratch space for - // decompression, reused across the calls of one worker. - void AddImage(const DataMessage &data, std::vector &buffer, size_t shard = 0); + // Accumulate one full converted-geometry image. Gap / masked pixels (the pixel type's sentinel + // extreme) are skipped. `buffer` is scratch space for decompression, reused across the calls of + // one worker. + void AddImage(const DataMessage &data, std::vector &buffer); // Compute the shadow mask (SHADOW, TRANSMITTING or 0 = keep), of the converted pixel count. // TRANSMITTING marks the pieces of hardware that let part of the beam through, added after the // shadow proper; both are masked, and a consumer that must not see those pieces can tell them apart. - // Recomputed on each call from the projection, which is summed over the shards on the first read - // of it (GetMask or GetMeanProjection): frames added after that are not seen. + // Recomputed on each call from the projection, which is put together on the first read of it + // (GetMask or GetMeanProjection): frames added after that are not seen. // nthreads = 0 asks for all hardware threads. The per-pixel passes over a 16M-pixel detector // dominate this, and they are all exactly parallel. [[nodiscard]] std::vector GetMask(size_t nthreads = 0) const; @@ -140,7 +147,7 @@ public: [[nodiscard]] std::vector GetMeanProjection() const; // Let go of the projection the two above read, once the caller has what it wants of it: on a - // 16 Mpx detector it is 360 MB. A later read sums the shards again. + // 16 Mpx detector it is 360 MB. A later read puts it together again. void ReleaseProjection(); [[nodiscard]] uint32_t GetFrameCount() const; diff --git a/image_analysis/beam_stop/ShadowFinderInternal.h b/image_analysis/beam_stop/ShadowFinderInternal.h new file mode 100644 index 000000000..63e292a70 --- /dev/null +++ b/image_analysis/beam_stop/ShadowFinderInternal.h @@ -0,0 +1,90 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#pragma once + +// What ShadowFinder::GetMask shares with its device twin (ShadowMaskGPU): the constants it tests +// against, and the two small fits made over whole rings and sectors, which stay on the host on both +// paths. + +#include +#include + +namespace shadow_finder { + +// The values of ShadowFinder::SHADOW and ShadowFinder::TRANSMITTING, for the device code, which does +// not include ShadowFinder.h. +inline constexpr uint32_t MASK_SHADOW = 1; +inline constexpr uint32_t MASK_TRANSMITTING = 2; + +// A pixel is shadow when its background is below this fraction of the background it is +// compared against. +inline constexpr float SHADOW_RATIO = 0.50f; + +// The boundary grows outward into partially shadowed pixels down to this fraction, but no +// further than PENUMBRA_MAX_PX from the core. A pin or a loop casts a wide half-shadow, so the +// reach is a good deal more than the beam stop's own edge needs. +inline constexpr float PENUMBRA_RATIO = 0.75f; +inline constexpr int PENUMBRA_MAX_PX = 30; + +// Bridge small breaks along the holder arm. A module gap wider than this is bridged separately for +// the arm search (see bridge_gaps). +inline constexpr int BRIDGE_PX = 6; + +// A pixel whose maximum reaches this recorded a real reflection and is never masked - a +// beam stop cannot block a reflection that was measured. +inline constexpr int64_t MIN_REFLECTION = 25; + +// How far below the background it is compared against a pixel must sit before the dip is +// believed, in standard deviations of the counts that back it. The counts are photons, so their +// scatter is Poisson and the deficit is measured against it rather than against a fixed number: +// on a well-exposed sweep a third of the background missing is overwhelming, and on a handful of +// low-background frames the same third is noise. Without this a six-frame pre-scan of a +// low-background sweep masks three quarters of the detector. +inline constexpr double MIN_DEFICIT_SIGMA = 6.0; + +// Smallest region the per-pixel test may return. A shadow is cast by something physical and is +// correspondingly large; an isolated patch this small is the background wandering, not hardware. +// This is what keeps the test specific now that a shadow no longer has to touch the direct beam. +inline constexpr int MIN_SHADOW_PIXELS = 2000; + +// ... and of those, how many must be deep (below SHADOW_RATIO) for a region of merely DIM pixels - +// hardware that lets part of the beam through - to count. A thin holder arm is dim along most of +// its length and deep in places; the background drifting over a detector's edge is dim everywhere +// and deep nowhere. +inline constexpr int MIN_CORE_PIXELS = 200; + +// The arm search compares a pixel with its ring as the ring actually varies around the beam. A ring +// of background is not flat once divided by the polarization factor when that factor is not the +// beam's: the remainder is a second harmonic in azimuth, cos 2phi, which reaches tens of percent at +// high angle. It is measured over radial bands of HARMONIC_BAND_PX, from the median of each of +// HARMONIC_SECTORS sectors that holds at least MIN_SECTOR_PIXELS pixels. +inline constexpr int HARMONIC_BAND_PX = 64; +inline constexpr int HARMONIC_SECTORS = 24; +inline constexpr int MIN_SECTOR_PIXELS = 200; + +// Side of the box the background is pooled over before testing. Its area is how many pixels back +// a ring's countability test, which decides where an azimuthal comparison is possible at all. +inline constexpr int POOL_PX = 5; +inline constexpr double MEAN_POOLED_PIXELS = POOL_PX * POOL_PX; + +// A ring with fewer valid pixels than this says nothing about whether it was counted. +inline constexpr int MIN_RING_PIXELS = 32; + +// A ring lies wholly inside the stop when its background is below this fraction of the background +// further out. This asks about a whole ring rather than about a pixel, so it keeps a threshold of +// its own and does not follow SHADOW_RATIO. +inline constexpr float BLOCKED_RING_RATIO = 0.35f; + +// The rings that lie wholly inside the stop: walking outward, every judgeable ring (at least +// MIN_RING_PIXELS pixels) before the first whose baseline reaches BLOCKED_RING_RATIO of the typical +// background. -1 when there is none. See GetMask. +int BlockedOutTo(const std::vector &baseline, const std::vector &ring_pixels); + +// The second harmonic in azimuth of each radial band, relative to its level, fitted to the medians of +// its HARMONIC_SECTORS sectors (sector_median[band * HARMONIC_SECTORS + k], negative where the sector +// has too few pixels). See GetMask. +void HarmonicFit(const std::vector §or_median, int n_bands, + std::vector &harm_c, std::vector &harm_s); + +} // namespace shadow_finder diff --git a/image_analysis/beam_stop/ShadowMaskGPU.cu b/image_analysis/beam_stop/ShadowMaskGPU.cu new file mode 100644 index 000000000..5961535c2 --- /dev/null +++ b/image_analysis/beam_stop/ShadowMaskGPU.cu @@ -0,0 +1,636 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#include "ShadowMaskGPU.h" + +#include + +#include "ShadowFinderInternal.h" +#include "../../common/JFJochMath.h" +#include "../indexing/CUDAMemHelpers.h" +#include "../../common/JFJochException.h" + +using namespace shadow_finder; + +namespace { + +constexpr int THREADS = 256; + +void check(cudaError_t err, const char *what) { + if (err != cudaSuccess) + throw JFJochException(JFJochExceptionCategory::GPUCUDAError, + std::string("Beam stop mask: ") + what + ": " + cudaGetErrorString(err)); +} + +unsigned grid(size_t n) { + return static_cast((n + THREADS - 1) / THREADS); +} + +// A float as an unsigned integer that sorts the same way, and back. +__device__ uint32_t float_key(float f) { + const uint32_t u = __float_as_uint(f); + return (u & 0x80000000u) ? ~u : (u | 0x80000000u); +} +__device__ float key_float(uint32_t k) { + return __uint_as_float((k & 0x80000000u) ? (k & 0x7fffffffu) : ~k); +} + +// The first index in a sorted key array whose key is not below `value`. +__device__ size_t lower_bound(const uint64_t *keys, size_t n, uint64_t value) { + size_t lo = 0, hi = n; + while (lo < hi) { + const size_t mid = (lo + hi) / 2; + if (keys[mid] < value) lo = mid + 1; + else hi = mid; + } + return lo; +} + +__device__ double poisson_deficit_sigma(double observed, double expected) { + if (expected <= 0.0 || observed >= expected) + return 0.0; + const double ll = 2.0 * (expected - observed + (observed > 0.0 ? observed * log(observed / expected) : 0.0)); + return ll > 0.0 ? sqrt(ll) : 0.0; +} + +// DiffractionGeometry::CalcAzIntPolarizationCorr about the centre the rings are drawn about. +__device__ float polarization_factor(const ShadowMaskSetup &s, float x, float y) { + const float u = (x - s.beam_x) * s.pixel_size_mm; + const float v = (y - s.beam_y) * s.pixel_size_mm; + const float *m = s.det_matrix; + const float lx = m[0] * u + m[1] * v + m[2] * s.distance_mm; + const float ly = m[3] * u + m[4] * v + m[5] * s.distance_mm; + const float lz = m[6] * u + m[7] * v + m[8] * s.distance_mm; + const float two_theta = atan2f(sqrtf(lx * lx + ly * ly), lz); + float phi = atan2f(ly, lx); + if (phi < 0) + phi += 2.0f * PI; + const float cos_2theta = cosf(two_theta); + const float cos_2theta_2 = cos_2theta * cos_2theta; + const float cos_2phi = cosf(2.0f * phi); + return 0.5f * (1.0f + cos_2theta_2 - s.polarization * cos_2phi * (1.0f - cos_2theta_2)); +} + +__global__ void setup_kernel(ShadowMaskSetup s, const uint32_t *__restrict__ pixel_mask, + const int64_t *__restrict__ sum_value, const uint32_t *__restrict__ valid_count, + float *__restrict__ pol, char *__restrict__ valid, int *__restrict__ radius, + double *__restrict__ num, int32_t *__restrict__ den, int *__restrict__ max_radius) { + const size_t n = static_cast(s.width) * s.height; + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) + return; + const int x = static_cast(i % s.width), y = static_cast(i / s.width); + const float dx = x - s.beam_x, dy = y - s.beam_y; + const float p = s.has_polarization ? polarization_factor(s, static_cast(x), static_cast(y)) : 1.0f; + pol[i] = p; + float mean = 0.0f; + char v = 0; + if (valid_count[i] > 0 && pixel_mask[i] == 0 && p > 0.0f) { + mean = static_cast(static_cast(sum_value[i]) / valid_count[i] / p); + v = 1; + } + valid[i] = v; + num[i] = v ? mean : 0.0; + den[i] = v ? 1 : 0; + const int r = static_cast(lroundf(sqrtf(dx * dx + dy * dy))); + radius[i] = r; + atomicMax(max_radius, r); +} + +// Sum over the k x k box about each pixel, zero outside the frame: one running sum per row, then one +// per column, each with exactly the terms and order of the host's (box_sum in ShadowFinder.cpp). +template +__global__ void box_rows(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) { + const int y = blockIdx.x * blockDim.x + threadIdx.x; + if (y >= H) return; + const T *src = in + static_cast(y) * W; + T *dst = out + static_cast(y) * W; + T s = 0; + for (int x = 0; x <= min(half, W - 1); x++) + s += src[x]; + for (int x = 0; x < W; x++) { + dst[x] = s; + if (x + half + 1 < W) s += src[x + half + 1]; + if (x - half >= 0) s -= src[x - half]; + } +} + +template +__global__ void box_columns(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) { + const int x = blockIdx.x * blockDim.x + threadIdx.x; + if (x >= W) return; + T s = 0; + for (int y = 0; y <= min(half, H - 1); y++) + s += in[static_cast(y) * W + x]; + for (int y = 0; y < H; y++) { + out[static_cast(y) * W + x] = s; + if (y + half + 1 < H) s += in[static_cast(y + half + 1) * W + x]; + if (y - half >= 0) s -= in[static_cast(y - half) * W + x]; + } +} + +// Dilation of a 0/1 plane by the (2r+1) square clipped to the frame, as a count over a sliding window +// along rows and then along columns (dilate in ShadowFinder.cpp). +__global__ void dilate_rows(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) { + const int y = blockIdx.x * blockDim.x + threadIdx.x; + if (y >= H) return; + const char *src = in + static_cast(y) * W; + char *dst = out + static_cast(y) * W; + int count = 0; + for (int x = 0; x <= min(r, W - 1); x++) + count += src[x]; + for (int x = 0; x < W; x++) { + dst[x] = count > 0; + if (x + r + 1 < W) count += src[x + r + 1]; + if (x - r >= 0) count -= src[x - r]; + } +} + +__global__ void dilate_columns(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) { + const int x = blockIdx.x * blockDim.x + threadIdx.x; + if (x >= W) return; + int count = 0; + for (int y = 0; y <= min(r, H - 1); y++) + count += in[static_cast(y) * W + x]; + for (int y = 0; y < H; y++) { + out[static_cast(y) * W + x] = count > 0; + if (y + r + 1 < H) count += in[static_cast(y + r + 1) * W + x]; + if (y - r >= 0) count -= in[static_cast(y - r) * W + x]; + } +} + +__global__ void pooled_kernel(size_t n, const double *__restrict__ pooled_sum, const int32_t *__restrict__ pooled_count, + const char *__restrict__ valid, const int *__restrict__ radius, + float *__restrict__ pooled, uint64_t *__restrict__ ring_key) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + const float p = pooled_count[i] > 0 ? static_cast(pooled_sum[i] / pooled_count[i]) : 0.0f; + pooled[i] = p; + ring_key[i] = valid[i] ? (static_cast(radius[i]) << 32) | float_key(p) : UINT64_MAX; +} + +// Where each of the keys' leading 32-bit groups (ring or sector) starts in a sorted key array. +__global__ void group_offsets(const uint64_t *__restrict__ keys, size_t n, int groups, int *__restrict__ offset) { + const int g = blockIdx.x * blockDim.x + threadIdx.x; + if (g > groups) return; + offset[g] = static_cast(lower_bound(keys, n, static_cast(g) << 32)); +} + +// The ring's baseline, three iterations of an order statistic over its sorted values (GetMask). +__global__ void baseline_kernel(int rings, const uint64_t *__restrict__ keys, const int *__restrict__ offset, + float *__restrict__ baseline) { + const int r = blockIdx.x * blockDim.x + threadIdx.x; + if (r >= rings) return; + const int lo = offset[r], n = offset[r + 1] - offset[r]; + int excluded = 0; + float b = 0.0f; + for (int iter = 0; iter < 3; iter++) { + const int avail = n - excluded; + b = (avail <= 0) ? 0.0f : key_float(static_cast(keys[lo + excluded + avail / 2])); + const float d = fmaxf(b, 1e-6f); + int excl = 0; + while (excl < n && key_float(static_cast(keys[lo + excl])) / d < SHADOW_RATIO) + excl++; + excluded = excl; + } + baseline[r] = b; +} + +__global__ void low_kernel(size_t n, uint32_t frames, const char *__restrict__ valid, const float *__restrict__ pooled, + const int32_t *__restrict__ pooled_count, const float *__restrict__ pol, + const int *__restrict__ radius, const float *__restrict__ baseline, + float *__restrict__ ratio, float *__restrict__ deficit, char *__restrict__ low) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + const float base = baseline[radius[i]]; + const float rt = valid[i] ? pooled[i] / fmaxf(base, 1e-6f) : 1.0f; + ratio[i] = rt; + float df = 0.0f; + char l = 0; + if (valid[i]) { + const double counted = static_cast(frames) * pooled_count[i] * pol[i]; + df = static_cast(poisson_deficit_sigma(pooled[i] * counted, base * counted)); + l = rt < SHADOW_RATIO && df > MIN_DEFICIT_SIGMA; + } + deficit[i] = df; + low[i] = l; +} + +// 8-connected components by union-find: every component ends up named by its smallest pixel index, +// whatever order the unions ran in. +__device__ int find_root(const int *parent, int x) { + while (parent[x] != x) + x = parent[x]; + return x; +} + +__global__ void cc_init(size_t n, const char *__restrict__ member, int *__restrict__ parent) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + parent[i] = member[i] ? static_cast(i) : -1; +} + +__device__ void cc_unite(int *parent, int a, int b) { + while (true) { + a = find_root(parent, a); + b = find_root(parent, b); + if (a == b) return; + if (a < b) { const int t = a; a = b; b = t; } + if (atomicCAS(&parent[a], a, b) == a) return; + } +} + +__global__ void cc_union(int W, int H, const char *__restrict__ member, int *parent) { + const size_t n = static_cast(W) * H; + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n || !member[i]) return; + const int x = static_cast(i % W), y = static_cast(i / W); + // The four neighbours before this pixel; the other four see it from their side. + if (x > 0 && member[i - 1]) cc_unite(parent, static_cast(i), static_cast(i - 1)); + if (y > 0) { + const size_t up = i - W; + if (member[up]) cc_unite(parent, static_cast(i), static_cast(up)); + if (x > 0 && member[up - 1]) cc_unite(parent, static_cast(i), static_cast(up - 1)); + if (x + 1 < W && member[up + 1]) cc_unite(parent, static_cast(i), static_cast(up + 1)); + } +} + +__global__ void cc_flatten(size_t n, int *parent) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n || parent[i] < 0) return; + parent[i] = find_root(parent, static_cast(i)); +} + +// Per component (by root): how many of its pixels are `a`, and how many are both `a` and `b`. +__global__ void cc_count(size_t n, const int *__restrict__ root, const char *__restrict__ a, const char *__restrict__ b, + int *__restrict__ count_a, int *__restrict__ count_ab) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n || root[i] < 0 || !a[i]) return; + atomicAdd(&count_a[root[i]], 1); + if (count_ab && b[i]) + atomicAdd(&count_ab[root[i]], 1); +} + +__global__ void region_kernel(size_t n, const int *__restrict__ root, const int *__restrict__ n_low, + const char *__restrict__ low, const char *__restrict__ valid, const int *__restrict__ radius, + int blocked_out_to, char *__restrict__ region) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + char r = root[i] >= 0 && n_low[root[i]] >= MIN_SHADOW_PIXELS ? low[i] : 0; + if (valid[i] && radius[i] <= blocked_out_to) + r = 1; + region[i] = r; +} + +__global__ void lit_kernel(size_t n, const uint32_t *__restrict__ valid_count, const int64_t *__restrict__ max_value, + char *__restrict__ lit) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + lit[i] = (valid_count[i] > 0) && (max_value[i] >= MIN_REFLECTION); +} + +__global__ void reflection_kernel(int W, int H, const char *__restrict__ lit, char *__restrict__ reflection) { + const size_t n = static_cast(W) * H; + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + const int x = static_cast(i % W), y = static_cast(i / W); + char r = 0; + if (lit[i]) { + int neighbours = 0; + for (int dy = -1; dy <= 1; dy++) + for (int dx = -1; dx <= 1; dx++) { + const int yy = y + dy, xx = x + dx; + if ((dx || dy) && yy >= 0 && yy < H && xx >= 0 && xx < W && lit[static_cast(yy) * W + xx]) + neighbours++; + } + r = neighbours >= 2; + } + reflection[i] = r; +} + +__global__ void penumbra_kernel(size_t n, const char *__restrict__ penumbra, const char *__restrict__ valid, + const float *__restrict__ ratio, const float *__restrict__ deficit, char *__restrict__ region) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + if (penumbra[i] && valid[i] && ratio[i] < PENUMBRA_RATIO && deficit[i] > MIN_DEFICIT_SIGMA) + region[i] = 1; +} + +__global__ void invert_kernel(size_t n, const char *__restrict__ in, char *__restrict__ out) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + out[i] = !in[i]; +} + +// Background components that touch the frame's edge are outside; the rest are holes. +__global__ void outside_kernel(int W, int H, const int *__restrict__ root, int *__restrict__ outside) { + const int k = blockIdx.x * blockDim.x + threadIdx.x; + const int perimeter = 2 * W + 2 * H; + if (k >= perimeter) return; + int x, y; + if (k < W) { x = k; y = 0; } + else if (k < 2 * W) { x = k - W; y = H - 1; } + else if (k < 2 * W + H) { x = 0; y = k - 2 * W; } + else { x = W - 1; y = k - 2 * W - H; } + const int r = root[static_cast(y) * W + x]; + if (r >= 0) outside[r] = 1; +} + +__global__ void final_kernel(size_t n, const int *__restrict__ background_root, const int *__restrict__ outside, + const char *__restrict__ reflection_grown, char *__restrict__ region, + uint32_t *__restrict__ mask) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + char r = region[i]; + if (background_root[i] >= 0 && !outside[background_root[i]]) + r = 1; // a hole + if (reflection_grown[i]) + r = 0; + region[i] = r; + mask[i] = r ? MASK_SHADOW : 0; +} + +__global__ void sector_key_kernel(ShadowMaskSetup s, const char *__restrict__ valid, const char *__restrict__ region, + const int *__restrict__ radius, const float *__restrict__ ratio, + uint64_t *__restrict__ key) { + const size_t n = static_cast(s.width) * s.height; + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + if (!valid[i] || region[i]) { + key[i] = UINT64_MAX; + return; + } + const float dx = static_cast(i % s.width) - s.beam_x, dy = static_cast(i / s.width) - s.beam_y; + const double phi = atan2f(dy, dx) + PI; + const int sector = min(HARMONIC_SECTORS - 1, static_cast(phi / (2.0 * PI) * HARMONIC_SECTORS)); + const uint64_t k = static_cast(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector; + key[i] = (k << 32) | float_key(ratio[i]); +} + +__global__ void sector_median_kernel(int sectors, const uint64_t *__restrict__ keys, const int *__restrict__ offset, + double *__restrict__ median) { + const int k = blockIdx.x * blockDim.x + threadIdx.x; + if (k >= sectors) return; + const int n = offset[k + 1] - offset[k]; + median[k] = n >= MIN_SECTOR_PIXELS ? key_float(static_cast(keys[offset[k] + n / 2])) : -1.0; +} + +__global__ void dim_kernel(ShadowMaskSetup s, uint32_t frames, const char *__restrict__ valid, + const char *__restrict__ region, const float *__restrict__ ratio, + const float *__restrict__ deficit, const int *__restrict__ radius, + const float *__restrict__ harm_c, const float *__restrict__ harm_s, + const int32_t *__restrict__ pooled_count, const float *__restrict__ pol, + const float *__restrict__ pooled, const float *__restrict__ baseline, + char *__restrict__ dim) { + const size_t n = static_cast(s.width) * s.height; + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + char d = 0; + if (valid[i] && !region[i] && ratio[i] < PENUMBRA_RATIO) { + // cos 2phi and sin 2phi from the offset to the beam. + const float dx = static_cast(i % s.width) - s.beam_x, dy = static_cast(i / s.width) - s.beam_y; + const float r2 = fmaxf(dx * dx + dy * dy, 1e-6f); + const int band = radius[i] / HARMONIC_BAND_PX; + const float model = fminf(1.0f, 1.0f + harm_c[band] * (dx * dx - dy * dy) / r2 + + harm_s[band] * 2.0f * dx * dy / r2); + if (model >= 1.0f) { + d = deficit[i] > MIN_DEFICIT_SIGMA; + } else { + const double counted = static_cast(frames) * pooled_count[i] * pol[i]; + d = ratio[i] < PENUMBRA_RATIO * model + && poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA; + } + } + dim[i] = d; +} + +// Join a region across the module gaps it crosses, one line per thread (bridge_gaps in +// ShadowFinder.cpp). Both directions read `region` and only ever set pixels of `out` to 1. +__global__ void bridge_rows(int W, int H, const char *__restrict__ region, const char *__restrict__ valid, + char *out) { + const int y = blockIdx.x * blockDim.x + threadIdx.x; + if (y >= H) return; + const size_t row = static_cast(y) * W; + int k = 0; + while (k < W) { + if (valid[row + k]) { k++; continue; } + const int start = k; + while (k < W && !valid[row + k]) k++; + if (start > 0 && k < W && region[row + start - 1] && region[row + k]) + for (int j = start; j < k; j++) out[row + j] = 1; + } +} + +__global__ void bridge_columns(int W, int H, const char *__restrict__ region, const char *__restrict__ valid, + char *out) { + const int x = blockIdx.x * blockDim.x + threadIdx.x; + if (x >= W) return; + const auto at = [&](int y) { return static_cast(y) * W + x; }; + int k = 0; + while (k < H) { + if (valid[at(k)]) { k++; continue; } + const int start = k; + while (k < H && !valid[at(k)]) k++; + if (start > 0 && k < H && region[at(start - 1)] && region[at(k)]) + for (int j = start; j < k; j++) out[at(j)] = 1; + } +} + +__global__ void transmitting_kernel(size_t n, const int *__restrict__ root, const char *__restrict__ dim, + const int *__restrict__ n_dim, const int *__restrict__ n_low, + uint32_t *__restrict__ mask) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + const int c = root[i]; + if (c >= 0 && dim[i] && n_dim[c] >= MIN_SHADOW_PIXELS && n_low[c] >= MIN_CORE_PIXELS) + mask[i] = MASK_TRANSMITTING; +} + +__global__ void mean_kernel(size_t n, const uint32_t *__restrict__ pixel_mask, const int64_t *__restrict__ sum_value, + const uint32_t *__restrict__ valid_count, float *__restrict__ mean) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= n) return; + mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0 + ? static_cast(static_cast(sum_value[i]) / valid_count[i]) : NAN; +} + +// The device side of one GetMask: planes, sort scratch and the stream to run on. +class MaskEngine { +public: + const ShadowMaskSetup s; + const int W, H; + const size_t n; + cudaStream_t stream; + + MaskEngine(const ShadowMaskSetup &setup, cudaStream_t st) + : s(setup), W(setup.width), H(setup.height), n(static_cast(setup.width) * setup.height), stream(st) {} + + void Check(const char *what) const { + check(cudaGetLastError(), what); + } + + void Dilate(const char *in, char *out, char *scratch, int r) const { + dilate_rows<<>>(in, scratch, W, H, r); + dilate_columns<<>>(scratch, out, W, H, r); + Check("dilate"); + } + + // Components of `member`, each pixel's root in `root` (-1 outside every component). + void Label(const char *member, int *root) const { + cc_init<<>>(n, member, root); + cc_union<<>>(W, H, member, root); + cc_flatten<<>>(n, root); + Check("components"); + } + + void SortKeys(uint64_t *keys, uint64_t *sorted) const { + size_t bytes = 0; + check(cub::DeviceRadixSort::SortKeys(nullptr, bytes, keys, sorted, n, 0, 64, stream), "sort size"); + CudaDevicePtr scratch(bytes); + check(cub::DeviceRadixSort::SortKeys(scratch.get(), bytes, keys, sorted, n, 0, 64, stream), "sort"); + // The scratch is freed on the allocation stream, which knows nothing of this one. + check(cudaStreamSynchronize(stream), "sort"); + } + + template + std::vector Download(const T *device, size_t count) const { + std::vector host(count); + check(cudaMemcpyAsync(host.data(), device, count * sizeof(T), cudaMemcpyDeviceToHost, stream), "download"); + check(cudaStreamSynchronize(stream), "download"); + return host; + } + + template + void Upload(T *device, const std::vector &host) const { + check(cudaMemcpyAsync(device, host.data(), host.size() * sizeof(T), cudaMemcpyHostToDevice, stream), "upload"); + } +}; + +} // namespace + +std::vector ShadowMaskOnDevice(const ShadowMaskSetup &setup, const std::vector &pixel_mask, + const int64_t *max_value, const int64_t *sum_value, + const uint32_t *valid_count, uint32_t frames, cudaStream_t stream) { + const MaskEngine e(setup, stream); + const size_t n = e.n; + const int W = e.W, H = e.H; + + CudaDevicePtr d_pixel_mask(n), mask(n); + e.Upload(d_pixel_mask.get(), pixel_mask); + + // Mean projection over the polarization factor, usable pixels and radius from the beam centre. + CudaDevicePtr pol(n), pooled(n), ratio(n), deficit(n); + CudaDevicePtr valid(n), low(n), region(n), a(n), b(n), c(n); + CudaDevicePtr radius(n), root(n), count_a(n), count_b(n), max_radius(1); + CudaDevicePtr num(n), dsum(n); + CudaDevicePtr den(n), dcount(n); + check(cudaMemsetAsync(max_radius.get(), 0, sizeof(int), stream), "memset"); + setup_kernel<<>>(setup, d_pixel_mask, sum_value, valid_count, pol, valid, radius, + num, den, max_radius); + e.Check("setup"); + const int max_r = e.Download(max_radius.get(), 1)[0]; + const int rings = max_r + 1; + + // The background pooled over a small box. + box_rows<<>>(num, dsum, W, H, POOL_PX / 2); + box_columns<<>>(dsum, num, W, H, POOL_PX / 2); + box_rows<<>>(den, dcount, W, H, POOL_PX / 2); + box_columns<<>>(dcount, den, W, H, POOL_PX / 2); + e.Check("pooling"); + const double *pooled_sum = num; + const int32_t *pooled_count = den; + + // The rings, each sorted once; the baseline is an order statistic of them. + CudaDevicePtr keys(n), sorted(n); + pooled_kernel<<>>(n, pooled_sum, pooled_count, valid, radius, pooled, keys); + e.Check("pooled"); + e.SortKeys(keys, sorted); + CudaDevicePtr ring_offset(rings + 1); + group_offsets<<>>(sorted, n, rings, ring_offset); + CudaDevicePtr baseline(rings); + baseline_kernel<<>>(rings, sorted, ring_offset, baseline); + e.Check("baseline"); + const auto host_baseline = e.Download(baseline.get(), rings); + const auto offsets = e.Download(ring_offset.get(), rings + 1); + std::vector ring_pixels(rings); + for (int r = 0; r < rings; r++) + ring_pixels[r] = offsets[r + 1] - offsets[r]; + const int blocked_out_to = BlockedOutTo(host_baseline, ring_pixels); + + // Low pixels, and the regions of them large enough to be hardware. + low_kernel<<>>(n, frames, valid, pooled, pooled_count, pol, radius, baseline, + ratio, deficit, low); + e.Check("low"); + e.Dilate(low, a, c, BRIDGE_PX); + e.Label(a, root); + check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset"); + cc_count<<>>(n, root, low, low, count_a, nullptr); + region_kernel<<>>(n, root, count_a, low, valid, radius, blocked_out_to, region); + e.Check("region"); + + // Recorded reflections; `b` holds them until they are given back at the end. + lit_kernel<<>>(n, valid_count, max_value, a); + reflection_kernel<<>>(W, H, a, b); + e.Check("reflections"); + + // Penumbra, round and fill. + e.Dilate(region, a, c, PENUMBRA_MAX_PX); + penumbra_kernel<<>>(n, a, valid, ratio, deficit, region); + e.Dilate(region, a, c, 2); + invert_kernel<<>>(n, a, region); + e.Dilate(region, a, c, 2); + invert_kernel<<>>(n, a, region); // region = erode(dilate(region)) + invert_kernel<<>>(n, region, a); // the background + e.Label(a, root); + check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset"); + outside_kernel<<>>(W, H, root, count_a); + e.Dilate(b, c, a, 1); // the reflections, grown by one + final_kernel<<>>(n, root, count_a, c, region, mask); + e.Check("fill"); + + // The arm search: sector medians of the ratio, the harmonic of each band, the dim pixels. + const int n_bands = max_r / HARMONIC_BAND_PX + 1; + const int n_sectors = n_bands * HARMONIC_SECTORS; + sector_key_kernel<<>>(setup, valid, region, radius, ratio, keys); + e.Check("sectors"); + e.SortKeys(keys, sorted); + CudaDevicePtr sector_offset(n_sectors + 1); + group_offsets<<>>(sorted, n, n_sectors, sector_offset); + CudaDevicePtr sector_median(n_sectors); + sector_median_kernel<<>>(n_sectors, sorted, sector_offset, sector_median); + e.Check("sector medians"); + std::vector harm_c, harm_s; + HarmonicFit(e.Download(sector_median.get(), n_sectors), n_bands, harm_c, harm_s); + CudaDevicePtr d_harm_c(n_bands), d_harm_s(n_bands); + e.Upload(d_harm_c.get(), harm_c); + e.Upload(d_harm_s.get(), harm_s); + dim_kernel<<>>(setup, frames, valid, region, ratio, deficit, radius, d_harm_c, d_harm_s, + pooled_count, pol, pooled, baseline, a); + e.Check("dim"); + e.Dilate(a, b, c, BRIDGE_PX); + check(cudaMemcpyAsync(c.get(), b.get(), n, cudaMemcpyDeviceToDevice, stream), "copy"); + bridge_rows<<>>(W, H, b, valid, c); + bridge_columns<<>>(W, H, b, valid, c); + e.Label(c, root); + check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset"); + check(cudaMemsetAsync(count_b.get(), 0, n * sizeof(int), stream), "memset"); + cc_count<<>>(n, root, a, low, count_a, count_b); + transmitting_kernel<<>>(n, root, a, count_a, count_b, mask); + e.Check("transmitting"); + + return e.Download(mask.get(), n); +} + +std::vector MeanProjectionOnDevice(const std::vector &pixel_mask, const int64_t *sum_value, + const uint32_t *valid_count, size_t npixels, cudaStream_t stream) { + CudaDevicePtr d_pixel_mask(npixels); + CudaDevicePtr mean(npixels); + check(cudaMemcpyAsync(d_pixel_mask.get(), pixel_mask.data(), npixels * sizeof(uint32_t), cudaMemcpyHostToDevice, + stream), "upload"); + mean_kernel<<>>(npixels, d_pixel_mask, sum_value, valid_count, mean); + check(cudaGetLastError(), "mean"); + std::vector host(npixels); + check(cudaMemcpyAsync(host.data(), mean.get(), npixels * sizeof(float), cudaMemcpyDeviceToHost, stream), "download"); + check(cudaStreamSynchronize(stream), "mean"); + return host; +} diff --git a/image_analysis/beam_stop/ShadowMaskGPU.h b/image_analysis/beam_stop/ShadowMaskGPU.h new file mode 100644 index 000000000..baa25a00f --- /dev/null +++ b/image_analysis/beam_stop/ShadowMaskGPU.h @@ -0,0 +1,39 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#pragma once + +// Included only under JFJOCH_USE_CUDA. + +#include +#include + +#include + +// What the mask is drawn about: the detector, the centre the rings are drawn about, and the geometry +// the polarization factor is read off (ShadowFinder keeps it as a DiffractionGeometry; the device +// takes it as numbers). +struct ShadowMaskSetup { + int width = 0, height = 0; + float beam_x = 0.0f, beam_y = 0.0f; + float det_matrix[9] = {}; // row major + float pixel_size_mm = 0.0f; + float distance_mm = 0.0f; + bool has_polarization = false; + float polarization = 0.0f; +}; + +// ShadowFinder::GetMask on the device, from the projection ShadowAccumulatorGPU holds there. Step for +// step the host's algorithm - the same pooling, ring medians, components, morphology and arm search - +// and the same answer wherever the arithmetic is exact: every integer, comparison, sort and component +// is. What is not is the floating point the two compilers evaluate differently - the polarization +// factor's trigonometry, the Poisson test's logarithm and the azimuth of the arm search - so a pixel +// within a rounding of one of those thresholds can come out the other way. +std::vector ShadowMaskOnDevice(const ShadowMaskSetup &setup, const std::vector &pixel_mask, + const int64_t *max_value, const int64_t *sum_value, + const uint32_t *valid_count, uint32_t frames, cudaStream_t stream); + +// The mean projection the host's GetMeanProjection makes, computed where the sums are: the same +// division, so the same bits, and a quarter of the bytes to bring back. +std::vector MeanProjectionOnDevice(const std::vector &pixel_mask, const int64_t *sum_value, + const uint32_t *valid_count, size_t npixels, cudaStream_t stream); diff --git a/image_analysis/bragg_prediction/BraggPredictionRot.cpp b/image_analysis/bragg_prediction/BraggPredictionRot.cpp index 33b022a6c..e9deb1428 100644 --- a/image_analysis/bragg_prediction/BraggPredictionRot.cpp +++ b/image_analysis/bragg_prediction/BraggPredictionRot.cpp @@ -45,6 +45,7 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys const Coord m3 = (m1 % m2).Normalize(); const float m2_S0 = m2 * S0; + const float four_S0_sq = 4 * S0 * S0; const float m3_S0 = m3 * S0; int i = 0; @@ -99,17 +100,24 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys cos_phi_limit = std::cos(phi_limit); } + // p0 = A* h + B* k + C* l, evaluated as ((A* h) + (B* k)) + (C* l) exactly as before, with the + // terms that do not change in the inner loops taken out of them. + std::vector Cstar_l(2 * settings.max_l + 1); + for (int l = -settings.max_l; l <= settings.max_l; l++) + Cstar_l[l + settings.max_l] = Cstar * l; + for (int h = -settings.max_h; h <= settings.max_h; h++) { - // Precompute A* h contribution + const Coord Astar_h = Astar * h; for (int k = -settings.max_k; k <= settings.max_k; k++) { - // Accumulate B* k contribution + const Coord AB = Astar_h + Bstar * k; for (int l = -settings.max_l; l <= settings.max_l; l++) { if (systematic_absence(h, k, l, settings.centering)) continue; - Coord p0 = Astar * h + Bstar * k + Cstar * l; + const Coord &Cl = Cstar_l[l + settings.max_l]; + const Coord p0(AB.x + Cl.x, AB.y + Cl.y, AB.z + Cl.z); float p0_sq = p0 * p0; if (p0_sq <= 0.0f || p0_sq > one_over_dmax_sq) @@ -129,7 +137,7 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys }; // No solution for Laue equations - if ((rho_sq < p_m3 * p_m3) || (p0_sq > 4 * S0 * S0)) + if ((rho_sq < p_m3 * p_m3) || (p0_sq > four_S0_sq)) continue; // Effective rocking width for this reflection: mosaicity broadened by the bandwidth diff --git a/image_analysis/geom_refinement/BackgroundBand.h b/image_analysis/geom_refinement/BackgroundBand.h new file mode 100644 index 000000000..394ab3e84 --- /dev/null +++ b/image_analysis/geom_refinement/BackgroundBand.h @@ -0,0 +1,107 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#pragma once + +// Where one pixel falls in the background beam-centre fit (FindBeamCenterFromBackground): which +// radial-bin x sector cell of the fitted band it lands in about a trial centre, and the derivative +// of its 2theta with respect to that centre. Written once and compiled both for the host fit and for +// its device twin (BeamCenterBackgroundGPU), so the two read the same formula. +// +// The same formula is the same number only when both sides evaluate it the same way. The library +// atan2f differs between glibc and CUDA in the last bit, and the compilers fuse multiply-adds each +// in their own way; either moves a pixel within a rounding of a cell edge into the neighbouring cell +// (measured on 16 Mpx sweeps: ~30 of 6.5 million pixels, enough to move the fitted centre by up to +// 0.05 px). So the angles come from BackgroundAtan2 below and both translation units are compiled +// without contraction (geom_refinement/CMakeLists.txt), and the host and the device then agree to +// the bit. + +#include +#include + +#include "../../common/JFJochMath.h" + +#ifdef __CUDACC__ +#define BACKGROUND_BAND_HD __host__ __device__ inline +#else +#define BACKGROUND_BAND_HD inline +#endif + +struct BackgroundBand { + static constexpr int SECTORS = 36; + static constexpr int RADIAL_BINS = 120; + static constexpr int CELLS = RADIAL_BINS * SECTORS; + + float rot[9]; // detector matrix, row major + float pixel_size; // mm + float distance; // mm + float tt_lo, tt_hi; // the band in 2theta + float d_tt; // width of one radial bin + double tan_lo, tan_hi; // the band in tan(2theta), widened for the quick rejection +}; + +// atan2(y, x) from IEEE operations alone - add, multiply, divide, square root, all correctly rounded on +// the host and on the device - so that, compiled without contraction (see CMakeLists.txt), the host +// and the device return the same bits. The library atan2f does not: glibc's and CUDA's differ in the +// last place. Two half-angle reductions take the argument below tan(pi/16), where eleven terms of the +// series leave an error under 1e-17. +BACKGROUND_BAND_HD double BackgroundAtan2(double y, double x) { + const double ax = x < 0 ? -x : x, ay = y < 0 ? -y : y; + if (ax == 0.0 && ay == 0.0) + return 0.0; + const bool swap = ay > ax; + double t = swap ? ax / ay : ay / ax; + t = t / (1.0 + sqrt(1.0 + t * t)); + t = t / (1.0 + sqrt(1.0 + t * t)); + const double t2 = t * t; + double series = 1.0 / 23.0; + for (int k = 21; k >= 1; k -= 2) + series = 1.0 / k - t2 * series; + double a = 4.0 * t * series; + if (swap) a = PI / 2 - a; + if (x < 0) a = PI - a; + return y < 0 ? -a : a; +} + +// The cell of pixel (x, y) about (beam_x, beam_y), or -1 when it is outside the band; for a pixel in +// it, also the two components of the derivative of its 2theta with respect to the centre. +BACKGROUND_BAND_HD int BackgroundBandCell(const BackgroundBand &b, int x, int y, float beam_x, float beam_y, + float &jac_x, float &jac_y) { + const float *rot = b.rot; + const float u = (x - beam_x) * b.pixel_size; + const float v = (y - beam_y) * b.pixel_size; + const float lx = rot[0] * u + rot[1] * v + rot[2] * b.distance; + const float ly = rot[3] * u + rot[4] * v + rot[5] * b.distance; + const float lz = rot[6] * u + rot[7] * v + rot[8] * b.distance; + const float rho_sq = lx * lx + ly * ly; + // Most pixels lie outside the band. Those clearly outside it in tan(2theta) = rho / lz - by a + // margin far above float rounding - skip the square root and both atan2 below; every pixel the + // exact test would keep still reaches it. + if (lz > 0.0f) { + const double lz_sq = static_cast(lz) * lz; + if (rho_sq < b.tan_lo * b.tan_lo * lz_sq || rho_sq > b.tan_hi * b.tan_hi * lz_sq) + return -1; + } + const float rho = sqrtf(rho_sq); + const float two_theta = static_cast(BackgroundAtan2(rho, lz)); + if (two_theta < b.tt_lo || two_theta >= b.tt_hi || rho == 0.0f) + return -1; + + const float phi = static_cast(BackgroundAtan2(ly, lx)); + // Both bins are clamped: a pixel one float ulp below the top of the band divides to exactly + // RADIAL_BINS, which is one cell past the end of every accumulator. + int r_bin = static_cast((two_theta - b.tt_lo) / b.d_tt); + r_bin = r_bin < 0 ? 0 : (r_bin > BackgroundBand::RADIAL_BINS - 1 ? BackgroundBand::RADIAL_BINS - 1 : r_bin); + int s_bin = static_cast((phi + PI) / (2 * PI) * BackgroundBand::SECTORS); + s_bin = s_bin < 0 ? 0 : (s_bin > BackgroundBand::SECTORS - 1 ? BackgroundBand::SECTORS - 1 : s_bin); + + // d(2theta)/d(beam), through the lab coordinate: the detector coordinate depends on the centre + // only as (x - beam_x), so moving the centre is moving the pixel. + const float denominator = rho * rho + lz * lz; + const float g_x = lz * lx / (rho * denominator); + const float g_y = lz * ly / (rho * denominator); + const float g_z = -rho / denominator; + jac_x = -b.pixel_size * (g_x * rot[0] + g_y * rot[3] + g_z * rot[6]); + jac_y = -b.pixel_size * (g_x * rot[1] + g_y * rot[4] + g_z * rot[7]); + return r_bin * BackgroundBand::SECTORS + s_bin; +} diff --git a/image_analysis/geom_refinement/BeamCenterBackgroundGPU.cu b/image_analysis/geom_refinement/BeamCenterBackgroundGPU.cu new file mode 100644 index 000000000..550593f1d --- /dev/null +++ b/image_analysis/geom_refinement/BeamCenterBackgroundGPU.cu @@ -0,0 +1,210 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#include "BeamCenterBackgroundGPU.h" + +#include + +#include "../indexing/CUDAMemHelpers.h" + +namespace { + +constexpr int THREADS = 256; + +void check(cudaError_t err, const char *what) { + if (err != cudaSuccess) + throw JFJochException(JFJochExceptionCategory::GPUCUDAError, + std::string("Beam centre from background: ") + what + ": " + cudaGetErrorString(err)); +} + +// Every pixel's cell (BackgroundBand::CELLS where it is outside the band or unusable), its index, +// and its derivatives. +__global__ void bin_kernel(BackgroundBand band, int width, size_t npixels, float beam_x, float beam_y, + const char *__restrict__ usable, int32_t *__restrict__ key, + int32_t *__restrict__ index, float *__restrict__ jac_x, + float *__restrict__ jac_y) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= npixels) + return; + int cell = BackgroundBand::CELLS; + float jx = 0.0f, jy = 0.0f; + if (usable[i]) { + const int c = BackgroundBandCell(band, static_cast(i % width), static_cast(i / width), + beam_x, beam_y, jx, jy); + if (c >= 0) + cell = c; + } + key[i] = cell; + index[i] = static_cast(i); + jac_x[i] = jx; + jac_y[i] = jy; +} + +// Where each cell's pixels start in the sorted keys; offset[CELLS] is the number of binned pixels. +__global__ void offset_kernel(const int32_t *__restrict__ sorted_key, size_t n, int32_t *__restrict__ offset) { + const int c = blockIdx.x * blockDim.x + threadIdx.x; + if (c > BackgroundBand::CELLS) + return; + size_t lo = 0, hi = n; + while (lo < hi) { + const size_t mid = (lo + hi) / 2; + if (sorted_key[mid] < c) lo = mid + 1; + else hi = mid; + } + offset[c] = static_cast(lo); +} + +// One thread per cell, walking its pixels in pixel order: a partial sum per host row block, added to +// the cell's total when the block changes - the host's arithmetic, step for step. A cell whose pixels +// are clipped (clip_limit != nullptr) skips the ones above its limit, as the host's clip rounds do. +__global__ void sum_kernel(int width, const int32_t *__restrict__ offset, const int32_t *__restrict__ index, + const int32_t *__restrict__ row_block, const float *__restrict__ mean, + const float *__restrict__ jac_x, const float *__restrict__ jac_y, + const float *__restrict__ clip_limit, + double *__restrict__ sum, double *__restrict__ sum_sq, + double *__restrict__ sum_jx, double *__restrict__ sum_jy, + int32_t *__restrict__ count) { + const int c = blockIdx.x * blockDim.x + threadIdx.x; + if (c >= BackgroundBand::CELLS) + return; + const bool clipped = clip_limit != nullptr; + const float limit = clipped ? clip_limit[c] : 0.0f; + + double s = 0, ss = 0, jx = 0, jy = 0; + double bs = 0, bss = 0, bjx = 0, bjy = 0; + int32_t n = 0; + int block = -1; + for (int32_t p = offset[c]; p < offset[c + 1]; p++) { + const int32_t i = index[p]; + const float value = mean[i]; + if (clipped && (limit < 0.0f || value > limit)) + continue; + const int b = row_block[i / width]; + if (b != block) { + s += bs; ss += bss; jx += bjx; jy += bjy; + bs = bss = bjx = bjy = 0; + block = b; + } + n++; + bs += value; + bss += static_cast(value) * value; + if (!clipped) { + bjx += jac_x[i]; + bjy += jac_y[i]; + } + } + s += bs; ss += bss; jx += bjx; jy += bjy; + sum[c] = s; + sum_sq[c] = ss; + count[c] = n; + if (!clipped) { + sum_jx[c] = jx; + sum_jy[c] = jy; + } +} + +} // namespace + +struct BeamCenterBackgroundGPU::Impl { + int width, height; + size_t npixels; + CudaStream stream; + CudaDevicePtr usable; + CudaDevicePtr mean; + CudaDevicePtr row_block; + CudaDevicePtr key, sorted_key, index, sorted_index; + CudaDevicePtr jac_x, jac_y; + CudaDevicePtr offset; + CudaDevicePtr clip_limit; + CudaDevicePtr sum, sum_sq, sum_jx, sum_jy; + CudaDevicePtr count; + CudaDevicePtr sort_scratch; + size_t sort_scratch_bytes = 0; + + Impl(int w, int h) + : width(w), height(h), npixels(static_cast(w) * h), + usable(npixels), mean(npixels), row_block(h), + key(npixels), sorted_key(npixels), index(npixels), sorted_index(npixels), + jac_x(npixels), jac_y(npixels), offset(BackgroundBand::CELLS + 1), + clip_limit(BackgroundBand::CELLS), + sum(BackgroundBand::CELLS), sum_sq(BackgroundBand::CELLS), + sum_jx(BackgroundBand::CELLS), sum_jy(BackgroundBand::CELLS), + count(BackgroundBand::CELLS) { + // Keys run to CELLS inclusive, the bin of everything outside the band. + check(cub::DeviceRadixSort::SortPairs(nullptr, sort_scratch_bytes, key.get(), sorted_key.get(), + index.get(), sorted_index.get(), npixels, 0, end_bit(), stream), + "sort size"); + sort_scratch = CudaDevicePtr(sort_scratch_bytes); + } + + static int end_bit() { + int bits = 0; + while ((1 << bits) <= BackgroundBand::CELLS) bits++; + return bits; + } + + void Download(std::vector &s, std::vector &ss, std::vector *jx, + std::vector *jy, std::vector &n) { + const size_t cells = BackgroundBand::CELLS; + check(cudaMemcpyAsync(s.data(), sum.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy"); + check(cudaMemcpyAsync(ss.data(), sum_sq.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy"); + if (jx) check(cudaMemcpyAsync(jx->data(), sum_jx.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy"); + if (jy) check(cudaMemcpyAsync(jy->data(), sum_jy.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy"); + check(cudaMemcpyAsync(n.data(), count.get(), cells * sizeof(int32_t), cudaMemcpyDeviceToHost, stream), "copy"); + check(cudaStreamSynchronize(stream), "sums"); + } +}; + +BeamCenterBackgroundGPU::BeamCenterBackgroundGPU(int width, int height, const std::vector &block_row, + const char *usable, const float *mean) + : impl(std::make_unique(width, height)) { + std::vector row_block(height); + for (size_t b = 0; b + 1 < block_row.size(); b++) + for (int y = block_row[b]; y < block_row[b + 1]; y++) + row_block[y] = static_cast(b); + check(cudaMemcpyAsync(impl->row_block.get(), row_block.data(), height * sizeof(int32_t), + cudaMemcpyHostToDevice, impl->stream), "upload"); + check(cudaMemcpyAsync(impl->usable.get(), usable, impl->npixels, cudaMemcpyHostToDevice, impl->stream), "upload"); + check(cudaMemcpyAsync(impl->mean.get(), mean, impl->npixels * sizeof(float), cudaMemcpyHostToDevice, + impl->stream), "upload"); + check(cudaStreamSynchronize(impl->stream), "upload"); +} + +BeamCenterBackgroundGPU::~BeamCenterBackgroundGPU() = default; + +void BeamCenterBackgroundGPU::Bin(const BackgroundBand &band, float beam_x, float beam_y, + std::vector &sum, std::vector &sum_sq, + std::vector &sum_jx, std::vector &sum_jy, + std::vector &count) { + Impl &d = *impl; + const auto blocks = static_cast((d.npixels + THREADS - 1) / THREADS); + bin_kernel<<>>(band, d.width, d.npixels, beam_x, beam_y, d.usable, + d.key, d.index, d.jac_x, d.jac_y); + check(cudaGetLastError(), "bin"); + // A radix sort is stable, so each cell's pixels come out in pixel order. + check(cub::DeviceRadixSort::SortPairs(d.sort_scratch.get(), d.sort_scratch_bytes, d.key.get(), + d.sorted_key.get(), d.index.get(), d.sorted_index.get(), + d.npixels, 0, Impl::end_bit(), d.stream), "sort"); + constexpr int cell_blocks = (BackgroundBand::CELLS + 1 + THREADS - 1) / THREADS; + offset_kernel<<>>(d.sorted_key, d.npixels, d.offset); + check(cudaGetLastError(), "offsets"); + sum_kernel<<>>(d.width, d.offset, d.sorted_index, d.row_block, d.mean, + d.jac_x, d.jac_y, nullptr, d.sum, d.sum_sq, + d.sum_jx, d.sum_jy, d.count); + check(cudaGetLastError(), "sums"); + d.Download(sum, sum_sq, &sum_jx, &sum_jy, count); +} + +void BeamCenterBackgroundGPU::Clip(const std::vector &clip_limit, + std::vector &sum, std::vector &sum_sq, + std::vector &count) { + Impl &d = *impl; + check(cudaMemcpyAsync(d.clip_limit.get(), clip_limit.data(), BackgroundBand::CELLS * sizeof(float), + cudaMemcpyHostToDevice, d.stream), "upload"); + constexpr int cell_blocks = (BackgroundBand::CELLS + THREADS - 1) / THREADS; + sum_kernel<<>>(d.width, d.offset, d.sorted_index, d.row_block, d.mean, + d.jac_x, d.jac_y, d.clip_limit, d.sum, d.sum_sq, + d.sum_jx, d.sum_jy, d.count); + check(cudaGetLastError(), "clip"); + d.Download(sum, sum_sq, nullptr, nullptr, count); +} diff --git a/image_analysis/geom_refinement/BeamCenterBackgroundGPU.h b/image_analysis/geom_refinement/BeamCenterBackgroundGPU.h new file mode 100644 index 000000000..72d47b266 --- /dev/null +++ b/image_analysis/geom_refinement/BeamCenterBackgroundGPU.h @@ -0,0 +1,40 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#pragma once + +// Included only under JFJOCH_USE_CUDA. Free of CUDA headers, so the host fit can hold one. + +#include +#include +#include + +#include "BackgroundBand.h" + +// The two passes over the pixels of the background beam-centre fit (FindBeamCenterFromBackground), +// on the device: binning every usable pixel into its cell about a trial centre, and summing the +// binned pixels again under a clip. They are all of the fit's cost; the fit itself stays on the host. +// +// Each cell is summed in the order the host sums it - pixel order within each of the host's row +// blocks, the blocks then added in block order - so a pixel that lands in the same cell on both +// sides adds the same rounding on both. Whatever differs comes from BackgroundBandCell (see there). +class BeamCenterBackgroundGPU { + struct Impl; + std::unique_ptr impl; +public: + // block_row: the first row of each of the host's row blocks, and one past the last row at the end. + BeamCenterBackgroundGPU(int width, int height, const std::vector &block_row, + const char *usable, const float *mean); + ~BeamCenterBackgroundGPU(); + + // Bin the band about (beam_x, beam_y) and sum each cell: the values, their squares, the two + // derivatives, and the count. + void Bin(const BackgroundBand &band, float beam_x, float beam_y, + std::vector &sum, std::vector &sum_sq, + std::vector &sum_jx, std::vector &sum_jy, std::vector &count); + + // Sum the pixels the last Bin put in each cell again, leaving out those above the cell's + // clip_limit and every pixel of a cell whose limit is negative. + void Clip(const std::vector &clip_limit, + std::vector &sum, std::vector &sum_sq, std::vector &count); +}; diff --git a/image_analysis/geom_refinement/BeamCenterFromBackground.cpp b/image_analysis/geom_refinement/BeamCenterFromBackground.cpp index b2c306b9e..0b99a030c 100644 --- a/image_analysis/geom_refinement/BeamCenterFromBackground.cpp +++ b/image_analysis/geom_refinement/BeamCenterFromBackground.cpp @@ -2,13 +2,19 @@ // SPDX-License-Identifier: GPL-3.0-only #include "BeamCenterFromBackground.h" +#include "BackgroundBand.h" #include #include #include +#include "../../common/CompressedImage.h" #include "../../common/JFJochMath.h" #include "../../common/ParallelFor.h" +#ifdef JFJOCH_USE_CUDA +#include "../../common/CUDAWrapper.h" +#include "BeamCenterBackgroundGPU.h" +#endif namespace { @@ -17,8 +23,8 @@ namespace { constexpr float BAND_LOW_RES_A = 12.0f; constexpr float BAND_HIGH_RES_A = 2.2f; -constexpr int SECTORS = 36; -constexpr int RADIAL_BINS = 120; +constexpr int SECTORS = BackgroundBand::SECTORS; +constexpr int RADIAL_BINS = BackgroundBand::RADIAL_BINS; // A cell with fewer pixels than this has no usable mean. constexpr int MIN_PIXELS_PER_CELL = 20; @@ -83,7 +89,7 @@ float median_of(std::vector &v) { std::optional FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const PixelMask &mask, const std::vector &mean, size_t nthreads, - std::optional> start) { + std::optional> start, bool allow_device) { if (nthreads == 0) nthreads = std::max(1u, std::thread::hardware_concurrency()); const auto W = static_cast(experiment.GetXPixelsNumConv()); @@ -111,7 +117,6 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe float beam_y = start ? start->second : geom.GetBeamY_pxl(); constexpr int n_cells = RADIAL_BINS * SECTORS; - std::vector cell_of(n_pixels); std::vector sum(n_cells), sum_sq(n_cells), sum_jx(n_cells), sum_jy(n_cells); std::vector count(n_cells), count_all(n_cells); std::vector profile(RADIAL_BINS), d_profile(RADIAL_BINS), clip_limit(n_cells); @@ -127,16 +132,69 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe std::vector block_jy(static_cast(BLOCKS) * n_cells); std::vector block_count(static_cast(BLOCKS) * n_cells); + // Whether a pixel can take part at all, which does not depend on the centre. + std::vector> usable(n_pixels); + ParallelFor(BLOCKS, nthreads, [&](int b) { + for (size_t i = static_cast(block_row[b]) * W; i < static_cast(block_row[b + 1]) * W; i++) + usable[i] = pixel_mask[i] == 0 && std::isfinite(mean[i]); + }); + // The pixels of each block that fall in the band at the current centre, with their cell and value, + // in pixel order from the block's first pixel on. The clipping rounds read these instead of the + // whole detector, in the same order. + std::vector> band_cell(n_pixels); + std::vector> band_value(n_pixels); + std::vector band_pixels(BLOCKS); + + // The blocks' cells folded in block order. Each cell is folded on its own, so the cells are split + // over the threads and every cell is still summed in the same order. + const auto fold = [&](bool with_jacobian) { + ParallelChunks(n_cells, nthreads, [&](int c0, int c1) { + for (int c = c0; c < c1; c++) { + double s = 0, ss = 0, jx = 0, jy = 0; + int32_t n = 0; + for (int b = 0; b < BLOCKS; b++) { + const size_t k = static_cast(b) * n_cells + c; + s += block_sum[k]; ss += block_sum_sq[k]; + if (with_jacobian) { jx += block_jx[k]; jy += block_jy[k]; } + n += block_count[k]; + } + sum[c] = s; sum_sq[c] = ss; count[c] = n; + if (with_jacobian) { sum_jx[c] = jx; sum_jy[c] = jy; } + } + }); + }; + // Most pixels lie outside the band. Those clearly outside it in tan(2theta) = rho / lz - by a // margin far above float rounding - skip the square root and both atan2 below; every pixel the exact // test would keep still reaches it. const double tan_lo = std::tan(static_cast(tt_lo)) * (1.0 - 1e-3); const double tan_hi = tt_hi < PI / 2 ? std::tan(static_cast(tt_hi)) * (1.0 + 1e-3) : INFINITY; + BackgroundBand band{}; + for (int k = 0; k < 9; k++) + band.rot[k] = rot[k]; + band.pixel_size = pixel_size; + band.distance = distance; + band.tt_lo = tt_lo; + band.tt_hi = tt_hi; + band.d_tt = d_tt; + band.tan_lo = tan_lo; + band.tan_hi = tan_hi; + +#ifndef JFJOCH_USE_CUDA + (void) allow_device; +#else + // With a GPU the two passes over the pixels run there, and only the cells come back. + std::unique_ptr gpu; + if (allow_device && get_gpu_count() > 0) + gpu = std::make_unique(W, H, block_row, usable.data(), mean.data()); +#endif + float step_x = 0.0f, step_y = 0.0f, sigma_x = 0.0f, sigma_y = 0.0f; float previous_x = 0.0f, previous_y = 0.0f; int reversals = 0; - for (int iteration = 0; iteration < MAX_ITERATIONS; iteration++) { + // The binning pass about the current centre, then the clip rounds over the pixels it binned. + const auto bin_cpu = [&] { ParallelFor(BLOCKS, nthreads, [&](int b) { double *b_sum = block_sum.data() + static_cast(b) * n_cells; double *b_sum_sq = block_sum_sq.data() + static_cast(b) * n_cells; @@ -148,61 +206,62 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe std::fill(b_jx, b_jx + n_cells, 0.0); std::fill(b_jy, b_jy + n_cells, 0.0); std::fill(b_count, b_count + n_cells, 0); + int32_t *cells = band_cell.data() + static_cast(block_row[b]) * W; + float *values = band_value.data() + static_cast(block_row[b]) * W; + size_t n_band = 0; for (int y = block_row[b]; y < block_row[b + 1]; y++) { for (int x = 0; x < W; x++) { const size_t i = static_cast(y) * W + x; - cell_of[i] = -1; - if (pixel_mask[i] != 0 || !std::isfinite(mean[i])) + if (!usable[i]) continue; - const float u = (x - beam_x) * pixel_size; - const float v = (y - beam_y) * pixel_size; - const float lx = rot[0] * u + rot[1] * v + rot[2] * distance; - const float ly = rot[3] * u + rot[4] * v + rot[5] * distance; - const float lz = rot[6] * u + rot[7] * v + rot[8] * distance; - const float rho_sq = lx * lx + ly * ly; - if (lz > 0.0f) { - const double lz_sq = static_cast(lz) * lz; - if (rho_sq < tan_lo * tan_lo * lz_sq || rho_sq > tan_hi * tan_hi * lz_sq) - continue; - } - const float rho = std::sqrt(rho_sq); - const float two_theta = std::atan2(rho, lz); - if (two_theta < tt_lo || two_theta >= tt_hi || rho == 0.0f) + float jac_x, jac_y; + const int cell = BackgroundBandCell(band, x, y, beam_x, beam_y, jac_x, jac_y); + if (cell < 0) continue; - - const float phi = std::atan2(ly, lx); - // Both bins are clamped: a pixel one float ulp below the top of the band divides - // to exactly RADIAL_BINS, which is one cell past the end of every accumulator. - const int r_bin = std::clamp(static_cast((two_theta - tt_lo) / d_tt), 0, RADIAL_BINS - 1); - const int s_bin = std::clamp(static_cast((phi + PI) / (2 * PI) * SECTORS), 0, SECTORS - 1); - const int cell = r_bin * SECTORS + s_bin; - - // d(2theta)/d(beam), through the lab coordinate: the detector coordinate depends - // on the centre only as (x - beam_x), so moving the centre is moving the pixel. - const float denominator = rho * rho + lz * lz; - const float g_x = lz * lx / (rho * denominator); - const float g_y = lz * ly / (rho * denominator); - const float g_z = -rho / denominator; - cell_of[i] = cell; + cells[n_band] = cell; + values[n_band] = mean[i]; + n_band++; b_count[cell]++; b_sum[cell] += mean[i]; b_sum_sq[cell] += static_cast(mean[i]) * mean[i]; - b_jx[cell] += -pixel_size * (g_x * rot[0] + g_y * rot[3] + g_z * rot[6]); - b_jy[cell] += -pixel_size * (g_x * rot[1] + g_y * rot[4] + g_z * rot[7]); + b_jx[cell] += jac_x; + b_jy[cell] += jac_y; } } + band_pixels[b] = n_band; }); - for (int c = 0; c < n_cells; c++) { - double s = 0, ss = 0, jx = 0, jy = 0; - int32_t n = 0; - for (int b = 0; b < BLOCKS; b++) { - const size_t k = static_cast(b) * n_cells + c; - s += block_sum[k]; ss += block_sum_sq[k]; - jx += block_jx[k]; jy += block_jy[k]; - n += block_count[k]; + fold(true); + }; + const auto clip_cpu = [&] { + ParallelFor(BLOCKS, nthreads, [&](int b) { + double *b_sum = block_sum.data() + static_cast(b) * n_cells; + double *b_sum_sq = block_sum_sq.data() + static_cast(b) * n_cells; + int32_t *b_count = block_count.data() + static_cast(b) * n_cells; + std::fill(b_sum, b_sum + n_cells, 0.0); + std::fill(b_sum_sq, b_sum_sq + n_cells, 0.0); + std::fill(b_count, b_count + n_cells, 0); + const int32_t *cells = band_cell.data() + static_cast(block_row[b]) * W; + const float *values = band_value.data() + static_cast(block_row[b]) * W; + for (size_t j = 0; j < band_pixels[b]; j++) { + const int32_t c = cells[j]; + const float value = values[j]; + if (clip_limit[c] < 0.0f || value > clip_limit[c]) + continue; + b_count[c]++; + b_sum[c] += value; + b_sum_sq[c] += static_cast(value) * value; } - sum[c] = s; sum_sq[c] = ss; sum_jx[c] = jx; sum_jy[c] = jy; count[c] = n; - } + }); + fold(false); + }; + + for (int iteration = 0; iteration < MAX_ITERATIONS; iteration++) { +#ifdef JFJOCH_USE_CUDA + if (gpu) + gpu->Bin(band, beam_x, beam_y, sum, sum_sq, sum_jx, sum_jy, count); + else +#endif + bin_cpu(); count_all = count; // the Jacobian sums belong to the unclipped pixel set @@ -213,33 +272,12 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe const double variance = std::max(sum_sq[c] / count[c] - m * m, 0.0); clip_limit[c] = static_cast(m + CLIP_SIGMA * std::sqrt(variance)); } - ParallelFor(BLOCKS, nthreads, [&](int b) { - double *b_sum = block_sum.data() + static_cast(b) * n_cells; - double *b_sum_sq = block_sum_sq.data() + static_cast(b) * n_cells; - int32_t *b_count = block_count.data() + static_cast(b) * n_cells; - std::fill(b_sum, b_sum + n_cells, 0.0); - std::fill(b_sum_sq, b_sum_sq + n_cells, 0.0); - std::fill(b_count, b_count + n_cells, 0); - const size_t lo = static_cast(block_row[b]) * W; - const size_t hi = static_cast(block_row[b + 1]) * W; - for (size_t i = lo; i < hi; i++) { - const int32_t c = cell_of[i]; - if (c < 0 || clip_limit[c] < 0.0f || mean[i] > clip_limit[c]) - continue; - b_count[c]++; - b_sum[c] += mean[i]; - b_sum_sq[c] += static_cast(mean[i]) * mean[i]; - } - }); - for (int c = 0; c < n_cells; c++) { - double s = 0, ss = 0; - int32_t n = 0; - for (int b = 0; b < BLOCKS; b++) { - const size_t k = static_cast(b) * n_cells + c; - s += block_sum[k]; ss += block_sum_sq[k]; n += block_count[k]; - } - sum[c] = s; sum_sq[c] = ss; count[c] = n; - } +#ifdef JFJOCH_USE_CUDA + if (gpu) + gpu->Clip(clip_limit, sum, sum_sq, count); + else +#endif + clip_cpu(); } // Radial profile: the median over the sectors that have a mean, on rings that are diff --git a/image_analysis/geom_refinement/BeamCenterFromBackground.h b/image_analysis/geom_refinement/BeamCenterFromBackground.h index 350cb6ddd..7f1df9efb 100644 --- a/image_analysis/geom_refinement/BeamCenterFromBackground.h +++ b/image_analysis/geom_refinement/BeamCenterFromBackground.h @@ -40,10 +40,14 @@ struct BeamCenterEstimate { // `start` is where the walk begins; the centre in the file when it is not given. The walk advances // by a bounded distance per iteration, so where it starts decides how much of its budget is spent // travelling and - on a surface with more than one basin - which fixed point it can reach at all. +// +// With a GPU the passes over the pixels run on it (BeamCenterBackgroundGPU); allow_device = false +// keeps them on the host, which is what the parity test compares against. std::optional FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const PixelMask &mask, const std::vector &mean, size_t nthreads = 0, - std::optional> start = {}); + std::optional> start = {}, + bool allow_device = true); // The precision of a centre that is the FFT capture alone, with no walk behind it. The capture is // a half-pixel grid position read off a surface, measured over 75 rotation datasets at a median diff --git a/image_analysis/geom_refinement/CMakeLists.txt b/image_analysis/geom_refinement/CMakeLists.txt index b004e8d4e..50971f3ad 100644 --- a/image_analysis/geom_refinement/CMakeLists.txt +++ b/image_analysis/geom_refinement/CMakeLists.txt @@ -24,6 +24,12 @@ ADD_LIBRARY(JFJochGeomRefinement STATIC XtalOptimizer.cpp XtalOptimizer.h XtalResidual.h + XtalRefine.cpp + XtalRefine.h + LMSolver.cpp + LMSolver.h + Dual.h + BackgroundBand.h PostRefine.cpp PostRefine.h GeometryRefiner.cpp @@ -37,7 +43,8 @@ ADD_LIBRARY(JFJochGeomRefinement STATIC TARGET_LINK_LIBRARIES(JFJochGeomRefinement Ceres::ceres Eigen3::Eigen JFJochCommon fftw3f) IF (JFJOCH_CUDA_AVAILABLE) - TARGET_SOURCES(JFJochGeomRefinement PRIVATE BeamCenterFFTGPU.cu BeamCenterFFTGPU.h) + TARGET_SOURCES(JFJochGeomRefinement PRIVATE BeamCenterFFTGPU.cu BeamCenterFFTGPU.h + BeamCenterBackgroundGPU.cu BeamCenterBackgroundGPU.h) # Same static/dynamic cuFFT choice as the FFT indexer, and for the same reasons - see the long # note in image_analysis/indexing/CMakeLists.txt. IF (JFJOCH_PORTABLE_ONLY AND TARGET CUDA::cufft_static) @@ -46,3 +53,13 @@ IF (JFJOCH_CUDA_AVAILABLE) TARGET_LINK_LIBRARIES(JFJochGeomRefinement CUDA::cufft) ENDIF() ENDIF() + +# The background beam-centre fit bins every pixel with BackgroundBand.h on the host and on the device +# and must get the same bits on both (see there). That needs no multiply-add contracted on either side: +# GCC and Clang contract by default and nvcc does too, while MSVC does not without /fp:contract. +IF (JFJOCH_CUDA_AVAILABLE) + SET_SOURCE_FILES_PROPERTIES(BeamCenterBackgroundGPU.cu PROPERTIES COMPILE_OPTIONS "--fmad=false") +ENDIF() +IF (CMAKE_CXX_COMPILER_ID MATCHES "GNU|Clang") + SET_SOURCE_FILES_PROPERTIES(BeamCenterFromBackground.cpp PROPERTIES COMPILE_OPTIONS "-ffp-contract=off") +ENDIF() diff --git a/image_analysis/geom_refinement/Dual.h b/image_analysis/geom_refinement/Dual.h new file mode 100644 index 000000000..932828058 --- /dev/null +++ b/image_analysis/geom_refinement/Dual.h @@ -0,0 +1,142 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#pragma once + +// A forward-mode dual number with N derivative lanes: a value and its gradient with respect to N +// parameters. The residuals of the crystal refinement are written as templates over their scalar type, +// so the same code runs on a plain double and on this. The value part of every operation is the plain +// double arithmetic of the same expression - written the way ceres::Jet writes it, division through the +// reciprocal - so a residual evaluated on a Dual has the same value as on a Jet. + +#include +#include + +#include + +template +struct Dual { + double a = 0.0; + double v[N] = {}; + + Dual() = default; + Dual(double value) : a(value) {} // NOLINT: implicit, a constant is a dual with zero derivatives + + static Dual Variable(double value, int lane) { + Dual d(value); + d.v[lane] = 1.0; + return d; + } + + Dual &operator+=(const Dual &o) { a += o.a; for (int i = 0; i < N; i++) v[i] += o.v[i]; return *this; } + Dual &operator-=(const Dual &o) { a -= o.a; for (int i = 0; i < N; i++) v[i] -= o.v[i]; return *this; } + Dual &operator*=(const Dual &o) { *this = *this * o; return *this; } + Dual &operator/=(const Dual &o) { *this = *this / o; return *this; } + + friend Dual operator+(const Dual &x) { return x; } + friend Dual operator-(const Dual &x) { + Dual r(-x.a); + for (int i = 0; i < N; i++) r.v[i] = -x.v[i]; + return r; + } + + friend Dual operator+(const Dual &x, const Dual &y) { + Dual r(x.a + y.a); + for (int i = 0; i < N; i++) r.v[i] = x.v[i] + y.v[i]; + return r; + } + friend Dual operator+(const Dual &x, double s) { Dual r = x; r.a += s; return r; } + friend Dual operator+(double s, const Dual &x) { Dual r = x; r.a += s; return r; } + + friend Dual operator-(const Dual &x, const Dual &y) { + Dual r(x.a - y.a); + for (int i = 0; i < N; i++) r.v[i] = x.v[i] - y.v[i]; + return r; + } + friend Dual operator-(const Dual &x, double s) { Dual r = x; r.a -= s; return r; } + friend Dual operator-(double s, const Dual &x) { + Dual r(s - x.a); + for (int i = 0; i < N; i++) r.v[i] = -x.v[i]; + return r; + } + + friend Dual operator*(const Dual &x, const Dual &y) { + Dual r(x.a * y.a); + for (int i = 0; i < N; i++) r.v[i] = x.a * y.v[i] + x.v[i] * y.a; + return r; + } + friend Dual operator*(const Dual &x, double s) { + Dual r(x.a * s); + for (int i = 0; i < N; i++) r.v[i] = x.v[i] * s; + return r; + } + friend Dual operator*(double s, const Dual &x) { return x * s; } + + friend Dual operator/(const Dual &x, const Dual &y) { + const double y_inv = 1.0 / y.a; + const double q = x.a * y_inv; + Dual r(q); + for (int i = 0; i < N; i++) r.v[i] = (x.v[i] - q * y.v[i]) * y_inv; + return r; + } + friend Dual operator/(const Dual &x, double s) { + const double s_inv = 1.0 / s; + return x * s_inv; + } + friend Dual operator/(double s, const Dual &y) { + const double y_inv = 1.0 / y.a; + const double d = -s * y_inv * y_inv; + Dual r(s * y_inv); + for (int i = 0; i < N; i++) r.v[i] = d * y.v[i]; + return r; + } + + friend bool operator<(const Dual &x, const Dual &y) { return x.a < y.a; } + friend bool operator>(const Dual &x, const Dual &y) { return x.a > y.a; } + friend bool operator<=(const Dual &x, const Dual &y) { return x.a <= y.a; } + friend bool operator>=(const Dual &x, const Dual &y) { return x.a >= y.a; } + friend bool operator==(const Dual &x, const Dual &y) { return x.a == y.a; } + friend bool operator!=(const Dual &x, const Dual &y) { return x.a != y.a; } + + // The chain rule for a function of one argument: value f, derivative df. + Dual Chain(double f, double df) const { + Dual r(f); + for (int i = 0; i < N; i++) r.v[i] = df * v[i]; + return r; + } + + friend Dual sqrt(const Dual &x) { + const double s = std::sqrt(x.a); + return x.Chain(s, 0.5 / s); + } + friend Dual cos(const Dual &x) { return x.Chain(std::cos(x.a), -std::sin(x.a)); } + friend Dual sin(const Dual &x) { return x.Chain(std::sin(x.a), std::cos(x.a)); } + friend Dual hypot(const Dual &x, const Dual &y, const Dual &z) { + // As ceres::hypot(Jet, Jet, Jet): the value is std::hypot, the derivative x/h dx + y/h dy + z/h dz. + const double h = std::hypot(x.a, y.a, z.a); + Dual r(h); + for (int i = 0; i < N; i++) r.v[i] = x.a / h * x.v[i] + y.a / h * y.v[i] + z.a / h * z.v[i]; + return r; + } + friend int fpclassify(const Dual &x) { return std::fpclassify(x.a); } +}; + +// What Eigen needs to hold a Dual in a fixed-size matrix (the reciprocal basis is built in one). +namespace Eigen { + template + struct NumTraits> : GenericNumTraits { + typedef Dual Real; + typedef Dual NonInteger; + typedef Dual Nested; + typedef Dual Literal; + enum { + IsComplex = 0, IsInteger = 0, IsSigned = 1, RequireInitialization = 1, + ReadCost = 1, AddCost = 1, MulCost = 1 + }; + static inline Real epsilon() { return Real(std::numeric_limits::epsilon()); } + static inline Real dummy_precision() { return Real(1e-12); } + static inline Real highest() { return Real(std::numeric_limits::max()); } + static inline Real lowest() { return Real(-std::numeric_limits::max()); } + static inline int digits10() { return NumTraits::digits10(); } + }; +} diff --git a/image_analysis/geom_refinement/LMSolver.cpp b/image_analysis/geom_refinement/LMSolver.cpp new file mode 100644 index 000000000..967ae965e --- /dev/null +++ b/image_analysis/geom_refinement/LMSolver.cpp @@ -0,0 +1,241 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +// Adapted from https://github.com/ceres-solver/ceres-solver (internal/ceres/polynomial.cc, +// include/ceres/internal/sphere_manifold_functions.h, householder_vector.h) +// Copyright 2023 Google Inc. All rights reserved. +// BSD-3-Clause, see licenses/ceres-solver.txt + +#include "LMSolver.h" + +#include + +namespace { + void HouseholderVector3(const double x[3], double v[3], double &beta) { + const double sigma = x[0] * x[0] + x[1] * x[1]; + v[0] = x[0]; + v[1] = x[1]; + v[2] = 1.0; + beta = 0.0; + const double x_pivot = x[2]; + if (sigma <= std::numeric_limits::epsilon()) { + if (x_pivot < 0.0) + beta = 2.0; + return; + } + const double mu = std::sqrt(x_pivot * x_pivot + sigma); + const double v_pivot = (x_pivot <= 0.0) ? x_pivot - mu : -sigma / (x_pivot + mu); + beta = 2.0 * v_pivot * v_pivot / (sigma + v_pivot * v_pivot); + v[0] /= v_pivot; + v[1] /= v_pivot; + } + + double Norm3(const double x[3]) { + return std::sqrt(x[0] * x[0] + x[1] * x[1] + x[2] * x[2]); + } + + using Vector = Eigen::VectorXd; + using Matrix = Eigen::MatrixXd; + + double EvaluatePolynomial(const Vector &polynomial, double x) { + double v = 0.0; + for (int i = 0; i < polynomial.size(); ++i) + v = v * x + polynomial(i); + return v; + } + + void BalanceCompanionMatrix(Matrix &companion_matrix) { + Matrix offdiagonal = companion_matrix; + offdiagonal.diagonal().setZero(); + const int degree = static_cast(companion_matrix.rows()); + const double gamma = 0.9; + bool scaling_has_changed; + do { + scaling_has_changed = false; + for (int i = 0; i < degree; ++i) { + const double col_norm = offdiagonal.col(i).lpNorm<1>(); + if (std::fpclassify(col_norm) != FP_ZERO) { + const double row_norm = offdiagonal.row(i).lpNorm<1>(); + int exponent = 0; + std::frexp(row_norm / col_norm, &exponent); + exponent /= 2; + if (exponent != 0) { + const double scaled_col_norm = std::ldexp(col_norm, exponent); + const double scaled_row_norm = std::ldexp(row_norm, -exponent); + if (scaled_col_norm + scaled_row_norm < gamma * (col_norm + row_norm)) { + scaling_has_changed = true; + offdiagonal.row(i) *= std::ldexp(1.0, -exponent); + offdiagonal.col(i) *= std::ldexp(1.0, exponent); + } + } + } + } + } while (scaling_has_changed); + offdiagonal.diagonal() = companion_matrix.diagonal(); + companion_matrix = offdiagonal; + } + + // Real parts of the roots, as Ceres' FindPolynomialRoots (the imaginary parts are not used here). + bool FindPolynomialRoots(const Vector &polynomial_in, Vector &real) { + if (polynomial_in.size() == 0) + return false; + int lead = 0; + while (lead < polynomial_in.size() - 1 && polynomial_in(lead) == 0.0) + ++lead; + Vector polynomial = polynomial_in.tail(polynomial_in.size() - lead); + const int degree = static_cast(polynomial.size()) - 1; + if (degree == 0) { + real.resize(0); + return true; + } + if (degree == 1) { + real.resize(1); + real(0) = -polynomial(1) / polynomial(0); + return true; + } + if (degree == 2) { + const double a = polynomial(0); + const double b = polynomial(1); + const double c = polynomial(2); + const double D = b * b - 4 * a * c; + const double sqrt_D = std::sqrt(std::fabs(D)); + real.setZero(2); + if (D >= 0) { + if (b >= 0) { + real(0) = (-b - sqrt_D) / (2.0 * a); + real(1) = (2.0 * c) / (-b - sqrt_D); + } else { + real(0) = (2.0 * c) / (-b + sqrt_D); + real(1) = (-b + sqrt_D) / (2.0 * a); + } + } else { + real(0) = -b / (2.0 * a); + real(1) = -b / (2.0 * a); + } + return true; + } + polynomial /= polynomial(0); + Matrix companion = Matrix::Zero(degree, degree); + companion.diagonal(-1).setOnes(); + companion.col(degree - 1) = -polynomial.reverse().head(degree); + BalanceCompanionMatrix(companion); + Eigen::EigenSolver solver(companion, false); + if (solver.info() != Eigen::Success) + return false; + real = solver.eigenvalues().real(); + return true; + } + + void MinimizePolynomial(const Vector &polynomial, double x_min, double x_max, + double &optimal_x, double &optimal_value) { + optimal_x = (x_min + x_max) / 2.0; + optimal_value = EvaluatePolynomial(polynomial, optimal_x); + const double x_min_value = EvaluatePolynomial(polynomial, x_min); + if (x_min_value < optimal_value) { + optimal_value = x_min_value; + optimal_x = x_min; + } + const double x_max_value = EvaluatePolynomial(polynomial, x_max); + if (x_max_value < optimal_value) { + optimal_value = x_max_value; + optimal_x = x_max; + } + if (polynomial.rows() <= 2) + return; + + const int degree = static_cast(polynomial.rows()) - 1; + Vector derivative(degree); + for (int i = 0; i < degree; ++i) + derivative(i) = (degree - i) * polynomial(i); + Vector roots_real; + if (!FindPolynomialRoots(derivative, roots_real)) + return; + for (int i = 0; i < roots_real.rows(); ++i) { + const double root = roots_real(i); + if (root < x_min || root > x_max) + continue; + const double value = EvaluatePolynomial(polynomial, root); + if (value < optimal_value) { + optimal_value = value; + optimal_x = root; + } + } + } + + Vector FindInterpolatingPolynomial(const std::vector &samples) { + int num_constraints = 0; + for (const auto &s: samples) + num_constraints += (s.value_is_valid ? 1 : 0) + (s.gradient_is_valid ? 1 : 0); + const int degree = num_constraints - 1; + Matrix lhs = Matrix::Zero(num_constraints, num_constraints); + Vector rhs = Vector::Zero(num_constraints); + int row = 0; + for (const auto &s: samples) { + if (s.value_is_valid) { + for (int j = 0; j <= degree; ++j) + lhs(row, j) = std::pow(s.x, degree - j); + rhs(row) = s.value; + ++row; + } + if (s.gradient_is_valid) { + for (int j = 0; j < degree; ++j) + lhs(row, j) = (degree - j) * std::pow(s.x, degree - j - 1); + rhs(row) = s.gradient; + ++row; + } + } + Eigen::FullPivLU lu(lhs); + return lu.setThreshold(0.0).solve(rhs); + } +} + +void SpherePlus3(const double x[3], const double delta[2], double out[3]) { + const double norm_delta = std::sqrt(delta[0] * delta[0] + delta[1] * delta[1]); + if (norm_delta == 0.0) { + out[0] = x[0]; + out[1] = x[1]; + out[2] = x[2]; + return; + } + double v[3], beta; + HouseholderVector3(x, v, beta); + const double sin_delta_by_delta = std::sin(norm_delta) / norm_delta; + const double y[3] = {sin_delta_by_delta * delta[0], sin_delta_by_delta * delta[1], std::cos(norm_delta)}; + const double vy = v[0] * y[0] + v[1] * y[1] + v[2] * y[2]; + const double x_norm = Norm3(x); + for (int i = 0; i < 3; i++) + out[i] = x_norm * (y[i] - v[i] * (beta * vy)); +} + +void SpherePlusJacobian3(const double x[3], double jacobian[3][2]) { + double v[3], beta; + HouseholderVector3(x, v, beta); + const double x_norm = Norm3(x); + for (int i = 0; i < 2; ++i) + for (int r = 0; r < 3; ++r) + jacobian[r][i] = (-beta * v[i] * v[r] + (r == i ? 1.0 : 0.0)) * x_norm; +} + +double LMInterpolatedStepSize(const LMLineSample &lowerbound, const LMLineSample &previous, + const LMLineSample ¤t, double min_step, double max_step) { + if (!current.value_is_valid) + return std::min(std::max(current.x * 0.5, min_step), max_step); + + std::vector samples{lowerbound, current}; + if (previous.value_is_valid) + samples.push_back(previous); + + const Vector polynomial = FindInterpolatingPolynomial(samples); + double step = 0.0, value = 0.0; + MinimizePolynomial(polynomial, min_step, max_step, step, value); + for (const auto &s: samples) { + if (s.x < min_step || s.x > max_step) + continue; + const double v = EvaluatePolynomial(polynomial, s.x); + if (v < value) { + step = s.x; + value = v; + } + } + return step; +} diff --git a/image_analysis/geom_refinement/LMSolver.h b/image_analysis/geom_refinement/LMSolver.h new file mode 100644 index 000000000..ee3b9657d --- /dev/null +++ b/image_analysis/geom_refinement/LMSolver.h @@ -0,0 +1,365 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +// The minimiser below follows Ceres Solver's trust-region Levenberg-Marquardt step for step - its +// options, its Jacobi scaling, its damping and radius updates, its stopping rules, its box projection, +// its projected Armijo line search on bounded problems and its SphereManifold - so that a problem moved +// off Ceres takes the same path to the same answer. Adapted from +// https://github.com/ceres-solver/ceres-solver (internal/ceres/trust_region_minimizer.cc, +// levenberg_marquardt_strategy.cc, trust_region_step_evaluator.cc, line_search.cc, polynomial.cc, +// include/ceres/internal/sphere_manifold_functions.h, householder_vector.h) +// Copyright 2023 Google Inc. All rights reserved. +// BSD-3-Clause, see licenses/ceres-solver.txt +// +// What it does NOT take from Ceres is the Jacobian: the caller hands over J^T J and J^T r directly, +// accumulated however it likes, and the minimiser never sees a row of J. Everything Ceres computes from +// the Jacobian - the column norms, the normal equations, the model cost change - is a function of those +// two alone. Only the options the crystal refinements use are reproduced: monotonic steps, no inner +// iterations, a dense Cholesky of the normal equations. + +#pragma once + +#include +#include +#include +#include + +#include + +struct LMBlock { + int offset = 0; // into the ambient parameter vector + int size = 0; // ambient size, at most 3 + bool constant = false; + bool sphere = false; // Ceres' SphereManifold: the norm is kept, the tangent has size - 1 coordinates + double lower[3] = {-std::numeric_limits::max(), -std::numeric_limits::max(), + -std::numeric_limits::max()}; + double upper[3] = {std::numeric_limits::max(), std::numeric_limits::max(), + std::numeric_limits::max()}; + + int TangentSize() const { return constant ? 0 : (sphere ? size - 1 : size); } +}; + +struct LMOptions { + int max_iterations = 50; + double max_time_s = 1e9; +}; + +enum class LMTermination { Convergence, NoConvergence, Failure }; + +struct LMSummary { + LMTermination termination = LMTermination::Failure; + // Iterations as Ceres counts them in Summary::iterations: iteration 0 included. + int iterations = 0; + int evaluations = 0; + int line_search_steps = 0; + double initial_cost = 0.0; + double final_cost = 0.0; + + bool IsSolutionUsable() const { return termination != LMTermination::Failure; } +}; + +// Ceres' SphereManifold<3> for a three-vector: the Householder reflection that takes x to the pole, the +// Plus that walks a tangent step along the sphere, and the 3x2 Jacobian of that Plus at zero step. +void SpherePlus3(const double x[3], const double delta[2], double out[3]); +void SpherePlusJacobian3(const double x[3], double jacobian[3][2]); + +// The polynomial step-size choice of Ceres' Armijo line search, cubic interpolation. +struct LMLineSample { + double x = 0.0; + double value = 0.0; + double gradient = 0.0; + bool value_is_valid = false; + bool gradient_is_valid = false; +}; +double LMInterpolatedStepSize(const LMLineSample &lowerbound, const LMLineSample &previous, + const LMLineSample ¤t, double min_step, double max_step); + +// Evaluate is called as eval(x, cost, g, H): x the ambient parameters, cost 1/2 sum of squared +// residuals, and - where g and H are not null - the gradient J^T r and J^T J in the TANGENT coordinates +// of the non-constant blocks, in block order. It returns false where anything came out non-finite. +// x is updated in place on success; it is left untouched on failure. +template +LMSummary SolveLM(std::vector &x_io, const std::vector &blocks, const LMOptions &options, + Evaluate &&eval) { + using Vec = Eigen::VectorXd; + using Mat = Eigen::MatrixXd; + constexpr double kMax = std::numeric_limits::max(); + + // Ceres' defaults, which is what the callers always ran with. + constexpr double initial_radius = 1e4; + constexpr double max_radius = 1e16; + constexpr double min_radius = 1e-32; + constexpr double min_relative_decrease = 1e-3; + constexpr double min_lm_diagonal = 1e-6; + constexpr double max_lm_diagonal = 1e32; + constexpr int max_consecutive_invalid_steps = 5; + constexpr double function_tolerance = 1e-6; + constexpr double gradient_tolerance = 1e-10; + constexpr double parameter_tolerance = 1e-8; + constexpr double sufficient_decrease = 1e-4; + constexpr double max_step_contraction = 1e-3; + constexpr double min_step_contraction = 0.6; + constexpr double min_line_search_step = 1e-9; + constexpr int max_line_search_iterations = 20; + + const auto start = std::chrono::steady_clock::now(); + LMSummary summary; + + int n = 0; + bool constrained = false; + for (const auto &b: blocks) { + for (int j = 0; j < b.size; j++) + if (!std::isfinite(x_io[b.offset + j])) + return summary; + n += b.TangentSize(); + for (int j = 0; j < b.size; j++) { + if (b.constant) { + if (x_io[b.offset + j] < b.lower[j] || x_io[b.offset + j] > b.upper[j]) + return summary; + } else { + if (b.lower[j] >= b.upper[j]) + return summary; + if (b.lower[j] > -kMax || b.upper[j] < kMax) + constrained = true; + } + } + } + + // x (+) delta, block by block, projected onto the bounds - Ceres' ParameterBlock::Plus. + const auto plus = [&](const Vec &x, const Vec &delta, Vec &out) { + out = x; + int t = 0; + for (const auto &b: blocks) { + if (b.constant) + continue; + if (b.sphere) { + SpherePlus3(x.data() + b.offset, delta.data() + t, out.data() + b.offset); + } else { + for (int j = 0; j < b.size; j++) + out[b.offset + j] = x[b.offset + j] + delta[t + j]; + } + for (int j = 0; j < b.size; j++) { + out[b.offset + j] = std::max(out[b.offset + j], b.lower[j]); + out[b.offset + j] = std::min(out[b.offset + j], b.upper[j]); + } + t += b.TangentSize(); + } + }; + // Norms over the parameters Ceres keeps in its state: the non-constant blocks only. + const auto free_norm = [&](const Vec &v) { + double s = 0.0; + for (const auto &b: blocks) + if (!b.constant) + for (int j = 0; j < b.size; j++) + s += v[b.offset + j] * v[b.offset + j]; + return std::sqrt(s); + }; + const auto free_max_norm = [&](const Vec &v) { + double m = 0.0; + for (const auto &b: blocks) + if (!b.constant) + for (int j = 0; j < b.size; j++) + m = std::max(m, std::fabs(v[b.offset + j])); + return m; + }; + + Vec x = Eigen::Map(x_io.data(), static_cast(x_io.size())); + if (constrained) { + Vec projected; + plus(x, Vec::Zero(n), projected); + x = projected; + } + + // One evaluation point with everything Ceres computes there. + struct Point { + Vec x; + double cost = kMax; + Vec g; + Mat H; + bool valid = false; + }; + const auto evaluate = [&](const Vec &at, Point &p) { + p.x = at; + p.g.setZero(n); + p.H.setZero(n, n); + summary.evaluations++; + p.valid = eval(at.data(), p.cost, &p.g, &p.H) && std::isfinite(p.cost); + if (!p.valid) + p.cost = kMax; + }; + + Point cur; + evaluate(x, cur); + if (!cur.valid) + return summary; + summary.initial_cost = cur.cost; + + // Jacobi scaling, fixed from the Jacobian at the starting point. + Vec scale(n); + for (int i = 0; i < n; i++) + scale[i] = 1.0 / (1.0 + std::sqrt(cur.H(i, i))); + + Vec gs, neg_g, projected; + Mat Hs; + double gradient_max_norm = 0.0; + const auto take_point = [&]() { + gs = scale.cwiseProduct(cur.g); + Hs = scale.asDiagonal() * cur.H * scale.asDiagonal(); + neg_g = -cur.g; + plus(cur.x, neg_g, projected); + gradient_max_norm = free_max_norm(cur.x - projected); + }; + take_point(); + + double radius = initial_radius; + double decrease_factor = 2.0; + bool reuse_diagonal = false; + Vec diagonal(n); + int consecutive_invalid = 0; + bool any_successful_step = false; + bool step_successful = true; // iteration 0 + int iteration = 0; + + Point trial; // the last point the line search evaluated, reused as the candidate when it is one + + const auto step_rejected = [&]() { + radius = radius / decrease_factor; + decrease_factor *= 2.0; + reuse_diagonal = true; + }; + + const auto finish = [&](LMTermination t) { + summary.termination = t; + summary.final_cost = cur.cost; + if (t != LMTermination::Failure) + for (int i = 0; i < x.size(); i++) + x_io[i] = cur.x[i]; + return summary; + }; + + for (;;) { + // FinalizeIterationAndCheckIfMinimizerCanContinue + summary.iterations++; + if (std::chrono::duration(std::chrono::steady_clock::now() - start).count() + >= options.max_time_s) + return finish(LMTermination::NoConvergence); + if (iteration >= options.max_iterations) + return finish(LMTermination::NoConvergence); + if (step_successful && gradient_max_norm <= gradient_tolerance) + return finish(LMTermination::Convergence); + if (radius <= min_radius) + return finish(LMTermination::Convergence); + + iteration++; + step_successful = false; + + // ComputeTrustRegionStep: the damped normal equations of the scaled Jacobian. + if (!reuse_diagonal) + for (int i = 0; i < n; i++) + diagonal[i] = std::min(std::max(Hs(i, i), min_lm_diagonal), max_lm_diagonal); + Mat lhs = Hs; + for (int i = 0; i < n; i++) { + const double d = std::sqrt(diagonal[i] / radius); + lhs(i, i) += d * d; + } + reuse_diagonal = true; + Eigen::LLT llt(lhs); + bool step_valid = false; + Vec step; + double model_cost_change = 0.0; + if (llt.info() == Eigen::Success) { + step = -llt.solve(gs); + if (step.allFinite()) { + model_cost_change = -(step.dot(gs) + 0.5 * step.dot(Hs * step)); + step_valid = model_cost_change > 0.0; + } + } + if (!step_valid) { + if (++consecutive_invalid >= max_consecutive_invalid_steps) + return finish(LMTermination::Failure); + step_rejected(); + continue; + } + consecutive_invalid = 0; + Vec delta = step.cwiseProduct(scale); + + bool have_trial = false; + if (constrained) { + // Projected Armijo line search along delta, cubic interpolation. + const double initial_gradient = cur.g.dot(delta); + const double direction_max_norm = delta.lpNorm(); + LMLineSample initial{0.0, cur.cost, initial_gradient, true, true}; + LMLineSample previous, current; + const auto line_eval = [&](double alpha, LMLineSample &s) { + s = LMLineSample{}; + s.x = alpha; + Vec moved; + plus(cur.x, Vec(alpha * delta), moved); + evaluate(moved, trial); + have_trial = true; + if (!trial.valid) + return; + s.value = trial.cost; + s.value_is_valid = true; + s.gradient = delta.dot(trial.g); + s.gradient_is_valid = std::isfinite(s.gradient); + }; + line_eval(1.0, current); + bool success = true; + int ls_iterations = 0; + while (!current.value_is_valid + || current.value > initial.value + sufficient_decrease * initial_gradient * current.x) { + ++ls_iterations; + if (ls_iterations >= max_line_search_iterations) { + success = false; + break; + } + const double alpha = LMInterpolatedStepSize(initial, previous, current, + max_step_contraction * current.x, + min_step_contraction * current.x); + if (alpha * direction_max_norm < min_line_search_step) { + success = false; + break; + } + previous = current; + line_eval(alpha, current); + } + summary.line_search_steps += ls_iterations; + if (success) + delta *= current.x; + } + + // ComputeCandidatePointAndEvaluateCost + Vec candidate_x; + plus(cur.x, delta, candidate_x); + Point cand; + if (have_trial && trial.x == candidate_x) + cand = std::move(trial); + else + evaluate(candidate_x, cand); + + if (any_successful_step) { + const double step_norm = free_norm(cur.x - cand.x); + if (step_norm <= parameter_tolerance * (free_norm(cur.x) + parameter_tolerance)) + return finish(LMTermination::Convergence); + } + if (std::fabs(cur.cost - cand.cost) <= function_tolerance * cur.cost) + return finish(LMTermination::Convergence); + + const double relative_decrease = (cand.cost >= kMax) + ? std::numeric_limits::lowest() + : (cur.cost - cand.cost) / model_cost_change; + if (relative_decrease > min_relative_decrease) { + any_successful_step = true; + step_successful = true; + cur = std::move(cand); + take_point(); + radius = radius / std::max(1.0 / 3.0, 1.0 - std::pow(2.0 * relative_decrease - 1.0, 3)); + radius = std::min(max_radius, radius); + decrease_factor = 2.0; + reuse_diagonal = false; + } else { + step_rejected(); + } + } +} diff --git a/image_analysis/geom_refinement/XtalOptimizer.cpp b/image_analysis/geom_refinement/XtalOptimizer.cpp index 58b049e10..23e671a2f 100644 --- a/image_analysis/geom_refinement/XtalOptimizer.cpp +++ b/image_analysis/geom_refinement/XtalOptimizer.cpp @@ -7,84 +7,11 @@ #include "XtalOptimizer.h" #include "XtalResidual.h" -#include "ceres/ceres.h" +#include "XtalRefine.h" #include "ceres/rotation.h" +#include "Dual.h" #include "LatticeReduction.h" -// Soft header prior on ONE beam-centre component (the spindle-parallel, gauge-weak one). Residual = w*(b - b0); -// the caller sets w so the prior behaves like a sigma-pixel restraint that competes with the (unit-weight) -// positional residuals - strong enough to pin the gauge direction, negligible in the well-constrained one. -// Soft restraint on one direction of a two-component block: g.(p - p0), weighted. Used for the beam -// centre and for the detector tilt, which are the same gauge seen twice (see the gauge block below), -// so they take the same direction g and cannot disagree about it. -struct GaugeDirectionPrior { - GaugeDirectionPrior(double gx, double gy, double p0, double weight) - : gx(gx), gy(gy), p0(p0), weight(weight) {} - template - bool operator()(const T *const p, T *residual) const { - residual[0] = T(weight) * (T(gx) * p[0] + T(gy) * p[1] - T(p0)); - return true; - } - double gx, gy, p0, weight; -}; - -struct XtalResidualRotationOnlyPrecomp { - XtalResidualRotationOnlyPrecomp(const Coord &recip_obs, - const CrystalLattice &latt, - double h, double k, double l) - : s_obs(recip_obs), - astar(latt.Astar()), bstar(latt.Bstar()), cstar(latt.Cstar()), - h(h), k(k), l(l) { - } - - template - bool operator()(const T *const rot_aa, T *residual) const { - const T astar_unrot[3] = {T(astar.x), T(astar.y), T(astar.z)}; - const T bstar_unrot[3] = {T(bstar.x), T(bstar.y), T(bstar.z)}; - const T cstar_unrot[3] = {T(cstar.x), T(cstar.y), T(cstar.z)}; - - T astar_rot[3], bstar_rot[3], cstar_rot[3]; - - const AngleAxisRotator rot(rot_aa); - rot.Rotate(astar_unrot, astar_rot); - rot.Rotate(bstar_unrot, bstar_rot); - rot.Rotate(cstar_unrot, cstar_rot); - - const Eigen::Matrix s_pred(T(h) * astar_rot[0] + T(k) * bstar_rot[0] + T(l) * cstar_rot[0], - T(h) * astar_rot[1] + T(k) * bstar_rot[1] + T(l) * cstar_rot[1], - T(h) * astar_rot[2] + T(k) * bstar_rot[2] + T(l) * cstar_rot[2] - ); - - // Residual in reciprocal space - residual[0] = T(s_obs.x) - s_pred[0]; - residual[1] = T(s_obs.y) - s_pred[1]; - residual[2] = T(s_obs.z) - s_pred[2]; - return true; - } - - const Coord s_obs; - const Coord astar, bstar, cstar; - const double h, k, l; -}; - -// Regularizer: penalises ||rot_aa|| to prefer the smallest rotation that -// explains the data. Weight should be chosen in the same units as the -// reciprocal-space residuals (Å⁻¹ per radian). A value of ~0.01–0.1 is -// typically enough to break degeneracy without biasing the solution. -struct RotationNormRegularizer { - explicit RotationNormRegularizer(double weight) : weight(weight) {} - - template - bool operator()(const T *const rot_aa, T *residual) const { - residual[0] = T(weight) * rot_aa[0]; - residual[1] = T(weight) * rot_aa[1]; - residual[2] = T(weight) * rot_aa[2]; - return true; - } - - const double weight; -}; - // Prior confidence weight per spot: how strong the spot is FOR ITS RESOLUTION. The frame's spots are // ordered by resolution and cut into equal-count shells, and each intensity is divided by its shell // median. Refinement needs the high-resolution spots (they carry the cell and distance information) and @@ -153,12 +80,11 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, const int num_threads) { try { // A coplanar basis has no reciprocal cell: 1/V is infinite, every predicted reciprocal vector - // comes out NaN, and Ceres fails on the very first evaluation - after dumping the offending - // block to stderr. There is nothing for the refinement to recover here, so refuse the lattice - // before the problem is built rather than let the solver discover it. The check has to be on - // the vectors: this close to flat, float cell angles no longer carry even the SIGN of the - // metric determinant, and the triclinic branch of XtalResidual then clamps c into the a-b - // plane and divides by the zero volume that makes. + // comes out NaN, and the solver fails on the very first evaluation. There is nothing for the + // refinement to recover here, so refuse the lattice before the problem is built rather than let + // the solver discover it. The check has to be on the vectors: this close to flat, float cell + // angles no longer carry even the SIGN of the metric determinant, and the triclinic branch of + // XtalResidual then clamps c into the a-b plane and divides by the zero volume that makes. if (data.latt.VolumeFraction() < MIN_BASIS_VOLUME_FRACTION) return false; @@ -168,24 +94,21 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, double beta = data.latt.GetUnitCell().beta; // Initial guess for the parameters - double beam[2] = {data.geom.GetBeamX_pxl(), data.geom.GetBeamY_pxl()}; - double distance_mm = data.geom.GetDetectorDistance_mm(); + const double distance_mm = data.geom.GetDetectorDistance_mm(); - double detector_rot[2] = {data.geom.GetPoniRot1_rad(), data.geom.GetPoniRot2_rad()}; - - // The per-frame constants of the reduced residual (see XtalFrameConstants), one entry per frame - // that contributes. Reserved up front and never grown past that, so the residual blocks' pointers - // into it stay valid, and declared before the problem so that it outlives it. - std::vector frame_const; - frame_const.reserve(spots.size()); - - ceres::Problem problem; - - double latt_vec0[3] = {0.0, 0.0, 0.0}; - double latt_vec1[3] = {0.0, 0.0, 0.0}; - double latt_vec2[3] = {0.0, 0.0, 0.0}; - - double rot_vec[3] = {1, 0, 0}; + XtalRefineProblem problem; + problem.crystal_system = data.crystal_system; + problem.distance_mm = distance_mm; + double *beam = problem.beam; + beam[0] = data.geom.GetBeamX_pxl(); + beam[1] = data.geom.GetBeamY_pxl(); + double *detector_rot = problem.detector_rot; + detector_rot[0] = data.geom.GetPoniRot1_rad(); + detector_rot[1] = data.geom.GetPoniRot2_rad(); + double *latt_vec0 = problem.latt_vec0; + double *latt_vec1 = problem.latt_vec1; + double *latt_vec2 = problem.latt_vec2; + double *rot_vec = problem.rot_vec; switch (data.crystal_system) { case gemmi::CrystalSystem::Orthorhombic: @@ -252,14 +175,15 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, const double sin_rot3 = std::sin(data.geom.GetPoniRot3_rad()); // Per-image rotation refinement frees only the beam and the orientation and holds the other five - // blocks constant, so the seven-block residual makes Ceres differentiate 17 parameters to use 5. - // Where that is the configuration, use the reduced residual instead - identical fit, Jet<5> - // autodiff. Any other combination (stills also free the cell, the offline refiner frees distance - // and detector angles) keeps the general form below. + // blocks constant. Where that is the configuration, the solver uses the reduced residual - the + // identical fit, with the crystal half worked out once (see XtalResidualBeamOrientation). Any + // other combination (stills also free the cell, the rotation indexer frees detector angles and + // spindle) keeps the general form. const bool beam_and_orientation_only = data.refine_beam_center && !data.refine_detector_angles && !data.refine_rotation_axis && !data.refine_unit_cell; + problem.beam_and_orientation_only = beam_and_orientation_only; // Sum of w^2 over the spots that entered - the beam prior below is scaled by it so that its // strength relative to the data is the same weighted or not. Equals the residual block count @@ -281,9 +205,8 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, rot_matr = data.axis->GetTransformationAngle(angle_deg); } - if (beam_and_orientation_only) - frame_const.emplace_back(detector_rot, rot_vec, angle_rad, latt_vec1, latt_vec2, - data.crystal_system); + const int frame_index = static_cast(problem.frame_angle_rad.size()); + problem.frame_angle_rad.push_back(angle_rad); // Add residuals for each point for (size_t j = 0; j < spots[i].size(); j++) { @@ -333,7 +256,7 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, const double weight_sq = weight.empty() ? 1.0 : weight[j] * weight[j]; effective_spots += weight_sq; - const XtalResidual residual(pt.x, pt.y, + problem.residuals.emplace_back(pt.x, pt.y, data.geom.GetWavelength_A(), data.geom.GetPixelSize_mm(), cos_rot3, sin_rot3, @@ -341,38 +264,14 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, h, k, l, data.crystal_system, data.geom.GetOrientation()); - - // Ceres has no per-residual weight; ScaledLoss(nullptr, a) multiplies the squared - // residual by the constant a, i.e. it applies a weight of sqrt(a) to the residual. - ceres::LossFunction *loss = weight.empty() - ? nullptr - : new ceres::ScaledLoss(nullptr, weight_sq, - ceres::TAKE_OWNERSHIP); - - if (beam_and_orientation_only) - problem.AddResidualBlock( - new ceres::AutoDiffCostFunction( - new XtalResidualBeamOrientation(residual, distance_mm, frame_const.back())), - loss, - beam, - latt_vec0 - ); - else - problem.AddResidualBlock( - new ceres::AutoDiffCostFunction( - new XtalResidualFixedDistance(residual, distance_mm)), - loss, - beam, - detector_rot, - rot_vec, - latt_vec0, - latt_vec1, - latt_vec2 - ); + problem.frame.push_back(frame_index); + // A per-residual weight w enters the squared residual as w^2. + if (!weight.empty()) + problem.weight_sq.push_back(weight_sq); } } - if (problem.NumResidualBlocks() < data.min_spots) + if (static_cast(problem.residuals.size()) < data.min_spots) return false; // The gauge direction of a single-axis rotation experiment - parallel to the spindle - written @@ -440,9 +339,8 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, const double gauge_w = data.geom.GetPixelSize_mm() / (distance_mm * data.geom.GetWavelength_A()) * std::sqrt(effective_spots) / sigma_px; - if (!data.refine_beam_center) - problem.SetParameterBlockConstant(beam); - else if (data.axis) { + problem.beam_constant = !data.refine_beam_center; + if (data.refine_beam_center && data.axis) { // Gauge handling (single-axis rotation): rotating the whole experiment about the spindle leaves every // spot position unchanged, so the beam-centre component PARALLEL to the spindle is a null/gauge-weak // direction. Refining it freely lets it wander (~+3 px) and absorb centroid systematics into a wrong @@ -450,23 +348,19 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, // does drift - it is only LaB6-monitored to ~a few px), RESTRAIN it toward the header with a soft // prior: the gauge direction has ~zero data sensitivity so the prior pins it near the header, while a // real, well-supported drift can still overcome it. - problem.AddResidualBlock( - new ceres::AutoDiffCostFunction( - new GaugeDirectionPrior(gauge_beam_x, gauge_beam_y, - gauge_beam_x * beam[0] + gauge_beam_y * beam[1], gauge_w)), - nullptr, beam); + problem.priors.push_back({XtalRefinePrior::Block::Beam, gauge_beam_x, gauge_beam_y, + gauge_beam_x * beam[0] + gauge_beam_y * beam[1], gauge_w}); } // Distance, detector angles, rotation axis and cell are parameter blocks only in the general // seven-block residual; the reduced one bakes them in, so there is nothing left to configure. if (!beam_and_orientation_only) { - if (!data.refine_detector_angles) { - problem.SetParameterBlockConstant(detector_rot); - } else { + problem.detector_rot_constant = !data.refine_detector_angles; + if (data.refine_detector_angles) { const double rot_range = 3.0 / 180.0 * PI; for (int i = 0; i < 2; ++i) { - problem.SetParameterLowerBound(detector_rot, i, detector_rot[i] - rot_range); - problem.SetParameterUpperBound(detector_rot, i, detector_rot[i] + rot_range); + problem.detector_rot_lower[i] = detector_rot[i] - rot_range; + problem.detector_rot_upper[i] = detector_rot[i] + rot_range; } // The same gauge as the beam prior above, described a second time: the tilt moves the // direct beam exactly as the beam centre does, at D/pixel px per radian, so leaving @@ -497,20 +391,15 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, for (int i = 0; i < 2; ++i) { if (budget[i] <= 0.0) continue; - problem.AddResidualBlock( - new ceres::AutoDiffCostFunction( - new GaugeDirectionPrior(dirs[i][0], dirs[i][1], - dirs[i][0] * detector_rot[0] - + dirs[i][1] * detector_rot[1], - gauge_w * (sigma_px / budget[i]) * lever)), - nullptr, detector_rot); + problem.priors.push_back({XtalRefinePrior::Block::DetectorRot, dirs[i][0], dirs[i][1], + dirs[i][0] * detector_rot[0] + dirs[i][1] * detector_rot[1], + gauge_w * (sigma_px / budget[i]) * lever}); } } } - if (!data.refine_rotation_axis) { - problem.SetParameterBlockConstant(rot_vec); - } else { + problem.rot_vec_constant = !data.refine_rotation_axis; + if (data.refine_rotation_axis) { // Only the DIRECTION of the goniometer axis is a parameter. The residual applies // angle_rad * |rot_vec|, so a free three-vector also fits a rotation SCALE - which // GoniometerAxis::Axis() then normalises away, leaving the candidate scored by @@ -520,63 +409,46 @@ bool XtalOptimizerInternal(XtalOptimizerData &data, // data it recovers 54 % of a known scale error, repeated first passes on one dataset // disagree with each other in SIGN, and on the one dataset with a real 1.3 % stage // fault it comes out negative. The rotation scale is measured properly, once, with - // four gates and a jackknife, in PostRefine. - problem.SetManifold(rot_vec, new ceres::SphereManifold<3>); + // four gates and a jackknife, in PostRefine. Refined on the sphere (see SolveXtalRefine). } - if (!data.refine_unit_cell) { - problem.SetParameterBlockConstant(latt_vec1); - problem.SetParameterBlockConstant(latt_vec2); - } else { + problem.latt_vec1_constant = !data.refine_unit_cell; + problem.latt_vec2_constant = !data.refine_unit_cell; + if (data.refine_unit_cell) { // Parameter bounds // Lengths for (int i = 0; i < 3; ++i) { - problem.SetParameterLowerBound(latt_vec1, i, data.min_length_A); - problem.SetParameterUpperBound(latt_vec1, i, data.max_length_A); + problem.latt_vec1_lower[i] = data.min_length_A; + problem.latt_vec1_upper[i] = data.max_length_A; } if (data.crystal_system == gemmi::CrystalSystem::Monoclinic) { - const double beta_lo = std::max(1e-6, PI * (data.min_angle_deg / 180.0)); - const double beta_hi = std::min(PI - 1e-6, PI * (data.max_angle_deg / 180.0)); - problem.SetParameterLowerBound(latt_vec2, 0, beta_lo); - problem.SetParameterUpperBound(latt_vec2, 0, beta_hi); + problem.latt_vec2_constant = false; + problem.latt_vec2_lower[0] = std::max(1e-6, PI * (data.min_angle_deg / 180.0)); + problem.latt_vec2_upper[0] = std::min(PI - 1e-6, PI * (data.max_angle_deg / 180.0)); } else if (data.crystal_system == gemmi::CrystalSystem::Triclinic) { // α, β, γ bounds (radians) const double alo = PI * (data.min_angle_deg / 180.0); const double ahi = PI * (data.max_angle_deg / 180.0); for (int i = 0; i < 3; ++i) { - problem.SetParameterLowerBound(latt_vec2, i, alo); - problem.SetParameterUpperBound(latt_vec2, i, ahi); + problem.latt_vec2_lower[i] = alo; + problem.latt_vec2_upper[i] = ahi; } } else { // Orthorhombic / Tetragonal / Cubic / Hexagonal: // latt_vec2 has no meaning for these systems — always freeze it. - problem.SetParameterBlockConstant(latt_vec2); + problem.latt_vec2_constant = true; } } } - // Configure solver - ceres::Solver::Options options; - // Normal equations, not QR. The problem is very tall and thin - thousands of spots against at - // most 17 parameters - and that is the shape DENSE_QR handles worst: it copies the Jacobian out - // of Ceres' row-major storage into a column-major buffer on every solve, and Eigen's blocked - // Householder then degenerates to the unblocked path because its block size is min(48, columns). - // Accumulating J^T J reads the Jacobian once instead. Both solve the same damped system, so the - // step is the same to round-off; the column scaling Ceres applies by default and the LM diagonal - // keep the squared condition number in hand. - options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY; - options.minimizer_progress_to_stdout = false; + // Stopping rule: a bound on iterations is reproducible, a bound on wall-clock time is not (see + // XtalOptimizerData::max_iterations). if (data.max_iterations > 0) - options.max_num_iterations = data.max_iterations; + problem.options.max_iterations = data.max_iterations; else - options.max_solver_time_in_seconds = data.max_time; - options.logging_type = ceres::LoggingType::SILENT; - options.num_threads = num_threads; // usually 1 (called from many threads); caller may raise it - ceres::Solver::Summary summary; - - // Run optimization - ceres::Solve(options, &problem, &summary); + problem.options.max_time_s = data.max_time; + const LMSummary summary = SolveXtalRefine(problem, num_threads); // Only a genuine numerical failure is rejected here: a solve that ran out of iterations or // out of time but still descended counts as usable, which is what the real-time caller @@ -653,7 +525,7 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data, return false; // Parameter: angle-axis for the extra rotation. Identity == {0,0,0}. - double rot_aa[3] = {0.0, 0.0, 0.0}; + std::vector rot_aa = {0.0, 0.0, 0.0}; // Spot selection by current indexing (same approach as XtalOptimizerInternal) const Coord a0 = data.latt.Vec0(); @@ -662,7 +534,12 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data, const float tol_sq = tolerance * tolerance; - ceres::Problem problem; + // Each selected spot: its observed reciprocal vector and the indices it is fitted to. + struct Observation { + Coord s_obs; + double h, k, l; + }; + std::vector observations; for (const auto &pt : spots) { if (!data.index_ice_rings && pt.ice_ring) @@ -697,42 +574,68 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data, if (data.axis.has_value()) s_obs = data.axis->GetTransformationAngle(pt.phi) * s_obs; - auto *cost = - new ceres::AutoDiffCostFunction( - new XtalResidualRotationOnlyPrecomp(s_obs, data.latt, h, k, l) - ); - - problem.AddResidualBlock(cost, nullptr, rot_aa); + observations.push_back({s_obs, h, k, l}); } - if (problem.NumResidualBlocks() < data.min_spots) + if (static_cast(observations.size()) < data.min_spots) return false; - // Regularization: prefer the smallest rotation correction that fits the - // data. This is essential when spots are nearly coplanar in reciprocal - // space (e.g. still images), where the rotation component perpendicular - // to the scattering plane is otherwise underdetermined. - // The weight is in Å⁻¹ rad⁻¹; tune relative to your typical residual. - { - const double reg_weight = 0.05; // e.g. 0.05 - problem.AddResidualBlock( - new ceres::AutoDiffCostFunction( - new RotationNormRegularizer(reg_weight)), - nullptr, rot_aa); - } + // Residual: s_obs - R(rot_aa) (h a* + k b* + l c*), the reciprocal basis rotated once per + // evaluation and shared by every spot. + // + // Regularization: prefer the smallest rotation correction that fits the data, w * rot_aa. This is + // essential when spots are nearly coplanar in reciprocal space (e.g. still images), where the + // rotation component perpendicular to the scattering plane is otherwise underdetermined. The + // weight is in A^-1 rad^-1, relative to the typical residual. + const double reg_weight = 0.05; + const Coord astar = data.latt.Astar(), bstar = data.latt.Bstar(), cstar = data.latt.Cstar(); + const auto evaluate = [&](const double *aa, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) { + using D = Dual<3>; + const D aa_d[3] = {D::Variable(aa[0], 0), D::Variable(aa[1], 1), D::Variable(aa[2], 2)}; + const AngleAxisRotator rot(aa_d); + const double astar_unrot[3] = {astar.x, astar.y, astar.z}; + const double bstar_unrot[3] = {bstar.x, bstar.y, bstar.z}; + const double cstar_unrot[3] = {cstar.x, cstar.y, cstar.z}; + D astar_rot[3], bstar_rot[3], cstar_rot[3]; + rot.Rotate(astar_unrot, astar_rot); + rot.Rotate(bstar_unrot, bstar_rot); + rot.Rotate(cstar_unrot, cstar_rot); - ceres::Solver::Options options; - options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY; // tall and thin, as above - options.minimizer_progress_to_stdout = false; + cost = 0.0; + const auto add = [&](double r, const double *J) { + cost += 0.5 * r * r; + if (!g) + return; + for (int i = 0; i < 3; i++) { + (*g)[i] += J[i] * r; + for (int j = 0; j < 3; j++) + (*H)(i, j) += J[i] * J[j]; + } + }; + for (const auto &o: observations) { + const double s_obs[3] = {o.s_obs.x, o.s_obs.y, o.s_obs.z}; + for (int c = 0; c < 3; c++) { + const D pred = o.h * astar_rot[c] + o.k * bstar_rot[c] + o.l * cstar_rot[c]; + const double J[3] = {-pred.v[0], -pred.v[1], -pred.v[2]}; + add(s_obs[c] - pred.a, J); + } + } + for (int c = 0; c < 3; c++) { + double J[3] = {0.0, 0.0, 0.0}; + J[c] = reg_weight; + add(reg_weight * aa[c], J); + } + return std::isfinite(cost) && (!g || (g->allFinite() && H->allFinite())); + }; + + std::vector blocks(1); + blocks[0].size = 3; + LMOptions options; if (data.max_iterations > 0) - options.max_num_iterations = data.max_iterations; + options.max_iterations = data.max_iterations; else - options.max_solver_time_in_seconds = data.max_time; - options.logging_type = ceres::LoggingType::SILENT; - options.num_threads = 1; - - ceres::Solver::Summary summary; - ceres::Solve(options, &problem, &summary); + options.max_time_s = data.max_time; + const LMSummary summary = SolveLM(rot_aa, blocks, options, evaluate); if (!summary.IsSolutionUsable()) return false; @@ -747,7 +650,7 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data, // rotating the reciprocal vectors (a*, b*, c*) by the same R. No // transpose or inversion of R is needed here. double R_raw[9]; - ceres::AngleAxisToRotationMatrix(rot_aa, R_raw); // row-major 3x3 + ceres::AngleAxisToRotationMatrix(rot_aa.data(), R_raw); // row-major 3x3 Eigen::Matrix3d R; R << R_raw[0], R_raw[3], R_raw[6], diff --git a/image_analysis/geom_refinement/XtalOptimizer.h b/image_analysis/geom_refinement/XtalOptimizer.h index 1bf9084de..0b64937bb 100644 --- a/image_analysis/geom_refinement/XtalOptimizer.h +++ b/image_analysis/geom_refinement/XtalOptimizer.h @@ -62,7 +62,7 @@ struct XtalOptimizerData { std::optional angle_axis; }; -// num_threads sets the Ceres solver thread count for the internal least-squares refine. It defaults +// num_threads sets the thread count of the internal least-squares refine (the answer does not depend on it). It defaults // to 1 because XtalOptimizer is usually called from many threads at once; raise it only when a caller // runs a small number of refinements concurrently and wants each to use several cores. bool XtalOptimizer(XtalOptimizerData &data, std::span> spots, diff --git a/image_analysis/geom_refinement/XtalRefine.cpp b/image_analysis/geom_refinement/XtalRefine.cpp new file mode 100644 index 000000000..3b466b97d --- /dev/null +++ b/image_analysis/geom_refinement/XtalRefine.cpp @@ -0,0 +1,334 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#include "XtalRefine.h" + +#include + +#include "Dual.h" +#include "../../common/ParallelFor.h" + +namespace { + // Ambient layout of the parameter vector and the order of the blocks in it. + constexpr int OFF_BEAM = 0, OFF_ROT = 2, OFF_AXIS = 4, OFF_P0 = 7, OFF_LEN = 10, OFF_ANG = 13, N_AMBIENT = 16; + + // Derivative lanes. The observed half of a residual depends on beam, detector angles and spindle + // (OBS lanes), the predicted half on orientation and cell (PRED lanes); each half is carried on a + // dual number of its own width and the two are put side by side only in the sums. + constexpr int OBS = 6, PRED = 9, LANES = OBS + PRED; + using DO = Dual; + using DP = Dual; + + // The reduced problem: beam (2) and orientation (3) only. + constexpr int R_OBS = 2, R_PRED = 3, R_LANES = R_OBS + R_PRED; + + // Sums over one block of residuals, in lane coordinates. + template + struct Sums { + double cost = 0.0; + double g[L] = {}; + double H[L][L] = {}; // upper triangle + + void Add(double r, const double *J, double w2) { + cost += 0.5 * w2 * r * r; + for (int i = 0; i < L; i++) { + const double wj = w2 * J[i]; + g[i] += wj * r; + for (int j = i; j < L; j++) + H[i][j] += wj * J[j]; + } + } + + void Add(const Sums &o) { + cost += o.cost; + for (int i = 0; i < L; i++) { + g[i] += o.g[i]; + for (int j = i; j < L; j++) + H[i][j] += o.H[i][j]; + } + } + }; + + std::vector MakeBlocks(const XtalRefineProblem &p) { + std::vector blocks(6); + blocks[0].offset = OFF_BEAM; + blocks[0].size = 2; + blocks[0].constant = p.beam_constant; + + blocks[1].offset = OFF_ROT; + blocks[1].size = 2; + blocks[1].constant = p.beam_and_orientation_only || p.detector_rot_constant; + for (int j = 0; j < 2; j++) { + blocks[1].lower[j] = p.detector_rot_lower[j]; + blocks[1].upper[j] = p.detector_rot_upper[j]; + } + + blocks[2].offset = OFF_AXIS; + blocks[2].size = 3; + blocks[2].constant = p.beam_and_orientation_only || p.rot_vec_constant; + blocks[2].sphere = true; + + blocks[3].offset = OFF_P0; + blocks[3].size = 3; + + blocks[4].offset = OFF_LEN; + blocks[4].size = 3; + blocks[4].constant = p.beam_and_orientation_only || p.latt_vec1_constant; + + blocks[5].offset = OFF_ANG; + blocks[5].size = 3; + blocks[5].constant = p.beam_and_orientation_only || p.latt_vec2_constant; + + for (int j = 0; j < 3; j++) { + blocks[4].lower[j] = p.latt_vec1_lower[j]; + blocks[4].upper[j] = p.latt_vec1_upper[j]; + blocks[5].lower[j] = p.latt_vec2_lower[j]; + blocks[5].upper[j] = p.latt_vec2_upper[j]; + } + return blocks; + } + + // Where each lane lands among the tangent coordinates of the free blocks; -1 for a held one. + template + std::array LaneToTangent(const std::vector &blocks, const std::array &lane_block, + const std::array &lane_index) { + std::array map{}; + std::array tangent_offset{}; + int t = 0; + for (size_t b = 0; b < blocks.size(); b++) { + tangent_offset[b] = t; + t += blocks[b].TangentSize(); + } + for (size_t l = 0; l < L; l++) + map[l] = blocks[lane_block[l]].constant ? -1 : tangent_offset[lane_block[l]] + lane_index[l]; + return map; + } + + template + void ToTangent(const Sums &s, const std::array &map, Eigen::VectorXd &g, Eigen::MatrixXd &H) { + for (int i = 0; i < L; i++) { + if (map[i] < 0) + continue; + g[map[i]] += s.g[i]; + for (int j = i; j < L; j++) { + if (map[j] < 0) + continue; + H(map[i], map[j]) += s.H[i][j]; + if (map[i] != map[j]) + H(map[j], map[i]) += s.H[i][j]; + } + } + } + + void AddPriors(const XtalRefineProblem &p, const double *x, double &cost, Eigen::VectorXd *g, + Eigen::MatrixXd *H, int beam_tangent, int rot_tangent) { + for (const auto &prior: p.priors) { + const bool on_beam = prior.block == XtalRefinePrior::Block::Beam; + const double *v = x + (on_beam ? OFF_BEAM : OFF_ROT); + const double r = prior.weight * (prior.gx * v[0] + prior.gy * v[1] - prior.p0); + cost += 0.5 * r * r; + const int t = on_beam ? beam_tangent : rot_tangent; + if (!g || t < 0) + continue; + const double J[2] = {prior.weight * prior.gx, prior.weight * prior.gy}; + for (int i = 0; i < 2; i++) { + (*g)[t + i] += J[i] * r; + for (int j = 0; j < 2; j++) + (*H)(t + i, t + j) += J[i] * J[j]; + } + } + } + + template + D Seed(double value, int lane, bool free) { + return free ? D::Variable(value, lane) : D(value); + } + + // Residual blocks of at least this many residuals; the cut depends on the count alone. + constexpr int MIN_RESIDUALS_PER_BLOCK = 256; + + template + Sums SumResiduals(const XtalRefineProblem &p, int num_threads, Fn &&residual) { + const int n = static_cast(p.residuals.size()); + std::vector> partial(ReductionBlocks(n, MIN_RESIDUALS_PER_BLOCK)); + ParallelBlocks(n, std::max(1, num_threads), [&](int b, int lo, int hi) { + for (int i = lo; i < hi; i++) + residual(i, partial[b]); + }, MIN_RESIDUALS_PER_BLOCK); + Sums total; + for (const auto &s: partial) + total.Add(s); + return total; + } + + double WeightSq(const XtalRefineProblem &p, int i) { + return p.weight_sq.empty() ? 1.0 : p.weight_sq[i]; + } + + bool AllFinite(double cost, const Eigen::VectorXd *g, const Eigen::MatrixXd *H) { + return std::isfinite(cost) && (!g || (g->allFinite() && H->allFinite())); + } + + // Seven-block residual (XtalResidualFixedDistance), with every block that depends on parameters + // alone - detector-angle sines and cosines, the spindle back-rotation of each frame, the reciprocal + // basis of the cell, the orientation's rotation - worked out once per evaluation. + bool EvaluateGeneral(const XtalRefineProblem &p, const std::array &map, int num_threads, + const double *x, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) { + const bool beam_free = map[0] >= 0, rot_free = map[2] >= 0, axis_free = map[4] >= 0; + const bool p0_free = map[OBS] >= 0, len_free = map[OBS + 3] >= 0, ang_free = map[OBS + 6] >= 0; + + const DO beam[2] = {Seed(x[OFF_BEAM], 0, beam_free), + Seed(x[OFF_BEAM + 1], 1, beam_free)}; + const DO rot1 = Seed(x[OFF_ROT], 2, rot_free); + const DO rot2 = Seed(x[OFF_ROT + 1], 3, rot_free); + const DO c1 = cos(rot1), s1 = sin(rot1), c2 = cos(rot2), s2 = sin(rot2); + + DO axis[3] = {x[OFF_AXIS], x[OFF_AXIS + 1], x[OFF_AXIS + 2]}; + if (axis_free) { + double J[3][2]; + SpherePlusJacobian3(x + OFF_AXIS, J); + for (int k = 0; k < 3; k++) { + axis[k].v[4] = J[k][0]; + axis[k].v[5] = J[k][1]; + } + } + std::vector> rot_back; + rot_back.reserve(p.frame_angle_rad.size()); + for (const double angle: p.frame_angle_rad) { + const DO aa_back[3] = {angle * axis[0], angle * axis[1], angle * axis[2]}; + rot_back.emplace_back(aa_back); + } + + DP p0[3], len[3], ang[3]; + for (int k = 0; k < 3; k++) { + p0[k] = Seed(x[OFF_P0 + k], k, p0_free); + len[k] = Seed(x[OFF_LEN + k], 3 + k, len_free); + ang[k] = Seed(x[OFF_ANG + k], 6 + k, ang_free); + } + Eigen::Matrix bxc, cxa, axb; + DP invV; + XtalResidual::ReciprocalBasis(len, ang, p.crystal_system, bxc, cxa, axb, invV); + const AngleAxisRotator rot_p0(p0); + + const Sums s = SumResiduals(p, num_threads, [&](int i, Sums &acc) { + const XtalResidual &res = p.residuals[i]; + DO obs[3]; + res.ObservedRecipCore(beam, p.distance_mm, c1, s1, c2, s2, rot_back[p.frame[i]], obs); + DP unrot[3], pred[3]; + res.CombineRecipUnrot(bxc, cxa, axb, invV, unrot); + rot_p0.Rotate(unrot, pred); + const double w2 = WeightSq(p, i); + for (int k = 0; k < 3; k++) { + double J[LANES]; + for (int l = 0; l < OBS; l++) + J[l] = obs[k].v[l]; + for (int l = 0; l < PRED; l++) + J[OBS + l] = -pred[k].v[l]; + acc.Add(obs[k].a - pred[k].a, J, w2); + } + }); + + cost = s.cost; + if (g) + ToTangent(s, map, *g, *H); + AddPriors(p, x, cost, g, H, map[0], map[2]); + return AllFinite(cost, g, H); + } + + // The reduced residual (XtalResidualBeamOrientation): detector, spindle and cell held, so the + // back-rotation of each frame and the unrotated prediction of each residual are constants. + bool EvaluateBeamOrientation(const XtalRefineProblem &p, const std::array &map, + const std::vector> &rot_back, + const std::vector> &unrot, double c1, double s1, + double c2, double s2, int num_threads, + const double *x, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) { + using DB = Dual; + using DR = Dual; + const bool beam_free = map[0] >= 0; + const DB beam[2] = {beam_free ? DB::Variable(x[OFF_BEAM], 0) : DB(x[OFF_BEAM]), + beam_free ? DB::Variable(x[OFF_BEAM + 1], 1) : DB(x[OFF_BEAM + 1])}; + const DR p0[3] = {DR::Variable(x[OFF_P0], 0), DR::Variable(x[OFF_P0 + 1], 1), + DR::Variable(x[OFF_P0 + 2], 2)}; + const AngleAxisRotator rot_p0(p0); + + const Sums s = SumResiduals(p, num_threads, [&](int i, Sums &acc) { + const XtalResidual &res = p.residuals[i]; + DB obs[3]; + res.ObservedRecipCore(beam, p.distance_mm, c1, s1, c2, s2, rot_back[p.frame[i]], obs); + DR pred[3]; + rot_p0.Rotate(unrot[i].data(), pred); + const double w2 = WeightSq(p, i); + for (int k = 0; k < 3; k++) { + double J[R_LANES]; + for (int l = 0; l < R_OBS; l++) + J[l] = obs[k].v[l]; + for (int l = 0; l < R_PRED; l++) + J[R_OBS + l] = -pred[k].v[l]; + acc.Add(obs[k].a - pred[k].a, J, w2); + } + }); + + cost = s.cost; + if (g) + ToTangent(s, map, *g, *H); + AddPriors(p, x, cost, g, H, map[0], -1); + return AllFinite(cost, g, H); + } +} + +LMSummary SolveXtalRefine(XtalRefineProblem &p, int num_threads) { + const std::vector blocks = MakeBlocks(p); + + std::vector x(N_AMBIENT); + const auto put = [&](int off, const double *v, int n) { for (int i = 0; i < n; i++) x[off + i] = v[i]; }; + put(OFF_BEAM, p.beam, 2); + put(OFF_ROT, p.detector_rot, 2); + put(OFF_AXIS, p.rot_vec, 3); + put(OFF_P0, p.latt_vec0, 3); + put(OFF_LEN, p.latt_vec1, 3); + put(OFF_ANG, p.latt_vec2, 3); + + LMSummary summary; + if (p.beam_and_orientation_only) { + const std::array lane_block = {0, 0, 3, 3, 3}; + const std::array lane_index = {0, 1, 0, 1, 2}; + const auto map = LaneToTangent(blocks, lane_block, lane_index); + + std::vector> rot_back; + rot_back.reserve(p.frame_angle_rad.size()); + for (const double angle: p.frame_angle_rad) + rot_back.push_back(XtalFrameConstants::BackRotator(angle, p.rot_vec)); + Eigen::Matrix bxc, cxa, axb; + double invV; + XtalResidual::ReciprocalBasis(p.latt_vec1, p.latt_vec2, p.crystal_system, bxc, cxa, axb, invV); + std::vector> unrot(p.residuals.size()); + for (size_t i = 0; i < p.residuals.size(); i++) + p.residuals[i].CombineRecipUnrot(bxc, cxa, axb, invV, unrot[i].data()); + const double c1 = std::cos(p.detector_rot[0]), s1 = std::sin(p.detector_rot[0]); + const double c2 = std::cos(p.detector_rot[1]), s2 = std::sin(p.detector_rot[1]); + + summary = SolveLM(x, blocks, p.options, [&](const double *at, double &cost, Eigen::VectorXd *g, + Eigen::MatrixXd *H) { + return EvaluateBeamOrientation(p, map, rot_back, unrot, c1, s1, c2, s2, num_threads, at, cost, g, H); + }); + } else { + const std::array lane_block = {0, 0, 1, 1, 2, 2, 3, 3, 3, 4, 4, 4, 5, 5, 5}; + const std::array lane_index = {0, 1, 0, 1, 0, 1, 0, 1, 2, 0, 1, 2, 0, 1, 2}; + const auto map = LaneToTangent(blocks, lane_block, lane_index); + summary = SolveLM(x, blocks, p.options, [&](const double *at, double &cost, Eigen::VectorXd *g, + Eigen::MatrixXd *H) { + return EvaluateGeneral(p, map, num_threads, at, cost, g, H); + }); + } + + if (summary.IsSolutionUsable()) { + const auto get = [&](int off, double *v, int n) { for (int i = 0; i < n; i++) v[i] = x[off + i]; }; + get(OFF_BEAM, p.beam, 2); + get(OFF_ROT, p.detector_rot, 2); + get(OFF_AXIS, p.rot_vec, 3); + get(OFF_P0, p.latt_vec0, 3); + get(OFF_LEN, p.latt_vec1, 3); + get(OFF_ANG, p.latt_vec2, 3); + } + return summary; +} diff --git a/image_analysis/geom_refinement/XtalRefine.h b/image_analysis/geom_refinement/XtalRefine.h new file mode 100644 index 000000000..4ee88870e --- /dev/null +++ b/image_analysis/geom_refinement/XtalRefine.h @@ -0,0 +1,63 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#pragma once + +#include +#include + +#include "XtalResidual.h" +#include "LMSolver.h" + +// A soft restraint w * (g . p - p0) on one direction g of a two-component block - the beam centre or +// the detector tilt, which are the same gauge seen twice and so take the same direction (see +// XtalOptimizer). +struct XtalRefinePrior { + enum class Block { Beam, DetectorRot } block = Block::Beam; + double gx = 0.0, gy = 0.0, p0 = 0.0, weight = 0.0; +}; + +// The least-squares problem XtalOptimizer solves, as data: the residuals with the frame each belongs +// to, the parameter blocks with what is held and what is bounded, and the priors. The parameter arrays +// are in/out. Parameter blocks: beam(2), detector_rot(2), rot_vec(3, the spindle, refined on the sphere), +// latt_vec0(3, orientation angle-axis), latt_vec1(3, cell lengths), latt_vec2(3, cell angles) - see +// XtalResidual. The distance is a constant of the problem. +struct XtalRefineProblem { + static constexpr double kNoBound = std::numeric_limits::max(); + + gemmi::CrystalSystem crystal_system = gemmi::CrystalSystem::Triclinic; + // Only beam and orientation free, everything else held: the reduced residual (see + // XtalResidualBeamOrientation), whose crystal half is a constant of the problem. + bool beam_and_orientation_only = false; + double distance_mm = 0.0; + + std::vector residuals; + std::vector frame; // per residual, an index into frame_angle_rad + std::vector frame_angle_rad; + std::vector weight_sq; // per residual; empty = unweighted + + double beam[2] = {0, 0}; + double detector_rot[2] = {0, 0}; + double rot_vec[3] = {1, 0, 0}; + double latt_vec0[3] = {0, 0, 0}; + double latt_vec1[3] = {0, 0, 0}; + double latt_vec2[3] = {0, 0, 0}; + + bool beam_constant = false; + bool detector_rot_constant = true; + bool rot_vec_constant = true; + bool latt_vec1_constant = true; + bool latt_vec2_constant = true; + + double detector_rot_lower[2] = {-kNoBound, -kNoBound}, detector_rot_upper[2] = {kNoBound, kNoBound}; + double latt_vec1_lower[3] = {-kNoBound, -kNoBound, -kNoBound}, latt_vec1_upper[3] = {kNoBound, kNoBound, kNoBound}; + double latt_vec2_lower[3] = {-kNoBound, -kNoBound, -kNoBound}, latt_vec2_upper[3] = {kNoBound, kNoBound, kNoBound}; + + std::vector priors; + + LMOptions options; +}; + +// Solves the problem in place. The residual sums are cut into blocks that depend on the problem +// alone, so the answer is the same at any thread count. +LMSummary SolveXtalRefine(XtalRefineProblem &problem, int num_threads); diff --git a/image_analysis/geom_refinement/XtalResidual.h b/image_analysis/geom_refinement/XtalResidual.h index b67e055c8..26c05a26a 100644 --- a/image_analysis/geom_refinement/XtalResidual.h +++ b/image_analysis/geom_refinement/XtalResidual.h @@ -7,8 +7,6 @@ #include -#include "ceres/ceres.h" -#include "ceres/rotation.h" #include "gemmi/symmetry.hpp" #include "../../common/JFJochException.h" @@ -110,6 +108,9 @@ inline void EffectiveCellFromParams(gemmi::CrystalSystem symmetry, const double } } +// The scalar type is a template parameter throughout: a double, a ceres::Jet or a Dual (Dual.h). The +// mathematical functions are called unqualified, so each type's own overload is found by lookup. +// // Detector -> reciprocal geometry residual, shared by the per-image XtalOptimizer (one lattice, one // frame) and the offline GeometryRefiner (shared beam/distance/cell blocks, one orientation block per // frame). Parameter blocks: beam(2), distance_mm(1), detector_rot(2 = rot1,rot2), rotation_axis(3), @@ -164,16 +165,18 @@ struct XtalResidual { // detector_rot[0] = rot1, detector_rot[1] = rot2 are refined; rot3 is fixed // (e.g. from a PONI import) and baked in here as a constant so that a non-zero // rot3 is not silently dropped during refinement. + using std::cos; + using std::sin; const C rot1 = detector_rot[0]; const C rot2 = detector_rot[1]; // Ry(+rot1): rotation around Y-axis - const C c1 = ceres::cos(rot1); - const C s1 = ceres::sin(rot1); + const C c1 = cos(rot1); + const C s1 = sin(rot1); // Rx(-rot2): rotation around X-axis with inverted sign (PyFAI left-handed) - const C c2 = ceres::cos(rot2); - const C s2 = ceres::sin(rot2); + const C c2 = cos(rot2); + const C s2 = sin(rot2); // Apply the goniometer "back-to-start" rotation of this frame's angle. const C aa_back[3] = { @@ -227,7 +230,8 @@ struct XtalResidual { const T z = t2_z; // convert to recip space - const T lab_norm = ceres::sqrt(x * x + y * y + z * z); + using std::sqrt; + const T lab_norm = sqrt(x * x + y * y + z * z); const T inv_norm = T(1) / lab_norm; T recip_raw[3]; @@ -258,6 +262,9 @@ struct XtalResidual { static void ReciprocalBasis(const C *const p1, const C *const p2, gemmi::CrystalSystem symmetry, Eigen::Matrix &bxc, Eigen::Matrix &cxa, Eigen::Matrix &axb, C &invV) { + using std::cos; + using std::sin; + using std::sqrt; // Build unit cell lengths and B (convention: columns are a, b, c prior to global rotation) Eigen::Matrix e_uc_len = Eigen::Matrix::Zero(); Eigen::Matrix B = Eigen::Matrix::Identity(); @@ -275,14 +282,14 @@ struct XtalResidual { } else if (symmetry == gemmi::CrystalSystem::Monoclinic) { // Unique axis b: alpha = gamma = 90°, beta free (angle between a and c) e_uc_len << p1[0], p1[1], p1[2]; - B(0, 2) = ceres::cos(p2[0]); - B(2, 2) = ceres::sin(p2[0]); + B(0, 2) = cos(p2[0]); + B(2, 2) = sin(p2[0]); } else { // Triclinic: p1 = (a,b,c), p2 = (alpha, beta, gamma) in radians - const C ca = ceres::cos(p2[0]); - const C cb = ceres::cos(p2[1]); - const C cg = ceres::cos(p2[2]); - const C sg = ceres::sin(p2[2]); + const C ca = cos(p2[0]); + const C cb = cos(p2[1]); + const C cg = cos(p2[2]); + const C sg = sin(p2[2]); e_uc_len << p1[0], p1[1], p1[2]; @@ -297,7 +304,7 @@ struct XtalResidual { const C cx = cb; const C cy = (ca - cb * cg) / sg; const C v = C(1) - cx * cx - cy * cy; - const C cz = (v >= C(0)) ? ceres::sqrt(v) : C(0); + const C cz = (v >= C(0)) ? sqrt(v) : C(0); B(0, 2) = cx; B(1, 2) = cy; @@ -399,12 +406,12 @@ struct XtalResidual { // reciprocal basis of the fixed cell. They are constants of the whole problem there - one frame per // image, one cell - and deriving them inside each residual costs six trigonometric calls, a hypot and a // division per evaluation for numbers that never change. Built by the caller, which has to keep it alive -// as long as the ceres::Problem that points at it. +// as long as the residuals that point at it. struct XtalFrameConstants { XtalFrameConstants(const double *detector_rot, const double *rotation_axis, double angle_rad, const double *uc_len, const double *uc_angle, gemmi::CrystalSystem symmetry) - : c1(ceres::cos(detector_rot[0])), s1(ceres::sin(detector_rot[0])), - c2(ceres::cos(detector_rot[1])), s2(ceres::sin(detector_rot[1])), + : c1(cos(detector_rot[0])), s1(sin(detector_rot[0])), + c2(cos(detector_rot[1])), s2(sin(detector_rot[1])), rot_back(BackRotator(angle_rad, rotation_axis)) { XtalResidual::ReciprocalBasis(uc_len, uc_angle, symmetry, bxc, cxa, axb, invV); } diff --git a/image_analysis/scale_merge/RotationScaleMerge.cpp b/image_analysis/scale_merge/RotationScaleMerge.cpp index c32a166f0..81dabcd22 100644 --- a/image_analysis/scale_merge/RotationScaleMerge.cpp +++ b/image_analysis/scale_merge/RotationScaleMerge.cpp @@ -3,6 +3,7 @@ #include "../../common/ParallelFor.h" #include "RotationScaleMerge.h" +#include "RotationScaleMergeGPU.h" // SurfaceTerm, the surface fit's term on both paths #include #include @@ -3334,7 +3335,7 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector &cell, int // traffic of the data it touched, over hundreds of megabytes. Copying once turns them into // sequential walks of a compact array. The copy keeps fulls order, so each sum below is formed // from exactly the same terms in exactly the same order. - struct Term { float I, sigma, corr, d; int32_t cell, group; }; + using Term = RotationScaleMergeGPU::SurfaceTerm; // float I, sigma, corr, d; int32_t cell, group std::vector term; std::vector term_parity; // frame parity, read only by the group-ordered copy below // Which terms each pass walks, as positions in `term`. The cross-validated halves then cost half a @@ -3437,22 +3438,27 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector &cell, int std::vector shw_shell(nshell, 0.0), shw_cell(nshell > 0 ? ncell : 0, 0.0); std::vector group_shell(n_groups, 0); // the shell each ASU group sits in, for the gate if (nshell > 0) { - std::vector s2; - s2.reserve(term.size()); - for (const Term &t : term) - s2.push_back(t.d > 0.0f ? 1.0f / (t.d * t.d) : 0.0f); + const int n_term = static_cast(term.size()); + std::vector s2(n_term); + ParallelChunks(n_term, nt, [&](int lo, int hi) { + for (int k = lo; k < hi; ++k) + s2[k] = term[k].d > 0.0f ? 1.0f / (term[k].d * term[k].d) : 0.0f; + }); + // The edges are order statistics of s2, so they are the same values however the sort orders + // equal elements among themselves. + std::vector sorted = s2; + ParallelSort(sorted.begin(), sorted.end(), nt, std::less()); std::vector edge(nshell - 1); - size_t prev = 0; - for (int i = 1; i < nshell; ++i) { - const size_t pos = s2.size() * static_cast(i) / static_cast(nshell); - std::nth_element(s2.begin() + prev, s2.begin() + pos, s2.end()); - edge[i - 1] = s2[pos]; - prev = pos; - } + for (int i = 1; i < nshell; ++i) + edge[i - 1] = sorted[sorted.size() * static_cast(i) / static_cast(nshell)]; + std::vector term_shell(n_term); + ParallelChunks(n_term, nt, [&](int lo, int hi) { + for (int k = lo; k < hi; ++k) + term_shell[k] = static_cast(std::upper_bound(edge.begin(), edge.end(), s2[k]) - edge.begin()); + }); for (size_t k = 0; k < term.size(); ++k) { const Term &t = term[k]; - const float v = t.d > 0.0f ? 1.0f / (t.d * t.d) : 0.0f; - const int s = static_cast(std::upper_bound(edge.begin(), edge.end(), v) - edge.begin()); + const int s = term_shell[k]; group_shell[t.group] = s; const double sc = static_cast(t.sigma) * t.corr; if (!(sc > 0.0)) continue; @@ -3475,8 +3481,41 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector &cell, int std::vector g_start(n_groups + 1, 0); for (const Term &t : term) ++g_start[t.group + 1]; for (int g = 0; g < n_groups; ++g) g_start[g + 1] += g_start[g]; - std::vector gterm(term.size()); - { + std::vector gterm; + // With a GPU the two passes of every round run there (RotationScaleMergeGPU::Surface*), on the same + // terms in the same order: the device gets the terms with the group order as a permutation, and each + // subset cut into this fit's reduction blocks with every block's terms ordered by cell, so that one + // device thread forms what one block of the host loop sums into one cell. + bool on_gpu = false; +#ifdef JFJOCH_USE_CUDA + on_gpu = gpu_active_; + if (on_gpu) { + std::vector gperm(term.size()); + std::vector fill(g_start.begin(), g_start.end() - 1); + for (size_t k = 0; k < term.size(); ++k) + gperm[fill[term[k].group]++] = static_cast(k); + gpu_->SurfaceSetTerms(static_cast(term.size()), term.data(), term_parity.data(), + n_groups, gperm.data(), g_start.data(), ncell); + for (int parity : {0, 1, -1}) { + const std::vector &sel = subset(parity); + const int n = static_cast(sel.size()); + const int nb = ReductionBlocks(n, SURFACE_BLOCK); + std::vector perm(n), seg_start(static_cast(nb) * ncell + 1, n); + ParallelFor(nb, nt, [&](int b) { + const int lo = static_cast(static_cast(n) * b / nb); + const int hi = static_cast(static_cast(n) * (b + 1) / nb); + std::vector pos(ncell + 1, 0); + for (int k = lo; k < hi; ++k) ++pos[term[sel[k]].cell + 1]; + for (int c = 0; c < ncell; ++c) pos[c + 1] += pos[c]; + for (int c = 0; c < ncell; ++c) seg_start[static_cast(b) * ncell + c] = lo + pos[c]; + for (int k = lo; k < hi; ++k) perm[lo + pos[term[sel[k]].cell]++] = sel[k]; + }); + gpu_->SurfaceSetSubset(parity < 0 ? 2 : parity, nb, perm.data(), seg_start.data()); + } + } +#endif + if (!on_gpu) { + gterm.resize(term.size()); std::vector fill(g_start.begin(), g_start.end() - 1); for (size_t k = 0; k < term.size(); ++k) { const Term &t = term[k]; @@ -3488,6 +3527,12 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector &cell, int // and the score. std::vector sw(n_groups), swI(n_groups); auto reference = [&](int parity, const std::vector &A) { +#ifdef JFJOCH_USE_CUDA + if (on_gpu) { + gpu_->SurfaceReference(parity, A.data()); + return; + } +#endif ParallelChunks(n_groups, nt, [&](int glo, int ghi) { for (int g = glo; g < ghi; ++g) { double s_w = 0.0, s_wI = 0.0; @@ -3520,7 +3565,7 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector &cell, int const std::vector &sel = subset(parity); std::vector A(ncell, 1.0); // Per-block cell accumulators, allocated once for the whole fit rather than per round. - const int nb = ReductionBlocks(static_cast(sel.size()), SURFACE_BLOCK); + const int nb = on_gpu ? 0 : ReductionBlocks(static_cast(sel.size()), SURFACE_BLOCK); std::vector> tcross(nb, std::vector(ncell)), tref2(nb, std::vector(ncell)); settled = false; n_clamped = 0; @@ -3555,22 +3600,28 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector &cell, int // there is no ordering that keeps threads off each other's bins, and there are only ncell // of them, so per-block copies are cheap and the fixed blocks keep it the same at any -N. std::vector cross(ncell, 0.0), ref2(ncell, 0.0); - ParallelBlocks(static_cast(sel.size()), nt, [&](int b, int lo, int hi) { - std::vector &xcross = tcross[b], &xref2 = tref2[b]; - std::fill(xcross.begin(), xcross.end(), 0.0); - std::fill(xref2.begin(), xref2.end(), 0.0); - for (int k = lo; k < hi; ++k) { - const Term &o = term[sel[k]]; - if (sw[o.group] <= 0.0) continue; - const double Iref = swI[o.group] / sw[o.group], a = A[o.cell]; - const double Is = static_cast(o.I) * o.corr * a, sc = static_cast(o.sigma) * o.corr * a; - if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue; - const double w = 1.0 / (sc * sc); - xcross[o.cell] += w * Is * Iref; xref2[o.cell] += w * Iref * Iref; - } - }, SURFACE_BLOCK); - for (int b = 0; b < nb; ++b) - for (int c = 0; c < ncell; ++c) { cross[c] += tcross[b][c]; ref2[c] += tref2[b][c]; } +#ifdef JFJOCH_USE_CUDA + if (on_gpu) + gpu_->SurfaceFitSums(parity < 0 ? 2 : parity, cross.data(), ref2.data()); +#endif + if (!on_gpu) { + ParallelBlocks(static_cast(sel.size()), nt, [&](int b, int lo, int hi) { + std::vector &xcross = tcross[b], &xref2 = tref2[b]; + std::fill(xcross.begin(), xcross.end(), 0.0); + std::fill(xref2.begin(), xref2.end(), 0.0); + for (int k = lo; k < hi; ++k) { + const Term &o = term[sel[k]]; + if (sw[o.group] <= 0.0) continue; + const double Iref = swI[o.group] / sw[o.group], a = A[o.cell]; + const double Is = static_cast(o.I) * o.corr * a, sc = static_cast(o.sigma) * o.corr * a; + if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue; + const double w = 1.0 / (sc * sc); + xcross[o.cell] += w * Is * Iref; xref2[o.cell] += w * Iref * Iref; + } + }, SURFACE_BLOCK); + for (int b = 0; b < nb; ++b) + for (int c = 0; c < ncell; ++c) { cross[c] += tcross[b][c]; ref2[c] += tref2[b][c]; } + } std::vector dsorted = cross; std::nth_element(dsorted.begin(), dsorted.begin() + dsorted.size() / 2, dsorted.end()); const double lambda = 0.1 * std::max(1e-30, dsorted[dsorted.size() / 2]); @@ -3672,20 +3723,33 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector &cell, int const int nsh_cc = nshell > 0 ? nshell : 1; auto half_means = [&](int parity, const std::vector &A) { reference(parity, A); +#ifdef JFJOCH_USE_CUDA + if (on_gpu) + gpu_->SurfaceGetReference(sw.data(), swI.data()); +#endif std::vector m(n_groups, std::numeric_limits::quiet_NaN()); for (int g = 0; g < n_groups; ++g) if (sw[g] > 0.0) m[g] = swI[g] / sw[g]; return m; }; - auto shell_cc = [&](const std::vector &x, const std::vector &y, int s) { - double n = 0, sx = 0, sy = 0, sxx = 0, syy = 0, sxy = 0; + // The correlation within every shell, in one walk over the groups: each shell still sums its own + // groups in group order. + auto shell_cc = [&](const std::vector &x, const std::vector &y) { + struct Sums { double n = 0, sx = 0, sy = 0, sxx = 0, syy = 0, sxy = 0; }; + std::vector sum(nsh_cc); for (int g = 0; g < n_groups; ++g) { - if (group_shell[g] != s || !std::isfinite(x[g]) || !std::isfinite(y[g])) continue; - n += 1; sx += x[g]; sy += y[g]; sxx += x[g] * x[g]; syy += y[g] * y[g]; sxy += x[g] * y[g]; + if (!std::isfinite(x[g]) || !std::isfinite(y[g])) continue; + Sums &u = sum[group_shell[g]]; + u.n += 1; u.sx += x[g]; u.sy += y[g]; u.sxx += x[g] * x[g]; u.syy += y[g] * y[g]; u.sxy += x[g] * y[g]; } - const double vx = sxx - sx * sx / n, vy = syy - sy * sy / n; - return (n >= 50 && vx > 0.0 && vy > 0.0) ? (sxy - sx * sy / n) / std::sqrt(vx * vy) - : std::numeric_limits::quiet_NaN(); + std::vector cc(nsh_cc, std::numeric_limits::quiet_NaN()); + for (int s = 0; s < nsh_cc; ++s) { + const Sums &u = sum[s]; + const double vx = u.sxx - u.sx * u.sx / u.n, vy = u.syy - u.sy * u.sy / u.n; + if (u.n >= 50 && vx > 0.0 && vy > 0.0) + cc[s] = (u.sxy - u.sx * u.sy / u.n) / std::sqrt(vx * vy); + } + return cc; }; const std::vector ident(ncell, 1.0); const std::vector A_even = fit_surface(0), A_odd = fit_surface(1); @@ -3702,8 +3766,9 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector &cell, int // Following Fisher (1915) Biometrika 10, 507-521 double gain = 0.0; int n_cc = 0; + const std::vector cc0 = shell_cc(odd0, even0), cc1 = shell_cc(odd1, even1); for (int sh = 0; sh < nsh_cc; ++sh) { - const double c0 = shell_cc(odd0, even0, sh), c1 = shell_cc(odd1, even1, sh); + const double c0 = cc0[sh], c1 = cc1[sh]; if (std::isfinite(c0) && std::isfinite(c1)) { gain += std::atanh(c1) - std::atanh(c0); ++n_cc; } } if (n_cc > 0) gain /= n_cc; @@ -5699,7 +5764,8 @@ RotationScaleMerge::Result RotationScaleMerge::Run(bool for_search, bool full_st ReducePartialGroupMeans(n_groups, partial_mean); ComputePerFrameCC(partial_mean, cc, cc_n); } - FinalizePerFrameScale(cc, cc_n, partial_scaled); + if (write_back_per_frame_scale) + FinalizePerFrameScale(cc, cc_n, partial_scaled); // The filters below remove observations by zeroing corr, which is what takes an observation out of // the 3D combine, the merge and the error model alike (excluding them from the ASU grouping is NOT diff --git a/image_analysis/scale_merge/RotationScaleMerge.h b/image_analysis/scale_merge/RotationScaleMerge.h index ea1947c8f..6b5a81216 100644 --- a/image_analysis/scale_merge/RotationScaleMerge.h +++ b/image_analysis/scale_merge/RotationScaleMerge.h @@ -119,6 +119,11 @@ public: // compares nothing, asks for it to be left out. Result Run(bool for_search, bool full_stats, bool measure_cc_before_corrections); + // Whether Run() writes the per-frame G / CC / mosaicity back onto the outcomes (on by default). Off + // for a merge that is not the run's answer - the P1 cross-check - so the per-image table and the + // unmerged MTZ describe the merge that was written. + void SetWriteBackPerFrameScale(bool on) { write_back_per_frame_scale = on; } + // Override the high-resolution cut for the next Run() - used to gate the de-novo P1 search pass at // >= 1 without cutting the final in-symmetry merge. Reset to the manual limit afterwards. void SetDMinLimit(std::optional d_min_A) { d_min_limit = d_min_A; } @@ -413,6 +418,7 @@ private: std::unique_ptr gpu_; bool gpu_active_ = false; #endif + bool write_back_per_frame_scale = true; // see SetWriteBackPerFrameScale // --- helpers (each a flat pass; see the .cpp) --- // Turn the per-frame mean background under the reflections (accumulated by the ingest fill loop) into diff --git a/image_analysis/scale_merge/RotationScaleMergeGPU.cu b/image_analysis/scale_merge/RotationScaleMergeGPU.cu index 8ed922149..8a0e97f85 100644 --- a/image_analysis/scale_merge/RotationScaleMergeGPU.cu +++ b/image_analysis/scale_merge/RotationScaleMergeGPU.cu @@ -18,13 +18,15 @@ namespace { constexpr int BLK = 256; constexpr int MIN_REFLECTIONS = 20; - // Every kernel and copy here is queued on the legacy NULL stream, so - as in BeamCenterFFTGPU - - // no buffer comes from the pool. A pooled buffer is freed with cudaFreeAsync on the thread's - // non-blocking allocation stream, which is not ordered after the NULL stream. Each entry point - // below waits for its own work before it returns, so no free here has yet overtaken a read, but - // that holds only by that convention, and not at all for a free while unwinding from a failed - // call; nor can compute-sanitizer --track-stream-ordered-races see it, and it reported every - // reassigned merge buffer as a use-after-free. cudaFree synchronises the device first. + // Every kernel and copy here is queued on the instance's own stream (Impl::stream), not on the + // legacy NULL stream: the merge and the image analysis of a probe pass beside it would otherwise + // wait for each other's work at every launch and every synchronisation. As in BeamCenterFFTGPU no buffer comes from the pool. A + // pooled buffer is freed with cudaFreeAsync on the thread's allocation stream, which is not + // ordered after this one. Each entry point below waits for its own work before it returns, so no + // free here has yet overtaken a read, but that holds only by that convention, and not at all for a + // free while unwinding from a failed call; nor can compute-sanitizer --track-stream-ordered-races + // see it, and it reported every reassigned merge buffer as a use-after-free. cudaFree + // synchronises the device first. constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous; __device__ __forceinline__ double SafeInvD(double x, double fallback) { @@ -636,6 +638,110 @@ namespace { } } + // --- correction-surface fit (ApplyCellSurface) --- + // The host fit is the reference here: these kernels form the same sums from the same terms in the + // same order, and every rounding is spelled out (__dmul_rn / __dadd_rn round each step on its own, + // fma rounds once) to be the one the host build makes - nvcc would otherwise contract a multiply + // and an add wherever it sees them, and GCC at -march=x86-64-v3 does so only in some of them. The + // products are the host's (I * corr) * a and (sigma * corr) * a. + using SurfaceTerm = RotationScaleMergeGPU::SurfaceTerm; + + __device__ __forceinline__ double SurfaceIs(const SurfaceTerm &t, double a) { + return __dmul_rn(__dmul_rn(double(t.I), double(t.corr)), a); + } + + __device__ __forceinline__ double SurfaceSigma(const SurfaceTerm &t, double a) { + return __dmul_rn(__dmul_rn(double(t.sigma), double(t.corr)), a); + } + + // One thread per ASU group: the group's inverse-variance sums over its terms of frame parity `parity` + // (< 0 = all), in fulls order - the host's `reference`. The host builds that loop twice, and only + // the parity-filtered copy fuses the multiply-add of swI; the unfiltered one rounds the product. + __global__ void SurfaceReferenceKernel(int n_groups, int parity, const int32_t *__restrict__ gperm, + const int32_t *__restrict__ gstart, + const SurfaceTerm *__restrict__ term, + const uint8_t *__restrict__ term_parity, + const double *__restrict__ A, + double *__restrict__ sw, double *__restrict__ swI) { + for (int g = blockIdx.x * blockDim.x + threadIdx.x; g < n_groups; g += gridDim.x * blockDim.x) { + double s_w = 0.0, s_wI = 0.0; + for (int k = gstart[g]; k < gstart[g + 1]; ++k) { + const int i = gperm[k]; + if (parity >= 0 && term_parity[i] != parity) continue; + const SurfaceTerm t = term[i]; + const double a = A[t.cell]; + const double Is = SurfaceIs(t, a), sc = SurfaceSigma(t, a); + const double w = 1.0 / __dmul_rn(sc, sc); + s_w = __dadd_rn(s_w, w); + s_wI = parity >= 0 ? fma(Is, w, s_wI) : __dadd_rn(s_wI, __dmul_rn(Is, w)); + } + sw[g] = s_w; swI[g] = s_wI; + } + } + + // The fit's sums of one round. Each term's contribution is formed on its own thread (one per term, + // in segment order), and only the two multiply-adds that sum them are left to the thread of each + // (reduction block, cell) - the per-block accumulators of the host fit, one slot at a time, walking + // its segment in term order. A term the host skips is marked by a NaN Iref (a kept one is finite). + __global__ void SurfaceFitTermKernel(int n, const int32_t *__restrict__ perm, + const SurfaceTerm *__restrict__ term, + const double *__restrict__ A, + const double *__restrict__ sw, const double *__restrict__ swI, + double *__restrict__ w_Is, double *__restrict__ w_Iref, + double *__restrict__ Iref_out) { + for (int k = blockIdx.x * blockDim.x + threadIdx.x; k < n; k += gridDim.x * blockDim.x) { + const SurfaceTerm t = term[perm[k]]; + Iref_out[k] = NAN; + if (sw[t.group] <= 0.0) continue; + const double Iref = swI[t.group] / sw[t.group], a = A[t.cell]; + const double Is = SurfaceIs(t, a), sc = SurfaceSigma(t, a); + if (!isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue; + const double w = 1.0 / __dmul_rn(sc, sc); + w_Is[k] = __dmul_rn(w, Is); + w_Iref[k] = __dmul_rn(w, Iref); + Iref_out[k] = Iref; + } + } + + __global__ void SurfaceFitSegmentKernel(int n_seg, const int32_t *__restrict__ seg_start, + const double *__restrict__ w_Is, const double *__restrict__ w_Iref, + const double *__restrict__ Iref, + double *__restrict__ tcross, double *__restrict__ tref2) { + for (int s = blockIdx.x * blockDim.x + threadIdx.x; s < n_seg; s += gridDim.x * blockDim.x) { + double xcross = 0.0, xref2 = 0.0; + for (int k = seg_start[s]; k < seg_start[s + 1]; ++k) { + if (isnan(Iref[k])) continue; + xcross = fma(w_Is[k], Iref[k], xcross); + xref2 = fma(w_Iref[k], Iref[k], xref2); + } + tcross[s] = xcross; tref2[s] = xref2; + } + } + + // One thread per cell: the blocks' sums added up in block order, as the host adds its slots. + __global__ void SurfaceFitCellKernel(int n_blocks, int ncell, const double *__restrict__ tcross, + const double *__restrict__ tref2, + double *__restrict__ cross, double *__restrict__ ref2) { + for (int c = blockIdx.x * blockDim.x + threadIdx.x; c < ncell; c += gridDim.x * blockDim.x) { + double sc = 0.0, sr = 0.0; + for (int b = 0; b < n_blocks; ++b) { + sc = __dadd_rn(sc, tcross[b * ncell + c]); + sr = __dadd_rn(sr, tref2[b * ncell + c]); + } + cross[c] = sc; ref2[c] = sr; + } + } + + void CudaCheck(cudaError_t e, const char *what); + + // A copy on the instance's stream, waited for - what cudaMemcpy on the NULL stream was, without also + // waiting for every other stream on the card. + void CopyAndWait(void *dst, const void *src, size_t bytes, cudaMemcpyKind kind, cudaStream_t s, + const char *what) { + CudaCheck(cudaMemcpyAsync(dst, src, bytes, kind, s), what); + CudaCheck(cudaStreamSynchronize(s), what); + } + void CudaCheck(cudaError_t e, const char *what) { if (e != cudaSuccess) throw JFJochException(JFJochExceptionCategory::GPUCUDAError, @@ -703,6 +809,9 @@ namespace { } struct RotationScaleMergeGPU::Impl { + // First, so it goes last: the buffers below are freed before the stream their work ran on. + std::unique_ptr stream; + cudaStream_t s() const { return stream->get(); } int device = 0; // the GPU this instance's buffers live on bool available = false; int n_obs = 0, n_frames = 0, n_groups = 0; @@ -736,7 +845,7 @@ struct RotationScaleMergeGPU::Impl { void Upload(CudaDevicePtr &dst, const T *src, int n) const { dst = Alloc(std::max(1, n)); if (n > 0) - CudaCheck(cudaMemcpy(dst.get(), src, size_t(n) * sizeof(T), cudaMemcpyHostToDevice), "upload"); + CopyAndWait(dst.get(), src, size_t(n) * sizeof(T), cudaMemcpyHostToDevice, s(), "upload"); } // immutable per-obs @@ -801,6 +910,18 @@ struct RotationScaleMergeGPU::Impl { CudaDevicePtr f_sco_ok; CudaDevicePtr f_frame_perm, f_frame_start, f_frame_count; CudaDevicePtr f_gperm, f_gstart, f_gcount; + // correction-surface fit (one ApplyCellSurface call at a time): its terms and their group CSR, the + // three subsets' per-(block, cell) segments, the surface and the sums of the round + int s_ncell = 0, s_n_groups = 0; + int s_n_blocks[3] = {0, 0, 0}; + int s_n_sel[3] = {0, 0, 0}; + size_t s_slots = 0; // capacity of s_tcross / s_tref2 + CudaDevicePtr s_term; + CudaDevicePtr s_parity; + CudaDevicePtr s_gperm, s_gstart; + CudaDevicePtr s_perm[3], s_seg_start[3]; + CudaDevicePtr s_A, s_sw, s_swI, s_tcross, s_tref2, s_cross, s_ref2; + CudaDevicePtr s_w_Is, s_w_Iref, s_Iref; // per term of a subset, in segment order }; // Set the device this instance's memory lives on for the duration of a call, and put the caller's @@ -833,6 +954,7 @@ RotationScaleMergeGPU::RotationScaleMergeGPU() : impl_(std::make_unique()) // can put the caller's device back instead of leaving the thread moved. impl_->device = 0; DeviceGuard guard(impl_->device, true); + impl_->stream = std::make_unique(); impl_->available = true; } } @@ -878,10 +1000,10 @@ void RotationScaleMergeGPU::SetPartialsLayout(int n_obs, int n_frames, namespace { template - void UploadChunk(CudaDevicePtr &dst, int offset, int count, const T *v) { + void UploadChunk(CudaDevicePtr &dst, int offset, int count, const T *v, cudaStream_t s) { if (count > 0) - CudaCheck(cudaMemcpy(dst.get() + offset, v, size_t(count) * sizeof(T), - cudaMemcpyHostToDevice), "upload chunk"); + CopyAndWait(dst.get() + offset, v, size_t(count) * sizeof(T), cudaMemcpyHostToDevice, s, + "upload chunk"); } } @@ -889,34 +1011,34 @@ void RotationScaleMergeGPU::SetObsField(ObsField f, int offset, int count, const DeviceGuard guard(impl_->device, impl_->available); auto &d = *impl_; switch (f) { - case ObsField::I: UploadChunk(d.I, offset, count, v); break; - case ObsField::Sigma: UploadChunk(d.sigma, offset, count, v); break; - case ObsField::PrescalingCorr: UploadChunk(d.prescaling_corr, offset, count, v); break; - case ObsField::Partiality: UploadChunk(d.partiality, offset, count, v); break; - case ObsField::Zeta: UploadChunk(d.zeta, offset, count, v); break; - case ObsField::Corr0: UploadChunk(d.corr, offset, count, v); break; - case ObsField::Bkg: UploadChunk(d.bkg, offset, count, v); break; - case ObsField::VarBkg: UploadChunk(d.var_bkg, offset, count, v); break; - case ObsField::ImageNumber: UploadChunk(d.image_number, offset, count, v); break; - case ObsField::D: UploadChunk(d.d_obs, offset, count, v); break; - case ObsField::Px: UploadChunk(d.px_obs, offset, count, v); break; - case ObsField::Py: UploadChunk(d.py_obs, offset, count, v); break; + case ObsField::I: UploadChunk(d.I, offset, count, v, impl_->s()); break; + case ObsField::Sigma: UploadChunk(d.sigma, offset, count, v, impl_->s()); break; + case ObsField::PrescalingCorr: UploadChunk(d.prescaling_corr, offset, count, v, impl_->s()); break; + case ObsField::Partiality: UploadChunk(d.partiality, offset, count, v, impl_->s()); break; + case ObsField::Zeta: UploadChunk(d.zeta, offset, count, v, impl_->s()); break; + case ObsField::Corr0: UploadChunk(d.corr, offset, count, v, impl_->s()); break; + case ObsField::Bkg: UploadChunk(d.bkg, offset, count, v, impl_->s()); break; + case ObsField::VarBkg: UploadChunk(d.var_bkg, offset, count, v, impl_->s()); break; + case ObsField::ImageNumber: UploadChunk(d.image_number, offset, count, v, impl_->s()); break; + case ObsField::D: UploadChunk(d.d_obs, offset, count, v, impl_->s()); break; + case ObsField::Px: UploadChunk(d.px_obs, offset, count, v, impl_->s()); break; + case ObsField::Py: UploadChunk(d.py_obs, offset, count, v, impl_->s()); break; } } void RotationScaleMergeGPU::SetObsFrame(int offset, int count, const int32_t *frame) { DeviceGuard guard(impl_->device, impl_->available); - UploadChunk(impl_->frame, offset, count, frame); + UploadChunk(impl_->frame, offset, count, frame, impl_->s()); } void RotationScaleMergeGPU::SetObsOnIce(int offset, int count, const uint8_t *on_ice) { DeviceGuard guard(impl_->device, impl_->available); - UploadChunk(impl_->on_ice, offset, count, on_ice); + UploadChunk(impl_->on_ice, offset, count, on_ice, impl_->s()); } void RotationScaleMergeGPU::SetObsClipped(int offset, int count, const uint8_t *clipped) { DeviceGuard guard(impl_->device, impl_->available); - UploadChunk(impl_->clipped, offset, count, clipped); + UploadChunk(impl_->clipped, offset, count, clipped, impl_->s()); } void RotationScaleMergeGPU::SetGroups(int n_groups, const int32_t *group, const int32_t *group_perm, @@ -934,8 +1056,8 @@ void RotationScaleMergeGPU::SetGroups(int n_groups, const int32_t *group, const void RotationScaleMergeGPU::SetCorr(const float *corr) { DeviceGuard guard(impl_->device, impl_->available); - CudaCheck(cudaMemcpy(impl_->corr.get(), corr, size_t(impl_->n_obs) * sizeof(float), - cudaMemcpyHostToDevice), "upload corr"); + CopyAndWait(impl_->corr.get(), corr, size_t(impl_->n_obs) * sizeof(float), + cudaMemcpyHostToDevice, impl_->s(), "upload corr"); } void RotationScaleMergeGPU::ScalePartials(int iters, double min_partiality, bool /*has_d_min*/) { @@ -943,42 +1065,42 @@ void RotationScaleMergeGPU::ScalePartials(int iters, double min_partiality, bool auto &d = *impl_; // Reset per call: the host keeps the G of a frame across calls (RunScalingLoop), so a frame this // call did not fit must read as unfitted, not as fitted with the value of the call before. - CudaCheck(cudaMemset(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t)), "memset scaled"); - CudaCheck(cudaMemset(d.g.get(), 0, size_t(d.n_frames) * sizeof(double)), "memset g"); // unscaled g unused + CudaCheck(cudaMemsetAsync(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t), impl_->s()), "memset scaled"); + CudaCheck(cudaMemsetAsync(d.g.get(), 0, size_t(d.n_frames) * sizeof(double), impl_->s()), "memset g"); // unscaled g unused const int obs_blocks = (d.n_obs + BLK - 1) / BLK; const int upd_blocks = std::min(65535, obs_blocks); const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK); for (int it = 0; it < iters; ++it) { - ReduceGroupMeansKernel<<>>(d.n_groups, min_partiality, + ReduceGroupMeansKernel<<s()>>>(d.n_groups, min_partiality, d.group_perm.get(), d.group_start.get(), d.group_count.get(), d.I.get(), d.sigma.get(), d.partiality.get(), d.corr.get(), d.group_mean.get()); CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch"); - PrepScaleObsKernel<<>>(d.n_obs, min_partiality, d.group.get(), d.partiality.get(), d.prescaling_corr.get(), + PrepScaleObsKernel<<s()>>>(d.n_obs, min_partiality, d.group.get(), d.partiality.get(), d.prescaling_corr.get(), d.zeta.get(), d.on_ice.get(), d.group_mean.get(), d.sigma.get(), d.inv_sigma.get(), d.sco_coeff.get(), d.sco_ok.get()); CudaCheck(cudaGetLastError(), "PrepScaleObsKernel launch"); - FitPerFrameGKernel<<>>(d.n_frames, d.frame_start.get(), d.frame_count.get(), + FitPerFrameGKernel<<s()>>>(d.n_frames, d.frame_start.get(), d.frame_count.get(), d.I.get(), d.inv_sigma.get(), d.sco_coeff.get(), d.sco_ok.get(), nullptr, d.g.get(), d.scaled.get()); CudaCheck(cudaGetLastError(), "FitPerFrameGKernel launch"); - UpdateCorrKernel<<>>(d.n_obs, d.frame.get(), d.prescaling_corr.get(), d.partiality.get(), + UpdateCorrKernel<<s()>>>(d.n_obs, d.frame.get(), d.prescaling_corr.get(), d.partiality.get(), d.g.get(), d.scaled.get(), d.corr.get()); } CudaCheck(cudaGetLastError(), "kernel launch"); - CudaCheck(cudaDeviceSynchronize(), "scale sync"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "scale sync"); } void RotationScaleMergeGPU::GetCorr(float *corr_out) const { DeviceGuard guard(impl_->device, impl_->available); - CudaCheck(cudaMemcpy(corr_out, impl_->corr.get(), size_t(impl_->n_obs) * sizeof(float), - cudaMemcpyDeviceToHost), "download corr"); + CopyAndWait(corr_out, impl_->corr.get(), size_t(impl_->n_obs) * sizeof(float), + cudaMemcpyDeviceToHost, impl_->s(), "download corr"); } void RotationScaleMergeGPU::GetG(double *g_out, uint8_t *scaled_out) const { DeviceGuard guard(impl_->device, impl_->available); - CudaCheck(cudaMemcpy(g_out, impl_->g.get(), size_t(impl_->n_frames) * sizeof(double), - cudaMemcpyDeviceToHost), "download g"); - CudaCheck(cudaMemcpy(scaled_out, impl_->scaled.get(), size_t(impl_->n_frames) * sizeof(uint8_t), - cudaMemcpyDeviceToHost), "download scaled"); + CopyAndWait(g_out, impl_->g.get(), size_t(impl_->n_frames) * sizeof(double), + cudaMemcpyDeviceToHost, impl_->s(), "download g"); + CopyAndWait(scaled_out, impl_->scaled.get(), size_t(impl_->n_frames) * sizeof(uint8_t), + cudaMemcpyDeviceToHost, impl_->s(), "download scaled"); } void RotationScaleMergeGPU::SetFrameCellOk(const uint8_t *frame_cell_ok) { @@ -1030,19 +1152,19 @@ void RotationScaleMergeGPU::MergeEmSamples(bool for_search, double min_partialit const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK); const int obs_blocks = std::min(65535, (nf + BLK - 1) / BLK); - MergeEmStatsKernel<<>>(p); - MergeSamplesKernel<<>>(nf, p); + MergeEmStatsKernel<<s()>>>(p); + MergeSamplesKernel<<s()>>>(nf, p); CudaCheck(cudaGetLastError(), "merge em/samples launch"); - CudaCheck(cudaDeviceSynchronize(), "merge em/samples sync"); - CudaCheck(cudaMemcpy(em_mean_out, d.m_em_mean.get(), size_t(ng) * sizeof(double), - cudaMemcpyDeviceToHost), "dl em_mean"); - CudaCheck(cudaMemcpy(cnt_out, d.m_cnt.get(), size_t(ng) * sizeof(int32_t), - cudaMemcpyDeviceToHost), "dl cnt"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "merge em/samples sync"); + CopyAndWait(em_mean_out, d.m_em_mean.get(), size_t(ng) * sizeof(double), + cudaMemcpyDeviceToHost, impl_->s(), "dl em_mean"); + CopyAndWait(cnt_out, d.m_cnt.get(), size_t(ng) * sizeof(int32_t), + cudaMemcpyDeviceToHost, impl_->s(), "dl cnt"); if (nf > 0) { - CudaCheck(cudaMemcpy(s2_out, d.m_s2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl s2"); - CudaCheck(cudaMemcpy(I2_out, d.m_I2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl I2"); - CudaCheck(cudaMemcpy(dev2_out, d.m_dev2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl dev2"); - CudaCheck(cudaMemcpy(valid_out, d.m_valid.get(), size_t(nf) * sizeof(uint8_t), cudaMemcpyDeviceToHost), "dl valid"); + CopyAndWait(s2_out, d.m_s2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl s2"); + CopyAndWait(I2_out, d.m_I2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl I2"); + CopyAndWait(dev2_out, d.m_dev2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl dev2"); + CopyAndWait(valid_out, d.m_valid.get(), size_t(nf) * sizeof(uint8_t), cudaMemcpyDeviceToHost, impl_->s(), "dl valid"); } } @@ -1091,10 +1213,10 @@ void RotationScaleMergeGPU::MergeAccum(double error_model_a, double error_model_ p.rejected_obs = d.m_rejected.get(); const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK); - MergeAccumKernel<<>>(p); + MergeAccumKernel<<s()>>>(p); CudaCheck(cudaGetLastError(), "merge accum launch"); - CudaCheck(cudaDeviceSynchronize(), "merge accum sync"); - CudaCheck(cudaMemcpy(rejected_obs, d.m_rejected.get(), size_t(d.n_fulls) * sizeof(uint8_t), cudaMemcpyDeviceToHost), + CudaCheck(cudaStreamSynchronize(impl_->s()), "merge accum sync"); + CopyAndWait(rejected_obs, d.m_rejected.get(), size_t(d.n_fulls) * sizeof(uint8_t), cudaMemcpyDeviceToHost, impl_->s(), "dl rejected_obs"); } @@ -1105,7 +1227,7 @@ void RotationScaleMergeGPU::MergeAccumRange(int g0, int n, double *swI, double * DeviceGuard guard(impl_->device, impl_->available); auto &d = *impl_; auto dl = [&](void *h, const auto &s) { - CudaCheck(cudaMemcpy(h, s.get() + g0, size_t(n) * sizeof(*s.get()), cudaMemcpyDeviceToHost), + CopyAndWait(h, s.get() + g0, size_t(n) * sizeof(*s.get()), cudaMemcpyDeviceToHost, impl_->s(), "dl accum"); }; dl(swI, d.a_swI); dl(sw, d.a_sw); dl(swIh0, d.a_swIh0); dl(swIh1, d.a_swIh1); dl(swh0, d.a_swh0); dl(swh1, d.a_swh1); dl(swh_typ0, d.a_swht0); dl(swh_typ1, d.a_swht1); @@ -1139,17 +1261,17 @@ void RotationScaleMergeGPU::MergeRmeas(const double *merged_I, double *absdev, d p.rejected_obs = d.m_rejected.get(); const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK); - MergeRmeasKernel<<>>(p); + MergeRmeasKernel<<s()>>>(p); CudaCheck(cudaGetLastError(), "merge rmeas launch"); - CudaCheck(cudaDeviceSynchronize(), "merge rmeas sync"); - CudaCheck(cudaMemcpy(absdev, d.r_absdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl absdev"); - CudaCheck(cudaMemcpy(sumI, d.r_sumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumI"); - CudaCheck(cudaMemcpy(wabsdev, d.r_wabsdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl wabsdev"); - CudaCheck(cudaMemcpy(wsumI, d.r_wsumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl wsumI"); - CudaCheck(cudaMemcpy(sumv, d.r_sumv.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumv"); - CudaCheck(cudaMemcpy(sumv2, d.r_sumv2.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumv2"); - CudaCheck(cudaMemcpy(n, d.r_n.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost), "dl rn"); - CudaCheck(cudaMemcpy(nusable, d.r_nusable.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost), "dl rnusable"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "merge rmeas sync"); + CopyAndWait(absdev, d.r_absdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl absdev"); + CopyAndWait(sumI, d.r_sumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumI"); + CopyAndWait(wabsdev, d.r_wabsdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl wabsdev"); + CopyAndWait(wsumI, d.r_wsumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl wsumI"); + CopyAndWait(sumv, d.r_sumv.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumv"); + CopyAndWait(sumv2, d.r_sumv2.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumv2"); + CopyAndWait(n, d.r_n.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost, impl_->s(), "dl rn"); + CopyAndWait(nusable, d.r_nusable.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost, impl_->s(), "dl rnusable"); } void RotationScaleMergeGPU::SmoothCorr(const uint8_t *apply, const double *ratio) { @@ -1158,10 +1280,10 @@ void RotationScaleMergeGPU::SmoothCorr(const uint8_t *apply, const double *ratio d.Upload(d.smooth_apply, apply, d.n_frames); d.Upload(d.smooth_ratio, ratio, d.n_frames); const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK); - SmoothCorrKernel<<>>(d.n_obs, d.frame.get(), d.smooth_apply.get(), + SmoothCorrKernel<<s()>>>(d.n_obs, d.frame.get(), d.smooth_apply.get(), d.smooth_ratio.get(), d.corr.get()); CudaCheck(cudaGetLastError(), "smooth corr launch"); - CudaCheck(cudaDeviceSynchronize(), "smooth corr sync"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "smooth corr sync"); } void RotationScaleMergeGPU::SmoothFullsCorr(const uint8_t *apply, const double *ratio) { @@ -1171,23 +1293,23 @@ void RotationScaleMergeGPU::SmoothFullsCorr(const uint8_t *apply, const double * d.Upload(d.smooth_apply, apply, d.n_frames); d.Upload(d.smooth_ratio, ratio, d.n_frames); const int blocks = std::min(65535, (d.n_fulls + BLK - 1) / BLK); - SmoothCorrKernel<<>>(d.n_fulls, d.f_frame.get(), d.smooth_apply.get(), + SmoothCorrKernel<<s()>>>(d.n_fulls, d.f_frame.get(), d.smooth_apply.get(), d.smooth_ratio.get(), d.f_corr.get()); CudaCheck(cudaGetLastError(), "smooth fulls corr launch"); - CudaCheck(cudaDeviceSynchronize(), "smooth fulls corr sync"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "smooth fulls corr sync"); } int64_t RotationScaleMergeGPU::FilterCorrByZeta(double min_zeta) { DeviceGuard guard(impl_->device, impl_->available); auto &d = *impl_; CudaDevicePtr dropped = d.Alloc(1); - CudaCheck(cudaMemset(dropped.get(), 0, sizeof(unsigned long long)), "zero zeta drop count"); + CudaCheck(cudaMemsetAsync(dropped.get(), 0, sizeof(unsigned long long), impl_->s()), "zero zeta drop count"); const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK); - FilterZetaKernel<<>>(d.n_obs, min_zeta, d.zeta.get(), d.corr.get(), dropped.get()); + FilterZetaKernel<<s()>>>(d.n_obs, min_zeta, d.zeta.get(), d.corr.get(), dropped.get()); CudaCheck(cudaGetLastError(), "zeta filter launch"); - CudaCheck(cudaDeviceSynchronize(), "zeta filter sync"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "zeta filter sync"); unsigned long long n = 0; - CudaCheck(cudaMemcpy(&n, dropped.get(), sizeof(unsigned long long), cudaMemcpyDeviceToHost), + CopyAndWait(&n, dropped.get(), sizeof(unsigned long long), cudaMemcpyDeviceToHost, impl_->s(), "dl zeta drop count"); return static_cast(n); } @@ -1197,9 +1319,9 @@ void RotationScaleMergeGPU::FilterCorrByFrame(const uint8_t *reject) { auto &d = *impl_; d.Upload(d.filter_reject, reject, d.n_frames); const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK); - FilterFrameKernel<<>>(d.n_obs, d.frame.get(), d.filter_reject.get(), d.corr.get()); + FilterFrameKernel<<s()>>>(d.n_obs, d.frame.get(), d.filter_reject.get(), d.corr.get()); CudaCheck(cudaGetLastError(), "frame filter launch"); - CudaCheck(cudaDeviceSynchronize(), "frame filter sync"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "frame filter sync"); } void RotationScaleMergeGPU::ComputePartialCC(double min_partiality, double *cc_out, int64_t *cc_n_out) { @@ -1208,19 +1330,19 @@ void RotationScaleMergeGPU::ComputePartialCC(double min_partiality, double *cc_o const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK); // Post-smooth group means (reuse the scaling reduce; reads the resident, smoothed corr), then the // per-frame CC over the resident partials. Only the tiny per-frame cc/cc_n come back to the host. - ReduceGroupMeansKernel<<>>(d.n_groups, min_partiality, + ReduceGroupMeansKernel<<s()>>>(d.n_groups, min_partiality, d.group_perm.get(), d.group_start.get(), d.group_count.get(), d.I.get(), d.sigma.get(), d.partiality.get(), d.corr.get(), d.group_mean.get()); CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch"); - PerFrameCCKernel<<>>(d.n_frames, min_partiality, + PerFrameCCKernel<<s()>>>(d.n_frames, min_partiality, d.frame_start.get(), d.frame_count.get(), d.I.get(), d.sigma.get(), d.partiality.get(), d.corr.get(), d.on_ice.get(), d.group.get(), d.group_mean.get(), d.cc.get(), d.cc_n.get()); CudaCheck(cudaGetLastError(), "partial CC launch"); - CudaCheck(cudaDeviceSynchronize(), "partial CC sync"); - CudaCheck(cudaMemcpy(cc_out, d.cc.get(), size_t(d.n_frames) * sizeof(double), - cudaMemcpyDeviceToHost), "download cc"); - CudaCheck(cudaMemcpy(cc_n_out, d.cc_n.get(), size_t(d.n_frames) * sizeof(int64_t), - cudaMemcpyDeviceToHost), "download cc_n"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "partial CC sync"); + CopyAndWait(cc_out, d.cc.get(), size_t(d.n_frames) * sizeof(double), + cudaMemcpyDeviceToHost, impl_->s(), "download cc"); + CopyAndWait(cc_n_out, d.cc_n.get(), size_t(d.n_frames) * sizeof(int64_t), + cudaMemcpyDeviceToHost, impl_->s(), "download cc_n"); } void RotationScaleMergeGPU::SetRawRuns(int n_runs, int n_perm, const int32_t *perm, @@ -1247,8 +1369,8 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti float max_frame_gap) { DeviceGuard guard(impl_->device, impl_->available); auto &d = *impl_; - CudaCheck(cudaMemcpy(d.rr_group.get(), rawrun_group, size_t(d.n_runs) * sizeof(int32_t), - cudaMemcpyHostToDevice), "upload rr_group"); + CopyAndWait(d.rr_group.get(), rawrun_group, size_t(d.n_runs) * sizeof(int32_t), + cudaMemcpyHostToDevice, impl_->s(), "upload rr_group"); CombineParams p{}; p.n_runs = d.n_runs; @@ -1267,13 +1389,13 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti const int blocks = std::min(65535, (d.n_runs + BLK - 1) / BLK); // Count pass: how many fulls each run emits. - CombineKernel<<>>(p); + CombineKernel<<s()>>>(p); CudaCheck(cudaGetLastError(), "combine count launch"); // Exclusive prefix sum on the host (deterministic) -> per-run output offset + total fulls. std::vector nevents(d.n_runs); - CudaCheck(cudaMemcpy(nevents.data(), d.rr_nevents.get(), size_t(d.n_runs) * sizeof(int32_t), - cudaMemcpyDeviceToHost), "download nevents"); + CopyAndWait(nevents.data(), d.rr_nevents.get(), size_t(d.n_runs) * sizeof(int32_t), + cudaMemcpyDeviceToHost, impl_->s(), "download nevents"); std::vector offset(d.n_runs); int64_t acc = 0; for (int r = 0; r < d.n_runs; ++r) { offset[r] = static_cast(acc); acc += nevents[r]; } @@ -1292,8 +1414,8 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti d.f_rlp = d.Alloc(nf); d.f_zeta = d.Alloc(nf); d.f_inv_sigma = d.Alloc(nf); d.f_sco_coeff = d.Alloc(nf); d.f_sco_ok = d.Alloc(nf); - CudaCheck(cudaMemcpy(d.rr_offset.get(), offset.data(), size_t(d.n_runs) * sizeof(int32_t), - cudaMemcpyHostToDevice), "upload offset"); + CopyAndWait(d.rr_offset.get(), offset.data(), size_t(d.n_runs) * sizeof(int32_t), + cudaMemcpyHostToDevice, impl_->s(), "upload offset"); p.rr_offset = d.rr_offset.get(); p.f_h = d.f_h.get(); p.f_k = d.f_k.get(); p.f_l = d.f_l.get(); @@ -1303,10 +1425,10 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti p.f_var_bkg = d.f_var_bkg.get(); p.f_var_per_I = d.f_var_per_I.get(); p.f_on_ice = d.f_on_ice.get(); p.f_clipped = d.f_clipped.get(); if (d.n_fulls > 0) { - CombineKernel<<>>(p); + CombineKernel<<s()>>>(p); CudaCheck(cudaGetLastError(), "combine emit launch"); } - CudaCheck(cudaDeviceSynchronize(), "combine sync"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "combine sync"); return d.n_fulls; } @@ -1318,7 +1440,7 @@ void RotationScaleMergeGPU::GetFulls(int32_t *h, int32_t *k, int32_t *l, float * const size_t n = static_cast(dd.n_fulls); if (n == 0) return; auto dl = [&](void *dst, const void *src, size_t bytes) { - CudaCheck(cudaMemcpy(dst, src, bytes, cudaMemcpyDeviceToHost), "download fulls"); + CopyAndWait(dst, src, bytes, cudaMemcpyDeviceToHost, impl_->s(), "download fulls"); }; dl(h, dd.f_h.get(), n * sizeof(int32_t)); dl(k, dd.f_k.get(), n * sizeof(int32_t)); dl(l, dd.f_l.get(), n * sizeof(int32_t)); dl(frame, dd.f_frame.get(), n * sizeof(int32_t)); @@ -1334,8 +1456,8 @@ void RotationScaleMergeGPU::GetFullsKeys(int32_t *frame, int32_t *group) const { const auto &d = *impl_; if (d.n_fulls == 0) return; const size_t bytes = size_t(d.n_fulls) * sizeof(int32_t); - CudaCheck(cudaMemcpy(frame, d.f_frame.get(), bytes, cudaMemcpyDeviceToHost), "download f_frame"); - CudaCheck(cudaMemcpy(group, d.f_group.get(), bytes, cudaMemcpyDeviceToHost), "download f_group"); + CopyAndWait(frame, d.f_frame.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_frame"); + CopyAndWait(group, d.f_group.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_group"); } void RotationScaleMergeGPU::SetFullsFrameCSR(const int32_t *frame_perm, int n_perm, @@ -1363,15 +1485,15 @@ void RotationScaleMergeGPU::ResetFullsScale() { if (nf == 0) return; const int obs_blocks = std::min(65535, (nf + BLK - 1) / BLK); // Unity model: partiality/prescaling_corr/zeta = 1 so coeff = mean; corr starts at 1. - FillKernel<<>>(d.f_corr.get(), nf, 1.0f); + FillKernel<<s()>>>(d.f_corr.get(), nf, 1.0f); CudaCheck(cudaGetLastError(), "FillKernel launch"); - FillKernel<<>>(d.f_partiality.get(), nf, 1.0f); + FillKernel<<s()>>>(d.f_partiality.get(), nf, 1.0f); CudaCheck(cudaGetLastError(), "FillKernel launch"); - FillKernel<<>>(d.f_rlp.get(), nf, 1.0f); + FillKernel<<s()>>>(d.f_rlp.get(), nf, 1.0f); CudaCheck(cudaGetLastError(), "FillKernel launch"); - FillKernel<<>>(d.f_zeta.get(), nf, 1.0f); + FillKernel<<s()>>>(d.f_zeta.get(), nf, 1.0f); CudaCheck(cudaGetLastError(), "FillKernel launch"); - CudaCheck(cudaDeviceSynchronize(), "reset fulls scale sync"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "reset fulls scale sync"); } void RotationScaleMergeGPU::ScaleFulls(int iters, double min_partiality) { @@ -1383,38 +1505,38 @@ void RotationScaleMergeGPU::ScaleFulls(int iters, double min_partiality) { const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK); // Reset per call, as ScalePartials: the host keeps the G of a frame across calls. - CudaCheck(cudaMemset(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t)), "memset f scaled"); - CudaCheck(cudaMemset(d.g.get(), 0, size_t(d.n_frames) * sizeof(double)), "memset f g"); + CudaCheck(cudaMemsetAsync(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t), impl_->s()), "memset f scaled"); + CudaCheck(cudaMemsetAsync(d.g.get(), 0, size_t(d.n_frames) * sizeof(double), impl_->s()), "memset f g"); for (int it = 0; it < iters; ++it) { - ReduceGroupMeansKernel<<>>(d.n_groups, min_partiality, + ReduceGroupMeansKernel<<s()>>>(d.n_groups, min_partiality, d.f_gperm.get(), d.f_gstart.get(), d.f_gcount.get(), d.f_I.get(), d.f_sigma.get(), d.f_partiality.get(), d.f_corr.get(), d.group_mean.get()); CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch"); // Not grid-stride, so its grid has to cover every full - unlike the grid-stride kernels // below, which the 65535 cap is there for. Capped, it would silently leave the tail of // sco_coeff/sco_ok stale above 16.8M fulls. - PrepScaleObsKernel<<<(nf + BLK - 1) / BLK, BLK>>>(nf, min_partiality, d.f_group.get(), d.f_partiality.get(), + PrepScaleObsKernel<<<(nf + BLK - 1) / BLK, BLK, 0, impl_->s()>>>(nf, min_partiality, d.f_group.get(), d.f_partiality.get(), d.f_rlp.get(), d.f_zeta.get(), d.f_on_ice.get(), d.group_mean.get(), d.f_sigma.get(), d.f_inv_sigma.get(), d.f_sco_coeff.get(), d.f_sco_ok.get()); CudaCheck(cudaGetLastError(), "PrepScaleObsKernel launch"); - FitPerFrameGKernel<<>>(d.n_frames, + FitPerFrameGKernel<<s()>>>(d.n_frames, d.f_frame_start.get(), d.f_frame_count.get(), d.f_I.get(), d.f_inv_sigma.get(), d.f_sco_coeff.get(), d.f_sco_ok.get(), d.f_frame_perm.get(), d.g.get(), d.scaled.get()); CudaCheck(cudaGetLastError(), "FitPerFrameGKernel launch"); - UpdateCorrKernel<<>>(nf, d.f_frame.get(), d.f_rlp.get(), d.f_partiality.get(), + UpdateCorrKernel<<s()>>>(nf, d.f_frame.get(), d.f_rlp.get(), d.f_partiality.get(), d.g.get(), d.scaled.get(), d.f_corr.get()); } CudaCheck(cudaGetLastError(), "scale fulls launch"); - CudaCheck(cudaDeviceSynchronize(), "scale fulls sync"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "scale fulls sync"); } void RotationScaleMergeGPU::GetFullsCorr(float *corr) const { DeviceGuard guard(impl_->device, impl_->available); const auto &d = *impl_; if (d.n_fulls == 0) return; - CudaCheck(cudaMemcpy(corr, d.f_corr.get(), size_t(d.n_fulls) * sizeof(float), - cudaMemcpyDeviceToHost), "download f_corr"); + CopyAndWait(corr, d.f_corr.get(), size_t(d.n_fulls) * sizeof(float), + cudaMemcpyDeviceToHost, impl_->s(), "download f_corr"); } void RotationScaleMergeGPU::GetFullsPxPy(float *px, float *py) const { @@ -1422,8 +1544,8 @@ void RotationScaleMergeGPU::GetFullsPxPy(float *px, float *py) const { const auto &d = *impl_; if (d.n_fulls == 0) return; const size_t bytes = size_t(d.n_fulls) * sizeof(float); - CudaCheck(cudaMemcpy(px, d.f_px.get(), bytes, cudaMemcpyDeviceToHost), "download f_px"); - CudaCheck(cudaMemcpy(py, d.f_py.get(), bytes, cudaMemcpyDeviceToHost), "download f_py"); + CopyAndWait(px, d.f_px.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_px"); + CopyAndWait(py, d.f_py.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_py"); } void RotationScaleMergeGPU::GetFullsVariance(float *var_bkg, float *var_per_I) const { @@ -1431,8 +1553,8 @@ void RotationScaleMergeGPU::GetFullsVariance(float *var_bkg, float *var_per_I) c const auto &d = *impl_; if (d.n_fulls == 0) return; const size_t bytes = size_t(d.n_fulls) * sizeof(float); - CudaCheck(cudaMemcpy(var_bkg, d.f_var_bkg.get(), bytes, cudaMemcpyDeviceToHost), "download f_var_bkg"); - CudaCheck(cudaMemcpy(var_per_I, d.f_var_per_I.get(), bytes, cudaMemcpyDeviceToHost), + CopyAndWait(var_bkg, d.f_var_bkg.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_var_bkg"); + CopyAndWait(var_per_I, d.f_var_per_I.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_var_per_I"); } @@ -1440,6 +1562,91 @@ void RotationScaleMergeGPU::SetFullsCorr(const float *corr) { DeviceGuard guard(impl_->device, impl_->available); auto &d = *impl_; if (d.n_fulls == 0) return; - CudaCheck(cudaMemcpy(d.f_corr.get(), corr, size_t(d.n_fulls) * sizeof(float), - cudaMemcpyHostToDevice), "upload f_corr"); + CopyAndWait(d.f_corr.get(), corr, size_t(d.n_fulls) * sizeof(float), + cudaMemcpyHostToDevice, impl_->s(), "upload f_corr"); +} + +void RotationScaleMergeGPU::SurfaceSetTerms(int n_terms, const SurfaceTerm *term, const uint8_t *parity, + int n_groups, const int32_t *gperm, const int32_t *gstart, + int ncell) { + DeviceGuard guard(impl_->device, impl_->available); + auto &d = *impl_; + d.s_ncell = ncell; + d.s_n_groups = n_groups; + d.Upload(d.s_term, term, n_terms); + d.Upload(d.s_parity, parity, n_terms); + d.Upload(d.s_gperm, gperm, n_terms); + d.Upload(d.s_gstart, gstart, n_groups + 1); + d.s_A = d.Alloc(std::max(1, ncell)); + d.s_cross = d.Alloc(std::max(1, ncell)); + d.s_ref2 = d.Alloc(std::max(1, ncell)); + d.s_sw = d.Alloc(std::max(1, n_groups)); + d.s_swI = d.Alloc(std::max(1, n_groups)); + d.s_w_Is = d.Alloc(std::max(1, n_terms)); + d.s_w_Iref = d.Alloc(std::max(1, n_terms)); + d.s_Iref = d.Alloc(std::max(1, n_terms)); + for (int &nb : d.s_n_blocks) nb = 0; +} + +void RotationScaleMergeGPU::SurfaceSetSubset(int subset, int n_blocks, const int32_t *perm, + const int32_t *seg_start) { + DeviceGuard guard(impl_->device, impl_->available); + auto &d = *impl_; + const int n_seg = n_blocks * d.s_ncell; + d.s_n_sel[subset] = n_blocks > 0 ? seg_start[n_seg] : 0; + d.Upload(d.s_perm[subset], perm, d.s_n_sel[subset]); + d.Upload(d.s_seg_start[subset], seg_start, n_seg + 1); + d.s_n_blocks[subset] = n_blocks; + if (size_t(n_seg) > d.s_slots) { + d.s_tcross = d.Alloc(n_seg); + d.s_tref2 = d.Alloc(n_seg); + d.s_slots = n_seg; + } +} + +void RotationScaleMergeGPU::SurfaceReference(int parity, const double *A) { + DeviceGuard guard(impl_->device, impl_->available); + auto &d = *impl_; + CopyAndWait(d.s_A.get(), A, size_t(d.s_ncell) * sizeof(double), cudaMemcpyHostToDevice, impl_->s(), + "upload surface"); + const int grp_blocks = std::min(65535, (d.s_n_groups + BLK - 1) / BLK); + if (grp_blocks > 0) + SurfaceReferenceKernel<<s()>>>(d.s_n_groups, parity, d.s_gperm.get(), + d.s_gstart.get(), d.s_term.get(), d.s_parity.get(), d.s_A.get(), d.s_sw.get(), d.s_swI.get()); + CudaCheck(cudaGetLastError(), "surface reference launch"); + CudaCheck(cudaStreamSynchronize(impl_->s()), "surface reference sync"); +} + +void RotationScaleMergeGPU::SurfaceGetReference(double *sw, double *swI) const { + DeviceGuard guard(impl_->device, impl_->available); + auto &d = *impl_; + const size_t bytes = size_t(d.s_n_groups) * sizeof(double); + CopyAndWait(sw, d.s_sw.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "dl surface sw"); + CopyAndWait(swI, d.s_swI.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "dl surface swI"); +} + +void RotationScaleMergeGPU::SurfaceFitSums(int subset, double *cross, double *ref2) { + DeviceGuard guard(impl_->device, impl_->available); + auto &d = *impl_; + const int ncell = d.s_ncell, nb = d.s_n_blocks[subset], n_seg = nb * ncell; + if (nb == 0) { + std::fill(cross, cross + ncell, 0.0); + std::fill(ref2, ref2 + ncell, 0.0); + return; + } + SurfaceFitTermKernel<<s()>>>( + d.s_n_sel[subset], d.s_perm[subset].get(), d.s_term.get(), d.s_A.get(), d.s_sw.get(), d.s_swI.get(), + d.s_w_Is.get(), d.s_w_Iref.get(), d.s_Iref.get()); + CudaCheck(cudaGetLastError(), "surface fit term launch"); + SurfaceFitSegmentKernel<<s()>>>(n_seg, + d.s_seg_start[subset].get(), d.s_w_Is.get(), d.s_w_Iref.get(), d.s_Iref.get(), + d.s_tcross.get(), d.s_tref2.get()); + CudaCheck(cudaGetLastError(), "surface fit segment launch"); + SurfaceFitCellKernel<<s()>>>(nb, ncell, + d.s_tcross.get(), d.s_tref2.get(), d.s_cross.get(), d.s_ref2.get()); + CudaCheck(cudaGetLastError(), "surface cell sum launch"); + CopyAndWait(cross, d.s_cross.get(), size_t(ncell) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), + "dl surface cross"); + CopyAndWait(ref2, d.s_ref2.get(), size_t(ncell) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), + "dl surface ref2"); } diff --git a/image_analysis/scale_merge/RotationScaleMergeGPU.h b/image_analysis/scale_merge/RotationScaleMergeGPU.h index 730a79add..766e2f640 100644 --- a/image_analysis/scale_merge/RotationScaleMergeGPU.h +++ b/image_analysis/scale_merge/RotationScaleMergeGPU.h @@ -192,6 +192,34 @@ public: // Download the fulls' working corr (length = n_fulls), valid after ScaleFulls. void GetFullsCorr(float *corr) const; + // --- correction-surface fit (RotationScaleMerge::ApplyCellSurface) --- + // The two passes every round of the fit makes over its terms, on the device; the host keeps the + // per-cell step, the gauge and the cross-validation. Both sums are formed in exactly the order the + // host forms them, so the fitted surface is the host's to the last bit (see the kernels). + + // One observation as the surface fit sees it: the host's term, uploaded as it stands. + struct SurfaceTerm { float I, sigma, corr, d; int32_t cell, group; }; + + // The terms (in fulls order) with each one's frame parity, the ASU-group CSR over them (gperm lists + // the terms of group g at [gstart[g], gstart[g+1]), in fulls order) and the cell count. + void SurfaceSetTerms(int n_terms, const SurfaceTerm *term, const uint8_t *parity, + int n_groups, const int32_t *gperm, const int32_t *gstart, int ncell); + + // One subset of the terms (0 = even frames, 1 = odd, 2 = all), cut into the host's n_blocks + // reduction blocks and ordered within each block by cell, keeping term order inside a cell: + // block b, cell c is perm[seg_start[b * ncell + c], seg_start[b * ncell + c + 1]). + void SurfaceSetSubset(int subset, int n_blocks, const int32_t *perm, const int32_t *seg_start); + + // The per-group reference sums sw / swI over the terms of frame parity `parity` (< 0 = all) with + // the surface A (length ncell) applied. They stay on the device for SurfaceFitSums; + // SurfaceGetReference downloads them (length n_groups each). + void SurfaceReference(int parity, const double *A); + void SurfaceGetReference(double *sw, double *swI) const; + + // The fit's per-cell sums over one subset against the last SurfaceReference and its A: + // cross = sum w Is Iref and ref2 = sum w Iref^2 (length ncell each). + void SurfaceFitSums(int subset, double *cross, double *ref2); + private: struct Impl; std::unique_ptr impl_; diff --git a/image_analysis/spot_finding/AdaptiveSpotFinderCPU.cpp b/image_analysis/spot_finding/AdaptiveSpotFinderCPU.cpp index 62474e7e3..995df923b 100644 --- a/image_analysis/spot_finding/AdaptiveSpotFinderCPU.cpp +++ b/image_analysis/spot_finding/AdaptiveSpotFinderCPU.cpp @@ -17,7 +17,7 @@ AdaptiveSpotFinderCPU::AdaptiveSpotFinderCPU(const AzimuthalIntegrationMapping & ring_cnt.assign(nbins, 0); ring_mean.assign(nbins, 0.0f); ring_sigma.assign(nbins, 0.0f); - ring_thr.assign(nbins, 0.0f); + ring_thr.assign(nbins + 1, INFINITY); // the last entry is for pixels outside every ring ring_bkg.assign(nbins, NAN); ring_bits.assign(OutputSize(), 0); ring_hist.assign(nbins * HIST_VALUES, 0); @@ -54,84 +54,54 @@ void AdaptiveSpotFinderCPU::BeginRings() { rings_from_blocks = true; } -// The plain pass over pixels [first, first + n). +// The plain pass over pixels [first, first + n): the histogram of each ring's values, from which +// PlainRings() takes the integer sums, and the fused profile when asked for. One loop reads each pixel +// once for both. void AdaptiveSpotFinderCPU::AccumulateRingsBlock(const ImagePreprocessorBuffer &image, size_t first, size_t n) { const auto &pixel_to_bin = mapping.GetPixelToBin(); const size_t nbins = ring_sum.size(); const float *corrections = mapping.Corrections().data(); - - // Consecutive pixels mostly share a ring, so a ring's sums are held in locals while they do and - // written back when the ring changes: the same additions in the same order, without a store and a - // reload of the same address on every pixel. The azimuthal-integration sums get a loop of their own - // over the block, so that neither loop runs out of registers for its sums. - if (fuse_azint) { - size_t cur = nbins; // the ring held in the locals below; nbins = none - float az_sum = 0.0f, az_sum2 = 0.0f; - uint32_t az_cnt = 0; - for (size_t pxl = first; pxl < first + n; ++pxl) { - const int32_t v = image[pxl]; - if (v == INT32_MIN || v == INT32_MAX) continue; // bad / saturated - const uint16_t b = pixel_to_bin[pxl]; - if (b >= nbins) continue; // masked / out of range (UINT16_MAX) - if (b != cur) { - if (cur != nbins) { - azint_sum[cur] = az_sum; - azint_sum2[cur] = az_sum2; - azint_count[cur] = az_cnt; - } - cur = b; - az_sum = azint_sum[b]; - az_sum2 = azint_sum2[b]; - az_cnt = azint_count[b]; - } - const float val = static_cast(v) * corrections[pxl]; - const float val_sq = val * val; - az_sum += val; - az_sum2 += val_sq; - ++az_cnt; - } - if (cur != nbins) { - azint_sum[cur] = az_sum; - azint_sum2[cur] = az_sum2; - azint_count[cur] = az_cnt; - } - } - - // Values outside the histogram are listed by a second loop, run only when the block has any: a - // call in this loop would leave the sums in memory again. uint32_t *hist = ring_hist.data(); + + // Consecutive pixels mostly share a ring, so the ring's profile sums are held in locals while they + // do and written back when the ring changes: the same additions in the same order, without a store + // and a reload of the same address on every pixel. Values outside the histogram are listed by a + // second loop, run only when the block has any. bool overflow = false; - size_t cur = nbins; - int64_t sum = 0, cnt = 0; - uint64_t sum2 = 0; + size_t cur = nbins; // the ring held in the locals below; nbins = none + float az_sum = 0.0f, az_sum2 = 0.0f; + uint32_t az_cnt = 0; for (size_t pxl = first; pxl < first + n; ++pxl) { const int32_t v = image[pxl]; if (v == INT32_MIN || v == INT32_MAX) continue; // bad / saturated const uint16_t b = pixel_to_bin[pxl]; if (b >= nbins) continue; // masked / out of range (UINT16_MAX) - if (b != cur) { - if (cur != nbins) { - ring_sum[cur] = sum; - ring_sum2[cur] = sum2; - ring_cnt[cur] = cnt; - } - cur = b; - sum = ring_sum[b]; - sum2 = ring_sum2[b]; - cnt = ring_cnt[b]; - } - sum += v; - sum2 += static_cast(static_cast(v) * v); - cnt += 1; - if (v >= 0 && v < HIST_VALUES) + if (static_cast(v) < HIST_VALUES) hist[b * HIST_VALUES + v] += 1; else overflow = true; + if (!fuse_azint) continue; + if (b != cur) { + if (cur != nbins) { + azint_sum[cur] = az_sum; + azint_sum2[cur] = az_sum2; + azint_count[cur] = az_cnt; + } + cur = b; + az_sum = azint_sum[b]; + az_sum2 = azint_sum2[b]; + az_cnt = azint_count[b]; + } + const float val = static_cast(v) * corrections[pxl]; + const float val_sq = val * val; + az_sum += val; + az_sum2 += val_sq; + ++az_cnt; } if (cur != nbins) { - ring_sum[cur] = sum; - ring_sum2[cur] = sum2; - ring_cnt[cur] = cnt; + azint_sum[cur] = az_sum; + azint_sum2[cur] = az_sum2; + azint_count[cur] = az_cnt; } if (overflow) @@ -146,7 +116,8 @@ void AdaptiveSpotFinderCPU::AccumulateRingsBlock(const ImagePreprocessorBuffer & } // A sigma-clip pass over the plain pass's values: each distinct value of a ring meets the same test the -// pixels holding it would, and its pixels are added as a count. +// pixels holding it would, and its pixels are added as a count. Integer sums, so with nothing clipped +// (clip_k = INFINITY) they are the plain sums over the pixels themselves. void AdaptiveSpotFinderCPU::ClipRings(float clip_k) { const size_t nbins = ring_sum.size(); @@ -155,6 +126,7 @@ void AdaptiveSpotFinderCPU::ClipRings(float clip_k) { std::fill(ring_cnt.begin(), ring_cnt.end(), 0); const auto keep = [&](uint16_t b, int32_t v) { + if (std::isinf(clip_k)) return true; const float lo = ring_mean[b] - clip_k * ring_sigma[b]; const float hi = ring_mean[b] + clip_k * ring_sigma[b]; return !(v < lo || v > hi); // exclude peaks / outliers @@ -197,6 +169,7 @@ void AdaptiveSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image, AccumulateRingsBlock(image, 0, static_cast(width) * height); } rings_from_blocks = false; + ClipRings(INFINITY); UpdateRingStatistics(); ClipRings(3.0f); UpdateRingStatistics(); @@ -259,19 +232,31 @@ void AdaptiveSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image, void AdaptiveSpotFinderCPU::FlagRow(const ImagePreprocessorBuffer &image, int32_t row) { const auto &pixel_to_bin = mapping.GetPixelToBin(); - const size_t nbins = ring_thr.size(); + const auto nbins = static_cast(ring_thr.size() - 1); // ring_thr[nbins] is +inf + const float *thr = ring_thr.data(); + const int32_t *img = image.data(); + const uint16_t *bin = pixel_to_bin.data(); const size_t first = static_cast(row) * width; + const size_t end = first + width; - for (size_t pxl = first; pxl < first + width; ++pxl) { - const int32_t v = image[pxl]; - const uint16_t b = pixel_to_bin[pxl]; - bool strong = false; - if (v == INT32_MAX) - strong = true; - else if (v != INT32_MIN && b < nbins && v >= ring_thr[b]) - strong = true; - - if (strong) - ring_bits[pxl / 32] |= 1U << (pxl % 32); + // Saturated is strong, bad is not, and a pixel outside every ring (bin >= nbins) meets the +inf + // threshold. Written without branches and a word of 32 pixels at a time, so that the loop vectorises. + const auto strong = [&](size_t pxl) -> uint32_t { + const int32_t v = img[pxl]; + const uint32_t b = std::min(bin[pxl], nbins); + return (v == INT32_MAX) | ((v != INT32_MIN) & (v >= thr[b])); + }; + size_t pxl = first; + while (pxl < end && pxl % 32 != 0) { + ring_bits[pxl / 32] |= strong(pxl) << (pxl % 32); + ++pxl; } + for (; pxl + 32 <= end; pxl += 32) { + uint32_t word = 0; + for (uint32_t j = 0; j < 32; ++j) + word |= strong(pxl + j) << j; + ring_bits[pxl / 32] |= word; + } + for (; pxl < end; ++pxl) + ring_bits[pxl / 32] |= strong(pxl) << (pxl % 32); } diff --git a/image_analysis/spot_finding/AdaptiveSpotFinderCPU.h b/image_analysis/spot_finding/AdaptiveSpotFinderCPU.h index d0a82aa5f..184ef2379 100644 --- a/image_analysis/spot_finding/AdaptiveSpotFinderCPU.h +++ b/image_analysis/spot_finding/AdaptiveSpotFinderCPU.h @@ -57,7 +57,7 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU { std::vector ring_cnt; std::vector ring_mean; std::vector ring_sigma; - std::vector ring_thr; + std::vector ring_thr; // nbins + 1: the last is +inf, the threshold of a pixel outside every ring // ring_mean of the last Detect(), NaN where the ring holds too few pixels to be its own background. // Kept separately because ring_mean carries the previous frame's value for an empty ring. std::vector ring_bkg; @@ -65,8 +65,8 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU { // local-box mask that ImageSpotFinderCPU::Detect leaves in output_buffer. std::vector ring_bits; // The plain pass's valid pixels as a per-ring histogram of their values (HIST_VALUES bins per - // ring) plus a list of the values outside it, so the two sigma-clip passes sum over distinct - // values instead of over the image again. Integer sums, so the same totals. + // ring) plus a list of the values outside it. The plain sums and the two sigma-clip passes are all + // taken from it, over distinct values instead of over the image. Integer sums, so the same totals. static constexpr int32_t HIST_VALUES = 1024; std::vector ring_hist; std::vector> ring_overflow; // (ring, value) @@ -84,7 +84,7 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU { // Zero the sums of the plain ring pass (and of the fused profile). void ResetRings(); - // One sigma-clip pass over the plain pass's values. + // One sigma-clip pass over the plain pass's values; clip_k = INFINITY gives the plain sums. void ClipRings(float clip_k); // ring_mean / ring_sigma from the current sums. void UpdateRingStatistics(); diff --git a/rugnux/HotPixels.cpp b/rugnux/HotPixels.cpp index 0af35f712..88fdadc9e 100644 --- a/rugnux/HotPixels.cpp +++ b/rugnux/HotPixels.cpp @@ -216,7 +216,9 @@ void HotPixelFinder::AddLevels(const std::vector §or_level, const s const int r = static_cast(k / SECTORS); level[k] = std::max(ring_level[r], sector_level[k]); const float noise = std::max(std::sqrt(static_cast(std::max(level[k], 0))), ring_spread[r]); - threshold[k] = static_cast(level[k]) + LIT_NSIGMA * noise + LIT_OFFSET; + // One rounding for level + nsigma * noise and one for the offset, written out so that every + // compiler takes the same two - the device takes them too (HotPixelsGPU.cu). + threshold[k] = std::fma(LIT_NSIGMA, noise, static_cast(level[k])) + LIT_OFFSET; } // Every sum is an integer, so the result does not depend on the order the frames arrive in. @@ -230,50 +232,24 @@ void HotPixelFinder::AddLevels(const std::vector §or_level, const s } #ifdef JFJOCH_USE_CUDA +void HotPixelFinder::PrepareDevice() { + std::lock_guard lock(m); + if (!gpu) + gpu = std::make_unique(key.get(), width * height, key_begin, nrings, SECTORS, + HotPixelLevelRules{MIN_SECTOR_PIXELS, MIN_RING_PIXELS, + LIT_NSIGMA, LIT_OFFSET}); +} + void HotPixelFinder::AddDeviceImage(const int32_t *device_image, HotPixelFinderGPU::Frame &frame) { - HotPixelFinderGPU *device; - { - std::lock_guard lock(m); - if (!gpu) - gpu = std::make_unique(key.get(), width * height, key_begin, nrings, SECTORS); - device = gpu.get(); - } - std::vector count; - std::vector sector_median, ring_median, ring_mad; - device->Statistics(device_image, frame, count, sector_median, ring_median, ring_mad); - - // The same levels AddImage takes off its scratch buffer, and under the same pixel minima. - const size_t nkeys = static_cast(nrings) * SECTORS; - std::vector sector_level(nkeys, 0); - for (size_t k = 0; k < nkeys; k++) - if (count[k] >= MIN_SECTOR_PIXELS) - sector_level[k] = sector_median[k]; - std::vector ring_level(nrings, 0); - std::vector ring_spread(nrings, 0.0f); - std::vector ring_ok(nrings, 0); - for (int r = 0; r < nrings; r++) { - size_t n = 0; - for (int s = 0; s < SECTORS; s++) - n += count[r * SECTORS + s]; - if (n < MIN_RING_PIXELS) continue; - ring_ok[r] = 1; - ring_level[r] = ring_median[r]; - ring_spread[r] = 1.4826f * static_cast(ring_mad[r]); - } - - std::vector level; - std::vector threshold; - AddLevels(sector_level, ring_level, ring_spread, ring_ok, level, threshold); - device->Accumulate(device_image, frame, level, threshold, ring_ok); + PrepareDevice(); + gpu->Add(device_image, frame); + std::lock_guard lock(m); + frames++; } #endif HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double spacing_deg, size_t nthreads) { std::lock_guard lock(m); -#ifdef JFJOCH_USE_CUDA - if (gpu) - gpu->Download(n_lit.get(), n_error.get(), sum_value.get(), n_error_ring_ok.get(), error_level_sum.get()); -#endif Result ret; ret.frames = frames; ret.mask.assign(width * height, 0); @@ -281,12 +257,38 @@ HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double sp // The chance rate per ring, from the pixels lit on no more than half of their frames: whatever // lights those - reflections, zingers, noise above the bound - lights a defect-free pixel too. + // Counted in integers by blocks of rows in parallel, so the totals do not depend on the split. std::vector lit(nrings, 0.0), seen(nrings, 0.0); - for (size_t i = 0; i < width * height; i++) - if (key[i] >= 0 && n_valid(i) > 0 && 2 * n_lit[i] <= n_valid(i)) { - lit[key[i] / SECTORS] += n_lit[i]; - seen[key[i] / SECTORS] += n_valid(i); +#ifdef JFJOCH_USE_CUDA + if (gpu) { + std::vector device_lit, device_seen; + gpu->ChanceCounts(device_lit, device_seen); + for (int r = 0; r < nrings; r++) { + lit[r] = static_cast(device_lit[r]); + seen[r] = static_cast(device_seen[r]); } + } else +#endif + { + std::vector> block_lit(BANDS), block_seen(BANDS); + const size_t rows_per_band = (height + BANDS - 1) / BANDS; + ParallelFor(static_cast(BANDS), nthreads, [&](int b) { + block_lit[b].assign(nrings, 0); + block_seen[b].assign(nrings, 0); + const size_t begin = std::min(width * height, b * rows_per_band * width); + const size_t end = std::min(width * height, (b + 1) * rows_per_band * width); + for (size_t i = begin; i < end; i++) + if (key[i] >= 0 && n_valid(i) > 0 && 2 * n_lit[i] <= n_valid(i)) { + block_lit[b][key[i] / SECTORS] += n_lit[i]; + block_seen[b][key[i] / SECTORS] += n_valid(i); + } + }); + for (int r = 0; r < nrings; r++) + for (size_t b = 0; b < BANDS; b++) { + lit[r] += static_cast(block_lit[b][r]); + seen[r] += static_cast(block_seen[b][r]); + } + } std::vector k_chance(nrings, n + 1); for (int r = 0; r < nrings; r++) if (seen[r] > 0.0) @@ -295,6 +297,26 @@ HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double sp // Persistent: lit on more frames than one reflection or chance explains. const int min_valid = std::max(10, n / 2); +#ifdef JFJOCH_USE_CUDA + // With a GPU the per-pixel sums stay there. Only the pixels that can be masked come back - those the + // tests below could pass (see GetCandidates) - into the host arrays, which are zero everywhere else, + // so the tests below run on them unchanged. + if (gpu) { + const auto c = gpu->GetCandidates(frames, min_valid, spacing_deg > 0.0, k_chance); + for (size_t j = 0; j < c.index.size(); j++) { + const size_t i = c.index[j]; + n_lit[i] = c.n_lit[j]; + n_error[i] = c.n_error[j]; + n_error_ring_ok[i] = c.n_error_ring_ok[j]; + sum_value[i] = c.sum_value[j]; + error_level_sum[i] = c.error_level_sum[j]; + } + for (size_t k = 0; k < key_frames.size(); k++) { + key_frames[k] = static_cast(c.key_frames[k]); + key_level_sum[k] = c.key_level_sum[k]; + } + } +#endif std::vector persistent(width * height, 0); ParallelChunks(static_cast(height), nthreads, [&](int y0, int y1) { for (size_t y = y0; y < static_cast(y1); y++) diff --git a/rugnux/HotPixels.h b/rugnux/HotPixels.h index 63e6bd7a8..b837ba32e 100644 --- a/rugnux/HotPixels.h +++ b/rugnux/HotPixels.h @@ -94,6 +94,10 @@ public: // `scratch` above. The per-pixel sums are then kept on the device and replace the host's when the // mask is read, so a finder is fed one way or the other, not both. Thread safe. void AddDeviceImage(const int32_t *device_image, HotPixelFinderGPU::Frame &frame); + + // Build the device half now rather than with the first device frame, so the workers do not wait on + // it one behind the other. + void PrepareDevice(); #endif // The mask, from the frames added so far. oscillation_deg is the rotation per image and @@ -132,11 +136,11 @@ private: std::unique_ptr n_error_ring_ok; std::unique_ptr error_level_sum; #ifdef JFJOCH_USE_CUDA - std::unique_ptr gpu; // built by the first device frame + std::unique_ptr gpu; // built by PrepareDevice or the first device frame #endif // Each ring-sector's level and lit threshold, from the frame's order statistics, and the frame's - // share of the per-key sums. The host and the device path both come through here. + // share of the per-key sums. The device does the same in HotPixelsGPU.cu, with the same roundings. void AddLevels(const std::vector §or_level, const std::vector &ring_level, const std::vector &ring_spread, const std::vector &ring_ok, std::vector &level, std::vector &threshold); diff --git a/rugnux/HotPixelsGPU.cu b/rugnux/HotPixelsGPU.cu index 2691c221f..dc6346dc6 100644 --- a/rugnux/HotPixelsGPU.cu +++ b/rugnux/HotPixelsGPU.cu @@ -3,6 +3,8 @@ #include "HotPixelsGPU.h" +#include + #include "../common/JFJochException.h" namespace { @@ -145,6 +147,84 @@ __global__ void accumulate_kernel(const int32_t *__restrict__ image, const int32 } } +// Each key's level and lit threshold from the frame's order statistics, and the frame's share of the +// per-key sums - HotPixelFinder::AddImage and AddLevels, step for step. The threshold is +// fma(nsigma, noise, level) + offset, the two roundings the host takes (see AddLevels). +__global__ void levels_kernel(size_t nkeys, int sectors, HotPixelLevelRules rules, const uint32_t *__restrict__ count, + const int32_t *__restrict__ sector_median, const int32_t *__restrict__ ring_median, + const int32_t *__restrict__ ring_mad, int32_t *__restrict__ level, + float *__restrict__ threshold, char *__restrict__ ring_ok, + uint32_t *__restrict__ key_frames, int64_t *__restrict__ key_level_sum) { + const size_t k = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (k >= nkeys) return; + const size_t r = k / sectors; + uint32_t n = 0; + for (int s = 0; s < sectors; s++) + n += count[r * sectors + s]; + const bool ok = n >= static_cast(rules.min_ring_pixels); + const int32_t ring_level = ok ? ring_median[r] : 0; + const float ring_spread = ok ? 1.4826f * static_cast(ring_mad[r]) : 0.0f; + const int32_t sector_level = count[k] >= static_cast(rules.min_sector_pixels) ? sector_median[k] : 0; + const int32_t lv = max(ring_level, sector_level); + const float root = sqrtf(static_cast(max(lv, 0))); + const float noise = root < ring_spread ? ring_spread : root; + level[k] = lv; + threshold[k] = __fadd_rn(__fmaf_rn(rules.lit_nsigma, noise, static_cast(lv)), rules.lit_offset); + if (k % sectors == 0) + ring_ok[r] = ok; + if (ok) { + atomicAdd(&key_frames[k], 1u); + atomicAdd(reinterpret_cast(&key_level_sum[k]), + static_cast(static_cast(lv))); + } +} + +__device__ int valid_frames(const int32_t *key, const uint32_t *key_frames, const uint16_t *n_error_ring_ok, size_t i) { + return static_cast(key_frames[key[i]]) - static_cast(n_error_ring_ok[i]); +} + +__global__ void chance_kernel(size_t npixels, int sectors, const int32_t *__restrict__ key, + const uint32_t *__restrict__ key_frames, const uint16_t *__restrict__ n_lit, + const uint16_t *__restrict__ n_error_ring_ok, unsigned long long *__restrict__ lit, + unsigned long long *__restrict__ seen) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= npixels || key[i] < 0) return; + const int nv = valid_frames(key, key_frames, n_error_ring_ok, i); + if (nv > 0 && 2 * static_cast(n_lit[i]) <= nv) { + atomicAdd(&lit[key[i] / sectors], static_cast(n_lit[i])); + atomicAdd(&seen[key[i] / sectors], static_cast(nv)); + } +} + +// Writes the candidates' indices from `out` on when `out` is given, and counts them either way. +__global__ void candidate_kernel(size_t npixels, int sectors, uint32_t frames, int min_valid, bool spacing_ok, + const int32_t *__restrict__ key, const uint32_t *__restrict__ key_frames, + const uint16_t *__restrict__ n_lit, const uint16_t *__restrict__ n_error, + const uint16_t *__restrict__ n_error_ring_ok, const int *__restrict__ k_chance, + uint32_t *__restrict__ count, uint32_t *__restrict__ out) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i >= npixels || key[i] < 0) return; + const int nv = valid_frames(key, key_frames, n_error_ring_ok, i); + const bool error = 2 * static_cast(n_error[i]) > frames; + const int lit = n_lit[i]; + const bool persistent = spacing_ok && nv >= min_valid && lit > 0 + && lit >= min(max(2, k_chance[key[i] / sectors]), nv); + if (!error && !persistent) return; + const uint32_t slot = atomicAdd(count, 1u); + if (out) out[slot] = static_cast(i); +} + +__global__ void iota_kernel(size_t n, uint32_t *__restrict__ out) { + const size_t i = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (i < n) out[i] = static_cast(i); +} + +template +__global__ void gather_kernel(size_t n, const uint32_t *__restrict__ index, const T *__restrict__ in, T *__restrict__ out) { + const size_t j = blockIdx.x * static_cast(blockDim.x) + threadIdx.x; + if (j < n) out[j] = in[index[j]]; +} + // The shared tables and sums are filled on a stream of their own, added to on the workers' streams // and downloaded on the NULL stream, so they are allocated synchronously rather than from the pool: // a pooled buffer is freed on the thread's allocation stream, which none of those is ordered before. @@ -156,22 +236,18 @@ constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous; } // namespace HotPixelFinderGPU::HotPixelFinderGPU(const int32_t *host_key, size_t npixels, - const std::vector &host_key_begin, int nrings, int sectors) + const std::vector &host_key_begin, int nrings, int sectors, + const HotPixelLevelRules &rules) : npixels(npixels), nkeys(static_cast(nrings) * sectors), nrings(nrings), - sectors(sectors), - key(npixels, ALLOC), pixels_by_key(host_key_begin.back(), ALLOC), key_begin(host_key_begin.size(), ALLOC), + sectors(sectors), rules(rules), + key(npixels, ALLOC), pixels_by_key(std::max(host_key_begin.back(), 1), ALLOC), + key_begin(host_key_begin.size(), ALLOC), n_lit(npixels, ALLOC), n_error(npixels, ALLOC), n_error_ring_ok(npixels, ALLOC), - sum_value(npixels, ALLOC), error_level_sum(npixels, ALLOC) { - std::vector pixels(host_key_begin.back()); - std::vector filled(host_key_begin.begin(), host_key_begin.end() - 1); - for (size_t i = 0; i < npixels; i++) - if (host_key[i] >= 0) - pixels[filled[host_key[i]]++] = static_cast(i); - + sum_value(npixels, ALLOC), error_level_sum(npixels, ALLOC), + key_frames(std::max(nkeys, 1), ALLOC), key_level_sum(std::max(nkeys, 1), ALLOC) { + cuda_err(cudaEventCreateWithFlags(&last_accumulate, cudaEventDisableTiming)); CudaStream stream; cuda_err(cudaMemcpyAsync(key, host_key, npixels * sizeof(int32_t), cudaMemcpyHostToDevice, stream)); - cuda_err(cudaMemcpyAsync(pixels_by_key, pixels.data(), pixels.size() * sizeof(uint32_t), - cudaMemcpyHostToDevice, stream)); cuda_err(cudaMemcpyAsync(key_begin, host_key_begin.data(), host_key_begin.size() * sizeof(uint32_t), cudaMemcpyHostToDevice, stream)); cuda_err(cudaMemsetAsync(n_lit, 0, npixels * sizeof(uint16_t), stream)); @@ -179,16 +255,34 @@ HotPixelFinderGPU::HotPixelFinderGPU(const int32_t *host_key, size_t npixels, cuda_err(cudaMemsetAsync(n_error_ring_ok, 0, npixels * sizeof(uint16_t), stream)); cuda_err(cudaMemsetAsync(sum_value, 0, npixels * sizeof(int64_t), stream)); cuda_err(cudaMemsetAsync(error_level_sum, 0, npixels * sizeof(int64_t), stream)); - cuda_err(cudaStreamSynchronize(stream)); + cuda_err(cudaMemsetAsync(key_frames, 0, std::max(nkeys, 1) * sizeof(uint32_t), stream)); + cuda_err(cudaMemsetAsync(key_level_sum, 0, std::max(nkeys, 1) * sizeof(int64_t), stream)); + + // The unmasked pixels grouped by key: a stable sort of the pixel indices by key, so each key's + // pixels stay in pixel order. A masked pixel's key, -1, is the largest as unsigned and sorts last. + { + CudaDevicePtr index(npixels, ALLOC), sorted_key(npixels, ALLOC), sorted_index(npixels, ALLOC); + iota_kernel<<((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>(npixels, index); + cuda_err(cudaGetLastError()); + const auto *keys_in = reinterpret_cast(key.get()); + size_t bytes = 0; + cuda_err(cub::DeviceRadixSort::SortPairs(nullptr, bytes, keys_in, sorted_key.get(), index.get(), + sorted_index.get(), npixels, 0, 32, stream)); + CudaDevicePtr scratch(bytes, ALLOC); + cuda_err(cub::DeviceRadixSort::SortPairs(scratch.get(), bytes, keys_in, sorted_key.get(), index.get(), + sorted_index.get(), npixels, 0, 32, stream)); + if (host_key_begin.back() > 0) + cuda_err(cudaMemcpyAsync(pixels_by_key, sorted_index, host_key_begin.back() * sizeof(uint32_t), + cudaMemcpyDeviceToDevice, stream)); + cuda_err(cudaStreamSynchronize(stream)); + } } -void HotPixelFinderGPU::Statistics(const int32_t *device_image, Frame &frame, std::vector &count, - std::vector §or_median, std::vector &ring_median, - std::vector &ring_mad) { - count.resize(nkeys); - sector_median.resize(nkeys); - ring_median.resize(nrings); - ring_mad.resize(nrings); +HotPixelFinderGPU::~HotPixelFinderGPU() { + if (last_accumulate) cudaEventDestroy(last_accumulate); +} + +void HotPixelFinderGPU::Add(const int32_t *device_image, Frame &frame) { if (nrings == 0) return; if (!frame.count.get()) { @@ -201,44 +295,101 @@ void HotPixelFinderGPU::Statistics(const int32_t *device_image, Frame &frame, st frame.ring_ok = CudaDevicePtr(nrings); } const cudaStream_t stream = *frame.stream; - sector_kernel<<(nkeys), THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin, frame.count, - frame.sector_median); + sector_kernel<<(nkeys), THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin, + frame.count, frame.sector_median); cuda_err(cudaGetLastError()); ring_kernel<<>>(device_image, pixels_by_key, key_begin, frame.count, sectors, frame.ring_median, frame.ring_mad); cuda_err(cudaGetLastError()); - - cuda_err(cudaMemcpyAsync(count.data(), frame.count, nkeys * sizeof(uint32_t), cudaMemcpyDeviceToHost, stream)); - cuda_err(cudaMemcpyAsync(sector_median.data(), frame.sector_median, nkeys * sizeof(int32_t), - cudaMemcpyDeviceToHost, stream)); - cuda_err(cudaMemcpyAsync(ring_median.data(), frame.ring_median, nrings * sizeof(int32_t), - cudaMemcpyDeviceToHost, stream)); - cuda_err(cudaMemcpyAsync(ring_mad.data(), frame.ring_mad, nrings * sizeof(int32_t), - cudaMemcpyDeviceToHost, stream)); - cuda_err(cudaStreamSynchronize(stream)); -} - -void HotPixelFinderGPU::Accumulate(const int32_t *device_image, Frame &frame, const std::vector &level, - const std::vector &threshold, const std::vector &ring_ok) { - const cudaStream_t stream = *frame.stream; - cuda_err(cudaMemcpyAsync(frame.level, level.data(), nkeys * sizeof(int32_t), cudaMemcpyHostToDevice, stream)); - cuda_err(cudaMemcpyAsync(frame.threshold, threshold.data(), nkeys * sizeof(float), cudaMemcpyHostToDevice, - stream)); - cuda_err(cudaMemcpyAsync(frame.ring_ok, ring_ok.data(), nrings * sizeof(char), cudaMemcpyHostToDevice, stream)); + levels_kernel<<((nkeys + THREADS - 1) / THREADS), THREADS, 0, stream>>>( + nkeys, sectors, rules, frame.count, frame.sector_median, frame.ring_median, frame.ring_mad, + frame.level, frame.threshold, frame.ring_ok, key_frames, key_level_sum); + cuda_err(cudaGetLastError()); std::lock_guard lock(accumulate_mutex); + cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0)); accumulate_kernel<<((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>( device_image, key, npixels, sectors, frame.level, frame.threshold, frame.ring_ok, n_lit, n_error, sum_value, n_error_ring_ok, error_level_sum); cuda_err(cudaGetLastError()); + cuda_err(cudaEventRecord(last_accumulate, stream)); +} + +void HotPixelFinderGPU::ChanceCounts(std::vector &lit, std::vector &seen) { + CudaStream stream; + cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0)); + const size_t rings = std::max(nrings, 1); + CudaDevicePtr d_lit(rings, ALLOC), d_seen(rings, ALLOC); + cuda_err(cudaMemsetAsync(d_lit, 0, rings * sizeof(unsigned long long), stream)); + cuda_err(cudaMemsetAsync(d_seen, 0, rings * sizeof(unsigned long long), stream)); + chance_kernel<<((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>( + npixels, sectors, key, key_frames, n_lit, n_error_ring_ok, d_lit, d_seen); + cuda_err(cudaGetLastError()); + lit.resize(nrings); + seen.resize(nrings); + cuda_err(cudaMemcpyAsync(lit.data(), d_lit, nrings * sizeof(int64_t), cudaMemcpyDeviceToHost, stream)); + cuda_err(cudaMemcpyAsync(seen.data(), d_seen, nrings * sizeof(int64_t), cudaMemcpyDeviceToHost, stream)); cuda_err(cudaStreamSynchronize(stream)); } -void HotPixelFinderGPU::Download(uint16_t *host_n_lit, uint16_t *host_n_error, int64_t *host_sum_value, - uint16_t *host_n_error_ring_ok, int64_t *host_error_level_sum) { - cuda_err(cudaMemcpy(host_n_lit, n_lit, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost)); - cuda_err(cudaMemcpy(host_n_error, n_error, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost)); - cuda_err(cudaMemcpy(host_sum_value, sum_value, npixels * sizeof(int64_t), cudaMemcpyDeviceToHost)); - cuda_err(cudaMemcpy(host_n_error_ring_ok, n_error_ring_ok, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost)); - cuda_err(cudaMemcpy(host_error_level_sum, error_level_sum, npixels * sizeof(int64_t), cudaMemcpyDeviceToHost)); +HotPixelFinderGPU::Candidates HotPixelFinderGPU::GetCandidates(uint32_t frames, int min_valid, bool spacing_ok, + const std::vector &k_chance) { + CudaStream stream; + cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0)); + const unsigned blocks = static_cast((npixels + THREADS - 1) / THREADS); + CudaDevicePtr d_k_chance(std::max(k_chance.size(), 1), ALLOC); + CudaDevicePtr d_count(1, ALLOC); + if (!k_chance.empty()) + cuda_err(cudaMemcpyAsync(d_k_chance, k_chance.data(), k_chance.size() * sizeof(int), cudaMemcpyHostToDevice, + stream)); + cuda_err(cudaMemsetAsync(d_count, 0, sizeof(uint32_t), stream)); + candidate_kernel<<>>(npixels, sectors, frames, min_valid, spacing_ok, key, key_frames, + n_lit, n_error, n_error_ring_ok, d_k_chance, d_count, nullptr); + cuda_err(cudaGetLastError()); + uint32_t n = 0; + cuda_err(cudaMemcpyAsync(&n, d_count, sizeof(uint32_t), cudaMemcpyDeviceToHost, stream)); + cuda_err(cudaStreamSynchronize(stream)); + + Candidates c; + c.key_frames.resize(nkeys); + c.key_level_sum.resize(nkeys); + if (nkeys > 0) { + cuda_err(cudaMemcpyAsync(c.key_frames.data(), key_frames, nkeys * sizeof(uint32_t), cudaMemcpyDeviceToHost, + stream)); + cuda_err(cudaMemcpyAsync(c.key_level_sum.data(), key_level_sum, nkeys * sizeof(int64_t), + cudaMemcpyDeviceToHost, stream)); + } + if (n > 0) { + CudaDevicePtr index(n, ALLOC); + CudaDevicePtr g_lit(n, ALLOC), g_error(n, ALLOC), g_error_ring_ok(n, ALLOC); + CudaDevicePtr g_sum(n, ALLOC), g_error_level(n, ALLOC); + cuda_err(cudaMemsetAsync(d_count, 0, sizeof(uint32_t), stream)); + candidate_kernel<<>>(npixels, sectors, frames, min_valid, spacing_ok, key, + key_frames, n_lit, n_error, n_error_ring_ok, d_k_chance, + d_count, index); + cuda_err(cudaGetLastError()); + const unsigned gb = (n + THREADS - 1) / THREADS; + gather_kernel<<>>(n, index, n_lit.get(), g_lit.get()); + gather_kernel<<>>(n, index, n_error.get(), g_error.get()); + gather_kernel<<>>(n, index, n_error_ring_ok.get(), g_error_ring_ok.get()); + gather_kernel<<>>(n, index, sum_value.get(), g_sum.get()); + gather_kernel<<>>(n, index, error_level_sum.get(), g_error_level.get()); + cuda_err(cudaGetLastError()); + c.index.resize(n); + c.n_lit.resize(n); + c.n_error.resize(n); + c.n_error_ring_ok.resize(n); + c.sum_value.resize(n); + c.error_level_sum.resize(n); + cuda_err(cudaMemcpyAsync(c.index.data(), index, n * sizeof(uint32_t), cudaMemcpyDeviceToHost, stream)); + cuda_err(cudaMemcpyAsync(c.n_lit.data(), g_lit, n * sizeof(uint16_t), cudaMemcpyDeviceToHost, stream)); + cuda_err(cudaMemcpyAsync(c.n_error.data(), g_error, n * sizeof(uint16_t), cudaMemcpyDeviceToHost, stream)); + cuda_err(cudaMemcpyAsync(c.n_error_ring_ok.data(), g_error_ring_ok, n * sizeof(uint16_t), + cudaMemcpyDeviceToHost, stream)); + cuda_err(cudaMemcpyAsync(c.sum_value.data(), g_sum, n * sizeof(int64_t), cudaMemcpyDeviceToHost, stream)); + cuda_err(cudaMemcpyAsync(c.error_level_sum.data(), g_error_level, n * sizeof(int64_t), + cudaMemcpyDeviceToHost, stream)); + } + cuda_err(cudaStreamSynchronize(stream)); + return c; } diff --git a/rugnux/HotPixelsGPU.h b/rugnux/HotPixelsGPU.h index f062e64cf..e68b6e4f0 100644 --- a/rugnux/HotPixelsGPU.h +++ b/rugnux/HotPixelsGPU.h @@ -10,33 +10,53 @@ #include "../image_analysis/indexing/CUDAMemHelpers.h" -// The device half of HotPixelFinder, for frames already preprocessed on the GPU: the per-frame order -// statistics (each ring-sector's median, each ring's median and median absolute deviation) and the -// per-pixel sums run where the image already is, and only the per-key statistics - tens of thousands -// of numbers - come to the host, which turns them into levels and thresholds with the very code the -// host path uses. Every statistic is an exact order statistic of integers and every sum an integer, -// so the sums, and with them the mask, are identical to what HotPixelFinder::AddImage produces. +// What a frame's levels and lit thresholds are made of (HotPixelFinder's constants), handed to the +// device so the two halves cannot drift apart. +struct HotPixelLevelRules { + int min_sector_pixels; + int min_ring_pixels; + float lit_nsigma; + float lit_offset; +}; + +// The device half of HotPixelFinder, for frames already preprocessed on the GPU. Everything a frame adds +// stays on the device: the per-frame order statistics (each ring-sector's median, each ring's median +// and median absolute deviation), the levels and thresholds made from them, and the per-pixel and +// per-key sums. Every statistic is an exact order statistic of integers, the threshold is computed +// with the same rounding steps as the host's (see HotPixelFinder::AddLevels) and every sum is an +// integer, so the sums are identical to what HotPixelFinder::AddImage produces. When the mask is read +// only what can decide it comes back: per-ring counts for the chance rate, then the few pixels that can +// be persistent or carry the error value. class HotPixelFinderGPU { const size_t npixels; const size_t nkeys; const int nrings; const int sectors; + const HotPixelLevelRules rules; CudaDevicePtr key; // ring * sectors + sector of each pixel, -1 masked - CudaDevicePtr pixels_by_key; // the unmasked pixels, grouped by key + CudaDevicePtr pixels_by_key; // the unmasked pixels, grouped by key, in pixel order CudaDevicePtr key_begin; // where each key's pixels start in pixels_by_key - // The per-pixel sums, exactly those of HotPixelFinder. + // The per-pixel and per-key sums, exactly those of HotPixelFinder. CudaDevicePtr n_lit, n_error, n_error_ring_ok; CudaDevicePtr sum_value, error_level_sum; - // Frames are selected on their workers' streams in parallel, but each pixel's sums are plain - // read-modify-writes, so one frame at a time adds to them. + CudaDevicePtr key_frames; + CudaDevicePtr key_level_sum; + + // Each pixel's sums are plain read-modify-writes, so the frames add to them one after another: + // every accumulation waits on the device for the one before it (an event, not the host). std::mutex accumulate_mutex; + cudaEvent_t last_accumulate = nullptr; public: // One worker's buffers, on the stream its frames are preprocessed on. struct Frame { explicit Frame(std::shared_ptr stream) : stream(std::move(stream)) {} + // Its frames are queued, not waited for (Add), so the buffers below must not be freed before + // the stream has used them. + ~Frame() { if (stream) cudaStreamSynchronize(*stream); } + Frame(Frame &&) = default; std::shared_ptr stream; CudaDevicePtr count; // valid pixels per key CudaDevicePtr sector_median; // per key @@ -46,21 +66,32 @@ public: CudaDevicePtr ring_ok; }; + // The keys of HotPixelFinder: one per pixel, and where each key's pixels start among the unmasked + // pixels sorted by key. HotPixelFinderGPU(const int32_t *key, size_t npixels, const std::vector &key_begin, int nrings, - int sectors); + int sectors, const HotPixelLevelRules &rules); + ~HotPixelFinderGPU(); + HotPixelFinderGPU(const HotPixelFinderGPU &) = delete; + HotPixelFinderGPU &operator=(const HotPixelFinderGPU &) = delete; - // The lower median of the valid values of each key (count[k] of them, 0 where there are none), and - // of each ring the median and the lower median of the absolute deviations from it. - void Statistics(const int32_t *device_image, Frame &frame, std::vector &count, - std::vector §or_median, std::vector &ring_median, - std::vector &ring_mad); + // Add one frame, preprocessed on frame.stream, as HotPixelFinder::AddImage does. Queued on that + // stream; nothing is waited for on the host. + void Add(const int32_t *device_image, Frame &frame); - // Add the frame to the per-pixel sums, with each key's level and lit threshold and each ring's - // verdict on whether it has a level at all - as HotPixelFinder::AddImage does. - void Accumulate(const int32_t *device_image, Frame &frame, const std::vector &level, - const std::vector &threshold, const std::vector &ring_ok); + // Per ring, over the pixels lit on no more than half of their valid frames: the lit frames and the + // valid frames, summed (HotPixelFinder::GetMask's chance rate). Waits for every frame added. + void ChanceCounts(std::vector &lit, std::vector &seen); - // The per-pixel sums, npixels each. - void Download(uint16_t *n_lit, uint16_t *n_error, int64_t *sum_value, uint16_t *n_error_ring_ok, - int64_t *error_level_sum); + // The pixels that can be masked: those holding the error value on more than half of `frames`, and, + // where spacing_ok, those lit on at least min(max(2, k_chance[ring]), valid frames) of at least + // min_valid valid frames - a persistent pixel is lit on at least that many, because one reflection + // explains at least two. For each, its index and per-pixel sums; and every key's sums. + struct Candidates { + std::vector index; + std::vector n_lit, n_error, n_error_ring_ok; + std::vector sum_value, error_level_sum; + std::vector key_frames; + std::vector key_level_sum; + }; + Candidates GetCandidates(uint32_t frames, int min_valid, bool spacing_ok, const std::vector &k_chance); }; diff --git a/rugnux/Rugnux.cpp b/rugnux/Rugnux.cpp index ca1dab038..aed1270d0 100644 --- a/rugnux/Rugnux.cpp +++ b/rugnux/Rugnux.cpp @@ -57,6 +57,7 @@ #ifdef JFJOCH_USE_CUDA #include "../image_analysis/image_preprocessing/ImagePreprocessorGPU.h" #include "../image_analysis/image_preprocessing/ImagePreprocessorBufferGPU.h" +#include "../image_analysis/spot_finding/AdaptiveSpotFinderGPU.h" #endif #include "../image_analysis/scale_merge/Merge.h" #include "../image_analysis/scale_merge/RfreeFlags.h" @@ -244,10 +245,10 @@ namespace { // as signal. The margin is capped at a tenth of the sweep so a short run still has a sample. constexpr int PRESCAN_END_MARGIN_IMAGES = 5; - // Workers reading the pre-scan sample. Each owns a shard of the beam-stop projection so no two - // threads touch the same accumulator, and a shard costs 20 bytes per pixel - 362 MB on a 16M - // detector - so this is capped well below the worker count of the run proper. The accumulation - // is memory-bound rather than compute-bound, so a handful of workers already saturates it. + // Workers reading the pre-scan sample. Each holds detector-sized buffers of its own, and pages of + // fresh memory are slow to fault in when many threads do it at once, so this is capped well below + // the worker count of the run proper. The beam-stop projection is memory-bound rather than + // compute-bound, so a handful of workers already saturates it. constexpr size_t PRESCAN_MAX_WORKERS = 8; // The spot width is measured on a GROWING share of the pre-scan sample: every eighth frame of @@ -929,12 +930,32 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru struct PreScanWorker { std::vector decompression_buffer; JFJochReaderRawImage raw_image; - std::unique_ptr preprocessor; + std::unique_ptr preprocessor; std::unique_ptr preprocessed; std::unique_ptr spot_finder; }; const auto make_worker = [&] { PreScanWorker w; +#ifdef JFJOCH_USE_CUDA + // On the card where there is one, as in the image loops: the frame is decoded and preprocessed + // there and the spots are found and extracted there. The device finders give the host's spot + // list to the bit - integer ring sums, the same connected components in the same order + // (AdaptiveSpotFinderGPU, SpotExtractorGPU) - so this is a choice of where, not of what. The + // preprocessed image comes back only for the spot width, which reads pixels around the spots. + if (want_spots && get_gpu_count() > 0) { + auto stream = std::make_shared(); + w.preprocessor = std::make_unique(prescan_x, prescan_mask, stream, + /*copy_image_to_host=*/want_width); + w.preprocessed = std::make_unique(prescan_x.GetPixelsNum(), + /*host_mirror=*/want_width); + if (config_.spot_finding.adaptive_threshold) + w.spot_finder = std::make_unique(*prescan_mapping, stream); + else + w.spot_finder = std::make_unique(prescan_x.GetXPixelsNumConv(), + prescan_x.GetYPixelsNumConv(), stream); + return w; + } +#endif if (want_spots) { w.preprocessor = std::make_unique(prescan_x, prescan_mask); w.preprocessed = std::make_unique(prescan_x.GetPixelsNum()); @@ -956,8 +977,20 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru bool for_width, std::vector &curves, std::vector &spot_q) { try { - w.preprocessor->Analyze(*w.preprocessed, - image.GetUncompressedPtr(w.decompression_buffer), image.GetMode()); + // As in the image loops: a frame the device cannot decode goes to the host decoder. + ImageStatistics stats; + bool decoded_on_device = false; + try { + decoded_on_device = w.preprocessor->AnalyzeCompressed(*w.preprocessed, image, stats); + } catch (const JFJochException &e) { + cuda_throw_if_context_lost(); + logger.Warning("Pre-scan: device decoding of image {} failed ({}), decompressing it on " + "the host", image_idx, e.what()); + cuda_clear_error(); + } + if (!decoded_on_device) + w.preprocessor->Analyze(*w.preprocessed, + image.GetUncompressedPtr(w.decompression_buffer), image.GetMode()); } catch (const std::exception &e) { if (IsFatalResourceError(e)) throw; logger.Warning("Pre-scan: failed to preprocess image {}: {}", image_idx, e.what()); @@ -1014,9 +1047,9 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru // Read the sample on several workers. The reader serialises on the HDF5 lock, but the // decompression, the projection and the spot finding - which is all of the cost on a large - // detector - run in parallel. Each worker accumulates into a shard of its own, so nothing is - // locked while an image is added, and the per-frame results are stitched together in sample - // order below so the beam centre sees the same input however the workers interleaved. + // detector - run in parallel. The projection is integer sums, so the order the workers add their + // frames in does not reach it, and the per-frame results are stitched together in sample order + // below so the beam centre sees the same input however the workers interleaved. // // Two passes over the sample. The first builds the projection, which is all the shadow, the // defective pixels and the beam-centre capture below read; the second finds the spots, for the @@ -1026,13 +1059,12 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru const std::vector ordinals(sample.begin(), sample.end()); const size_t nworkers = std::min(std::max(config_.nthreads, 1), std::min(PRESCAN_MAX_WORKERS, ordinals.size())); - finder.SetShardCount(nworkers); { std::atomic next{0}; std::vector> futures; futures.reserve(nworkers); for (size_t t = 0; t < nworkers; t++) - futures.emplace_back(std::async(std::launch::async, [&, t] { + futures.emplace_back(std::async(std::launch::async, [&] { std::vector shadow_buffer; JFJochReaderRawImage raw_image; for (size_t i = next.fetch_add(1); i < ordinals.size(); i = next.fetch_add(1)) { @@ -1057,7 +1089,7 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru msg.image = raw_image.image; msg.number = ordinal; msg.original_number = image_idx; - finder.AddImage(msg, shadow_buffer, t); + finder.AddImage(msg, shadow_buffer); } } })); @@ -1670,6 +1702,11 @@ void Rugnux::MaskDefectivePixels(int start_image, const std::vector &sample const size_t nthreads = static_cast(std::max(config_.nthreads, 1)); HotPixelFinder finder(about, pixel_mask_, nthreads); +#ifdef JFJOCH_USE_CUDA + // Built here, once, so the workers' first frames do not queue behind it. + if (get_gpu_count() > 0) + finder.PrepareDevice(); +#endif std::atomic next{0}; std::vector> futures; const size_t nworkers = std::min(nthreads, std::min(PRESCAN_MAX_WORKERS, sample.size())); @@ -8389,6 +8426,27 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b const auto &twin_sg_opt = experiment_.GetGemmiSpaceGroup(); const gemmi::SpaceGroup *twin_sg = twin_sg_opt ? &*twin_sg_opt : nullptr; + + // Diffraction anisotropy (see where it is reported, below), made beside the analyses that come + // before it: it reads the merge, the integrated observations and the per-frame scales the merge + // wrote back, none of which they change, and most of it is gathering the observations. + std::future anisotropy; + if (!geometry_prepass && !superseded && result.consensus_cell) { + AnisotropyRunInfo aniso_run; + if (sm.statistics.sweep_quality.measured && sm.statistics.sweep_quality.sweep_deg > 0.0f) + aniso_run.observed_rotation_deg = sm.statistics.sweep_quality.sweep_deg; + aniso_run.dose_term_in_scale_model = experiment_.GetScalingSettings().GetCorrectionSurfaces(); + aniso_run.radiation_damage_relative_b = sm.statistics.radiation_damage_delta_b; + const float wedge_deg = experiment_.GetGoniometer() ? experiment_.GetGoniometer()->GetWedge_deg() : 0.0f; + anisotropy = std::async(std::launch::async, + [&, aniso_run, wedge_deg, cell = *result.consensus_cell, + rotation = experiment_.IsRotationIndexing()] { + return AnalyzeAnisotropy(sm.merged, + ScaledObservations(indexer->GetIntegrationOutcome(), rotation, twin_sg, + wedge_deg, 0.5, config_.nthreads), + cell, twin_sg, aniso_run); + }); + } // Not on the geometry pre-pass, nor on a superseded one: the analysis goes into that pass's // statistics text and its written reflections, and neither survives the run. The promotion flag // below is a different thing - it is what the SEARCH did, the second pass reads it, and it is @@ -8619,21 +8677,8 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b // unmerged observations, because a merge has exact Laue symmetry by construction and the // tensor directions the symmetry forbids - the only place a dataset measures its own // systematic error - are identically zero in it. - if (result.consensus_cell) { - AnisotropyRunInfo aniso_run; - if (sm.statistics.sweep_quality.measured && sm.statistics.sweep_quality.sweep_deg > 0.0f) - aniso_run.observed_rotation_deg = sm.statistics.sweep_quality.sweep_deg; - aniso_run.dose_term_in_scale_model = - experiment_.GetScalingSettings().GetCorrectionSurfaces(); - aniso_run.radiation_damage_relative_b = sm.statistics.radiation_damage_delta_b; - sm.statistics.anisotropy = AnalyzeAnisotropy( - sm.merged, - ScaledObservations(indexer->GetIntegrationOutcome(), - experiment_.IsRotationIndexing(), twin_sg, - experiment_.GetGoniometer() - ? experiment_.GetGoniometer()->GetWedge_deg() : 0.0f, - 0.5, config_.nthreads), - *result.consensus_cell, twin_sg, aniso_run); + if (anisotropy.valid()) { + sm.statistics.anisotropy = anisotropy.get(); stats_text << AnisotropyToText(sm.statistics.anisotropy) << "\n"; } } @@ -8869,8 +8914,10 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b Logger held = Logger::Buffered(); auto pending = std::async(std::launch::async, validate, std::ref(held)); experiment_.SpaceGroupNumber(1); + rsm->SetWriteBackPerFrameScale(false); p1_merged_early = rsm->Run(/*for_search=*/false, /*full_stats=*/true, /*measure_cc_before_corrections=*/false); + rsm->SetWriteBackPerFrameScale(true); experiment_.SetSpaceGroup(data_sg); validation = pending.get(); held.ReplayInto(logger); @@ -9046,6 +9093,9 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b const double em_a = result.error_model_a; const double em_b = result.error_model_b; const auto res_fit = result.resolution_fit_A; + const int iter_partials = result.scaling_iterations_partials; + const int iter_fulls = result.scaling_iterations_fulls; + const bool converged = result.scaling_converged; // The unmerged MTZ below does not depend on this merge, so it is built meanwhile, from // the experiment as it stands in the determined group. The merge rewrites each image's // mosaicity, which the file's batch headers carry, so that is filled in only after it. @@ -9062,7 +9112,10 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b // Both the merge and the MTZ read the group from the experiment, so it is set for // the whole of it and restored after. experiment_.SpaceGroupNumber(1); + if (!p1_merged_early) + rsm->SetWriteBackPerFrameScale(false); auto p1 = scale_and_merge("P1 cross-check", false, false, std::move(p1_merged_early)); + rsm->SetWriteBackPerFrameScale(true); // The scaler still holds the observations in the indexing they were merged in, so every // relabelling since (the written setting, the model's indexing) is applied to this merge // too - or it would describe the dataset on other axes than the merged output beside it, @@ -9135,6 +9188,9 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b result.error_model_a = em_a; result.error_model_b = em_b; result.resolution_fit_A = res_fit; + result.scaling_iterations_partials = iter_partials; + result.scaling_iterations_fulls = iter_fulls; + result.scaling_converged = converged; if (determined != nullptr && determined->number > 1) logger.Info("P1 cross-check dataset written to {} ({} unique reflections): the " "same observations merged in P1 instead of {}, so a wrong space group " diff --git a/tests/AdaptiveSpotFinderGPUTest.cpp b/tests/AdaptiveSpotFinderGPUTest.cpp index 87d9d9e44..efee443fd 100644 --- a/tests/AdaptiveSpotFinderGPUTest.cpp +++ b/tests/AdaptiveSpotFinderGPUTest.cpp @@ -12,6 +12,7 @@ #include "../common/AzimuthalIntegrationMapping.h" #include "../common/AzimuthalIntegrationProfile.h" +#include "../image_analysis/azint/AzIntEngineCPU.h" #include "../image_analysis/azint/AzIntEngineGPU.h" #include "../image_analysis/spot_finding/AdaptiveSpotFinderCPU.h" #include "../image_analysis/spot_finding/AdaptiveSpotFinderGPU.h" @@ -161,6 +162,46 @@ TEST_CASE("AdaptiveSpotFinderGPU_AzimuthalIntegration", "[AdaptiveSpotFinderGPU] } } +// The GPU azimuthal integration against the CPU one, on a pixel count that is not a multiple of four (the +// kernel reads four pixels at a time and does the rest one by one) and with masked and saturated pixels +// in it. The per-ring pixel counts are integers and must agree exactly; the float sums only to rounding. +TEST_CASE("AzIntEngineGPU_MatchesCPU", "[AdaptiveSpotFinderGPU]") { + if (get_gpu_count() == 0) { + WARN("No CUDA GPU present. Skipping AzIntEngineGPU_MatchesCPU"); + return; + } + + DiffractionExperiment x(DetDECTRIS(1031, 1063, "Test", {})); + x.DetectorDistance_mm(80).BeamX_pxl(515).BeamY_pxl(530); + x.QSpacingForAzimInt_recipA(0.05).QRangeForAzimInt_recipA(0.05, 5.0); + REQUIRE(x.GetPixelsNum() % 4 != 0); + PixelMask pixel_mask(x); + AzimuthalIntegrationMapping mapping(x, pixel_mask); + + ImagePreprocessorBufferGPU buffer(x.GetPixelsNum()); + for (size_t i = 0; i < x.GetPixelsNum(); i++) + buffer[i] = (i % 997 == 0) ? INT32_MIN : (i % 1009 == 0) ? INT32_MAX + : 8 + static_cast((i * 7919) % 23); + REQUIRE(cudaMemcpy(buffer.getGPUBuffer(), buffer.getBuffer().data(), + x.GetPixelsNum() * sizeof(int32_t), cudaMemcpyHostToDevice) == cudaSuccess); + REQUIRE(cudaDeviceSynchronize() == cudaSuccess); + + AzimuthalIntegrationProfile cpu_profile(mapping), gpu_profile(mapping); + AzIntEngineCPU(mapping).Run(buffer, cpu_profile); + AzIntEngineGPU(mapping, std::make_shared()).Run(buffer, gpu_profile); + + REQUIRE(gpu_profile.GetPixelCount() == cpu_profile.GetPixelCount()); + const auto ref = cpu_profile.GetResult(); + const auto got = gpu_profile.GetResult(); + REQUIRE(ref.size() == got.size()); + for (size_t b = 0; b < ref.size(); b++) { + if (std::isnan(ref[b])) + CHECK(std::isnan(got[b])); + else + CHECK(got[b] == Catch::Approx(ref[b]).epsilon(1e-5)); + } +} + // The ring sums are built by atomics, which arrive in an arbitrary order, so the same frame has to be // re-run to show the engine agrees with itself: detection is a hard "value >= threshold" on integer // counts, and a threshold that wobbles between runs flips pixels on the boundary and with them the size diff --git a/tests/BeamCenterFromBackgroundTest.cpp b/tests/BeamCenterFromBackgroundTest.cpp index 73a547e92..d38db14e2 100644 --- a/tests/BeamCenterFromBackgroundTest.cpp +++ b/tests/BeamCenterFromBackgroundTest.cpp @@ -307,3 +307,29 @@ TEST_CASE("FindBeamCenter_BlanksTheBeamStopOutOfTheCapture", "[BeamCenter]") { CHECK(masked_error < 12.0f); CHECK(masked_error <= unmasked_error); } + +#ifdef JFJOCH_USE_CUDA +#include "../common/CUDAWrapper.h" + +// With a GPU the passes over the pixels run on it. The cell a pixel lands in is computed from IEEE +// operations alone and without contraction on both sides, and each cell is summed in the host's order, +// so the walk ends on the same bits - a tilted detector and an offset centre included. +TEST_CASE("BeamCenterFromBackground_DeviceMatchesHost", "[BeamCenter]") { + if (get_gpu_count() == 0) + SKIP("no GPU"); + DiffractionExperiment x = TestExperiment(); + x.PoniRot1_rad(0.005f).PoniRot2_rad(-0.003f); + PixelMask pixel_mask(x); + + const DiffractionGeometry geom_true = OffsetBy(x.GetDiffractionGeometry(), 7.0f, -4.5f); + const auto projection = SynthesiseProjection(x, pixel_mask, geom_true, 60.0f, 0.5f); + const auto host = FindBeamCenterFromBackground(x, pixel_mask, projection, 0, {}, /*allow_device=*/false); + const auto device = FindBeamCenterFromBackground(x, pixel_mask, projection, 0, {}, /*allow_device=*/true); + + REQUIRE(host.has_value()); + REQUIRE(device.has_value()); + CHECK(device->beam_x_pxl == host->beam_x_pxl); + CHECK(device->beam_y_pxl == host->beam_y_pxl); + CHECK(device->sigma_pxl == host->sigma_pxl); +} +#endif diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index f01b89506..c6d6df0bb 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -112,6 +112,8 @@ ADD_EXECUTABLE(jfjoch_test RingsFromProfileTest.cpp CalibrationTest.cpp XtalOptimizerTest.cpp + XtalRefineTest.cpp + XtalRefineCeres.h CrystalLatticeTest.cpp FPGAPTPTest.cpp ResolutionShellsTest.cpp @@ -146,6 +148,7 @@ ADD_EXECUTABLE(jfjoch_test AnisotropyAnalysisTest.cpp ModelScalingTest.cpp ModelScaleGPUTest.cpp + CorrectionSurfaceGPUTest.cpp TwinningAnalysisTest.cpp TranslationalNCSTest.cpp RfreeFlagsTest.cpp diff --git a/tests/CorrectionSurfaceGPUTest.cpp b/tests/CorrectionSurfaceGPUTest.cpp new file mode 100644 index 000000000..bfe767db0 --- /dev/null +++ b/tests/CorrectionSurfaceGPUTest.cpp @@ -0,0 +1,167 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#include +#include "../common/CUDAWrapper.h" + +#ifdef JFJOCH_USE_CUDA + +#include +#include +#include +#include + +#include "../common/ParallelFor.h" +#include "../image_analysis/scale_merge/RotationScaleMergeGPU.h" + +namespace { + +using Term = RotationScaleMergeGPU::SurfaceTerm; + +// The roundings RotationScaleMerge::ApplyCellSurface makes on the host (x86-64-v3 build), spelled out: +// a volatile result is rounded on its own and never fused into the next operation, std::fma is fused. +double Mul(double a, double b) { volatile double r = a * b; return r; } +double Add(double a, double b) { volatile double r = a + b; return r; } + +constexpr int SURFACE_BLOCK = 32768; // ApplyCellSurface's reduction block + +struct Surface { + int n_groups = 0, ncell = 0; + std::vector term; + std::vector parity; + std::vector gperm, gstart; + std::vector sel[3]; // even, odd, all - in term order +}; + +Surface MakeSurface(int n_terms, int n_groups, int ncell, uint32_t seed) { + std::mt19937 rng(seed); + std::uniform_int_distribution group(0, n_groups - 1), cell(0, ncell - 1), bit(0, 1); + std::uniform_real_distribution u(0.0f, 1.0f); + Surface s; + s.n_groups = n_groups; + s.ncell = ncell; + for (int i = 0; i < n_terms; ++i) { + // Negative intensities, and now and then a sigma of zero: both reach the host sums as they are. + const float I = 1000.0f * u(rng) - 100.0f; + const float sigma = (i % 997 == 0) ? 0.0f : 1.0f + 30.0f * u(rng); + // Every 50th group gets no terms at all, so its reference is empty. + int g = group(rng); + if (g % 50 == 0) g = (g + 1) % n_groups; + s.term.push_back({I, sigma, 0.5f + u(rng), 1.0f + 3.0f * u(rng), cell(rng), g}); + s.parity.push_back(static_cast(bit(rng))); + s.sel[s.parity.back()].push_back(i); + s.sel[2].push_back(i); + } + s.gstart.assign(n_groups + 1, 0); + for (const Term &t : s.term) ++s.gstart[t.group + 1]; + for (int g = 0; g < n_groups; ++g) s.gstart[g + 1] += s.gstart[g]; + s.gperm.resize(n_terms); + std::vector fill(s.gstart.begin(), s.gstart.end() - 1); + for (int i = 0; i < n_terms; ++i) s.gperm[fill[s.term[i].group]++] = i; + return s; +} + +void HostReference(const Surface &s, int parity, const std::vector &A, + std::vector &sw, std::vector &swI) { + sw.assign(s.n_groups, 0.0); + swI.assign(s.n_groups, 0.0); + for (int g = 0; g < s.n_groups; ++g) { + double s_w = 0.0, s_wI = 0.0; + for (int k = s.gstart[g]; k < s.gstart[g + 1]; ++k) { + const int i = s.gperm[k]; + if (parity >= 0 && s.parity[i] != parity) continue; + const Term &t = s.term[i]; + const double a = A[t.cell]; + const double Is = Mul(Mul(t.I, t.corr), a), sc = Mul(Mul(t.sigma, t.corr), a); + const double w = 1.0 / Mul(sc, sc); + s_w = Add(s_w, w); + s_wI = parity >= 0 ? std::fma(Is, w, s_wI) : Add(s_wI, Mul(Is, w)); + } + sw[g] = s_w; + swI[g] = s_wI; + } +} + +void HostFitSums(const Surface &s, const std::vector &sel, const std::vector &A, + const std::vector &sw, const std::vector &swI, + std::vector &cross, std::vector &ref2) { + cross.assign(s.ncell, 0.0); + ref2.assign(s.ncell, 0.0); + const int n = static_cast(sel.size()), nb = ReductionBlocks(n, SURFACE_BLOCK); + for (int b = 0; b < nb; ++b) { + std::vector xcross(s.ncell, 0.0), xref2(s.ncell, 0.0); + const int lo = static_cast(int64_t(n) * b / nb), hi = static_cast(int64_t(n) * (b + 1) / nb); + for (int k = lo; k < hi; ++k) { + const Term &t = s.term[sel[k]]; + if (sw[t.group] <= 0.0) continue; + const double Iref = swI[t.group] / sw[t.group], a = A[t.cell]; + const double Is = Mul(Mul(t.I, t.corr), a), sc = Mul(Mul(t.sigma, t.corr), a); + if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue; + const double w = 1.0 / Mul(sc, sc); + xcross[t.cell] = std::fma(Mul(w, Is), Iref, xcross[t.cell]); + xref2[t.cell] = std::fma(Mul(w, Iref), Iref, xref2[t.cell]); + } + for (int c = 0; c < s.ncell; ++c) { + cross[c] = Add(cross[c], xcross[c]); + ref2[c] = Add(ref2[c], xref2[c]); + } + } +} + +// The device side of one subset: ApplyCellSurface's per-block counting sort by cell. +void UploadSubset(RotationScaleMergeGPU &gpu, const Surface &s, int id, const std::vector &sel) { + const int n = static_cast(sel.size()), nb = ReductionBlocks(n, SURFACE_BLOCK); + std::vector perm(n), seg_start(static_cast(nb) * s.ncell + 1, n); + for (int b = 0; b < nb; ++b) { + const int lo = static_cast(int64_t(n) * b / nb), hi = static_cast(int64_t(n) * (b + 1) / nb); + std::vector pos(s.ncell + 1, 0); + for (int k = lo; k < hi; ++k) ++pos[s.term[sel[k]].cell + 1]; + for (int c = 0; c < s.ncell; ++c) pos[c + 1] += pos[c]; + for (int c = 0; c < s.ncell; ++c) seg_start[size_t(b) * s.ncell + c] = lo + pos[c]; + for (int k = lo; k < hi; ++k) perm[lo + pos[s.term[sel[k]].cell]++] = sel[k]; + } + gpu.SurfaceSetSubset(id, nb, perm.data(), seg_start.data()); +} + +bool SameBits(const std::vector &a, const std::vector &b) { + return a.size() == b.size() && std::memcmp(a.data(), b.data(), a.size() * sizeof(double)) == 0; +} + +} // namespace + +TEST_CASE("CorrectionSurfaceGPU_SumsMatchHostBitForBit", "[RotationScale][gpu]") { + if (get_gpu_count() == 0) + SKIP("No GPU"); + // Enough terms for several reduction blocks in every subset. + const Surface s = MakeSurface(250000, 4000, 144, 7); + RotationScaleMergeGPU gpu; + REQUIRE(gpu.Available()); + gpu.SurfaceSetTerms(static_cast(s.term.size()), s.term.data(), s.parity.data(), s.n_groups, + s.gperm.data(), s.gstart.data(), s.ncell); + for (int id = 0; id < 3; ++id) + UploadSubset(gpu, s, id, s.sel[id]); + + std::mt19937 rng(11); + std::uniform_real_distribution u(0.7, 1.4); + std::vector A(s.ncell); + for (double &a : A) a = u(rng); + + for (int parity : {0, 1, -1}) { + const int id = parity < 0 ? 2 : parity; + std::vector sw, swI, cross, ref2; + HostReference(s, parity, A, sw, swI); + HostFitSums(s, s.sel[id], A, sw, swI, cross, ref2); + REQUIRE(ReductionBlocks(static_cast(s.sel[id].size()), SURFACE_BLOCK) > 1); + + gpu.SurfaceReference(parity, A.data()); + std::vector dsw(s.n_groups), dswI(s.n_groups), dcross(s.ncell), dref2(s.ncell); + gpu.SurfaceGetReference(dsw.data(), dswI.data()); + gpu.SurfaceFitSums(id, dcross.data(), dref2.data()); + CHECK(SameBits(sw, dsw)); + CHECK(SameBits(swI, dswI)); + CHECK(SameBits(cross, dcross)); + CHECK(SameBits(ref2, dref2)); + } +} + +#endif diff --git a/tests/ShadowFinderTest.cpp b/tests/ShadowFinderTest.cpp index 99473260e..f01e2c0b4 100644 --- a/tests/ShadowFinderTest.cpp +++ b/tests/ShadowFinderTest.cpp @@ -8,6 +8,7 @@ #include #include #include +#include #include #include "../common/DetectorSetup.h" @@ -168,16 +169,15 @@ TEST_CASE("ShadowFinder_MaskDoesNotDependOnTheThreadCount", "[ShadowFinder]") { CHECK(finder.GetMask(8) == one); } -// Workers accumulate into shards of their own and the shards are summed when the projection is read, -// so which worker saw which frame must not reach the answer - including the maximum, which only one -// shard holds when the reflection is on a single frame. -TEST_CASE("ShadowFinder_ShardingDoesNotChangeTheProjection", "[ShadowFinder]") { +// Workers add their frames concurrently, each starting at a different band of the projection, so +// which worker added which frame, and in what order, must not reach the answer - including the +// maximum, which one frame alone holds when the reflection is on a single frame. +TEST_CASE("ShadowFinder_ConcurrentWorkersDoNotChangeTheProjection", "[ShadowFinder]") { const DiffractionExperiment x = TestExperiment(); const PixelMask pixel_mask(x); ShadowFinder serial(x, pixel_mask); - ShadowFinder sharded(x, pixel_mask); - sharded.SetShardCount(4); + ShadowFinder concurrent(x, pixel_mask); std::vector> frames; std::vector buffer; @@ -185,22 +185,32 @@ TEST_CASE("ShadowFinder_ShardingDoesNotChangeTheProjection", "[ShadowFinder]") { frames.push_back(Scene(/*cross=*/false, /*reflection=*/f == 0)); DataMessage msg{}; msg.image = CompressedImage(frames.back(), W, H); - serial.AddImage(msg, buffer, 0); - sharded.AddImage(msg, buffer, static_cast(f) % 4); + serial.AddImage(msg, buffer); } + std::vector workers; + for (int t = 0; t < 4; t++) + workers.emplace_back([&, t] { + std::vector worker_buffer; + for (int f = NFRAMES - 1 - t; f >= 0; f -= 4) { + DataMessage msg{}; + msg.image = CompressedImage(frames[f], W, H); + concurrent.AddImage(msg, worker_buffer); + } + }); + for (auto &w : workers) w.join(); - CHECK(serial.GetFrameCount() == sharded.GetFrameCount()); + CHECK(serial.GetFrameCount() == concurrent.GetFrameCount()); const auto a = serial.GetMeanProjection(); - const auto b = sharded.GetMeanProjection(); + const auto b = concurrent.GetMeanProjection(); REQUIRE(a.size() == b.size()); // NAN marks a pixel nothing counted, and NAN != NAN, so compare the bits rather than the values. CHECK(memcmp(a.data(), b.data(), a.size() * sizeof(float)) == 0); - // The reflection is on one frame, so its maximum lives in a single shard. If the fold lost it, - // the mask would swallow the reflection instead of giving it back. - CHECK(serial.GetMask(1) == sharded.GetMask(1)); - CHECK(sharded.GetMask(1)[I(C - 14, C - 2)] == 0); + // The reflection is on one frame, so only that frame's maximum sees it. If it were lost, the mask + // would swallow the reflection instead of giving it back. + CHECK(serial.GetMask(1) == concurrent.GetMask(1)); + CHECK(concurrent.GetMask(1)[I(C - 14, C - 2)] == 0); } // Four opaque arms and a centred disk: the scene is invariant under a quarter turn, so the mask must @@ -461,3 +471,91 @@ TEST_CASE("ShadowFinder_ABrightRingIsNotABeamStop", "[ShadowFinder]") { // Anything much beyond the stop, its arm and their penumbra means the walk ran away. CHECK(std::count(mask.begin(), mask.end(), 1u) < 6000); } + +#ifdef JFJOCH_USE_CUDA +#include "../common/CUDAWrapper.h" +#include "../compression/JFJochCompressor.h" + +namespace { + // The mask and the mean projection of the same frames, once from the host projection (the frames + // handed over uncompressed) and once from the device's (handed over as bitshuffle+LZ4, which is + // what sends them to the GPU). + struct HostAndDevice { + std::vector host_mask, device_mask; + std::vector host_mean, device_mean; + }; + + HostAndDevice MaskBothWays(const DiffractionExperiment &x, const std::vector> &frames) { + const PixelMask pixel_mask(x); + HostAndDevice out; + std::vector buffer; + { + ShadowFinder finder(x, pixel_mask); + for (const auto &frame : frames) { + DataMessage msg{}; + msg.image = CompressedImage(frame, W, H); + finder.AddImage(msg, buffer); + } + out.host_mask = finder.GetMask(); + out.host_mean = finder.GetMeanProjection(); + } + { + ShadowFinder finder(x, pixel_mask); + JFJochBitShuffleCompressor compressor(CompressionAlgorithm::BSHUF_LZ4); + std::vector> compressed; + for (const auto &frame : frames) { + compressed.push_back(compressor.Compress(frame)); + DataMessage msg{}; + msg.image = CompressedImage(compressed.back().data(), compressed.back().size(), W, H, + CompressedImageMode::Int32, CompressionAlgorithm::BSHUF_LZ4); + finder.AddImage(msg, buffer); + } + out.device_mask = finder.GetMask(); + out.device_mean = finder.GetMeanProjection(); + } + return out; + } +} + +// The mask is made on the GPU wherever the projection is there. On scenes where no pixel sits within a +// rounding of a threshold the two must agree to the pixel: every step but the polarization factor, +// the Poisson test's logarithm and the arm search's azimuth is exact on both, and these scenes are +// built so that none of the three decides anything at an edge. +TEST_CASE("ShadowFinder_DeviceMaskMatchesHost", "[ShadowFinder]") { + if (get_gpu_count() == 0) + SKIP("no GPU"); + + SECTION("a beam stop with a reflection behind it") { + std::vector> frames; + for (int f = 0; f < NFRAMES; f++) + frames.push_back(Scene(false, f == 0)); + const auto r = MaskBothWays(TestExperiment(), frames); + CHECK(std::count(r.host_mask.begin(), r.host_mask.end(), 1u) == 2612); + CHECK(r.device_mask == r.host_mask); + CHECK(std::memcmp(r.device_mean.data(), r.host_mean.data(), r.host_mean.size() * sizeof(float)) == 0); + } + + SECTION("an arm that lets part of the beam through, across a module gap") { + constexpr int32_t BRIGHT = 50; + constexpr int ARM_HALF_WIDE = 15, GAP_X0 = 200, GAP_X1 = 216, OPAQUE_FROM_X = 232; + std::vector> frames; + for (int f = 0; f < NFRAMES; f++) { + frames.emplace_back(static_cast(W) * H, BRIGHT); + auto &frame = frames.back(); + for (int y = 0; y < H; y++) + for (int xi = 0; xi < W; xi++) { + const int dx = xi - C, dy = y - C; + if (dx * dx + dy * dy <= STOP_R * STOP_R) + frame[I(xi, y)] = 0; + else if (dx >= 0 && std::abs(dy) <= ARM_HALF_WIDE) + frame[I(xi, y)] = xi >= OPAQUE_FROM_X ? 0 : BRIGHT * 6 / 10; + if (xi >= GAP_X0 && xi <= GAP_X1) + frame[I(xi, y)] = INT32_MIN; + } + } + const auto r = MaskBothWays(TestExperiment(), frames); + CHECK(std::count(r.host_mask.begin(), r.host_mask.end(), ShadowFinder::TRANSMITTING) > 0); + CHECK(r.device_mask == r.host_mask); + } +} +#endif diff --git a/tests/XtalRefineCeres.h b/tests/XtalRefineCeres.h new file mode 100644 index 000000000..77f34b4f9 --- /dev/null +++ b/tests/XtalRefineCeres.h @@ -0,0 +1,92 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +#pragma once + +// The reference the LM solver of XtalRefine is checked against: the same XtalRefineProblem handed to +// Ceres exactly as XtalOptimizer used to build it - the same residual functors, losses, priors, bounds, +// manifold and options. + +#include "ceres/ceres.h" +#include "../image_analysis/geom_refinement/XtalRefine.h" + +struct XtalRefineCeresPrior { + XtalRefineCeresPrior(double gx, double gy, double p0, double weight) + : gx(gx), gy(gy), p0(p0), weight(weight) {} + template + bool operator()(const T *const p, T *residual) const { + residual[0] = T(weight) * (T(gx) * p[0] + T(gy) * p[1] - T(p0)); + return true; + } + double gx, gy, p0, weight; +}; + +inline ceres::Solver::Summary SolveXtalRefineCeres(XtalRefineProblem &p, int num_threads) { + std::vector frame_const; + frame_const.reserve(p.frame_angle_rad.size()); + if (p.beam_and_orientation_only) + for (const double angle: p.frame_angle_rad) + frame_const.emplace_back(p.detector_rot, p.rot_vec, angle, p.latt_vec1, p.latt_vec2, p.crystal_system); + + ceres::Problem problem; + for (size_t i = 0; i < p.residuals.size(); i++) { + ceres::LossFunction *loss = p.weight_sq.empty() + ? nullptr + : new ceres::ScaledLoss(nullptr, p.weight_sq[i], ceres::TAKE_OWNERSHIP); + if (p.beam_and_orientation_only) + problem.AddResidualBlock( + new ceres::AutoDiffCostFunction( + new XtalResidualBeamOrientation(p.residuals[i], p.distance_mm, frame_const[p.frame[i]])), + loss, p.beam, p.latt_vec0); + else + problem.AddResidualBlock( + new ceres::AutoDiffCostFunction( + new XtalResidualFixedDistance(p.residuals[i], p.distance_mm)), + loss, p.beam, p.detector_rot, p.rot_vec, p.latt_vec0, p.latt_vec1, p.latt_vec2); + } + for (const auto &prior: p.priors) + problem.AddResidualBlock( + new ceres::AutoDiffCostFunction( + new XtalRefineCeresPrior(prior.gx, prior.gy, prior.p0, prior.weight)), + nullptr, prior.block == XtalRefinePrior::Block::Beam ? p.beam : p.detector_rot); + + const auto bounds = [&](double *block, const double *lo, const double *hi, int n) { + for (int i = 0; i < n; i++) { + if (lo[i] > -XtalRefineProblem::kNoBound) + problem.SetParameterLowerBound(block, i, lo[i]); + if (hi[i] < XtalRefineProblem::kNoBound) + problem.SetParameterUpperBound(block, i, hi[i]); + } + }; + if (p.beam_constant) + problem.SetParameterBlockConstant(p.beam); + if (!p.beam_and_orientation_only) { + if (p.detector_rot_constant) + problem.SetParameterBlockConstant(p.detector_rot); + else + bounds(p.detector_rot, p.detector_rot_lower, p.detector_rot_upper, 2); + if (p.rot_vec_constant) + problem.SetParameterBlockConstant(p.rot_vec); + else + problem.SetManifold(p.rot_vec, new ceres::SphereManifold<3>); + if (p.latt_vec1_constant) + problem.SetParameterBlockConstant(p.latt_vec1); + else + bounds(p.latt_vec1, p.latt_vec1_lower, p.latt_vec1_upper, 3); + if (p.latt_vec2_constant) + problem.SetParameterBlockConstant(p.latt_vec2); + else + bounds(p.latt_vec2, p.latt_vec2_lower, p.latt_vec2_upper, 3); + } + + ceres::Solver::Options options; + options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY; + options.minimizer_progress_to_stdout = false; + options.max_num_iterations = p.options.max_iterations; + options.max_solver_time_in_seconds = p.options.max_time_s; + options.logging_type = ceres::LoggingType::SILENT; + options.num_threads = num_threads; + ceres::Solver::Summary summary; + ceres::Solve(options, &problem, &summary); + return summary; +} diff --git a/tests/XtalRefineTest.cpp b/tests/XtalRefineTest.cpp new file mode 100644 index 000000000..a26021f91 --- /dev/null +++ b/tests/XtalRefineTest.cpp @@ -0,0 +1,127 @@ +// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute +// SPDX-License-Identifier: GPL-3.0-only + +// Ceres first: its logging header defines a CHECK macro of its own, which Catch's must replace here. +#include "XtalRefineCeres.h" +#undef CHECK +#include + +#include "../image_analysis/geom_refinement/LatticeReduction.h" +#include "../image_analysis/bragg_prediction/BraggPrediction.h" + +// The LM solver of XtalRefine is meant to take Ceres' path to Ceres' answer: same steps accepted, same +// stopping rule, same point. These cases hand one problem to both and compare. +namespace { + // A monoclinic crystal rotated about X over ten 3-degree frames, its predicted spots as observations; + // the refinement starts from a perturbed beam, tilt and cell. + XtalRefineProblem RotationProblem(bool reduced, bool weighted) { + DiffractionExperiment exp; + exp.IncidentEnergy_keV(WVL_1A_IN_KEV).BeamX_pxl(1000).BeamY_pxl(1000) + .PoniRot1_rad(0.01).PoniRot2_rad(0.02).DetectorDistance_mm(200); + const auto geom = exp.GetDiffractionGeometry(); + const CrystalLattice latt(40, 50, 80, 90, 95, 90); + const GoniometerAxis axis("omega", 0.0f, 3.0f, Coord(1, 0, 0), std::nullopt); + const gemmi::CrystalSystem sys = gemmi::CrystalSystem::Monoclinic; + + XtalRefineProblem p; + p.crystal_system = sys; + p.beam_and_orientation_only = reduced; + p.distance_mm = 200; + + BraggPrediction prediction; + const BraggPredictionSettings settings{.high_res_A = 1.5, .ewald_dist_cutoff = 0.002}; + for (int img = 0; img < 10; img++) { + const float angle_deg = axis.GetAngle_deg(img) + axis.GetWedge_deg() / 2.0f; + const auto n = prediction.Calc(exp, latt.Multiply(axis.GetTransformationAngle(angle_deg).transpose()), + settings); + p.frame_angle_rad.push_back(angle_deg * PI / 180.0); + for (int i = 0; i < n; i++) { + const auto &r = prediction.GetReflections().at(i); + p.residuals.emplace_back(r.predicted_x + 0.3 * std::sin(i), r.predicted_y + 0.3 * std::cos(i), + geom.GetWavelength_A(), geom.GetPixelSize_mm(), 1.0, 0.0, + angle_deg * PI / 180.0, r.h, r.k, r.l, sys); + p.frame.push_back(img); + if (weighted) + p.weight_sq.push_back(0.2 + 0.6 * (i % 5) / 4.0); + } + } + + p.beam[0] = 1000.0; + p.beam[1] = 997.0; + p.detector_rot[0] = 0.012; + p.detector_rot[1] = 0.018; + p.rot_vec[0] = 1.0; + p.rot_vec[1] = 0.0; + p.rot_vec[2] = 0.0; + double beta = 0; + LatticeToRodriguesLengthsBeta_Mono(CrystalLattice(39.7f, 50.6f, 79.6f, 90.0f, 94.5f, 90.0f), + p.latt_vec0, p.latt_vec1, beta); + p.latt_vec2[0] = beta; + + if (!reduced) { + p.detector_rot_constant = false; + p.rot_vec_constant = false; + p.latt_vec1_constant = false; + p.latt_vec2_constant = false; + for (int i = 0; i < 2; i++) { + p.detector_rot_lower[i] = p.detector_rot[i] - 0.05; + p.detector_rot_upper[i] = p.detector_rot[i] + 0.05; + } + for (int i = 0; i < 3; i++) { + p.latt_vec1_lower[i] = 5.0; + p.latt_vec1_upper[i] = 100.0; + } + p.latt_vec2_lower[0] = PI / 3; + p.latt_vec2_upper[0] = 2 * PI / 3; + p.priors.push_back({XtalRefinePrior::Block::Beam, 1.0, 0.0, p.beam[0], 0.5}); + p.priors.push_back({XtalRefinePrior::Block::DetectorRot, 0.0, 1.0, p.detector_rot[1], 50.0}); + } + p.options.max_iterations = 50; + return p; + } + + void CompareWithCeres(XtalRefineProblem p) { + XtalRefineProblem q = p; + const LMSummary lm = SolveXtalRefine(p, 4); + const ceres::Solver::Summary ref = SolveXtalRefineCeres(q, 4); + + REQUIRE(lm.IsSolutionUsable() == ref.IsSolutionUsable()); + CHECK(lm.iterations == static_cast(ref.iterations.size())); + CHECK(lm.final_cost == Catch::Approx(ref.final_cost).epsilon(1e-9)); + const auto same = [](const double *a, const double *b, int n, double tol) { + for (int i = 0; i < n; i++) + CHECK(a[i] == Catch::Approx(b[i]).margin(tol)); + }; + same(p.beam, q.beam, 2, 1e-7); + same(p.detector_rot, q.detector_rot, 2, 1e-10); + same(p.rot_vec, q.rot_vec, 3, 1e-10); + same(p.latt_vec0, q.latt_vec0, 3, 1e-10); + same(p.latt_vec1, q.latt_vec1, 3, 1e-8); + same(p.latt_vec2, q.latt_vec2, 3, 1e-10); + } +} + +TEST_CASE("XtalRefine_matches_Ceres_full", "[XtalOptimizer]") { + CompareWithCeres(RotationProblem(false, false)); +} + +TEST_CASE("XtalRefine_matches_Ceres_full_weighted", "[XtalOptimizer]") { + CompareWithCeres(RotationProblem(false, true)); +} + +TEST_CASE("XtalRefine_matches_Ceres_beam_orientation", "[XtalOptimizer]") { + CompareWithCeres(RotationProblem(true, false)); +} + +TEST_CASE("XtalRefine_same_answer_at_any_thread_count", "[XtalOptimizer]") { + XtalRefineProblem a = RotationProblem(false, false); + XtalRefineProblem b = a; + SolveXtalRefine(a, 1); + SolveXtalRefine(b, 7); + for (int i = 0; i < 3; i++) { + CHECK(a.latt_vec0[i] == b.latt_vec0[i]); + CHECK(a.latt_vec1[i] == b.latt_vec1[i]); + } + CHECK(a.beam[0] == b.beam[0]); + CHECK(a.beam[1] == b.beam[1]); +}