Merge perf-merge: rotation performance round (own LM solver, GPU/parallel pre-scan, tail on own GPU stream, GPU correction surfaces, CPU spot-finder/prediction, faster GPU azint kernel)

Battery: 20261003-1424_581e1c_perf-merge-full (+_private) vs rc174 built with the same flags:
no verdict change except the second scaling engine's GPU OOM on the three largest sets,
removed in 0f728ecab (8a1a/8qaw/8tyy pass again: 20261003-2032_581e1c_perf-oom-fix).
One-time result change from the GPU/CPU-parity beam-centre walk (b2).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
This commit is contained in:
2026-10-03 21:08:43 +02:00
co-authored by Claude Opus 5.5
43 changed files with 4592 additions and 1113 deletions
+6 -1
View File
@@ -50,7 +50,7 @@ either way. Eigen is header-only: only its headers reach the binaries, and no Ei
## Vendored directly in the repository
These live in the source tree (see the path) rather than being fetched; traccc is the exception - code adapted into first-party files rather than a vendored directory, see the note at the end of this file.
These live in the source tree (see the path) rather than being fetched; traccc and the Ceres-derived minimiser are the exceptions - code adapted into first-party files rather than a vendored directory, see the notes at the end of this file.
| Component | Path | Copyright | License (SPDX) | License text |
|---|---|---|---|---|
@@ -69,6 +69,7 @@ These live in the source tree (see the path) rather than being fetched; traccc i
| [pocketfft](https://github.com/mreineck/pocketfft) | `gemmi_gph/gemmi/third_party/pocketfft_hdronly.h` | Max-Planck-Society; Peter Bell; MIT (FFTW-derived parts) | BSD-3-Clause | [pocketfft.txt](licenses/pocketfft.txt) |
| [tinydir](https://github.com/cxong/tinydir) | `gemmi_gph/gemmi/third_party/tinydir.h` | Cong Xu, Lautis Sun, Baudouin Feildel, Andargor | BSD-2-Clause | [tinydir.txt](licenses/tinydir.txt) |
| [traccc (ACTS)](https://github.com/acts-project/traccc) | `image_analysis/spot_finding/StrongPixelSet.cpp`, `SpotExtractorGPU.cu` | CERN, for the benefit of the ACTS project | MPL-2.0 | [traccc.txt](licenses/traccc.txt) |
| [Ceres Solver](https://github.com/ceres-solver/ceres-solver) (adapted) | `image_analysis/geom_refinement/LMSolver.h`, `LMSolver.cpp` | Google Inc. | BSD-3-Clause | [ceres-solver.txt](licenses/ceres-solver.txt) |
| [xbflash.qspi](https://github.com/Xilinx/XRT) | `tools/xbflash.qspi/` | Xilinx / AMD | Apache-2.0 | [xbflash-qspi.txt](licenses/xbflash-qspi.txt) |
| [wingetopt](https://github.com/alex85k/wingetopt) | `tools/wingetopt/` | Todd C. Miller; The NetBSD Foundation | ISC AND BSD-2-Clause | [wingetopt.txt](licenses/wingetopt.txt) |
@@ -107,6 +108,10 @@ served frontend, so the shipped web UI carries its own attribution.
and `SpotExtractorGPU.cu` follows the design of its GPU counterpart. MPL-2.0 is file-level, so both
files name the origin at the top and are covered by `licenses/traccc.txt`. See
[ACKNOWLEDGEMENT.md](docs/ACKNOWLEDGEMENT.md) for the citation.
* **Ceres Solver** is also fetched and linked (table above); separately, `LMSolver.h`/`.cpp` re-implement
its trust-region Levenberg-Marquardt minimiser, projected line search, polynomial step choice and
sphere manifold for the crystal refinement, following its source. Both files name the origin at the
top and are covered by `licenses/ceres-solver.txt`.
* **FFTW** is GPL-2.0-or-later — compatible with, and absorbed by, this project's GPL-3.0 license.
* **Apache-2.0** components: where upstream ships a `NOTICE` file, it is reproduced in the
corresponding `licenses/` text.
+3
View File
@@ -99,7 +99,10 @@ ADD_LIBRARY(JFJochImageAnalysis STATIC
beam_stop/ShadowFinder.cpp
beam_stop/ShadowFinder.h
$<$<BOOL:${JFJOCH_CUDA_AVAILABLE}>:beam_stop/ShadowAccumulatorGPU.cu>
$<$<BOOL:${JFJOCH_CUDA_AVAILABLE}>:beam_stop/ShadowMaskGPU.cu>
beam_stop/ShadowAccumulatorGPU.h
beam_stop/ShadowFinderInternal.h
beam_stop/ShadowMaskGPU.h
rotation_indexer/RotationIndexer.cpp
rotation_indexer/RotationIndexer.h
WriteReflections.cpp
+48 -9
View File
@@ -8,6 +8,15 @@ inline void cuda_err(cudaError_t val) {
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
}
// Pushes one ring run's totals to the shared accumulators; nothing for an empty run.
__device__ __forceinline__ void flush_azim_run(float *s_sum, float *s_sum2, uint32_t *s_count,
int b, float r_sum, float r_sum2, uint32_t r_count) {
if (r_count == 0) return; // also covers the initial "no ring yet"
atomicAdd(&s_sum[b], r_sum);
atomicAdd(&s_sum2[b], r_sum2);
atomicAdd(&s_count[b], r_count);
}
__global__
void gpu_azim_shared(
const uint16_t *__restrict__ pixel_to_bin,
@@ -33,19 +42,49 @@ void gpu_azim_shared(
__syncthreads();
for (size_t idx = blockIdx.x * blockDim.x + threadIdx.x;
idx < num_pixels;
idx += blockDim.x * gridDim.x) {
uint16_t bin = pixel_to_bin[idx];
// Four pixels per thread, read as vector loads, and a running total per ring pushed to shared
// memory only when the ring changes: consecutive pixels along a row mostly share a ring, and an
// atomic per pixel on the same few addresses is what this kernel was limited by. The same scheme
// as the adaptive spot finder's ring pass (reduce_rings_shared). The buffers come straight from
// cudaMalloc, aligned for int4/float4; the npix % 4 leftovers are done one at a time below.
const size_t stride = static_cast<size_t>(blockDim.x) * gridDim.x;
const size_t nquad = num_pixels / 4;
for (size_t q = blockIdx.x * blockDim.x + threadIdx.x; q < nquad; q += stride) {
const int4 v4 = reinterpret_cast<const int4 *>(input_buffer)[q];
const ushort4 b4 = reinterpret_cast<const ushort4 *>(pixel_to_bin)[q];
const float4 c4 = reinterpret_cast<const float4 *>(corrections)[q];
const int32_t vq[4] = {v4.x, v4.y, v4.z, v4.w};
const uint16_t bq[4] = {b4.x, b4.y, b4.z, b4.w};
const float cq[4] = {c4.x, c4.y, c4.z, c4.w};
int32_t v = input_buffer[idx];
bool valid = (v != INT32_MIN) & (v != INT32_MAX);
int r_b = -1;
float r_sum = 0.0f, r_sum2 = 0.0f;
uint32_t r_count = 0;
#pragma unroll
for (int k = 0; k < 4; k++) {
const int32_t v = vq[k];
const int b = bq[k];
if (v == INT32_MIN || v == INT32_MAX || b >= azint_bins) continue;
if (b != r_b) {
flush_azim_run(s_sum, s_sum2, s_count, r_b, r_sum, r_sum2, r_count);
r_b = b;
r_sum = 0.0f; r_sum2 = 0.0f; r_count = 0;
}
const float val = static_cast<float>(v) * cq[k];
r_sum += val;
r_sum2 += val * val;
r_count += 1;
}
flush_azim_run(s_sum, s_sum2, s_count, r_b, r_sum, r_sum2, r_count);
}
if (bin < azint_bins && valid) {
for (size_t idx = 4 * nquad + blockIdx.x * blockDim.x + threadIdx.x; idx < num_pixels; idx += stride) {
const uint16_t bin = pixel_to_bin[idx];
const int32_t v = input_buffer[idx];
if (bin < azint_bins && v != INT32_MIN && v != INT32_MAX) {
const float val = static_cast<float>(v) * corrections[idx];
const float val2 = val * val;
atomicAdd(&s_sum[bin], val);
atomicAdd(&s_sum2[bin], val2);
atomicAdd(&s_sum2[bin], val * val);
atomicAdd(&s_count[bin], 1);
}
}
@@ -190,3 +190,13 @@ void ShadowAccumulatorGPU::Download(std::vector<int64_t> &max_value, std::vector
cudaMemcpyDeviceToHost, *stream));
cuda_err(cudaStreamSynchronize(*stream));
}
std::vector<uint32_t> ShadowAccumulatorGPU::Mask(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask) {
FoldPending();
return ShadowMaskOnDevice(setup, pixel_mask, gpu_max, gpu_sum, gpu_count, frames, *stream);
}
std::vector<float> ShadowAccumulatorGPU::MeanProjection(const std::vector<uint32_t> &pixel_mask) {
FoldPending();
return MeanProjectionOnDevice(pixel_mask, gpu_sum, gpu_count, npixels, *stream);
}
@@ -10,6 +10,7 @@
#include "../../common/CompressedImage.h"
#include "../image_preprocessing/BSLZ4DecoderGPU.h"
#include "../indexing/CUDAMemHelpers.h"
#include "ShadowMaskGPU.h"
// The beam-stop projection accumulated on the device: only the compressed chunk crosses PCIe, and
// both the decode and the per-pixel maximum / sum / count run on the GPU. The projection comes back
@@ -59,6 +60,11 @@ public:
[[nodiscard]] uint32_t GetFrameCount() const { return frames; }
// The beam-stop mask and the mean projection, made where the projection is (ShadowMaskGPU.h), so
// that it does not have to come back at all.
std::vector<uint32_t> Mask(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask);
std::vector<float> MeanProjection(const std::vector<uint32_t> &pixel_mask);
// Bring the projection back to the host, folding in whatever the last batch still holds. Cheap
// to call once; it moves 20 bytes per pixel.
void Download(std::vector<int64_t> &max_value, std::vector<int64_t> &sum_value,
+415 -360
View File
@@ -2,6 +2,7 @@
// SPDX-License-Identifier: GPL-3.0-only
#include "ShadowFinder.h"
#include "ShadowFinderInternal.h"
#include <algorithm>
#include <cmath>
@@ -10,7 +11,6 @@
#include <future>
#include <thread>
#include <numbers>
#include <queue>
#include <type_traits>
#include <spdlog/spdlog.h>
@@ -19,67 +19,12 @@
#include "../../common/ParallelFor.h"
#include "../../common/JFJochException.h"
// A pixel is shadow when its background is below this fraction of the background it is
// compared against.
constexpr float SHADOW_RATIO = 0.50f;
using namespace shadow_finder;
// The boundary grows outward into partially shadowed pixels down to this fraction, but no
// further than PENUMBRA_MAX_PX from the core. A pin or a loop casts a wide half-shadow, so the
// reach is a good deal more than the beam stop's own edge needs.
constexpr float PENUMBRA_RATIO = 0.75f;
constexpr int PENUMBRA_MAX_PX = 30;
// Bridge small breaks along the holder arm. A module gap wider than this is bridged separately for
// the arm search (see bridge_gaps).
constexpr int BRIDGE_PX = 6;
// A pixel whose maximum reaches this recorded a real reflection and is never masked - a
// beam stop cannot block a reflection that was measured.
constexpr int64_t MIN_REFLECTION = 25;
// How far below the background it is compared against a pixel must sit before the dip is
// believed, in standard deviations of the counts that back it. The counts are photons, so their
// scatter is Poisson and the deficit is measured against it rather than against a fixed number:
// on a well-exposed sweep a third of the background missing is overwhelming, and on a handful of
// low-background frames the same third is noise. Without this a six-frame pre-scan of a
// low-background sweep masks three quarters of the detector.
constexpr double MIN_DEFICIT_SIGMA = 6.0;
// Smallest region the per-pixel test may return. A shadow is cast by something physical and is
// correspondingly large; an isolated patch this small is the background wandering, not hardware.
// This is what keeps the test specific now that a shadow no longer has to touch the direct beam.
constexpr int MIN_SHADOW_PIXELS = 2000;
// ... and of those, how many must be deep (below SHADOW_RATIO) for a region of merely DIM pixels -
// hardware that lets part of the beam through - to count. A thin holder arm is dim along most of
// its length and deep in places; the background drifting over a detector's edge is dim everywhere
// and deep nowhere.
constexpr int MIN_CORE_PIXELS = 200;
// The arm search compares a pixel with its ring as the ring actually varies around the beam. A ring
// of background is not flat once divided by the polarization factor when that factor is not the
// beam's: the remainder is a second harmonic in azimuth, cos 2phi, which reaches tens of percent at
// high angle. It is measured over radial bands of HARMONIC_BAND_PX, from the median of each of
// HARMONIC_SECTORS sectors that holds at least MIN_SECTOR_PIXELS pixels.
constexpr int HARMONIC_BAND_PX = 64;
constexpr int HARMONIC_SECTORS = 24;
constexpr int MIN_SECTOR_PIXELS = 200;
// Side of the box the background is pooled over before testing. Its area is how many pixels back
// a ring's countability test, which decides where an azimuthal comparison is possible at all.
constexpr int POOL_PX = 5;
constexpr double MEAN_POOLED_PIXELS = POOL_PX * POOL_PX;
// A ring with fewer valid pixels than this says nothing about whether it was counted.
constexpr int MIN_RING_PIXELS = 32;
// A ring lies wholly inside the stop when its background is below this fraction of the background
// further out. This asks about a whole ring rather than about a pixel, so it keeps a threshold of
// its own and does not follow SHADOW_RATIO.
constexpr float BLOCKED_RING_RATIO = 0.35f;
static_assert(ShadowFinder::SHADOW == MASK_SHADOW && ShadowFinder::TRANSMITTING == MASK_TRANSMITTING);
// Binary-image helpers on a width*height frame stored row-major as char (0/1). All run once,
// at GetMask() time; the BFS forms keep them O(pixels) rather than O(pixels * radius).
// at GetMask() time, and all are O(pixels) rather than O(pixels * radius).
namespace {
// A per-pixel array of GetMask(). A std::vector zeroes what it allocates on the thread that makes it,
@@ -186,10 +131,12 @@ Plane<char> erode(const Plane<char> &in, int W, int H, int r, size_t nthreads) {
// continues on both sides of it is one shadow - but a gap can be wider than BRIDGE_PX reaches (17 px
// between the rows of PILATUS modules), and an arm crossing one fell apart into pieces each too small
// to be believed.
Plane<char> bridge_gaps(const Plane<char> &region, const Plane<char> &valid, int W, int H) {
Plane<char> bridge_gaps(const Plane<char> &region, const Plane<char> &valid, int W, int H, size_t nthreads) {
Plane<char> out = region;
// The lines of one direction are independent: each reads `region` and only ever sets its own pixels.
auto walk = [&](int n_lines, int len, auto index) {
for (int line = 0; line < n_lines; line++) {
ParallelChunks(n_lines, nthreads, [&](int lo, int hi) {
for (int line = lo; line < hi; line++) {
int k = 0;
while (k < len) {
if (valid[index(line, k)]) { k++; continue; }
@@ -199,12 +146,99 @@ Plane<char> bridge_gaps(const Plane<char> &region, const Plane<char> &valid, int
for (int j = start; j < k; j++) out[index(line, j)] = 1;
}
}
});
};
walk(H, W, [W](int y, int x) { return static_cast<size_t>(y) * W + x; });
walk(W, H, [W](int x, int y) { return static_cast<size_t>(y) * W + x; });
return out;
}
// The 8-connected components of `member`: each member pixel gets the index of its component, dense
// from 0 and in no particular order, and every other pixel -1.
//
// Labelled in parallel. Each band of rows is flooded on its own, then the pieces that touch across a
// band boundary are joined. Which pixels share a component is all a caller reads, and that does not
// depend on how the rows were split.
struct Components {
Plane<int> id;
int count = 0;
};
Components label_components(const Plane<char> &member, int W, int H, size_t nthreads) {
const int bands = std::min(64, H); // never more bands than rows, so none is empty
std::vector<int> band_row(bands + 1);
for (int b = 0; b <= bands; b++)
band_row[b] = static_cast<int>(static_cast<int64_t>(b) * H / bands);
Components out;
out.id = Plane<int>(member.size());
std::vector<int> band_pieces(bands, 0);
ParallelFor(bands, nthreads, [&](int b) {
const size_t lo = static_cast<size_t>(band_row[b]) * W, hi = static_cast<size_t>(band_row[b + 1]) * W;
std::fill(out.id.begin() + lo, out.id.begin() + hi, -1);
std::vector<size_t> stack;
int pieces = 0;
for (size_t start = lo; start < hi; start++) {
if (!member[start] || out.id[start] >= 0)
continue;
out.id[start] = pieces;
stack.push_back(start);
while (!stack.empty()) {
const size_t i = stack.back(); stack.pop_back();
const int y = static_cast<int>(i / W), x = static_cast<int>(i % W);
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if (yy < band_row[b] || yy >= band_row[b + 1] || xx < 0 || xx >= W)
continue;
const size_t j = static_cast<size_t>(yy) * W + xx;
if (member[j] && out.id[j] < 0) { out.id[j] = pieces; stack.push_back(j); }
}
}
pieces++;
}
band_pieces[b] = pieces;
});
// A piece is named by its band's first index plus its number in the band, and the pieces are
// joined across each boundary row by union-find.
std::vector<int> first(bands + 1, 0);
for (int b = 0; b < bands; b++)
first[b + 1] = first[b] + band_pieces[b];
std::vector<int> parent(first[bands]);
for (size_t k = 0; k < parent.size(); k++)
parent[k] = static_cast<int>(k);
const auto find = [&](int k) {
while (parent[k] != k) { parent[k] = parent[parent[k]]; k = parent[k]; }
return k;
};
for (int b = 0; b + 1 < bands; b++) {
const size_t above = static_cast<size_t>(band_row[b + 1] - 1) * W, below = above + W;
for (int x = 0; x < W; x++) {
if (!member[above + x])
continue;
for (int xx = std::max(0, x - 1); xx <= std::min(W - 1, x + 1); xx++)
if (member[below + xx]) {
const int ra = find(first[b] + out.id[above + x]);
const int rb = find(first[b + 1] + out.id[below + xx]);
if (ra != rb) parent[std::max(ra, rb)] = std::min(ra, rb);
}
}
}
std::vector<int> dense(parent.size(), -1), component(parent.size());
for (size_t k = 0; k < parent.size(); k++) {
const int root = find(static_cast<int>(k));
if (dense[root] < 0) dense[root] = out.count++;
component[k] = dense[root];
}
ParallelFor(bands, nthreads, [&](int b) {
const size_t lo = static_cast<size_t>(band_row[b]) * W, hi = static_cast<size_t>(band_row[b + 1]) * W;
for (size_t i = lo; i < hi; i++)
if (out.id[i] >= 0) out.id[i] = component[first[b] + out.id[i]];
});
return out;
}
// Fill holes: background not reachable from the image border becomes region.
//
// The flood is run over the bounding box of `region` grown by one, not the whole detector. Outside
@@ -212,52 +246,53 @@ Plane<char> bridge_gaps(const Plane<char> &region, const Plane<char> &valid, int
// outside is one border-connected component: a background pixel inside the box is border-connected
// exactly when it reaches the ring. The beam stop occupies a small part of a detector, so this is
// the same answer over a fraction of the pixels.
Plane<char> fill_holes(const Plane<char> &region, int W, int H) {
Plane<char> fill_holes(const Plane<char> &region, int W, int H, size_t nthreads) {
std::vector<int> row_x0(H, W), row_x1(H, -1);
ParallelChunks(H, nthreads, [&](int ylo, int yhi) {
for (int y = ylo; y < yhi; y++)
for (int x = 0; x < W; x++)
if (region[static_cast<size_t>(y) * W + x]) {
row_x0[y] = std::min(row_x0[y], x);
row_x1[y] = x;
}
});
int x0 = W, x1 = -1, y0 = H, y1 = -1;
for (int y = 0; y < H; y++)
for (int x = 0; x < W; x++)
if (region[static_cast<size_t>(y) * W + x]) {
x0 = std::min(x0, x); x1 = std::max(x1, x);
y0 = std::min(y0, y); y1 = std::max(y1, y);
}
if (row_x1[y] >= 0) {
x0 = std::min(x0, row_x0[y]); x1 = std::max(x1, row_x1[y]);
y0 = std::min(y0, y); y1 = y;
}
if (x1 < 0)
return region; // nothing to enclose
x0 = std::max(0, x0 - 1); x1 = std::min(W - 1, x1 + 1);
y0 = std::max(0, y0 - 1); y1 = std::min(H - 1, y1 + 1);
// The background of the box, in components; one that reaches the box's edge is outside.
const int BW = x1 - x0 + 1, BH = y1 - y0 + 1;
std::vector<char> bg_visited(static_cast<size_t>(BW) * BH, 0);
std::queue<int> q; // indices into the box
auto push = [&](int bx, int by) {
const int j = by * BW + bx;
if (!region[static_cast<size_t>(by + y0) * W + bx + x0] && !bg_visited[j]) {
bg_visited[j] = 1; q.push(j);
}
Plane<char> background(static_cast<size_t>(BW) * BH);
ParallelChunks(BH, nthreads, [&](int lo, int hi) {
for (int by = lo; by < hi; by++)
for (int bx = 0; bx < BW; bx++)
background[static_cast<size_t>(by) * BW + bx] = !region[static_cast<size_t>(by + y0) * W + bx + x0];
});
const auto pieces = label_components(background, BW, BH, nthreads);
std::vector<char> outside(pieces.count, 0);
const auto edge = [&](int bx, int by) {
const int c = pieces.id[static_cast<size_t>(by) * BW + bx];
if (c >= 0) outside[c] = 1;
};
for (int bx = 0; bx < BW; bx++) { push(bx, 0); push(bx, BH - 1); }
for (int by = 0; by < BH; by++) { push(0, by); push(BW - 1, by); }
while (!q.empty()) {
const int i = q.front(); q.pop();
const int by = i / BW, bx = i % BW;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = by + dy, xx = bx + dx;
if (yy < 0 || yy >= BH || xx < 0 || xx >= BW)
continue;
const int j = yy * BW + xx;
if (!region[static_cast<size_t>(yy + y0) * W + xx + x0] && !bg_visited[j]) {
bg_visited[j] = 1; q.push(j);
}
}
}
for (int bx = 0; bx < BW; bx++) { edge(bx, 0); edge(bx, BH - 1); }
for (int by = 0; by < BH; by++) { edge(0, by); edge(BW - 1, by); }
Plane<char> out = region;
for (int by = 0; by < BH; by++)
for (int bx = 0; bx < BW; bx++) {
const size_t i = static_cast<size_t>(by + y0) * W + bx + x0;
if (!region[i] && !bg_visited[by * BW + bx])
out[i] = 1;
}
ParallelChunks(BH, nthreads, [&](int lo, int hi) {
for (int by = lo; by < hi; by++)
for (int bx = 0; bx < BW; bx++) {
const int c = pieces.id[static_cast<size_t>(by) * BW + bx];
if (c >= 0 && !outside[c])
out[static_cast<size_t>(by + y0) * W + bx + x0] = 1;
}
});
return out;
}
@@ -321,19 +356,40 @@ struct RingValues {
RingValues bin_by_ring(const Plane<float> &values, const Plane<char> &valid,
const Plane<int> &radius, int max_radius, size_t nthreads) {
// Counted and scattered by blocks of pixels in parallel: each block writes its values of a ring
// after those of the blocks before it, so every ring holds its values in pixel order, as a single
// pass would leave them - and they are sorted below in any case.
constexpr int BLOCKS = 64;
const size_t n = values.size();
const auto block_begin = [n](int b) { return n * b / BLOCKS; };
const size_t rings = static_cast<size_t>(max_radius) + 1;
std::vector<int> cursor(BLOCKS * rings, 0);
ParallelFor(BLOCKS, nthreads, [&](int b) {
int *count = cursor.data() + b * rings;
for (size_t i = block_begin(b); i < block_begin(b + 1); i++)
if (valid[i])
count[radius[i]]++;
});
RingValues rv;
rv.offset.assign(max_radius + 2, 0);
for (size_t i = 0; i < values.size(); i++)
if (valid[i])
rv.offset[radius[i] + 1]++;
for (int r = 0; r <= max_radius; r++)
rv.offset[r + 1] += rv.offset[r];
for (size_t r = 0; r < rings; r++) {
int at = rv.offset[r];
for (int b = 0; b < BLOCKS; b++) {
const int count = cursor[b * rings + r];
cursor[b * rings + r] = at;
at += count;
}
rv.offset[r + 1] = at;
}
rv.values.resize(rv.offset[max_radius + 1]);
std::vector<int> cursor(rv.offset.begin(), rv.offset.end() - 1);
for (size_t i = 0; i < values.size(); i++)
if (valid[i])
rv.values[cursor[radius[i]]++] = values[i];
ParallelFor(BLOCKS, nthreads, [&](int b) {
int *next = cursor.data() + b * rings;
for (size_t i = block_begin(b); i < block_begin(b + 1); i++)
if (valid[i])
rv.values[next[radius[i]]++] = values[i];
});
// Sorted once; the three iterations then only pick a rank and count a prefix.
ParallelFor(max_radius + 1, nthreads, [&](int r) {
@@ -355,7 +411,6 @@ ShadowFinder::ShadowFinder(const DiffractionExperiment &experiment, const PixelM
if (pixel_mask.size() != static_cast<size_t>(width) * height)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: pixel mask does not match the detector");
SetShardCount(1);
#ifdef JFJOCH_USE_CUDA
if (get_gpu_count() > 0) {
const size_t npixels = static_cast<size_t>(width) * height;
@@ -372,98 +427,37 @@ void ShadowFinder::BeamCenter(float x, float y) {
beam_y = y;
}
// A shard's accumulators are allocated when a frame is first added to it, not here: with a GPU they
// are never used at all, and on a 16 Mpx detector eight of them are 2.9 GB to allocate and clear -
// which measured 0.8 s of the pre-scan, all of it wasted.
void ShadowFinder::SetShardCount(size_t n) {
shards.clear();
shards.resize(std::max<size_t>(1, n));
}
#ifdef JFJOCH_USE_CUDA
ShadowFinder::Projection ShadowFinder::Reduce() const {
#ifdef JFJOCH_USE_CUDA
// The device holds its own projection. Bring it back and let it take part in the fold below as
// one more shard; when every frame went to the GPU it is the whole answer.
Projection device;
if (Gpu() && gpu->GetFrameCount() > 0) {
gpu->Download(device.max_value, device.sum_value, device.valid_count);
device.frames = gpu->GetFrameCount();
bool host_empty = true;
for (const auto &p : shards)
host_empty = host_empty && (p.frames == 0);
if (host_empty)
return device;
}
#endif
// Only when that shard actually holds something: its accumulators are allocated on first use, so
// an unused shard is empty rather than zeroed, and returning it would hand the callers below a
// projection they index by pixel.
if (shards.size() == 1 && shards[0].frames > 0
#ifdef JFJOCH_USE_CUDA
&& !(gpu && gpu->GetFrameCount() > 0)
#endif
)
return shards[0];
// The device holds its own projection. Bring it back; when every frame went to the GPU it is the
// whole answer.
Projection out;
const size_t npixels = static_cast<size_t>(width) * height;
out.max_value.assign(npixels, 0);
out.sum_value.assign(npixels, 0);
out.valid_count.assign(npixels, 0);
for (const auto &p : shards)
out.frames += p.frames;
#ifdef JFJOCH_USE_CUDA
out.frames += device.frames;
#endif
// Each worker owns a slice of the pixels and folds every shard into it. The sums and counts are
// integers and a pixel is touched by one worker only, so the result is the same as folding them
// one shard at a time on one thread - this is several hundred megabytes per shard and is limited
// by memory rather than by arithmetic.
const size_t nthreads = std::max<size_t>(1, std::min<size_t>(std::thread::hardware_concurrency(),
shards.size() * 2));
const size_t chunk = (npixels + nthreads - 1) / nthreads;
std::vector<std::future<void>> futures;
futures.reserve(nthreads);
for (size_t t = 0; t < nthreads; t++) {
const size_t lo = t * chunk, hi = std::min(npixels, lo + chunk);
if (lo >= hi) break;
futures.emplace_back(std::async(std::launch::async, [&, lo, hi] {
#ifdef JFJOCH_USE_CUDA
const Projection *extra[1] = {&device};
for (const auto *pp : extra) {
const auto &p = *pp;
if (p.frames == 0) continue;
for (size_t i = lo; i < hi; i++) {
if (p.valid_count[i] == 0)
continue;
if (out.valid_count[i] == 0 || p.max_value[i] > out.max_value[i])
out.max_value[i] = p.max_value[i];
out.sum_value[i] += p.sum_value[i];
out.valid_count[i] += p.valid_count[i];
}
}
#endif
for (const auto &p : shards) {
if (p.frames == 0) continue; // never used, and its accumulators were never allocated
for (size_t i = lo; i < hi; i++) {
if (p.valid_count[i] == 0)
continue;
if (out.valid_count[i] == 0 || p.max_value[i] > out.max_value[i])
out.max_value[i] = p.max_value[i];
out.sum_value[i] += p.sum_value[i];
out.valid_count[i] += p.valid_count[i];
}
}
}));
if (Gpu() && gpu->GetFrameCount() > 0) {
gpu->Download(out.max_value, out.sum_value, out.valid_count);
out.frames = gpu->GetFrameCount();
}
for (auto &f : futures) f.get();
if (host.frames == 0)
return out;
// A pixel is touched by one worker only and the sums and counts are integers, so the result is
// the same as folding on one thread.
out.frames += host.frames;
ParallelChunks(static_cast<int>(out.max_value.size()), std::thread::hardware_concurrency(), [&](int lo, int hi) {
for (int i = lo; i < hi; i++) {
if (host.valid_count[i] == 0)
continue;
if (out.valid_count[i] == 0 || host.max_value[i] > out.max_value[i])
out.max_value[i] = host.max_value[i];
out.sum_value[i] += host.sum_value[i];
out.valid_count[i] += host.valid_count[i];
}
});
return out;
}
#endif
template<class T>
void ShadowFinder::Add(const T *ptr, Projection &p) {
void ShadowFinder::Add(const T *ptr, size_t begin, size_t end) {
// The pixel type's sentinel extreme marks "no data" (module gap / masked): the
// preprocessor/writer stores INT*_MIN for signed and UINT*_MAX for unsigned. For signed
// types the opposite extreme is a genuine saturated value and is kept, so a saturated
@@ -474,27 +468,23 @@ void ShadowFinder::Add(const T *ptr, Projection &p) {
else
masked = std::numeric_limits<T>::max();
for (size_t i = 0; i < p.max_value.size(); i++) {
for (size_t i = begin; i < end; i++) {
const T v = ptr[i];
if (v == masked)
continue;
const int64_t vi = static_cast<int64_t>(v);
if (p.valid_count[i] == 0 || vi > p.max_value[i])
p.max_value[i] = vi;
p.sum_value[i] += vi;
p.valid_count[i]++;
if (host.valid_count[i] == 0 || vi > host.max_value[i])
host.max_value[i] = vi;
host.sum_value[i] += vi;
host.valid_count[i]++;
}
p.frames++;
}
void ShadowFinder::AddImage(const DataMessage &data, std::vector<uint8_t> &buffer, size_t shard) {
void ShadowFinder::AddImage(const DataMessage &data, std::vector<uint8_t> &buffer) {
if (static_cast<size_t>(data.image.GetWidth()) * data.image.GetHeight()
!= static_cast<size_t>(width) * height)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: image size does not match the detector");
if (shard >= shards.size())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: shard out of range");
#ifdef JFJOCH_USE_CUDA
// One device, so the frames queue here - but each is only a chunk upload plus two kernels, and
@@ -511,25 +501,37 @@ void ShadowFinder::AddImage(const DataMessage &data, std::vector<uint8_t> &buffe
}
#endif
Projection &p = shards[shard];
if (p.max_value.empty()) {
const size_t npixels = static_cast<size_t>(width) * height;
p.max_value.assign(npixels, 0);
p.sum_value.assign(npixels, 0);
p.valid_count.assign(npixels, 0);
const size_t npixels = static_cast<size_t>(width) * height;
{
std::unique_lock ul(host_mutex);
if (host.max_value.empty()) {
host.max_value.resize(npixels);
host.sum_value.resize(npixels);
host.valid_count.resize(npixels);
}
}
const auto ptr = data.image.GetUncompressedPtr(buffer);
switch (data.image.GetMode()) {
case CompressedImageMode::Int8: Add(reinterpret_cast<const int8_t *>(ptr), p); break;
case CompressedImageMode::Uint8: Add(reinterpret_cast<const uint8_t *>(ptr), p); break;
case CompressedImageMode::Int16: Add(reinterpret_cast<const int16_t *>(ptr), p); break;
case CompressedImageMode::Uint16: Add(reinterpret_cast<const uint16_t *>(ptr), p); break;
case CompressedImageMode::Int32: Add(reinterpret_cast<const int32_t *>(ptr), p); break;
case CompressedImageMode::Uint32: Add(reinterpret_cast<const uint32_t *>(ptr), p); break;
default:
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: unsupported image mode");
const size_t rows_per_band = (static_cast<size_t>(height) + BANDS - 1) / BANDS;
const size_t first = next_band.fetch_add(1);
for (size_t b = 0; b < BANDS; b++) {
const size_t band = (first + b) % BANDS;
const size_t begin = std::min(npixels, band * rows_per_band * width);
const size_t end = std::min(npixels, (band + 1) * rows_per_band * width);
std::lock_guard lock(band_mutex[band]);
switch (data.image.GetMode()) {
case CompressedImageMode::Int8: Add(reinterpret_cast<const int8_t *>(ptr), begin, end); break;
case CompressedImageMode::Uint8: Add(reinterpret_cast<const uint8_t *>(ptr), begin, end); break;
case CompressedImageMode::Int16: Add(reinterpret_cast<const int16_t *>(ptr), begin, end); break;
case CompressedImageMode::Uint16: Add(reinterpret_cast<const uint16_t *>(ptr), begin, end); break;
case CompressedImageMode::Int32: Add(reinterpret_cast<const int32_t *>(ptr), begin, end); break;
case CompressedImageMode::Uint32: Add(reinterpret_cast<const uint32_t *>(ptr), begin, end); break;
default:
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: unsupported image mode");
}
}
std::unique_lock ul(host_mutex);
host.frames++;
}
#ifdef JFJOCH_USE_CUDA
@@ -552,8 +554,7 @@ ShadowAccumulatorGPU *ShadowFinder::Gpu() const {
uint32_t ShadowFinder::GetFrameCount() const {
std::unique_lock ul(m);
uint32_t frames = 0;
for (const auto &p : shards) frames += p.frames;
uint32_t frames = host.frames;
#ifdef JFJOCH_USE_CUDA
if (gpu) frames += gpu->GetFrameCount();
#endif
@@ -561,26 +562,39 @@ uint32_t ShadowFinder::GetFrameCount() const {
}
const ShadowFinder::Projection &ShadowFinder::Reduced() const {
if (!reduced)
reduced = Reduce();
return *reduced;
#ifdef JFJOCH_USE_CUDA
if (gpu && gpu->GetFrameCount() > 0) {
if (!reduced)
reduced = Reduce();
return *reduced;
}
#endif
return host;
}
void ShadowFinder::ReleaseProjection() {
#ifdef JFJOCH_USE_CUDA
std::unique_lock ul(m);
reduced.reset();
#endif
}
std::vector<float> ShadowFinder::GetMeanProjection() const {
std::unique_lock ul(m);
#ifdef JFJOCH_USE_CUDA
if (Gpu() && gpu->GetFrameCount() > 0 && host.frames == 0)
return gpu->MeanProjection(pixel_mask);
#endif
const Projection &p = Reduced();
const auto &sum_value = p.sum_value;
const auto &valid_count = p.valid_count;
std::vector<float> mean(static_cast<size_t>(width) * height, NAN);
for (size_t i = 0; i < mean.size(); i++)
if (valid_count[i] > 0 && pixel_mask[i] == 0)
mean[i] = static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]);
std::vector<float> mean(static_cast<size_t>(width) * height);
ParallelChunks(static_cast<int>(mean.size()), std::thread::hardware_concurrency(), [&](int lo, int hi) {
for (int i = lo; i < hi; i++)
mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0
? static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]) : NAN;
});
return mean;
}
@@ -588,6 +602,28 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
std::unique_lock ul(m);
if (nthreads == 0)
nthreads = std::max(1u, std::thread::hardware_concurrency());
#ifdef JFJOCH_USE_CUDA
// Where every frame went to the device the mask is made there too, from the projection as it lies.
if (Gpu() && gpu->GetFrameCount() > 0 && host.frames == 0) {
const float diag = std::hypot(static_cast<float>(width), static_cast<float>(height));
if (!std::isfinite(beam_x) || !std::isfinite(beam_y)
|| std::fabs(beam_x - width * 0.5f) > 4.0f * diag || std::fabs(beam_y - height * 0.5f) > 4.0f * diag)
return std::vector<uint32_t>(static_cast<size_t>(width) * height, 0);
ShadowMaskSetup setup;
setup.width = width;
setup.height = height;
setup.beam_x = beam_x;
setup.beam_y = beam_y;
const auto rot = geometry.GetDetectorMatrix().arr();
for (int k = 0; k < 9; k++)
setup.det_matrix[k] = rot[k];
setup.pixel_size_mm = geometry.GetPixelSize_mm();
setup.distance_mm = geometry.GetDetectorDistance_mm();
setup.has_polarization = polarization.has_value();
setup.polarization = polarization.value_or(0.0f);
return gpu->Mask(setup, pixel_mask);
}
#endif
const Projection &p = Reduced();
const auto &max_value = p.max_value;
const auto &sum_value = p.sum_value;
@@ -727,9 +763,8 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
// Innermost rings hold only a handful of pixels, too few to judge, so they are stepped over
// rather than allowed to end the walk.
std::vector<int> ring_pixels(max_radius + 1, 0);
for (int i = 0; i < n_pixels; i++)
if (valid[i])
ring_pixels[radius[i]]++;
for (int rad = 0; rad <= max_radius; rad++)
ring_pixels[rad] = rings.offset[rad + 1] - rings.offset[rad];
// A ring lies inside the stop when its background is a fraction of what this detector's
// background typically is. Counting statistics cannot decide this: on a bright dataset the
@@ -741,25 +776,7 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
// disk it masked. The median over the rings this walk is willing to judge is what "typically"
// means, and it is robust from both sides - to a bright ring, and to a corner ring of a handful
// of pixels whose median is one pixel's mean.
std::vector<float> judgeable;
for (int rad = 0; rad <= max_radius; rad++)
if (ring_pixels[rad] >= MIN_RING_PIXELS)
judgeable.push_back(baseline[rad]);
float typical_background = 0.0f;
if (!judgeable.empty()) {
const auto middle = judgeable.begin() + judgeable.size() / 2;
std::nth_element(judgeable.begin(), middle, judgeable.end());
typical_background = *middle;
}
int blocked_out_to = -1;
for (int rad = 0; rad <= max_radius; rad++) {
if (ring_pixels[rad] < MIN_RING_PIXELS)
continue;
if (baseline[rad] >= BLOCKED_RING_RATIO * typical_background)
break;
blocked_out_to = rad;
}
const int blocked_out_to = BlockedOutTo(baseline, ring_pixels);
// The counts a pixel's pooled background is made of, and the counts the ring says it should
// have had. The test is on the deficit between them, in units of its own Poisson scatter.
@@ -789,36 +806,20 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
// and the stop. What keeps the test specific instead is size, since the background wanders by a
// pixel or two at a time and hardware does not.
const Plane<char> bridged = dilate(low, W, H, BRIDGE_PX, nthreads);
Plane<char> region = filled_plane<char>(n_pixels, 0, nthreads);
Plane<char> region(n_pixels);
{
Plane<char> seen = filled_plane<char>(n_pixels, 0, nthreads);
std::vector<int> component;
std::queue<int> q;
for (int start = 0; start < n_pixels; start++) {
if (!bridged[start] || seen[start])
continue;
component.clear();
int n_low = 0;
seen[start] = 1;
q.push(start);
while (!q.empty()) {
const int i = q.front(); q.pop();
component.push_back(i);
n_low += low[i];
const int y = i / W, x = i % W;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if (yy < 0 || yy >= H || xx < 0 || xx >= W)
continue;
const int j = yy * W + xx;
if (bridged[j] && !seen[j]) { seen[j] = 1; q.push(j); }
}
}
if (n_low >= MIN_SHADOW_PIXELS)
for (const int i : component)
region[i] = low[i];
}
const auto pieces = label_components(bridged, W, H, nthreads);
std::vector<std::atomic<int>> n_low(pieces.count);
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
for (int i = lo; i < hi; i++)
if (pieces.id[i] >= 0 && low[i])
n_low[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
});
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
for (int i = lo; i < hi; i++)
region[i] = pieces.id[i] >= 0 && n_low[pieces.id[i]].load(std::memory_order_relaxed) >= MIN_SHADOW_PIXELS
? low[i] : 0;
});
}
// The rings that lie wholly inside the stop are decided by the ring walk above rather than by
@@ -866,7 +867,7 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
region = erode(dilate(region, W, H, 2, nthreads), W, H, 2, nthreads);
region = fill_holes(region, W, H);
region = fill_holes(region, W, H, nthreads);
// Expose recorded reflections - done last, with no fill afterwards, so a spot the shadow
@@ -903,60 +904,42 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
// allowed to explain a dim sector away - where it would ask for more than the median, the median
// stands - so the step can only drop what it found before, never find something new.
const int n_bands = max_radius / HARMONIC_BAND_PX + 1;
std::vector<std::vector<float>> sector_values(static_cast<size_t>(n_bands) * HARMONIC_SECTORS);
for (int i = 0; i < n_pixels; i++) {
if (!valid[i] || region[i])
continue;
const float dx = static_cast<float>(i % W) - beam_x, dy = static_cast<float>(i / W) - beam_y;
const double phi = std::atan2(dy, dx) + std::numbers::pi;
const int sector = std::min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * std::numbers::pi) * HARMONIC_SECTORS));
sector_values[static_cast<size_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector].push_back(ratio[i]);
}
std::vector<float> harm_c(n_bands, 0.0f), harm_s(n_bands, 0.0f);
ParallelFor(n_bands, nthreads, [&](int band) {
std::vector<double> med(HARMONIC_SECTORS, -1.0), c(HARMONIC_SECTORS), s(HARMONIC_SECTORS);
for (int k = 0; k < HARMONIC_SECTORS; k++) {
auto &v = sector_values[static_cast<size_t>(band) * HARMONIC_SECTORS + k];
if (v.size() >= MIN_SECTOR_PIXELS) {
std::nth_element(v.begin(), v.begin() + v.size() / 2, v.end());
med[k] = v[v.size() / 2];
}
const double phi = (k + 0.5) * 2.0 * std::numbers::pi / HARMONIC_SECTORS - std::numbers::pi;
c[k] = std::cos(2 * phi);
s[k] = std::sin(2 * phi);
}
// Least squares of med = m + p cos 2phi + q sin 2phi over the sectors that are not themselves
// dim against the fit, three times over as the baseline is; the model is relative, 1 + (p cos + q sin)/m.
std::vector<char> use(HARMONIC_SECTORS);
for (int k = 0; k < HARMONIC_SECTORS; k++)
use[k] = med[k] >= 0;
for (int iter = 0; iter < 3; iter++) {
double n = 0, sc = 0, ss = 0, scc = 0, sss = 0, scs = 0, y = 0, yc = 0, ys = 0;
for (int k = 0; k < HARMONIC_SECTORS; k++) {
if (!use[k]) continue;
n++; sc += c[k]; ss += s[k]; scc += c[k] * c[k]; sss += s[k] * s[k]; scs += c[k] * s[k];
y += med[k]; yc += med[k] * c[k]; ys += med[k] * s[k];
}
// Sectors crowded into a narrow arc cannot tell a harmonic from a level; the determinant of
// the normal matrix, per sector cubed, is 1/4 on a full ring.
const double det = n * (scc * sss - scs * scs) - sc * (sc * sss - scs * ss) + ss * (sc * scs - scc * ss);
if (n < 6 || det < 0.01 * n * n * n) {
harm_c[band] = harm_s[band] = 0.0f;
return;
}
const double m = (y * (scc * sss - scs * scs) - sc * (yc * sss - scs * ys) + ss * (yc * scs - scc * ys)) / det;
const double p = (n * (yc * sss - ys * scs) - y * (sc * sss - scs * ss) + ss * (sc * ys - yc * ss)) / det;
const double q = (n * (scc * ys - scs * yc) - sc * (sc * ys - yc * ss) + y * (sc * scs - scc * ss)) / det;
if (m <= 0) {
harm_c[band] = harm_s[band] = 0.0f;
return;
}
harm_c[band] = static_cast<float>(p / m);
harm_s[band] = static_cast<float>(q / m);
for (int k = 0; k < HARMONIC_SECTORS; k++)
use[k] = med[k] >= 0 && med[k] >= PENUMBRA_RATIO * (1.0 + harm_c[band] * c[k] + harm_s[band] * s[k]);
const size_t n_sectors = static_cast<size_t>(n_bands) * HARMONIC_SECTORS;
// Gathered by blocks of rows in parallel and joined in block order. Only the median of each
// sector is read, and that is the same whatever order its values were gathered in.
constexpr int SECTOR_BLOCKS = 64;
std::vector<std::vector<std::vector<float>>> block_values(SECTOR_BLOCKS);
ParallelFor(SECTOR_BLOCKS, nthreads, [&](int b) {
auto &values = block_values[b];
values.resize(n_sectors);
const int lo = static_cast<int>(static_cast<int64_t>(n_pixels) * b / SECTOR_BLOCKS);
const int hi = static_cast<int>(static_cast<int64_t>(n_pixels) * (b + 1) / SECTOR_BLOCKS);
for (int i = lo; i < hi; i++) {
if (!valid[i] || region[i])
continue;
const float dx = static_cast<float>(i % W) - beam_x, dy = static_cast<float>(i / W) - beam_y;
const double phi = std::atan2(dy, dx) + std::numbers::pi;
const int sector = std::min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * std::numbers::pi) * HARMONIC_SECTORS));
values[static_cast<size_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector].push_back(ratio[i]);
}
});
std::vector<std::vector<float>> sector_values(n_sectors);
ParallelFor(static_cast<int>(n_sectors), nthreads, [&](int k) {
for (const auto &values : block_values)
sector_values[k].insert(sector_values[k].end(), values[k].begin(), values[k].end());
});
block_values.clear();
// The median of each sector with enough pixels to have one.
std::vector<double> sector_median(n_sectors, -1.0);
ParallelFor(static_cast<int>(n_sectors), nthreads, [&](int k) {
auto &v = sector_values[k];
if (v.size() >= MIN_SECTOR_PIXELS) {
std::nth_element(v.begin(), v.begin() + v.size() / 2, v.end());
sector_median[k] = v[v.size() / 2];
}
});
std::vector<float> harm_c, harm_s;
HarmonicFit(sector_median, n_bands, harm_c, harm_s);
Plane<char> dim = filled_plane<char>(n_pixels, 0, nthreads);
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
@@ -978,36 +961,108 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
&& poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA;
}
});
const Plane<char> joined = bridge_gaps(dilate(dim, W, H, BRIDGE_PX, nthreads), valid, W, H);
Plane<char> seen = filled_plane<char>(n_pixels, 0, nthreads);
std::vector<int> component;
std::queue<int> q;
for (int start = 0; start < n_pixels; start++) {
if (!joined[start] || seen[start])
continue;
component.clear();
int n_dim = 0, n_low = 0;
seen[start] = 1;
q.push(start);
while (!q.empty()) {
const int i = q.front(); q.pop();
component.push_back(i);
n_dim += dim[i];
n_low += dim[i] && low[i];
const int y = i / W, x = i % W;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if (yy < 0 || yy >= H || xx < 0 || xx >= W)
continue;
const int j = yy * W + xx;
if (joined[j] && !seen[j]) { seen[j] = 1; q.push(j); }
}
const Plane<char> joined = bridge_gaps(dilate(dim, W, H, BRIDGE_PX, nthreads), valid, W, H, nthreads);
const auto pieces = label_components(joined, W, H, nthreads);
std::vector<std::atomic<int>> n_dim(pieces.count), n_low(pieces.count);
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
for (int i = lo; i < hi; i++)
if (pieces.id[i] >= 0 && dim[i]) {
n_dim[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
if (low[i])
n_low[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
}
});
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
for (int i = lo; i < hi; i++) {
const int c = pieces.id[i];
if (c >= 0 && dim[i] && n_dim[c].load(std::memory_order_relaxed) >= MIN_SHADOW_PIXELS
&& n_low[c].load(std::memory_order_relaxed) >= MIN_CORE_PIXELS)
mask[i] = TRANSMITTING;
}
if (n_dim >= MIN_SHADOW_PIXELS && n_low >= MIN_CORE_PIXELS)
for (const int i : component)
if (dim[i])
mask[i] = TRANSMITTING;
}
});
return mask;
}
namespace shadow_finder {
int BlockedOutTo(const std::vector<float> &baseline, const std::vector<int> &ring_pixels) {
const int max_radius = static_cast<int>(baseline.size()) - 1;
// A ring lies inside the stop when its background is a fraction of what this detector's
// background typically is. Counting statistics cannot decide this: on a bright dataset the
// shadow is still well counted. The comparison used to be against the LARGEST background of any
// ring further out, and that reads a sample whose background peaks in a strong ring away from
// the beam - a powder standard, a strong solvent ring - as a beam stop the size of that ring:
// the ordinary background inside it is legitimately below a third of the peak. On one corpus
// dataset it declared 16 % of the detector to be stop, with diffraction rings visible inside the
// disk it masked. The median over the rings this walk is willing to judge is what "typically"
// means, and it is robust from both sides - to a bright ring, and to a corner ring of a handful
// of pixels whose median is one pixel's mean.
std::vector<float> judgeable;
for (int rad = 0; rad <= max_radius; rad++)
if (ring_pixels[rad] >= MIN_RING_PIXELS)
judgeable.push_back(baseline[rad]);
float typical_background = 0.0f;
if (!judgeable.empty()) {
const auto middle = judgeable.begin() + judgeable.size() / 2;
std::nth_element(judgeable.begin(), middle, judgeable.end());
typical_background = *middle;
}
int blocked_out_to = -1;
for (int rad = 0; rad <= max_radius; rad++) {
if (ring_pixels[rad] < MIN_RING_PIXELS)
continue;
if (baseline[rad] >= BLOCKED_RING_RATIO * typical_background)
break;
blocked_out_to = rad;
}
return blocked_out_to;
}
void HarmonicFit(const std::vector<double> &sector_median, int n_bands,
std::vector<float> &harm_c, std::vector<float> &harm_s) {
harm_c.assign(n_bands, 0.0f);
harm_s.assign(n_bands, 0.0f);
for (int band = 0; band < n_bands; band++) {
std::vector<double> med(HARMONIC_SECTORS), c(HARMONIC_SECTORS), s(HARMONIC_SECTORS);
for (int k = 0; k < HARMONIC_SECTORS; k++) {
med[k] = sector_median[static_cast<size_t>(band) * HARMONIC_SECTORS + k];
const double phi = (k + 0.5) * 2.0 * std::numbers::pi / HARMONIC_SECTORS - std::numbers::pi;
c[k] = std::cos(2 * phi);
s[k] = std::sin(2 * phi);
}
// Least squares of med = m + p cos 2phi + q sin 2phi over the sectors that are not themselves
// dim against the fit, three times over as the baseline is; the model is relative, 1 + (p cos + q sin)/m.
std::vector<char> use(HARMONIC_SECTORS);
for (int k = 0; k < HARMONIC_SECTORS; k++)
use[k] = med[k] >= 0;
for (int iter = 0; iter < 3; iter++) {
double n = 0, sc = 0, ss = 0, scc = 0, sss = 0, scs = 0, y = 0, yc = 0, ys = 0;
for (int k = 0; k < HARMONIC_SECTORS; k++) {
if (!use[k]) continue;
n++; sc += c[k]; ss += s[k]; scc += c[k] * c[k]; sss += s[k] * s[k]; scs += c[k] * s[k];
y += med[k]; yc += med[k] * c[k]; ys += med[k] * s[k];
}
// Sectors crowded into a narrow arc cannot tell a harmonic from a level; the determinant of
// the normal matrix, per sector cubed, is 1/4 on a full ring.
const double det = n * (scc * sss - scs * scs) - sc * (sc * sss - scs * ss) + ss * (sc * scs - scc * ss);
if (n < 6 || det < 0.01 * n * n * n) {
harm_c[band] = harm_s[band] = 0.0f;
break;
}
const double m = (y * (scc * sss - scs * scs) - sc * (yc * sss - scs * ys) + ss * (yc * scs - scc * ys)) / det;
const double p = (n * (yc * sss - ys * scs) - y * (sc * sss - scs * ss) + ss * (sc * ys - yc * ss)) / det;
const double q = (n * (scc * ys - scs * yc) - sc * (sc * ys - yc * ss) + y * (sc * scs - scc * ss)) / det;
if (m <= 0) {
harm_c[band] = harm_s[band] = 0.0f;
break;
}
harm_c[band] = static_cast<float>(p / m);
harm_s[band] = static_cast<float>(q / m);
for (int k = 0; k < HARMONIC_SECTORS; k++)
use[k] = med[k] >= 0 && med[k] >= PENUMBRA_RATIO * (1.0 + harm_c[band] * c[k] + harm_s[band] * s[k]);
}
}
}
} // namespace shadow_finder
+33 -26
View File
@@ -4,6 +4,7 @@
#pragma once
#include <cstdint>
#include <atomic>
#include <future>
#include <memory>
#include <mutex>
@@ -36,9 +37,9 @@
//
// Frames are chosen by the caller; the detection needs enough of them that the background
// is counted rather than guessed (see MIN_EXPECTED_COUNTS in the .cpp).
// Thread-safe: workers call AddImage concurrently, each naming a shard of its own (see
// SetShardCount) - so no two threads touch the same accumulator and nothing is locked while
// an image is added. The shards are summed when the projection is read.
// Thread-safe: workers call AddImage concurrently. The projection is split into bands of rows, each
// with a lock of its own, and a worker adding a frame starts at a different band from the one before
// it, so workers meet only when they reach the same band.
class ShadowFinder {
mutable std::mutex m;
@@ -62,22 +63,27 @@ class ShadowFinder {
std::vector<uint32_t> pixel_mask; // pixels already masked carry no background to test
// Per-pixel projection over the frames added so far (converted geometry). One set per shard:
// the sums and counts are integers, so summing the shards is exact and the result does not
// depend on how the frames were spread over them.
// Per-pixel projection over the frames added so far (converted geometry). The sums and counts are
// integers and the maximum is a maximum, so the result does not depend on the order the frames
// arrive in.
struct Projection {
std::vector<int64_t> max_value;
std::vector<int64_t> sum_value;
std::vector<uint32_t> valid_count;
uint32_t frames = 0;
};
std::vector<Projection> shards;
// The frames added on the host. Allocated by the first of them: with a GPU there are usually none.
Projection host;
std::mutex host_mutex; // guards the allocation and the frame count, not the sums
static constexpr size_t BANDS = 64;
std::mutex band_mutex[BANDS];
std::atomic<size_t> next_band{0};
#ifdef JFJOCH_USE_CUDA
// Present when a GPU is available. Frames it can decode are accumulated there instead of on the
// host - only the compressed chunk crosses PCIe - and its projection is folded in with the
// shards when the mask is read. Frames it cannot take (anything but bitshuffle+LZ4) still go to
// a host shard, so a run mixing compressions is handled without a second code path.
// host projection when the mask is read. Frames it cannot take (anything but bitshuffle+LZ4) still
// go to the host, so a run mixing compressions is handled without a second code path.
// Built on a thread of its own: it allocates and clears several hundred megabytes of device
// memory, and cudaMalloc synchronises the whole device, so doing it in the constructor would
// stall the caller before it has read its first frame. The first AddImage waits for it, by
@@ -90,17 +96,22 @@ class ShadowFinder {
[[nodiscard]] ShadowAccumulatorGPU *Gpu() const;
#endif
template<class T> void Add(const T *ptr, Projection &p);
// Add the pixels [begin, end) of one frame to the host projection.
template<class T> void Add(const T *ptr, size_t begin, size_t end);
// Sum the shards into one projection. max_value is only taken from a shard that actually
// counted the pixel - a shard that never saw it holds 0, which would beat a genuinely
#ifdef JFJOCH_USE_CUDA
// The device's projection with the host's folded in. max_value is only taken from a projection
// that actually counted the pixel - one that never saw it holds 0, which would beat a genuinely
// negative maximum.
[[nodiscard]] Projection Reduce() const;
// The projection, reduced on its first read and kept. The ring-centre fit, the mask and the
// beam-centre capture all read the same one, and on a 16 Mpx detector each reduction is 360 MB
// brought back from the device into fresh memory. Called with `m` held.
// That projection, made on its first read and kept. The ring-centre fit, the mask and the
// beam-centre capture all read the same one, and on a 16 Mpx detector each is 360 MB brought back
// from the device into fresh memory.
mutable std::optional<Projection> reduced;
#endif
// The projection the frames added so far make: the host's, or the one above where the device
// took frames. Called with `m` held.
[[nodiscard]] const Projection &Reduced() const;
public:
@@ -116,20 +127,16 @@ public:
// hardware. The projection is not centred on anything, so this may be set after the frames.
void BeamCenter(float x, float y);
// Give each worker a shard to accumulate into. Must be called before the first AddImage,
// and costs 20 bytes per pixel per shard.
void SetShardCount(size_t n);
// Accumulate one full converted-geometry image into shard `shard`. Gap / masked pixels
// (the pixel type's sentinel extreme) are skipped. `buffer` is scratch space for
// decompression, reused across the calls of one worker.
void AddImage(const DataMessage &data, std::vector<uint8_t> &buffer, size_t shard = 0);
// Accumulate one full converted-geometry image. Gap / masked pixels (the pixel type's sentinel
// extreme) are skipped. `buffer` is scratch space for decompression, reused across the calls of
// one worker.
void AddImage(const DataMessage &data, std::vector<uint8_t> &buffer);
// Compute the shadow mask (SHADOW, TRANSMITTING or 0 = keep), of the converted pixel count.
// TRANSMITTING marks the pieces of hardware that let part of the beam through, added after the
// shadow proper; both are masked, and a consumer that must not see those pieces can tell them apart.
// Recomputed on each call from the projection, which is summed over the shards on the first read
// of it (GetMask or GetMeanProjection): frames added after that are not seen.
// Recomputed on each call from the projection, which is put together on the first read of it
// (GetMask or GetMeanProjection): frames added after that are not seen.
// nthreads = 0 asks for all hardware threads. The per-pixel passes over a 16M-pixel detector
// dominate this, and they are all exactly parallel.
[[nodiscard]] std::vector<uint32_t> GetMask(size_t nthreads = 0) const;
@@ -140,7 +147,7 @@ public:
[[nodiscard]] std::vector<float> GetMeanProjection() const;
// Let go of the projection the two above read, once the caller has what it wants of it: on a
// 16 Mpx detector it is 360 MB. A later read sums the shards again.
// 16 Mpx detector it is 360 MB. A later read puts it together again.
void ReleaseProjection();
[[nodiscard]] uint32_t GetFrameCount() const;
@@ -0,0 +1,90 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
// What ShadowFinder::GetMask shares with its device twin (ShadowMaskGPU): the constants it tests
// against, and the two small fits made over whole rings and sectors, which stay on the host on both
// paths.
#include <cstdint>
#include <vector>
namespace shadow_finder {
// The values of ShadowFinder::SHADOW and ShadowFinder::TRANSMITTING, for the device code, which does
// not include ShadowFinder.h.
inline constexpr uint32_t MASK_SHADOW = 1;
inline constexpr uint32_t MASK_TRANSMITTING = 2;
// A pixel is shadow when its background is below this fraction of the background it is
// compared against.
inline constexpr float SHADOW_RATIO = 0.50f;
// The boundary grows outward into partially shadowed pixels down to this fraction, but no
// further than PENUMBRA_MAX_PX from the core. A pin or a loop casts a wide half-shadow, so the
// reach is a good deal more than the beam stop's own edge needs.
inline constexpr float PENUMBRA_RATIO = 0.75f;
inline constexpr int PENUMBRA_MAX_PX = 30;
// Bridge small breaks along the holder arm. A module gap wider than this is bridged separately for
// the arm search (see bridge_gaps).
inline constexpr int BRIDGE_PX = 6;
// A pixel whose maximum reaches this recorded a real reflection and is never masked - a
// beam stop cannot block a reflection that was measured.
inline constexpr int64_t MIN_REFLECTION = 25;
// How far below the background it is compared against a pixel must sit before the dip is
// believed, in standard deviations of the counts that back it. The counts are photons, so their
// scatter is Poisson and the deficit is measured against it rather than against a fixed number:
// on a well-exposed sweep a third of the background missing is overwhelming, and on a handful of
// low-background frames the same third is noise. Without this a six-frame pre-scan of a
// low-background sweep masks three quarters of the detector.
inline constexpr double MIN_DEFICIT_SIGMA = 6.0;
// Smallest region the per-pixel test may return. A shadow is cast by something physical and is
// correspondingly large; an isolated patch this small is the background wandering, not hardware.
// This is what keeps the test specific now that a shadow no longer has to touch the direct beam.
inline constexpr int MIN_SHADOW_PIXELS = 2000;
// ... and of those, how many must be deep (below SHADOW_RATIO) for a region of merely DIM pixels -
// hardware that lets part of the beam through - to count. A thin holder arm is dim along most of
// its length and deep in places; the background drifting over a detector's edge is dim everywhere
// and deep nowhere.
inline constexpr int MIN_CORE_PIXELS = 200;
// The arm search compares a pixel with its ring as the ring actually varies around the beam. A ring
// of background is not flat once divided by the polarization factor when that factor is not the
// beam's: the remainder is a second harmonic in azimuth, cos 2phi, which reaches tens of percent at
// high angle. It is measured over radial bands of HARMONIC_BAND_PX, from the median of each of
// HARMONIC_SECTORS sectors that holds at least MIN_SECTOR_PIXELS pixels.
inline constexpr int HARMONIC_BAND_PX = 64;
inline constexpr int HARMONIC_SECTORS = 24;
inline constexpr int MIN_SECTOR_PIXELS = 200;
// Side of the box the background is pooled over before testing. Its area is how many pixels back
// a ring's countability test, which decides where an azimuthal comparison is possible at all.
inline constexpr int POOL_PX = 5;
inline constexpr double MEAN_POOLED_PIXELS = POOL_PX * POOL_PX;
// A ring with fewer valid pixels than this says nothing about whether it was counted.
inline constexpr int MIN_RING_PIXELS = 32;
// A ring lies wholly inside the stop when its background is below this fraction of the background
// further out. This asks about a whole ring rather than about a pixel, so it keeps a threshold of
// its own and does not follow SHADOW_RATIO.
inline constexpr float BLOCKED_RING_RATIO = 0.35f;
// The rings that lie wholly inside the stop: walking outward, every judgeable ring (at least
// MIN_RING_PIXELS pixels) before the first whose baseline reaches BLOCKED_RING_RATIO of the typical
// background. -1 when there is none. See GetMask.
int BlockedOutTo(const std::vector<float> &baseline, const std::vector<int> &ring_pixels);
// The second harmonic in azimuth of each radial band, relative to its level, fitted to the medians of
// its HARMONIC_SECTORS sectors (sector_median[band * HARMONIC_SECTORS + k], negative where the sector
// has too few pixels). See GetMask.
void HarmonicFit(const std::vector<double> &sector_median, int n_bands,
std::vector<float> &harm_c, std::vector<float> &harm_s);
} // namespace shadow_finder
+636
View File
@@ -0,0 +1,636 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "ShadowMaskGPU.h"
#include <cub/device/device_radix_sort.cuh>
#include "ShadowFinderInternal.h"
#include "../../common/JFJochMath.h"
#include "../indexing/CUDAMemHelpers.h"
#include "../../common/JFJochException.h"
using namespace shadow_finder;
namespace {
constexpr int THREADS = 256;
void check(cudaError_t err, const char *what) {
if (err != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
std::string("Beam stop mask: ") + what + ": " + cudaGetErrorString(err));
}
unsigned grid(size_t n) {
return static_cast<unsigned>((n + THREADS - 1) / THREADS);
}
// A float as an unsigned integer that sorts the same way, and back.
__device__ uint32_t float_key(float f) {
const uint32_t u = __float_as_uint(f);
return (u & 0x80000000u) ? ~u : (u | 0x80000000u);
}
__device__ float key_float(uint32_t k) {
return __uint_as_float((k & 0x80000000u) ? (k & 0x7fffffffu) : ~k);
}
// The first index in a sorted key array whose key is not below `value`.
__device__ size_t lower_bound(const uint64_t *keys, size_t n, uint64_t value) {
size_t lo = 0, hi = n;
while (lo < hi) {
const size_t mid = (lo + hi) / 2;
if (keys[mid] < value) lo = mid + 1;
else hi = mid;
}
return lo;
}
__device__ double poisson_deficit_sigma(double observed, double expected) {
if (expected <= 0.0 || observed >= expected)
return 0.0;
const double ll = 2.0 * (expected - observed + (observed > 0.0 ? observed * log(observed / expected) : 0.0));
return ll > 0.0 ? sqrt(ll) : 0.0;
}
// DiffractionGeometry::CalcAzIntPolarizationCorr about the centre the rings are drawn about.
__device__ float polarization_factor(const ShadowMaskSetup &s, float x, float y) {
const float u = (x - s.beam_x) * s.pixel_size_mm;
const float v = (y - s.beam_y) * s.pixel_size_mm;
const float *m = s.det_matrix;
const float lx = m[0] * u + m[1] * v + m[2] * s.distance_mm;
const float ly = m[3] * u + m[4] * v + m[5] * s.distance_mm;
const float lz = m[6] * u + m[7] * v + m[8] * s.distance_mm;
const float two_theta = atan2f(sqrtf(lx * lx + ly * ly), lz);
float phi = atan2f(ly, lx);
if (phi < 0)
phi += 2.0f * PI;
const float cos_2theta = cosf(two_theta);
const float cos_2theta_2 = cos_2theta * cos_2theta;
const float cos_2phi = cosf(2.0f * phi);
return 0.5f * (1.0f + cos_2theta_2 - s.polarization * cos_2phi * (1.0f - cos_2theta_2));
}
__global__ void setup_kernel(ShadowMaskSetup s, const uint32_t *__restrict__ pixel_mask,
const int64_t *__restrict__ sum_value, const uint32_t *__restrict__ valid_count,
float *__restrict__ pol, char *__restrict__ valid, int *__restrict__ radius,
double *__restrict__ num, int32_t *__restrict__ den, int *__restrict__ max_radius) {
const size_t n = static_cast<size_t>(s.width) * s.height;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n)
return;
const int x = static_cast<int>(i % s.width), y = static_cast<int>(i / s.width);
const float dx = x - s.beam_x, dy = y - s.beam_y;
const float p = s.has_polarization ? polarization_factor(s, static_cast<float>(x), static_cast<float>(y)) : 1.0f;
pol[i] = p;
float mean = 0.0f;
char v = 0;
if (valid_count[i] > 0 && pixel_mask[i] == 0 && p > 0.0f) {
mean = static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i] / p);
v = 1;
}
valid[i] = v;
num[i] = v ? mean : 0.0;
den[i] = v ? 1 : 0;
const int r = static_cast<int>(lroundf(sqrtf(dx * dx + dy * dy)));
radius[i] = r;
atomicMax(max_radius, r);
}
// Sum over the k x k box about each pixel, zero outside the frame: one running sum per row, then one
// per column, each with exactly the terms and order of the host's (box_sum in ShadowFinder.cpp).
template <typename T>
__global__ void box_rows(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) {
const int y = blockIdx.x * blockDim.x + threadIdx.x;
if (y >= H) return;
const T *src = in + static_cast<size_t>(y) * W;
T *dst = out + static_cast<size_t>(y) * W;
T s = 0;
for (int x = 0; x <= min(half, W - 1); x++)
s += src[x];
for (int x = 0; x < W; x++) {
dst[x] = s;
if (x + half + 1 < W) s += src[x + half + 1];
if (x - half >= 0) s -= src[x - half];
}
}
template <typename T>
__global__ void box_columns(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) {
const int x = blockIdx.x * blockDim.x + threadIdx.x;
if (x >= W) return;
T s = 0;
for (int y = 0; y <= min(half, H - 1); y++)
s += in[static_cast<size_t>(y) * W + x];
for (int y = 0; y < H; y++) {
out[static_cast<size_t>(y) * W + x] = s;
if (y + half + 1 < H) s += in[static_cast<size_t>(y + half + 1) * W + x];
if (y - half >= 0) s -= in[static_cast<size_t>(y - half) * W + x];
}
}
// Dilation of a 0/1 plane by the (2r+1) square clipped to the frame, as a count over a sliding window
// along rows and then along columns (dilate in ShadowFinder.cpp).
__global__ void dilate_rows(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) {
const int y = blockIdx.x * blockDim.x + threadIdx.x;
if (y >= H) return;
const char *src = in + static_cast<size_t>(y) * W;
char *dst = out + static_cast<size_t>(y) * W;
int count = 0;
for (int x = 0; x <= min(r, W - 1); x++)
count += src[x];
for (int x = 0; x < W; x++) {
dst[x] = count > 0;
if (x + r + 1 < W) count += src[x + r + 1];
if (x - r >= 0) count -= src[x - r];
}
}
__global__ void dilate_columns(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) {
const int x = blockIdx.x * blockDim.x + threadIdx.x;
if (x >= W) return;
int count = 0;
for (int y = 0; y <= min(r, H - 1); y++)
count += in[static_cast<size_t>(y) * W + x];
for (int y = 0; y < H; y++) {
out[static_cast<size_t>(y) * W + x] = count > 0;
if (y + r + 1 < H) count += in[static_cast<size_t>(y + r + 1) * W + x];
if (y - r >= 0) count -= in[static_cast<size_t>(y - r) * W + x];
}
}
__global__ void pooled_kernel(size_t n, const double *__restrict__ pooled_sum, const int32_t *__restrict__ pooled_count,
const char *__restrict__ valid, const int *__restrict__ radius,
float *__restrict__ pooled, uint64_t *__restrict__ ring_key) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
const float p = pooled_count[i] > 0 ? static_cast<float>(pooled_sum[i] / pooled_count[i]) : 0.0f;
pooled[i] = p;
ring_key[i] = valid[i] ? (static_cast<uint64_t>(radius[i]) << 32) | float_key(p) : UINT64_MAX;
}
// Where each of the keys' leading 32-bit groups (ring or sector) starts in a sorted key array.
__global__ void group_offsets(const uint64_t *__restrict__ keys, size_t n, int groups, int *__restrict__ offset) {
const int g = blockIdx.x * blockDim.x + threadIdx.x;
if (g > groups) return;
offset[g] = static_cast<int>(lower_bound(keys, n, static_cast<uint64_t>(g) << 32));
}
// The ring's baseline, three iterations of an order statistic over its sorted values (GetMask).
__global__ void baseline_kernel(int rings, const uint64_t *__restrict__ keys, const int *__restrict__ offset,
float *__restrict__ baseline) {
const int r = blockIdx.x * blockDim.x + threadIdx.x;
if (r >= rings) return;
const int lo = offset[r], n = offset[r + 1] - offset[r];
int excluded = 0;
float b = 0.0f;
for (int iter = 0; iter < 3; iter++) {
const int avail = n - excluded;
b = (avail <= 0) ? 0.0f : key_float(static_cast<uint32_t>(keys[lo + excluded + avail / 2]));
const float d = fmaxf(b, 1e-6f);
int excl = 0;
while (excl < n && key_float(static_cast<uint32_t>(keys[lo + excl])) / d < SHADOW_RATIO)
excl++;
excluded = excl;
}
baseline[r] = b;
}
__global__ void low_kernel(size_t n, uint32_t frames, const char *__restrict__ valid, const float *__restrict__ pooled,
const int32_t *__restrict__ pooled_count, const float *__restrict__ pol,
const int *__restrict__ radius, const float *__restrict__ baseline,
float *__restrict__ ratio, float *__restrict__ deficit, char *__restrict__ low) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
const float base = baseline[radius[i]];
const float rt = valid[i] ? pooled[i] / fmaxf(base, 1e-6f) : 1.0f;
ratio[i] = rt;
float df = 0.0f;
char l = 0;
if (valid[i]) {
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
df = static_cast<float>(poisson_deficit_sigma(pooled[i] * counted, base * counted));
l = rt < SHADOW_RATIO && df > MIN_DEFICIT_SIGMA;
}
deficit[i] = df;
low[i] = l;
}
// 8-connected components by union-find: every component ends up named by its smallest pixel index,
// whatever order the unions ran in.
__device__ int find_root(const int *parent, int x) {
while (parent[x] != x)
x = parent[x];
return x;
}
__global__ void cc_init(size_t n, const char *__restrict__ member, int *__restrict__ parent) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
parent[i] = member[i] ? static_cast<int>(i) : -1;
}
__device__ void cc_unite(int *parent, int a, int b) {
while (true) {
a = find_root(parent, a);
b = find_root(parent, b);
if (a == b) return;
if (a < b) { const int t = a; a = b; b = t; }
if (atomicCAS(&parent[a], a, b) == a) return;
}
}
__global__ void cc_union(int W, int H, const char *__restrict__ member, int *parent) {
const size_t n = static_cast<size_t>(W) * H;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n || !member[i]) return;
const int x = static_cast<int>(i % W), y = static_cast<int>(i / W);
// The four neighbours before this pixel; the other four see it from their side.
if (x > 0 && member[i - 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(i - 1));
if (y > 0) {
const size_t up = i - W;
if (member[up]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up));
if (x > 0 && member[up - 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up - 1));
if (x + 1 < W && member[up + 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up + 1));
}
}
__global__ void cc_flatten(size_t n, int *parent) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n || parent[i] < 0) return;
parent[i] = find_root(parent, static_cast<int>(i));
}
// Per component (by root): how many of its pixels are `a`, and how many are both `a` and `b`.
__global__ void cc_count(size_t n, const int *__restrict__ root, const char *__restrict__ a, const char *__restrict__ b,
int *__restrict__ count_a, int *__restrict__ count_ab) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n || root[i] < 0 || !a[i]) return;
atomicAdd(&count_a[root[i]], 1);
if (count_ab && b[i])
atomicAdd(&count_ab[root[i]], 1);
}
__global__ void region_kernel(size_t n, const int *__restrict__ root, const int *__restrict__ n_low,
const char *__restrict__ low, const char *__restrict__ valid, const int *__restrict__ radius,
int blocked_out_to, char *__restrict__ region) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
char r = root[i] >= 0 && n_low[root[i]] >= MIN_SHADOW_PIXELS ? low[i] : 0;
if (valid[i] && radius[i] <= blocked_out_to)
r = 1;
region[i] = r;
}
__global__ void lit_kernel(size_t n, const uint32_t *__restrict__ valid_count, const int64_t *__restrict__ max_value,
char *__restrict__ lit) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
lit[i] = (valid_count[i] > 0) && (max_value[i] >= MIN_REFLECTION);
}
__global__ void reflection_kernel(int W, int H, const char *__restrict__ lit, char *__restrict__ reflection) {
const size_t n = static_cast<size_t>(W) * H;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
const int x = static_cast<int>(i % W), y = static_cast<int>(i / W);
char r = 0;
if (lit[i]) {
int neighbours = 0;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if ((dx || dy) && yy >= 0 && yy < H && xx >= 0 && xx < W && lit[static_cast<size_t>(yy) * W + xx])
neighbours++;
}
r = neighbours >= 2;
}
reflection[i] = r;
}
__global__ void penumbra_kernel(size_t n, const char *__restrict__ penumbra, const char *__restrict__ valid,
const float *__restrict__ ratio, const float *__restrict__ deficit, char *__restrict__ region) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
if (penumbra[i] && valid[i] && ratio[i] < PENUMBRA_RATIO && deficit[i] > MIN_DEFICIT_SIGMA)
region[i] = 1;
}
__global__ void invert_kernel(size_t n, const char *__restrict__ in, char *__restrict__ out) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
out[i] = !in[i];
}
// Background components that touch the frame's edge are outside; the rest are holes.
__global__ void outside_kernel(int W, int H, const int *__restrict__ root, int *__restrict__ outside) {
const int k = blockIdx.x * blockDim.x + threadIdx.x;
const int perimeter = 2 * W + 2 * H;
if (k >= perimeter) return;
int x, y;
if (k < W) { x = k; y = 0; }
else if (k < 2 * W) { x = k - W; y = H - 1; }
else if (k < 2 * W + H) { x = 0; y = k - 2 * W; }
else { x = W - 1; y = k - 2 * W - H; }
const int r = root[static_cast<size_t>(y) * W + x];
if (r >= 0) outside[r] = 1;
}
__global__ void final_kernel(size_t n, const int *__restrict__ background_root, const int *__restrict__ outside,
const char *__restrict__ reflection_grown, char *__restrict__ region,
uint32_t *__restrict__ mask) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
char r = region[i];
if (background_root[i] >= 0 && !outside[background_root[i]])
r = 1; // a hole
if (reflection_grown[i])
r = 0;
region[i] = r;
mask[i] = r ? MASK_SHADOW : 0;
}
__global__ void sector_key_kernel(ShadowMaskSetup s, const char *__restrict__ valid, const char *__restrict__ region,
const int *__restrict__ radius, const float *__restrict__ ratio,
uint64_t *__restrict__ key) {
const size_t n = static_cast<size_t>(s.width) * s.height;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
if (!valid[i] || region[i]) {
key[i] = UINT64_MAX;
return;
}
const float dx = static_cast<float>(i % s.width) - s.beam_x, dy = static_cast<float>(i / s.width) - s.beam_y;
const double phi = atan2f(dy, dx) + PI;
const int sector = min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * PI) * HARMONIC_SECTORS));
const uint64_t k = static_cast<uint64_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector;
key[i] = (k << 32) | float_key(ratio[i]);
}
__global__ void sector_median_kernel(int sectors, const uint64_t *__restrict__ keys, const int *__restrict__ offset,
double *__restrict__ median) {
const int k = blockIdx.x * blockDim.x + threadIdx.x;
if (k >= sectors) return;
const int n = offset[k + 1] - offset[k];
median[k] = n >= MIN_SECTOR_PIXELS ? key_float(static_cast<uint32_t>(keys[offset[k] + n / 2])) : -1.0;
}
__global__ void dim_kernel(ShadowMaskSetup s, uint32_t frames, const char *__restrict__ valid,
const char *__restrict__ region, const float *__restrict__ ratio,
const float *__restrict__ deficit, const int *__restrict__ radius,
const float *__restrict__ harm_c, const float *__restrict__ harm_s,
const int32_t *__restrict__ pooled_count, const float *__restrict__ pol,
const float *__restrict__ pooled, const float *__restrict__ baseline,
char *__restrict__ dim) {
const size_t n = static_cast<size_t>(s.width) * s.height;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
char d = 0;
if (valid[i] && !region[i] && ratio[i] < PENUMBRA_RATIO) {
// cos 2phi and sin 2phi from the offset to the beam.
const float dx = static_cast<float>(i % s.width) - s.beam_x, dy = static_cast<float>(i / s.width) - s.beam_y;
const float r2 = fmaxf(dx * dx + dy * dy, 1e-6f);
const int band = radius[i] / HARMONIC_BAND_PX;
const float model = fminf(1.0f, 1.0f + harm_c[band] * (dx * dx - dy * dy) / r2
+ harm_s[band] * 2.0f * dx * dy / r2);
if (model >= 1.0f) {
d = deficit[i] > MIN_DEFICIT_SIGMA;
} else {
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
d = ratio[i] < PENUMBRA_RATIO * model
&& poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA;
}
}
dim[i] = d;
}
// Join a region across the module gaps it crosses, one line per thread (bridge_gaps in
// ShadowFinder.cpp). Both directions read `region` and only ever set pixels of `out` to 1.
__global__ void bridge_rows(int W, int H, const char *__restrict__ region, const char *__restrict__ valid,
char *out) {
const int y = blockIdx.x * blockDim.x + threadIdx.x;
if (y >= H) return;
const size_t row = static_cast<size_t>(y) * W;
int k = 0;
while (k < W) {
if (valid[row + k]) { k++; continue; }
const int start = k;
while (k < W && !valid[row + k]) k++;
if (start > 0 && k < W && region[row + start - 1] && region[row + k])
for (int j = start; j < k; j++) out[row + j] = 1;
}
}
__global__ void bridge_columns(int W, int H, const char *__restrict__ region, const char *__restrict__ valid,
char *out) {
const int x = blockIdx.x * blockDim.x + threadIdx.x;
if (x >= W) return;
const auto at = [&](int y) { return static_cast<size_t>(y) * W + x; };
int k = 0;
while (k < H) {
if (valid[at(k)]) { k++; continue; }
const int start = k;
while (k < H && !valid[at(k)]) k++;
if (start > 0 && k < H && region[at(start - 1)] && region[at(k)])
for (int j = start; j < k; j++) out[at(j)] = 1;
}
}
__global__ void transmitting_kernel(size_t n, const int *__restrict__ root, const char *__restrict__ dim,
const int *__restrict__ n_dim, const int *__restrict__ n_low,
uint32_t *__restrict__ mask) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
const int c = root[i];
if (c >= 0 && dim[i] && n_dim[c] >= MIN_SHADOW_PIXELS && n_low[c] >= MIN_CORE_PIXELS)
mask[i] = MASK_TRANSMITTING;
}
__global__ void mean_kernel(size_t n, const uint32_t *__restrict__ pixel_mask, const int64_t *__restrict__ sum_value,
const uint32_t *__restrict__ valid_count, float *__restrict__ mean) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0
? static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]) : NAN;
}
// The device side of one GetMask: planes, sort scratch and the stream to run on.
class MaskEngine {
public:
const ShadowMaskSetup s;
const int W, H;
const size_t n;
cudaStream_t stream;
MaskEngine(const ShadowMaskSetup &setup, cudaStream_t st)
: s(setup), W(setup.width), H(setup.height), n(static_cast<size_t>(setup.width) * setup.height), stream(st) {}
void Check(const char *what) const {
check(cudaGetLastError(), what);
}
void Dilate(const char *in, char *out, char *scratch, int r) const {
dilate_rows<<<grid(H), THREADS, 0, stream>>>(in, scratch, W, H, r);
dilate_columns<<<grid(W), THREADS, 0, stream>>>(scratch, out, W, H, r);
Check("dilate");
}
// Components of `member`, each pixel's root in `root` (-1 outside every component).
void Label(const char *member, int *root) const {
cc_init<<<grid(n), THREADS, 0, stream>>>(n, member, root);
cc_union<<<grid(n), THREADS, 0, stream>>>(W, H, member, root);
cc_flatten<<<grid(n), THREADS, 0, stream>>>(n, root);
Check("components");
}
void SortKeys(uint64_t *keys, uint64_t *sorted) const {
size_t bytes = 0;
check(cub::DeviceRadixSort::SortKeys(nullptr, bytes, keys, sorted, n, 0, 64, stream), "sort size");
CudaDevicePtr<uint8_t> scratch(bytes);
check(cub::DeviceRadixSort::SortKeys(scratch.get(), bytes, keys, sorted, n, 0, 64, stream), "sort");
// The scratch is freed on the allocation stream, which knows nothing of this one.
check(cudaStreamSynchronize(stream), "sort");
}
template <typename T>
std::vector<T> Download(const T *device, size_t count) const {
std::vector<T> host(count);
check(cudaMemcpyAsync(host.data(), device, count * sizeof(T), cudaMemcpyDeviceToHost, stream), "download");
check(cudaStreamSynchronize(stream), "download");
return host;
}
template <typename T>
void Upload(T *device, const std::vector<T> &host) const {
check(cudaMemcpyAsync(device, host.data(), host.size() * sizeof(T), cudaMemcpyHostToDevice, stream), "upload");
}
};
} // namespace
std::vector<uint32_t> ShadowMaskOnDevice(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask,
const int64_t *max_value, const int64_t *sum_value,
const uint32_t *valid_count, uint32_t frames, cudaStream_t stream) {
const MaskEngine e(setup, stream);
const size_t n = e.n;
const int W = e.W, H = e.H;
CudaDevicePtr<uint32_t> d_pixel_mask(n), mask(n);
e.Upload(d_pixel_mask.get(), pixel_mask);
// Mean projection over the polarization factor, usable pixels and radius from the beam centre.
CudaDevicePtr<float> pol(n), pooled(n), ratio(n), deficit(n);
CudaDevicePtr<char> valid(n), low(n), region(n), a(n), b(n), c(n);
CudaDevicePtr<int> radius(n), root(n), count_a(n), count_b(n), max_radius(1);
CudaDevicePtr<double> num(n), dsum(n);
CudaDevicePtr<int32_t> den(n), dcount(n);
check(cudaMemsetAsync(max_radius.get(), 0, sizeof(int), stream), "memset");
setup_kernel<<<grid(n), THREADS, 0, stream>>>(setup, d_pixel_mask, sum_value, valid_count, pol, valid, radius,
num, den, max_radius);
e.Check("setup");
const int max_r = e.Download(max_radius.get(), 1)[0];
const int rings = max_r + 1;
// The background pooled over a small box.
box_rows<double><<<grid(H), THREADS, 0, stream>>>(num, dsum, W, H, POOL_PX / 2);
box_columns<double><<<grid(W), THREADS, 0, stream>>>(dsum, num, W, H, POOL_PX / 2);
box_rows<int32_t><<<grid(H), THREADS, 0, stream>>>(den, dcount, W, H, POOL_PX / 2);
box_columns<int32_t><<<grid(W), THREADS, 0, stream>>>(dcount, den, W, H, POOL_PX / 2);
e.Check("pooling");
const double *pooled_sum = num;
const int32_t *pooled_count = den;
// The rings, each sorted once; the baseline is an order statistic of them.
CudaDevicePtr<uint64_t> keys(n), sorted(n);
pooled_kernel<<<grid(n), THREADS, 0, stream>>>(n, pooled_sum, pooled_count, valid, radius, pooled, keys);
e.Check("pooled");
e.SortKeys(keys, sorted);
CudaDevicePtr<int> ring_offset(rings + 1);
group_offsets<<<grid(rings + 1), THREADS, 0, stream>>>(sorted, n, rings, ring_offset);
CudaDevicePtr<float> baseline(rings);
baseline_kernel<<<grid(rings), THREADS, 0, stream>>>(rings, sorted, ring_offset, baseline);
e.Check("baseline");
const auto host_baseline = e.Download(baseline.get(), rings);
const auto offsets = e.Download(ring_offset.get(), rings + 1);
std::vector<int> ring_pixels(rings);
for (int r = 0; r < rings; r++)
ring_pixels[r] = offsets[r + 1] - offsets[r];
const int blocked_out_to = BlockedOutTo(host_baseline, ring_pixels);
// Low pixels, and the regions of them large enough to be hardware.
low_kernel<<<grid(n), THREADS, 0, stream>>>(n, frames, valid, pooled, pooled_count, pol, radius, baseline,
ratio, deficit, low);
e.Check("low");
e.Dilate(low, a, c, BRIDGE_PX);
e.Label(a, root);
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
cc_count<<<grid(n), THREADS, 0, stream>>>(n, root, low, low, count_a, nullptr);
region_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, count_a, low, valid, radius, blocked_out_to, region);
e.Check("region");
// Recorded reflections; `b` holds them until they are given back at the end.
lit_kernel<<<grid(n), THREADS, 0, stream>>>(n, valid_count, max_value, a);
reflection_kernel<<<grid(n), THREADS, 0, stream>>>(W, H, a, b);
e.Check("reflections");
// Penumbra, round and fill.
e.Dilate(region, a, c, PENUMBRA_MAX_PX);
penumbra_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, valid, ratio, deficit, region);
e.Dilate(region, a, c, 2);
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, region);
e.Dilate(region, a, c, 2);
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, region); // region = erode(dilate(region))
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, region, a); // the background
e.Label(a, root);
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
outside_kernel<<<grid(2 * W + 2 * H), THREADS, 0, stream>>>(W, H, root, count_a);
e.Dilate(b, c, a, 1); // the reflections, grown by one
final_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, count_a, c, region, mask);
e.Check("fill");
// The arm search: sector medians of the ratio, the harmonic of each band, the dim pixels.
const int n_bands = max_r / HARMONIC_BAND_PX + 1;
const int n_sectors = n_bands * HARMONIC_SECTORS;
sector_key_kernel<<<grid(n), THREADS, 0, stream>>>(setup, valid, region, radius, ratio, keys);
e.Check("sectors");
e.SortKeys(keys, sorted);
CudaDevicePtr<int> sector_offset(n_sectors + 1);
group_offsets<<<grid(n_sectors + 1), THREADS, 0, stream>>>(sorted, n, n_sectors, sector_offset);
CudaDevicePtr<double> sector_median(n_sectors);
sector_median_kernel<<<grid(n_sectors), THREADS, 0, stream>>>(n_sectors, sorted, sector_offset, sector_median);
e.Check("sector medians");
std::vector<float> harm_c, harm_s;
HarmonicFit(e.Download(sector_median.get(), n_sectors), n_bands, harm_c, harm_s);
CudaDevicePtr<float> d_harm_c(n_bands), d_harm_s(n_bands);
e.Upload(d_harm_c.get(), harm_c);
e.Upload(d_harm_s.get(), harm_s);
dim_kernel<<<grid(n), THREADS, 0, stream>>>(setup, frames, valid, region, ratio, deficit, radius, d_harm_c, d_harm_s,
pooled_count, pol, pooled, baseline, a);
e.Check("dim");
e.Dilate(a, b, c, BRIDGE_PX);
check(cudaMemcpyAsync(c.get(), b.get(), n, cudaMemcpyDeviceToDevice, stream), "copy");
bridge_rows<<<grid(H), THREADS, 0, stream>>>(W, H, b, valid, c);
bridge_columns<<<grid(W), THREADS, 0, stream>>>(W, H, b, valid, c);
e.Label(c, root);
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
check(cudaMemsetAsync(count_b.get(), 0, n * sizeof(int), stream), "memset");
cc_count<<<grid(n), THREADS, 0, stream>>>(n, root, a, low, count_a, count_b);
transmitting_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, a, count_a, count_b, mask);
e.Check("transmitting");
return e.Download(mask.get(), n);
}
std::vector<float> MeanProjectionOnDevice(const std::vector<uint32_t> &pixel_mask, const int64_t *sum_value,
const uint32_t *valid_count, size_t npixels, cudaStream_t stream) {
CudaDevicePtr<uint32_t> d_pixel_mask(npixels);
CudaDevicePtr<float> mean(npixels);
check(cudaMemcpyAsync(d_pixel_mask.get(), pixel_mask.data(), npixels * sizeof(uint32_t), cudaMemcpyHostToDevice,
stream), "upload");
mean_kernel<<<grid(npixels), THREADS, 0, stream>>>(npixels, d_pixel_mask, sum_value, valid_count, mean);
check(cudaGetLastError(), "mean");
std::vector<float> host(npixels);
check(cudaMemcpyAsync(host.data(), mean.get(), npixels * sizeof(float), cudaMemcpyDeviceToHost, stream), "download");
check(cudaStreamSynchronize(stream), "mean");
return host;
}
+39
View File
@@ -0,0 +1,39 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
// Included only under JFJOCH_USE_CUDA.
#include <cstdint>
#include <vector>
#include <cuda_runtime.h>
// What the mask is drawn about: the detector, the centre the rings are drawn about, and the geometry
// the polarization factor is read off (ShadowFinder keeps it as a DiffractionGeometry; the device
// takes it as numbers).
struct ShadowMaskSetup {
int width = 0, height = 0;
float beam_x = 0.0f, beam_y = 0.0f;
float det_matrix[9] = {}; // row major
float pixel_size_mm = 0.0f;
float distance_mm = 0.0f;
bool has_polarization = false;
float polarization = 0.0f;
};
// ShadowFinder::GetMask on the device, from the projection ShadowAccumulatorGPU holds there. Step for
// step the host's algorithm - the same pooling, ring medians, components, morphology and arm search -
// and the same answer wherever the arithmetic is exact: every integer, comparison, sort and component
// is. What is not is the floating point the two compilers evaluate differently - the polarization
// factor's trigonometry, the Poisson test's logarithm and the azimuth of the arm search - so a pixel
// within a rounding of one of those thresholds can come out the other way.
std::vector<uint32_t> ShadowMaskOnDevice(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask,
const int64_t *max_value, const int64_t *sum_value,
const uint32_t *valid_count, uint32_t frames, cudaStream_t stream);
// The mean projection the host's GetMeanProjection makes, computed where the sums are: the same
// division, so the same bits, and a quarter of the bytes to bring back.
std::vector<float> MeanProjectionOnDevice(const std::vector<uint32_t> &pixel_mask, const int64_t *sum_value,
const uint32_t *valid_count, size_t npixels, cudaStream_t stream);
@@ -45,6 +45,7 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys
const Coord m3 = (m1 % m2).Normalize();
const float m2_S0 = m2 * S0;
const float four_S0_sq = 4 * S0 * S0;
const float m3_S0 = m3 * S0;
int i = 0;
@@ -99,17 +100,24 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys
cos_phi_limit = std::cos(phi_limit);
}
// p0 = A* h + B* k + C* l, evaluated as ((A* h) + (B* k)) + (C* l) exactly as before, with the
// terms that do not change in the inner loops taken out of them.
std::vector<Coord> Cstar_l(2 * settings.max_l + 1);
for (int l = -settings.max_l; l <= settings.max_l; l++)
Cstar_l[l + settings.max_l] = Cstar * l;
for (int h = -settings.max_h; h <= settings.max_h; h++) {
// Precompute A* h contribution
const Coord Astar_h = Astar * h;
for (int k = -settings.max_k; k <= settings.max_k; k++) {
// Accumulate B* k contribution
const Coord AB = Astar_h + Bstar * k;
for (int l = -settings.max_l; l <= settings.max_l; l++) {
if (systematic_absence(h, k, l, settings.centering))
continue;
Coord p0 = Astar * h + Bstar * k + Cstar * l;
const Coord &Cl = Cstar_l[l + settings.max_l];
const Coord p0(AB.x + Cl.x, AB.y + Cl.y, AB.z + Cl.z);
float p0_sq = p0 * p0;
if (p0_sq <= 0.0f || p0_sq > one_over_dmax_sq)
@@ -129,7 +137,7 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys
};
// No solution for Laue equations
if ((rho_sq < p_m3 * p_m3) || (p0_sq > 4 * S0 * S0))
if ((rho_sq < p_m3 * p_m3) || (p0_sq > four_S0_sq))
continue;
// Effective rocking width for this reflection: mosaicity broadened by the bandwidth
@@ -0,0 +1,107 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
// Where one pixel falls in the background beam-centre fit (FindBeamCenterFromBackground): which
// radial-bin x sector cell of the fitted band it lands in about a trial centre, and the derivative
// of its 2theta with respect to that centre. Written once and compiled both for the host fit and for
// its device twin (BeamCenterBackgroundGPU), so the two read the same formula.
//
// The same formula is the same number only when both sides evaluate it the same way. The library
// atan2f differs between glibc and CUDA in the last bit, and the compilers fuse multiply-adds each
// in their own way; either moves a pixel within a rounding of a cell edge into the neighbouring cell
// (measured on 16 Mpx sweeps: ~30 of 6.5 million pixels, enough to move the fitted centre by up to
// 0.05 px). So the angles come from BackgroundAtan2 below and both translation units are compiled
// without contraction (geom_refinement/CMakeLists.txt), and the host and the device then agree to
// the bit.
#include <cmath>
#include <cstdint>
#include "../../common/JFJochMath.h"
#ifdef __CUDACC__
#define BACKGROUND_BAND_HD __host__ __device__ inline
#else
#define BACKGROUND_BAND_HD inline
#endif
struct BackgroundBand {
static constexpr int SECTORS = 36;
static constexpr int RADIAL_BINS = 120;
static constexpr int CELLS = RADIAL_BINS * SECTORS;
float rot[9]; // detector matrix, row major
float pixel_size; // mm
float distance; // mm
float tt_lo, tt_hi; // the band in 2theta
float d_tt; // width of one radial bin
double tan_lo, tan_hi; // the band in tan(2theta), widened for the quick rejection
};
// atan2(y, x) from IEEE operations alone - add, multiply, divide, square root, all correctly rounded on
// the host and on the device - so that, compiled without contraction (see CMakeLists.txt), the host
// and the device return the same bits. The library atan2f does not: glibc's and CUDA's differ in the
// last place. Two half-angle reductions take the argument below tan(pi/16), where eleven terms of the
// series leave an error under 1e-17.
BACKGROUND_BAND_HD double BackgroundAtan2(double y, double x) {
const double ax = x < 0 ? -x : x, ay = y < 0 ? -y : y;
if (ax == 0.0 && ay == 0.0)
return 0.0;
const bool swap = ay > ax;
double t = swap ? ax / ay : ay / ax;
t = t / (1.0 + sqrt(1.0 + t * t));
t = t / (1.0 + sqrt(1.0 + t * t));
const double t2 = t * t;
double series = 1.0 / 23.0;
for (int k = 21; k >= 1; k -= 2)
series = 1.0 / k - t2 * series;
double a = 4.0 * t * series;
if (swap) a = PI / 2 - a;
if (x < 0) a = PI - a;
return y < 0 ? -a : a;
}
// The cell of pixel (x, y) about (beam_x, beam_y), or -1 when it is outside the band; for a pixel in
// it, also the two components of the derivative of its 2theta with respect to the centre.
BACKGROUND_BAND_HD int BackgroundBandCell(const BackgroundBand &b, int x, int y, float beam_x, float beam_y,
float &jac_x, float &jac_y) {
const float *rot = b.rot;
const float u = (x - beam_x) * b.pixel_size;
const float v = (y - beam_y) * b.pixel_size;
const float lx = rot[0] * u + rot[1] * v + rot[2] * b.distance;
const float ly = rot[3] * u + rot[4] * v + rot[5] * b.distance;
const float lz = rot[6] * u + rot[7] * v + rot[8] * b.distance;
const float rho_sq = lx * lx + ly * ly;
// Most pixels lie outside the band. Those clearly outside it in tan(2theta) = rho / lz - by a
// margin far above float rounding - skip the square root and both atan2 below; every pixel the
// exact test would keep still reaches it.
if (lz > 0.0f) {
const double lz_sq = static_cast<double>(lz) * lz;
if (rho_sq < b.tan_lo * b.tan_lo * lz_sq || rho_sq > b.tan_hi * b.tan_hi * lz_sq)
return -1;
}
const float rho = sqrtf(rho_sq);
const float two_theta = static_cast<float>(BackgroundAtan2(rho, lz));
if (two_theta < b.tt_lo || two_theta >= b.tt_hi || rho == 0.0f)
return -1;
const float phi = static_cast<float>(BackgroundAtan2(ly, lx));
// Both bins are clamped: a pixel one float ulp below the top of the band divides to exactly
// RADIAL_BINS, which is one cell past the end of every accumulator.
int r_bin = static_cast<int>((two_theta - b.tt_lo) / b.d_tt);
r_bin = r_bin < 0 ? 0 : (r_bin > BackgroundBand::RADIAL_BINS - 1 ? BackgroundBand::RADIAL_BINS - 1 : r_bin);
int s_bin = static_cast<int>((phi + PI) / (2 * PI) * BackgroundBand::SECTORS);
s_bin = s_bin < 0 ? 0 : (s_bin > BackgroundBand::SECTORS - 1 ? BackgroundBand::SECTORS - 1 : s_bin);
// d(2theta)/d(beam), through the lab coordinate: the detector coordinate depends on the centre
// only as (x - beam_x), so moving the centre is moving the pixel.
const float denominator = rho * rho + lz * lz;
const float g_x = lz * lx / (rho * denominator);
const float g_y = lz * ly / (rho * denominator);
const float g_z = -rho / denominator;
jac_x = -b.pixel_size * (g_x * rot[0] + g_y * rot[3] + g_z * rot[6]);
jac_y = -b.pixel_size * (g_x * rot[1] + g_y * rot[4] + g_z * rot[7]);
return r_bin * BackgroundBand::SECTORS + s_bin;
}
@@ -0,0 +1,210 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "BeamCenterBackgroundGPU.h"
#include <cub/device/device_radix_sort.cuh>
#include "../indexing/CUDAMemHelpers.h"
namespace {
constexpr int THREADS = 256;
void check(cudaError_t err, const char *what) {
if (err != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
std::string("Beam centre from background: ") + what + ": " + cudaGetErrorString(err));
}
// Every pixel's cell (BackgroundBand::CELLS where it is outside the band or unusable), its index,
// and its derivatives.
__global__ void bin_kernel(BackgroundBand band, int width, size_t npixels, float beam_x, float beam_y,
const char *__restrict__ usable, int32_t *__restrict__ key,
int32_t *__restrict__ index, float *__restrict__ jac_x,
float *__restrict__ jac_y) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= npixels)
return;
int cell = BackgroundBand::CELLS;
float jx = 0.0f, jy = 0.0f;
if (usable[i]) {
const int c = BackgroundBandCell(band, static_cast<int>(i % width), static_cast<int>(i / width),
beam_x, beam_y, jx, jy);
if (c >= 0)
cell = c;
}
key[i] = cell;
index[i] = static_cast<int32_t>(i);
jac_x[i] = jx;
jac_y[i] = jy;
}
// Where each cell's pixels start in the sorted keys; offset[CELLS] is the number of binned pixels.
__global__ void offset_kernel(const int32_t *__restrict__ sorted_key, size_t n, int32_t *__restrict__ offset) {
const int c = blockIdx.x * blockDim.x + threadIdx.x;
if (c > BackgroundBand::CELLS)
return;
size_t lo = 0, hi = n;
while (lo < hi) {
const size_t mid = (lo + hi) / 2;
if (sorted_key[mid] < c) lo = mid + 1;
else hi = mid;
}
offset[c] = static_cast<int32_t>(lo);
}
// One thread per cell, walking its pixels in pixel order: a partial sum per host row block, added to
// the cell's total when the block changes - the host's arithmetic, step for step. A cell whose pixels
// are clipped (clip_limit != nullptr) skips the ones above its limit, as the host's clip rounds do.
__global__ void sum_kernel(int width, const int32_t *__restrict__ offset, const int32_t *__restrict__ index,
const int32_t *__restrict__ row_block, const float *__restrict__ mean,
const float *__restrict__ jac_x, const float *__restrict__ jac_y,
const float *__restrict__ clip_limit,
double *__restrict__ sum, double *__restrict__ sum_sq,
double *__restrict__ sum_jx, double *__restrict__ sum_jy,
int32_t *__restrict__ count) {
const int c = blockIdx.x * blockDim.x + threadIdx.x;
if (c >= BackgroundBand::CELLS)
return;
const bool clipped = clip_limit != nullptr;
const float limit = clipped ? clip_limit[c] : 0.0f;
double s = 0, ss = 0, jx = 0, jy = 0;
double bs = 0, bss = 0, bjx = 0, bjy = 0;
int32_t n = 0;
int block = -1;
for (int32_t p = offset[c]; p < offset[c + 1]; p++) {
const int32_t i = index[p];
const float value = mean[i];
if (clipped && (limit < 0.0f || value > limit))
continue;
const int b = row_block[i / width];
if (b != block) {
s += bs; ss += bss; jx += bjx; jy += bjy;
bs = bss = bjx = bjy = 0;
block = b;
}
n++;
bs += value;
bss += static_cast<double>(value) * value;
if (!clipped) {
bjx += jac_x[i];
bjy += jac_y[i];
}
}
s += bs; ss += bss; jx += bjx; jy += bjy;
sum[c] = s;
sum_sq[c] = ss;
count[c] = n;
if (!clipped) {
sum_jx[c] = jx;
sum_jy[c] = jy;
}
}
} // namespace
struct BeamCenterBackgroundGPU::Impl {
int width, height;
size_t npixels;
CudaStream stream;
CudaDevicePtr<char> usable;
CudaDevicePtr<float> mean;
CudaDevicePtr<int32_t> row_block;
CudaDevicePtr<int32_t> key, sorted_key, index, sorted_index;
CudaDevicePtr<float> jac_x, jac_y;
CudaDevicePtr<int32_t> offset;
CudaDevicePtr<float> clip_limit;
CudaDevicePtr<double> sum, sum_sq, sum_jx, sum_jy;
CudaDevicePtr<int32_t> count;
CudaDevicePtr<uint8_t> sort_scratch;
size_t sort_scratch_bytes = 0;
Impl(int w, int h)
: width(w), height(h), npixels(static_cast<size_t>(w) * h),
usable(npixels), mean(npixels), row_block(h),
key(npixels), sorted_key(npixels), index(npixels), sorted_index(npixels),
jac_x(npixels), jac_y(npixels), offset(BackgroundBand::CELLS + 1),
clip_limit(BackgroundBand::CELLS),
sum(BackgroundBand::CELLS), sum_sq(BackgroundBand::CELLS),
sum_jx(BackgroundBand::CELLS), sum_jy(BackgroundBand::CELLS),
count(BackgroundBand::CELLS) {
// Keys run to CELLS inclusive, the bin of everything outside the band.
check(cub::DeviceRadixSort::SortPairs(nullptr, sort_scratch_bytes, key.get(), sorted_key.get(),
index.get(), sorted_index.get(), npixels, 0, end_bit(), stream),
"sort size");
sort_scratch = CudaDevicePtr<uint8_t>(sort_scratch_bytes);
}
static int end_bit() {
int bits = 0;
while ((1 << bits) <= BackgroundBand::CELLS) bits++;
return bits;
}
void Download(std::vector<double> &s, std::vector<double> &ss, std::vector<double> *jx,
std::vector<double> *jy, std::vector<int32_t> &n) {
const size_t cells = BackgroundBand::CELLS;
check(cudaMemcpyAsync(s.data(), sum.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy");
check(cudaMemcpyAsync(ss.data(), sum_sq.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy");
if (jx) check(cudaMemcpyAsync(jx->data(), sum_jx.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy");
if (jy) check(cudaMemcpyAsync(jy->data(), sum_jy.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy");
check(cudaMemcpyAsync(n.data(), count.get(), cells * sizeof(int32_t), cudaMemcpyDeviceToHost, stream), "copy");
check(cudaStreamSynchronize(stream), "sums");
}
};
BeamCenterBackgroundGPU::BeamCenterBackgroundGPU(int width, int height, const std::vector<int> &block_row,
const char *usable, const float *mean)
: impl(std::make_unique<Impl>(width, height)) {
std::vector<int32_t> row_block(height);
for (size_t b = 0; b + 1 < block_row.size(); b++)
for (int y = block_row[b]; y < block_row[b + 1]; y++)
row_block[y] = static_cast<int32_t>(b);
check(cudaMemcpyAsync(impl->row_block.get(), row_block.data(), height * sizeof(int32_t),
cudaMemcpyHostToDevice, impl->stream), "upload");
check(cudaMemcpyAsync(impl->usable.get(), usable, impl->npixels, cudaMemcpyHostToDevice, impl->stream), "upload");
check(cudaMemcpyAsync(impl->mean.get(), mean, impl->npixels * sizeof(float), cudaMemcpyHostToDevice,
impl->stream), "upload");
check(cudaStreamSynchronize(impl->stream), "upload");
}
BeamCenterBackgroundGPU::~BeamCenterBackgroundGPU() = default;
void BeamCenterBackgroundGPU::Bin(const BackgroundBand &band, float beam_x, float beam_y,
std::vector<double> &sum, std::vector<double> &sum_sq,
std::vector<double> &sum_jx, std::vector<double> &sum_jy,
std::vector<int32_t> &count) {
Impl &d = *impl;
const auto blocks = static_cast<unsigned>((d.npixels + THREADS - 1) / THREADS);
bin_kernel<<<blocks, THREADS, 0, d.stream>>>(band, d.width, d.npixels, beam_x, beam_y, d.usable,
d.key, d.index, d.jac_x, d.jac_y);
check(cudaGetLastError(), "bin");
// A radix sort is stable, so each cell's pixels come out in pixel order.
check(cub::DeviceRadixSort::SortPairs(d.sort_scratch.get(), d.sort_scratch_bytes, d.key.get(),
d.sorted_key.get(), d.index.get(), d.sorted_index.get(),
d.npixels, 0, Impl::end_bit(), d.stream), "sort");
constexpr int cell_blocks = (BackgroundBand::CELLS + 1 + THREADS - 1) / THREADS;
offset_kernel<<<cell_blocks, THREADS, 0, d.stream>>>(d.sorted_key, d.npixels, d.offset);
check(cudaGetLastError(), "offsets");
sum_kernel<<<cell_blocks, THREADS, 0, d.stream>>>(d.width, d.offset, d.sorted_index, d.row_block, d.mean,
d.jac_x, d.jac_y, nullptr, d.sum, d.sum_sq,
d.sum_jx, d.sum_jy, d.count);
check(cudaGetLastError(), "sums");
d.Download(sum, sum_sq, &sum_jx, &sum_jy, count);
}
void BeamCenterBackgroundGPU::Clip(const std::vector<float> &clip_limit,
std::vector<double> &sum, std::vector<double> &sum_sq,
std::vector<int32_t> &count) {
Impl &d = *impl;
check(cudaMemcpyAsync(d.clip_limit.get(), clip_limit.data(), BackgroundBand::CELLS * sizeof(float),
cudaMemcpyHostToDevice, d.stream), "upload");
constexpr int cell_blocks = (BackgroundBand::CELLS + THREADS - 1) / THREADS;
sum_kernel<<<cell_blocks, THREADS, 0, d.stream>>>(d.width, d.offset, d.sorted_index, d.row_block, d.mean,
d.jac_x, d.jac_y, d.clip_limit, d.sum, d.sum_sq,
d.sum_jx, d.sum_jy, d.count);
check(cudaGetLastError(), "clip");
d.Download(sum, sum_sq, nullptr, nullptr, count);
}
@@ -0,0 +1,40 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
// Included only under JFJOCH_USE_CUDA. Free of CUDA headers, so the host fit can hold one.
#include <cstdint>
#include <memory>
#include <vector>
#include "BackgroundBand.h"
// The two passes over the pixels of the background beam-centre fit (FindBeamCenterFromBackground),
// on the device: binning every usable pixel into its cell about a trial centre, and summing the
// binned pixels again under a clip. They are all of the fit's cost; the fit itself stays on the host.
//
// Each cell is summed in the order the host sums it - pixel order within each of the host's row
// blocks, the blocks then added in block order - so a pixel that lands in the same cell on both
// sides adds the same rounding on both. Whatever differs comes from BackgroundBandCell (see there).
class BeamCenterBackgroundGPU {
struct Impl;
std::unique_ptr<Impl> impl;
public:
// block_row: the first row of each of the host's row blocks, and one past the last row at the end.
BeamCenterBackgroundGPU(int width, int height, const std::vector<int> &block_row,
const char *usable, const float *mean);
~BeamCenterBackgroundGPU();
// Bin the band about (beam_x, beam_y) and sum each cell: the values, their squares, the two
// derivatives, and the count.
void Bin(const BackgroundBand &band, float beam_x, float beam_y,
std::vector<double> &sum, std::vector<double> &sum_sq,
std::vector<double> &sum_jx, std::vector<double> &sum_jy, std::vector<int32_t> &count);
// Sum the pixels the last Bin put in each cell again, leaving out those above the cell's
// clip_limit and every pixel of a cell whose limit is negative.
void Clip(const std::vector<float> &clip_limit,
std::vector<double> &sum, std::vector<double> &sum_sq, std::vector<int32_t> &count);
};
@@ -2,13 +2,19 @@
// SPDX-License-Identifier: GPL-3.0-only
#include "BeamCenterFromBackground.h"
#include "BackgroundBand.h"
#include <algorithm>
#include <cmath>
#include <thread>
#include "../../common/CompressedImage.h"
#include "../../common/JFJochMath.h"
#include "../../common/ParallelFor.h"
#ifdef JFJOCH_USE_CUDA
#include "../../common/CUDAWrapper.h"
#include "BeamCenterBackgroundGPU.h"
#endif
namespace {
@@ -17,8 +23,8 @@ namespace {
constexpr float BAND_LOW_RES_A = 12.0f;
constexpr float BAND_HIGH_RES_A = 2.2f;
constexpr int SECTORS = 36;
constexpr int RADIAL_BINS = 120;
constexpr int SECTORS = BackgroundBand::SECTORS;
constexpr int RADIAL_BINS = BackgroundBand::RADIAL_BINS;
// A cell with fewer pixels than this has no usable mean.
constexpr int MIN_PIXELS_PER_CELL = 20;
@@ -83,7 +89,7 @@ float median_of(std::vector<float> &v) {
std::optional<BeamCenterEstimate>
FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const PixelMask &mask,
const std::vector<float> &mean, size_t nthreads,
std::optional<std::pair<float, float>> start) {
std::optional<std::pair<float, float>> start, bool allow_device) {
if (nthreads == 0)
nthreads = std::max(1u, std::thread::hardware_concurrency());
const auto W = static_cast<int>(experiment.GetXPixelsNumConv());
@@ -111,7 +117,6 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe
float beam_y = start ? start->second : geom.GetBeamY_pxl();
constexpr int n_cells = RADIAL_BINS * SECTORS;
std::vector<int32_t> cell_of(n_pixels);
std::vector<double> sum(n_cells), sum_sq(n_cells), sum_jx(n_cells), sum_jy(n_cells);
std::vector<int32_t> count(n_cells), count_all(n_cells);
std::vector<float> profile(RADIAL_BINS), d_profile(RADIAL_BINS), clip_limit(n_cells);
@@ -127,16 +132,69 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe
std::vector<double> block_jy(static_cast<size_t>(BLOCKS) * n_cells);
std::vector<int32_t> block_count(static_cast<size_t>(BLOCKS) * n_cells);
// Whether a pixel can take part at all, which does not depend on the centre.
std::vector<char, NoInitAllocator<char>> usable(n_pixels);
ParallelFor(BLOCKS, nthreads, [&](int b) {
for (size_t i = static_cast<size_t>(block_row[b]) * W; i < static_cast<size_t>(block_row[b + 1]) * W; i++)
usable[i] = pixel_mask[i] == 0 && std::isfinite(mean[i]);
});
// The pixels of each block that fall in the band at the current centre, with their cell and value,
// in pixel order from the block's first pixel on. The clipping rounds read these instead of the
// whole detector, in the same order.
std::vector<int32_t, NoInitAllocator<int32_t>> band_cell(n_pixels);
std::vector<float, NoInitAllocator<float>> band_value(n_pixels);
std::vector<size_t> band_pixels(BLOCKS);
// The blocks' cells folded in block order. Each cell is folded on its own, so the cells are split
// over the threads and every cell is still summed in the same order.
const auto fold = [&](bool with_jacobian) {
ParallelChunks(n_cells, nthreads, [&](int c0, int c1) {
for (int c = c0; c < c1; c++) {
double s = 0, ss = 0, jx = 0, jy = 0;
int32_t n = 0;
for (int b = 0; b < BLOCKS; b++) {
const size_t k = static_cast<size_t>(b) * n_cells + c;
s += block_sum[k]; ss += block_sum_sq[k];
if (with_jacobian) { jx += block_jx[k]; jy += block_jy[k]; }
n += block_count[k];
}
sum[c] = s; sum_sq[c] = ss; count[c] = n;
if (with_jacobian) { sum_jx[c] = jx; sum_jy[c] = jy; }
}
});
};
// Most pixels lie outside the band. Those clearly outside it in tan(2theta) = rho / lz - by a
// margin far above float rounding - skip the square root and both atan2 below; every pixel the exact
// test would keep still reaches it.
const double tan_lo = std::tan(static_cast<double>(tt_lo)) * (1.0 - 1e-3);
const double tan_hi = tt_hi < PI / 2 ? std::tan(static_cast<double>(tt_hi)) * (1.0 + 1e-3) : INFINITY;
BackgroundBand band{};
for (int k = 0; k < 9; k++)
band.rot[k] = rot[k];
band.pixel_size = pixel_size;
band.distance = distance;
band.tt_lo = tt_lo;
band.tt_hi = tt_hi;
band.d_tt = d_tt;
band.tan_lo = tan_lo;
band.tan_hi = tan_hi;
#ifndef JFJOCH_USE_CUDA
(void) allow_device;
#else
// With a GPU the two passes over the pixels run there, and only the cells come back.
std::unique_ptr<BeamCenterBackgroundGPU> gpu;
if (allow_device && get_gpu_count() > 0)
gpu = std::make_unique<BeamCenterBackgroundGPU>(W, H, block_row, usable.data(), mean.data());
#endif
float step_x = 0.0f, step_y = 0.0f, sigma_x = 0.0f, sigma_y = 0.0f;
float previous_x = 0.0f, previous_y = 0.0f;
int reversals = 0;
for (int iteration = 0; iteration < MAX_ITERATIONS; iteration++) {
// The binning pass about the current centre, then the clip rounds over the pixels it binned.
const auto bin_cpu = [&] {
ParallelFor(BLOCKS, nthreads, [&](int b) {
double *b_sum = block_sum.data() + static_cast<size_t>(b) * n_cells;
double *b_sum_sq = block_sum_sq.data() + static_cast<size_t>(b) * n_cells;
@@ -148,61 +206,62 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe
std::fill(b_jx, b_jx + n_cells, 0.0);
std::fill(b_jy, b_jy + n_cells, 0.0);
std::fill(b_count, b_count + n_cells, 0);
int32_t *cells = band_cell.data() + static_cast<size_t>(block_row[b]) * W;
float *values = band_value.data() + static_cast<size_t>(block_row[b]) * W;
size_t n_band = 0;
for (int y = block_row[b]; y < block_row[b + 1]; y++) {
for (int x = 0; x < W; x++) {
const size_t i = static_cast<size_t>(y) * W + x;
cell_of[i] = -1;
if (pixel_mask[i] != 0 || !std::isfinite(mean[i]))
if (!usable[i])
continue;
const float u = (x - beam_x) * pixel_size;
const float v = (y - beam_y) * pixel_size;
const float lx = rot[0] * u + rot[1] * v + rot[2] * distance;
const float ly = rot[3] * u + rot[4] * v + rot[5] * distance;
const float lz = rot[6] * u + rot[7] * v + rot[8] * distance;
const float rho_sq = lx * lx + ly * ly;
if (lz > 0.0f) {
const double lz_sq = static_cast<double>(lz) * lz;
if (rho_sq < tan_lo * tan_lo * lz_sq || rho_sq > tan_hi * tan_hi * lz_sq)
continue;
}
const float rho = std::sqrt(rho_sq);
const float two_theta = std::atan2(rho, lz);
if (two_theta < tt_lo || two_theta >= tt_hi || rho == 0.0f)
float jac_x, jac_y;
const int cell = BackgroundBandCell(band, x, y, beam_x, beam_y, jac_x, jac_y);
if (cell < 0)
continue;
const float phi = std::atan2(ly, lx);
// Both bins are clamped: a pixel one float ulp below the top of the band divides
// to exactly RADIAL_BINS, which is one cell past the end of every accumulator.
const int r_bin = std::clamp(static_cast<int>((two_theta - tt_lo) / d_tt), 0, RADIAL_BINS - 1);
const int s_bin = std::clamp(static_cast<int>((phi + PI) / (2 * PI) * SECTORS), 0, SECTORS - 1);
const int cell = r_bin * SECTORS + s_bin;
// d(2theta)/d(beam), through the lab coordinate: the detector coordinate depends
// on the centre only as (x - beam_x), so moving the centre is moving the pixel.
const float denominator = rho * rho + lz * lz;
const float g_x = lz * lx / (rho * denominator);
const float g_y = lz * ly / (rho * denominator);
const float g_z = -rho / denominator;
cell_of[i] = cell;
cells[n_band] = cell;
values[n_band] = mean[i];
n_band++;
b_count[cell]++;
b_sum[cell] += mean[i];
b_sum_sq[cell] += static_cast<double>(mean[i]) * mean[i];
b_jx[cell] += -pixel_size * (g_x * rot[0] + g_y * rot[3] + g_z * rot[6]);
b_jy[cell] += -pixel_size * (g_x * rot[1] + g_y * rot[4] + g_z * rot[7]);
b_jx[cell] += jac_x;
b_jy[cell] += jac_y;
}
}
band_pixels[b] = n_band;
});
for (int c = 0; c < n_cells; c++) {
double s = 0, ss = 0, jx = 0, jy = 0;
int32_t n = 0;
for (int b = 0; b < BLOCKS; b++) {
const size_t k = static_cast<size_t>(b) * n_cells + c;
s += block_sum[k]; ss += block_sum_sq[k];
jx += block_jx[k]; jy += block_jy[k];
n += block_count[k];
fold(true);
};
const auto clip_cpu = [&] {
ParallelFor(BLOCKS, nthreads, [&](int b) {
double *b_sum = block_sum.data() + static_cast<size_t>(b) * n_cells;
double *b_sum_sq = block_sum_sq.data() + static_cast<size_t>(b) * n_cells;
int32_t *b_count = block_count.data() + static_cast<size_t>(b) * n_cells;
std::fill(b_sum, b_sum + n_cells, 0.0);
std::fill(b_sum_sq, b_sum_sq + n_cells, 0.0);
std::fill(b_count, b_count + n_cells, 0);
const int32_t *cells = band_cell.data() + static_cast<size_t>(block_row[b]) * W;
const float *values = band_value.data() + static_cast<size_t>(block_row[b]) * W;
for (size_t j = 0; j < band_pixels[b]; j++) {
const int32_t c = cells[j];
const float value = values[j];
if (clip_limit[c] < 0.0f || value > clip_limit[c])
continue;
b_count[c]++;
b_sum[c] += value;
b_sum_sq[c] += static_cast<double>(value) * value;
}
sum[c] = s; sum_sq[c] = ss; sum_jx[c] = jx; sum_jy[c] = jy; count[c] = n;
}
});
fold(false);
};
for (int iteration = 0; iteration < MAX_ITERATIONS; iteration++) {
#ifdef JFJOCH_USE_CUDA
if (gpu)
gpu->Bin(band, beam_x, beam_y, sum, sum_sq, sum_jx, sum_jy, count);
else
#endif
bin_cpu();
count_all = count; // the Jacobian sums belong to the unclipped pixel set
@@ -213,33 +272,12 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe
const double variance = std::max(sum_sq[c] / count[c] - m * m, 0.0);
clip_limit[c] = static_cast<float>(m + CLIP_SIGMA * std::sqrt(variance));
}
ParallelFor(BLOCKS, nthreads, [&](int b) {
double *b_sum = block_sum.data() + static_cast<size_t>(b) * n_cells;
double *b_sum_sq = block_sum_sq.data() + static_cast<size_t>(b) * n_cells;
int32_t *b_count = block_count.data() + static_cast<size_t>(b) * n_cells;
std::fill(b_sum, b_sum + n_cells, 0.0);
std::fill(b_sum_sq, b_sum_sq + n_cells, 0.0);
std::fill(b_count, b_count + n_cells, 0);
const size_t lo = static_cast<size_t>(block_row[b]) * W;
const size_t hi = static_cast<size_t>(block_row[b + 1]) * W;
for (size_t i = lo; i < hi; i++) {
const int32_t c = cell_of[i];
if (c < 0 || clip_limit[c] < 0.0f || mean[i] > clip_limit[c])
continue;
b_count[c]++;
b_sum[c] += mean[i];
b_sum_sq[c] += static_cast<double>(mean[i]) * mean[i];
}
});
for (int c = 0; c < n_cells; c++) {
double s = 0, ss = 0;
int32_t n = 0;
for (int b = 0; b < BLOCKS; b++) {
const size_t k = static_cast<size_t>(b) * n_cells + c;
s += block_sum[k]; ss += block_sum_sq[k]; n += block_count[k];
}
sum[c] = s; sum_sq[c] = ss; count[c] = n;
}
#ifdef JFJOCH_USE_CUDA
if (gpu)
gpu->Clip(clip_limit, sum, sum_sq, count);
else
#endif
clip_cpu();
}
// Radial profile: the median over the sectors that have a mean, on rings that are
@@ -40,10 +40,14 @@ struct BeamCenterEstimate {
// `start` is where the walk begins; the centre in the file when it is not given. The walk advances
// by a bounded distance per iteration, so where it starts decides how much of its budget is spent
// travelling and - on a surface with more than one basin - which fixed point it can reach at all.
//
// With a GPU the passes over the pixels run on it (BeamCenterBackgroundGPU); allow_device = false
// keeps them on the host, which is what the parity test compares against.
std::optional<BeamCenterEstimate>
FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const PixelMask &mask,
const std::vector<float> &mean, size_t nthreads = 0,
std::optional<std::pair<float, float>> start = {});
std::optional<std::pair<float, float>> start = {},
bool allow_device = true);
// The precision of a centre that is the FFT capture alone, with no walk behind it. The capture is
// a half-pixel grid position read off a surface, measured over 75 rotation datasets at a median
+18 -1
View File
@@ -24,6 +24,12 @@ ADD_LIBRARY(JFJochGeomRefinement STATIC
XtalOptimizer.cpp
XtalOptimizer.h
XtalResidual.h
XtalRefine.cpp
XtalRefine.h
LMSolver.cpp
LMSolver.h
Dual.h
BackgroundBand.h
PostRefine.cpp
PostRefine.h
GeometryRefiner.cpp
@@ -37,7 +43,8 @@ ADD_LIBRARY(JFJochGeomRefinement STATIC
TARGET_LINK_LIBRARIES(JFJochGeomRefinement Ceres::ceres Eigen3::Eigen JFJochCommon fftw3f)
IF (JFJOCH_CUDA_AVAILABLE)
TARGET_SOURCES(JFJochGeomRefinement PRIVATE BeamCenterFFTGPU.cu BeamCenterFFTGPU.h)
TARGET_SOURCES(JFJochGeomRefinement PRIVATE BeamCenterFFTGPU.cu BeamCenterFFTGPU.h
BeamCenterBackgroundGPU.cu BeamCenterBackgroundGPU.h)
# Same static/dynamic cuFFT choice as the FFT indexer, and for the same reasons - see the long
# note in image_analysis/indexing/CMakeLists.txt.
IF (JFJOCH_PORTABLE_ONLY AND TARGET CUDA::cufft_static)
@@ -46,3 +53,13 @@ IF (JFJOCH_CUDA_AVAILABLE)
TARGET_LINK_LIBRARIES(JFJochGeomRefinement CUDA::cufft)
ENDIF()
ENDIF()
# The background beam-centre fit bins every pixel with BackgroundBand.h on the host and on the device
# and must get the same bits on both (see there). That needs no multiply-add contracted on either side:
# GCC and Clang contract by default and nvcc does too, while MSVC does not without /fp:contract.
IF (JFJOCH_CUDA_AVAILABLE)
SET_SOURCE_FILES_PROPERTIES(BeamCenterBackgroundGPU.cu PROPERTIES COMPILE_OPTIONS "--fmad=false")
ENDIF()
IF (CMAKE_CXX_COMPILER_ID MATCHES "GNU|Clang")
SET_SOURCE_FILES_PROPERTIES(BeamCenterFromBackground.cpp PROPERTIES COMPILE_OPTIONS "-ffp-contract=off")
ENDIF()
+142
View File
@@ -0,0 +1,142 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
// A forward-mode dual number with N derivative lanes: a value and its gradient with respect to N
// parameters. The residuals of the crystal refinement are written as templates over their scalar type,
// so the same code runs on a plain double and on this. The value part of every operation is the plain
// double arithmetic of the same expression - written the way ceres::Jet writes it, division through the
// reciprocal - so a residual evaluated on a Dual has the same value as on a Jet.
#include <cmath>
#include <limits>
#include <Eigen/Core>
template<int N>
struct Dual {
double a = 0.0;
double v[N] = {};
Dual() = default;
Dual(double value) : a(value) {} // NOLINT: implicit, a constant is a dual with zero derivatives
static Dual Variable(double value, int lane) {
Dual d(value);
d.v[lane] = 1.0;
return d;
}
Dual &operator+=(const Dual &o) { a += o.a; for (int i = 0; i < N; i++) v[i] += o.v[i]; return *this; }
Dual &operator-=(const Dual &o) { a -= o.a; for (int i = 0; i < N; i++) v[i] -= o.v[i]; return *this; }
Dual &operator*=(const Dual &o) { *this = *this * o; return *this; }
Dual &operator/=(const Dual &o) { *this = *this / o; return *this; }
friend Dual operator+(const Dual &x) { return x; }
friend Dual operator-(const Dual &x) {
Dual r(-x.a);
for (int i = 0; i < N; i++) r.v[i] = -x.v[i];
return r;
}
friend Dual operator+(const Dual &x, const Dual &y) {
Dual r(x.a + y.a);
for (int i = 0; i < N; i++) r.v[i] = x.v[i] + y.v[i];
return r;
}
friend Dual operator+(const Dual &x, double s) { Dual r = x; r.a += s; return r; }
friend Dual operator+(double s, const Dual &x) { Dual r = x; r.a += s; return r; }
friend Dual operator-(const Dual &x, const Dual &y) {
Dual r(x.a - y.a);
for (int i = 0; i < N; i++) r.v[i] = x.v[i] - y.v[i];
return r;
}
friend Dual operator-(const Dual &x, double s) { Dual r = x; r.a -= s; return r; }
friend Dual operator-(double s, const Dual &x) {
Dual r(s - x.a);
for (int i = 0; i < N; i++) r.v[i] = -x.v[i];
return r;
}
friend Dual operator*(const Dual &x, const Dual &y) {
Dual r(x.a * y.a);
for (int i = 0; i < N; i++) r.v[i] = x.a * y.v[i] + x.v[i] * y.a;
return r;
}
friend Dual operator*(const Dual &x, double s) {
Dual r(x.a * s);
for (int i = 0; i < N; i++) r.v[i] = x.v[i] * s;
return r;
}
friend Dual operator*(double s, const Dual &x) { return x * s; }
friend Dual operator/(const Dual &x, const Dual &y) {
const double y_inv = 1.0 / y.a;
const double q = x.a * y_inv;
Dual r(q);
for (int i = 0; i < N; i++) r.v[i] = (x.v[i] - q * y.v[i]) * y_inv;
return r;
}
friend Dual operator/(const Dual &x, double s) {
const double s_inv = 1.0 / s;
return x * s_inv;
}
friend Dual operator/(double s, const Dual &y) {
const double y_inv = 1.0 / y.a;
const double d = -s * y_inv * y_inv;
Dual r(s * y_inv);
for (int i = 0; i < N; i++) r.v[i] = d * y.v[i];
return r;
}
friend bool operator<(const Dual &x, const Dual &y) { return x.a < y.a; }
friend bool operator>(const Dual &x, const Dual &y) { return x.a > y.a; }
friend bool operator<=(const Dual &x, const Dual &y) { return x.a <= y.a; }
friend bool operator>=(const Dual &x, const Dual &y) { return x.a >= y.a; }
friend bool operator==(const Dual &x, const Dual &y) { return x.a == y.a; }
friend bool operator!=(const Dual &x, const Dual &y) { return x.a != y.a; }
// The chain rule for a function of one argument: value f, derivative df.
Dual Chain(double f, double df) const {
Dual r(f);
for (int i = 0; i < N; i++) r.v[i] = df * v[i];
return r;
}
friend Dual sqrt(const Dual &x) {
const double s = std::sqrt(x.a);
return x.Chain(s, 0.5 / s);
}
friend Dual cos(const Dual &x) { return x.Chain(std::cos(x.a), -std::sin(x.a)); }
friend Dual sin(const Dual &x) { return x.Chain(std::sin(x.a), std::cos(x.a)); }
friend Dual hypot(const Dual &x, const Dual &y, const Dual &z) {
// As ceres::hypot(Jet, Jet, Jet): the value is std::hypot, the derivative x/h dx + y/h dy + z/h dz.
const double h = std::hypot(x.a, y.a, z.a);
Dual r(h);
for (int i = 0; i < N; i++) r.v[i] = x.a / h * x.v[i] + y.a / h * y.v[i] + z.a / h * z.v[i];
return r;
}
friend int fpclassify(const Dual &x) { return std::fpclassify(x.a); }
};
// What Eigen needs to hold a Dual in a fixed-size matrix (the reciprocal basis is built in one).
namespace Eigen {
template<int N>
struct NumTraits<Dual<N>> : GenericNumTraits<double> {
typedef Dual<N> Real;
typedef Dual<N> NonInteger;
typedef Dual<N> Nested;
typedef Dual<N> Literal;
enum {
IsComplex = 0, IsInteger = 0, IsSigned = 1, RequireInitialization = 1,
ReadCost = 1, AddCost = 1, MulCost = 1
};
static inline Real epsilon() { return Real(std::numeric_limits<double>::epsilon()); }
static inline Real dummy_precision() { return Real(1e-12); }
static inline Real highest() { return Real(std::numeric_limits<double>::max()); }
static inline Real lowest() { return Real(-std::numeric_limits<double>::max()); }
static inline int digits10() { return NumTraits<double>::digits10(); }
};
}
+241
View File
@@ -0,0 +1,241 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
// Adapted from https://github.com/ceres-solver/ceres-solver (internal/ceres/polynomial.cc,
// include/ceres/internal/sphere_manifold_functions.h, householder_vector.h)
// Copyright 2023 Google Inc. All rights reserved.
// BSD-3-Clause, see licenses/ceres-solver.txt
#include "LMSolver.h"
#include <Eigen/Eigenvalues>
namespace {
void HouseholderVector3(const double x[3], double v[3], double &beta) {
const double sigma = x[0] * x[0] + x[1] * x[1];
v[0] = x[0];
v[1] = x[1];
v[2] = 1.0;
beta = 0.0;
const double x_pivot = x[2];
if (sigma <= std::numeric_limits<double>::epsilon()) {
if (x_pivot < 0.0)
beta = 2.0;
return;
}
const double mu = std::sqrt(x_pivot * x_pivot + sigma);
const double v_pivot = (x_pivot <= 0.0) ? x_pivot - mu : -sigma / (x_pivot + mu);
beta = 2.0 * v_pivot * v_pivot / (sigma + v_pivot * v_pivot);
v[0] /= v_pivot;
v[1] /= v_pivot;
}
double Norm3(const double x[3]) {
return std::sqrt(x[0] * x[0] + x[1] * x[1] + x[2] * x[2]);
}
using Vector = Eigen::VectorXd;
using Matrix = Eigen::MatrixXd;
double EvaluatePolynomial(const Vector &polynomial, double x) {
double v = 0.0;
for (int i = 0; i < polynomial.size(); ++i)
v = v * x + polynomial(i);
return v;
}
void BalanceCompanionMatrix(Matrix &companion_matrix) {
Matrix offdiagonal = companion_matrix;
offdiagonal.diagonal().setZero();
const int degree = static_cast<int>(companion_matrix.rows());
const double gamma = 0.9;
bool scaling_has_changed;
do {
scaling_has_changed = false;
for (int i = 0; i < degree; ++i) {
const double col_norm = offdiagonal.col(i).lpNorm<1>();
if (std::fpclassify(col_norm) != FP_ZERO) {
const double row_norm = offdiagonal.row(i).lpNorm<1>();
int exponent = 0;
std::frexp(row_norm / col_norm, &exponent);
exponent /= 2;
if (exponent != 0) {
const double scaled_col_norm = std::ldexp(col_norm, exponent);
const double scaled_row_norm = std::ldexp(row_norm, -exponent);
if (scaled_col_norm + scaled_row_norm < gamma * (col_norm + row_norm)) {
scaling_has_changed = true;
offdiagonal.row(i) *= std::ldexp(1.0, -exponent);
offdiagonal.col(i) *= std::ldexp(1.0, exponent);
}
}
}
}
} while (scaling_has_changed);
offdiagonal.diagonal() = companion_matrix.diagonal();
companion_matrix = offdiagonal;
}
// Real parts of the roots, as Ceres' FindPolynomialRoots (the imaginary parts are not used here).
bool FindPolynomialRoots(const Vector &polynomial_in, Vector &real) {
if (polynomial_in.size() == 0)
return false;
int lead = 0;
while (lead < polynomial_in.size() - 1 && polynomial_in(lead) == 0.0)
++lead;
Vector polynomial = polynomial_in.tail(polynomial_in.size() - lead);
const int degree = static_cast<int>(polynomial.size()) - 1;
if (degree == 0) {
real.resize(0);
return true;
}
if (degree == 1) {
real.resize(1);
real(0) = -polynomial(1) / polynomial(0);
return true;
}
if (degree == 2) {
const double a = polynomial(0);
const double b = polynomial(1);
const double c = polynomial(2);
const double D = b * b - 4 * a * c;
const double sqrt_D = std::sqrt(std::fabs(D));
real.setZero(2);
if (D >= 0) {
if (b >= 0) {
real(0) = (-b - sqrt_D) / (2.0 * a);
real(1) = (2.0 * c) / (-b - sqrt_D);
} else {
real(0) = (2.0 * c) / (-b + sqrt_D);
real(1) = (-b + sqrt_D) / (2.0 * a);
}
} else {
real(0) = -b / (2.0 * a);
real(1) = -b / (2.0 * a);
}
return true;
}
polynomial /= polynomial(0);
Matrix companion = Matrix::Zero(degree, degree);
companion.diagonal(-1).setOnes();
companion.col(degree - 1) = -polynomial.reverse().head(degree);
BalanceCompanionMatrix(companion);
Eigen::EigenSolver<Matrix> solver(companion, false);
if (solver.info() != Eigen::Success)
return false;
real = solver.eigenvalues().real();
return true;
}
void MinimizePolynomial(const Vector &polynomial, double x_min, double x_max,
double &optimal_x, double &optimal_value) {
optimal_x = (x_min + x_max) / 2.0;
optimal_value = EvaluatePolynomial(polynomial, optimal_x);
const double x_min_value = EvaluatePolynomial(polynomial, x_min);
if (x_min_value < optimal_value) {
optimal_value = x_min_value;
optimal_x = x_min;
}
const double x_max_value = EvaluatePolynomial(polynomial, x_max);
if (x_max_value < optimal_value) {
optimal_value = x_max_value;
optimal_x = x_max;
}
if (polynomial.rows() <= 2)
return;
const int degree = static_cast<int>(polynomial.rows()) - 1;
Vector derivative(degree);
for (int i = 0; i < degree; ++i)
derivative(i) = (degree - i) * polynomial(i);
Vector roots_real;
if (!FindPolynomialRoots(derivative, roots_real))
return;
for (int i = 0; i < roots_real.rows(); ++i) {
const double root = roots_real(i);
if (root < x_min || root > x_max)
continue;
const double value = EvaluatePolynomial(polynomial, root);
if (value < optimal_value) {
optimal_value = value;
optimal_x = root;
}
}
}
Vector FindInterpolatingPolynomial(const std::vector<LMLineSample> &samples) {
int num_constraints = 0;
for (const auto &s: samples)
num_constraints += (s.value_is_valid ? 1 : 0) + (s.gradient_is_valid ? 1 : 0);
const int degree = num_constraints - 1;
Matrix lhs = Matrix::Zero(num_constraints, num_constraints);
Vector rhs = Vector::Zero(num_constraints);
int row = 0;
for (const auto &s: samples) {
if (s.value_is_valid) {
for (int j = 0; j <= degree; ++j)
lhs(row, j) = std::pow(s.x, degree - j);
rhs(row) = s.value;
++row;
}
if (s.gradient_is_valid) {
for (int j = 0; j < degree; ++j)
lhs(row, j) = (degree - j) * std::pow(s.x, degree - j - 1);
rhs(row) = s.gradient;
++row;
}
}
Eigen::FullPivLU<Matrix> lu(lhs);
return lu.setThreshold(0.0).solve(rhs);
}
}
void SpherePlus3(const double x[3], const double delta[2], double out[3]) {
const double norm_delta = std::sqrt(delta[0] * delta[0] + delta[1] * delta[1]);
if (norm_delta == 0.0) {
out[0] = x[0];
out[1] = x[1];
out[2] = x[2];
return;
}
double v[3], beta;
HouseholderVector3(x, v, beta);
const double sin_delta_by_delta = std::sin(norm_delta) / norm_delta;
const double y[3] = {sin_delta_by_delta * delta[0], sin_delta_by_delta * delta[1], std::cos(norm_delta)};
const double vy = v[0] * y[0] + v[1] * y[1] + v[2] * y[2];
const double x_norm = Norm3(x);
for (int i = 0; i < 3; i++)
out[i] = x_norm * (y[i] - v[i] * (beta * vy));
}
void SpherePlusJacobian3(const double x[3], double jacobian[3][2]) {
double v[3], beta;
HouseholderVector3(x, v, beta);
const double x_norm = Norm3(x);
for (int i = 0; i < 2; ++i)
for (int r = 0; r < 3; ++r)
jacobian[r][i] = (-beta * v[i] * v[r] + (r == i ? 1.0 : 0.0)) * x_norm;
}
double LMInterpolatedStepSize(const LMLineSample &lowerbound, const LMLineSample &previous,
const LMLineSample &current, double min_step, double max_step) {
if (!current.value_is_valid)
return std::min(std::max(current.x * 0.5, min_step), max_step);
std::vector<LMLineSample> samples{lowerbound, current};
if (previous.value_is_valid)
samples.push_back(previous);
const Vector polynomial = FindInterpolatingPolynomial(samples);
double step = 0.0, value = 0.0;
MinimizePolynomial(polynomial, min_step, max_step, step, value);
for (const auto &s: samples) {
if (s.x < min_step || s.x > max_step)
continue;
const double v = EvaluatePolynomial(polynomial, s.x);
if (v < value) {
step = s.x;
value = v;
}
}
return step;
}
+365
View File
@@ -0,0 +1,365 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
// The minimiser below follows Ceres Solver's trust-region Levenberg-Marquardt step for step - its
// options, its Jacobi scaling, its damping and radius updates, its stopping rules, its box projection,
// its projected Armijo line search on bounded problems and its SphereManifold - so that a problem moved
// off Ceres takes the same path to the same answer. Adapted from
// https://github.com/ceres-solver/ceres-solver (internal/ceres/trust_region_minimizer.cc,
// levenberg_marquardt_strategy.cc, trust_region_step_evaluator.cc, line_search.cc, polynomial.cc,
// include/ceres/internal/sphere_manifold_functions.h, householder_vector.h)
// Copyright 2023 Google Inc. All rights reserved.
// BSD-3-Clause, see licenses/ceres-solver.txt
//
// What it does NOT take from Ceres is the Jacobian: the caller hands over J^T J and J^T r directly,
// accumulated however it likes, and the minimiser never sees a row of J. Everything Ceres computes from
// the Jacobian - the column norms, the normal equations, the model cost change - is a function of those
// two alone. Only the options the crystal refinements use are reproduced: monotonic steps, no inner
// iterations, a dense Cholesky of the normal equations.
#pragma once
#include <chrono>
#include <cmath>
#include <limits>
#include <vector>
#include <Eigen/Dense>
struct LMBlock {
int offset = 0; // into the ambient parameter vector
int size = 0; // ambient size, at most 3
bool constant = false;
bool sphere = false; // Ceres' SphereManifold: the norm is kept, the tangent has size - 1 coordinates
double lower[3] = {-std::numeric_limits<double>::max(), -std::numeric_limits<double>::max(),
-std::numeric_limits<double>::max()};
double upper[3] = {std::numeric_limits<double>::max(), std::numeric_limits<double>::max(),
std::numeric_limits<double>::max()};
int TangentSize() const { return constant ? 0 : (sphere ? size - 1 : size); }
};
struct LMOptions {
int max_iterations = 50;
double max_time_s = 1e9;
};
enum class LMTermination { Convergence, NoConvergence, Failure };
struct LMSummary {
LMTermination termination = LMTermination::Failure;
// Iterations as Ceres counts them in Summary::iterations: iteration 0 included.
int iterations = 0;
int evaluations = 0;
int line_search_steps = 0;
double initial_cost = 0.0;
double final_cost = 0.0;
bool IsSolutionUsable() const { return termination != LMTermination::Failure; }
};
// Ceres' SphereManifold<3> for a three-vector: the Householder reflection that takes x to the pole, the
// Plus that walks a tangent step along the sphere, and the 3x2 Jacobian of that Plus at zero step.
void SpherePlus3(const double x[3], const double delta[2], double out[3]);
void SpherePlusJacobian3(const double x[3], double jacobian[3][2]);
// The polynomial step-size choice of Ceres' Armijo line search, cubic interpolation.
struct LMLineSample {
double x = 0.0;
double value = 0.0;
double gradient = 0.0;
bool value_is_valid = false;
bool gradient_is_valid = false;
};
double LMInterpolatedStepSize(const LMLineSample &lowerbound, const LMLineSample &previous,
const LMLineSample &current, double min_step, double max_step);
// Evaluate is called as eval(x, cost, g, H): x the ambient parameters, cost 1/2 sum of squared
// residuals, and - where g and H are not null - the gradient J^T r and J^T J in the TANGENT coordinates
// of the non-constant blocks, in block order. It returns false where anything came out non-finite.
// x is updated in place on success; it is left untouched on failure.
template<class Evaluate>
LMSummary SolveLM(std::vector<double> &x_io, const std::vector<LMBlock> &blocks, const LMOptions &options,
Evaluate &&eval) {
using Vec = Eigen::VectorXd;
using Mat = Eigen::MatrixXd;
constexpr double kMax = std::numeric_limits<double>::max();
// Ceres' defaults, which is what the callers always ran with.
constexpr double initial_radius = 1e4;
constexpr double max_radius = 1e16;
constexpr double min_radius = 1e-32;
constexpr double min_relative_decrease = 1e-3;
constexpr double min_lm_diagonal = 1e-6;
constexpr double max_lm_diagonal = 1e32;
constexpr int max_consecutive_invalid_steps = 5;
constexpr double function_tolerance = 1e-6;
constexpr double gradient_tolerance = 1e-10;
constexpr double parameter_tolerance = 1e-8;
constexpr double sufficient_decrease = 1e-4;
constexpr double max_step_contraction = 1e-3;
constexpr double min_step_contraction = 0.6;
constexpr double min_line_search_step = 1e-9;
constexpr int max_line_search_iterations = 20;
const auto start = std::chrono::steady_clock::now();
LMSummary summary;
int n = 0;
bool constrained = false;
for (const auto &b: blocks) {
for (int j = 0; j < b.size; j++)
if (!std::isfinite(x_io[b.offset + j]))
return summary;
n += b.TangentSize();
for (int j = 0; j < b.size; j++) {
if (b.constant) {
if (x_io[b.offset + j] < b.lower[j] || x_io[b.offset + j] > b.upper[j])
return summary;
} else {
if (b.lower[j] >= b.upper[j])
return summary;
if (b.lower[j] > -kMax || b.upper[j] < kMax)
constrained = true;
}
}
}
// x (+) delta, block by block, projected onto the bounds - Ceres' ParameterBlock::Plus.
const auto plus = [&](const Vec &x, const Vec &delta, Vec &out) {
out = x;
int t = 0;
for (const auto &b: blocks) {
if (b.constant)
continue;
if (b.sphere) {
SpherePlus3(x.data() + b.offset, delta.data() + t, out.data() + b.offset);
} else {
for (int j = 0; j < b.size; j++)
out[b.offset + j] = x[b.offset + j] + delta[t + j];
}
for (int j = 0; j < b.size; j++) {
out[b.offset + j] = std::max(out[b.offset + j], b.lower[j]);
out[b.offset + j] = std::min(out[b.offset + j], b.upper[j]);
}
t += b.TangentSize();
}
};
// Norms over the parameters Ceres keeps in its state: the non-constant blocks only.
const auto free_norm = [&](const Vec &v) {
double s = 0.0;
for (const auto &b: blocks)
if (!b.constant)
for (int j = 0; j < b.size; j++)
s += v[b.offset + j] * v[b.offset + j];
return std::sqrt(s);
};
const auto free_max_norm = [&](const Vec &v) {
double m = 0.0;
for (const auto &b: blocks)
if (!b.constant)
for (int j = 0; j < b.size; j++)
m = std::max(m, std::fabs(v[b.offset + j]));
return m;
};
Vec x = Eigen::Map<const Vec>(x_io.data(), static_cast<Eigen::Index>(x_io.size()));
if (constrained) {
Vec projected;
plus(x, Vec::Zero(n), projected);
x = projected;
}
// One evaluation point with everything Ceres computes there.
struct Point {
Vec x;
double cost = kMax;
Vec g;
Mat H;
bool valid = false;
};
const auto evaluate = [&](const Vec &at, Point &p) {
p.x = at;
p.g.setZero(n);
p.H.setZero(n, n);
summary.evaluations++;
p.valid = eval(at.data(), p.cost, &p.g, &p.H) && std::isfinite(p.cost);
if (!p.valid)
p.cost = kMax;
};
Point cur;
evaluate(x, cur);
if (!cur.valid)
return summary;
summary.initial_cost = cur.cost;
// Jacobi scaling, fixed from the Jacobian at the starting point.
Vec scale(n);
for (int i = 0; i < n; i++)
scale[i] = 1.0 / (1.0 + std::sqrt(cur.H(i, i)));
Vec gs, neg_g, projected;
Mat Hs;
double gradient_max_norm = 0.0;
const auto take_point = [&]() {
gs = scale.cwiseProduct(cur.g);
Hs = scale.asDiagonal() * cur.H * scale.asDiagonal();
neg_g = -cur.g;
plus(cur.x, neg_g, projected);
gradient_max_norm = free_max_norm(cur.x - projected);
};
take_point();
double radius = initial_radius;
double decrease_factor = 2.0;
bool reuse_diagonal = false;
Vec diagonal(n);
int consecutive_invalid = 0;
bool any_successful_step = false;
bool step_successful = true; // iteration 0
int iteration = 0;
Point trial; // the last point the line search evaluated, reused as the candidate when it is one
const auto step_rejected = [&]() {
radius = radius / decrease_factor;
decrease_factor *= 2.0;
reuse_diagonal = true;
};
const auto finish = [&](LMTermination t) {
summary.termination = t;
summary.final_cost = cur.cost;
if (t != LMTermination::Failure)
for (int i = 0; i < x.size(); i++)
x_io[i] = cur.x[i];
return summary;
};
for (;;) {
// FinalizeIterationAndCheckIfMinimizerCanContinue
summary.iterations++;
if (std::chrono::duration<double>(std::chrono::steady_clock::now() - start).count()
>= options.max_time_s)
return finish(LMTermination::NoConvergence);
if (iteration >= options.max_iterations)
return finish(LMTermination::NoConvergence);
if (step_successful && gradient_max_norm <= gradient_tolerance)
return finish(LMTermination::Convergence);
if (radius <= min_radius)
return finish(LMTermination::Convergence);
iteration++;
step_successful = false;
// ComputeTrustRegionStep: the damped normal equations of the scaled Jacobian.
if (!reuse_diagonal)
for (int i = 0; i < n; i++)
diagonal[i] = std::min(std::max(Hs(i, i), min_lm_diagonal), max_lm_diagonal);
Mat lhs = Hs;
for (int i = 0; i < n; i++) {
const double d = std::sqrt(diagonal[i] / radius);
lhs(i, i) += d * d;
}
reuse_diagonal = true;
Eigen::LLT<Mat, Eigen::Upper> llt(lhs);
bool step_valid = false;
Vec step;
double model_cost_change = 0.0;
if (llt.info() == Eigen::Success) {
step = -llt.solve(gs);
if (step.allFinite()) {
model_cost_change = -(step.dot(gs) + 0.5 * step.dot(Hs * step));
step_valid = model_cost_change > 0.0;
}
}
if (!step_valid) {
if (++consecutive_invalid >= max_consecutive_invalid_steps)
return finish(LMTermination::Failure);
step_rejected();
continue;
}
consecutive_invalid = 0;
Vec delta = step.cwiseProduct(scale);
bool have_trial = false;
if (constrained) {
// Projected Armijo line search along delta, cubic interpolation.
const double initial_gradient = cur.g.dot(delta);
const double direction_max_norm = delta.lpNorm<Eigen::Infinity>();
LMLineSample initial{0.0, cur.cost, initial_gradient, true, true};
LMLineSample previous, current;
const auto line_eval = [&](double alpha, LMLineSample &s) {
s = LMLineSample{};
s.x = alpha;
Vec moved;
plus(cur.x, Vec(alpha * delta), moved);
evaluate(moved, trial);
have_trial = true;
if (!trial.valid)
return;
s.value = trial.cost;
s.value_is_valid = true;
s.gradient = delta.dot(trial.g);
s.gradient_is_valid = std::isfinite(s.gradient);
};
line_eval(1.0, current);
bool success = true;
int ls_iterations = 0;
while (!current.value_is_valid
|| current.value > initial.value + sufficient_decrease * initial_gradient * current.x) {
++ls_iterations;
if (ls_iterations >= max_line_search_iterations) {
success = false;
break;
}
const double alpha = LMInterpolatedStepSize(initial, previous, current,
max_step_contraction * current.x,
min_step_contraction * current.x);
if (alpha * direction_max_norm < min_line_search_step) {
success = false;
break;
}
previous = current;
line_eval(alpha, current);
}
summary.line_search_steps += ls_iterations;
if (success)
delta *= current.x;
}
// ComputeCandidatePointAndEvaluateCost
Vec candidate_x;
plus(cur.x, delta, candidate_x);
Point cand;
if (have_trial && trial.x == candidate_x)
cand = std::move(trial);
else
evaluate(candidate_x, cand);
if (any_successful_step) {
const double step_norm = free_norm(cur.x - cand.x);
if (step_norm <= parameter_tolerance * (free_norm(cur.x) + parameter_tolerance))
return finish(LMTermination::Convergence);
}
if (std::fabs(cur.cost - cand.cost) <= function_tolerance * cur.cost)
return finish(LMTermination::Convergence);
const double relative_decrease = (cand.cost >= kMax)
? std::numeric_limits<double>::lowest()
: (cur.cost - cand.cost) / model_cost_change;
if (relative_decrease > min_relative_decrease) {
any_successful_step = true;
step_successful = true;
cur = std::move(cand);
take_point();
radius = radius / std::max(1.0 / 3.0, 1.0 - std::pow(2.0 * relative_decrease - 1.0, 3));
radius = std::min(max_radius, radius);
decrease_factor = 2.0;
reuse_diagonal = false;
} else {
step_rejected();
}
}
}
+127 -224
View File
@@ -7,84 +7,11 @@
#include "XtalOptimizer.h"
#include "XtalResidual.h"
#include "ceres/ceres.h"
#include "XtalRefine.h"
#include "ceres/rotation.h"
#include "Dual.h"
#include "LatticeReduction.h"
// Soft header prior on ONE beam-centre component (the spindle-parallel, gauge-weak one). Residual = w*(b - b0);
// the caller sets w so the prior behaves like a sigma-pixel restraint that competes with the (unit-weight)
// positional residuals - strong enough to pin the gauge direction, negligible in the well-constrained one.
// Soft restraint on one direction of a two-component block: g.(p - p0), weighted. Used for the beam
// centre and for the detector tilt, which are the same gauge seen twice (see the gauge block below),
// so they take the same direction g and cannot disagree about it.
struct GaugeDirectionPrior {
GaugeDirectionPrior(double gx, double gy, double p0, double weight)
: gx(gx), gy(gy), p0(p0), weight(weight) {}
template<typename T>
bool operator()(const T *const p, T *residual) const {
residual[0] = T(weight) * (T(gx) * p[0] + T(gy) * p[1] - T(p0));
return true;
}
double gx, gy, p0, weight;
};
struct XtalResidualRotationOnlyPrecomp {
XtalResidualRotationOnlyPrecomp(const Coord &recip_obs,
const CrystalLattice &latt,
double h, double k, double l)
: s_obs(recip_obs),
astar(latt.Astar()), bstar(latt.Bstar()), cstar(latt.Cstar()),
h(h), k(k), l(l) {
}
template<typename T>
bool operator()(const T *const rot_aa, T *residual) const {
const T astar_unrot[3] = {T(astar.x), T(astar.y), T(astar.z)};
const T bstar_unrot[3] = {T(bstar.x), T(bstar.y), T(bstar.z)};
const T cstar_unrot[3] = {T(cstar.x), T(cstar.y), T(cstar.z)};
T astar_rot[3], bstar_rot[3], cstar_rot[3];
const AngleAxisRotator<T> rot(rot_aa);
rot.Rotate(astar_unrot, astar_rot);
rot.Rotate(bstar_unrot, bstar_rot);
rot.Rotate(cstar_unrot, cstar_rot);
const Eigen::Matrix<T, 3, 1> s_pred(T(h) * astar_rot[0] + T(k) * bstar_rot[0] + T(l) * cstar_rot[0],
T(h) * astar_rot[1] + T(k) * bstar_rot[1] + T(l) * cstar_rot[1],
T(h) * astar_rot[2] + T(k) * bstar_rot[2] + T(l) * cstar_rot[2]
);
// Residual in reciprocal space
residual[0] = T(s_obs.x) - s_pred[0];
residual[1] = T(s_obs.y) - s_pred[1];
residual[2] = T(s_obs.z) - s_pred[2];
return true;
}
const Coord s_obs;
const Coord astar, bstar, cstar;
const double h, k, l;
};
// Regularizer: penalises ||rot_aa|| to prefer the smallest rotation that
// explains the data. Weight should be chosen in the same units as the
// reciprocal-space residuals (Å⁻¹ per radian). A value of ~0.01–0.1 is
// typically enough to break degeneracy without biasing the solution.
struct RotationNormRegularizer {
explicit RotationNormRegularizer(double weight) : weight(weight) {}
template<typename T>
bool operator()(const T *const rot_aa, T *residual) const {
residual[0] = T(weight) * rot_aa[0];
residual[1] = T(weight) * rot_aa[1];
residual[2] = T(weight) * rot_aa[2];
return true;
}
const double weight;
};
// Prior confidence weight per spot: how strong the spot is FOR ITS RESOLUTION. The frame's spots are
// ordered by resolution and cut into equal-count shells, and each intensity is divided by its shell
// median. Refinement needs the high-resolution spots (they carry the cell and distance information) and
@@ -153,12 +80,11 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
const int num_threads) {
try {
// A coplanar basis has no reciprocal cell: 1/V is infinite, every predicted reciprocal vector
// comes out NaN, and Ceres fails on the very first evaluation - after dumping the offending
// block to stderr. There is nothing for the refinement to recover here, so refuse the lattice
// before the problem is built rather than let the solver discover it. The check has to be on
// the vectors: this close to flat, float cell angles no longer carry even the SIGN of the
// metric determinant, and the triclinic branch of XtalResidual then clamps c into the a-b
// plane and divides by the zero volume that makes.
// comes out NaN, and the solver fails on the very first evaluation. There is nothing for the
// refinement to recover here, so refuse the lattice before the problem is built rather than let
// the solver discover it. The check has to be on the vectors: this close to flat, float cell
// angles no longer carry even the SIGN of the metric determinant, and the triclinic branch of
// XtalResidual then clamps c into the a-b plane and divides by the zero volume that makes.
if (data.latt.VolumeFraction() < MIN_BASIS_VOLUME_FRACTION)
return false;
@@ -168,24 +94,21 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
double beta = data.latt.GetUnitCell().beta;
// Initial guess for the parameters
double beam[2] = {data.geom.GetBeamX_pxl(), data.geom.GetBeamY_pxl()};
double distance_mm = data.geom.GetDetectorDistance_mm();
const double distance_mm = data.geom.GetDetectorDistance_mm();
double detector_rot[2] = {data.geom.GetPoniRot1_rad(), data.geom.GetPoniRot2_rad()};
// The per-frame constants of the reduced residual (see XtalFrameConstants), one entry per frame
// that contributes. Reserved up front and never grown past that, so the residual blocks' pointers
// into it stay valid, and declared before the problem so that it outlives it.
std::vector<XtalFrameConstants> frame_const;
frame_const.reserve(spots.size());
ceres::Problem problem;
double latt_vec0[3] = {0.0, 0.0, 0.0};
double latt_vec1[3] = {0.0, 0.0, 0.0};
double latt_vec2[3] = {0.0, 0.0, 0.0};
double rot_vec[3] = {1, 0, 0};
XtalRefineProblem problem;
problem.crystal_system = data.crystal_system;
problem.distance_mm = distance_mm;
double *beam = problem.beam;
beam[0] = data.geom.GetBeamX_pxl();
beam[1] = data.geom.GetBeamY_pxl();
double *detector_rot = problem.detector_rot;
detector_rot[0] = data.geom.GetPoniRot1_rad();
detector_rot[1] = data.geom.GetPoniRot2_rad();
double *latt_vec0 = problem.latt_vec0;
double *latt_vec1 = problem.latt_vec1;
double *latt_vec2 = problem.latt_vec2;
double *rot_vec = problem.rot_vec;
switch (data.crystal_system) {
case gemmi::CrystalSystem::Orthorhombic:
@@ -252,14 +175,15 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
const double sin_rot3 = std::sin(data.geom.GetPoniRot3_rad());
// Per-image rotation refinement frees only the beam and the orientation and holds the other five
// blocks constant, so the seven-block residual makes Ceres differentiate 17 parameters to use 5.
// Where that is the configuration, use the reduced residual instead - identical fit, Jet<5>
// autodiff. Any other combination (stills also free the cell, the offline refiner frees distance
// and detector angles) keeps the general form below.
// blocks constant. Where that is the configuration, the solver uses the reduced residual - the
// identical fit, with the crystal half worked out once (see XtalResidualBeamOrientation). Any
// other combination (stills also free the cell, the rotation indexer frees detector angles and
// spindle) keeps the general form.
const bool beam_and_orientation_only = data.refine_beam_center
&& !data.refine_detector_angles
&& !data.refine_rotation_axis
&& !data.refine_unit_cell;
problem.beam_and_orientation_only = beam_and_orientation_only;
// Sum of w^2 over the spots that entered - the beam prior below is scaled by it so that its
// strength relative to the data is the same weighted or not. Equals the residual block count
@@ -281,9 +205,8 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
rot_matr = data.axis->GetTransformationAngle(angle_deg);
}
if (beam_and_orientation_only)
frame_const.emplace_back(detector_rot, rot_vec, angle_rad, latt_vec1, latt_vec2,
data.crystal_system);
const int frame_index = static_cast<int>(problem.frame_angle_rad.size());
problem.frame_angle_rad.push_back(angle_rad);
// Add residuals for each point
for (size_t j = 0; j < spots[i].size(); j++) {
@@ -333,7 +256,7 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
const double weight_sq = weight.empty() ? 1.0 : weight[j] * weight[j];
effective_spots += weight_sq;
const XtalResidual residual(pt.x, pt.y,
problem.residuals.emplace_back(pt.x, pt.y,
data.geom.GetWavelength_A(),
data.geom.GetPixelSize_mm(),
cos_rot3, sin_rot3,
@@ -341,38 +264,14 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
h, k, l,
data.crystal_system,
data.geom.GetOrientation());
// Ceres has no per-residual weight; ScaledLoss(nullptr, a) multiplies the squared
// residual by the constant a, i.e. it applies a weight of sqrt(a) to the residual.
ceres::LossFunction *loss = weight.empty()
? nullptr
: new ceres::ScaledLoss(nullptr, weight_sq,
ceres::TAKE_OWNERSHIP);
if (beam_and_orientation_only)
problem.AddResidualBlock(
new ceres::AutoDiffCostFunction<XtalResidualBeamOrientation, 3, 2, 3>(
new XtalResidualBeamOrientation(residual, distance_mm, frame_const.back())),
loss,
beam,
latt_vec0
);
else
problem.AddResidualBlock(
new ceres::AutoDiffCostFunction<XtalResidualFixedDistance, 3, 2, 2, 3, 3, 3, 3>(
new XtalResidualFixedDistance(residual, distance_mm)),
loss,
beam,
detector_rot,
rot_vec,
latt_vec0,
latt_vec1,
latt_vec2
);
problem.frame.push_back(frame_index);
// A per-residual weight w enters the squared residual as w^2.
if (!weight.empty())
problem.weight_sq.push_back(weight_sq);
}
}
if (problem.NumResidualBlocks() < data.min_spots)
if (static_cast<int64_t>(problem.residuals.size()) < data.min_spots)
return false;
// The gauge direction of a single-axis rotation experiment - parallel to the spindle - written
@@ -440,9 +339,8 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
const double gauge_w = data.geom.GetPixelSize_mm() / (distance_mm * data.geom.GetWavelength_A())
* std::sqrt(effective_spots) / sigma_px;
if (!data.refine_beam_center)
problem.SetParameterBlockConstant(beam);
else if (data.axis) {
problem.beam_constant = !data.refine_beam_center;
if (data.refine_beam_center && data.axis) {
// Gauge handling (single-axis rotation): rotating the whole experiment about the spindle leaves every
// spot position unchanged, so the beam-centre component PARALLEL to the spindle is a null/gauge-weak
// direction. Refining it freely lets it wander (~+3 px) and absorb centroid systematics into a wrong
@@ -450,23 +348,19 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
// does drift - it is only LaB6-monitored to ~a few px), RESTRAIN it toward the header with a soft
// prior: the gauge direction has ~zero data sensitivity so the prior pins it near the header, while a
// real, well-supported drift can still overcome it.
problem.AddResidualBlock(
new ceres::AutoDiffCostFunction<GaugeDirectionPrior, 1, 2>(
new GaugeDirectionPrior(gauge_beam_x, gauge_beam_y,
gauge_beam_x * beam[0] + gauge_beam_y * beam[1], gauge_w)),
nullptr, beam);
problem.priors.push_back({XtalRefinePrior::Block::Beam, gauge_beam_x, gauge_beam_y,
gauge_beam_x * beam[0] + gauge_beam_y * beam[1], gauge_w});
}
// Distance, detector angles, rotation axis and cell are parameter blocks only in the general
// seven-block residual; the reduced one bakes them in, so there is nothing left to configure.
if (!beam_and_orientation_only) {
if (!data.refine_detector_angles) {
problem.SetParameterBlockConstant(detector_rot);
} else {
problem.detector_rot_constant = !data.refine_detector_angles;
if (data.refine_detector_angles) {
const double rot_range = 3.0 / 180.0 * PI;
for (int i = 0; i < 2; ++i) {
problem.SetParameterLowerBound(detector_rot, i, detector_rot[i] - rot_range);
problem.SetParameterUpperBound(detector_rot, i, detector_rot[i] + rot_range);
problem.detector_rot_lower[i] = detector_rot[i] - rot_range;
problem.detector_rot_upper[i] = detector_rot[i] + rot_range;
}
// The same gauge as the beam prior above, described a second time: the tilt moves the
// direct beam exactly as the beam centre does, at D/pixel px per radian, so leaving
@@ -497,20 +391,15 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
for (int i = 0; i < 2; ++i) {
if (budget[i] <= 0.0)
continue;
problem.AddResidualBlock(
new ceres::AutoDiffCostFunction<GaugeDirectionPrior, 1, 2>(
new GaugeDirectionPrior(dirs[i][0], dirs[i][1],
dirs[i][0] * detector_rot[0]
+ dirs[i][1] * detector_rot[1],
gauge_w * (sigma_px / budget[i]) * lever)),
nullptr, detector_rot);
problem.priors.push_back({XtalRefinePrior::Block::DetectorRot, dirs[i][0], dirs[i][1],
dirs[i][0] * detector_rot[0] + dirs[i][1] * detector_rot[1],
gauge_w * (sigma_px / budget[i]) * lever});
}
}
}
if (!data.refine_rotation_axis) {
problem.SetParameterBlockConstant(rot_vec);
} else {
problem.rot_vec_constant = !data.refine_rotation_axis;
if (data.refine_rotation_axis) {
// Only the DIRECTION of the goniometer axis is a parameter. The residual applies
// angle_rad * |rot_vec|, so a free three-vector also fits a rotation SCALE - which
// GoniometerAxis::Axis() then normalises away, leaving the candidate scored by
@@ -520,63 +409,46 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
// data it recovers 54 % of a known scale error, repeated first passes on one dataset
// disagree with each other in SIGN, and on the one dataset with a real 1.3 % stage
// fault it comes out negative. The rotation scale is measured properly, once, with
// four gates and a jackknife, in PostRefine.
problem.SetManifold(rot_vec, new ceres::SphereManifold<3>);
// four gates and a jackknife, in PostRefine. Refined on the sphere (see SolveXtalRefine).
}
if (!data.refine_unit_cell) {
problem.SetParameterBlockConstant(latt_vec1);
problem.SetParameterBlockConstant(latt_vec2);
} else {
problem.latt_vec1_constant = !data.refine_unit_cell;
problem.latt_vec2_constant = !data.refine_unit_cell;
if (data.refine_unit_cell) {
// Parameter bounds
// Lengths
for (int i = 0; i < 3; ++i) {
problem.SetParameterLowerBound(latt_vec1, i, data.min_length_A);
problem.SetParameterUpperBound(latt_vec1, i, data.max_length_A);
problem.latt_vec1_lower[i] = data.min_length_A;
problem.latt_vec1_upper[i] = data.max_length_A;
}
if (data.crystal_system == gemmi::CrystalSystem::Monoclinic) {
const double beta_lo = std::max(1e-6, PI * (data.min_angle_deg / 180.0));
const double beta_hi = std::min(PI - 1e-6, PI * (data.max_angle_deg / 180.0));
problem.SetParameterLowerBound(latt_vec2, 0, beta_lo);
problem.SetParameterUpperBound(latt_vec2, 0, beta_hi);
problem.latt_vec2_constant = false;
problem.latt_vec2_lower[0] = std::max(1e-6, PI * (data.min_angle_deg / 180.0));
problem.latt_vec2_upper[0] = std::min(PI - 1e-6, PI * (data.max_angle_deg / 180.0));
} else if (data.crystal_system == gemmi::CrystalSystem::Triclinic) {
// α, β, γ bounds (radians)
const double alo = PI * (data.min_angle_deg / 180.0);
const double ahi = PI * (data.max_angle_deg / 180.0);
for (int i = 0; i < 3; ++i) {
problem.SetParameterLowerBound(latt_vec2, i, alo);
problem.SetParameterUpperBound(latt_vec2, i, ahi);
problem.latt_vec2_lower[i] = alo;
problem.latt_vec2_upper[i] = ahi;
}
} else {
// Orthorhombic / Tetragonal / Cubic / Hexagonal:
// latt_vec2 has no meaning for these systems — always freeze it.
problem.SetParameterBlockConstant(latt_vec2);
problem.latt_vec2_constant = true;
}
}
}
// Configure solver
ceres::Solver::Options options;
// Normal equations, not QR. The problem is very tall and thin - thousands of spots against at
// most 17 parameters - and that is the shape DENSE_QR handles worst: it copies the Jacobian out
// of Ceres' row-major storage into a column-major buffer on every solve, and Eigen's blocked
// Householder then degenerates to the unblocked path because its block size is min(48, columns).
// Accumulating J^T J reads the Jacobian once instead. Both solve the same damped system, so the
// step is the same to round-off; the column scaling Ceres applies by default and the LM diagonal
// keep the squared condition number in hand.
options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY;
options.minimizer_progress_to_stdout = false;
// Stopping rule: a bound on iterations is reproducible, a bound on wall-clock time is not (see
// XtalOptimizerData::max_iterations).
if (data.max_iterations > 0)
options.max_num_iterations = data.max_iterations;
problem.options.max_iterations = data.max_iterations;
else
options.max_solver_time_in_seconds = data.max_time;
options.logging_type = ceres::LoggingType::SILENT;
options.num_threads = num_threads; // usually 1 (called from many threads); caller may raise it
ceres::Solver::Summary summary;
// Run optimization
ceres::Solve(options, &problem, &summary);
problem.options.max_time_s = data.max_time;
const LMSummary summary = SolveXtalRefine(problem, num_threads);
// Only a genuine numerical failure is rejected here: a solve that ran out of iterations or
// out of time but still descended counts as usable, which is what the real-time caller
@@ -653,7 +525,7 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data,
return false;
// Parameter: angle-axis for the extra rotation. Identity == {0,0,0}.
double rot_aa[3] = {0.0, 0.0, 0.0};
std::vector<double> rot_aa = {0.0, 0.0, 0.0};
// Spot selection by current indexing (same approach as XtalOptimizerInternal)
const Coord a0 = data.latt.Vec0();
@@ -662,7 +534,12 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data,
const float tol_sq = tolerance * tolerance;
ceres::Problem problem;
// Each selected spot: its observed reciprocal vector and the indices it is fitted to.
struct Observation {
Coord s_obs;
double h, k, l;
};
std::vector<Observation> observations;
for (const auto &pt : spots) {
if (!data.index_ice_rings && pt.ice_ring)
@@ -697,42 +574,68 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data,
if (data.axis.has_value())
s_obs = data.axis->GetTransformationAngle(pt.phi) * s_obs;
auto *cost =
new ceres::AutoDiffCostFunction<XtalResidualRotationOnlyPrecomp, 3, 3>(
new XtalResidualRotationOnlyPrecomp(s_obs, data.latt, h, k, l)
);
problem.AddResidualBlock(cost, nullptr, rot_aa);
observations.push_back({s_obs, h, k, l});
}
if (problem.NumResidualBlocks() < data.min_spots)
if (static_cast<int64_t>(observations.size()) < data.min_spots)
return false;
// Regularization: prefer the smallest rotation correction that fits the
// data. This is essential when spots are nearly coplanar in reciprocal
// space (e.g. still images), where the rotation component perpendicular
// to the scattering plane is otherwise underdetermined.
// The weight is in Å⁻¹ rad⁻¹; tune relative to your typical residual.
{
const double reg_weight = 0.05; // e.g. 0.05
problem.AddResidualBlock(
new ceres::AutoDiffCostFunction<RotationNormRegularizer, 3, 3>(
new RotationNormRegularizer(reg_weight)),
nullptr, rot_aa);
}
// Residual: s_obs - R(rot_aa) (h a* + k b* + l c*), the reciprocal basis rotated once per
// evaluation and shared by every spot.
//
// Regularization: prefer the smallest rotation correction that fits the data, w * rot_aa. This is
// essential when spots are nearly coplanar in reciprocal space (e.g. still images), where the
// rotation component perpendicular to the scattering plane is otherwise underdetermined. The
// weight is in A^-1 rad^-1, relative to the typical residual.
const double reg_weight = 0.05;
const Coord astar = data.latt.Astar(), bstar = data.latt.Bstar(), cstar = data.latt.Cstar();
const auto evaluate = [&](const double *aa, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) {
using D = Dual<3>;
const D aa_d[3] = {D::Variable(aa[0], 0), D::Variable(aa[1], 1), D::Variable(aa[2], 2)};
const AngleAxisRotator<D> rot(aa_d);
const double astar_unrot[3] = {astar.x, astar.y, astar.z};
const double bstar_unrot[3] = {bstar.x, bstar.y, bstar.z};
const double cstar_unrot[3] = {cstar.x, cstar.y, cstar.z};
D astar_rot[3], bstar_rot[3], cstar_rot[3];
rot.Rotate(astar_unrot, astar_rot);
rot.Rotate(bstar_unrot, bstar_rot);
rot.Rotate(cstar_unrot, cstar_rot);
ceres::Solver::Options options;
options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY; // tall and thin, as above
options.minimizer_progress_to_stdout = false;
cost = 0.0;
const auto add = [&](double r, const double *J) {
cost += 0.5 * r * r;
if (!g)
return;
for (int i = 0; i < 3; i++) {
(*g)[i] += J[i] * r;
for (int j = 0; j < 3; j++)
(*H)(i, j) += J[i] * J[j];
}
};
for (const auto &o: observations) {
const double s_obs[3] = {o.s_obs.x, o.s_obs.y, o.s_obs.z};
for (int c = 0; c < 3; c++) {
const D pred = o.h * astar_rot[c] + o.k * bstar_rot[c] + o.l * cstar_rot[c];
const double J[3] = {-pred.v[0], -pred.v[1], -pred.v[2]};
add(s_obs[c] - pred.a, J);
}
}
for (int c = 0; c < 3; c++) {
double J[3] = {0.0, 0.0, 0.0};
J[c] = reg_weight;
add(reg_weight * aa[c], J);
}
return std::isfinite(cost) && (!g || (g->allFinite() && H->allFinite()));
};
std::vector<LMBlock> blocks(1);
blocks[0].size = 3;
LMOptions options;
if (data.max_iterations > 0)
options.max_num_iterations = data.max_iterations;
options.max_iterations = data.max_iterations;
else
options.max_solver_time_in_seconds = data.max_time;
options.logging_type = ceres::LoggingType::SILENT;
options.num_threads = 1;
ceres::Solver::Summary summary;
ceres::Solve(options, &problem, &summary);
options.max_time_s = data.max_time;
const LMSummary summary = SolveLM(rot_aa, blocks, options, evaluate);
if (!summary.IsSolutionUsable())
return false;
@@ -747,7 +650,7 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data,
// rotating the reciprocal vectors (a*, b*, c*) by the same R. No
// transpose or inversion of R is needed here.
double R_raw[9];
ceres::AngleAxisToRotationMatrix(rot_aa, R_raw); // row-major 3x3
ceres::AngleAxisToRotationMatrix(rot_aa.data(), R_raw); // row-major 3x3
Eigen::Matrix3d R;
R << R_raw[0], R_raw[3], R_raw[6],
@@ -62,7 +62,7 @@ struct XtalOptimizerData {
std::optional<Coord> angle_axis;
};
// num_threads sets the Ceres solver thread count for the internal least-squares refine. It defaults
// num_threads sets the thread count of the internal least-squares refine (the answer does not depend on it). It defaults
// to 1 because XtalOptimizer is usually called from many threads at once; raise it only when a caller
// runs a small number of refinements concurrently and wants each to use several cores.
bool XtalOptimizer(XtalOptimizerData &data, std::span<const std::vector<SpotToSave>> spots,
@@ -0,0 +1,334 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "XtalRefine.h"
#include <array>
#include "Dual.h"
#include "../../common/ParallelFor.h"
namespace {
// Ambient layout of the parameter vector and the order of the blocks in it.
constexpr int OFF_BEAM = 0, OFF_ROT = 2, OFF_AXIS = 4, OFF_P0 = 7, OFF_LEN = 10, OFF_ANG = 13, N_AMBIENT = 16;
// Derivative lanes. The observed half of a residual depends on beam, detector angles and spindle
// (OBS lanes), the predicted half on orientation and cell (PRED lanes); each half is carried on a
// dual number of its own width and the two are put side by side only in the sums.
constexpr int OBS = 6, PRED = 9, LANES = OBS + PRED;
using DO = Dual<OBS>;
using DP = Dual<PRED>;
// The reduced problem: beam (2) and orientation (3) only.
constexpr int R_OBS = 2, R_PRED = 3, R_LANES = R_OBS + R_PRED;
// Sums over one block of residuals, in lane coordinates.
template<int L>
struct Sums {
double cost = 0.0;
double g[L] = {};
double H[L][L] = {}; // upper triangle
void Add(double r, const double *J, double w2) {
cost += 0.5 * w2 * r * r;
for (int i = 0; i < L; i++) {
const double wj = w2 * J[i];
g[i] += wj * r;
for (int j = i; j < L; j++)
H[i][j] += wj * J[j];
}
}
void Add(const Sums &o) {
cost += o.cost;
for (int i = 0; i < L; i++) {
g[i] += o.g[i];
for (int j = i; j < L; j++)
H[i][j] += o.H[i][j];
}
}
};
std::vector<LMBlock> MakeBlocks(const XtalRefineProblem &p) {
std::vector<LMBlock> blocks(6);
blocks[0].offset = OFF_BEAM;
blocks[0].size = 2;
blocks[0].constant = p.beam_constant;
blocks[1].offset = OFF_ROT;
blocks[1].size = 2;
blocks[1].constant = p.beam_and_orientation_only || p.detector_rot_constant;
for (int j = 0; j < 2; j++) {
blocks[1].lower[j] = p.detector_rot_lower[j];
blocks[1].upper[j] = p.detector_rot_upper[j];
}
blocks[2].offset = OFF_AXIS;
blocks[2].size = 3;
blocks[2].constant = p.beam_and_orientation_only || p.rot_vec_constant;
blocks[2].sphere = true;
blocks[3].offset = OFF_P0;
blocks[3].size = 3;
blocks[4].offset = OFF_LEN;
blocks[4].size = 3;
blocks[4].constant = p.beam_and_orientation_only || p.latt_vec1_constant;
blocks[5].offset = OFF_ANG;
blocks[5].size = 3;
blocks[5].constant = p.beam_and_orientation_only || p.latt_vec2_constant;
for (int j = 0; j < 3; j++) {
blocks[4].lower[j] = p.latt_vec1_lower[j];
blocks[4].upper[j] = p.latt_vec1_upper[j];
blocks[5].lower[j] = p.latt_vec2_lower[j];
blocks[5].upper[j] = p.latt_vec2_upper[j];
}
return blocks;
}
// Where each lane lands among the tangent coordinates of the free blocks; -1 for a held one.
template<size_t L>
std::array<int, L> LaneToTangent(const std::vector<LMBlock> &blocks, const std::array<int, L> &lane_block,
const std::array<int, L> &lane_index) {
std::array<int, L> map{};
std::array<int, 6> tangent_offset{};
int t = 0;
for (size_t b = 0; b < blocks.size(); b++) {
tangent_offset[b] = t;
t += blocks[b].TangentSize();
}
for (size_t l = 0; l < L; l++)
map[l] = blocks[lane_block[l]].constant ? -1 : tangent_offset[lane_block[l]] + lane_index[l];
return map;
}
template<int L>
void ToTangent(const Sums<L> &s, const std::array<int, L> &map, Eigen::VectorXd &g, Eigen::MatrixXd &H) {
for (int i = 0; i < L; i++) {
if (map[i] < 0)
continue;
g[map[i]] += s.g[i];
for (int j = i; j < L; j++) {
if (map[j] < 0)
continue;
H(map[i], map[j]) += s.H[i][j];
if (map[i] != map[j])
H(map[j], map[i]) += s.H[i][j];
}
}
}
void AddPriors(const XtalRefineProblem &p, const double *x, double &cost, Eigen::VectorXd *g,
Eigen::MatrixXd *H, int beam_tangent, int rot_tangent) {
for (const auto &prior: p.priors) {
const bool on_beam = prior.block == XtalRefinePrior::Block::Beam;
const double *v = x + (on_beam ? OFF_BEAM : OFF_ROT);
const double r = prior.weight * (prior.gx * v[0] + prior.gy * v[1] - prior.p0);
cost += 0.5 * r * r;
const int t = on_beam ? beam_tangent : rot_tangent;
if (!g || t < 0)
continue;
const double J[2] = {prior.weight * prior.gx, prior.weight * prior.gy};
for (int i = 0; i < 2; i++) {
(*g)[t + i] += J[i] * r;
for (int j = 0; j < 2; j++)
(*H)(t + i, t + j) += J[i] * J[j];
}
}
}
template<class D>
D Seed(double value, int lane, bool free) {
return free ? D::Variable(value, lane) : D(value);
}
// Residual blocks of at least this many residuals; the cut depends on the count alone.
constexpr int MIN_RESIDUALS_PER_BLOCK = 256;
template<int L, class Fn>
Sums<L> SumResiduals(const XtalRefineProblem &p, int num_threads, Fn &&residual) {
const int n = static_cast<int>(p.residuals.size());
std::vector<Sums<L>> partial(ReductionBlocks(n, MIN_RESIDUALS_PER_BLOCK));
ParallelBlocks(n, std::max(1, num_threads), [&](int b, int lo, int hi) {
for (int i = lo; i < hi; i++)
residual(i, partial[b]);
}, MIN_RESIDUALS_PER_BLOCK);
Sums<L> total;
for (const auto &s: partial)
total.Add(s);
return total;
}
double WeightSq(const XtalRefineProblem &p, int i) {
return p.weight_sq.empty() ? 1.0 : p.weight_sq[i];
}
bool AllFinite(double cost, const Eigen::VectorXd *g, const Eigen::MatrixXd *H) {
return std::isfinite(cost) && (!g || (g->allFinite() && H->allFinite()));
}
// Seven-block residual (XtalResidualFixedDistance), with every block that depends on parameters
// alone - detector-angle sines and cosines, the spindle back-rotation of each frame, the reciprocal
// basis of the cell, the orientation's rotation - worked out once per evaluation.
bool EvaluateGeneral(const XtalRefineProblem &p, const std::array<int, LANES> &map, int num_threads,
const double *x, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) {
const bool beam_free = map[0] >= 0, rot_free = map[2] >= 0, axis_free = map[4] >= 0;
const bool p0_free = map[OBS] >= 0, len_free = map[OBS + 3] >= 0, ang_free = map[OBS + 6] >= 0;
const DO beam[2] = {Seed<DO>(x[OFF_BEAM], 0, beam_free),
Seed<DO>(x[OFF_BEAM + 1], 1, beam_free)};
const DO rot1 = Seed<DO>(x[OFF_ROT], 2, rot_free);
const DO rot2 = Seed<DO>(x[OFF_ROT + 1], 3, rot_free);
const DO c1 = cos(rot1), s1 = sin(rot1), c2 = cos(rot2), s2 = sin(rot2);
DO axis[3] = {x[OFF_AXIS], x[OFF_AXIS + 1], x[OFF_AXIS + 2]};
if (axis_free) {
double J[3][2];
SpherePlusJacobian3(x + OFF_AXIS, J);
for (int k = 0; k < 3; k++) {
axis[k].v[4] = J[k][0];
axis[k].v[5] = J[k][1];
}
}
std::vector<AngleAxisRotator<DO>> rot_back;
rot_back.reserve(p.frame_angle_rad.size());
for (const double angle: p.frame_angle_rad) {
const DO aa_back[3] = {angle * axis[0], angle * axis[1], angle * axis[2]};
rot_back.emplace_back(aa_back);
}
DP p0[3], len[3], ang[3];
for (int k = 0; k < 3; k++) {
p0[k] = Seed<DP>(x[OFF_P0 + k], k, p0_free);
len[k] = Seed<DP>(x[OFF_LEN + k], 3 + k, len_free);
ang[k] = Seed<DP>(x[OFF_ANG + k], 6 + k, ang_free);
}
Eigen::Matrix<DP, 3, 1> bxc, cxa, axb;
DP invV;
XtalResidual::ReciprocalBasis(len, ang, p.crystal_system, bxc, cxa, axb, invV);
const AngleAxisRotator<DP> rot_p0(p0);
const Sums<LANES> s = SumResiduals<LANES>(p, num_threads, [&](int i, Sums<LANES> &acc) {
const XtalResidual &res = p.residuals[i];
DO obs[3];
res.ObservedRecipCore(beam, p.distance_mm, c1, s1, c2, s2, rot_back[p.frame[i]], obs);
DP unrot[3], pred[3];
res.CombineRecipUnrot(bxc, cxa, axb, invV, unrot);
rot_p0.Rotate(unrot, pred);
const double w2 = WeightSq(p, i);
for (int k = 0; k < 3; k++) {
double J[LANES];
for (int l = 0; l < OBS; l++)
J[l] = obs[k].v[l];
for (int l = 0; l < PRED; l++)
J[OBS + l] = -pred[k].v[l];
acc.Add(obs[k].a - pred[k].a, J, w2);
}
});
cost = s.cost;
if (g)
ToTangent<LANES>(s, map, *g, *H);
AddPriors(p, x, cost, g, H, map[0], map[2]);
return AllFinite(cost, g, H);
}
// The reduced residual (XtalResidualBeamOrientation): detector, spindle and cell held, so the
// back-rotation of each frame and the unrotated prediction of each residual are constants.
bool EvaluateBeamOrientation(const XtalRefineProblem &p, const std::array<int, R_LANES> &map,
const std::vector<AngleAxisRotator<double>> &rot_back,
const std::vector<std::array<double, 3>> &unrot, double c1, double s1,
double c2, double s2, int num_threads,
const double *x, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) {
using DB = Dual<R_OBS>;
using DR = Dual<R_PRED>;
const bool beam_free = map[0] >= 0;
const DB beam[2] = {beam_free ? DB::Variable(x[OFF_BEAM], 0) : DB(x[OFF_BEAM]),
beam_free ? DB::Variable(x[OFF_BEAM + 1], 1) : DB(x[OFF_BEAM + 1])};
const DR p0[3] = {DR::Variable(x[OFF_P0], 0), DR::Variable(x[OFF_P0 + 1], 1),
DR::Variable(x[OFF_P0 + 2], 2)};
const AngleAxisRotator<DR> rot_p0(p0);
const Sums<R_LANES> s = SumResiduals<R_LANES>(p, num_threads, [&](int i, Sums<R_LANES> &acc) {
const XtalResidual &res = p.residuals[i];
DB obs[3];
res.ObservedRecipCore(beam, p.distance_mm, c1, s1, c2, s2, rot_back[p.frame[i]], obs);
DR pred[3];
rot_p0.Rotate(unrot[i].data(), pred);
const double w2 = WeightSq(p, i);
for (int k = 0; k < 3; k++) {
double J[R_LANES];
for (int l = 0; l < R_OBS; l++)
J[l] = obs[k].v[l];
for (int l = 0; l < R_PRED; l++)
J[R_OBS + l] = -pred[k].v[l];
acc.Add(obs[k].a - pred[k].a, J, w2);
}
});
cost = s.cost;
if (g)
ToTangent<R_LANES>(s, map, *g, *H);
AddPriors(p, x, cost, g, H, map[0], -1);
return AllFinite(cost, g, H);
}
}
LMSummary SolveXtalRefine(XtalRefineProblem &p, int num_threads) {
const std::vector<LMBlock> blocks = MakeBlocks(p);
std::vector<double> x(N_AMBIENT);
const auto put = [&](int off, const double *v, int n) { for (int i = 0; i < n; i++) x[off + i] = v[i]; };
put(OFF_BEAM, p.beam, 2);
put(OFF_ROT, p.detector_rot, 2);
put(OFF_AXIS, p.rot_vec, 3);
put(OFF_P0, p.latt_vec0, 3);
put(OFF_LEN, p.latt_vec1, 3);
put(OFF_ANG, p.latt_vec2, 3);
LMSummary summary;
if (p.beam_and_orientation_only) {
const std::array<int, R_LANES> lane_block = {0, 0, 3, 3, 3};
const std::array<int, R_LANES> lane_index = {0, 1, 0, 1, 2};
const auto map = LaneToTangent<R_LANES>(blocks, lane_block, lane_index);
std::vector<AngleAxisRotator<double>> rot_back;
rot_back.reserve(p.frame_angle_rad.size());
for (const double angle: p.frame_angle_rad)
rot_back.push_back(XtalFrameConstants::BackRotator(angle, p.rot_vec));
Eigen::Matrix<double, 3, 1> bxc, cxa, axb;
double invV;
XtalResidual::ReciprocalBasis(p.latt_vec1, p.latt_vec2, p.crystal_system, bxc, cxa, axb, invV);
std::vector<std::array<double, 3>> unrot(p.residuals.size());
for (size_t i = 0; i < p.residuals.size(); i++)
p.residuals[i].CombineRecipUnrot(bxc, cxa, axb, invV, unrot[i].data());
const double c1 = std::cos(p.detector_rot[0]), s1 = std::sin(p.detector_rot[0]);
const double c2 = std::cos(p.detector_rot[1]), s2 = std::sin(p.detector_rot[1]);
summary = SolveLM(x, blocks, p.options, [&](const double *at, double &cost, Eigen::VectorXd *g,
Eigen::MatrixXd *H) {
return EvaluateBeamOrientation(p, map, rot_back, unrot, c1, s1, c2, s2, num_threads, at, cost, g, H);
});
} else {
const std::array<int, LANES> lane_block = {0, 0, 1, 1, 2, 2, 3, 3, 3, 4, 4, 4, 5, 5, 5};
const std::array<int, LANES> lane_index = {0, 1, 0, 1, 0, 1, 0, 1, 2, 0, 1, 2, 0, 1, 2};
const auto map = LaneToTangent<LANES>(blocks, lane_block, lane_index);
summary = SolveLM(x, blocks, p.options, [&](const double *at, double &cost, Eigen::VectorXd *g,
Eigen::MatrixXd *H) {
return EvaluateGeneral(p, map, num_threads, at, cost, g, H);
});
}
if (summary.IsSolutionUsable()) {
const auto get = [&](int off, double *v, int n) { for (int i = 0; i < n; i++) v[i] = x[off + i]; };
get(OFF_BEAM, p.beam, 2);
get(OFF_ROT, p.detector_rot, 2);
get(OFF_AXIS, p.rot_vec, 3);
get(OFF_P0, p.latt_vec0, 3);
get(OFF_LEN, p.latt_vec1, 3);
get(OFF_ANG, p.latt_vec2, 3);
}
return summary;
}
@@ -0,0 +1,63 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
#include <limits>
#include <vector>
#include "XtalResidual.h"
#include "LMSolver.h"
// A soft restraint w * (g . p - p0) on one direction g of a two-component block - the beam centre or
// the detector tilt, which are the same gauge seen twice and so take the same direction (see
// XtalOptimizer).
struct XtalRefinePrior {
enum class Block { Beam, DetectorRot } block = Block::Beam;
double gx = 0.0, gy = 0.0, p0 = 0.0, weight = 0.0;
};
// The least-squares problem XtalOptimizer solves, as data: the residuals with the frame each belongs
// to, the parameter blocks with what is held and what is bounded, and the priors. The parameter arrays
// are in/out. Parameter blocks: beam(2), detector_rot(2), rot_vec(3, the spindle, refined on the sphere),
// latt_vec0(3, orientation angle-axis), latt_vec1(3, cell lengths), latt_vec2(3, cell angles) - see
// XtalResidual. The distance is a constant of the problem.
struct XtalRefineProblem {
static constexpr double kNoBound = std::numeric_limits<double>::max();
gemmi::CrystalSystem crystal_system = gemmi::CrystalSystem::Triclinic;
// Only beam and orientation free, everything else held: the reduced residual (see
// XtalResidualBeamOrientation), whose crystal half is a constant of the problem.
bool beam_and_orientation_only = false;
double distance_mm = 0.0;
std::vector<XtalResidual> residuals;
std::vector<int> frame; // per residual, an index into frame_angle_rad
std::vector<double> frame_angle_rad;
std::vector<double> weight_sq; // per residual; empty = unweighted
double beam[2] = {0, 0};
double detector_rot[2] = {0, 0};
double rot_vec[3] = {1, 0, 0};
double latt_vec0[3] = {0, 0, 0};
double latt_vec1[3] = {0, 0, 0};
double latt_vec2[3] = {0, 0, 0};
bool beam_constant = false;
bool detector_rot_constant = true;
bool rot_vec_constant = true;
bool latt_vec1_constant = true;
bool latt_vec2_constant = true;
double detector_rot_lower[2] = {-kNoBound, -kNoBound}, detector_rot_upper[2] = {kNoBound, kNoBound};
double latt_vec1_lower[3] = {-kNoBound, -kNoBound, -kNoBound}, latt_vec1_upper[3] = {kNoBound, kNoBound, kNoBound};
double latt_vec2_lower[3] = {-kNoBound, -kNoBound, -kNoBound}, latt_vec2_upper[3] = {kNoBound, kNoBound, kNoBound};
std::vector<XtalRefinePrior> priors;
LMOptions options;
};
// Solves the problem in place. The residual sums are cut into blocks that depend on the problem
// alone, so the answer is the same at any thread count.
LMSummary SolveXtalRefine(XtalRefineProblem &problem, int num_threads);
+24 -17
View File
@@ -7,8 +7,6 @@
#include <Eigen/Dense>
#include "ceres/ceres.h"
#include "ceres/rotation.h"
#include "gemmi/symmetry.hpp"
#include "../../common/JFJochException.h"
@@ -110,6 +108,9 @@ inline void EffectiveCellFromParams(gemmi::CrystalSystem symmetry, const double
}
}
// The scalar type is a template parameter throughout: a double, a ceres::Jet or a Dual (Dual.h). The
// mathematical functions are called unqualified, so each type's own overload is found by lookup.
//
// Detector -> reciprocal geometry residual, shared by the per-image XtalOptimizer (one lattice, one
// frame) and the offline GeometryRefiner (shared beam/distance/cell blocks, one orientation block per
// frame). Parameter blocks: beam(2), distance_mm(1), detector_rot(2 = rot1,rot2), rotation_axis(3),
@@ -164,16 +165,18 @@ struct XtalResidual {
// detector_rot[0] = rot1, detector_rot[1] = rot2 are refined; rot3 is fixed
// (e.g. from a PONI import) and baked in here as a constant so that a non-zero
// rot3 is not silently dropped during refinement.
using std::cos;
using std::sin;
const C rot1 = detector_rot[0];
const C rot2 = detector_rot[1];
// Ry(+rot1): rotation around Y-axis
const C c1 = ceres::cos(rot1);
const C s1 = ceres::sin(rot1);
const C c1 = cos(rot1);
const C s1 = sin(rot1);
// Rx(-rot2): rotation around X-axis with inverted sign (PyFAI left-handed)
const C c2 = ceres::cos(rot2);
const C s2 = ceres::sin(rot2);
const C c2 = cos(rot2);
const C s2 = sin(rot2);
// Apply the goniometer "back-to-start" rotation of this frame's angle.
const C aa_back[3] = {
@@ -227,7 +230,8 @@ struct XtalResidual {
const T z = t2_z;
// convert to recip space
const T lab_norm = ceres::sqrt(x * x + y * y + z * z);
using std::sqrt;
const T lab_norm = sqrt(x * x + y * y + z * z);
const T inv_norm = T(1) / lab_norm;
T recip_raw[3];
@@ -258,6 +262,9 @@ struct XtalResidual {
static void ReciprocalBasis(const C *const p1, const C *const p2, gemmi::CrystalSystem symmetry,
Eigen::Matrix<C, 3, 1> &bxc, Eigen::Matrix<C, 3, 1> &cxa,
Eigen::Matrix<C, 3, 1> &axb, C &invV) {
using std::cos;
using std::sin;
using std::sqrt;
// Build unit cell lengths and B (convention: columns are a, b, c prior to global rotation)
Eigen::Matrix<C, 3, 1> e_uc_len = Eigen::Matrix<C, 3, 1>::Zero();
Eigen::Matrix<C, 3, 3> B = Eigen::Matrix<C, 3, 3>::Identity();
@@ -275,14 +282,14 @@ struct XtalResidual {
} else if (symmetry == gemmi::CrystalSystem::Monoclinic) {
// Unique axis b: alpha = gamma = 90°, beta free (angle between a and c)
e_uc_len << p1[0], p1[1], p1[2];
B(0, 2) = ceres::cos(p2[0]);
B(2, 2) = ceres::sin(p2[0]);
B(0, 2) = cos(p2[0]);
B(2, 2) = sin(p2[0]);
} else {
// Triclinic: p1 = (a,b,c), p2 = (alpha, beta, gamma) in radians
const C ca = ceres::cos(p2[0]);
const C cb = ceres::cos(p2[1]);
const C cg = ceres::cos(p2[2]);
const C sg = ceres::sin(p2[2]);
const C ca = cos(p2[0]);
const C cb = cos(p2[1]);
const C cg = cos(p2[2]);
const C sg = sin(p2[2]);
e_uc_len << p1[0], p1[1], p1[2];
@@ -297,7 +304,7 @@ struct XtalResidual {
const C cx = cb;
const C cy = (ca - cb * cg) / sg;
const C v = C(1) - cx * cx - cy * cy;
const C cz = (v >= C(0)) ? ceres::sqrt(v) : C(0);
const C cz = (v >= C(0)) ? sqrt(v) : C(0);
B(0, 2) = cx;
B(1, 2) = cy;
@@ -399,12 +406,12 @@ struct XtalResidual {
// reciprocal basis of the fixed cell. They are constants of the whole problem there - one frame per
// image, one cell - and deriving them inside each residual costs six trigonometric calls, a hypot and a
// division per evaluation for numbers that never change. Built by the caller, which has to keep it alive
// as long as the ceres::Problem that points at it.
// as long as the residuals that point at it.
struct XtalFrameConstants {
XtalFrameConstants(const double *detector_rot, const double *rotation_axis, double angle_rad,
const double *uc_len, const double *uc_angle, gemmi::CrystalSystem symmetry)
: c1(ceres::cos(detector_rot[0])), s1(ceres::sin(detector_rot[0])),
c2(ceres::cos(detector_rot[1])), s2(ceres::sin(detector_rot[1])),
: c1(cos(detector_rot[0])), s1(sin(detector_rot[0])),
c2(cos(detector_rot[1])), s2(sin(detector_rot[1])),
rot_back(BackRotator(angle_rad, rotation_axis)) {
XtalResidual::ReciprocalBasis(uc_len, uc_angle, symmetry, bxc, cxa, axb, invV);
}
+108 -42
View File
@@ -3,6 +3,7 @@
#include "../../common/ParallelFor.h"
#include "RotationScaleMerge.h"
#include "RotationScaleMergeGPU.h" // SurfaceTerm, the surface fit's term on both paths
#include <algorithm>
#include <atomic>
@@ -3334,7 +3335,7 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
// traffic of the data it touched, over hundreds of megabytes. Copying once turns them into
// sequential walks of a compact array. The copy keeps fulls order, so each sum below is formed
// from exactly the same terms in exactly the same order.
struct Term { float I, sigma, corr, d; int32_t cell, group; };
using Term = RotationScaleMergeGPU::SurfaceTerm; // float I, sigma, corr, d; int32_t cell, group
std::vector<Term> term;
std::vector<uint8_t> term_parity; // frame parity, read only by the group-ordered copy below
// Which terms each pass walks, as positions in `term`. The cross-validated halves then cost half a
@@ -3437,22 +3438,27 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
std::vector<double> shw_shell(nshell, 0.0), shw_cell(nshell > 0 ? ncell : 0, 0.0);
std::vector<int32_t> group_shell(n_groups, 0); // the shell each ASU group sits in, for the gate
if (nshell > 0) {
std::vector<float> s2;
s2.reserve(term.size());
for (const Term &t : term)
s2.push_back(t.d > 0.0f ? 1.0f / (t.d * t.d) : 0.0f);
const int n_term = static_cast<int>(term.size());
std::vector<float> s2(n_term);
ParallelChunks(n_term, nt, [&](int lo, int hi) {
for (int k = lo; k < hi; ++k)
s2[k] = term[k].d > 0.0f ? 1.0f / (term[k].d * term[k].d) : 0.0f;
});
// The edges are order statistics of s2, so they are the same values however the sort orders
// equal elements among themselves.
std::vector<float> sorted = s2;
ParallelSort(sorted.begin(), sorted.end(), nt, std::less<float>());
std::vector<float> edge(nshell - 1);
size_t prev = 0;
for (int i = 1; i < nshell; ++i) {
const size_t pos = s2.size() * static_cast<size_t>(i) / static_cast<size_t>(nshell);
std::nth_element(s2.begin() + prev, s2.begin() + pos, s2.end());
edge[i - 1] = s2[pos];
prev = pos;
}
for (int i = 1; i < nshell; ++i)
edge[i - 1] = sorted[sorted.size() * static_cast<size_t>(i) / static_cast<size_t>(nshell)];
std::vector<uint8_t> term_shell(n_term);
ParallelChunks(n_term, nt, [&](int lo, int hi) {
for (int k = lo; k < hi; ++k)
term_shell[k] = static_cast<uint8_t>(std::upper_bound(edge.begin(), edge.end(), s2[k]) - edge.begin());
});
for (size_t k = 0; k < term.size(); ++k) {
const Term &t = term[k];
const float v = t.d > 0.0f ? 1.0f / (t.d * t.d) : 0.0f;
const int s = static_cast<int>(std::upper_bound(edge.begin(), edge.end(), v) - edge.begin());
const int s = term_shell[k];
group_shell[t.group] = s;
const double sc = static_cast<double>(t.sigma) * t.corr;
if (!(sc > 0.0)) continue;
@@ -3475,8 +3481,41 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
std::vector<int32_t> g_start(n_groups + 1, 0);
for (const Term &t : term) ++g_start[t.group + 1];
for (int g = 0; g < n_groups; ++g) g_start[g + 1] += g_start[g];
std::vector<RefTerm> gterm(term.size());
{
std::vector<RefTerm> gterm;
// With a GPU the two passes of every round run there (RotationScaleMergeGPU::Surface*), on the same
// terms in the same order: the device gets the terms with the group order as a permutation, and each
// subset cut into this fit's reduction blocks with every block's terms ordered by cell, so that one
// device thread forms what one block of the host loop sums into one cell.
bool on_gpu = false;
#ifdef JFJOCH_USE_CUDA
on_gpu = gpu_active_;
if (on_gpu) {
std::vector<int32_t> gperm(term.size());
std::vector<int32_t> fill(g_start.begin(), g_start.end() - 1);
for (size_t k = 0; k < term.size(); ++k)
gperm[fill[term[k].group]++] = static_cast<int32_t>(k);
gpu_->SurfaceSetTerms(static_cast<int>(term.size()), term.data(), term_parity.data(),
n_groups, gperm.data(), g_start.data(), ncell);
for (int parity : {0, 1, -1}) {
const std::vector<int32_t> &sel = subset(parity);
const int n = static_cast<int>(sel.size());
const int nb = ReductionBlocks(n, SURFACE_BLOCK);
std::vector<int32_t> perm(n), seg_start(static_cast<size_t>(nb) * ncell + 1, n);
ParallelFor(nb, nt, [&](int b) {
const int lo = static_cast<int>(static_cast<int64_t>(n) * b / nb);
const int hi = static_cast<int>(static_cast<int64_t>(n) * (b + 1) / nb);
std::vector<int32_t> pos(ncell + 1, 0);
for (int k = lo; k < hi; ++k) ++pos[term[sel[k]].cell + 1];
for (int c = 0; c < ncell; ++c) pos[c + 1] += pos[c];
for (int c = 0; c < ncell; ++c) seg_start[static_cast<size_t>(b) * ncell + c] = lo + pos[c];
for (int k = lo; k < hi; ++k) perm[lo + pos[term[sel[k]].cell]++] = sel[k];
});
gpu_->SurfaceSetSubset(parity < 0 ? 2 : parity, nb, perm.data(), seg_start.data());
}
}
#endif
if (!on_gpu) {
gterm.resize(term.size());
std::vector<int32_t> fill(g_start.begin(), g_start.end() - 1);
for (size_t k = 0; k < term.size(); ++k) {
const Term &t = term[k];
@@ -3488,6 +3527,12 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
// and the score.
std::vector<double> sw(n_groups), swI(n_groups);
auto reference = [&](int parity, const std::vector<double> &A) {
#ifdef JFJOCH_USE_CUDA
if (on_gpu) {
gpu_->SurfaceReference(parity, A.data());
return;
}
#endif
ParallelChunks(n_groups, nt, [&](int glo, int ghi) {
for (int g = glo; g < ghi; ++g) {
double s_w = 0.0, s_wI = 0.0;
@@ -3520,7 +3565,7 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
const std::vector<int32_t> &sel = subset(parity);
std::vector<double> A(ncell, 1.0);
// Per-block cell accumulators, allocated once for the whole fit rather than per round.
const int nb = ReductionBlocks(static_cast<int>(sel.size()), SURFACE_BLOCK);
const int nb = on_gpu ? 0 : ReductionBlocks(static_cast<int>(sel.size()), SURFACE_BLOCK);
std::vector<std::vector<double>> tcross(nb, std::vector<double>(ncell)), tref2(nb, std::vector<double>(ncell));
settled = false;
n_clamped = 0;
@@ -3555,22 +3600,28 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
// there is no ordering that keeps threads off each other's bins, and there are only ncell
// of them, so per-block copies are cheap and the fixed blocks keep it the same at any -N.
std::vector<double> cross(ncell, 0.0), ref2(ncell, 0.0);
ParallelBlocks(static_cast<int>(sel.size()), nt, [&](int b, int lo, int hi) {
std::vector<double> &xcross = tcross[b], &xref2 = tref2[b];
std::fill(xcross.begin(), xcross.end(), 0.0);
std::fill(xref2.begin(), xref2.end(), 0.0);
for (int k = lo; k < hi; ++k) {
const Term &o = term[sel[k]];
if (sw[o.group] <= 0.0) continue;
const double Iref = swI[o.group] / sw[o.group], a = A[o.cell];
const double Is = static_cast<double>(o.I) * o.corr * a, sc = static_cast<double>(o.sigma) * o.corr * a;
if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue;
const double w = 1.0 / (sc * sc);
xcross[o.cell] += w * Is * Iref; xref2[o.cell] += w * Iref * Iref;
}
}, SURFACE_BLOCK);
for (int b = 0; b < nb; ++b)
for (int c = 0; c < ncell; ++c) { cross[c] += tcross[b][c]; ref2[c] += tref2[b][c]; }
#ifdef JFJOCH_USE_CUDA
if (on_gpu)
gpu_->SurfaceFitSums(parity < 0 ? 2 : parity, cross.data(), ref2.data());
#endif
if (!on_gpu) {
ParallelBlocks(static_cast<int>(sel.size()), nt, [&](int b, int lo, int hi) {
std::vector<double> &xcross = tcross[b], &xref2 = tref2[b];
std::fill(xcross.begin(), xcross.end(), 0.0);
std::fill(xref2.begin(), xref2.end(), 0.0);
for (int k = lo; k < hi; ++k) {
const Term &o = term[sel[k]];
if (sw[o.group] <= 0.0) continue;
const double Iref = swI[o.group] / sw[o.group], a = A[o.cell];
const double Is = static_cast<double>(o.I) * o.corr * a, sc = static_cast<double>(o.sigma) * o.corr * a;
if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue;
const double w = 1.0 / (sc * sc);
xcross[o.cell] += w * Is * Iref; xref2[o.cell] += w * Iref * Iref;
}
}, SURFACE_BLOCK);
for (int b = 0; b < nb; ++b)
for (int c = 0; c < ncell; ++c) { cross[c] += tcross[b][c]; ref2[c] += tref2[b][c]; }
}
std::vector<double> dsorted = cross;
std::nth_element(dsorted.begin(), dsorted.begin() + dsorted.size() / 2, dsorted.end());
const double lambda = 0.1 * std::max(1e-30, dsorted[dsorted.size() / 2]);
@@ -3672,20 +3723,33 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
const int nsh_cc = nshell > 0 ? nshell : 1;
auto half_means = [&](int parity, const std::vector<double> &A) {
reference(parity, A);
#ifdef JFJOCH_USE_CUDA
if (on_gpu)
gpu_->SurfaceGetReference(sw.data(), swI.data());
#endif
std::vector<double> m(n_groups, std::numeric_limits<double>::quiet_NaN());
for (int g = 0; g < n_groups; ++g)
if (sw[g] > 0.0) m[g] = swI[g] / sw[g];
return m;
};
auto shell_cc = [&](const std::vector<double> &x, const std::vector<double> &y, int s) {
double n = 0, sx = 0, sy = 0, sxx = 0, syy = 0, sxy = 0;
// The correlation within every shell, in one walk over the groups: each shell still sums its own
// groups in group order.
auto shell_cc = [&](const std::vector<double> &x, const std::vector<double> &y) {
struct Sums { double n = 0, sx = 0, sy = 0, sxx = 0, syy = 0, sxy = 0; };
std::vector<Sums> sum(nsh_cc);
for (int g = 0; g < n_groups; ++g) {
if (group_shell[g] != s || !std::isfinite(x[g]) || !std::isfinite(y[g])) continue;
n += 1; sx += x[g]; sy += y[g]; sxx += x[g] * x[g]; syy += y[g] * y[g]; sxy += x[g] * y[g];
if (!std::isfinite(x[g]) || !std::isfinite(y[g])) continue;
Sums &u = sum[group_shell[g]];
u.n += 1; u.sx += x[g]; u.sy += y[g]; u.sxx += x[g] * x[g]; u.syy += y[g] * y[g]; u.sxy += x[g] * y[g];
}
const double vx = sxx - sx * sx / n, vy = syy - sy * sy / n;
return (n >= 50 && vx > 0.0 && vy > 0.0) ? (sxy - sx * sy / n) / std::sqrt(vx * vy)
: std::numeric_limits<double>::quiet_NaN();
std::vector<double> cc(nsh_cc, std::numeric_limits<double>::quiet_NaN());
for (int s = 0; s < nsh_cc; ++s) {
const Sums &u = sum[s];
const double vx = u.sxx - u.sx * u.sx / u.n, vy = u.syy - u.sy * u.sy / u.n;
if (u.n >= 50 && vx > 0.0 && vy > 0.0)
cc[s] = (u.sxy - u.sx * u.sy / u.n) / std::sqrt(vx * vy);
}
return cc;
};
const std::vector<double> ident(ncell, 1.0);
const std::vector<double> A_even = fit_surface(0), A_odd = fit_surface(1);
@@ -3702,8 +3766,9 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
// Following Fisher (1915) Biometrika 10, 507-521
double gain = 0.0;
int n_cc = 0;
const std::vector<double> cc0 = shell_cc(odd0, even0), cc1 = shell_cc(odd1, even1);
for (int sh = 0; sh < nsh_cc; ++sh) {
const double c0 = shell_cc(odd0, even0, sh), c1 = shell_cc(odd1, even1, sh);
const double c0 = cc0[sh], c1 = cc1[sh];
if (std::isfinite(c0) && std::isfinite(c1)) { gain += std::atanh(c1) - std::atanh(c0); ++n_cc; }
}
if (n_cc > 0) gain /= n_cc;
@@ -5699,7 +5764,8 @@ RotationScaleMerge::Result RotationScaleMerge::Run(bool for_search, bool full_st
ReducePartialGroupMeans(n_groups, partial_mean);
ComputePerFrameCC(partial_mean, cc, cc_n);
}
FinalizePerFrameScale(cc, cc_n, partial_scaled);
if (write_back_per_frame_scale)
FinalizePerFrameScale(cc, cc_n, partial_scaled);
// The filters below remove observations by zeroing corr, which is what takes an observation out of
// the 3D combine, the merge and the error model alike (excluding them from the ASU grouping is NOT
@@ -119,6 +119,11 @@ public:
// compares nothing, asks for it to be left out.
Result Run(bool for_search, bool full_stats, bool measure_cc_before_corrections);
// Whether Run() writes the per-frame G / CC / mosaicity back onto the outcomes (on by default). Off
// for a merge that is not the run's answer - the P1 cross-check - so the per-image table and the
// unmerged MTZ describe the merge that was written.
void SetWriteBackPerFrameScale(bool on) { write_back_per_frame_scale = on; }
// Override the high-resolution cut for the next Run() - used to gate the de-novo P1 search pass at
// <I/sigma> >= 1 without cutting the final in-symmetry merge. Reset to the manual limit afterwards.
void SetDMinLimit(std::optional<double> d_min_A) { d_min_limit = d_min_A; }
@@ -413,6 +418,7 @@ private:
std::unique_ptr<RotationScaleMergeGPU> gpu_;
bool gpu_active_ = false;
#endif
bool write_back_per_frame_scale = true; // see SetWriteBackPerFrameScale
// --- helpers (each a flat pass; see the .cpp) ---
// Turn the per-frame mean background under the reflections (accumulated by the ingest fill loop) into
@@ -18,13 +18,15 @@ namespace {
constexpr int BLK = 256;
constexpr int MIN_REFLECTIONS = 20;
// Every kernel and copy here is queued on the legacy NULL stream, so - as in BeamCenterFFTGPU -
// no buffer comes from the pool. A pooled buffer is freed with cudaFreeAsync on the thread's
// non-blocking allocation stream, which is not ordered after the NULL stream. Each entry point
// below waits for its own work before it returns, so no free here has yet overtaken a read, but
// that holds only by that convention, and not at all for a free while unwinding from a failed
// call; nor can compute-sanitizer --track-stream-ordered-races see it, and it reported every
// reassigned merge buffer as a use-after-free. cudaFree synchronises the device first.
// Every kernel and copy here is queued on the instance's own stream (Impl::stream), not on the
// legacy NULL stream: the merge and the image analysis of a probe pass beside it would otherwise
// wait for each other's work at every launch and every synchronisation. As in BeamCenterFFTGPU no buffer comes from the pool. A
// pooled buffer is freed with cudaFreeAsync on the thread's allocation stream, which is not
// ordered after this one. Each entry point below waits for its own work before it returns, so no
// free here has yet overtaken a read, but that holds only by that convention, and not at all for a
// free while unwinding from a failed call; nor can compute-sanitizer --track-stream-ordered-races
// see it, and it reported every reassigned merge buffer as a use-after-free. cudaFree
// synchronises the device first.
constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous;
__device__ __forceinline__ double SafeInvD(double x, double fallback) {
@@ -636,6 +638,110 @@ namespace {
}
}
// --- correction-surface fit (ApplyCellSurface) ---
// The host fit is the reference here: these kernels form the same sums from the same terms in the
// same order, and every rounding is spelled out (__dmul_rn / __dadd_rn round each step on its own,
// fma rounds once) to be the one the host build makes - nvcc would otherwise contract a multiply
// and an add wherever it sees them, and GCC at -march=x86-64-v3 does so only in some of them. The
// products are the host's (I * corr) * a and (sigma * corr) * a.
using SurfaceTerm = RotationScaleMergeGPU::SurfaceTerm;
__device__ __forceinline__ double SurfaceIs(const SurfaceTerm &t, double a) {
return __dmul_rn(__dmul_rn(double(t.I), double(t.corr)), a);
}
__device__ __forceinline__ double SurfaceSigma(const SurfaceTerm &t, double a) {
return __dmul_rn(__dmul_rn(double(t.sigma), double(t.corr)), a);
}
// One thread per ASU group: the group's inverse-variance sums over its terms of frame parity `parity`
// (< 0 = all), in fulls order - the host's `reference`. The host builds that loop twice, and only
// the parity-filtered copy fuses the multiply-add of swI; the unfiltered one rounds the product.
__global__ void SurfaceReferenceKernel(int n_groups, int parity, const int32_t *__restrict__ gperm,
const int32_t *__restrict__ gstart,
const SurfaceTerm *__restrict__ term,
const uint8_t *__restrict__ term_parity,
const double *__restrict__ A,
double *__restrict__ sw, double *__restrict__ swI) {
for (int g = blockIdx.x * blockDim.x + threadIdx.x; g < n_groups; g += gridDim.x * blockDim.x) {
double s_w = 0.0, s_wI = 0.0;
for (int k = gstart[g]; k < gstart[g + 1]; ++k) {
const int i = gperm[k];
if (parity >= 0 && term_parity[i] != parity) continue;
const SurfaceTerm t = term[i];
const double a = A[t.cell];
const double Is = SurfaceIs(t, a), sc = SurfaceSigma(t, a);
const double w = 1.0 / __dmul_rn(sc, sc);
s_w = __dadd_rn(s_w, w);
s_wI = parity >= 0 ? fma(Is, w, s_wI) : __dadd_rn(s_wI, __dmul_rn(Is, w));
}
sw[g] = s_w; swI[g] = s_wI;
}
}
// The fit's sums of one round. Each term's contribution is formed on its own thread (one per term,
// in segment order), and only the two multiply-adds that sum them are left to the thread of each
// (reduction block, cell) - the per-block accumulators of the host fit, one slot at a time, walking
// its segment in term order. A term the host skips is marked by a NaN Iref (a kept one is finite).
__global__ void SurfaceFitTermKernel(int n, const int32_t *__restrict__ perm,
const SurfaceTerm *__restrict__ term,
const double *__restrict__ A,
const double *__restrict__ sw, const double *__restrict__ swI,
double *__restrict__ w_Is, double *__restrict__ w_Iref,
double *__restrict__ Iref_out) {
for (int k = blockIdx.x * blockDim.x + threadIdx.x; k < n; k += gridDim.x * blockDim.x) {
const SurfaceTerm t = term[perm[k]];
Iref_out[k] = NAN;
if (sw[t.group] <= 0.0) continue;
const double Iref = swI[t.group] / sw[t.group], a = A[t.cell];
const double Is = SurfaceIs(t, a), sc = SurfaceSigma(t, a);
if (!isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue;
const double w = 1.0 / __dmul_rn(sc, sc);
w_Is[k] = __dmul_rn(w, Is);
w_Iref[k] = __dmul_rn(w, Iref);
Iref_out[k] = Iref;
}
}
__global__ void SurfaceFitSegmentKernel(int n_seg, const int32_t *__restrict__ seg_start,
const double *__restrict__ w_Is, const double *__restrict__ w_Iref,
const double *__restrict__ Iref,
double *__restrict__ tcross, double *__restrict__ tref2) {
for (int s = blockIdx.x * blockDim.x + threadIdx.x; s < n_seg; s += gridDim.x * blockDim.x) {
double xcross = 0.0, xref2 = 0.0;
for (int k = seg_start[s]; k < seg_start[s + 1]; ++k) {
if (isnan(Iref[k])) continue;
xcross = fma(w_Is[k], Iref[k], xcross);
xref2 = fma(w_Iref[k], Iref[k], xref2);
}
tcross[s] = xcross; tref2[s] = xref2;
}
}
// One thread per cell: the blocks' sums added up in block order, as the host adds its slots.
__global__ void SurfaceFitCellKernel(int n_blocks, int ncell, const double *__restrict__ tcross,
const double *__restrict__ tref2,
double *__restrict__ cross, double *__restrict__ ref2) {
for (int c = blockIdx.x * blockDim.x + threadIdx.x; c < ncell; c += gridDim.x * blockDim.x) {
double sc = 0.0, sr = 0.0;
for (int b = 0; b < n_blocks; ++b) {
sc = __dadd_rn(sc, tcross[b * ncell + c]);
sr = __dadd_rn(sr, tref2[b * ncell + c]);
}
cross[c] = sc; ref2[c] = sr;
}
}
void CudaCheck(cudaError_t e, const char *what);
// A copy on the instance's stream, waited for - what cudaMemcpy on the NULL stream was, without also
// waiting for every other stream on the card.
void CopyAndWait(void *dst, const void *src, size_t bytes, cudaMemcpyKind kind, cudaStream_t s,
const char *what) {
CudaCheck(cudaMemcpyAsync(dst, src, bytes, kind, s), what);
CudaCheck(cudaStreamSynchronize(s), what);
}
void CudaCheck(cudaError_t e, const char *what) {
if (e != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
@@ -703,6 +809,9 @@ namespace {
}
struct RotationScaleMergeGPU::Impl {
// First, so it goes last: the buffers below are freed before the stream their work ran on.
std::unique_ptr<CudaStream> stream;
cudaStream_t s() const { return stream->get(); }
int device = 0; // the GPU this instance's buffers live on
bool available = false;
int n_obs = 0, n_frames = 0, n_groups = 0;
@@ -736,7 +845,7 @@ struct RotationScaleMergeGPU::Impl {
void Upload(CudaDevicePtr<T> &dst, const T *src, int n) const {
dst = Alloc<T>(std::max(1, n));
if (n > 0)
CudaCheck(cudaMemcpy(dst.get(), src, size_t(n) * sizeof(T), cudaMemcpyHostToDevice), "upload");
CopyAndWait(dst.get(), src, size_t(n) * sizeof(T), cudaMemcpyHostToDevice, s(), "upload");
}
// immutable per-obs
@@ -801,6 +910,18 @@ struct RotationScaleMergeGPU::Impl {
CudaDevicePtr<uint8_t> f_sco_ok;
CudaDevicePtr<int32_t> f_frame_perm, f_frame_start, f_frame_count;
CudaDevicePtr<int32_t> f_gperm, f_gstart, f_gcount;
// correction-surface fit (one ApplyCellSurface call at a time): its terms and their group CSR, the
// three subsets' per-(block, cell) segments, the surface and the sums of the round
int s_ncell = 0, s_n_groups = 0;
int s_n_blocks[3] = {0, 0, 0};
int s_n_sel[3] = {0, 0, 0};
size_t s_slots = 0; // capacity of s_tcross / s_tref2
CudaDevicePtr<RotationScaleMergeGPU::SurfaceTerm> s_term;
CudaDevicePtr<uint8_t> s_parity;
CudaDevicePtr<int32_t> s_gperm, s_gstart;
CudaDevicePtr<int32_t> s_perm[3], s_seg_start[3];
CudaDevicePtr<double> s_A, s_sw, s_swI, s_tcross, s_tref2, s_cross, s_ref2;
CudaDevicePtr<double> s_w_Is, s_w_Iref, s_Iref; // per term of a subset, in segment order
};
// Set the device this instance's memory lives on for the duration of a call, and put the caller's
@@ -833,6 +954,7 @@ RotationScaleMergeGPU::RotationScaleMergeGPU() : impl_(std::make_unique<Impl>())
// can put the caller's device back instead of leaving the thread moved.
impl_->device = 0;
DeviceGuard guard(impl_->device, true);
impl_->stream = std::make_unique<CudaStream>();
impl_->available = true;
}
}
@@ -878,10 +1000,10 @@ void RotationScaleMergeGPU::SetPartialsLayout(int n_obs, int n_frames,
namespace {
template <typename T>
void UploadChunk(CudaDevicePtr<T> &dst, int offset, int count, const T *v) {
void UploadChunk(CudaDevicePtr<T> &dst, int offset, int count, const T *v, cudaStream_t s) {
if (count > 0)
CudaCheck(cudaMemcpy(dst.get() + offset, v, size_t(count) * sizeof(T),
cudaMemcpyHostToDevice), "upload chunk");
CopyAndWait(dst.get() + offset, v, size_t(count) * sizeof(T), cudaMemcpyHostToDevice, s,
"upload chunk");
}
}
@@ -889,34 +1011,34 @@ void RotationScaleMergeGPU::SetObsField(ObsField f, int offset, int count, const
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
switch (f) {
case ObsField::I: UploadChunk(d.I, offset, count, v); break;
case ObsField::Sigma: UploadChunk(d.sigma, offset, count, v); break;
case ObsField::PrescalingCorr: UploadChunk(d.prescaling_corr, offset, count, v); break;
case ObsField::Partiality: UploadChunk(d.partiality, offset, count, v); break;
case ObsField::Zeta: UploadChunk(d.zeta, offset, count, v); break;
case ObsField::Corr0: UploadChunk(d.corr, offset, count, v); break;
case ObsField::Bkg: UploadChunk(d.bkg, offset, count, v); break;
case ObsField::VarBkg: UploadChunk(d.var_bkg, offset, count, v); break;
case ObsField::ImageNumber: UploadChunk(d.image_number, offset, count, v); break;
case ObsField::D: UploadChunk(d.d_obs, offset, count, v); break;
case ObsField::Px: UploadChunk(d.px_obs, offset, count, v); break;
case ObsField::Py: UploadChunk(d.py_obs, offset, count, v); break;
case ObsField::I: UploadChunk(d.I, offset, count, v, impl_->s()); break;
case ObsField::Sigma: UploadChunk(d.sigma, offset, count, v, impl_->s()); break;
case ObsField::PrescalingCorr: UploadChunk(d.prescaling_corr, offset, count, v, impl_->s()); break;
case ObsField::Partiality: UploadChunk(d.partiality, offset, count, v, impl_->s()); break;
case ObsField::Zeta: UploadChunk(d.zeta, offset, count, v, impl_->s()); break;
case ObsField::Corr0: UploadChunk(d.corr, offset, count, v, impl_->s()); break;
case ObsField::Bkg: UploadChunk(d.bkg, offset, count, v, impl_->s()); break;
case ObsField::VarBkg: UploadChunk(d.var_bkg, offset, count, v, impl_->s()); break;
case ObsField::ImageNumber: UploadChunk(d.image_number, offset, count, v, impl_->s()); break;
case ObsField::D: UploadChunk(d.d_obs, offset, count, v, impl_->s()); break;
case ObsField::Px: UploadChunk(d.px_obs, offset, count, v, impl_->s()); break;
case ObsField::Py: UploadChunk(d.py_obs, offset, count, v, impl_->s()); break;
}
}
void RotationScaleMergeGPU::SetObsFrame(int offset, int count, const int32_t *frame) {
DeviceGuard guard(impl_->device, impl_->available);
UploadChunk(impl_->frame, offset, count, frame);
UploadChunk(impl_->frame, offset, count, frame, impl_->s());
}
void RotationScaleMergeGPU::SetObsOnIce(int offset, int count, const uint8_t *on_ice) {
DeviceGuard guard(impl_->device, impl_->available);
UploadChunk(impl_->on_ice, offset, count, on_ice);
UploadChunk(impl_->on_ice, offset, count, on_ice, impl_->s());
}
void RotationScaleMergeGPU::SetObsClipped(int offset, int count, const uint8_t *clipped) {
DeviceGuard guard(impl_->device, impl_->available);
UploadChunk(impl_->clipped, offset, count, clipped);
UploadChunk(impl_->clipped, offset, count, clipped, impl_->s());
}
void RotationScaleMergeGPU::SetGroups(int n_groups, const int32_t *group, const int32_t *group_perm,
@@ -934,8 +1056,8 @@ void RotationScaleMergeGPU::SetGroups(int n_groups, const int32_t *group, const
void RotationScaleMergeGPU::SetCorr(const float *corr) {
DeviceGuard guard(impl_->device, impl_->available);
CudaCheck(cudaMemcpy(impl_->corr.get(), corr, size_t(impl_->n_obs) * sizeof(float),
cudaMemcpyHostToDevice), "upload corr");
CopyAndWait(impl_->corr.get(), corr, size_t(impl_->n_obs) * sizeof(float),
cudaMemcpyHostToDevice, impl_->s(), "upload corr");
}
void RotationScaleMergeGPU::ScalePartials(int iters, double min_partiality, bool /*has_d_min*/) {
@@ -943,42 +1065,42 @@ void RotationScaleMergeGPU::ScalePartials(int iters, double min_partiality, bool
auto &d = *impl_;
// Reset per call: the host keeps the G of a frame across calls (RunScalingLoop), so a frame this
// call did not fit must read as unfitted, not as fitted with the value of the call before.
CudaCheck(cudaMemset(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t)), "memset scaled");
CudaCheck(cudaMemset(d.g.get(), 0, size_t(d.n_frames) * sizeof(double)), "memset g"); // unscaled g unused
CudaCheck(cudaMemsetAsync(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t), impl_->s()), "memset scaled");
CudaCheck(cudaMemsetAsync(d.g.get(), 0, size_t(d.n_frames) * sizeof(double), impl_->s()), "memset g"); // unscaled g unused
const int obs_blocks = (d.n_obs + BLK - 1) / BLK;
const int upd_blocks = std::min(65535, obs_blocks);
const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK);
for (int it = 0; it < iters; ++it) {
ReduceGroupMeansKernel<<<grp_blocks, BLK>>>(d.n_groups, min_partiality,
ReduceGroupMeansKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(d.n_groups, min_partiality,
d.group_perm.get(), d.group_start.get(), d.group_count.get(),
d.I.get(), d.sigma.get(), d.partiality.get(), d.corr.get(), d.group_mean.get());
CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch");
PrepScaleObsKernel<<<obs_blocks, BLK>>>(d.n_obs, min_partiality, d.group.get(), d.partiality.get(), d.prescaling_corr.get(),
PrepScaleObsKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.n_obs, min_partiality, d.group.get(), d.partiality.get(), d.prescaling_corr.get(),
d.zeta.get(), d.on_ice.get(), d.group_mean.get(), d.sigma.get(), d.inv_sigma.get(),
d.sco_coeff.get(), d.sco_ok.get());
CudaCheck(cudaGetLastError(), "PrepScaleObsKernel launch");
FitPerFrameGKernel<<<d.n_frames, BLK>>>(d.n_frames, d.frame_start.get(), d.frame_count.get(),
FitPerFrameGKernel<<<d.n_frames, BLK, 0, impl_->s()>>>(d.n_frames, d.frame_start.get(), d.frame_count.get(),
d.I.get(), d.inv_sigma.get(), d.sco_coeff.get(), d.sco_ok.get(), nullptr, d.g.get(), d.scaled.get());
CudaCheck(cudaGetLastError(), "FitPerFrameGKernel launch");
UpdateCorrKernel<<<upd_blocks, BLK>>>(d.n_obs, d.frame.get(), d.prescaling_corr.get(), d.partiality.get(),
UpdateCorrKernel<<<upd_blocks, BLK, 0, impl_->s()>>>(d.n_obs, d.frame.get(), d.prescaling_corr.get(), d.partiality.get(),
d.g.get(), d.scaled.get(), d.corr.get());
}
CudaCheck(cudaGetLastError(), "kernel launch");
CudaCheck(cudaDeviceSynchronize(), "scale sync");
CudaCheck(cudaStreamSynchronize(impl_->s()), "scale sync");
}
void RotationScaleMergeGPU::GetCorr(float *corr_out) const {
DeviceGuard guard(impl_->device, impl_->available);
CudaCheck(cudaMemcpy(corr_out, impl_->corr.get(), size_t(impl_->n_obs) * sizeof(float),
cudaMemcpyDeviceToHost), "download corr");
CopyAndWait(corr_out, impl_->corr.get(), size_t(impl_->n_obs) * sizeof(float),
cudaMemcpyDeviceToHost, impl_->s(), "download corr");
}
void RotationScaleMergeGPU::GetG(double *g_out, uint8_t *scaled_out) const {
DeviceGuard guard(impl_->device, impl_->available);
CudaCheck(cudaMemcpy(g_out, impl_->g.get(), size_t(impl_->n_frames) * sizeof(double),
cudaMemcpyDeviceToHost), "download g");
CudaCheck(cudaMemcpy(scaled_out, impl_->scaled.get(), size_t(impl_->n_frames) * sizeof(uint8_t),
cudaMemcpyDeviceToHost), "download scaled");
CopyAndWait(g_out, impl_->g.get(), size_t(impl_->n_frames) * sizeof(double),
cudaMemcpyDeviceToHost, impl_->s(), "download g");
CopyAndWait(scaled_out, impl_->scaled.get(), size_t(impl_->n_frames) * sizeof(uint8_t),
cudaMemcpyDeviceToHost, impl_->s(), "download scaled");
}
void RotationScaleMergeGPU::SetFrameCellOk(const uint8_t *frame_cell_ok) {
@@ -1030,19 +1152,19 @@ void RotationScaleMergeGPU::MergeEmSamples(bool for_search, double min_partialit
const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK);
const int obs_blocks = std::min(65535, (nf + BLK - 1) / BLK);
MergeEmStatsKernel<<<grp_blocks, BLK>>>(p);
MergeSamplesKernel<<<obs_blocks, BLK>>>(nf, p);
MergeEmStatsKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(p);
MergeSamplesKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(nf, p);
CudaCheck(cudaGetLastError(), "merge em/samples launch");
CudaCheck(cudaDeviceSynchronize(), "merge em/samples sync");
CudaCheck(cudaMemcpy(em_mean_out, d.m_em_mean.get(), size_t(ng) * sizeof(double),
cudaMemcpyDeviceToHost), "dl em_mean");
CudaCheck(cudaMemcpy(cnt_out, d.m_cnt.get(), size_t(ng) * sizeof(int32_t),
cudaMemcpyDeviceToHost), "dl cnt");
CudaCheck(cudaStreamSynchronize(impl_->s()), "merge em/samples sync");
CopyAndWait(em_mean_out, d.m_em_mean.get(), size_t(ng) * sizeof(double),
cudaMemcpyDeviceToHost, impl_->s(), "dl em_mean");
CopyAndWait(cnt_out, d.m_cnt.get(), size_t(ng) * sizeof(int32_t),
cudaMemcpyDeviceToHost, impl_->s(), "dl cnt");
if (nf > 0) {
CudaCheck(cudaMemcpy(s2_out, d.m_s2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl s2");
CudaCheck(cudaMemcpy(I2_out, d.m_I2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl I2");
CudaCheck(cudaMemcpy(dev2_out, d.m_dev2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl dev2");
CudaCheck(cudaMemcpy(valid_out, d.m_valid.get(), size_t(nf) * sizeof(uint8_t), cudaMemcpyDeviceToHost), "dl valid");
CopyAndWait(s2_out, d.m_s2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl s2");
CopyAndWait(I2_out, d.m_I2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl I2");
CopyAndWait(dev2_out, d.m_dev2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl dev2");
CopyAndWait(valid_out, d.m_valid.get(), size_t(nf) * sizeof(uint8_t), cudaMemcpyDeviceToHost, impl_->s(), "dl valid");
}
}
@@ -1091,10 +1213,10 @@ void RotationScaleMergeGPU::MergeAccum(double error_model_a, double error_model_
p.rejected_obs = d.m_rejected.get();
const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK);
MergeAccumKernel<<<grp_blocks, BLK>>>(p);
MergeAccumKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(p);
CudaCheck(cudaGetLastError(), "merge accum launch");
CudaCheck(cudaDeviceSynchronize(), "merge accum sync");
CudaCheck(cudaMemcpy(rejected_obs, d.m_rejected.get(), size_t(d.n_fulls) * sizeof(uint8_t), cudaMemcpyDeviceToHost),
CudaCheck(cudaStreamSynchronize(impl_->s()), "merge accum sync");
CopyAndWait(rejected_obs, d.m_rejected.get(), size_t(d.n_fulls) * sizeof(uint8_t), cudaMemcpyDeviceToHost, impl_->s(),
"dl rejected_obs");
}
@@ -1105,7 +1227,7 @@ void RotationScaleMergeGPU::MergeAccumRange(int g0, int n, double *swI, double *
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
auto dl = [&](void *h, const auto &s) {
CudaCheck(cudaMemcpy(h, s.get() + g0, size_t(n) * sizeof(*s.get()), cudaMemcpyDeviceToHost),
CopyAndWait(h, s.get() + g0, size_t(n) * sizeof(*s.get()), cudaMemcpyDeviceToHost, impl_->s(),
"dl accum"); };
dl(swI, d.a_swI); dl(sw, d.a_sw); dl(swIh0, d.a_swIh0); dl(swIh1, d.a_swIh1);
dl(swh0, d.a_swh0); dl(swh1, d.a_swh1); dl(swh_typ0, d.a_swht0); dl(swh_typ1, d.a_swht1);
@@ -1139,17 +1261,17 @@ void RotationScaleMergeGPU::MergeRmeas(const double *merged_I, double *absdev, d
p.rejected_obs = d.m_rejected.get();
const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK);
MergeRmeasKernel<<<grp_blocks, BLK>>>(p);
MergeRmeasKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(p);
CudaCheck(cudaGetLastError(), "merge rmeas launch");
CudaCheck(cudaDeviceSynchronize(), "merge rmeas sync");
CudaCheck(cudaMemcpy(absdev, d.r_absdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl absdev");
CudaCheck(cudaMemcpy(sumI, d.r_sumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumI");
CudaCheck(cudaMemcpy(wabsdev, d.r_wabsdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl wabsdev");
CudaCheck(cudaMemcpy(wsumI, d.r_wsumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl wsumI");
CudaCheck(cudaMemcpy(sumv, d.r_sumv.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumv");
CudaCheck(cudaMemcpy(sumv2, d.r_sumv2.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumv2");
CudaCheck(cudaMemcpy(n, d.r_n.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost), "dl rn");
CudaCheck(cudaMemcpy(nusable, d.r_nusable.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost), "dl rnusable");
CudaCheck(cudaStreamSynchronize(impl_->s()), "merge rmeas sync");
CopyAndWait(absdev, d.r_absdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl absdev");
CopyAndWait(sumI, d.r_sumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumI");
CopyAndWait(wabsdev, d.r_wabsdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl wabsdev");
CopyAndWait(wsumI, d.r_wsumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl wsumI");
CopyAndWait(sumv, d.r_sumv.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumv");
CopyAndWait(sumv2, d.r_sumv2.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumv2");
CopyAndWait(n, d.r_n.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost, impl_->s(), "dl rn");
CopyAndWait(nusable, d.r_nusable.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost, impl_->s(), "dl rnusable");
}
void RotationScaleMergeGPU::SmoothCorr(const uint8_t *apply, const double *ratio) {
@@ -1158,10 +1280,10 @@ void RotationScaleMergeGPU::SmoothCorr(const uint8_t *apply, const double *ratio
d.Upload(d.smooth_apply, apply, d.n_frames);
d.Upload(d.smooth_ratio, ratio, d.n_frames);
const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK);
SmoothCorrKernel<<<blocks, BLK>>>(d.n_obs, d.frame.get(), d.smooth_apply.get(),
SmoothCorrKernel<<<blocks, BLK, 0, impl_->s()>>>(d.n_obs, d.frame.get(), d.smooth_apply.get(),
d.smooth_ratio.get(), d.corr.get());
CudaCheck(cudaGetLastError(), "smooth corr launch");
CudaCheck(cudaDeviceSynchronize(), "smooth corr sync");
CudaCheck(cudaStreamSynchronize(impl_->s()), "smooth corr sync");
}
void RotationScaleMergeGPU::SmoothFullsCorr(const uint8_t *apply, const double *ratio) {
@@ -1171,23 +1293,23 @@ void RotationScaleMergeGPU::SmoothFullsCorr(const uint8_t *apply, const double *
d.Upload(d.smooth_apply, apply, d.n_frames);
d.Upload(d.smooth_ratio, ratio, d.n_frames);
const int blocks = std::min(65535, (d.n_fulls + BLK - 1) / BLK);
SmoothCorrKernel<<<blocks, BLK>>>(d.n_fulls, d.f_frame.get(), d.smooth_apply.get(),
SmoothCorrKernel<<<blocks, BLK, 0, impl_->s()>>>(d.n_fulls, d.f_frame.get(), d.smooth_apply.get(),
d.smooth_ratio.get(), d.f_corr.get());
CudaCheck(cudaGetLastError(), "smooth fulls corr launch");
CudaCheck(cudaDeviceSynchronize(), "smooth fulls corr sync");
CudaCheck(cudaStreamSynchronize(impl_->s()), "smooth fulls corr sync");
}
int64_t RotationScaleMergeGPU::FilterCorrByZeta(double min_zeta) {
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
CudaDevicePtr<unsigned long long> dropped = d.Alloc<unsigned long long>(1);
CudaCheck(cudaMemset(dropped.get(), 0, sizeof(unsigned long long)), "zero zeta drop count");
CudaCheck(cudaMemsetAsync(dropped.get(), 0, sizeof(unsigned long long), impl_->s()), "zero zeta drop count");
const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK);
FilterZetaKernel<<<blocks, BLK>>>(d.n_obs, min_zeta, d.zeta.get(), d.corr.get(), dropped.get());
FilterZetaKernel<<<blocks, BLK, 0, impl_->s()>>>(d.n_obs, min_zeta, d.zeta.get(), d.corr.get(), dropped.get());
CudaCheck(cudaGetLastError(), "zeta filter launch");
CudaCheck(cudaDeviceSynchronize(), "zeta filter sync");
CudaCheck(cudaStreamSynchronize(impl_->s()), "zeta filter sync");
unsigned long long n = 0;
CudaCheck(cudaMemcpy(&n, dropped.get(), sizeof(unsigned long long), cudaMemcpyDeviceToHost),
CopyAndWait(&n, dropped.get(), sizeof(unsigned long long), cudaMemcpyDeviceToHost, impl_->s(),
"dl zeta drop count");
return static_cast<int64_t>(n);
}
@@ -1197,9 +1319,9 @@ void RotationScaleMergeGPU::FilterCorrByFrame(const uint8_t *reject) {
auto &d = *impl_;
d.Upload(d.filter_reject, reject, d.n_frames);
const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK);
FilterFrameKernel<<<blocks, BLK>>>(d.n_obs, d.frame.get(), d.filter_reject.get(), d.corr.get());
FilterFrameKernel<<<blocks, BLK, 0, impl_->s()>>>(d.n_obs, d.frame.get(), d.filter_reject.get(), d.corr.get());
CudaCheck(cudaGetLastError(), "frame filter launch");
CudaCheck(cudaDeviceSynchronize(), "frame filter sync");
CudaCheck(cudaStreamSynchronize(impl_->s()), "frame filter sync");
}
void RotationScaleMergeGPU::ComputePartialCC(double min_partiality, double *cc_out, int64_t *cc_n_out) {
@@ -1208,19 +1330,19 @@ void RotationScaleMergeGPU::ComputePartialCC(double min_partiality, double *cc_o
const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK);
// Post-smooth group means (reuse the scaling reduce; reads the resident, smoothed corr), then the
// per-frame CC over the resident partials. Only the tiny per-frame cc/cc_n come back to the host.
ReduceGroupMeansKernel<<<grp_blocks, BLK>>>(d.n_groups, min_partiality,
ReduceGroupMeansKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(d.n_groups, min_partiality,
d.group_perm.get(), d.group_start.get(), d.group_count.get(),
d.I.get(), d.sigma.get(), d.partiality.get(), d.corr.get(), d.group_mean.get());
CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch");
PerFrameCCKernel<<<d.n_frames, BLK>>>(d.n_frames, min_partiality,
PerFrameCCKernel<<<d.n_frames, BLK, 0, impl_->s()>>>(d.n_frames, min_partiality,
d.frame_start.get(), d.frame_count.get(), d.I.get(), d.sigma.get(), d.partiality.get(),
d.corr.get(), d.on_ice.get(), d.group.get(), d.group_mean.get(), d.cc.get(), d.cc_n.get());
CudaCheck(cudaGetLastError(), "partial CC launch");
CudaCheck(cudaDeviceSynchronize(), "partial CC sync");
CudaCheck(cudaMemcpy(cc_out, d.cc.get(), size_t(d.n_frames) * sizeof(double),
cudaMemcpyDeviceToHost), "download cc");
CudaCheck(cudaMemcpy(cc_n_out, d.cc_n.get(), size_t(d.n_frames) * sizeof(int64_t),
cudaMemcpyDeviceToHost), "download cc_n");
CudaCheck(cudaStreamSynchronize(impl_->s()), "partial CC sync");
CopyAndWait(cc_out, d.cc.get(), size_t(d.n_frames) * sizeof(double),
cudaMemcpyDeviceToHost, impl_->s(), "download cc");
CopyAndWait(cc_n_out, d.cc_n.get(), size_t(d.n_frames) * sizeof(int64_t),
cudaMemcpyDeviceToHost, impl_->s(), "download cc_n");
}
void RotationScaleMergeGPU::SetRawRuns(int n_runs, int n_perm, const int32_t *perm,
@@ -1247,8 +1369,8 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti
float max_frame_gap) {
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
CudaCheck(cudaMemcpy(d.rr_group.get(), rawrun_group, size_t(d.n_runs) * sizeof(int32_t),
cudaMemcpyHostToDevice), "upload rr_group");
CopyAndWait(d.rr_group.get(), rawrun_group, size_t(d.n_runs) * sizeof(int32_t),
cudaMemcpyHostToDevice, impl_->s(), "upload rr_group");
CombineParams p{};
p.n_runs = d.n_runs;
@@ -1267,13 +1389,13 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti
const int blocks = std::min(65535, (d.n_runs + BLK - 1) / BLK);
// Count pass: how many fulls each run emits.
CombineKernel<false><<<blocks, BLK>>>(p);
CombineKernel<false><<<blocks, BLK, 0, impl_->s()>>>(p);
CudaCheck(cudaGetLastError(), "combine count launch");
// Exclusive prefix sum on the host (deterministic) -> per-run output offset + total fulls.
std::vector<int32_t> nevents(d.n_runs);
CudaCheck(cudaMemcpy(nevents.data(), d.rr_nevents.get(), size_t(d.n_runs) * sizeof(int32_t),
cudaMemcpyDeviceToHost), "download nevents");
CopyAndWait(nevents.data(), d.rr_nevents.get(), size_t(d.n_runs) * sizeof(int32_t),
cudaMemcpyDeviceToHost, impl_->s(), "download nevents");
std::vector<int32_t> offset(d.n_runs);
int64_t acc = 0;
for (int r = 0; r < d.n_runs; ++r) { offset[r] = static_cast<int32_t>(acc); acc += nevents[r]; }
@@ -1292,8 +1414,8 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti
d.f_rlp = d.Alloc<float>(nf); d.f_zeta = d.Alloc<float>(nf);
d.f_inv_sigma = d.Alloc<double>(nf);
d.f_sco_coeff = d.Alloc<float>(nf); d.f_sco_ok = d.Alloc<uint8_t>(nf);
CudaCheck(cudaMemcpy(d.rr_offset.get(), offset.data(), size_t(d.n_runs) * sizeof(int32_t),
cudaMemcpyHostToDevice), "upload offset");
CopyAndWait(d.rr_offset.get(), offset.data(), size_t(d.n_runs) * sizeof(int32_t),
cudaMemcpyHostToDevice, impl_->s(), "upload offset");
p.rr_offset = d.rr_offset.get();
p.f_h = d.f_h.get(); p.f_k = d.f_k.get(); p.f_l = d.f_l.get();
@@ -1303,10 +1425,10 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti
p.f_var_bkg = d.f_var_bkg.get(); p.f_var_per_I = d.f_var_per_I.get();
p.f_on_ice = d.f_on_ice.get(); p.f_clipped = d.f_clipped.get();
if (d.n_fulls > 0) {
CombineKernel<true><<<blocks, BLK>>>(p);
CombineKernel<true><<<blocks, BLK, 0, impl_->s()>>>(p);
CudaCheck(cudaGetLastError(), "combine emit launch");
}
CudaCheck(cudaDeviceSynchronize(), "combine sync");
CudaCheck(cudaStreamSynchronize(impl_->s()), "combine sync");
return d.n_fulls;
}
@@ -1318,7 +1440,7 @@ void RotationScaleMergeGPU::GetFulls(int32_t *h, int32_t *k, int32_t *l, float *
const size_t n = static_cast<size_t>(dd.n_fulls);
if (n == 0) return;
auto dl = [&](void *dst, const void *src, size_t bytes) {
CudaCheck(cudaMemcpy(dst, src, bytes, cudaMemcpyDeviceToHost), "download fulls");
CopyAndWait(dst, src, bytes, cudaMemcpyDeviceToHost, impl_->s(), "download fulls");
};
dl(h, dd.f_h.get(), n * sizeof(int32_t)); dl(k, dd.f_k.get(), n * sizeof(int32_t));
dl(l, dd.f_l.get(), n * sizeof(int32_t)); dl(frame, dd.f_frame.get(), n * sizeof(int32_t));
@@ -1334,8 +1456,8 @@ void RotationScaleMergeGPU::GetFullsKeys(int32_t *frame, int32_t *group) const {
const auto &d = *impl_;
if (d.n_fulls == 0) return;
const size_t bytes = size_t(d.n_fulls) * sizeof(int32_t);
CudaCheck(cudaMemcpy(frame, d.f_frame.get(), bytes, cudaMemcpyDeviceToHost), "download f_frame");
CudaCheck(cudaMemcpy(group, d.f_group.get(), bytes, cudaMemcpyDeviceToHost), "download f_group");
CopyAndWait(frame, d.f_frame.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_frame");
CopyAndWait(group, d.f_group.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_group");
}
void RotationScaleMergeGPU::SetFullsFrameCSR(const int32_t *frame_perm, int n_perm,
@@ -1363,15 +1485,15 @@ void RotationScaleMergeGPU::ResetFullsScale() {
if (nf == 0) return;
const int obs_blocks = std::min(65535, (nf + BLK - 1) / BLK);
// Unity model: partiality/prescaling_corr/zeta = 1 so coeff = mean; corr starts at 1.
FillKernel<<<obs_blocks, BLK>>>(d.f_corr.get(), nf, 1.0f);
FillKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.f_corr.get(), nf, 1.0f);
CudaCheck(cudaGetLastError(), "FillKernel launch");
FillKernel<<<obs_blocks, BLK>>>(d.f_partiality.get(), nf, 1.0f);
FillKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.f_partiality.get(), nf, 1.0f);
CudaCheck(cudaGetLastError(), "FillKernel launch");
FillKernel<<<obs_blocks, BLK>>>(d.f_rlp.get(), nf, 1.0f);
FillKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.f_rlp.get(), nf, 1.0f);
CudaCheck(cudaGetLastError(), "FillKernel launch");
FillKernel<<<obs_blocks, BLK>>>(d.f_zeta.get(), nf, 1.0f);
FillKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.f_zeta.get(), nf, 1.0f);
CudaCheck(cudaGetLastError(), "FillKernel launch");
CudaCheck(cudaDeviceSynchronize(), "reset fulls scale sync");
CudaCheck(cudaStreamSynchronize(impl_->s()), "reset fulls scale sync");
}
void RotationScaleMergeGPU::ScaleFulls(int iters, double min_partiality) {
@@ -1383,38 +1505,38 @@ void RotationScaleMergeGPU::ScaleFulls(int iters, double min_partiality) {
const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK);
// Reset per call, as ScalePartials: the host keeps the G of a frame across calls.
CudaCheck(cudaMemset(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t)), "memset f scaled");
CudaCheck(cudaMemset(d.g.get(), 0, size_t(d.n_frames) * sizeof(double)), "memset f g");
CudaCheck(cudaMemsetAsync(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t), impl_->s()), "memset f scaled");
CudaCheck(cudaMemsetAsync(d.g.get(), 0, size_t(d.n_frames) * sizeof(double), impl_->s()), "memset f g");
for (int it = 0; it < iters; ++it) {
ReduceGroupMeansKernel<<<grp_blocks, BLK>>>(d.n_groups, min_partiality,
ReduceGroupMeansKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(d.n_groups, min_partiality,
d.f_gperm.get(), d.f_gstart.get(), d.f_gcount.get(),
d.f_I.get(), d.f_sigma.get(), d.f_partiality.get(), d.f_corr.get(), d.group_mean.get());
CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch");
// Not grid-stride, so its grid has to cover every full - unlike the grid-stride kernels
// below, which the 65535 cap is there for. Capped, it would silently leave the tail of
// sco_coeff/sco_ok stale above 16.8M fulls.
PrepScaleObsKernel<<<(nf + BLK - 1) / BLK, BLK>>>(nf, min_partiality, d.f_group.get(), d.f_partiality.get(),
PrepScaleObsKernel<<<(nf + BLK - 1) / BLK, BLK, 0, impl_->s()>>>(nf, min_partiality, d.f_group.get(), d.f_partiality.get(),
d.f_rlp.get(), d.f_zeta.get(), d.f_on_ice.get(), d.group_mean.get(),
d.f_sigma.get(), d.f_inv_sigma.get(), d.f_sco_coeff.get(), d.f_sco_ok.get());
CudaCheck(cudaGetLastError(), "PrepScaleObsKernel launch");
FitPerFrameGKernel<<<d.n_frames, BLK>>>(d.n_frames,
FitPerFrameGKernel<<<d.n_frames, BLK, 0, impl_->s()>>>(d.n_frames,
d.f_frame_start.get(), d.f_frame_count.get(), d.f_I.get(), d.f_inv_sigma.get(),
d.f_sco_coeff.get(), d.f_sco_ok.get(), d.f_frame_perm.get(), d.g.get(), d.scaled.get());
CudaCheck(cudaGetLastError(), "FitPerFrameGKernel launch");
UpdateCorrKernel<<<obs_blocks, BLK>>>(nf, d.f_frame.get(), d.f_rlp.get(), d.f_partiality.get(),
UpdateCorrKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(nf, d.f_frame.get(), d.f_rlp.get(), d.f_partiality.get(),
d.g.get(), d.scaled.get(), d.f_corr.get());
}
CudaCheck(cudaGetLastError(), "scale fulls launch");
CudaCheck(cudaDeviceSynchronize(), "scale fulls sync");
CudaCheck(cudaStreamSynchronize(impl_->s()), "scale fulls sync");
}
void RotationScaleMergeGPU::GetFullsCorr(float *corr) const {
DeviceGuard guard(impl_->device, impl_->available);
const auto &d = *impl_;
if (d.n_fulls == 0) return;
CudaCheck(cudaMemcpy(corr, d.f_corr.get(), size_t(d.n_fulls) * sizeof(float),
cudaMemcpyDeviceToHost), "download f_corr");
CopyAndWait(corr, d.f_corr.get(), size_t(d.n_fulls) * sizeof(float),
cudaMemcpyDeviceToHost, impl_->s(), "download f_corr");
}
void RotationScaleMergeGPU::GetFullsPxPy(float *px, float *py) const {
@@ -1422,8 +1544,8 @@ void RotationScaleMergeGPU::GetFullsPxPy(float *px, float *py) const {
const auto &d = *impl_;
if (d.n_fulls == 0) return;
const size_t bytes = size_t(d.n_fulls) * sizeof(float);
CudaCheck(cudaMemcpy(px, d.f_px.get(), bytes, cudaMemcpyDeviceToHost), "download f_px");
CudaCheck(cudaMemcpy(py, d.f_py.get(), bytes, cudaMemcpyDeviceToHost), "download f_py");
CopyAndWait(px, d.f_px.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_px");
CopyAndWait(py, d.f_py.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_py");
}
void RotationScaleMergeGPU::GetFullsVariance(float *var_bkg, float *var_per_I) const {
@@ -1431,8 +1553,8 @@ void RotationScaleMergeGPU::GetFullsVariance(float *var_bkg, float *var_per_I) c
const auto &d = *impl_;
if (d.n_fulls == 0) return;
const size_t bytes = size_t(d.n_fulls) * sizeof(float);
CudaCheck(cudaMemcpy(var_bkg, d.f_var_bkg.get(), bytes, cudaMemcpyDeviceToHost), "download f_var_bkg");
CudaCheck(cudaMemcpy(var_per_I, d.f_var_per_I.get(), bytes, cudaMemcpyDeviceToHost),
CopyAndWait(var_bkg, d.f_var_bkg.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_var_bkg");
CopyAndWait(var_per_I, d.f_var_per_I.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(),
"download f_var_per_I");
}
@@ -1440,6 +1562,91 @@ void RotationScaleMergeGPU::SetFullsCorr(const float *corr) {
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
if (d.n_fulls == 0) return;
CudaCheck(cudaMemcpy(d.f_corr.get(), corr, size_t(d.n_fulls) * sizeof(float),
cudaMemcpyHostToDevice), "upload f_corr");
CopyAndWait(d.f_corr.get(), corr, size_t(d.n_fulls) * sizeof(float),
cudaMemcpyHostToDevice, impl_->s(), "upload f_corr");
}
void RotationScaleMergeGPU::SurfaceSetTerms(int n_terms, const SurfaceTerm *term, const uint8_t *parity,
int n_groups, const int32_t *gperm, const int32_t *gstart,
int ncell) {
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
d.s_ncell = ncell;
d.s_n_groups = n_groups;
d.Upload(d.s_term, term, n_terms);
d.Upload(d.s_parity, parity, n_terms);
d.Upload(d.s_gperm, gperm, n_terms);
d.Upload(d.s_gstart, gstart, n_groups + 1);
d.s_A = d.Alloc<double>(std::max(1, ncell));
d.s_cross = d.Alloc<double>(std::max(1, ncell));
d.s_ref2 = d.Alloc<double>(std::max(1, ncell));
d.s_sw = d.Alloc<double>(std::max(1, n_groups));
d.s_swI = d.Alloc<double>(std::max(1, n_groups));
d.s_w_Is = d.Alloc<double>(std::max(1, n_terms));
d.s_w_Iref = d.Alloc<double>(std::max(1, n_terms));
d.s_Iref = d.Alloc<double>(std::max(1, n_terms));
for (int &nb : d.s_n_blocks) nb = 0;
}
void RotationScaleMergeGPU::SurfaceSetSubset(int subset, int n_blocks, const int32_t *perm,
const int32_t *seg_start) {
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
const int n_seg = n_blocks * d.s_ncell;
d.s_n_sel[subset] = n_blocks > 0 ? seg_start[n_seg] : 0;
d.Upload(d.s_perm[subset], perm, d.s_n_sel[subset]);
d.Upload(d.s_seg_start[subset], seg_start, n_seg + 1);
d.s_n_blocks[subset] = n_blocks;
if (size_t(n_seg) > d.s_slots) {
d.s_tcross = d.Alloc<double>(n_seg);
d.s_tref2 = d.Alloc<double>(n_seg);
d.s_slots = n_seg;
}
}
void RotationScaleMergeGPU::SurfaceReference(int parity, const double *A) {
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
CopyAndWait(d.s_A.get(), A, size_t(d.s_ncell) * sizeof(double), cudaMemcpyHostToDevice, impl_->s(),
"upload surface");
const int grp_blocks = std::min(65535, (d.s_n_groups + BLK - 1) / BLK);
if (grp_blocks > 0)
SurfaceReferenceKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(d.s_n_groups, parity, d.s_gperm.get(),
d.s_gstart.get(), d.s_term.get(), d.s_parity.get(), d.s_A.get(), d.s_sw.get(), d.s_swI.get());
CudaCheck(cudaGetLastError(), "surface reference launch");
CudaCheck(cudaStreamSynchronize(impl_->s()), "surface reference sync");
}
void RotationScaleMergeGPU::SurfaceGetReference(double *sw, double *swI) const {
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
const size_t bytes = size_t(d.s_n_groups) * sizeof(double);
CopyAndWait(sw, d.s_sw.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "dl surface sw");
CopyAndWait(swI, d.s_swI.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "dl surface swI");
}
void RotationScaleMergeGPU::SurfaceFitSums(int subset, double *cross, double *ref2) {
DeviceGuard guard(impl_->device, impl_->available);
auto &d = *impl_;
const int ncell = d.s_ncell, nb = d.s_n_blocks[subset], n_seg = nb * ncell;
if (nb == 0) {
std::fill(cross, cross + ncell, 0.0);
std::fill(ref2, ref2 + ncell, 0.0);
return;
}
SurfaceFitTermKernel<<<std::min(65535, (d.s_n_sel[subset] + BLK - 1) / BLK), BLK, 0, impl_->s()>>>(
d.s_n_sel[subset], d.s_perm[subset].get(), d.s_term.get(), d.s_A.get(), d.s_sw.get(), d.s_swI.get(),
d.s_w_Is.get(), d.s_w_Iref.get(), d.s_Iref.get());
CudaCheck(cudaGetLastError(), "surface fit term launch");
SurfaceFitSegmentKernel<<<std::min(65535, (n_seg + BLK - 1) / BLK), BLK, 0, impl_->s()>>>(n_seg,
d.s_seg_start[subset].get(), d.s_w_Is.get(), d.s_w_Iref.get(), d.s_Iref.get(),
d.s_tcross.get(), d.s_tref2.get());
CudaCheck(cudaGetLastError(), "surface fit segment launch");
SurfaceFitCellKernel<<<std::min(65535, (ncell + BLK - 1) / BLK), BLK, 0, impl_->s()>>>(nb, ncell,
d.s_tcross.get(), d.s_tref2.get(), d.s_cross.get(), d.s_ref2.get());
CudaCheck(cudaGetLastError(), "surface cell sum launch");
CopyAndWait(cross, d.s_cross.get(), size_t(ncell) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(),
"dl surface cross");
CopyAndWait(ref2, d.s_ref2.get(), size_t(ncell) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(),
"dl surface ref2");
}
@@ -192,6 +192,34 @@ public:
// Download the fulls' working corr (length = n_fulls), valid after ScaleFulls.
void GetFullsCorr(float *corr) const;
// --- correction-surface fit (RotationScaleMerge::ApplyCellSurface) ---
// The two passes every round of the fit makes over its terms, on the device; the host keeps the
// per-cell step, the gauge and the cross-validation. Both sums are formed in exactly the order the
// host forms them, so the fitted surface is the host's to the last bit (see the kernels).
// One observation as the surface fit sees it: the host's term, uploaded as it stands.
struct SurfaceTerm { float I, sigma, corr, d; int32_t cell, group; };
// The terms (in fulls order) with each one's frame parity, the ASU-group CSR over them (gperm lists
// the terms of group g at [gstart[g], gstart[g+1]), in fulls order) and the cell count.
void SurfaceSetTerms(int n_terms, const SurfaceTerm *term, const uint8_t *parity,
int n_groups, const int32_t *gperm, const int32_t *gstart, int ncell);
// One subset of the terms (0 = even frames, 1 = odd, 2 = all), cut into the host's n_blocks
// reduction blocks and ordered within each block by cell, keeping term order inside a cell:
// block b, cell c is perm[seg_start[b * ncell + c], seg_start[b * ncell + c + 1]).
void SurfaceSetSubset(int subset, int n_blocks, const int32_t *perm, const int32_t *seg_start);
// The per-group reference sums sw / swI over the terms of frame parity `parity` (< 0 = all) with
// the surface A (length ncell) applied. They stay on the device for SurfaceFitSums;
// SurfaceGetReference downloads them (length n_groups each).
void SurfaceReference(int parity, const double *A);
void SurfaceGetReference(double *sw, double *swI) const;
// The fit's per-cell sums over one subset against the last SurfaceReference and its A:
// cross = sum w Is Iref and ref2 = sum w Iref^2 (length ncell each).
void SurfaceFitSums(int subset, double *cross, double *ref2);
private:
struct Impl;
std::unique_ptr<Impl> impl_;
@@ -17,7 +17,7 @@ AdaptiveSpotFinderCPU::AdaptiveSpotFinderCPU(const AzimuthalIntegrationMapping &
ring_cnt.assign(nbins, 0);
ring_mean.assign(nbins, 0.0f);
ring_sigma.assign(nbins, 0.0f);
ring_thr.assign(nbins, 0.0f);
ring_thr.assign(nbins + 1, INFINITY); // the last entry is for pixels outside every ring
ring_bkg.assign(nbins, NAN);
ring_bits.assign(OutputSize(), 0);
ring_hist.assign(nbins * HIST_VALUES, 0);
@@ -54,84 +54,54 @@ void AdaptiveSpotFinderCPU::BeginRings() {
rings_from_blocks = true;
}
// The plain pass over pixels [first, first + n).
// The plain pass over pixels [first, first + n): the histogram of each ring's values, from which
// PlainRings() takes the integer sums, and the fused profile when asked for. One loop reads each pixel
// once for both.
void AdaptiveSpotFinderCPU::AccumulateRingsBlock(const ImagePreprocessorBuffer &image, size_t first, size_t n) {
const auto &pixel_to_bin = mapping.GetPixelToBin();
const size_t nbins = ring_sum.size();
const float *corrections = mapping.Corrections().data();
// Consecutive pixels mostly share a ring, so a ring's sums are held in locals while they do and
// written back when the ring changes: the same additions in the same order, without a store and a
// reload of the same address on every pixel. The azimuthal-integration sums get a loop of their own
// over the block, so that neither loop runs out of registers for its sums.
if (fuse_azint) {
size_t cur = nbins; // the ring held in the locals below; nbins = none
float az_sum = 0.0f, az_sum2 = 0.0f;
uint32_t az_cnt = 0;
for (size_t pxl = first; pxl < first + n; ++pxl) {
const int32_t v = image[pxl];
if (v == INT32_MIN || v == INT32_MAX) continue; // bad / saturated
const uint16_t b = pixel_to_bin[pxl];
if (b >= nbins) continue; // masked / out of range (UINT16_MAX)
if (b != cur) {
if (cur != nbins) {
azint_sum[cur] = az_sum;
azint_sum2[cur] = az_sum2;
azint_count[cur] = az_cnt;
}
cur = b;
az_sum = azint_sum[b];
az_sum2 = azint_sum2[b];
az_cnt = azint_count[b];
}
const float val = static_cast<float>(v) * corrections[pxl];
const float val_sq = val * val;
az_sum += val;
az_sum2 += val_sq;
++az_cnt;
}
if (cur != nbins) {
azint_sum[cur] = az_sum;
azint_sum2[cur] = az_sum2;
azint_count[cur] = az_cnt;
}
}
// Values outside the histogram are listed by a second loop, run only when the block has any: a
// call in this loop would leave the sums in memory again.
uint32_t *hist = ring_hist.data();
// Consecutive pixels mostly share a ring, so the ring's profile sums are held in locals while they
// do and written back when the ring changes: the same additions in the same order, without a store
// and a reload of the same address on every pixel. Values outside the histogram are listed by a
// second loop, run only when the block has any.
bool overflow = false;
size_t cur = nbins;
int64_t sum = 0, cnt = 0;
uint64_t sum2 = 0;
size_t cur = nbins; // the ring held in the locals below; nbins = none
float az_sum = 0.0f, az_sum2 = 0.0f;
uint32_t az_cnt = 0;
for (size_t pxl = first; pxl < first + n; ++pxl) {
const int32_t v = image[pxl];
if (v == INT32_MIN || v == INT32_MAX) continue; // bad / saturated
const uint16_t b = pixel_to_bin[pxl];
if (b >= nbins) continue; // masked / out of range (UINT16_MAX)
if (b != cur) {
if (cur != nbins) {
ring_sum[cur] = sum;
ring_sum2[cur] = sum2;
ring_cnt[cur] = cnt;
}
cur = b;
sum = ring_sum[b];
sum2 = ring_sum2[b];
cnt = ring_cnt[b];
}
sum += v;
sum2 += static_cast<uint64_t>(static_cast<int64_t>(v) * v);
cnt += 1;
if (v >= 0 && v < HIST_VALUES)
if (static_cast<uint32_t>(v) < HIST_VALUES)
hist[b * HIST_VALUES + v] += 1;
else
overflow = true;
if (!fuse_azint) continue;
if (b != cur) {
if (cur != nbins) {
azint_sum[cur] = az_sum;
azint_sum2[cur] = az_sum2;
azint_count[cur] = az_cnt;
}
cur = b;
az_sum = azint_sum[b];
az_sum2 = azint_sum2[b];
az_cnt = azint_count[b];
}
const float val = static_cast<float>(v) * corrections[pxl];
const float val_sq = val * val;
az_sum += val;
az_sum2 += val_sq;
++az_cnt;
}
if (cur != nbins) {
ring_sum[cur] = sum;
ring_sum2[cur] = sum2;
ring_cnt[cur] = cnt;
azint_sum[cur] = az_sum;
azint_sum2[cur] = az_sum2;
azint_count[cur] = az_cnt;
}
if (overflow)
@@ -146,7 +116,8 @@ void AdaptiveSpotFinderCPU::AccumulateRingsBlock(const ImagePreprocessorBuffer &
}
// A sigma-clip pass over the plain pass's values: each distinct value of a ring meets the same test the
// pixels holding it would, and its pixels are added as a count.
// pixels holding it would, and its pixels are added as a count. Integer sums, so with nothing clipped
// (clip_k = INFINITY) they are the plain sums over the pixels themselves.
void AdaptiveSpotFinderCPU::ClipRings(float clip_k) {
const size_t nbins = ring_sum.size();
@@ -155,6 +126,7 @@ void AdaptiveSpotFinderCPU::ClipRings(float clip_k) {
std::fill(ring_cnt.begin(), ring_cnt.end(), 0);
const auto keep = [&](uint16_t b, int32_t v) {
if (std::isinf(clip_k)) return true;
const float lo = ring_mean[b] - clip_k * ring_sigma[b];
const float hi = ring_mean[b] + clip_k * ring_sigma[b];
return !(v < lo || v > hi); // exclude peaks / outliers
@@ -197,6 +169,7 @@ void AdaptiveSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image,
AccumulateRingsBlock(image, 0, static_cast<size_t>(width) * height);
}
rings_from_blocks = false;
ClipRings(INFINITY);
UpdateRingStatistics();
ClipRings(3.0f);
UpdateRingStatistics();
@@ -259,19 +232,31 @@ void AdaptiveSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image,
void AdaptiveSpotFinderCPU::FlagRow(const ImagePreprocessorBuffer &image, int32_t row) {
const auto &pixel_to_bin = mapping.GetPixelToBin();
const size_t nbins = ring_thr.size();
const auto nbins = static_cast<uint32_t>(ring_thr.size() - 1); // ring_thr[nbins] is +inf
const float *thr = ring_thr.data();
const int32_t *img = image.data();
const uint16_t *bin = pixel_to_bin.data();
const size_t first = static_cast<size_t>(row) * width;
const size_t end = first + width;
for (size_t pxl = first; pxl < first + width; ++pxl) {
const int32_t v = image[pxl];
const uint16_t b = pixel_to_bin[pxl];
bool strong = false;
if (v == INT32_MAX)
strong = true;
else if (v != INT32_MIN && b < nbins && v >= ring_thr[b])
strong = true;
if (strong)
ring_bits[pxl / 32] |= 1U << (pxl % 32);
// Saturated is strong, bad is not, and a pixel outside every ring (bin >= nbins) meets the +inf
// threshold. Written without branches and a word of 32 pixels at a time, so that the loop vectorises.
const auto strong = [&](size_t pxl) -> uint32_t {
const int32_t v = img[pxl];
const uint32_t b = std::min<uint32_t>(bin[pxl], nbins);
return (v == INT32_MAX) | ((v != INT32_MIN) & (v >= thr[b]));
};
size_t pxl = first;
while (pxl < end && pxl % 32 != 0) {
ring_bits[pxl / 32] |= strong(pxl) << (pxl % 32);
++pxl;
}
for (; pxl + 32 <= end; pxl += 32) {
uint32_t word = 0;
for (uint32_t j = 0; j < 32; ++j)
word |= strong(pxl + j) << j;
ring_bits[pxl / 32] |= word;
}
for (; pxl < end; ++pxl)
ring_bits[pxl / 32] |= strong(pxl) << (pxl % 32);
}
@@ -57,7 +57,7 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU {
std::vector<int64_t> ring_cnt;
std::vector<float> ring_mean;
std::vector<float> ring_sigma;
std::vector<float> ring_thr;
std::vector<float> ring_thr; // nbins + 1: the last is +inf, the threshold of a pixel outside every ring
// ring_mean of the last Detect(), NaN where the ring holds too few pixels to be its own background.
// Kept separately because ring_mean carries the previous frame's value for an empty ring.
std::vector<float> ring_bkg;
@@ -65,8 +65,8 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU {
// local-box mask that ImageSpotFinderCPU::Detect leaves in output_buffer.
std::vector<uint32_t> ring_bits;
// The plain pass's valid pixels as a per-ring histogram of their values (HIST_VALUES bins per
// ring) plus a list of the values outside it, so the two sigma-clip passes sum over distinct
// values instead of over the image again. Integer sums, so the same totals.
// ring) plus a list of the values outside it. The plain sums and the two sigma-clip passes are all
// taken from it, over distinct values instead of over the image. Integer sums, so the same totals.
static constexpr int32_t HIST_VALUES = 1024;
std::vector<uint32_t> ring_hist;
std::vector<std::pair<uint16_t, int32_t>> ring_overflow; // (ring, value)
@@ -84,7 +84,7 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU {
// Zero the sums of the plain ring pass (and of the fused profile).
void ResetRings();
// One sigma-clip pass over the plain pass's values.
// One sigma-clip pass over the plain pass's values; clip_k = INFINITY gives the plain sums.
void ClipRings(float clip_k);
// ring_mean / ring_sigma from the current sums.
void UpdateRingStatistics();
+65 -43
View File
@@ -216,7 +216,9 @@ void HotPixelFinder::AddLevels(const std::vector<int32_t> &sector_level, const s
const int r = static_cast<int>(k / SECTORS);
level[k] = std::max(ring_level[r], sector_level[k]);
const float noise = std::max(std::sqrt(static_cast<float>(std::max(level[k], 0))), ring_spread[r]);
threshold[k] = static_cast<float>(level[k]) + LIT_NSIGMA * noise + LIT_OFFSET;
// One rounding for level + nsigma * noise and one for the offset, written out so that every
// compiler takes the same two - the device takes them too (HotPixelsGPU.cu).
threshold[k] = std::fma(LIT_NSIGMA, noise, static_cast<float>(level[k])) + LIT_OFFSET;
}
// Every sum is an integer, so the result does not depend on the order the frames arrive in.
@@ -230,50 +232,24 @@ void HotPixelFinder::AddLevels(const std::vector<int32_t> &sector_level, const s
}
#ifdef JFJOCH_USE_CUDA
void HotPixelFinder::PrepareDevice() {
std::lock_guard lock(m);
if (!gpu)
gpu = std::make_unique<HotPixelFinderGPU>(key.get(), width * height, key_begin, nrings, SECTORS,
HotPixelLevelRules{MIN_SECTOR_PIXELS, MIN_RING_PIXELS,
LIT_NSIGMA, LIT_OFFSET});
}
void HotPixelFinder::AddDeviceImage(const int32_t *device_image, HotPixelFinderGPU::Frame &frame) {
HotPixelFinderGPU *device;
{
std::lock_guard lock(m);
if (!gpu)
gpu = std::make_unique<HotPixelFinderGPU>(key.get(), width * height, key_begin, nrings, SECTORS);
device = gpu.get();
}
std::vector<uint32_t> count;
std::vector<int32_t> sector_median, ring_median, ring_mad;
device->Statistics(device_image, frame, count, sector_median, ring_median, ring_mad);
// The same levels AddImage takes off its scratch buffer, and under the same pixel minima.
const size_t nkeys = static_cast<size_t>(nrings) * SECTORS;
std::vector<int32_t> sector_level(nkeys, 0);
for (size_t k = 0; k < nkeys; k++)
if (count[k] >= MIN_SECTOR_PIXELS)
sector_level[k] = sector_median[k];
std::vector<int32_t> ring_level(nrings, 0);
std::vector<float> ring_spread(nrings, 0.0f);
std::vector<char> ring_ok(nrings, 0);
for (int r = 0; r < nrings; r++) {
size_t n = 0;
for (int s = 0; s < SECTORS; s++)
n += count[r * SECTORS + s];
if (n < MIN_RING_PIXELS) continue;
ring_ok[r] = 1;
ring_level[r] = ring_median[r];
ring_spread[r] = 1.4826f * static_cast<float>(ring_mad[r]);
}
std::vector<int32_t> level;
std::vector<float> threshold;
AddLevels(sector_level, ring_level, ring_spread, ring_ok, level, threshold);
device->Accumulate(device_image, frame, level, threshold, ring_ok);
PrepareDevice();
gpu->Add(device_image, frame);
std::lock_guard lock(m);
frames++;
}
#endif
HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double spacing_deg, size_t nthreads) {
std::lock_guard lock(m);
#ifdef JFJOCH_USE_CUDA
if (gpu)
gpu->Download(n_lit.get(), n_error.get(), sum_value.get(), n_error_ring_ok.get(), error_level_sum.get());
#endif
Result ret;
ret.frames = frames;
ret.mask.assign(width * height, 0);
@@ -281,12 +257,38 @@ HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double sp
// The chance rate per ring, from the pixels lit on no more than half of their frames: whatever
// lights those - reflections, zingers, noise above the bound - lights a defect-free pixel too.
// Counted in integers by blocks of rows in parallel, so the totals do not depend on the split.
std::vector<double> lit(nrings, 0.0), seen(nrings, 0.0);
for (size_t i = 0; i < width * height; i++)
if (key[i] >= 0 && n_valid(i) > 0 && 2 * n_lit[i] <= n_valid(i)) {
lit[key[i] / SECTORS] += n_lit[i];
seen[key[i] / SECTORS] += n_valid(i);
#ifdef JFJOCH_USE_CUDA
if (gpu) {
std::vector<int64_t> device_lit, device_seen;
gpu->ChanceCounts(device_lit, device_seen);
for (int r = 0; r < nrings; r++) {
lit[r] = static_cast<double>(device_lit[r]);
seen[r] = static_cast<double>(device_seen[r]);
}
} else
#endif
{
std::vector<std::vector<int64_t>> block_lit(BANDS), block_seen(BANDS);
const size_t rows_per_band = (height + BANDS - 1) / BANDS;
ParallelFor(static_cast<int>(BANDS), nthreads, [&](int b) {
block_lit[b].assign(nrings, 0);
block_seen[b].assign(nrings, 0);
const size_t begin = std::min(width * height, b * rows_per_band * width);
const size_t end = std::min(width * height, (b + 1) * rows_per_band * width);
for (size_t i = begin; i < end; i++)
if (key[i] >= 0 && n_valid(i) > 0 && 2 * n_lit[i] <= n_valid(i)) {
block_lit[b][key[i] / SECTORS] += n_lit[i];
block_seen[b][key[i] / SECTORS] += n_valid(i);
}
});
for (int r = 0; r < nrings; r++)
for (size_t b = 0; b < BANDS; b++) {
lit[r] += static_cast<double>(block_lit[b][r]);
seen[r] += static_cast<double>(block_seen[b][r]);
}
}
std::vector<int> k_chance(nrings, n + 1);
for (int r = 0; r < nrings; r++)
if (seen[r] > 0.0)
@@ -295,6 +297,26 @@ HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double sp
// Persistent: lit on more frames than one reflection or chance explains.
const int min_valid = std::max(10, n / 2);
#ifdef JFJOCH_USE_CUDA
// With a GPU the per-pixel sums stay there. Only the pixels that can be masked come back - those the
// tests below could pass (see GetCandidates) - into the host arrays, which are zero everywhere else,
// so the tests below run on them unchanged.
if (gpu) {
const auto c = gpu->GetCandidates(frames, min_valid, spacing_deg > 0.0, k_chance);
for (size_t j = 0; j < c.index.size(); j++) {
const size_t i = c.index[j];
n_lit[i] = c.n_lit[j];
n_error[i] = c.n_error[j];
n_error_ring_ok[i] = c.n_error_ring_ok[j];
sum_value[i] = c.sum_value[j];
error_level_sum[i] = c.error_level_sum[j];
}
for (size_t k = 0; k < key_frames.size(); k++) {
key_frames[k] = static_cast<uint16_t>(c.key_frames[k]);
key_level_sum[k] = c.key_level_sum[k];
}
}
#endif
std::vector<uint8_t> persistent(width * height, 0);
ParallelChunks(static_cast<int>(height), nthreads, [&](int y0, int y1) {
for (size_t y = y0; y < static_cast<size_t>(y1); y++)
+6 -2
View File
@@ -94,6 +94,10 @@ public:
// `scratch` above. The per-pixel sums are then kept on the device and replace the host's when the
// mask is read, so a finder is fed one way or the other, not both. Thread safe.
void AddDeviceImage(const int32_t *device_image, HotPixelFinderGPU::Frame &frame);
// Build the device half now rather than with the first device frame, so the workers do not wait on
// it one behind the other.
void PrepareDevice();
#endif
// The mask, from the frames added so far. oscillation_deg is the rotation per image and
@@ -132,11 +136,11 @@ private:
std::unique_ptr<uint16_t[]> n_error_ring_ok;
std::unique_ptr<int64_t[]> error_level_sum;
#ifdef JFJOCH_USE_CUDA
std::unique_ptr<HotPixelFinderGPU> gpu; // built by the first device frame
std::unique_ptr<HotPixelFinderGPU> gpu; // built by PrepareDevice or the first device frame
#endif
// Each ring-sector's level and lit threshold, from the frame's order statistics, and the frame's
// share of the per-key sums. The host and the device path both come through here.
// share of the per-key sums. The device does the same in HotPixelsGPU.cu, with the same roundings.
void AddLevels(const std::vector<int32_t> &sector_level, const std::vector<int32_t> &ring_level,
const std::vector<float> &ring_spread, const std::vector<char> &ring_ok,
std::vector<int32_t> &level, std::vector<float> &threshold);
+198 -47
View File
@@ -3,6 +3,8 @@
#include "HotPixelsGPU.h"
#include <cub/device/device_radix_sort.cuh>
#include "../common/JFJochException.h"
namespace {
@@ -145,6 +147,84 @@ __global__ void accumulate_kernel(const int32_t *__restrict__ image, const int32
}
}
// Each key's level and lit threshold from the frame's order statistics, and the frame's share of the
// per-key sums - HotPixelFinder::AddImage and AddLevels, step for step. The threshold is
// fma(nsigma, noise, level) + offset, the two roundings the host takes (see AddLevels).
__global__ void levels_kernel(size_t nkeys, int sectors, HotPixelLevelRules rules, const uint32_t *__restrict__ count,
const int32_t *__restrict__ sector_median, const int32_t *__restrict__ ring_median,
const int32_t *__restrict__ ring_mad, int32_t *__restrict__ level,
float *__restrict__ threshold, char *__restrict__ ring_ok,
uint32_t *__restrict__ key_frames, int64_t *__restrict__ key_level_sum) {
const size_t k = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (k >= nkeys) return;
const size_t r = k / sectors;
uint32_t n = 0;
for (int s = 0; s < sectors; s++)
n += count[r * sectors + s];
const bool ok = n >= static_cast<uint32_t>(rules.min_ring_pixels);
const int32_t ring_level = ok ? ring_median[r] : 0;
const float ring_spread = ok ? 1.4826f * static_cast<float>(ring_mad[r]) : 0.0f;
const int32_t sector_level = count[k] >= static_cast<uint32_t>(rules.min_sector_pixels) ? sector_median[k] : 0;
const int32_t lv = max(ring_level, sector_level);
const float root = sqrtf(static_cast<float>(max(lv, 0)));
const float noise = root < ring_spread ? ring_spread : root;
level[k] = lv;
threshold[k] = __fadd_rn(__fmaf_rn(rules.lit_nsigma, noise, static_cast<float>(lv)), rules.lit_offset);
if (k % sectors == 0)
ring_ok[r] = ok;
if (ok) {
atomicAdd(&key_frames[k], 1u);
atomicAdd(reinterpret_cast<unsigned long long *>(&key_level_sum[k]),
static_cast<unsigned long long>(static_cast<int64_t>(lv)));
}
}
__device__ int valid_frames(const int32_t *key, const uint32_t *key_frames, const uint16_t *n_error_ring_ok, size_t i) {
return static_cast<int>(key_frames[key[i]]) - static_cast<int>(n_error_ring_ok[i]);
}
__global__ void chance_kernel(size_t npixels, int sectors, const int32_t *__restrict__ key,
const uint32_t *__restrict__ key_frames, const uint16_t *__restrict__ n_lit,
const uint16_t *__restrict__ n_error_ring_ok, unsigned long long *__restrict__ lit,
unsigned long long *__restrict__ seen) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= npixels || key[i] < 0) return;
const int nv = valid_frames(key, key_frames, n_error_ring_ok, i);
if (nv > 0 && 2 * static_cast<int>(n_lit[i]) <= nv) {
atomicAdd(&lit[key[i] / sectors], static_cast<unsigned long long>(n_lit[i]));
atomicAdd(&seen[key[i] / sectors], static_cast<unsigned long long>(nv));
}
}
// Writes the candidates' indices from `out` on when `out` is given, and counts them either way.
__global__ void candidate_kernel(size_t npixels, int sectors, uint32_t frames, int min_valid, bool spacing_ok,
const int32_t *__restrict__ key, const uint32_t *__restrict__ key_frames,
const uint16_t *__restrict__ n_lit, const uint16_t *__restrict__ n_error,
const uint16_t *__restrict__ n_error_ring_ok, const int *__restrict__ k_chance,
uint32_t *__restrict__ count, uint32_t *__restrict__ out) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= npixels || key[i] < 0) return;
const int nv = valid_frames(key, key_frames, n_error_ring_ok, i);
const bool error = 2 * static_cast<uint32_t>(n_error[i]) > frames;
const int lit = n_lit[i];
const bool persistent = spacing_ok && nv >= min_valid && lit > 0
&& lit >= min(max(2, k_chance[key[i] / sectors]), nv);
if (!error && !persistent) return;
const uint32_t slot = atomicAdd(count, 1u);
if (out) out[slot] = static_cast<uint32_t>(i);
}
__global__ void iota_kernel(size_t n, uint32_t *__restrict__ out) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i < n) out[i] = static_cast<uint32_t>(i);
}
template <class T>
__global__ void gather_kernel(size_t n, const uint32_t *__restrict__ index, const T *__restrict__ in, T *__restrict__ out) {
const size_t j = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (j < n) out[j] = in[index[j]];
}
// The shared tables and sums are filled on a stream of their own, added to on the workers' streams
// and downloaded on the NULL stream, so they are allocated synchronously rather than from the pool:
// a pooled buffer is freed on the thread's allocation stream, which none of those is ordered before.
@@ -156,22 +236,18 @@ constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous;
} // namespace
HotPixelFinderGPU::HotPixelFinderGPU(const int32_t *host_key, size_t npixels,
const std::vector<uint32_t> &host_key_begin, int nrings, int sectors)
const std::vector<uint32_t> &host_key_begin, int nrings, int sectors,
const HotPixelLevelRules &rules)
: npixels(npixels), nkeys(static_cast<size_t>(nrings) * sectors), nrings(nrings),
sectors(sectors),
key(npixels, ALLOC), pixels_by_key(host_key_begin.back(), ALLOC), key_begin(host_key_begin.size(), ALLOC),
sectors(sectors), rules(rules),
key(npixels, ALLOC), pixels_by_key(std::max<size_t>(host_key_begin.back(), 1), ALLOC),
key_begin(host_key_begin.size(), ALLOC),
n_lit(npixels, ALLOC), n_error(npixels, ALLOC), n_error_ring_ok(npixels, ALLOC),
sum_value(npixels, ALLOC), error_level_sum(npixels, ALLOC) {
std::vector<uint32_t> pixels(host_key_begin.back());
std::vector<uint32_t> filled(host_key_begin.begin(), host_key_begin.end() - 1);
for (size_t i = 0; i < npixels; i++)
if (host_key[i] >= 0)
pixels[filled[host_key[i]]++] = static_cast<uint32_t>(i);
sum_value(npixels, ALLOC), error_level_sum(npixels, ALLOC),
key_frames(std::max<size_t>(nkeys, 1), ALLOC), key_level_sum(std::max<size_t>(nkeys, 1), ALLOC) {
cuda_err(cudaEventCreateWithFlags(&last_accumulate, cudaEventDisableTiming));
CudaStream stream;
cuda_err(cudaMemcpyAsync(key, host_key, npixels * sizeof(int32_t), cudaMemcpyHostToDevice, stream));
cuda_err(cudaMemcpyAsync(pixels_by_key, pixels.data(), pixels.size() * sizeof(uint32_t),
cudaMemcpyHostToDevice, stream));
cuda_err(cudaMemcpyAsync(key_begin, host_key_begin.data(), host_key_begin.size() * sizeof(uint32_t),
cudaMemcpyHostToDevice, stream));
cuda_err(cudaMemsetAsync(n_lit, 0, npixels * sizeof(uint16_t), stream));
@@ -179,16 +255,34 @@ HotPixelFinderGPU::HotPixelFinderGPU(const int32_t *host_key, size_t npixels,
cuda_err(cudaMemsetAsync(n_error_ring_ok, 0, npixels * sizeof(uint16_t), stream));
cuda_err(cudaMemsetAsync(sum_value, 0, npixels * sizeof(int64_t), stream));
cuda_err(cudaMemsetAsync(error_level_sum, 0, npixels * sizeof(int64_t), stream));
cuda_err(cudaStreamSynchronize(stream));
cuda_err(cudaMemsetAsync(key_frames, 0, std::max<size_t>(nkeys, 1) * sizeof(uint32_t), stream));
cuda_err(cudaMemsetAsync(key_level_sum, 0, std::max<size_t>(nkeys, 1) * sizeof(int64_t), stream));
// The unmasked pixels grouped by key: a stable sort of the pixel indices by key, so each key's
// pixels stay in pixel order. A masked pixel's key, -1, is the largest as unsigned and sorts last.
{
CudaDevicePtr<uint32_t> index(npixels, ALLOC), sorted_key(npixels, ALLOC), sorted_index(npixels, ALLOC);
iota_kernel<<<static_cast<unsigned>((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>(npixels, index);
cuda_err(cudaGetLastError());
const auto *keys_in = reinterpret_cast<const uint32_t *>(key.get());
size_t bytes = 0;
cuda_err(cub::DeviceRadixSort::SortPairs(nullptr, bytes, keys_in, sorted_key.get(), index.get(),
sorted_index.get(), npixels, 0, 32, stream));
CudaDevicePtr<uint8_t> scratch(bytes, ALLOC);
cuda_err(cub::DeviceRadixSort::SortPairs(scratch.get(), bytes, keys_in, sorted_key.get(), index.get(),
sorted_index.get(), npixels, 0, 32, stream));
if (host_key_begin.back() > 0)
cuda_err(cudaMemcpyAsync(pixels_by_key, sorted_index, host_key_begin.back() * sizeof(uint32_t),
cudaMemcpyDeviceToDevice, stream));
cuda_err(cudaStreamSynchronize(stream));
}
}
void HotPixelFinderGPU::Statistics(const int32_t *device_image, Frame &frame, std::vector<uint32_t> &count,
std::vector<int32_t> &sector_median, std::vector<int32_t> &ring_median,
std::vector<int32_t> &ring_mad) {
count.resize(nkeys);
sector_median.resize(nkeys);
ring_median.resize(nrings);
ring_mad.resize(nrings);
HotPixelFinderGPU::~HotPixelFinderGPU() {
if (last_accumulate) cudaEventDestroy(last_accumulate);
}
void HotPixelFinderGPU::Add(const int32_t *device_image, Frame &frame) {
if (nrings == 0)
return;
if (!frame.count.get()) {
@@ -201,44 +295,101 @@ void HotPixelFinderGPU::Statistics(const int32_t *device_image, Frame &frame, st
frame.ring_ok = CudaDevicePtr<char>(nrings);
}
const cudaStream_t stream = *frame.stream;
sector_kernel<<<static_cast<unsigned>(nkeys), THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin, frame.count,
frame.sector_median);
sector_kernel<<<static_cast<unsigned>(nkeys), THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin,
frame.count, frame.sector_median);
cuda_err(cudaGetLastError());
ring_kernel<<<nrings, THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin, frame.count, sectors,
frame.ring_median, frame.ring_mad);
cuda_err(cudaGetLastError());
cuda_err(cudaMemcpyAsync(count.data(), frame.count, nkeys * sizeof(uint32_t), cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(sector_median.data(), frame.sector_median, nkeys * sizeof(int32_t),
cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(ring_median.data(), frame.ring_median, nrings * sizeof(int32_t),
cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(ring_mad.data(), frame.ring_mad, nrings * sizeof(int32_t),
cudaMemcpyDeviceToHost, stream));
cuda_err(cudaStreamSynchronize(stream));
}
void HotPixelFinderGPU::Accumulate(const int32_t *device_image, Frame &frame, const std::vector<int32_t> &level,
const std::vector<float> &threshold, const std::vector<char> &ring_ok) {
const cudaStream_t stream = *frame.stream;
cuda_err(cudaMemcpyAsync(frame.level, level.data(), nkeys * sizeof(int32_t), cudaMemcpyHostToDevice, stream));
cuda_err(cudaMemcpyAsync(frame.threshold, threshold.data(), nkeys * sizeof(float), cudaMemcpyHostToDevice,
stream));
cuda_err(cudaMemcpyAsync(frame.ring_ok, ring_ok.data(), nrings * sizeof(char), cudaMemcpyHostToDevice, stream));
levels_kernel<<<static_cast<unsigned>((nkeys + THREADS - 1) / THREADS), THREADS, 0, stream>>>(
nkeys, sectors, rules, frame.count, frame.sector_median, frame.ring_median, frame.ring_mad,
frame.level, frame.threshold, frame.ring_ok, key_frames, key_level_sum);
cuda_err(cudaGetLastError());
std::lock_guard lock(accumulate_mutex);
cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0));
accumulate_kernel<<<static_cast<unsigned>((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>(
device_image, key, npixels, sectors, frame.level, frame.threshold, frame.ring_ok,
n_lit, n_error, sum_value, n_error_ring_ok, error_level_sum);
cuda_err(cudaGetLastError());
cuda_err(cudaEventRecord(last_accumulate, stream));
}
void HotPixelFinderGPU::ChanceCounts(std::vector<int64_t> &lit, std::vector<int64_t> &seen) {
CudaStream stream;
cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0));
const size_t rings = std::max(nrings, 1);
CudaDevicePtr<unsigned long long> d_lit(rings, ALLOC), d_seen(rings, ALLOC);
cuda_err(cudaMemsetAsync(d_lit, 0, rings * sizeof(unsigned long long), stream));
cuda_err(cudaMemsetAsync(d_seen, 0, rings * sizeof(unsigned long long), stream));
chance_kernel<<<static_cast<unsigned>((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>(
npixels, sectors, key, key_frames, n_lit, n_error_ring_ok, d_lit, d_seen);
cuda_err(cudaGetLastError());
lit.resize(nrings);
seen.resize(nrings);
cuda_err(cudaMemcpyAsync(lit.data(), d_lit, nrings * sizeof(int64_t), cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(seen.data(), d_seen, nrings * sizeof(int64_t), cudaMemcpyDeviceToHost, stream));
cuda_err(cudaStreamSynchronize(stream));
}
void HotPixelFinderGPU::Download(uint16_t *host_n_lit, uint16_t *host_n_error, int64_t *host_sum_value,
uint16_t *host_n_error_ring_ok, int64_t *host_error_level_sum) {
cuda_err(cudaMemcpy(host_n_lit, n_lit, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
cuda_err(cudaMemcpy(host_n_error, n_error, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
cuda_err(cudaMemcpy(host_sum_value, sum_value, npixels * sizeof(int64_t), cudaMemcpyDeviceToHost));
cuda_err(cudaMemcpy(host_n_error_ring_ok, n_error_ring_ok, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
cuda_err(cudaMemcpy(host_error_level_sum, error_level_sum, npixels * sizeof(int64_t), cudaMemcpyDeviceToHost));
HotPixelFinderGPU::Candidates HotPixelFinderGPU::GetCandidates(uint32_t frames, int min_valid, bool spacing_ok,
const std::vector<int> &k_chance) {
CudaStream stream;
cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0));
const unsigned blocks = static_cast<unsigned>((npixels + THREADS - 1) / THREADS);
CudaDevicePtr<int> d_k_chance(std::max<size_t>(k_chance.size(), 1), ALLOC);
CudaDevicePtr<uint32_t> d_count(1, ALLOC);
if (!k_chance.empty())
cuda_err(cudaMemcpyAsync(d_k_chance, k_chance.data(), k_chance.size() * sizeof(int), cudaMemcpyHostToDevice,
stream));
cuda_err(cudaMemsetAsync(d_count, 0, sizeof(uint32_t), stream));
candidate_kernel<<<blocks, THREADS, 0, stream>>>(npixels, sectors, frames, min_valid, spacing_ok, key, key_frames,
n_lit, n_error, n_error_ring_ok, d_k_chance, d_count, nullptr);
cuda_err(cudaGetLastError());
uint32_t n = 0;
cuda_err(cudaMemcpyAsync(&n, d_count, sizeof(uint32_t), cudaMemcpyDeviceToHost, stream));
cuda_err(cudaStreamSynchronize(stream));
Candidates c;
c.key_frames.resize(nkeys);
c.key_level_sum.resize(nkeys);
if (nkeys > 0) {
cuda_err(cudaMemcpyAsync(c.key_frames.data(), key_frames, nkeys * sizeof(uint32_t), cudaMemcpyDeviceToHost,
stream));
cuda_err(cudaMemcpyAsync(c.key_level_sum.data(), key_level_sum, nkeys * sizeof(int64_t),
cudaMemcpyDeviceToHost, stream));
}
if (n > 0) {
CudaDevicePtr<uint32_t> index(n, ALLOC);
CudaDevicePtr<uint16_t> g_lit(n, ALLOC), g_error(n, ALLOC), g_error_ring_ok(n, ALLOC);
CudaDevicePtr<int64_t> g_sum(n, ALLOC), g_error_level(n, ALLOC);
cuda_err(cudaMemsetAsync(d_count, 0, sizeof(uint32_t), stream));
candidate_kernel<<<blocks, THREADS, 0, stream>>>(npixels, sectors, frames, min_valid, spacing_ok, key,
key_frames, n_lit, n_error, n_error_ring_ok, d_k_chance,
d_count, index);
cuda_err(cudaGetLastError());
const unsigned gb = (n + THREADS - 1) / THREADS;
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, n_lit.get(), g_lit.get());
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, n_error.get(), g_error.get());
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, n_error_ring_ok.get(), g_error_ring_ok.get());
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, sum_value.get(), g_sum.get());
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, error_level_sum.get(), g_error_level.get());
cuda_err(cudaGetLastError());
c.index.resize(n);
c.n_lit.resize(n);
c.n_error.resize(n);
c.n_error_ring_ok.resize(n);
c.sum_value.resize(n);
c.error_level_sum.resize(n);
cuda_err(cudaMemcpyAsync(c.index.data(), index, n * sizeof(uint32_t), cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(c.n_lit.data(), g_lit, n * sizeof(uint16_t), cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(c.n_error.data(), g_error, n * sizeof(uint16_t), cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(c.n_error_ring_ok.data(), g_error_ring_ok, n * sizeof(uint16_t),
cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(c.sum_value.data(), g_sum, n * sizeof(int64_t), cudaMemcpyDeviceToHost, stream));
cuda_err(cudaMemcpyAsync(c.error_level_sum.data(), g_error_level, n * sizeof(int64_t),
cudaMemcpyDeviceToHost, stream));
}
cuda_err(cudaStreamSynchronize(stream));
return c;
}
+54 -23
View File
@@ -10,33 +10,53 @@
#include "../image_analysis/indexing/CUDAMemHelpers.h"
// The device half of HotPixelFinder, for frames already preprocessed on the GPU: the per-frame order
// statistics (each ring-sector's median, each ring's median and median absolute deviation) and the
// per-pixel sums run where the image already is, and only the per-key statistics - tens of thousands
// of numbers - come to the host, which turns them into levels and thresholds with the very code the
// host path uses. Every statistic is an exact order statistic of integers and every sum an integer,
// so the sums, and with them the mask, are identical to what HotPixelFinder::AddImage produces.
// What a frame's levels and lit thresholds are made of (HotPixelFinder's constants), handed to the
// device so the two halves cannot drift apart.
struct HotPixelLevelRules {
int min_sector_pixels;
int min_ring_pixels;
float lit_nsigma;
float lit_offset;
};
// The device half of HotPixelFinder, for frames already preprocessed on the GPU. Everything a frame adds
// stays on the device: the per-frame order statistics (each ring-sector's median, each ring's median
// and median absolute deviation), the levels and thresholds made from them, and the per-pixel and
// per-key sums. Every statistic is an exact order statistic of integers, the threshold is computed
// with the same rounding steps as the host's (see HotPixelFinder::AddLevels) and every sum is an
// integer, so the sums are identical to what HotPixelFinder::AddImage produces. When the mask is read
// only what can decide it comes back: per-ring counts for the chance rate, then the few pixels that can
// be persistent or carry the error value.
class HotPixelFinderGPU {
const size_t npixels;
const size_t nkeys;
const int nrings;
const int sectors;
const HotPixelLevelRules rules;
CudaDevicePtr<int32_t> key; // ring * sectors + sector of each pixel, -1 masked
CudaDevicePtr<uint32_t> pixels_by_key; // the unmasked pixels, grouped by key
CudaDevicePtr<uint32_t> pixels_by_key; // the unmasked pixels, grouped by key, in pixel order
CudaDevicePtr<uint32_t> key_begin; // where each key's pixels start in pixels_by_key
// The per-pixel sums, exactly those of HotPixelFinder.
// The per-pixel and per-key sums, exactly those of HotPixelFinder.
CudaDevicePtr<uint16_t> n_lit, n_error, n_error_ring_ok;
CudaDevicePtr<int64_t> sum_value, error_level_sum;
// Frames are selected on their workers' streams in parallel, but each pixel's sums are plain
// read-modify-writes, so one frame at a time adds to them.
CudaDevicePtr<uint32_t> key_frames;
CudaDevicePtr<int64_t> key_level_sum;
// Each pixel's sums are plain read-modify-writes, so the frames add to them one after another:
// every accumulation waits on the device for the one before it (an event, not the host).
std::mutex accumulate_mutex;
cudaEvent_t last_accumulate = nullptr;
public:
// One worker's buffers, on the stream its frames are preprocessed on.
struct Frame {
explicit Frame(std::shared_ptr<CudaStream> stream) : stream(std::move(stream)) {}
// Its frames are queued, not waited for (Add), so the buffers below must not be freed before
// the stream has used them.
~Frame() { if (stream) cudaStreamSynchronize(*stream); }
Frame(Frame &&) = default;
std::shared_ptr<CudaStream> stream;
CudaDevicePtr<uint32_t> count; // valid pixels per key
CudaDevicePtr<int32_t> sector_median; // per key
@@ -46,21 +66,32 @@ public:
CudaDevicePtr<char> ring_ok;
};
// The keys of HotPixelFinder: one per pixel, and where each key's pixels start among the unmasked
// pixels sorted by key.
HotPixelFinderGPU(const int32_t *key, size_t npixels, const std::vector<uint32_t> &key_begin, int nrings,
int sectors);
int sectors, const HotPixelLevelRules &rules);
~HotPixelFinderGPU();
HotPixelFinderGPU(const HotPixelFinderGPU &) = delete;
HotPixelFinderGPU &operator=(const HotPixelFinderGPU &) = delete;
// The lower median of the valid values of each key (count[k] of them, 0 where there are none), and
// of each ring the median and the lower median of the absolute deviations from it.
void Statistics(const int32_t *device_image, Frame &frame, std::vector<uint32_t> &count,
std::vector<int32_t> &sector_median, std::vector<int32_t> &ring_median,
std::vector<int32_t> &ring_mad);
// Add one frame, preprocessed on frame.stream, as HotPixelFinder::AddImage does. Queued on that
// stream; nothing is waited for on the host.
void Add(const int32_t *device_image, Frame &frame);
// Add the frame to the per-pixel sums, with each key's level and lit threshold and each ring's
// verdict on whether it has a level at all - as HotPixelFinder::AddImage does.
void Accumulate(const int32_t *device_image, Frame &frame, const std::vector<int32_t> &level,
const std::vector<float> &threshold, const std::vector<char> &ring_ok);
// Per ring, over the pixels lit on no more than half of their valid frames: the lit frames and the
// valid frames, summed (HotPixelFinder::GetMask's chance rate). Waits for every frame added.
void ChanceCounts(std::vector<int64_t> &lit, std::vector<int64_t> &seen);
// The per-pixel sums, npixels each.
void Download(uint16_t *n_lit, uint16_t *n_error, int64_t *sum_value, uint16_t *n_error_ring_ok,
int64_t *error_level_sum);
// The pixels that can be masked: those holding the error value on more than half of `frames`, and,
// where spacing_ok, those lit on at least min(max(2, k_chance[ring]), valid frames) of at least
// min_valid valid frames - a persistent pixel is lit on at least that many, because one reflection
// explains at least two. For each, its index and per-pixel sums; and every key's sums.
struct Candidates {
std::vector<uint32_t> index;
std::vector<uint16_t> n_lit, n_error, n_error_ring_ok;
std::vector<int64_t> sum_value, error_level_sum;
std::vector<uint32_t> key_frames;
std::vector<int64_t> key_level_sum;
};
Candidates GetCandidates(uint32_t frames, int min_valid, bool spacing_ok, const std::vector<int> &k_chance);
};
+84 -28
View File
@@ -57,6 +57,7 @@
#ifdef JFJOCH_USE_CUDA
#include "../image_analysis/image_preprocessing/ImagePreprocessorGPU.h"
#include "../image_analysis/image_preprocessing/ImagePreprocessorBufferGPU.h"
#include "../image_analysis/spot_finding/AdaptiveSpotFinderGPU.h"
#endif
#include "../image_analysis/scale_merge/Merge.h"
#include "../image_analysis/scale_merge/RfreeFlags.h"
@@ -244,10 +245,10 @@ namespace {
// as signal. The margin is capped at a tenth of the sweep so a short run still has a sample.
constexpr int PRESCAN_END_MARGIN_IMAGES = 5;
// Workers reading the pre-scan sample. Each owns a shard of the beam-stop projection so no two
// threads touch the same accumulator, and a shard costs 20 bytes per pixel - 362 MB on a 16M
// detector - so this is capped well below the worker count of the run proper. The accumulation
// is memory-bound rather than compute-bound, so a handful of workers already saturates it.
// Workers reading the pre-scan sample. Each holds detector-sized buffers of its own, and pages of
// fresh memory are slow to fault in when many threads do it at once, so this is capped well below
// the worker count of the run proper. The beam-stop projection is memory-bound rather than
// compute-bound, so a handful of workers already saturates it.
constexpr size_t PRESCAN_MAX_WORKERS = 8;
// The spot width is measured on a GROWING share of the pre-scan sample: every eighth frame of
@@ -929,12 +930,32 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
struct PreScanWorker {
std::vector<uint8_t> decompression_buffer;
JFJochReaderRawImage raw_image;
std::unique_ptr<ImagePreprocessorCPU> preprocessor;
std::unique_ptr<ImagePreprocessor> preprocessor;
std::unique_ptr<ImagePreprocessorBuffer> preprocessed;
std::unique_ptr<ImageSpotFinder> spot_finder;
};
const auto make_worker = [&] {
PreScanWorker w;
#ifdef JFJOCH_USE_CUDA
// On the card where there is one, as in the image loops: the frame is decoded and preprocessed
// there and the spots are found and extracted there. The device finders give the host's spot
// list to the bit - integer ring sums, the same connected components in the same order
// (AdaptiveSpotFinderGPU, SpotExtractorGPU) - so this is a choice of where, not of what. The
// preprocessed image comes back only for the spot width, which reads pixels around the spots.
if (want_spots && get_gpu_count() > 0) {
auto stream = std::make_shared<CudaStream>();
w.preprocessor = std::make_unique<ImagePreprocessorGPU>(prescan_x, prescan_mask, stream,
/*copy_image_to_host=*/want_width);
w.preprocessed = std::make_unique<ImagePreprocessorBufferGPU>(prescan_x.GetPixelsNum(),
/*host_mirror=*/want_width);
if (config_.spot_finding.adaptive_threshold)
w.spot_finder = std::make_unique<AdaptiveSpotFinderGPU>(*prescan_mapping, stream);
else
w.spot_finder = std::make_unique<ImageSpotFinderGPU>(prescan_x.GetXPixelsNumConv(),
prescan_x.GetYPixelsNumConv(), stream);
return w;
}
#endif
if (want_spots) {
w.preprocessor = std::make_unique<ImagePreprocessorCPU>(prescan_x, prescan_mask);
w.preprocessed = std::make_unique<ImagePreprocessorBuffer>(prescan_x.GetPixelsNum());
@@ -956,8 +977,20 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
bool for_width, std::vector<spot_width::FluxCurve> &curves,
std::vector<float> &spot_q) {
try {
w.preprocessor->Analyze(*w.preprocessed,
image.GetUncompressedPtr(w.decompression_buffer), image.GetMode());
// As in the image loops: a frame the device cannot decode goes to the host decoder.
ImageStatistics stats;
bool decoded_on_device = false;
try {
decoded_on_device = w.preprocessor->AnalyzeCompressed(*w.preprocessed, image, stats);
} catch (const JFJochException &e) {
cuda_throw_if_context_lost();
logger.Warning("Pre-scan: device decoding of image {} failed ({}), decompressing it on "
"the host", image_idx, e.what());
cuda_clear_error();
}
if (!decoded_on_device)
w.preprocessor->Analyze(*w.preprocessed,
image.GetUncompressedPtr(w.decompression_buffer), image.GetMode());
} catch (const std::exception &e) {
if (IsFatalResourceError(e)) throw;
logger.Warning("Pre-scan: failed to preprocess image {}: {}", image_idx, e.what());
@@ -1014,9 +1047,9 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
// Read the sample on several workers. The reader serialises on the HDF5 lock, but the
// decompression, the projection and the spot finding - which is all of the cost on a large
// detector - run in parallel. Each worker accumulates into a shard of its own, so nothing is
// locked while an image is added, and the per-frame results are stitched together in sample
// order below so the beam centre sees the same input however the workers interleaved.
// detector - run in parallel. The projection is integer sums, so the order the workers add their
// frames in does not reach it, and the per-frame results are stitched together in sample order
// below so the beam centre sees the same input however the workers interleaved.
//
// Two passes over the sample. The first builds the projection, which is all the shadow, the
// defective pixels and the beam-centre capture below read; the second finds the spots, for the
@@ -1026,13 +1059,12 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
const std::vector<int> ordinals(sample.begin(), sample.end());
const size_t nworkers = std::min<size_t>(std::max<size_t>(config_.nthreads, 1),
std::min(PRESCAN_MAX_WORKERS, ordinals.size()));
finder.SetShardCount(nworkers);
{
std::atomic<size_t> next{0};
std::vector<std::future<void>> futures;
futures.reserve(nworkers);
for (size_t t = 0; t < nworkers; t++)
futures.emplace_back(std::async(std::launch::async, [&, t] {
futures.emplace_back(std::async(std::launch::async, [&] {
std::vector<uint8_t> shadow_buffer;
JFJochReaderRawImage raw_image;
for (size_t i = next.fetch_add(1); i < ordinals.size(); i = next.fetch_add(1)) {
@@ -1057,7 +1089,7 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
msg.image = raw_image.image;
msg.number = ordinal;
msg.original_number = image_idx;
finder.AddImage(msg, shadow_buffer, t);
finder.AddImage(msg, shadow_buffer);
}
}
}));
@@ -1670,6 +1702,11 @@ void Rugnux::MaskDefectivePixels(int start_image, const std::vector<int> &sample
const size_t nthreads = static_cast<size_t>(std::max(config_.nthreads, 1));
HotPixelFinder finder(about, pixel_mask_, nthreads);
#ifdef JFJOCH_USE_CUDA
// Built here, once, so the workers' first frames do not queue behind it.
if (get_gpu_count() > 0)
finder.PrepareDevice();
#endif
std::atomic<size_t> next{0};
std::vector<std::future<void>> futures;
const size_t nworkers = std::min<size_t>(nthreads, std::min(PRESCAN_MAX_WORKERS, sample.size()));
@@ -8389,6 +8426,27 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
const auto &twin_sg_opt = experiment_.GetGemmiSpaceGroup();
const gemmi::SpaceGroup *twin_sg = twin_sg_opt ? &*twin_sg_opt : nullptr;
// Diffraction anisotropy (see where it is reported, below), made beside the analyses that come
// before it: it reads the merge, the integrated observations and the per-frame scales the merge
// wrote back, none of which they change, and most of it is gathering the observations.
std::future<decltype(sm.statistics.anisotropy)> anisotropy;
if (!geometry_prepass && !superseded && result.consensus_cell) {
AnisotropyRunInfo aniso_run;
if (sm.statistics.sweep_quality.measured && sm.statistics.sweep_quality.sweep_deg > 0.0f)
aniso_run.observed_rotation_deg = sm.statistics.sweep_quality.sweep_deg;
aniso_run.dose_term_in_scale_model = experiment_.GetScalingSettings().GetCorrectionSurfaces();
aniso_run.radiation_damage_relative_b = sm.statistics.radiation_damage_delta_b;
const float wedge_deg = experiment_.GetGoniometer() ? experiment_.GetGoniometer()->GetWedge_deg() : 0.0f;
anisotropy = std::async(std::launch::async,
[&, aniso_run, wedge_deg, cell = *result.consensus_cell,
rotation = experiment_.IsRotationIndexing()] {
return AnalyzeAnisotropy(sm.merged,
ScaledObservations(indexer->GetIntegrationOutcome(), rotation, twin_sg,
wedge_deg, 0.5, config_.nthreads),
cell, twin_sg, aniso_run);
});
}
// Not on the geometry pre-pass, nor on a superseded one: the analysis goes into that pass's
// statistics text and its written reflections, and neither survives the run. The promotion flag
// below is a different thing - it is what the SEARCH did, the second pass reads it, and it is
@@ -8619,21 +8677,8 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
// unmerged observations, because a merge has exact Laue symmetry by construction and the
// tensor directions the symmetry forbids - the only place a dataset measures its own
// systematic error - are identically zero in it.
if (result.consensus_cell) {
AnisotropyRunInfo aniso_run;
if (sm.statistics.sweep_quality.measured && sm.statistics.sweep_quality.sweep_deg > 0.0f)
aniso_run.observed_rotation_deg = sm.statistics.sweep_quality.sweep_deg;
aniso_run.dose_term_in_scale_model =
experiment_.GetScalingSettings().GetCorrectionSurfaces();
aniso_run.radiation_damage_relative_b = sm.statistics.radiation_damage_delta_b;
sm.statistics.anisotropy = AnalyzeAnisotropy(
sm.merged,
ScaledObservations(indexer->GetIntegrationOutcome(),
experiment_.IsRotationIndexing(), twin_sg,
experiment_.GetGoniometer()
? experiment_.GetGoniometer()->GetWedge_deg() : 0.0f,
0.5, config_.nthreads),
*result.consensus_cell, twin_sg, aniso_run);
if (anisotropy.valid()) {
sm.statistics.anisotropy = anisotropy.get();
stats_text << AnisotropyToText(sm.statistics.anisotropy) << "\n";
}
}
@@ -8869,8 +8914,10 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
Logger held = Logger::Buffered();
auto pending = std::async(std::launch::async, validate, std::ref(held));
experiment_.SpaceGroupNumber(1);
rsm->SetWriteBackPerFrameScale(false);
p1_merged_early = rsm->Run(/*for_search=*/false, /*full_stats=*/true,
/*measure_cc_before_corrections=*/false);
rsm->SetWriteBackPerFrameScale(true);
experiment_.SetSpaceGroup(data_sg);
validation = pending.get();
held.ReplayInto(logger);
@@ -9046,6 +9093,9 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
const double em_a = result.error_model_a;
const double em_b = result.error_model_b;
const auto res_fit = result.resolution_fit_A;
const int iter_partials = result.scaling_iterations_partials;
const int iter_fulls = result.scaling_iterations_fulls;
const bool converged = result.scaling_converged;
// The unmerged MTZ below does not depend on this merge, so it is built meanwhile, from
// the experiment as it stands in the determined group. The merge rewrites each image's
// mosaicity, which the file's batch headers carry, so that is filled in only after it.
@@ -9062,7 +9112,10 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
// Both the merge and the MTZ read the group from the experiment, so it is set for
// the whole of it and restored after.
experiment_.SpaceGroupNumber(1);
if (!p1_merged_early)
rsm->SetWriteBackPerFrameScale(false);
auto p1 = scale_and_merge("P1 cross-check", false, false, std::move(p1_merged_early));
rsm->SetWriteBackPerFrameScale(true);
// The scaler still holds the observations in the indexing they were merged in, so every
// relabelling since (the written setting, the model's indexing) is applied to this merge
// too - or it would describe the dataset on other axes than the merged output beside it,
@@ -9135,6 +9188,9 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
result.error_model_a = em_a;
result.error_model_b = em_b;
result.resolution_fit_A = res_fit;
result.scaling_iterations_partials = iter_partials;
result.scaling_iterations_fulls = iter_fulls;
result.scaling_converged = converged;
if (determined != nullptr && determined->number > 1)
logger.Info("P1 cross-check dataset written to {} ({} unique reflections): the "
"same observations merged in P1 instead of {}, so a wrong space group "
+41
View File
@@ -12,6 +12,7 @@
#include "../common/AzimuthalIntegrationMapping.h"
#include "../common/AzimuthalIntegrationProfile.h"
#include "../image_analysis/azint/AzIntEngineCPU.h"
#include "../image_analysis/azint/AzIntEngineGPU.h"
#include "../image_analysis/spot_finding/AdaptiveSpotFinderCPU.h"
#include "../image_analysis/spot_finding/AdaptiveSpotFinderGPU.h"
@@ -161,6 +162,46 @@ TEST_CASE("AdaptiveSpotFinderGPU_AzimuthalIntegration", "[AdaptiveSpotFinderGPU]
}
}
// The GPU azimuthal integration against the CPU one, on a pixel count that is not a multiple of four (the
// kernel reads four pixels at a time and does the rest one by one) and with masked and saturated pixels
// in it. The per-ring pixel counts are integers and must agree exactly; the float sums only to rounding.
TEST_CASE("AzIntEngineGPU_MatchesCPU", "[AdaptiveSpotFinderGPU]") {
if (get_gpu_count() == 0) {
WARN("No CUDA GPU present. Skipping AzIntEngineGPU_MatchesCPU");
return;
}
DiffractionExperiment x(DetDECTRIS(1031, 1063, "Test", {}));
x.DetectorDistance_mm(80).BeamX_pxl(515).BeamY_pxl(530);
x.QSpacingForAzimInt_recipA(0.05).QRangeForAzimInt_recipA(0.05, 5.0);
REQUIRE(x.GetPixelsNum() % 4 != 0);
PixelMask pixel_mask(x);
AzimuthalIntegrationMapping mapping(x, pixel_mask);
ImagePreprocessorBufferGPU buffer(x.GetPixelsNum());
for (size_t i = 0; i < x.GetPixelsNum(); i++)
buffer[i] = (i % 997 == 0) ? INT32_MIN : (i % 1009 == 0) ? INT32_MAX
: 8 + static_cast<int32_t>((i * 7919) % 23);
REQUIRE(cudaMemcpy(buffer.getGPUBuffer(), buffer.getBuffer().data(),
x.GetPixelsNum() * sizeof(int32_t), cudaMemcpyHostToDevice) == cudaSuccess);
REQUIRE(cudaDeviceSynchronize() == cudaSuccess);
AzimuthalIntegrationProfile cpu_profile(mapping), gpu_profile(mapping);
AzIntEngineCPU(mapping).Run(buffer, cpu_profile);
AzIntEngineGPU(mapping, std::make_shared<CudaStream>()).Run(buffer, gpu_profile);
REQUIRE(gpu_profile.GetPixelCount() == cpu_profile.GetPixelCount());
const auto ref = cpu_profile.GetResult();
const auto got = gpu_profile.GetResult();
REQUIRE(ref.size() == got.size());
for (size_t b = 0; b < ref.size(); b++) {
if (std::isnan(ref[b]))
CHECK(std::isnan(got[b]));
else
CHECK(got[b] == Catch::Approx(ref[b]).epsilon(1e-5));
}
}
// The ring sums are built by atomics, which arrive in an arbitrary order, so the same frame has to be
// re-run to show the engine agrees with itself: detection is a hard "value >= threshold" on integer
// counts, and a threshold that wobbles between runs flips pixels on the boundary and with them the size
+26
View File
@@ -307,3 +307,29 @@ TEST_CASE("FindBeamCenter_BlanksTheBeamStopOutOfTheCapture", "[BeamCenter]") {
CHECK(masked_error < 12.0f);
CHECK(masked_error <= unmasked_error);
}
#ifdef JFJOCH_USE_CUDA
#include "../common/CUDAWrapper.h"
// With a GPU the passes over the pixels run on it. The cell a pixel lands in is computed from IEEE
// operations alone and without contraction on both sides, and each cell is summed in the host's order,
// so the walk ends on the same bits - a tilted detector and an offset centre included.
TEST_CASE("BeamCenterFromBackground_DeviceMatchesHost", "[BeamCenter]") {
if (get_gpu_count() == 0)
SKIP("no GPU");
DiffractionExperiment x = TestExperiment();
x.PoniRot1_rad(0.005f).PoniRot2_rad(-0.003f);
PixelMask pixel_mask(x);
const DiffractionGeometry geom_true = OffsetBy(x.GetDiffractionGeometry(), 7.0f, -4.5f);
const auto projection = SynthesiseProjection(x, pixel_mask, geom_true, 60.0f, 0.5f);
const auto host = FindBeamCenterFromBackground(x, pixel_mask, projection, 0, {}, /*allow_device=*/false);
const auto device = FindBeamCenterFromBackground(x, pixel_mask, projection, 0, {}, /*allow_device=*/true);
REQUIRE(host.has_value());
REQUIRE(device.has_value());
CHECK(device->beam_x_pxl == host->beam_x_pxl);
CHECK(device->beam_y_pxl == host->beam_y_pxl);
CHECK(device->sigma_pxl == host->sigma_pxl);
}
#endif
+3
View File
@@ -112,6 +112,8 @@ ADD_EXECUTABLE(jfjoch_test
RingsFromProfileTest.cpp
CalibrationTest.cpp
XtalOptimizerTest.cpp
XtalRefineTest.cpp
XtalRefineCeres.h
CrystalLatticeTest.cpp
FPGAPTPTest.cpp
ResolutionShellsTest.cpp
@@ -146,6 +148,7 @@ ADD_EXECUTABLE(jfjoch_test
AnisotropyAnalysisTest.cpp
ModelScalingTest.cpp
ModelScaleGPUTest.cpp
CorrectionSurfaceGPUTest.cpp
TwinningAnalysisTest.cpp
TranslationalNCSTest.cpp
RfreeFlagsTest.cpp
+167
View File
@@ -0,0 +1,167 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <catch2/catch_all.hpp>
#include "../common/CUDAWrapper.h"
#ifdef JFJOCH_USE_CUDA
#include <cmath>
#include <cstring>
#include <random>
#include <vector>
#include "../common/ParallelFor.h"
#include "../image_analysis/scale_merge/RotationScaleMergeGPU.h"
namespace {
using Term = RotationScaleMergeGPU::SurfaceTerm;
// The roundings RotationScaleMerge::ApplyCellSurface makes on the host (x86-64-v3 build), spelled out:
// a volatile result is rounded on its own and never fused into the next operation, std::fma is fused.
double Mul(double a, double b) { volatile double r = a * b; return r; }
double Add(double a, double b) { volatile double r = a + b; return r; }
constexpr int SURFACE_BLOCK = 32768; // ApplyCellSurface's reduction block
struct Surface {
int n_groups = 0, ncell = 0;
std::vector<Term> term;
std::vector<uint8_t> parity;
std::vector<int32_t> gperm, gstart;
std::vector<int32_t> sel[3]; // even, odd, all - in term order
};
Surface MakeSurface(int n_terms, int n_groups, int ncell, uint32_t seed) {
std::mt19937 rng(seed);
std::uniform_int_distribution<int> group(0, n_groups - 1), cell(0, ncell - 1), bit(0, 1);
std::uniform_real_distribution<float> u(0.0f, 1.0f);
Surface s;
s.n_groups = n_groups;
s.ncell = ncell;
for (int i = 0; i < n_terms; ++i) {
// Negative intensities, and now and then a sigma of zero: both reach the host sums as they are.
const float I = 1000.0f * u(rng) - 100.0f;
const float sigma = (i % 997 == 0) ? 0.0f : 1.0f + 30.0f * u(rng);
// Every 50th group gets no terms at all, so its reference is empty.
int g = group(rng);
if (g % 50 == 0) g = (g + 1) % n_groups;
s.term.push_back({I, sigma, 0.5f + u(rng), 1.0f + 3.0f * u(rng), cell(rng), g});
s.parity.push_back(static_cast<uint8_t>(bit(rng)));
s.sel[s.parity.back()].push_back(i);
s.sel[2].push_back(i);
}
s.gstart.assign(n_groups + 1, 0);
for (const Term &t : s.term) ++s.gstart[t.group + 1];
for (int g = 0; g < n_groups; ++g) s.gstart[g + 1] += s.gstart[g];
s.gperm.resize(n_terms);
std::vector<int32_t> fill(s.gstart.begin(), s.gstart.end() - 1);
for (int i = 0; i < n_terms; ++i) s.gperm[fill[s.term[i].group]++] = i;
return s;
}
void HostReference(const Surface &s, int parity, const std::vector<double> &A,
std::vector<double> &sw, std::vector<double> &swI) {
sw.assign(s.n_groups, 0.0);
swI.assign(s.n_groups, 0.0);
for (int g = 0; g < s.n_groups; ++g) {
double s_w = 0.0, s_wI = 0.0;
for (int k = s.gstart[g]; k < s.gstart[g + 1]; ++k) {
const int i = s.gperm[k];
if (parity >= 0 && s.parity[i] != parity) continue;
const Term &t = s.term[i];
const double a = A[t.cell];
const double Is = Mul(Mul(t.I, t.corr), a), sc = Mul(Mul(t.sigma, t.corr), a);
const double w = 1.0 / Mul(sc, sc);
s_w = Add(s_w, w);
s_wI = parity >= 0 ? std::fma(Is, w, s_wI) : Add(s_wI, Mul(Is, w));
}
sw[g] = s_w;
swI[g] = s_wI;
}
}
void HostFitSums(const Surface &s, const std::vector<int32_t> &sel, const std::vector<double> &A,
const std::vector<double> &sw, const std::vector<double> &swI,
std::vector<double> &cross, std::vector<double> &ref2) {
cross.assign(s.ncell, 0.0);
ref2.assign(s.ncell, 0.0);
const int n = static_cast<int>(sel.size()), nb = ReductionBlocks(n, SURFACE_BLOCK);
for (int b = 0; b < nb; ++b) {
std::vector<double> xcross(s.ncell, 0.0), xref2(s.ncell, 0.0);
const int lo = static_cast<int>(int64_t(n) * b / nb), hi = static_cast<int>(int64_t(n) * (b + 1) / nb);
for (int k = lo; k < hi; ++k) {
const Term &t = s.term[sel[k]];
if (sw[t.group] <= 0.0) continue;
const double Iref = swI[t.group] / sw[t.group], a = A[t.cell];
const double Is = Mul(Mul(t.I, t.corr), a), sc = Mul(Mul(t.sigma, t.corr), a);
if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue;
const double w = 1.0 / Mul(sc, sc);
xcross[t.cell] = std::fma(Mul(w, Is), Iref, xcross[t.cell]);
xref2[t.cell] = std::fma(Mul(w, Iref), Iref, xref2[t.cell]);
}
for (int c = 0; c < s.ncell; ++c) {
cross[c] = Add(cross[c], xcross[c]);
ref2[c] = Add(ref2[c], xref2[c]);
}
}
}
// The device side of one subset: ApplyCellSurface's per-block counting sort by cell.
void UploadSubset(RotationScaleMergeGPU &gpu, const Surface &s, int id, const std::vector<int32_t> &sel) {
const int n = static_cast<int>(sel.size()), nb = ReductionBlocks(n, SURFACE_BLOCK);
std::vector<int32_t> perm(n), seg_start(static_cast<size_t>(nb) * s.ncell + 1, n);
for (int b = 0; b < nb; ++b) {
const int lo = static_cast<int>(int64_t(n) * b / nb), hi = static_cast<int>(int64_t(n) * (b + 1) / nb);
std::vector<int32_t> pos(s.ncell + 1, 0);
for (int k = lo; k < hi; ++k) ++pos[s.term[sel[k]].cell + 1];
for (int c = 0; c < s.ncell; ++c) pos[c + 1] += pos[c];
for (int c = 0; c < s.ncell; ++c) seg_start[size_t(b) * s.ncell + c] = lo + pos[c];
for (int k = lo; k < hi; ++k) perm[lo + pos[s.term[sel[k]].cell]++] = sel[k];
}
gpu.SurfaceSetSubset(id, nb, perm.data(), seg_start.data());
}
bool SameBits(const std::vector<double> &a, const std::vector<double> &b) {
return a.size() == b.size() && std::memcmp(a.data(), b.data(), a.size() * sizeof(double)) == 0;
}
} // namespace
TEST_CASE("CorrectionSurfaceGPU_SumsMatchHostBitForBit", "[RotationScale][gpu]") {
if (get_gpu_count() == 0)
SKIP("No GPU");
// Enough terms for several reduction blocks in every subset.
const Surface s = MakeSurface(250000, 4000, 144, 7);
RotationScaleMergeGPU gpu;
REQUIRE(gpu.Available());
gpu.SurfaceSetTerms(static_cast<int>(s.term.size()), s.term.data(), s.parity.data(), s.n_groups,
s.gperm.data(), s.gstart.data(), s.ncell);
for (int id = 0; id < 3; ++id)
UploadSubset(gpu, s, id, s.sel[id]);
std::mt19937 rng(11);
std::uniform_real_distribution<double> u(0.7, 1.4);
std::vector<double> A(s.ncell);
for (double &a : A) a = u(rng);
for (int parity : {0, 1, -1}) {
const int id = parity < 0 ? 2 : parity;
std::vector<double> sw, swI, cross, ref2;
HostReference(s, parity, A, sw, swI);
HostFitSums(s, s.sel[id], A, sw, swI, cross, ref2);
REQUIRE(ReductionBlocks(static_cast<int>(s.sel[id].size()), SURFACE_BLOCK) > 1);
gpu.SurfaceReference(parity, A.data());
std::vector<double> dsw(s.n_groups), dswI(s.n_groups), dcross(s.ncell), dref2(s.ncell);
gpu.SurfaceGetReference(dsw.data(), dswI.data());
gpu.SurfaceFitSums(id, dcross.data(), dref2.data());
CHECK(SameBits(sw, dsw));
CHECK(SameBits(swI, dswI));
CHECK(SameBits(cross, dcross));
CHECK(SameBits(ref2, dref2));
}
}
#endif
+112 -14
View File
@@ -8,6 +8,7 @@
#include <cmath>
#include <cstring>
#include <optional>
#include <thread>
#include <vector>
#include "../common/DetectorSetup.h"
@@ -168,16 +169,15 @@ TEST_CASE("ShadowFinder_MaskDoesNotDependOnTheThreadCount", "[ShadowFinder]") {
CHECK(finder.GetMask(8) == one);
}
// Workers accumulate into shards of their own and the shards are summed when the projection is read,
// so which worker saw which frame must not reach the answer - including the maximum, which only one
// shard holds when the reflection is on a single frame.
TEST_CASE("ShadowFinder_ShardingDoesNotChangeTheProjection", "[ShadowFinder]") {
// Workers add their frames concurrently, each starting at a different band of the projection, so
// which worker added which frame, and in what order, must not reach the answer - including the
// maximum, which one frame alone holds when the reflection is on a single frame.
TEST_CASE("ShadowFinder_ConcurrentWorkersDoNotChangeTheProjection", "[ShadowFinder]") {
const DiffractionExperiment x = TestExperiment();
const PixelMask pixel_mask(x);
ShadowFinder serial(x, pixel_mask);
ShadowFinder sharded(x, pixel_mask);
sharded.SetShardCount(4);
ShadowFinder concurrent(x, pixel_mask);
std::vector<std::vector<int32_t>> frames;
std::vector<uint8_t> buffer;
@@ -185,22 +185,32 @@ TEST_CASE("ShadowFinder_ShardingDoesNotChangeTheProjection", "[ShadowFinder]") {
frames.push_back(Scene(/*cross=*/false, /*reflection=*/f == 0));
DataMessage msg{};
msg.image = CompressedImage(frames.back(), W, H);
serial.AddImage(msg, buffer, 0);
sharded.AddImage(msg, buffer, static_cast<size_t>(f) % 4);
serial.AddImage(msg, buffer);
}
std::vector<std::thread> workers;
for (int t = 0; t < 4; t++)
workers.emplace_back([&, t] {
std::vector<uint8_t> worker_buffer;
for (int f = NFRAMES - 1 - t; f >= 0; f -= 4) {
DataMessage msg{};
msg.image = CompressedImage(frames[f], W, H);
concurrent.AddImage(msg, worker_buffer);
}
});
for (auto &w : workers) w.join();
CHECK(serial.GetFrameCount() == sharded.GetFrameCount());
CHECK(serial.GetFrameCount() == concurrent.GetFrameCount());
const auto a = serial.GetMeanProjection();
const auto b = sharded.GetMeanProjection();
const auto b = concurrent.GetMeanProjection();
REQUIRE(a.size() == b.size());
// NAN marks a pixel nothing counted, and NAN != NAN, so compare the bits rather than the values.
CHECK(memcmp(a.data(), b.data(), a.size() * sizeof(float)) == 0);
// The reflection is on one frame, so its maximum lives in a single shard. If the fold lost it,
// the mask would swallow the reflection instead of giving it back.
CHECK(serial.GetMask(1) == sharded.GetMask(1));
CHECK(sharded.GetMask(1)[I(C - 14, C - 2)] == 0);
// The reflection is on one frame, so only that frame's maximum sees it. If it were lost, the mask
// would swallow the reflection instead of giving it back.
CHECK(serial.GetMask(1) == concurrent.GetMask(1));
CHECK(concurrent.GetMask(1)[I(C - 14, C - 2)] == 0);
}
// Four opaque arms and a centred disk: the scene is invariant under a quarter turn, so the mask must
@@ -461,3 +471,91 @@ TEST_CASE("ShadowFinder_ABrightRingIsNotABeamStop", "[ShadowFinder]") {
// Anything much beyond the stop, its arm and their penumbra means the walk ran away.
CHECK(std::count(mask.begin(), mask.end(), 1u) < 6000);
}
#ifdef JFJOCH_USE_CUDA
#include "../common/CUDAWrapper.h"
#include "../compression/JFJochCompressor.h"
namespace {
// The mask and the mean projection of the same frames, once from the host projection (the frames
// handed over uncompressed) and once from the device's (handed over as bitshuffle+LZ4, which is
// what sends them to the GPU).
struct HostAndDevice {
std::vector<uint32_t> host_mask, device_mask;
std::vector<float> host_mean, device_mean;
};
HostAndDevice MaskBothWays(const DiffractionExperiment &x, const std::vector<std::vector<int32_t>> &frames) {
const PixelMask pixel_mask(x);
HostAndDevice out;
std::vector<uint8_t> buffer;
{
ShadowFinder finder(x, pixel_mask);
for (const auto &frame : frames) {
DataMessage msg{};
msg.image = CompressedImage(frame, W, H);
finder.AddImage(msg, buffer);
}
out.host_mask = finder.GetMask();
out.host_mean = finder.GetMeanProjection();
}
{
ShadowFinder finder(x, pixel_mask);
JFJochBitShuffleCompressor compressor(CompressionAlgorithm::BSHUF_LZ4);
std::vector<std::vector<uint8_t>> compressed;
for (const auto &frame : frames) {
compressed.push_back(compressor.Compress(frame));
DataMessage msg{};
msg.image = CompressedImage(compressed.back().data(), compressed.back().size(), W, H,
CompressedImageMode::Int32, CompressionAlgorithm::BSHUF_LZ4);
finder.AddImage(msg, buffer);
}
out.device_mask = finder.GetMask();
out.device_mean = finder.GetMeanProjection();
}
return out;
}
}
// The mask is made on the GPU wherever the projection is there. On scenes where no pixel sits within a
// rounding of a threshold the two must agree to the pixel: every step but the polarization factor,
// the Poisson test's logarithm and the arm search's azimuth is exact on both, and these scenes are
// built so that none of the three decides anything at an edge.
TEST_CASE("ShadowFinder_DeviceMaskMatchesHost", "[ShadowFinder]") {
if (get_gpu_count() == 0)
SKIP("no GPU");
SECTION("a beam stop with a reflection behind it") {
std::vector<std::vector<int32_t>> frames;
for (int f = 0; f < NFRAMES; f++)
frames.push_back(Scene(false, f == 0));
const auto r = MaskBothWays(TestExperiment(), frames);
CHECK(std::count(r.host_mask.begin(), r.host_mask.end(), 1u) == 2612);
CHECK(r.device_mask == r.host_mask);
CHECK(std::memcmp(r.device_mean.data(), r.host_mean.data(), r.host_mean.size() * sizeof(float)) == 0);
}
SECTION("an arm that lets part of the beam through, across a module gap") {
constexpr int32_t BRIGHT = 50;
constexpr int ARM_HALF_WIDE = 15, GAP_X0 = 200, GAP_X1 = 216, OPAQUE_FROM_X = 232;
std::vector<std::vector<int32_t>> frames;
for (int f = 0; f < NFRAMES; f++) {
frames.emplace_back(static_cast<size_t>(W) * H, BRIGHT);
auto &frame = frames.back();
for (int y = 0; y < H; y++)
for (int xi = 0; xi < W; xi++) {
const int dx = xi - C, dy = y - C;
if (dx * dx + dy * dy <= STOP_R * STOP_R)
frame[I(xi, y)] = 0;
else if (dx >= 0 && std::abs(dy) <= ARM_HALF_WIDE)
frame[I(xi, y)] = xi >= OPAQUE_FROM_X ? 0 : BRIGHT * 6 / 10;
if (xi >= GAP_X0 && xi <= GAP_X1)
frame[I(xi, y)] = INT32_MIN;
}
}
const auto r = MaskBothWays(TestExperiment(), frames);
CHECK(std::count(r.host_mask.begin(), r.host_mask.end(), ShadowFinder::TRANSMITTING) > 0);
CHECK(r.device_mask == r.host_mask);
}
}
#endif
+92
View File
@@ -0,0 +1,92 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
// The reference the LM solver of XtalRefine is checked against: the same XtalRefineProblem handed to
// Ceres exactly as XtalOptimizer used to build it - the same residual functors, losses, priors, bounds,
// manifold and options.
#include "ceres/ceres.h"
#include "../image_analysis/geom_refinement/XtalRefine.h"
struct XtalRefineCeresPrior {
XtalRefineCeresPrior(double gx, double gy, double p0, double weight)
: gx(gx), gy(gy), p0(p0), weight(weight) {}
template<typename T>
bool operator()(const T *const p, T *residual) const {
residual[0] = T(weight) * (T(gx) * p[0] + T(gy) * p[1] - T(p0));
return true;
}
double gx, gy, p0, weight;
};
inline ceres::Solver::Summary SolveXtalRefineCeres(XtalRefineProblem &p, int num_threads) {
std::vector<XtalFrameConstants> frame_const;
frame_const.reserve(p.frame_angle_rad.size());
if (p.beam_and_orientation_only)
for (const double angle: p.frame_angle_rad)
frame_const.emplace_back(p.detector_rot, p.rot_vec, angle, p.latt_vec1, p.latt_vec2, p.crystal_system);
ceres::Problem problem;
for (size_t i = 0; i < p.residuals.size(); i++) {
ceres::LossFunction *loss = p.weight_sq.empty()
? nullptr
: new ceres::ScaledLoss(nullptr, p.weight_sq[i], ceres::TAKE_OWNERSHIP);
if (p.beam_and_orientation_only)
problem.AddResidualBlock(
new ceres::AutoDiffCostFunction<XtalResidualBeamOrientation, 3, 2, 3>(
new XtalResidualBeamOrientation(p.residuals[i], p.distance_mm, frame_const[p.frame[i]])),
loss, p.beam, p.latt_vec0);
else
problem.AddResidualBlock(
new ceres::AutoDiffCostFunction<XtalResidualFixedDistance, 3, 2, 2, 3, 3, 3, 3>(
new XtalResidualFixedDistance(p.residuals[i], p.distance_mm)),
loss, p.beam, p.detector_rot, p.rot_vec, p.latt_vec0, p.latt_vec1, p.latt_vec2);
}
for (const auto &prior: p.priors)
problem.AddResidualBlock(
new ceres::AutoDiffCostFunction<XtalRefineCeresPrior, 1, 2>(
new XtalRefineCeresPrior(prior.gx, prior.gy, prior.p0, prior.weight)),
nullptr, prior.block == XtalRefinePrior::Block::Beam ? p.beam : p.detector_rot);
const auto bounds = [&](double *block, const double *lo, const double *hi, int n) {
for (int i = 0; i < n; i++) {
if (lo[i] > -XtalRefineProblem::kNoBound)
problem.SetParameterLowerBound(block, i, lo[i]);
if (hi[i] < XtalRefineProblem::kNoBound)
problem.SetParameterUpperBound(block, i, hi[i]);
}
};
if (p.beam_constant)
problem.SetParameterBlockConstant(p.beam);
if (!p.beam_and_orientation_only) {
if (p.detector_rot_constant)
problem.SetParameterBlockConstant(p.detector_rot);
else
bounds(p.detector_rot, p.detector_rot_lower, p.detector_rot_upper, 2);
if (p.rot_vec_constant)
problem.SetParameterBlockConstant(p.rot_vec);
else
problem.SetManifold(p.rot_vec, new ceres::SphereManifold<3>);
if (p.latt_vec1_constant)
problem.SetParameterBlockConstant(p.latt_vec1);
else
bounds(p.latt_vec1, p.latt_vec1_lower, p.latt_vec1_upper, 3);
if (p.latt_vec2_constant)
problem.SetParameterBlockConstant(p.latt_vec2);
else
bounds(p.latt_vec2, p.latt_vec2_lower, p.latt_vec2_upper, 3);
}
ceres::Solver::Options options;
options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY;
options.minimizer_progress_to_stdout = false;
options.max_num_iterations = p.options.max_iterations;
options.max_solver_time_in_seconds = p.options.max_time_s;
options.logging_type = ceres::LoggingType::SILENT;
options.num_threads = num_threads;
ceres::Solver::Summary summary;
ceres::Solve(options, &problem, &summary);
return summary;
}
+127
View File
@@ -0,0 +1,127 @@
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
// Ceres first: its logging header defines a CHECK macro of its own, which Catch's must replace here.
#include "XtalRefineCeres.h"
#undef CHECK
#include <catch2/catch_all.hpp>
#include "../image_analysis/geom_refinement/LatticeReduction.h"
#include "../image_analysis/bragg_prediction/BraggPrediction.h"
// The LM solver of XtalRefine is meant to take Ceres' path to Ceres' answer: same steps accepted, same
// stopping rule, same point. These cases hand one problem to both and compare.
namespace {
// A monoclinic crystal rotated about X over ten 3-degree frames, its predicted spots as observations;
// the refinement starts from a perturbed beam, tilt and cell.
XtalRefineProblem RotationProblem(bool reduced, bool weighted) {
DiffractionExperiment exp;
exp.IncidentEnergy_keV(WVL_1A_IN_KEV).BeamX_pxl(1000).BeamY_pxl(1000)
.PoniRot1_rad(0.01).PoniRot2_rad(0.02).DetectorDistance_mm(200);
const auto geom = exp.GetDiffractionGeometry();
const CrystalLattice latt(40, 50, 80, 90, 95, 90);
const GoniometerAxis axis("omega", 0.0f, 3.0f, Coord(1, 0, 0), std::nullopt);
const gemmi::CrystalSystem sys = gemmi::CrystalSystem::Monoclinic;
XtalRefineProblem p;
p.crystal_system = sys;
p.beam_and_orientation_only = reduced;
p.distance_mm = 200;
BraggPrediction prediction;
const BraggPredictionSettings settings{.high_res_A = 1.5, .ewald_dist_cutoff = 0.002};
for (int img = 0; img < 10; img++) {
const float angle_deg = axis.GetAngle_deg(img) + axis.GetWedge_deg() / 2.0f;
const auto n = prediction.Calc(exp, latt.Multiply(axis.GetTransformationAngle(angle_deg).transpose()),
settings);
p.frame_angle_rad.push_back(angle_deg * PI / 180.0);
for (int i = 0; i < n; i++) {
const auto &r = prediction.GetReflections().at(i);
p.residuals.emplace_back(r.predicted_x + 0.3 * std::sin(i), r.predicted_y + 0.3 * std::cos(i),
geom.GetWavelength_A(), geom.GetPixelSize_mm(), 1.0, 0.0,
angle_deg * PI / 180.0, r.h, r.k, r.l, sys);
p.frame.push_back(img);
if (weighted)
p.weight_sq.push_back(0.2 + 0.6 * (i % 5) / 4.0);
}
}
p.beam[0] = 1000.0;
p.beam[1] = 997.0;
p.detector_rot[0] = 0.012;
p.detector_rot[1] = 0.018;
p.rot_vec[0] = 1.0;
p.rot_vec[1] = 0.0;
p.rot_vec[2] = 0.0;
double beta = 0;
LatticeToRodriguesLengthsBeta_Mono(CrystalLattice(39.7f, 50.6f, 79.6f, 90.0f, 94.5f, 90.0f),
p.latt_vec0, p.latt_vec1, beta);
p.latt_vec2[0] = beta;
if (!reduced) {
p.detector_rot_constant = false;
p.rot_vec_constant = false;
p.latt_vec1_constant = false;
p.latt_vec2_constant = false;
for (int i = 0; i < 2; i++) {
p.detector_rot_lower[i] = p.detector_rot[i] - 0.05;
p.detector_rot_upper[i] = p.detector_rot[i] + 0.05;
}
for (int i = 0; i < 3; i++) {
p.latt_vec1_lower[i] = 5.0;
p.latt_vec1_upper[i] = 100.0;
}
p.latt_vec2_lower[0] = PI / 3;
p.latt_vec2_upper[0] = 2 * PI / 3;
p.priors.push_back({XtalRefinePrior::Block::Beam, 1.0, 0.0, p.beam[0], 0.5});
p.priors.push_back({XtalRefinePrior::Block::DetectorRot, 0.0, 1.0, p.detector_rot[1], 50.0});
}
p.options.max_iterations = 50;
return p;
}
void CompareWithCeres(XtalRefineProblem p) {
XtalRefineProblem q = p;
const LMSummary lm = SolveXtalRefine(p, 4);
const ceres::Solver::Summary ref = SolveXtalRefineCeres(q, 4);
REQUIRE(lm.IsSolutionUsable() == ref.IsSolutionUsable());
CHECK(lm.iterations == static_cast<int>(ref.iterations.size()));
CHECK(lm.final_cost == Catch::Approx(ref.final_cost).epsilon(1e-9));
const auto same = [](const double *a, const double *b, int n, double tol) {
for (int i = 0; i < n; i++)
CHECK(a[i] == Catch::Approx(b[i]).margin(tol));
};
same(p.beam, q.beam, 2, 1e-7);
same(p.detector_rot, q.detector_rot, 2, 1e-10);
same(p.rot_vec, q.rot_vec, 3, 1e-10);
same(p.latt_vec0, q.latt_vec0, 3, 1e-10);
same(p.latt_vec1, q.latt_vec1, 3, 1e-8);
same(p.latt_vec2, q.latt_vec2, 3, 1e-10);
}
}
TEST_CASE("XtalRefine_matches_Ceres_full", "[XtalOptimizer]") {
CompareWithCeres(RotationProblem(false, false));
}
TEST_CASE("XtalRefine_matches_Ceres_full_weighted", "[XtalOptimizer]") {
CompareWithCeres(RotationProblem(false, true));
}
TEST_CASE("XtalRefine_matches_Ceres_beam_orientation", "[XtalOptimizer]") {
CompareWithCeres(RotationProblem(true, false));
}
TEST_CASE("XtalRefine_same_answer_at_any_thread_count", "[XtalOptimizer]") {
XtalRefineProblem a = RotationProblem(false, false);
XtalRefineProblem b = a;
SolveXtalRefine(a, 1);
SolveXtalRefine(b, 7);
for (int i = 0; i < 3; i++) {
CHECK(a.latt_vec0[i] == b.latt_vec0[i]);
CHECK(a.latt_vec1[i] == b.latt_vec1[i]);
}
CHECK(a.beam[0] == b.beam[0]);
CHECK(a.beam[1] == b.beam[1]);
}