Merge perf-merge: rotation performance round (own LM solver, GPU/parallel pre-scan, tail on own GPU stream, GPU correction surfaces, CPU spot-finder/prediction, faster GPU azint kernel)
Battery: 20261003-1424_581e1c_perf-merge-full (+_private) vs rc174 built with the same flags:
no verdict change except the second scaling engine's GPU OOM on the three largest sets,
removed in 0f728ecab (8a1a/8qaw/8tyy pass again: 20261003-2032_581e1c_perf-oom-fix).
One-time result change from the GPU/CPU-parity beam-centre walk (b2).
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
This commit is contained in:
@@ -50,7 +50,7 @@ either way. Eigen is header-only: only its headers reach the binaries, and no Ei
|
||||
|
||||
## Vendored directly in the repository
|
||||
|
||||
These live in the source tree (see the path) rather than being fetched; traccc is the exception - code adapted into first-party files rather than a vendored directory, see the note at the end of this file.
|
||||
These live in the source tree (see the path) rather than being fetched; traccc and the Ceres-derived minimiser are the exceptions - code adapted into first-party files rather than a vendored directory, see the notes at the end of this file.
|
||||
|
||||
| Component | Path | Copyright | License (SPDX) | License text |
|
||||
|---|---|---|---|---|
|
||||
@@ -69,6 +69,7 @@ These live in the source tree (see the path) rather than being fetched; traccc i
|
||||
| [pocketfft](https://github.com/mreineck/pocketfft) | `gemmi_gph/gemmi/third_party/pocketfft_hdronly.h` | Max-Planck-Society; Peter Bell; MIT (FFTW-derived parts) | BSD-3-Clause | [pocketfft.txt](licenses/pocketfft.txt) |
|
||||
| [tinydir](https://github.com/cxong/tinydir) | `gemmi_gph/gemmi/third_party/tinydir.h` | Cong Xu, Lautis Sun, Baudouin Feildel, Andargor | BSD-2-Clause | [tinydir.txt](licenses/tinydir.txt) |
|
||||
| [traccc (ACTS)](https://github.com/acts-project/traccc) | `image_analysis/spot_finding/StrongPixelSet.cpp`, `SpotExtractorGPU.cu` | CERN, for the benefit of the ACTS project | MPL-2.0 | [traccc.txt](licenses/traccc.txt) |
|
||||
| [Ceres Solver](https://github.com/ceres-solver/ceres-solver) (adapted) | `image_analysis/geom_refinement/LMSolver.h`, `LMSolver.cpp` | Google Inc. | BSD-3-Clause | [ceres-solver.txt](licenses/ceres-solver.txt) |
|
||||
| [xbflash.qspi](https://github.com/Xilinx/XRT) | `tools/xbflash.qspi/` | Xilinx / AMD | Apache-2.0 | [xbflash-qspi.txt](licenses/xbflash-qspi.txt) |
|
||||
| [wingetopt](https://github.com/alex85k/wingetopt) | `tools/wingetopt/` | Todd C. Miller; The NetBSD Foundation | ISC AND BSD-2-Clause | [wingetopt.txt](licenses/wingetopt.txt) |
|
||||
|
||||
@@ -107,6 +108,10 @@ served frontend, so the shipped web UI carries its own attribution.
|
||||
and `SpotExtractorGPU.cu` follows the design of its GPU counterpart. MPL-2.0 is file-level, so both
|
||||
files name the origin at the top and are covered by `licenses/traccc.txt`. See
|
||||
[ACKNOWLEDGEMENT.md](docs/ACKNOWLEDGEMENT.md) for the citation.
|
||||
* **Ceres Solver** is also fetched and linked (table above); separately, `LMSolver.h`/`.cpp` re-implement
|
||||
its trust-region Levenberg-Marquardt minimiser, projected line search, polynomial step choice and
|
||||
sphere manifold for the crystal refinement, following its source. Both files name the origin at the
|
||||
top and are covered by `licenses/ceres-solver.txt`.
|
||||
* **FFTW** is GPL-2.0-or-later — compatible with, and absorbed by, this project's GPL-3.0 license.
|
||||
* **Apache-2.0** components: where upstream ships a `NOTICE` file, it is reproduced in the
|
||||
corresponding `licenses/` text.
|
||||
|
||||
@@ -99,7 +99,10 @@ ADD_LIBRARY(JFJochImageAnalysis STATIC
|
||||
beam_stop/ShadowFinder.cpp
|
||||
beam_stop/ShadowFinder.h
|
||||
$<$<BOOL:${JFJOCH_CUDA_AVAILABLE}>:beam_stop/ShadowAccumulatorGPU.cu>
|
||||
$<$<BOOL:${JFJOCH_CUDA_AVAILABLE}>:beam_stop/ShadowMaskGPU.cu>
|
||||
beam_stop/ShadowAccumulatorGPU.h
|
||||
beam_stop/ShadowFinderInternal.h
|
||||
beam_stop/ShadowMaskGPU.h
|
||||
rotation_indexer/RotationIndexer.cpp
|
||||
rotation_indexer/RotationIndexer.h
|
||||
WriteReflections.cpp
|
||||
|
||||
@@ -8,6 +8,15 @@ inline void cuda_err(cudaError_t val) {
|
||||
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
|
||||
}
|
||||
|
||||
// Pushes one ring run's totals to the shared accumulators; nothing for an empty run.
|
||||
__device__ __forceinline__ void flush_azim_run(float *s_sum, float *s_sum2, uint32_t *s_count,
|
||||
int b, float r_sum, float r_sum2, uint32_t r_count) {
|
||||
if (r_count == 0) return; // also covers the initial "no ring yet"
|
||||
atomicAdd(&s_sum[b], r_sum);
|
||||
atomicAdd(&s_sum2[b], r_sum2);
|
||||
atomicAdd(&s_count[b], r_count);
|
||||
}
|
||||
|
||||
__global__
|
||||
void gpu_azim_shared(
|
||||
const uint16_t *__restrict__ pixel_to_bin,
|
||||
@@ -33,19 +42,49 @@ void gpu_azim_shared(
|
||||
|
||||
__syncthreads();
|
||||
|
||||
for (size_t idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
idx < num_pixels;
|
||||
idx += blockDim.x * gridDim.x) {
|
||||
uint16_t bin = pixel_to_bin[idx];
|
||||
// Four pixels per thread, read as vector loads, and a running total per ring pushed to shared
|
||||
// memory only when the ring changes: consecutive pixels along a row mostly share a ring, and an
|
||||
// atomic per pixel on the same few addresses is what this kernel was limited by. The same scheme
|
||||
// as the adaptive spot finder's ring pass (reduce_rings_shared). The buffers come straight from
|
||||
// cudaMalloc, aligned for int4/float4; the npix % 4 leftovers are done one at a time below.
|
||||
const size_t stride = static_cast<size_t>(blockDim.x) * gridDim.x;
|
||||
const size_t nquad = num_pixels / 4;
|
||||
for (size_t q = blockIdx.x * blockDim.x + threadIdx.x; q < nquad; q += stride) {
|
||||
const int4 v4 = reinterpret_cast<const int4 *>(input_buffer)[q];
|
||||
const ushort4 b4 = reinterpret_cast<const ushort4 *>(pixel_to_bin)[q];
|
||||
const float4 c4 = reinterpret_cast<const float4 *>(corrections)[q];
|
||||
const int32_t vq[4] = {v4.x, v4.y, v4.z, v4.w};
|
||||
const uint16_t bq[4] = {b4.x, b4.y, b4.z, b4.w};
|
||||
const float cq[4] = {c4.x, c4.y, c4.z, c4.w};
|
||||
|
||||
int32_t v = input_buffer[idx];
|
||||
bool valid = (v != INT32_MIN) & (v != INT32_MAX);
|
||||
int r_b = -1;
|
||||
float r_sum = 0.0f, r_sum2 = 0.0f;
|
||||
uint32_t r_count = 0;
|
||||
#pragma unroll
|
||||
for (int k = 0; k < 4; k++) {
|
||||
const int32_t v = vq[k];
|
||||
const int b = bq[k];
|
||||
if (v == INT32_MIN || v == INT32_MAX || b >= azint_bins) continue;
|
||||
if (b != r_b) {
|
||||
flush_azim_run(s_sum, s_sum2, s_count, r_b, r_sum, r_sum2, r_count);
|
||||
r_b = b;
|
||||
r_sum = 0.0f; r_sum2 = 0.0f; r_count = 0;
|
||||
}
|
||||
const float val = static_cast<float>(v) * cq[k];
|
||||
r_sum += val;
|
||||
r_sum2 += val * val;
|
||||
r_count += 1;
|
||||
}
|
||||
flush_azim_run(s_sum, s_sum2, s_count, r_b, r_sum, r_sum2, r_count);
|
||||
}
|
||||
|
||||
if (bin < azint_bins && valid) {
|
||||
for (size_t idx = 4 * nquad + blockIdx.x * blockDim.x + threadIdx.x; idx < num_pixels; idx += stride) {
|
||||
const uint16_t bin = pixel_to_bin[idx];
|
||||
const int32_t v = input_buffer[idx];
|
||||
if (bin < azint_bins && v != INT32_MIN && v != INT32_MAX) {
|
||||
const float val = static_cast<float>(v) * corrections[idx];
|
||||
const float val2 = val * val;
|
||||
atomicAdd(&s_sum[bin], val);
|
||||
atomicAdd(&s_sum2[bin], val2);
|
||||
atomicAdd(&s_sum2[bin], val * val);
|
||||
atomicAdd(&s_count[bin], 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -190,3 +190,13 @@ void ShadowAccumulatorGPU::Download(std::vector<int64_t> &max_value, std::vector
|
||||
cudaMemcpyDeviceToHost, *stream));
|
||||
cuda_err(cudaStreamSynchronize(*stream));
|
||||
}
|
||||
|
||||
std::vector<uint32_t> ShadowAccumulatorGPU::Mask(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask) {
|
||||
FoldPending();
|
||||
return ShadowMaskOnDevice(setup, pixel_mask, gpu_max, gpu_sum, gpu_count, frames, *stream);
|
||||
}
|
||||
|
||||
std::vector<float> ShadowAccumulatorGPU::MeanProjection(const std::vector<uint32_t> &pixel_mask) {
|
||||
FoldPending();
|
||||
return MeanProjectionOnDevice(pixel_mask, gpu_sum, gpu_count, npixels, *stream);
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
#include "../../common/CompressedImage.h"
|
||||
#include "../image_preprocessing/BSLZ4DecoderGPU.h"
|
||||
#include "../indexing/CUDAMemHelpers.h"
|
||||
#include "ShadowMaskGPU.h"
|
||||
|
||||
// The beam-stop projection accumulated on the device: only the compressed chunk crosses PCIe, and
|
||||
// both the decode and the per-pixel maximum / sum / count run on the GPU. The projection comes back
|
||||
@@ -59,6 +60,11 @@ public:
|
||||
|
||||
[[nodiscard]] uint32_t GetFrameCount() const { return frames; }
|
||||
|
||||
// The beam-stop mask and the mean projection, made where the projection is (ShadowMaskGPU.h), so
|
||||
// that it does not have to come back at all.
|
||||
std::vector<uint32_t> Mask(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask);
|
||||
std::vector<float> MeanProjection(const std::vector<uint32_t> &pixel_mask);
|
||||
|
||||
// Bring the projection back to the host, folding in whatever the last batch still holds. Cheap
|
||||
// to call once; it moves 20 bytes per pixel.
|
||||
void Download(std::vector<int64_t> &max_value, std::vector<int64_t> &sum_value,
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#include "ShadowFinder.h"
|
||||
#include "ShadowFinderInternal.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
@@ -10,7 +11,6 @@
|
||||
#include <future>
|
||||
#include <thread>
|
||||
#include <numbers>
|
||||
#include <queue>
|
||||
#include <type_traits>
|
||||
|
||||
#include <spdlog/spdlog.h>
|
||||
@@ -19,67 +19,12 @@
|
||||
#include "../../common/ParallelFor.h"
|
||||
#include "../../common/JFJochException.h"
|
||||
|
||||
// A pixel is shadow when its background is below this fraction of the background it is
|
||||
// compared against.
|
||||
constexpr float SHADOW_RATIO = 0.50f;
|
||||
using namespace shadow_finder;
|
||||
|
||||
// The boundary grows outward into partially shadowed pixels down to this fraction, but no
|
||||
// further than PENUMBRA_MAX_PX from the core. A pin or a loop casts a wide half-shadow, so the
|
||||
// reach is a good deal more than the beam stop's own edge needs.
|
||||
constexpr float PENUMBRA_RATIO = 0.75f;
|
||||
constexpr int PENUMBRA_MAX_PX = 30;
|
||||
|
||||
// Bridge small breaks along the holder arm. A module gap wider than this is bridged separately for
|
||||
// the arm search (see bridge_gaps).
|
||||
constexpr int BRIDGE_PX = 6;
|
||||
|
||||
// A pixel whose maximum reaches this recorded a real reflection and is never masked - a
|
||||
// beam stop cannot block a reflection that was measured.
|
||||
constexpr int64_t MIN_REFLECTION = 25;
|
||||
|
||||
// How far below the background it is compared against a pixel must sit before the dip is
|
||||
// believed, in standard deviations of the counts that back it. The counts are photons, so their
|
||||
// scatter is Poisson and the deficit is measured against it rather than against a fixed number:
|
||||
// on a well-exposed sweep a third of the background missing is overwhelming, and on a handful of
|
||||
// low-background frames the same third is noise. Without this a six-frame pre-scan of a
|
||||
// low-background sweep masks three quarters of the detector.
|
||||
constexpr double MIN_DEFICIT_SIGMA = 6.0;
|
||||
|
||||
// Smallest region the per-pixel test may return. A shadow is cast by something physical and is
|
||||
// correspondingly large; an isolated patch this small is the background wandering, not hardware.
|
||||
// This is what keeps the test specific now that a shadow no longer has to touch the direct beam.
|
||||
constexpr int MIN_SHADOW_PIXELS = 2000;
|
||||
|
||||
// ... and of those, how many must be deep (below SHADOW_RATIO) for a region of merely DIM pixels -
|
||||
// hardware that lets part of the beam through - to count. A thin holder arm is dim along most of
|
||||
// its length and deep in places; the background drifting over a detector's edge is dim everywhere
|
||||
// and deep nowhere.
|
||||
constexpr int MIN_CORE_PIXELS = 200;
|
||||
|
||||
// The arm search compares a pixel with its ring as the ring actually varies around the beam. A ring
|
||||
// of background is not flat once divided by the polarization factor when that factor is not the
|
||||
// beam's: the remainder is a second harmonic in azimuth, cos 2phi, which reaches tens of percent at
|
||||
// high angle. It is measured over radial bands of HARMONIC_BAND_PX, from the median of each of
|
||||
// HARMONIC_SECTORS sectors that holds at least MIN_SECTOR_PIXELS pixels.
|
||||
constexpr int HARMONIC_BAND_PX = 64;
|
||||
constexpr int HARMONIC_SECTORS = 24;
|
||||
constexpr int MIN_SECTOR_PIXELS = 200;
|
||||
|
||||
// Side of the box the background is pooled over before testing. Its area is how many pixels back
|
||||
// a ring's countability test, which decides where an azimuthal comparison is possible at all.
|
||||
constexpr int POOL_PX = 5;
|
||||
constexpr double MEAN_POOLED_PIXELS = POOL_PX * POOL_PX;
|
||||
|
||||
// A ring with fewer valid pixels than this says nothing about whether it was counted.
|
||||
constexpr int MIN_RING_PIXELS = 32;
|
||||
|
||||
// A ring lies wholly inside the stop when its background is below this fraction of the background
|
||||
// further out. This asks about a whole ring rather than about a pixel, so it keeps a threshold of
|
||||
// its own and does not follow SHADOW_RATIO.
|
||||
constexpr float BLOCKED_RING_RATIO = 0.35f;
|
||||
static_assert(ShadowFinder::SHADOW == MASK_SHADOW && ShadowFinder::TRANSMITTING == MASK_TRANSMITTING);
|
||||
|
||||
// Binary-image helpers on a width*height frame stored row-major as char (0/1). All run once,
|
||||
// at GetMask() time; the BFS forms keep them O(pixels) rather than O(pixels * radius).
|
||||
// at GetMask() time, and all are O(pixels) rather than O(pixels * radius).
|
||||
namespace {
|
||||
|
||||
// A per-pixel array of GetMask(). A std::vector zeroes what it allocates on the thread that makes it,
|
||||
@@ -186,10 +131,12 @@ Plane<char> erode(const Plane<char> &in, int W, int H, int r, size_t nthreads) {
|
||||
// continues on both sides of it is one shadow - but a gap can be wider than BRIDGE_PX reaches (17 px
|
||||
// between the rows of PILATUS modules), and an arm crossing one fell apart into pieces each too small
|
||||
// to be believed.
|
||||
Plane<char> bridge_gaps(const Plane<char> ®ion, const Plane<char> &valid, int W, int H) {
|
||||
Plane<char> bridge_gaps(const Plane<char> ®ion, const Plane<char> &valid, int W, int H, size_t nthreads) {
|
||||
Plane<char> out = region;
|
||||
// The lines of one direction are independent: each reads `region` and only ever sets its own pixels.
|
||||
auto walk = [&](int n_lines, int len, auto index) {
|
||||
for (int line = 0; line < n_lines; line++) {
|
||||
ParallelChunks(n_lines, nthreads, [&](int lo, int hi) {
|
||||
for (int line = lo; line < hi; line++) {
|
||||
int k = 0;
|
||||
while (k < len) {
|
||||
if (valid[index(line, k)]) { k++; continue; }
|
||||
@@ -199,12 +146,99 @@ Plane<char> bridge_gaps(const Plane<char> ®ion, const Plane<char> &valid, int
|
||||
for (int j = start; j < k; j++) out[index(line, j)] = 1;
|
||||
}
|
||||
}
|
||||
});
|
||||
};
|
||||
walk(H, W, [W](int y, int x) { return static_cast<size_t>(y) * W + x; });
|
||||
walk(W, H, [W](int x, int y) { return static_cast<size_t>(y) * W + x; });
|
||||
return out;
|
||||
}
|
||||
|
||||
// The 8-connected components of `member`: each member pixel gets the index of its component, dense
|
||||
// from 0 and in no particular order, and every other pixel -1.
|
||||
//
|
||||
// Labelled in parallel. Each band of rows is flooded on its own, then the pieces that touch across a
|
||||
// band boundary are joined. Which pixels share a component is all a caller reads, and that does not
|
||||
// depend on how the rows were split.
|
||||
struct Components {
|
||||
Plane<int> id;
|
||||
int count = 0;
|
||||
};
|
||||
|
||||
Components label_components(const Plane<char> &member, int W, int H, size_t nthreads) {
|
||||
const int bands = std::min(64, H); // never more bands than rows, so none is empty
|
||||
std::vector<int> band_row(bands + 1);
|
||||
for (int b = 0; b <= bands; b++)
|
||||
band_row[b] = static_cast<int>(static_cast<int64_t>(b) * H / bands);
|
||||
|
||||
Components out;
|
||||
out.id = Plane<int>(member.size());
|
||||
std::vector<int> band_pieces(bands, 0);
|
||||
ParallelFor(bands, nthreads, [&](int b) {
|
||||
const size_t lo = static_cast<size_t>(band_row[b]) * W, hi = static_cast<size_t>(band_row[b + 1]) * W;
|
||||
std::fill(out.id.begin() + lo, out.id.begin() + hi, -1);
|
||||
std::vector<size_t> stack;
|
||||
int pieces = 0;
|
||||
for (size_t start = lo; start < hi; start++) {
|
||||
if (!member[start] || out.id[start] >= 0)
|
||||
continue;
|
||||
out.id[start] = pieces;
|
||||
stack.push_back(start);
|
||||
while (!stack.empty()) {
|
||||
const size_t i = stack.back(); stack.pop_back();
|
||||
const int y = static_cast<int>(i / W), x = static_cast<int>(i % W);
|
||||
for (int dy = -1; dy <= 1; dy++)
|
||||
for (int dx = -1; dx <= 1; dx++) {
|
||||
const int yy = y + dy, xx = x + dx;
|
||||
if (yy < band_row[b] || yy >= band_row[b + 1] || xx < 0 || xx >= W)
|
||||
continue;
|
||||
const size_t j = static_cast<size_t>(yy) * W + xx;
|
||||
if (member[j] && out.id[j] < 0) { out.id[j] = pieces; stack.push_back(j); }
|
||||
}
|
||||
}
|
||||
pieces++;
|
||||
}
|
||||
band_pieces[b] = pieces;
|
||||
});
|
||||
|
||||
// A piece is named by its band's first index plus its number in the band, and the pieces are
|
||||
// joined across each boundary row by union-find.
|
||||
std::vector<int> first(bands + 1, 0);
|
||||
for (int b = 0; b < bands; b++)
|
||||
first[b + 1] = first[b] + band_pieces[b];
|
||||
std::vector<int> parent(first[bands]);
|
||||
for (size_t k = 0; k < parent.size(); k++)
|
||||
parent[k] = static_cast<int>(k);
|
||||
const auto find = [&](int k) {
|
||||
while (parent[k] != k) { parent[k] = parent[parent[k]]; k = parent[k]; }
|
||||
return k;
|
||||
};
|
||||
for (int b = 0; b + 1 < bands; b++) {
|
||||
const size_t above = static_cast<size_t>(band_row[b + 1] - 1) * W, below = above + W;
|
||||
for (int x = 0; x < W; x++) {
|
||||
if (!member[above + x])
|
||||
continue;
|
||||
for (int xx = std::max(0, x - 1); xx <= std::min(W - 1, x + 1); xx++)
|
||||
if (member[below + xx]) {
|
||||
const int ra = find(first[b] + out.id[above + x]);
|
||||
const int rb = find(first[b + 1] + out.id[below + xx]);
|
||||
if (ra != rb) parent[std::max(ra, rb)] = std::min(ra, rb);
|
||||
}
|
||||
}
|
||||
}
|
||||
std::vector<int> dense(parent.size(), -1), component(parent.size());
|
||||
for (size_t k = 0; k < parent.size(); k++) {
|
||||
const int root = find(static_cast<int>(k));
|
||||
if (dense[root] < 0) dense[root] = out.count++;
|
||||
component[k] = dense[root];
|
||||
}
|
||||
ParallelFor(bands, nthreads, [&](int b) {
|
||||
const size_t lo = static_cast<size_t>(band_row[b]) * W, hi = static_cast<size_t>(band_row[b + 1]) * W;
|
||||
for (size_t i = lo; i < hi; i++)
|
||||
if (out.id[i] >= 0) out.id[i] = component[first[b] + out.id[i]];
|
||||
});
|
||||
return out;
|
||||
}
|
||||
|
||||
// Fill holes: background not reachable from the image border becomes region.
|
||||
//
|
||||
// The flood is run over the bounding box of `region` grown by one, not the whole detector. Outside
|
||||
@@ -212,52 +246,53 @@ Plane<char> bridge_gaps(const Plane<char> ®ion, const Plane<char> &valid, int
|
||||
// outside is one border-connected component: a background pixel inside the box is border-connected
|
||||
// exactly when it reaches the ring. The beam stop occupies a small part of a detector, so this is
|
||||
// the same answer over a fraction of the pixels.
|
||||
Plane<char> fill_holes(const Plane<char> ®ion, int W, int H) {
|
||||
Plane<char> fill_holes(const Plane<char> ®ion, int W, int H, size_t nthreads) {
|
||||
std::vector<int> row_x0(H, W), row_x1(H, -1);
|
||||
ParallelChunks(H, nthreads, [&](int ylo, int yhi) {
|
||||
for (int y = ylo; y < yhi; y++)
|
||||
for (int x = 0; x < W; x++)
|
||||
if (region[static_cast<size_t>(y) * W + x]) {
|
||||
row_x0[y] = std::min(row_x0[y], x);
|
||||
row_x1[y] = x;
|
||||
}
|
||||
});
|
||||
int x0 = W, x1 = -1, y0 = H, y1 = -1;
|
||||
for (int y = 0; y < H; y++)
|
||||
for (int x = 0; x < W; x++)
|
||||
if (region[static_cast<size_t>(y) * W + x]) {
|
||||
x0 = std::min(x0, x); x1 = std::max(x1, x);
|
||||
y0 = std::min(y0, y); y1 = std::max(y1, y);
|
||||
}
|
||||
if (row_x1[y] >= 0) {
|
||||
x0 = std::min(x0, row_x0[y]); x1 = std::max(x1, row_x1[y]);
|
||||
y0 = std::min(y0, y); y1 = y;
|
||||
}
|
||||
if (x1 < 0)
|
||||
return region; // nothing to enclose
|
||||
x0 = std::max(0, x0 - 1); x1 = std::min(W - 1, x1 + 1);
|
||||
y0 = std::max(0, y0 - 1); y1 = std::min(H - 1, y1 + 1);
|
||||
|
||||
// The background of the box, in components; one that reaches the box's edge is outside.
|
||||
const int BW = x1 - x0 + 1, BH = y1 - y0 + 1;
|
||||
std::vector<char> bg_visited(static_cast<size_t>(BW) * BH, 0);
|
||||
std::queue<int> q; // indices into the box
|
||||
auto push = [&](int bx, int by) {
|
||||
const int j = by * BW + bx;
|
||||
if (!region[static_cast<size_t>(by + y0) * W + bx + x0] && !bg_visited[j]) {
|
||||
bg_visited[j] = 1; q.push(j);
|
||||
}
|
||||
Plane<char> background(static_cast<size_t>(BW) * BH);
|
||||
ParallelChunks(BH, nthreads, [&](int lo, int hi) {
|
||||
for (int by = lo; by < hi; by++)
|
||||
for (int bx = 0; bx < BW; bx++)
|
||||
background[static_cast<size_t>(by) * BW + bx] = !region[static_cast<size_t>(by + y0) * W + bx + x0];
|
||||
});
|
||||
const auto pieces = label_components(background, BW, BH, nthreads);
|
||||
std::vector<char> outside(pieces.count, 0);
|
||||
const auto edge = [&](int bx, int by) {
|
||||
const int c = pieces.id[static_cast<size_t>(by) * BW + bx];
|
||||
if (c >= 0) outside[c] = 1;
|
||||
};
|
||||
for (int bx = 0; bx < BW; bx++) { push(bx, 0); push(bx, BH - 1); }
|
||||
for (int by = 0; by < BH; by++) { push(0, by); push(BW - 1, by); }
|
||||
while (!q.empty()) {
|
||||
const int i = q.front(); q.pop();
|
||||
const int by = i / BW, bx = i % BW;
|
||||
for (int dy = -1; dy <= 1; dy++)
|
||||
for (int dx = -1; dx <= 1; dx++) {
|
||||
const int yy = by + dy, xx = bx + dx;
|
||||
if (yy < 0 || yy >= BH || xx < 0 || xx >= BW)
|
||||
continue;
|
||||
const int j = yy * BW + xx;
|
||||
if (!region[static_cast<size_t>(yy + y0) * W + xx + x0] && !bg_visited[j]) {
|
||||
bg_visited[j] = 1; q.push(j);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int bx = 0; bx < BW; bx++) { edge(bx, 0); edge(bx, BH - 1); }
|
||||
for (int by = 0; by < BH; by++) { edge(0, by); edge(BW - 1, by); }
|
||||
|
||||
Plane<char> out = region;
|
||||
for (int by = 0; by < BH; by++)
|
||||
for (int bx = 0; bx < BW; bx++) {
|
||||
const size_t i = static_cast<size_t>(by + y0) * W + bx + x0;
|
||||
if (!region[i] && !bg_visited[by * BW + bx])
|
||||
out[i] = 1;
|
||||
}
|
||||
ParallelChunks(BH, nthreads, [&](int lo, int hi) {
|
||||
for (int by = lo; by < hi; by++)
|
||||
for (int bx = 0; bx < BW; bx++) {
|
||||
const int c = pieces.id[static_cast<size_t>(by) * BW + bx];
|
||||
if (c >= 0 && !outside[c])
|
||||
out[static_cast<size_t>(by + y0) * W + bx + x0] = 1;
|
||||
}
|
||||
});
|
||||
return out;
|
||||
}
|
||||
|
||||
@@ -321,19 +356,40 @@ struct RingValues {
|
||||
|
||||
RingValues bin_by_ring(const Plane<float> &values, const Plane<char> &valid,
|
||||
const Plane<int> &radius, int max_radius, size_t nthreads) {
|
||||
// Counted and scattered by blocks of pixels in parallel: each block writes its values of a ring
|
||||
// after those of the blocks before it, so every ring holds its values in pixel order, as a single
|
||||
// pass would leave them - and they are sorted below in any case.
|
||||
constexpr int BLOCKS = 64;
|
||||
const size_t n = values.size();
|
||||
const auto block_begin = [n](int b) { return n * b / BLOCKS; };
|
||||
const size_t rings = static_cast<size_t>(max_radius) + 1;
|
||||
std::vector<int> cursor(BLOCKS * rings, 0);
|
||||
ParallelFor(BLOCKS, nthreads, [&](int b) {
|
||||
int *count = cursor.data() + b * rings;
|
||||
for (size_t i = block_begin(b); i < block_begin(b + 1); i++)
|
||||
if (valid[i])
|
||||
count[radius[i]]++;
|
||||
});
|
||||
|
||||
RingValues rv;
|
||||
rv.offset.assign(max_radius + 2, 0);
|
||||
for (size_t i = 0; i < values.size(); i++)
|
||||
if (valid[i])
|
||||
rv.offset[radius[i] + 1]++;
|
||||
for (int r = 0; r <= max_radius; r++)
|
||||
rv.offset[r + 1] += rv.offset[r];
|
||||
for (size_t r = 0; r < rings; r++) {
|
||||
int at = rv.offset[r];
|
||||
for (int b = 0; b < BLOCKS; b++) {
|
||||
const int count = cursor[b * rings + r];
|
||||
cursor[b * rings + r] = at;
|
||||
at += count;
|
||||
}
|
||||
rv.offset[r + 1] = at;
|
||||
}
|
||||
|
||||
rv.values.resize(rv.offset[max_radius + 1]);
|
||||
std::vector<int> cursor(rv.offset.begin(), rv.offset.end() - 1);
|
||||
for (size_t i = 0; i < values.size(); i++)
|
||||
if (valid[i])
|
||||
rv.values[cursor[radius[i]]++] = values[i];
|
||||
ParallelFor(BLOCKS, nthreads, [&](int b) {
|
||||
int *next = cursor.data() + b * rings;
|
||||
for (size_t i = block_begin(b); i < block_begin(b + 1); i++)
|
||||
if (valid[i])
|
||||
rv.values[next[radius[i]]++] = values[i];
|
||||
});
|
||||
|
||||
// Sorted once; the three iterations then only pick a rank and count a prefix.
|
||||
ParallelFor(max_radius + 1, nthreads, [&](int r) {
|
||||
@@ -355,7 +411,6 @@ ShadowFinder::ShadowFinder(const DiffractionExperiment &experiment, const PixelM
|
||||
if (pixel_mask.size() != static_cast<size_t>(width) * height)
|
||||
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
|
||||
"ShadowFinder: pixel mask does not match the detector");
|
||||
SetShardCount(1);
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (get_gpu_count() > 0) {
|
||||
const size_t npixels = static_cast<size_t>(width) * height;
|
||||
@@ -372,98 +427,37 @@ void ShadowFinder::BeamCenter(float x, float y) {
|
||||
beam_y = y;
|
||||
}
|
||||
|
||||
// A shard's accumulators are allocated when a frame is first added to it, not here: with a GPU they
|
||||
// are never used at all, and on a 16 Mpx detector eight of them are 2.9 GB to allocate and clear -
|
||||
// which measured 0.8 s of the pre-scan, all of it wasted.
|
||||
void ShadowFinder::SetShardCount(size_t n) {
|
||||
shards.clear();
|
||||
shards.resize(std::max<size_t>(1, n));
|
||||
}
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
ShadowFinder::Projection ShadowFinder::Reduce() const {
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
// The device holds its own projection. Bring it back and let it take part in the fold below as
|
||||
// one more shard; when every frame went to the GPU it is the whole answer.
|
||||
Projection device;
|
||||
if (Gpu() && gpu->GetFrameCount() > 0) {
|
||||
gpu->Download(device.max_value, device.sum_value, device.valid_count);
|
||||
device.frames = gpu->GetFrameCount();
|
||||
bool host_empty = true;
|
||||
for (const auto &p : shards)
|
||||
host_empty = host_empty && (p.frames == 0);
|
||||
if (host_empty)
|
||||
return device;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Only when that shard actually holds something: its accumulators are allocated on first use, so
|
||||
// an unused shard is empty rather than zeroed, and returning it would hand the callers below a
|
||||
// projection they index by pixel.
|
||||
if (shards.size() == 1 && shards[0].frames > 0
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
&& !(gpu && gpu->GetFrameCount() > 0)
|
||||
#endif
|
||||
)
|
||||
return shards[0];
|
||||
|
||||
// The device holds its own projection. Bring it back; when every frame went to the GPU it is the
|
||||
// whole answer.
|
||||
Projection out;
|
||||
const size_t npixels = static_cast<size_t>(width) * height;
|
||||
out.max_value.assign(npixels, 0);
|
||||
out.sum_value.assign(npixels, 0);
|
||||
out.valid_count.assign(npixels, 0);
|
||||
for (const auto &p : shards)
|
||||
out.frames += p.frames;
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
out.frames += device.frames;
|
||||
#endif
|
||||
|
||||
// Each worker owns a slice of the pixels and folds every shard into it. The sums and counts are
|
||||
// integers and a pixel is touched by one worker only, so the result is the same as folding them
|
||||
// one shard at a time on one thread - this is several hundred megabytes per shard and is limited
|
||||
// by memory rather than by arithmetic.
|
||||
const size_t nthreads = std::max<size_t>(1, std::min<size_t>(std::thread::hardware_concurrency(),
|
||||
shards.size() * 2));
|
||||
const size_t chunk = (npixels + nthreads - 1) / nthreads;
|
||||
std::vector<std::future<void>> futures;
|
||||
futures.reserve(nthreads);
|
||||
for (size_t t = 0; t < nthreads; t++) {
|
||||
const size_t lo = t * chunk, hi = std::min(npixels, lo + chunk);
|
||||
if (lo >= hi) break;
|
||||
futures.emplace_back(std::async(std::launch::async, [&, lo, hi] {
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
const Projection *extra[1] = {&device};
|
||||
for (const auto *pp : extra) {
|
||||
const auto &p = *pp;
|
||||
if (p.frames == 0) continue;
|
||||
for (size_t i = lo; i < hi; i++) {
|
||||
if (p.valid_count[i] == 0)
|
||||
continue;
|
||||
if (out.valid_count[i] == 0 || p.max_value[i] > out.max_value[i])
|
||||
out.max_value[i] = p.max_value[i];
|
||||
out.sum_value[i] += p.sum_value[i];
|
||||
out.valid_count[i] += p.valid_count[i];
|
||||
}
|
||||
}
|
||||
#endif
|
||||
for (const auto &p : shards) {
|
||||
if (p.frames == 0) continue; // never used, and its accumulators were never allocated
|
||||
for (size_t i = lo; i < hi; i++) {
|
||||
if (p.valid_count[i] == 0)
|
||||
continue;
|
||||
if (out.valid_count[i] == 0 || p.max_value[i] > out.max_value[i])
|
||||
out.max_value[i] = p.max_value[i];
|
||||
out.sum_value[i] += p.sum_value[i];
|
||||
out.valid_count[i] += p.valid_count[i];
|
||||
}
|
||||
}
|
||||
}));
|
||||
if (Gpu() && gpu->GetFrameCount() > 0) {
|
||||
gpu->Download(out.max_value, out.sum_value, out.valid_count);
|
||||
out.frames = gpu->GetFrameCount();
|
||||
}
|
||||
for (auto &f : futures) f.get();
|
||||
if (host.frames == 0)
|
||||
return out;
|
||||
|
||||
// A pixel is touched by one worker only and the sums and counts are integers, so the result is
|
||||
// the same as folding on one thread.
|
||||
out.frames += host.frames;
|
||||
ParallelChunks(static_cast<int>(out.max_value.size()), std::thread::hardware_concurrency(), [&](int lo, int hi) {
|
||||
for (int i = lo; i < hi; i++) {
|
||||
if (host.valid_count[i] == 0)
|
||||
continue;
|
||||
if (out.valid_count[i] == 0 || host.max_value[i] > out.max_value[i])
|
||||
out.max_value[i] = host.max_value[i];
|
||||
out.sum_value[i] += host.sum_value[i];
|
||||
out.valid_count[i] += host.valid_count[i];
|
||||
}
|
||||
});
|
||||
return out;
|
||||
}
|
||||
#endif
|
||||
|
||||
template<class T>
|
||||
void ShadowFinder::Add(const T *ptr, Projection &p) {
|
||||
void ShadowFinder::Add(const T *ptr, size_t begin, size_t end) {
|
||||
// The pixel type's sentinel extreme marks "no data" (module gap / masked): the
|
||||
// preprocessor/writer stores INT*_MIN for signed and UINT*_MAX for unsigned. For signed
|
||||
// types the opposite extreme is a genuine saturated value and is kept, so a saturated
|
||||
@@ -474,27 +468,23 @@ void ShadowFinder::Add(const T *ptr, Projection &p) {
|
||||
else
|
||||
masked = std::numeric_limits<T>::max();
|
||||
|
||||
for (size_t i = 0; i < p.max_value.size(); i++) {
|
||||
for (size_t i = begin; i < end; i++) {
|
||||
const T v = ptr[i];
|
||||
if (v == masked)
|
||||
continue;
|
||||
const int64_t vi = static_cast<int64_t>(v);
|
||||
if (p.valid_count[i] == 0 || vi > p.max_value[i])
|
||||
p.max_value[i] = vi;
|
||||
p.sum_value[i] += vi;
|
||||
p.valid_count[i]++;
|
||||
if (host.valid_count[i] == 0 || vi > host.max_value[i])
|
||||
host.max_value[i] = vi;
|
||||
host.sum_value[i] += vi;
|
||||
host.valid_count[i]++;
|
||||
}
|
||||
p.frames++;
|
||||
}
|
||||
|
||||
void ShadowFinder::AddImage(const DataMessage &data, std::vector<uint8_t> &buffer, size_t shard) {
|
||||
void ShadowFinder::AddImage(const DataMessage &data, std::vector<uint8_t> &buffer) {
|
||||
if (static_cast<size_t>(data.image.GetWidth()) * data.image.GetHeight()
|
||||
!= static_cast<size_t>(width) * height)
|
||||
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
|
||||
"ShadowFinder: image size does not match the detector");
|
||||
if (shard >= shards.size())
|
||||
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
|
||||
"ShadowFinder: shard out of range");
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
// One device, so the frames queue here - but each is only a chunk upload plus two kernels, and
|
||||
@@ -511,25 +501,37 @@ void ShadowFinder::AddImage(const DataMessage &data, std::vector<uint8_t> &buffe
|
||||
}
|
||||
#endif
|
||||
|
||||
Projection &p = shards[shard];
|
||||
if (p.max_value.empty()) {
|
||||
const size_t npixels = static_cast<size_t>(width) * height;
|
||||
p.max_value.assign(npixels, 0);
|
||||
p.sum_value.assign(npixels, 0);
|
||||
p.valid_count.assign(npixels, 0);
|
||||
const size_t npixels = static_cast<size_t>(width) * height;
|
||||
{
|
||||
std::unique_lock ul(host_mutex);
|
||||
if (host.max_value.empty()) {
|
||||
host.max_value.resize(npixels);
|
||||
host.sum_value.resize(npixels);
|
||||
host.valid_count.resize(npixels);
|
||||
}
|
||||
}
|
||||
const auto ptr = data.image.GetUncompressedPtr(buffer);
|
||||
switch (data.image.GetMode()) {
|
||||
case CompressedImageMode::Int8: Add(reinterpret_cast<const int8_t *>(ptr), p); break;
|
||||
case CompressedImageMode::Uint8: Add(reinterpret_cast<const uint8_t *>(ptr), p); break;
|
||||
case CompressedImageMode::Int16: Add(reinterpret_cast<const int16_t *>(ptr), p); break;
|
||||
case CompressedImageMode::Uint16: Add(reinterpret_cast<const uint16_t *>(ptr), p); break;
|
||||
case CompressedImageMode::Int32: Add(reinterpret_cast<const int32_t *>(ptr), p); break;
|
||||
case CompressedImageMode::Uint32: Add(reinterpret_cast<const uint32_t *>(ptr), p); break;
|
||||
default:
|
||||
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
|
||||
"ShadowFinder: unsupported image mode");
|
||||
const size_t rows_per_band = (static_cast<size_t>(height) + BANDS - 1) / BANDS;
|
||||
const size_t first = next_band.fetch_add(1);
|
||||
for (size_t b = 0; b < BANDS; b++) {
|
||||
const size_t band = (first + b) % BANDS;
|
||||
const size_t begin = std::min(npixels, band * rows_per_band * width);
|
||||
const size_t end = std::min(npixels, (band + 1) * rows_per_band * width);
|
||||
std::lock_guard lock(band_mutex[band]);
|
||||
switch (data.image.GetMode()) {
|
||||
case CompressedImageMode::Int8: Add(reinterpret_cast<const int8_t *>(ptr), begin, end); break;
|
||||
case CompressedImageMode::Uint8: Add(reinterpret_cast<const uint8_t *>(ptr), begin, end); break;
|
||||
case CompressedImageMode::Int16: Add(reinterpret_cast<const int16_t *>(ptr), begin, end); break;
|
||||
case CompressedImageMode::Uint16: Add(reinterpret_cast<const uint16_t *>(ptr), begin, end); break;
|
||||
case CompressedImageMode::Int32: Add(reinterpret_cast<const int32_t *>(ptr), begin, end); break;
|
||||
case CompressedImageMode::Uint32: Add(reinterpret_cast<const uint32_t *>(ptr), begin, end); break;
|
||||
default:
|
||||
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
|
||||
"ShadowFinder: unsupported image mode");
|
||||
}
|
||||
}
|
||||
std::unique_lock ul(host_mutex);
|
||||
host.frames++;
|
||||
}
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
@@ -552,8 +554,7 @@ ShadowAccumulatorGPU *ShadowFinder::Gpu() const {
|
||||
|
||||
uint32_t ShadowFinder::GetFrameCount() const {
|
||||
std::unique_lock ul(m);
|
||||
uint32_t frames = 0;
|
||||
for (const auto &p : shards) frames += p.frames;
|
||||
uint32_t frames = host.frames;
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (gpu) frames += gpu->GetFrameCount();
|
||||
#endif
|
||||
@@ -561,26 +562,39 @@ uint32_t ShadowFinder::GetFrameCount() const {
|
||||
}
|
||||
|
||||
const ShadowFinder::Projection &ShadowFinder::Reduced() const {
|
||||
if (!reduced)
|
||||
reduced = Reduce();
|
||||
return *reduced;
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (gpu && gpu->GetFrameCount() > 0) {
|
||||
if (!reduced)
|
||||
reduced = Reduce();
|
||||
return *reduced;
|
||||
}
|
||||
#endif
|
||||
return host;
|
||||
}
|
||||
|
||||
void ShadowFinder::ReleaseProjection() {
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
std::unique_lock ul(m);
|
||||
reduced.reset();
|
||||
#endif
|
||||
}
|
||||
|
||||
std::vector<float> ShadowFinder::GetMeanProjection() const {
|
||||
std::unique_lock ul(m);
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (Gpu() && gpu->GetFrameCount() > 0 && host.frames == 0)
|
||||
return gpu->MeanProjection(pixel_mask);
|
||||
#endif
|
||||
const Projection &p = Reduced();
|
||||
const auto &sum_value = p.sum_value;
|
||||
const auto &valid_count = p.valid_count;
|
||||
|
||||
std::vector<float> mean(static_cast<size_t>(width) * height, NAN);
|
||||
for (size_t i = 0; i < mean.size(); i++)
|
||||
if (valid_count[i] > 0 && pixel_mask[i] == 0)
|
||||
mean[i] = static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]);
|
||||
std::vector<float> mean(static_cast<size_t>(width) * height);
|
||||
ParallelChunks(static_cast<int>(mean.size()), std::thread::hardware_concurrency(), [&](int lo, int hi) {
|
||||
for (int i = lo; i < hi; i++)
|
||||
mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0
|
||||
? static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]) : NAN;
|
||||
});
|
||||
return mean;
|
||||
}
|
||||
|
||||
@@ -588,6 +602,28 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
|
||||
std::unique_lock ul(m);
|
||||
if (nthreads == 0)
|
||||
nthreads = std::max(1u, std::thread::hardware_concurrency());
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
// Where every frame went to the device the mask is made there too, from the projection as it lies.
|
||||
if (Gpu() && gpu->GetFrameCount() > 0 && host.frames == 0) {
|
||||
const float diag = std::hypot(static_cast<float>(width), static_cast<float>(height));
|
||||
if (!std::isfinite(beam_x) || !std::isfinite(beam_y)
|
||||
|| std::fabs(beam_x - width * 0.5f) > 4.0f * diag || std::fabs(beam_y - height * 0.5f) > 4.0f * diag)
|
||||
return std::vector<uint32_t>(static_cast<size_t>(width) * height, 0);
|
||||
ShadowMaskSetup setup;
|
||||
setup.width = width;
|
||||
setup.height = height;
|
||||
setup.beam_x = beam_x;
|
||||
setup.beam_y = beam_y;
|
||||
const auto rot = geometry.GetDetectorMatrix().arr();
|
||||
for (int k = 0; k < 9; k++)
|
||||
setup.det_matrix[k] = rot[k];
|
||||
setup.pixel_size_mm = geometry.GetPixelSize_mm();
|
||||
setup.distance_mm = geometry.GetDetectorDistance_mm();
|
||||
setup.has_polarization = polarization.has_value();
|
||||
setup.polarization = polarization.value_or(0.0f);
|
||||
return gpu->Mask(setup, pixel_mask);
|
||||
}
|
||||
#endif
|
||||
const Projection &p = Reduced();
|
||||
const auto &max_value = p.max_value;
|
||||
const auto &sum_value = p.sum_value;
|
||||
@@ -727,9 +763,8 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
|
||||
// Innermost rings hold only a handful of pixels, too few to judge, so they are stepped over
|
||||
// rather than allowed to end the walk.
|
||||
std::vector<int> ring_pixels(max_radius + 1, 0);
|
||||
for (int i = 0; i < n_pixels; i++)
|
||||
if (valid[i])
|
||||
ring_pixels[radius[i]]++;
|
||||
for (int rad = 0; rad <= max_radius; rad++)
|
||||
ring_pixels[rad] = rings.offset[rad + 1] - rings.offset[rad];
|
||||
|
||||
// A ring lies inside the stop when its background is a fraction of what this detector's
|
||||
// background typically is. Counting statistics cannot decide this: on a bright dataset the
|
||||
@@ -741,25 +776,7 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
|
||||
// disk it masked. The median over the rings this walk is willing to judge is what "typically"
|
||||
// means, and it is robust from both sides - to a bright ring, and to a corner ring of a handful
|
||||
// of pixels whose median is one pixel's mean.
|
||||
std::vector<float> judgeable;
|
||||
for (int rad = 0; rad <= max_radius; rad++)
|
||||
if (ring_pixels[rad] >= MIN_RING_PIXELS)
|
||||
judgeable.push_back(baseline[rad]);
|
||||
float typical_background = 0.0f;
|
||||
if (!judgeable.empty()) {
|
||||
const auto middle = judgeable.begin() + judgeable.size() / 2;
|
||||
std::nth_element(judgeable.begin(), middle, judgeable.end());
|
||||
typical_background = *middle;
|
||||
}
|
||||
|
||||
int blocked_out_to = -1;
|
||||
for (int rad = 0; rad <= max_radius; rad++) {
|
||||
if (ring_pixels[rad] < MIN_RING_PIXELS)
|
||||
continue;
|
||||
if (baseline[rad] >= BLOCKED_RING_RATIO * typical_background)
|
||||
break;
|
||||
blocked_out_to = rad;
|
||||
}
|
||||
const int blocked_out_to = BlockedOutTo(baseline, ring_pixels);
|
||||
|
||||
// The counts a pixel's pooled background is made of, and the counts the ring says it should
|
||||
// have had. The test is on the deficit between them, in units of its own Poisson scatter.
|
||||
@@ -789,36 +806,20 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
|
||||
// and the stop. What keeps the test specific instead is size, since the background wanders by a
|
||||
// pixel or two at a time and hardware does not.
|
||||
const Plane<char> bridged = dilate(low, W, H, BRIDGE_PX, nthreads);
|
||||
Plane<char> region = filled_plane<char>(n_pixels, 0, nthreads);
|
||||
Plane<char> region(n_pixels);
|
||||
{
|
||||
Plane<char> seen = filled_plane<char>(n_pixels, 0, nthreads);
|
||||
std::vector<int> component;
|
||||
std::queue<int> q;
|
||||
for (int start = 0; start < n_pixels; start++) {
|
||||
if (!bridged[start] || seen[start])
|
||||
continue;
|
||||
component.clear();
|
||||
int n_low = 0;
|
||||
seen[start] = 1;
|
||||
q.push(start);
|
||||
while (!q.empty()) {
|
||||
const int i = q.front(); q.pop();
|
||||
component.push_back(i);
|
||||
n_low += low[i];
|
||||
const int y = i / W, x = i % W;
|
||||
for (int dy = -1; dy <= 1; dy++)
|
||||
for (int dx = -1; dx <= 1; dx++) {
|
||||
const int yy = y + dy, xx = x + dx;
|
||||
if (yy < 0 || yy >= H || xx < 0 || xx >= W)
|
||||
continue;
|
||||
const int j = yy * W + xx;
|
||||
if (bridged[j] && !seen[j]) { seen[j] = 1; q.push(j); }
|
||||
}
|
||||
}
|
||||
if (n_low >= MIN_SHADOW_PIXELS)
|
||||
for (const int i : component)
|
||||
region[i] = low[i];
|
||||
}
|
||||
const auto pieces = label_components(bridged, W, H, nthreads);
|
||||
std::vector<std::atomic<int>> n_low(pieces.count);
|
||||
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
||||
for (int i = lo; i < hi; i++)
|
||||
if (pieces.id[i] >= 0 && low[i])
|
||||
n_low[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
|
||||
});
|
||||
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
||||
for (int i = lo; i < hi; i++)
|
||||
region[i] = pieces.id[i] >= 0 && n_low[pieces.id[i]].load(std::memory_order_relaxed) >= MIN_SHADOW_PIXELS
|
||||
? low[i] : 0;
|
||||
});
|
||||
}
|
||||
|
||||
// The rings that lie wholly inside the stop are decided by the ring walk above rather than by
|
||||
@@ -866,7 +867,7 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
|
||||
|
||||
region = erode(dilate(region, W, H, 2, nthreads), W, H, 2, nthreads);
|
||||
|
||||
region = fill_holes(region, W, H);
|
||||
region = fill_holes(region, W, H, nthreads);
|
||||
|
||||
|
||||
// Expose recorded reflections - done last, with no fill afterwards, so a spot the shadow
|
||||
@@ -903,60 +904,42 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
|
||||
// allowed to explain a dim sector away - where it would ask for more than the median, the median
|
||||
// stands - so the step can only drop what it found before, never find something new.
|
||||
const int n_bands = max_radius / HARMONIC_BAND_PX + 1;
|
||||
std::vector<std::vector<float>> sector_values(static_cast<size_t>(n_bands) * HARMONIC_SECTORS);
|
||||
for (int i = 0; i < n_pixels; i++) {
|
||||
if (!valid[i] || region[i])
|
||||
continue;
|
||||
const float dx = static_cast<float>(i % W) - beam_x, dy = static_cast<float>(i / W) - beam_y;
|
||||
const double phi = std::atan2(dy, dx) + std::numbers::pi;
|
||||
const int sector = std::min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * std::numbers::pi) * HARMONIC_SECTORS));
|
||||
sector_values[static_cast<size_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector].push_back(ratio[i]);
|
||||
}
|
||||
std::vector<float> harm_c(n_bands, 0.0f), harm_s(n_bands, 0.0f);
|
||||
ParallelFor(n_bands, nthreads, [&](int band) {
|
||||
std::vector<double> med(HARMONIC_SECTORS, -1.0), c(HARMONIC_SECTORS), s(HARMONIC_SECTORS);
|
||||
for (int k = 0; k < HARMONIC_SECTORS; k++) {
|
||||
auto &v = sector_values[static_cast<size_t>(band) * HARMONIC_SECTORS + k];
|
||||
if (v.size() >= MIN_SECTOR_PIXELS) {
|
||||
std::nth_element(v.begin(), v.begin() + v.size() / 2, v.end());
|
||||
med[k] = v[v.size() / 2];
|
||||
}
|
||||
const double phi = (k + 0.5) * 2.0 * std::numbers::pi / HARMONIC_SECTORS - std::numbers::pi;
|
||||
c[k] = std::cos(2 * phi);
|
||||
s[k] = std::sin(2 * phi);
|
||||
}
|
||||
// Least squares of med = m + p cos 2phi + q sin 2phi over the sectors that are not themselves
|
||||
// dim against the fit, three times over as the baseline is; the model is relative, 1 + (p cos + q sin)/m.
|
||||
std::vector<char> use(HARMONIC_SECTORS);
|
||||
for (int k = 0; k < HARMONIC_SECTORS; k++)
|
||||
use[k] = med[k] >= 0;
|
||||
for (int iter = 0; iter < 3; iter++) {
|
||||
double n = 0, sc = 0, ss = 0, scc = 0, sss = 0, scs = 0, y = 0, yc = 0, ys = 0;
|
||||
for (int k = 0; k < HARMONIC_SECTORS; k++) {
|
||||
if (!use[k]) continue;
|
||||
n++; sc += c[k]; ss += s[k]; scc += c[k] * c[k]; sss += s[k] * s[k]; scs += c[k] * s[k];
|
||||
y += med[k]; yc += med[k] * c[k]; ys += med[k] * s[k];
|
||||
}
|
||||
// Sectors crowded into a narrow arc cannot tell a harmonic from a level; the determinant of
|
||||
// the normal matrix, per sector cubed, is 1/4 on a full ring.
|
||||
const double det = n * (scc * sss - scs * scs) - sc * (sc * sss - scs * ss) + ss * (sc * scs - scc * ss);
|
||||
if (n < 6 || det < 0.01 * n * n * n) {
|
||||
harm_c[band] = harm_s[band] = 0.0f;
|
||||
return;
|
||||
}
|
||||
const double m = (y * (scc * sss - scs * scs) - sc * (yc * sss - scs * ys) + ss * (yc * scs - scc * ys)) / det;
|
||||
const double p = (n * (yc * sss - ys * scs) - y * (sc * sss - scs * ss) + ss * (sc * ys - yc * ss)) / det;
|
||||
const double q = (n * (scc * ys - scs * yc) - sc * (sc * ys - yc * ss) + y * (sc * scs - scc * ss)) / det;
|
||||
if (m <= 0) {
|
||||
harm_c[band] = harm_s[band] = 0.0f;
|
||||
return;
|
||||
}
|
||||
harm_c[band] = static_cast<float>(p / m);
|
||||
harm_s[band] = static_cast<float>(q / m);
|
||||
for (int k = 0; k < HARMONIC_SECTORS; k++)
|
||||
use[k] = med[k] >= 0 && med[k] >= PENUMBRA_RATIO * (1.0 + harm_c[band] * c[k] + harm_s[band] * s[k]);
|
||||
const size_t n_sectors = static_cast<size_t>(n_bands) * HARMONIC_SECTORS;
|
||||
// Gathered by blocks of rows in parallel and joined in block order. Only the median of each
|
||||
// sector is read, and that is the same whatever order its values were gathered in.
|
||||
constexpr int SECTOR_BLOCKS = 64;
|
||||
std::vector<std::vector<std::vector<float>>> block_values(SECTOR_BLOCKS);
|
||||
ParallelFor(SECTOR_BLOCKS, nthreads, [&](int b) {
|
||||
auto &values = block_values[b];
|
||||
values.resize(n_sectors);
|
||||
const int lo = static_cast<int>(static_cast<int64_t>(n_pixels) * b / SECTOR_BLOCKS);
|
||||
const int hi = static_cast<int>(static_cast<int64_t>(n_pixels) * (b + 1) / SECTOR_BLOCKS);
|
||||
for (int i = lo; i < hi; i++) {
|
||||
if (!valid[i] || region[i])
|
||||
continue;
|
||||
const float dx = static_cast<float>(i % W) - beam_x, dy = static_cast<float>(i / W) - beam_y;
|
||||
const double phi = std::atan2(dy, dx) + std::numbers::pi;
|
||||
const int sector = std::min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * std::numbers::pi) * HARMONIC_SECTORS));
|
||||
values[static_cast<size_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector].push_back(ratio[i]);
|
||||
}
|
||||
});
|
||||
std::vector<std::vector<float>> sector_values(n_sectors);
|
||||
ParallelFor(static_cast<int>(n_sectors), nthreads, [&](int k) {
|
||||
for (const auto &values : block_values)
|
||||
sector_values[k].insert(sector_values[k].end(), values[k].begin(), values[k].end());
|
||||
});
|
||||
block_values.clear();
|
||||
// The median of each sector with enough pixels to have one.
|
||||
std::vector<double> sector_median(n_sectors, -1.0);
|
||||
ParallelFor(static_cast<int>(n_sectors), nthreads, [&](int k) {
|
||||
auto &v = sector_values[k];
|
||||
if (v.size() >= MIN_SECTOR_PIXELS) {
|
||||
std::nth_element(v.begin(), v.begin() + v.size() / 2, v.end());
|
||||
sector_median[k] = v[v.size() / 2];
|
||||
}
|
||||
});
|
||||
std::vector<float> harm_c, harm_s;
|
||||
HarmonicFit(sector_median, n_bands, harm_c, harm_s);
|
||||
|
||||
Plane<char> dim = filled_plane<char>(n_pixels, 0, nthreads);
|
||||
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
||||
@@ -978,36 +961,108 @@ std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
|
||||
&& poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA;
|
||||
}
|
||||
});
|
||||
const Plane<char> joined = bridge_gaps(dilate(dim, W, H, BRIDGE_PX, nthreads), valid, W, H);
|
||||
Plane<char> seen = filled_plane<char>(n_pixels, 0, nthreads);
|
||||
std::vector<int> component;
|
||||
std::queue<int> q;
|
||||
for (int start = 0; start < n_pixels; start++) {
|
||||
if (!joined[start] || seen[start])
|
||||
continue;
|
||||
component.clear();
|
||||
int n_dim = 0, n_low = 0;
|
||||
seen[start] = 1;
|
||||
q.push(start);
|
||||
while (!q.empty()) {
|
||||
const int i = q.front(); q.pop();
|
||||
component.push_back(i);
|
||||
n_dim += dim[i];
|
||||
n_low += dim[i] && low[i];
|
||||
const int y = i / W, x = i % W;
|
||||
for (int dy = -1; dy <= 1; dy++)
|
||||
for (int dx = -1; dx <= 1; dx++) {
|
||||
const int yy = y + dy, xx = x + dx;
|
||||
if (yy < 0 || yy >= H || xx < 0 || xx >= W)
|
||||
continue;
|
||||
const int j = yy * W + xx;
|
||||
if (joined[j] && !seen[j]) { seen[j] = 1; q.push(j); }
|
||||
}
|
||||
const Plane<char> joined = bridge_gaps(dilate(dim, W, H, BRIDGE_PX, nthreads), valid, W, H, nthreads);
|
||||
const auto pieces = label_components(joined, W, H, nthreads);
|
||||
std::vector<std::atomic<int>> n_dim(pieces.count), n_low(pieces.count);
|
||||
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
||||
for (int i = lo; i < hi; i++)
|
||||
if (pieces.id[i] >= 0 && dim[i]) {
|
||||
n_dim[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
|
||||
if (low[i])
|
||||
n_low[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
|
||||
}
|
||||
});
|
||||
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
||||
for (int i = lo; i < hi; i++) {
|
||||
const int c = pieces.id[i];
|
||||
if (c >= 0 && dim[i] && n_dim[c].load(std::memory_order_relaxed) >= MIN_SHADOW_PIXELS
|
||||
&& n_low[c].load(std::memory_order_relaxed) >= MIN_CORE_PIXELS)
|
||||
mask[i] = TRANSMITTING;
|
||||
}
|
||||
if (n_dim >= MIN_SHADOW_PIXELS && n_low >= MIN_CORE_PIXELS)
|
||||
for (const int i : component)
|
||||
if (dim[i])
|
||||
mask[i] = TRANSMITTING;
|
||||
}
|
||||
});
|
||||
return mask;
|
||||
}
|
||||
|
||||
namespace shadow_finder {
|
||||
|
||||
int BlockedOutTo(const std::vector<float> &baseline, const std::vector<int> &ring_pixels) {
|
||||
const int max_radius = static_cast<int>(baseline.size()) - 1;
|
||||
// A ring lies inside the stop when its background is a fraction of what this detector's
|
||||
// background typically is. Counting statistics cannot decide this: on a bright dataset the
|
||||
// shadow is still well counted. The comparison used to be against the LARGEST background of any
|
||||
// ring further out, and that reads a sample whose background peaks in a strong ring away from
|
||||
// the beam - a powder standard, a strong solvent ring - as a beam stop the size of that ring:
|
||||
// the ordinary background inside it is legitimately below a third of the peak. On one corpus
|
||||
// dataset it declared 16 % of the detector to be stop, with diffraction rings visible inside the
|
||||
// disk it masked. The median over the rings this walk is willing to judge is what "typically"
|
||||
// means, and it is robust from both sides - to a bright ring, and to a corner ring of a handful
|
||||
// of pixels whose median is one pixel's mean.
|
||||
std::vector<float> judgeable;
|
||||
for (int rad = 0; rad <= max_radius; rad++)
|
||||
if (ring_pixels[rad] >= MIN_RING_PIXELS)
|
||||
judgeable.push_back(baseline[rad]);
|
||||
float typical_background = 0.0f;
|
||||
if (!judgeable.empty()) {
|
||||
const auto middle = judgeable.begin() + judgeable.size() / 2;
|
||||
std::nth_element(judgeable.begin(), middle, judgeable.end());
|
||||
typical_background = *middle;
|
||||
}
|
||||
|
||||
int blocked_out_to = -1;
|
||||
for (int rad = 0; rad <= max_radius; rad++) {
|
||||
if (ring_pixels[rad] < MIN_RING_PIXELS)
|
||||
continue;
|
||||
if (baseline[rad] >= BLOCKED_RING_RATIO * typical_background)
|
||||
break;
|
||||
blocked_out_to = rad;
|
||||
}
|
||||
return blocked_out_to;
|
||||
}
|
||||
|
||||
void HarmonicFit(const std::vector<double> §or_median, int n_bands,
|
||||
std::vector<float> &harm_c, std::vector<float> &harm_s) {
|
||||
harm_c.assign(n_bands, 0.0f);
|
||||
harm_s.assign(n_bands, 0.0f);
|
||||
for (int band = 0; band < n_bands; band++) {
|
||||
std::vector<double> med(HARMONIC_SECTORS), c(HARMONIC_SECTORS), s(HARMONIC_SECTORS);
|
||||
for (int k = 0; k < HARMONIC_SECTORS; k++) {
|
||||
med[k] = sector_median[static_cast<size_t>(band) * HARMONIC_SECTORS + k];
|
||||
const double phi = (k + 0.5) * 2.0 * std::numbers::pi / HARMONIC_SECTORS - std::numbers::pi;
|
||||
c[k] = std::cos(2 * phi);
|
||||
s[k] = std::sin(2 * phi);
|
||||
}
|
||||
// Least squares of med = m + p cos 2phi + q sin 2phi over the sectors that are not themselves
|
||||
// dim against the fit, three times over as the baseline is; the model is relative, 1 + (p cos + q sin)/m.
|
||||
std::vector<char> use(HARMONIC_SECTORS);
|
||||
for (int k = 0; k < HARMONIC_SECTORS; k++)
|
||||
use[k] = med[k] >= 0;
|
||||
for (int iter = 0; iter < 3; iter++) {
|
||||
double n = 0, sc = 0, ss = 0, scc = 0, sss = 0, scs = 0, y = 0, yc = 0, ys = 0;
|
||||
for (int k = 0; k < HARMONIC_SECTORS; k++) {
|
||||
if (!use[k]) continue;
|
||||
n++; sc += c[k]; ss += s[k]; scc += c[k] * c[k]; sss += s[k] * s[k]; scs += c[k] * s[k];
|
||||
y += med[k]; yc += med[k] * c[k]; ys += med[k] * s[k];
|
||||
}
|
||||
// Sectors crowded into a narrow arc cannot tell a harmonic from a level; the determinant of
|
||||
// the normal matrix, per sector cubed, is 1/4 on a full ring.
|
||||
const double det = n * (scc * sss - scs * scs) - sc * (sc * sss - scs * ss) + ss * (sc * scs - scc * ss);
|
||||
if (n < 6 || det < 0.01 * n * n * n) {
|
||||
harm_c[band] = harm_s[band] = 0.0f;
|
||||
break;
|
||||
}
|
||||
const double m = (y * (scc * sss - scs * scs) - sc * (yc * sss - scs * ys) + ss * (yc * scs - scc * ys)) / det;
|
||||
const double p = (n * (yc * sss - ys * scs) - y * (sc * sss - scs * ss) + ss * (sc * ys - yc * ss)) / det;
|
||||
const double q = (n * (scc * ys - scs * yc) - sc * (sc * ys - yc * ss) + y * (sc * scs - scc * ss)) / det;
|
||||
if (m <= 0) {
|
||||
harm_c[band] = harm_s[band] = 0.0f;
|
||||
break;
|
||||
}
|
||||
harm_c[band] = static_cast<float>(p / m);
|
||||
harm_s[band] = static_cast<float>(q / m);
|
||||
for (int k = 0; k < HARMONIC_SECTORS; k++)
|
||||
use[k] = med[k] >= 0 && med[k] >= PENUMBRA_RATIO * (1.0 + harm_c[band] * c[k] + harm_s[band] * s[k]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace shadow_finder
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <atomic>
|
||||
#include <future>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
@@ -36,9 +37,9 @@
|
||||
//
|
||||
// Frames are chosen by the caller; the detection needs enough of them that the background
|
||||
// is counted rather than guessed (see MIN_EXPECTED_COUNTS in the .cpp).
|
||||
// Thread-safe: workers call AddImage concurrently, each naming a shard of its own (see
|
||||
// SetShardCount) - so no two threads touch the same accumulator and nothing is locked while
|
||||
// an image is added. The shards are summed when the projection is read.
|
||||
// Thread-safe: workers call AddImage concurrently. The projection is split into bands of rows, each
|
||||
// with a lock of its own, and a worker adding a frame starts at a different band from the one before
|
||||
// it, so workers meet only when they reach the same band.
|
||||
class ShadowFinder {
|
||||
mutable std::mutex m;
|
||||
|
||||
@@ -62,22 +63,27 @@ class ShadowFinder {
|
||||
|
||||
std::vector<uint32_t> pixel_mask; // pixels already masked carry no background to test
|
||||
|
||||
// Per-pixel projection over the frames added so far (converted geometry). One set per shard:
|
||||
// the sums and counts are integers, so summing the shards is exact and the result does not
|
||||
// depend on how the frames were spread over them.
|
||||
// Per-pixel projection over the frames added so far (converted geometry). The sums and counts are
|
||||
// integers and the maximum is a maximum, so the result does not depend on the order the frames
|
||||
// arrive in.
|
||||
struct Projection {
|
||||
std::vector<int64_t> max_value;
|
||||
std::vector<int64_t> sum_value;
|
||||
std::vector<uint32_t> valid_count;
|
||||
uint32_t frames = 0;
|
||||
};
|
||||
std::vector<Projection> shards;
|
||||
// The frames added on the host. Allocated by the first of them: with a GPU there are usually none.
|
||||
Projection host;
|
||||
std::mutex host_mutex; // guards the allocation and the frame count, not the sums
|
||||
static constexpr size_t BANDS = 64;
|
||||
std::mutex band_mutex[BANDS];
|
||||
std::atomic<size_t> next_band{0};
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
// Present when a GPU is available. Frames it can decode are accumulated there instead of on the
|
||||
// host - only the compressed chunk crosses PCIe - and its projection is folded in with the
|
||||
// shards when the mask is read. Frames it cannot take (anything but bitshuffle+LZ4) still go to
|
||||
// a host shard, so a run mixing compressions is handled without a second code path.
|
||||
// host projection when the mask is read. Frames it cannot take (anything but bitshuffle+LZ4) still
|
||||
// go to the host, so a run mixing compressions is handled without a second code path.
|
||||
// Built on a thread of its own: it allocates and clears several hundred megabytes of device
|
||||
// memory, and cudaMalloc synchronises the whole device, so doing it in the constructor would
|
||||
// stall the caller before it has read its first frame. The first AddImage waits for it, by
|
||||
@@ -90,17 +96,22 @@ class ShadowFinder {
|
||||
[[nodiscard]] ShadowAccumulatorGPU *Gpu() const;
|
||||
#endif
|
||||
|
||||
template<class T> void Add(const T *ptr, Projection &p);
|
||||
// Add the pixels [begin, end) of one frame to the host projection.
|
||||
template<class T> void Add(const T *ptr, size_t begin, size_t end);
|
||||
|
||||
// Sum the shards into one projection. max_value is only taken from a shard that actually
|
||||
// counted the pixel - a shard that never saw it holds 0, which would beat a genuinely
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
// The device's projection with the host's folded in. max_value is only taken from a projection
|
||||
// that actually counted the pixel - one that never saw it holds 0, which would beat a genuinely
|
||||
// negative maximum.
|
||||
[[nodiscard]] Projection Reduce() const;
|
||||
|
||||
// The projection, reduced on its first read and kept. The ring-centre fit, the mask and the
|
||||
// beam-centre capture all read the same one, and on a 16 Mpx detector each reduction is 360 MB
|
||||
// brought back from the device into fresh memory. Called with `m` held.
|
||||
// That projection, made on its first read and kept. The ring-centre fit, the mask and the
|
||||
// beam-centre capture all read the same one, and on a 16 Mpx detector each is 360 MB brought back
|
||||
// from the device into fresh memory.
|
||||
mutable std::optional<Projection> reduced;
|
||||
#endif
|
||||
// The projection the frames added so far make: the host's, or the one above where the device
|
||||
// took frames. Called with `m` held.
|
||||
[[nodiscard]] const Projection &Reduced() const;
|
||||
|
||||
public:
|
||||
@@ -116,20 +127,16 @@ public:
|
||||
// hardware. The projection is not centred on anything, so this may be set after the frames.
|
||||
void BeamCenter(float x, float y);
|
||||
|
||||
// Give each worker a shard to accumulate into. Must be called before the first AddImage,
|
||||
// and costs 20 bytes per pixel per shard.
|
||||
void SetShardCount(size_t n);
|
||||
|
||||
// Accumulate one full converted-geometry image into shard `shard`. Gap / masked pixels
|
||||
// (the pixel type's sentinel extreme) are skipped. `buffer` is scratch space for
|
||||
// decompression, reused across the calls of one worker.
|
||||
void AddImage(const DataMessage &data, std::vector<uint8_t> &buffer, size_t shard = 0);
|
||||
// Accumulate one full converted-geometry image. Gap / masked pixels (the pixel type's sentinel
|
||||
// extreme) are skipped. `buffer` is scratch space for decompression, reused across the calls of
|
||||
// one worker.
|
||||
void AddImage(const DataMessage &data, std::vector<uint8_t> &buffer);
|
||||
|
||||
// Compute the shadow mask (SHADOW, TRANSMITTING or 0 = keep), of the converted pixel count.
|
||||
// TRANSMITTING marks the pieces of hardware that let part of the beam through, added after the
|
||||
// shadow proper; both are masked, and a consumer that must not see those pieces can tell them apart.
|
||||
// Recomputed on each call from the projection, which is summed over the shards on the first read
|
||||
// of it (GetMask or GetMeanProjection): frames added after that are not seen.
|
||||
// Recomputed on each call from the projection, which is put together on the first read of it
|
||||
// (GetMask or GetMeanProjection): frames added after that are not seen.
|
||||
// nthreads = 0 asks for all hardware threads. The per-pixel passes over a 16M-pixel detector
|
||||
// dominate this, and they are all exactly parallel.
|
||||
[[nodiscard]] std::vector<uint32_t> GetMask(size_t nthreads = 0) const;
|
||||
@@ -140,7 +147,7 @@ public:
|
||||
[[nodiscard]] std::vector<float> GetMeanProjection() const;
|
||||
|
||||
// Let go of the projection the two above read, once the caller has what it wants of it: on a
|
||||
// 16 Mpx detector it is 360 MB. A later read sums the shards again.
|
||||
// 16 Mpx detector it is 360 MB. A later read puts it together again.
|
||||
void ReleaseProjection();
|
||||
|
||||
[[nodiscard]] uint32_t GetFrameCount() const;
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
// What ShadowFinder::GetMask shares with its device twin (ShadowMaskGPU): the constants it tests
|
||||
// against, and the two small fits made over whole rings and sectors, which stay on the host on both
|
||||
// paths.
|
||||
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
namespace shadow_finder {
|
||||
|
||||
// The values of ShadowFinder::SHADOW and ShadowFinder::TRANSMITTING, for the device code, which does
|
||||
// not include ShadowFinder.h.
|
||||
inline constexpr uint32_t MASK_SHADOW = 1;
|
||||
inline constexpr uint32_t MASK_TRANSMITTING = 2;
|
||||
|
||||
// A pixel is shadow when its background is below this fraction of the background it is
|
||||
// compared against.
|
||||
inline constexpr float SHADOW_RATIO = 0.50f;
|
||||
|
||||
// The boundary grows outward into partially shadowed pixels down to this fraction, but no
|
||||
// further than PENUMBRA_MAX_PX from the core. A pin or a loop casts a wide half-shadow, so the
|
||||
// reach is a good deal more than the beam stop's own edge needs.
|
||||
inline constexpr float PENUMBRA_RATIO = 0.75f;
|
||||
inline constexpr int PENUMBRA_MAX_PX = 30;
|
||||
|
||||
// Bridge small breaks along the holder arm. A module gap wider than this is bridged separately for
|
||||
// the arm search (see bridge_gaps).
|
||||
inline constexpr int BRIDGE_PX = 6;
|
||||
|
||||
// A pixel whose maximum reaches this recorded a real reflection and is never masked - a
|
||||
// beam stop cannot block a reflection that was measured.
|
||||
inline constexpr int64_t MIN_REFLECTION = 25;
|
||||
|
||||
// How far below the background it is compared against a pixel must sit before the dip is
|
||||
// believed, in standard deviations of the counts that back it. The counts are photons, so their
|
||||
// scatter is Poisson and the deficit is measured against it rather than against a fixed number:
|
||||
// on a well-exposed sweep a third of the background missing is overwhelming, and on a handful of
|
||||
// low-background frames the same third is noise. Without this a six-frame pre-scan of a
|
||||
// low-background sweep masks three quarters of the detector.
|
||||
inline constexpr double MIN_DEFICIT_SIGMA = 6.0;
|
||||
|
||||
// Smallest region the per-pixel test may return. A shadow is cast by something physical and is
|
||||
// correspondingly large; an isolated patch this small is the background wandering, not hardware.
|
||||
// This is what keeps the test specific now that a shadow no longer has to touch the direct beam.
|
||||
inline constexpr int MIN_SHADOW_PIXELS = 2000;
|
||||
|
||||
// ... and of those, how many must be deep (below SHADOW_RATIO) for a region of merely DIM pixels -
|
||||
// hardware that lets part of the beam through - to count. A thin holder arm is dim along most of
|
||||
// its length and deep in places; the background drifting over a detector's edge is dim everywhere
|
||||
// and deep nowhere.
|
||||
inline constexpr int MIN_CORE_PIXELS = 200;
|
||||
|
||||
// The arm search compares a pixel with its ring as the ring actually varies around the beam. A ring
|
||||
// of background is not flat once divided by the polarization factor when that factor is not the
|
||||
// beam's: the remainder is a second harmonic in azimuth, cos 2phi, which reaches tens of percent at
|
||||
// high angle. It is measured over radial bands of HARMONIC_BAND_PX, from the median of each of
|
||||
// HARMONIC_SECTORS sectors that holds at least MIN_SECTOR_PIXELS pixels.
|
||||
inline constexpr int HARMONIC_BAND_PX = 64;
|
||||
inline constexpr int HARMONIC_SECTORS = 24;
|
||||
inline constexpr int MIN_SECTOR_PIXELS = 200;
|
||||
|
||||
// Side of the box the background is pooled over before testing. Its area is how many pixels back
|
||||
// a ring's countability test, which decides where an azimuthal comparison is possible at all.
|
||||
inline constexpr int POOL_PX = 5;
|
||||
inline constexpr double MEAN_POOLED_PIXELS = POOL_PX * POOL_PX;
|
||||
|
||||
// A ring with fewer valid pixels than this says nothing about whether it was counted.
|
||||
inline constexpr int MIN_RING_PIXELS = 32;
|
||||
|
||||
// A ring lies wholly inside the stop when its background is below this fraction of the background
|
||||
// further out. This asks about a whole ring rather than about a pixel, so it keeps a threshold of
|
||||
// its own and does not follow SHADOW_RATIO.
|
||||
inline constexpr float BLOCKED_RING_RATIO = 0.35f;
|
||||
|
||||
// The rings that lie wholly inside the stop: walking outward, every judgeable ring (at least
|
||||
// MIN_RING_PIXELS pixels) before the first whose baseline reaches BLOCKED_RING_RATIO of the typical
|
||||
// background. -1 when there is none. See GetMask.
|
||||
int BlockedOutTo(const std::vector<float> &baseline, const std::vector<int> &ring_pixels);
|
||||
|
||||
// The second harmonic in azimuth of each radial band, relative to its level, fitted to the medians of
|
||||
// its HARMONIC_SECTORS sectors (sector_median[band * HARMONIC_SECTORS + k], negative where the sector
|
||||
// has too few pixels). See GetMask.
|
||||
void HarmonicFit(const std::vector<double> §or_median, int n_bands,
|
||||
std::vector<float> &harm_c, std::vector<float> &harm_s);
|
||||
|
||||
} // namespace shadow_finder
|
||||
@@ -0,0 +1,636 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#include "ShadowMaskGPU.h"
|
||||
|
||||
#include <cub/device/device_radix_sort.cuh>
|
||||
|
||||
#include "ShadowFinderInternal.h"
|
||||
#include "../../common/JFJochMath.h"
|
||||
#include "../indexing/CUDAMemHelpers.h"
|
||||
#include "../../common/JFJochException.h"
|
||||
|
||||
using namespace shadow_finder;
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr int THREADS = 256;
|
||||
|
||||
void check(cudaError_t err, const char *what) {
|
||||
if (err != cudaSuccess)
|
||||
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
|
||||
std::string("Beam stop mask: ") + what + ": " + cudaGetErrorString(err));
|
||||
}
|
||||
|
||||
unsigned grid(size_t n) {
|
||||
return static_cast<unsigned>((n + THREADS - 1) / THREADS);
|
||||
}
|
||||
|
||||
// A float as an unsigned integer that sorts the same way, and back.
|
||||
__device__ uint32_t float_key(float f) {
|
||||
const uint32_t u = __float_as_uint(f);
|
||||
return (u & 0x80000000u) ? ~u : (u | 0x80000000u);
|
||||
}
|
||||
__device__ float key_float(uint32_t k) {
|
||||
return __uint_as_float((k & 0x80000000u) ? (k & 0x7fffffffu) : ~k);
|
||||
}
|
||||
|
||||
// The first index in a sorted key array whose key is not below `value`.
|
||||
__device__ size_t lower_bound(const uint64_t *keys, size_t n, uint64_t value) {
|
||||
size_t lo = 0, hi = n;
|
||||
while (lo < hi) {
|
||||
const size_t mid = (lo + hi) / 2;
|
||||
if (keys[mid] < value) lo = mid + 1;
|
||||
else hi = mid;
|
||||
}
|
||||
return lo;
|
||||
}
|
||||
|
||||
__device__ double poisson_deficit_sigma(double observed, double expected) {
|
||||
if (expected <= 0.0 || observed >= expected)
|
||||
return 0.0;
|
||||
const double ll = 2.0 * (expected - observed + (observed > 0.0 ? observed * log(observed / expected) : 0.0));
|
||||
return ll > 0.0 ? sqrt(ll) : 0.0;
|
||||
}
|
||||
|
||||
// DiffractionGeometry::CalcAzIntPolarizationCorr about the centre the rings are drawn about.
|
||||
__device__ float polarization_factor(const ShadowMaskSetup &s, float x, float y) {
|
||||
const float u = (x - s.beam_x) * s.pixel_size_mm;
|
||||
const float v = (y - s.beam_y) * s.pixel_size_mm;
|
||||
const float *m = s.det_matrix;
|
||||
const float lx = m[0] * u + m[1] * v + m[2] * s.distance_mm;
|
||||
const float ly = m[3] * u + m[4] * v + m[5] * s.distance_mm;
|
||||
const float lz = m[6] * u + m[7] * v + m[8] * s.distance_mm;
|
||||
const float two_theta = atan2f(sqrtf(lx * lx + ly * ly), lz);
|
||||
float phi = atan2f(ly, lx);
|
||||
if (phi < 0)
|
||||
phi += 2.0f * PI;
|
||||
const float cos_2theta = cosf(two_theta);
|
||||
const float cos_2theta_2 = cos_2theta * cos_2theta;
|
||||
const float cos_2phi = cosf(2.0f * phi);
|
||||
return 0.5f * (1.0f + cos_2theta_2 - s.polarization * cos_2phi * (1.0f - cos_2theta_2));
|
||||
}
|
||||
|
||||
__global__ void setup_kernel(ShadowMaskSetup s, const uint32_t *__restrict__ pixel_mask,
|
||||
const int64_t *__restrict__ sum_value, const uint32_t *__restrict__ valid_count,
|
||||
float *__restrict__ pol, char *__restrict__ valid, int *__restrict__ radius,
|
||||
double *__restrict__ num, int32_t *__restrict__ den, int *__restrict__ max_radius) {
|
||||
const size_t n = static_cast<size_t>(s.width) * s.height;
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n)
|
||||
return;
|
||||
const int x = static_cast<int>(i % s.width), y = static_cast<int>(i / s.width);
|
||||
const float dx = x - s.beam_x, dy = y - s.beam_y;
|
||||
const float p = s.has_polarization ? polarization_factor(s, static_cast<float>(x), static_cast<float>(y)) : 1.0f;
|
||||
pol[i] = p;
|
||||
float mean = 0.0f;
|
||||
char v = 0;
|
||||
if (valid_count[i] > 0 && pixel_mask[i] == 0 && p > 0.0f) {
|
||||
mean = static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i] / p);
|
||||
v = 1;
|
||||
}
|
||||
valid[i] = v;
|
||||
num[i] = v ? mean : 0.0;
|
||||
den[i] = v ? 1 : 0;
|
||||
const int r = static_cast<int>(lroundf(sqrtf(dx * dx + dy * dy)));
|
||||
radius[i] = r;
|
||||
atomicMax(max_radius, r);
|
||||
}
|
||||
|
||||
// Sum over the k x k box about each pixel, zero outside the frame: one running sum per row, then one
|
||||
// per column, each with exactly the terms and order of the host's (box_sum in ShadowFinder.cpp).
|
||||
template <typename T>
|
||||
__global__ void box_rows(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) {
|
||||
const int y = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (y >= H) return;
|
||||
const T *src = in + static_cast<size_t>(y) * W;
|
||||
T *dst = out + static_cast<size_t>(y) * W;
|
||||
T s = 0;
|
||||
for (int x = 0; x <= min(half, W - 1); x++)
|
||||
s += src[x];
|
||||
for (int x = 0; x < W; x++) {
|
||||
dst[x] = s;
|
||||
if (x + half + 1 < W) s += src[x + half + 1];
|
||||
if (x - half >= 0) s -= src[x - half];
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void box_columns(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) {
|
||||
const int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (x >= W) return;
|
||||
T s = 0;
|
||||
for (int y = 0; y <= min(half, H - 1); y++)
|
||||
s += in[static_cast<size_t>(y) * W + x];
|
||||
for (int y = 0; y < H; y++) {
|
||||
out[static_cast<size_t>(y) * W + x] = s;
|
||||
if (y + half + 1 < H) s += in[static_cast<size_t>(y + half + 1) * W + x];
|
||||
if (y - half >= 0) s -= in[static_cast<size_t>(y - half) * W + x];
|
||||
}
|
||||
}
|
||||
|
||||
// Dilation of a 0/1 plane by the (2r+1) square clipped to the frame, as a count over a sliding window
|
||||
// along rows and then along columns (dilate in ShadowFinder.cpp).
|
||||
__global__ void dilate_rows(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) {
|
||||
const int y = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (y >= H) return;
|
||||
const char *src = in + static_cast<size_t>(y) * W;
|
||||
char *dst = out + static_cast<size_t>(y) * W;
|
||||
int count = 0;
|
||||
for (int x = 0; x <= min(r, W - 1); x++)
|
||||
count += src[x];
|
||||
for (int x = 0; x < W; x++) {
|
||||
dst[x] = count > 0;
|
||||
if (x + r + 1 < W) count += src[x + r + 1];
|
||||
if (x - r >= 0) count -= src[x - r];
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void dilate_columns(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) {
|
||||
const int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (x >= W) return;
|
||||
int count = 0;
|
||||
for (int y = 0; y <= min(r, H - 1); y++)
|
||||
count += in[static_cast<size_t>(y) * W + x];
|
||||
for (int y = 0; y < H; y++) {
|
||||
out[static_cast<size_t>(y) * W + x] = count > 0;
|
||||
if (y + r + 1 < H) count += in[static_cast<size_t>(y + r + 1) * W + x];
|
||||
if (y - r >= 0) count -= in[static_cast<size_t>(y - r) * W + x];
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void pooled_kernel(size_t n, const double *__restrict__ pooled_sum, const int32_t *__restrict__ pooled_count,
|
||||
const char *__restrict__ valid, const int *__restrict__ radius,
|
||||
float *__restrict__ pooled, uint64_t *__restrict__ ring_key) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
const float p = pooled_count[i] > 0 ? static_cast<float>(pooled_sum[i] / pooled_count[i]) : 0.0f;
|
||||
pooled[i] = p;
|
||||
ring_key[i] = valid[i] ? (static_cast<uint64_t>(radius[i]) << 32) | float_key(p) : UINT64_MAX;
|
||||
}
|
||||
|
||||
// Where each of the keys' leading 32-bit groups (ring or sector) starts in a sorted key array.
|
||||
__global__ void group_offsets(const uint64_t *__restrict__ keys, size_t n, int groups, int *__restrict__ offset) {
|
||||
const int g = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (g > groups) return;
|
||||
offset[g] = static_cast<int>(lower_bound(keys, n, static_cast<uint64_t>(g) << 32));
|
||||
}
|
||||
|
||||
// The ring's baseline, three iterations of an order statistic over its sorted values (GetMask).
|
||||
__global__ void baseline_kernel(int rings, const uint64_t *__restrict__ keys, const int *__restrict__ offset,
|
||||
float *__restrict__ baseline) {
|
||||
const int r = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (r >= rings) return;
|
||||
const int lo = offset[r], n = offset[r + 1] - offset[r];
|
||||
int excluded = 0;
|
||||
float b = 0.0f;
|
||||
for (int iter = 0; iter < 3; iter++) {
|
||||
const int avail = n - excluded;
|
||||
b = (avail <= 0) ? 0.0f : key_float(static_cast<uint32_t>(keys[lo + excluded + avail / 2]));
|
||||
const float d = fmaxf(b, 1e-6f);
|
||||
int excl = 0;
|
||||
while (excl < n && key_float(static_cast<uint32_t>(keys[lo + excl])) / d < SHADOW_RATIO)
|
||||
excl++;
|
||||
excluded = excl;
|
||||
}
|
||||
baseline[r] = b;
|
||||
}
|
||||
|
||||
__global__ void low_kernel(size_t n, uint32_t frames, const char *__restrict__ valid, const float *__restrict__ pooled,
|
||||
const int32_t *__restrict__ pooled_count, const float *__restrict__ pol,
|
||||
const int *__restrict__ radius, const float *__restrict__ baseline,
|
||||
float *__restrict__ ratio, float *__restrict__ deficit, char *__restrict__ low) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
const float base = baseline[radius[i]];
|
||||
const float rt = valid[i] ? pooled[i] / fmaxf(base, 1e-6f) : 1.0f;
|
||||
ratio[i] = rt;
|
||||
float df = 0.0f;
|
||||
char l = 0;
|
||||
if (valid[i]) {
|
||||
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
|
||||
df = static_cast<float>(poisson_deficit_sigma(pooled[i] * counted, base * counted));
|
||||
l = rt < SHADOW_RATIO && df > MIN_DEFICIT_SIGMA;
|
||||
}
|
||||
deficit[i] = df;
|
||||
low[i] = l;
|
||||
}
|
||||
|
||||
// 8-connected components by union-find: every component ends up named by its smallest pixel index,
|
||||
// whatever order the unions ran in.
|
||||
__device__ int find_root(const int *parent, int x) {
|
||||
while (parent[x] != x)
|
||||
x = parent[x];
|
||||
return x;
|
||||
}
|
||||
|
||||
__global__ void cc_init(size_t n, const char *__restrict__ member, int *__restrict__ parent) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
parent[i] = member[i] ? static_cast<int>(i) : -1;
|
||||
}
|
||||
|
||||
__device__ void cc_unite(int *parent, int a, int b) {
|
||||
while (true) {
|
||||
a = find_root(parent, a);
|
||||
b = find_root(parent, b);
|
||||
if (a == b) return;
|
||||
if (a < b) { const int t = a; a = b; b = t; }
|
||||
if (atomicCAS(&parent[a], a, b) == a) return;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void cc_union(int W, int H, const char *__restrict__ member, int *parent) {
|
||||
const size_t n = static_cast<size_t>(W) * H;
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n || !member[i]) return;
|
||||
const int x = static_cast<int>(i % W), y = static_cast<int>(i / W);
|
||||
// The four neighbours before this pixel; the other four see it from their side.
|
||||
if (x > 0 && member[i - 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(i - 1));
|
||||
if (y > 0) {
|
||||
const size_t up = i - W;
|
||||
if (member[up]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up));
|
||||
if (x > 0 && member[up - 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up - 1));
|
||||
if (x + 1 < W && member[up + 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up + 1));
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void cc_flatten(size_t n, int *parent) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n || parent[i] < 0) return;
|
||||
parent[i] = find_root(parent, static_cast<int>(i));
|
||||
}
|
||||
|
||||
// Per component (by root): how many of its pixels are `a`, and how many are both `a` and `b`.
|
||||
__global__ void cc_count(size_t n, const int *__restrict__ root, const char *__restrict__ a, const char *__restrict__ b,
|
||||
int *__restrict__ count_a, int *__restrict__ count_ab) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n || root[i] < 0 || !a[i]) return;
|
||||
atomicAdd(&count_a[root[i]], 1);
|
||||
if (count_ab && b[i])
|
||||
atomicAdd(&count_ab[root[i]], 1);
|
||||
}
|
||||
|
||||
__global__ void region_kernel(size_t n, const int *__restrict__ root, const int *__restrict__ n_low,
|
||||
const char *__restrict__ low, const char *__restrict__ valid, const int *__restrict__ radius,
|
||||
int blocked_out_to, char *__restrict__ region) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
char r = root[i] >= 0 && n_low[root[i]] >= MIN_SHADOW_PIXELS ? low[i] : 0;
|
||||
if (valid[i] && radius[i] <= blocked_out_to)
|
||||
r = 1;
|
||||
region[i] = r;
|
||||
}
|
||||
|
||||
__global__ void lit_kernel(size_t n, const uint32_t *__restrict__ valid_count, const int64_t *__restrict__ max_value,
|
||||
char *__restrict__ lit) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
lit[i] = (valid_count[i] > 0) && (max_value[i] >= MIN_REFLECTION);
|
||||
}
|
||||
|
||||
__global__ void reflection_kernel(int W, int H, const char *__restrict__ lit, char *__restrict__ reflection) {
|
||||
const size_t n = static_cast<size_t>(W) * H;
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
const int x = static_cast<int>(i % W), y = static_cast<int>(i / W);
|
||||
char r = 0;
|
||||
if (lit[i]) {
|
||||
int neighbours = 0;
|
||||
for (int dy = -1; dy <= 1; dy++)
|
||||
for (int dx = -1; dx <= 1; dx++) {
|
||||
const int yy = y + dy, xx = x + dx;
|
||||
if ((dx || dy) && yy >= 0 && yy < H && xx >= 0 && xx < W && lit[static_cast<size_t>(yy) * W + xx])
|
||||
neighbours++;
|
||||
}
|
||||
r = neighbours >= 2;
|
||||
}
|
||||
reflection[i] = r;
|
||||
}
|
||||
|
||||
__global__ void penumbra_kernel(size_t n, const char *__restrict__ penumbra, const char *__restrict__ valid,
|
||||
const float *__restrict__ ratio, const float *__restrict__ deficit, char *__restrict__ region) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
if (penumbra[i] && valid[i] && ratio[i] < PENUMBRA_RATIO && deficit[i] > MIN_DEFICIT_SIGMA)
|
||||
region[i] = 1;
|
||||
}
|
||||
|
||||
__global__ void invert_kernel(size_t n, const char *__restrict__ in, char *__restrict__ out) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
out[i] = !in[i];
|
||||
}
|
||||
|
||||
// Background components that touch the frame's edge are outside; the rest are holes.
|
||||
__global__ void outside_kernel(int W, int H, const int *__restrict__ root, int *__restrict__ outside) {
|
||||
const int k = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
const int perimeter = 2 * W + 2 * H;
|
||||
if (k >= perimeter) return;
|
||||
int x, y;
|
||||
if (k < W) { x = k; y = 0; }
|
||||
else if (k < 2 * W) { x = k - W; y = H - 1; }
|
||||
else if (k < 2 * W + H) { x = 0; y = k - 2 * W; }
|
||||
else { x = W - 1; y = k - 2 * W - H; }
|
||||
const int r = root[static_cast<size_t>(y) * W + x];
|
||||
if (r >= 0) outside[r] = 1;
|
||||
}
|
||||
|
||||
__global__ void final_kernel(size_t n, const int *__restrict__ background_root, const int *__restrict__ outside,
|
||||
const char *__restrict__ reflection_grown, char *__restrict__ region,
|
||||
uint32_t *__restrict__ mask) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
char r = region[i];
|
||||
if (background_root[i] >= 0 && !outside[background_root[i]])
|
||||
r = 1; // a hole
|
||||
if (reflection_grown[i])
|
||||
r = 0;
|
||||
region[i] = r;
|
||||
mask[i] = r ? MASK_SHADOW : 0;
|
||||
}
|
||||
|
||||
__global__ void sector_key_kernel(ShadowMaskSetup s, const char *__restrict__ valid, const char *__restrict__ region,
|
||||
const int *__restrict__ radius, const float *__restrict__ ratio,
|
||||
uint64_t *__restrict__ key) {
|
||||
const size_t n = static_cast<size_t>(s.width) * s.height;
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
if (!valid[i] || region[i]) {
|
||||
key[i] = UINT64_MAX;
|
||||
return;
|
||||
}
|
||||
const float dx = static_cast<float>(i % s.width) - s.beam_x, dy = static_cast<float>(i / s.width) - s.beam_y;
|
||||
const double phi = atan2f(dy, dx) + PI;
|
||||
const int sector = min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * PI) * HARMONIC_SECTORS));
|
||||
const uint64_t k = static_cast<uint64_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector;
|
||||
key[i] = (k << 32) | float_key(ratio[i]);
|
||||
}
|
||||
|
||||
__global__ void sector_median_kernel(int sectors, const uint64_t *__restrict__ keys, const int *__restrict__ offset,
|
||||
double *__restrict__ median) {
|
||||
const int k = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (k >= sectors) return;
|
||||
const int n = offset[k + 1] - offset[k];
|
||||
median[k] = n >= MIN_SECTOR_PIXELS ? key_float(static_cast<uint32_t>(keys[offset[k] + n / 2])) : -1.0;
|
||||
}
|
||||
|
||||
__global__ void dim_kernel(ShadowMaskSetup s, uint32_t frames, const char *__restrict__ valid,
|
||||
const char *__restrict__ region, const float *__restrict__ ratio,
|
||||
const float *__restrict__ deficit, const int *__restrict__ radius,
|
||||
const float *__restrict__ harm_c, const float *__restrict__ harm_s,
|
||||
const int32_t *__restrict__ pooled_count, const float *__restrict__ pol,
|
||||
const float *__restrict__ pooled, const float *__restrict__ baseline,
|
||||
char *__restrict__ dim) {
|
||||
const size_t n = static_cast<size_t>(s.width) * s.height;
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
char d = 0;
|
||||
if (valid[i] && !region[i] && ratio[i] < PENUMBRA_RATIO) {
|
||||
// cos 2phi and sin 2phi from the offset to the beam.
|
||||
const float dx = static_cast<float>(i % s.width) - s.beam_x, dy = static_cast<float>(i / s.width) - s.beam_y;
|
||||
const float r2 = fmaxf(dx * dx + dy * dy, 1e-6f);
|
||||
const int band = radius[i] / HARMONIC_BAND_PX;
|
||||
const float model = fminf(1.0f, 1.0f + harm_c[band] * (dx * dx - dy * dy) / r2
|
||||
+ harm_s[band] * 2.0f * dx * dy / r2);
|
||||
if (model >= 1.0f) {
|
||||
d = deficit[i] > MIN_DEFICIT_SIGMA;
|
||||
} else {
|
||||
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
|
||||
d = ratio[i] < PENUMBRA_RATIO * model
|
||||
&& poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA;
|
||||
}
|
||||
}
|
||||
dim[i] = d;
|
||||
}
|
||||
|
||||
// Join a region across the module gaps it crosses, one line per thread (bridge_gaps in
|
||||
// ShadowFinder.cpp). Both directions read `region` and only ever set pixels of `out` to 1.
|
||||
__global__ void bridge_rows(int W, int H, const char *__restrict__ region, const char *__restrict__ valid,
|
||||
char *out) {
|
||||
const int y = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (y >= H) return;
|
||||
const size_t row = static_cast<size_t>(y) * W;
|
||||
int k = 0;
|
||||
while (k < W) {
|
||||
if (valid[row + k]) { k++; continue; }
|
||||
const int start = k;
|
||||
while (k < W && !valid[row + k]) k++;
|
||||
if (start > 0 && k < W && region[row + start - 1] && region[row + k])
|
||||
for (int j = start; j < k; j++) out[row + j] = 1;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void bridge_columns(int W, int H, const char *__restrict__ region, const char *__restrict__ valid,
|
||||
char *out) {
|
||||
const int x = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (x >= W) return;
|
||||
const auto at = [&](int y) { return static_cast<size_t>(y) * W + x; };
|
||||
int k = 0;
|
||||
while (k < H) {
|
||||
if (valid[at(k)]) { k++; continue; }
|
||||
const int start = k;
|
||||
while (k < H && !valid[at(k)]) k++;
|
||||
if (start > 0 && k < H && region[at(start - 1)] && region[at(k)])
|
||||
for (int j = start; j < k; j++) out[at(j)] = 1;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void transmitting_kernel(size_t n, const int *__restrict__ root, const char *__restrict__ dim,
|
||||
const int *__restrict__ n_dim, const int *__restrict__ n_low,
|
||||
uint32_t *__restrict__ mask) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
const int c = root[i];
|
||||
if (c >= 0 && dim[i] && n_dim[c] >= MIN_SHADOW_PIXELS && n_low[c] >= MIN_CORE_PIXELS)
|
||||
mask[i] = MASK_TRANSMITTING;
|
||||
}
|
||||
|
||||
__global__ void mean_kernel(size_t n, const uint32_t *__restrict__ pixel_mask, const int64_t *__restrict__ sum_value,
|
||||
const uint32_t *__restrict__ valid_count, float *__restrict__ mean) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= n) return;
|
||||
mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0
|
||||
? static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]) : NAN;
|
||||
}
|
||||
|
||||
// The device side of one GetMask: planes, sort scratch and the stream to run on.
|
||||
class MaskEngine {
|
||||
public:
|
||||
const ShadowMaskSetup s;
|
||||
const int W, H;
|
||||
const size_t n;
|
||||
cudaStream_t stream;
|
||||
|
||||
MaskEngine(const ShadowMaskSetup &setup, cudaStream_t st)
|
||||
: s(setup), W(setup.width), H(setup.height), n(static_cast<size_t>(setup.width) * setup.height), stream(st) {}
|
||||
|
||||
void Check(const char *what) const {
|
||||
check(cudaGetLastError(), what);
|
||||
}
|
||||
|
||||
void Dilate(const char *in, char *out, char *scratch, int r) const {
|
||||
dilate_rows<<<grid(H), THREADS, 0, stream>>>(in, scratch, W, H, r);
|
||||
dilate_columns<<<grid(W), THREADS, 0, stream>>>(scratch, out, W, H, r);
|
||||
Check("dilate");
|
||||
}
|
||||
|
||||
// Components of `member`, each pixel's root in `root` (-1 outside every component).
|
||||
void Label(const char *member, int *root) const {
|
||||
cc_init<<<grid(n), THREADS, 0, stream>>>(n, member, root);
|
||||
cc_union<<<grid(n), THREADS, 0, stream>>>(W, H, member, root);
|
||||
cc_flatten<<<grid(n), THREADS, 0, stream>>>(n, root);
|
||||
Check("components");
|
||||
}
|
||||
|
||||
void SortKeys(uint64_t *keys, uint64_t *sorted) const {
|
||||
size_t bytes = 0;
|
||||
check(cub::DeviceRadixSort::SortKeys(nullptr, bytes, keys, sorted, n, 0, 64, stream), "sort size");
|
||||
CudaDevicePtr<uint8_t> scratch(bytes);
|
||||
check(cub::DeviceRadixSort::SortKeys(scratch.get(), bytes, keys, sorted, n, 0, 64, stream), "sort");
|
||||
// The scratch is freed on the allocation stream, which knows nothing of this one.
|
||||
check(cudaStreamSynchronize(stream), "sort");
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
std::vector<T> Download(const T *device, size_t count) const {
|
||||
std::vector<T> host(count);
|
||||
check(cudaMemcpyAsync(host.data(), device, count * sizeof(T), cudaMemcpyDeviceToHost, stream), "download");
|
||||
check(cudaStreamSynchronize(stream), "download");
|
||||
return host;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void Upload(T *device, const std::vector<T> &host) const {
|
||||
check(cudaMemcpyAsync(device, host.data(), host.size() * sizeof(T), cudaMemcpyHostToDevice, stream), "upload");
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace
|
||||
|
||||
std::vector<uint32_t> ShadowMaskOnDevice(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask,
|
||||
const int64_t *max_value, const int64_t *sum_value,
|
||||
const uint32_t *valid_count, uint32_t frames, cudaStream_t stream) {
|
||||
const MaskEngine e(setup, stream);
|
||||
const size_t n = e.n;
|
||||
const int W = e.W, H = e.H;
|
||||
|
||||
CudaDevicePtr<uint32_t> d_pixel_mask(n), mask(n);
|
||||
e.Upload(d_pixel_mask.get(), pixel_mask);
|
||||
|
||||
// Mean projection over the polarization factor, usable pixels and radius from the beam centre.
|
||||
CudaDevicePtr<float> pol(n), pooled(n), ratio(n), deficit(n);
|
||||
CudaDevicePtr<char> valid(n), low(n), region(n), a(n), b(n), c(n);
|
||||
CudaDevicePtr<int> radius(n), root(n), count_a(n), count_b(n), max_radius(1);
|
||||
CudaDevicePtr<double> num(n), dsum(n);
|
||||
CudaDevicePtr<int32_t> den(n), dcount(n);
|
||||
check(cudaMemsetAsync(max_radius.get(), 0, sizeof(int), stream), "memset");
|
||||
setup_kernel<<<grid(n), THREADS, 0, stream>>>(setup, d_pixel_mask, sum_value, valid_count, pol, valid, radius,
|
||||
num, den, max_radius);
|
||||
e.Check("setup");
|
||||
const int max_r = e.Download(max_radius.get(), 1)[0];
|
||||
const int rings = max_r + 1;
|
||||
|
||||
// The background pooled over a small box.
|
||||
box_rows<double><<<grid(H), THREADS, 0, stream>>>(num, dsum, W, H, POOL_PX / 2);
|
||||
box_columns<double><<<grid(W), THREADS, 0, stream>>>(dsum, num, W, H, POOL_PX / 2);
|
||||
box_rows<int32_t><<<grid(H), THREADS, 0, stream>>>(den, dcount, W, H, POOL_PX / 2);
|
||||
box_columns<int32_t><<<grid(W), THREADS, 0, stream>>>(dcount, den, W, H, POOL_PX / 2);
|
||||
e.Check("pooling");
|
||||
const double *pooled_sum = num;
|
||||
const int32_t *pooled_count = den;
|
||||
|
||||
// The rings, each sorted once; the baseline is an order statistic of them.
|
||||
CudaDevicePtr<uint64_t> keys(n), sorted(n);
|
||||
pooled_kernel<<<grid(n), THREADS, 0, stream>>>(n, pooled_sum, pooled_count, valid, radius, pooled, keys);
|
||||
e.Check("pooled");
|
||||
e.SortKeys(keys, sorted);
|
||||
CudaDevicePtr<int> ring_offset(rings + 1);
|
||||
group_offsets<<<grid(rings + 1), THREADS, 0, stream>>>(sorted, n, rings, ring_offset);
|
||||
CudaDevicePtr<float> baseline(rings);
|
||||
baseline_kernel<<<grid(rings), THREADS, 0, stream>>>(rings, sorted, ring_offset, baseline);
|
||||
e.Check("baseline");
|
||||
const auto host_baseline = e.Download(baseline.get(), rings);
|
||||
const auto offsets = e.Download(ring_offset.get(), rings + 1);
|
||||
std::vector<int> ring_pixels(rings);
|
||||
for (int r = 0; r < rings; r++)
|
||||
ring_pixels[r] = offsets[r + 1] - offsets[r];
|
||||
const int blocked_out_to = BlockedOutTo(host_baseline, ring_pixels);
|
||||
|
||||
// Low pixels, and the regions of them large enough to be hardware.
|
||||
low_kernel<<<grid(n), THREADS, 0, stream>>>(n, frames, valid, pooled, pooled_count, pol, radius, baseline,
|
||||
ratio, deficit, low);
|
||||
e.Check("low");
|
||||
e.Dilate(low, a, c, BRIDGE_PX);
|
||||
e.Label(a, root);
|
||||
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
|
||||
cc_count<<<grid(n), THREADS, 0, stream>>>(n, root, low, low, count_a, nullptr);
|
||||
region_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, count_a, low, valid, radius, blocked_out_to, region);
|
||||
e.Check("region");
|
||||
|
||||
// Recorded reflections; `b` holds them until they are given back at the end.
|
||||
lit_kernel<<<grid(n), THREADS, 0, stream>>>(n, valid_count, max_value, a);
|
||||
reflection_kernel<<<grid(n), THREADS, 0, stream>>>(W, H, a, b);
|
||||
e.Check("reflections");
|
||||
|
||||
// Penumbra, round and fill.
|
||||
e.Dilate(region, a, c, PENUMBRA_MAX_PX);
|
||||
penumbra_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, valid, ratio, deficit, region);
|
||||
e.Dilate(region, a, c, 2);
|
||||
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, region);
|
||||
e.Dilate(region, a, c, 2);
|
||||
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, region); // region = erode(dilate(region))
|
||||
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, region, a); // the background
|
||||
e.Label(a, root);
|
||||
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
|
||||
outside_kernel<<<grid(2 * W + 2 * H), THREADS, 0, stream>>>(W, H, root, count_a);
|
||||
e.Dilate(b, c, a, 1); // the reflections, grown by one
|
||||
final_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, count_a, c, region, mask);
|
||||
e.Check("fill");
|
||||
|
||||
// The arm search: sector medians of the ratio, the harmonic of each band, the dim pixels.
|
||||
const int n_bands = max_r / HARMONIC_BAND_PX + 1;
|
||||
const int n_sectors = n_bands * HARMONIC_SECTORS;
|
||||
sector_key_kernel<<<grid(n), THREADS, 0, stream>>>(setup, valid, region, radius, ratio, keys);
|
||||
e.Check("sectors");
|
||||
e.SortKeys(keys, sorted);
|
||||
CudaDevicePtr<int> sector_offset(n_sectors + 1);
|
||||
group_offsets<<<grid(n_sectors + 1), THREADS, 0, stream>>>(sorted, n, n_sectors, sector_offset);
|
||||
CudaDevicePtr<double> sector_median(n_sectors);
|
||||
sector_median_kernel<<<grid(n_sectors), THREADS, 0, stream>>>(n_sectors, sorted, sector_offset, sector_median);
|
||||
e.Check("sector medians");
|
||||
std::vector<float> harm_c, harm_s;
|
||||
HarmonicFit(e.Download(sector_median.get(), n_sectors), n_bands, harm_c, harm_s);
|
||||
CudaDevicePtr<float> d_harm_c(n_bands), d_harm_s(n_bands);
|
||||
e.Upload(d_harm_c.get(), harm_c);
|
||||
e.Upload(d_harm_s.get(), harm_s);
|
||||
dim_kernel<<<grid(n), THREADS, 0, stream>>>(setup, frames, valid, region, ratio, deficit, radius, d_harm_c, d_harm_s,
|
||||
pooled_count, pol, pooled, baseline, a);
|
||||
e.Check("dim");
|
||||
e.Dilate(a, b, c, BRIDGE_PX);
|
||||
check(cudaMemcpyAsync(c.get(), b.get(), n, cudaMemcpyDeviceToDevice, stream), "copy");
|
||||
bridge_rows<<<grid(H), THREADS, 0, stream>>>(W, H, b, valid, c);
|
||||
bridge_columns<<<grid(W), THREADS, 0, stream>>>(W, H, b, valid, c);
|
||||
e.Label(c, root);
|
||||
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
|
||||
check(cudaMemsetAsync(count_b.get(), 0, n * sizeof(int), stream), "memset");
|
||||
cc_count<<<grid(n), THREADS, 0, stream>>>(n, root, a, low, count_a, count_b);
|
||||
transmitting_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, a, count_a, count_b, mask);
|
||||
e.Check("transmitting");
|
||||
|
||||
return e.Download(mask.get(), n);
|
||||
}
|
||||
|
||||
std::vector<float> MeanProjectionOnDevice(const std::vector<uint32_t> &pixel_mask, const int64_t *sum_value,
|
||||
const uint32_t *valid_count, size_t npixels, cudaStream_t stream) {
|
||||
CudaDevicePtr<uint32_t> d_pixel_mask(npixels);
|
||||
CudaDevicePtr<float> mean(npixels);
|
||||
check(cudaMemcpyAsync(d_pixel_mask.get(), pixel_mask.data(), npixels * sizeof(uint32_t), cudaMemcpyHostToDevice,
|
||||
stream), "upload");
|
||||
mean_kernel<<<grid(npixels), THREADS, 0, stream>>>(npixels, d_pixel_mask, sum_value, valid_count, mean);
|
||||
check(cudaGetLastError(), "mean");
|
||||
std::vector<float> host(npixels);
|
||||
check(cudaMemcpyAsync(host.data(), mean.get(), npixels * sizeof(float), cudaMemcpyDeviceToHost, stream), "download");
|
||||
check(cudaStreamSynchronize(stream), "mean");
|
||||
return host;
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
// Included only under JFJOCH_USE_CUDA.
|
||||
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
// What the mask is drawn about: the detector, the centre the rings are drawn about, and the geometry
|
||||
// the polarization factor is read off (ShadowFinder keeps it as a DiffractionGeometry; the device
|
||||
// takes it as numbers).
|
||||
struct ShadowMaskSetup {
|
||||
int width = 0, height = 0;
|
||||
float beam_x = 0.0f, beam_y = 0.0f;
|
||||
float det_matrix[9] = {}; // row major
|
||||
float pixel_size_mm = 0.0f;
|
||||
float distance_mm = 0.0f;
|
||||
bool has_polarization = false;
|
||||
float polarization = 0.0f;
|
||||
};
|
||||
|
||||
// ShadowFinder::GetMask on the device, from the projection ShadowAccumulatorGPU holds there. Step for
|
||||
// step the host's algorithm - the same pooling, ring medians, components, morphology and arm search -
|
||||
// and the same answer wherever the arithmetic is exact: every integer, comparison, sort and component
|
||||
// is. What is not is the floating point the two compilers evaluate differently - the polarization
|
||||
// factor's trigonometry, the Poisson test's logarithm and the azimuth of the arm search - so a pixel
|
||||
// within a rounding of one of those thresholds can come out the other way.
|
||||
std::vector<uint32_t> ShadowMaskOnDevice(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask,
|
||||
const int64_t *max_value, const int64_t *sum_value,
|
||||
const uint32_t *valid_count, uint32_t frames, cudaStream_t stream);
|
||||
|
||||
// The mean projection the host's GetMeanProjection makes, computed where the sums are: the same
|
||||
// division, so the same bits, and a quarter of the bytes to bring back.
|
||||
std::vector<float> MeanProjectionOnDevice(const std::vector<uint32_t> &pixel_mask, const int64_t *sum_value,
|
||||
const uint32_t *valid_count, size_t npixels, cudaStream_t stream);
|
||||
@@ -45,6 +45,7 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys
|
||||
const Coord m3 = (m1 % m2).Normalize();
|
||||
|
||||
const float m2_S0 = m2 * S0;
|
||||
const float four_S0_sq = 4 * S0 * S0;
|
||||
const float m3_S0 = m3 * S0;
|
||||
|
||||
int i = 0;
|
||||
@@ -99,17 +100,24 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys
|
||||
cos_phi_limit = std::cos(phi_limit);
|
||||
}
|
||||
|
||||
// p0 = A* h + B* k + C* l, evaluated as ((A* h) + (B* k)) + (C* l) exactly as before, with the
|
||||
// terms that do not change in the inner loops taken out of them.
|
||||
std::vector<Coord> Cstar_l(2 * settings.max_l + 1);
|
||||
for (int l = -settings.max_l; l <= settings.max_l; l++)
|
||||
Cstar_l[l + settings.max_l] = Cstar * l;
|
||||
|
||||
for (int h = -settings.max_h; h <= settings.max_h; h++) {
|
||||
// Precompute A* h contribution
|
||||
const Coord Astar_h = Astar * h;
|
||||
|
||||
for (int k = -settings.max_k; k <= settings.max_k; k++) {
|
||||
// Accumulate B* k contribution
|
||||
const Coord AB = Astar_h + Bstar * k;
|
||||
|
||||
for (int l = -settings.max_l; l <= settings.max_l; l++) {
|
||||
if (systematic_absence(h, k, l, settings.centering))
|
||||
continue;
|
||||
|
||||
Coord p0 = Astar * h + Bstar * k + Cstar * l;
|
||||
const Coord &Cl = Cstar_l[l + settings.max_l];
|
||||
const Coord p0(AB.x + Cl.x, AB.y + Cl.y, AB.z + Cl.z);
|
||||
|
||||
float p0_sq = p0 * p0;
|
||||
if (p0_sq <= 0.0f || p0_sq > one_over_dmax_sq)
|
||||
@@ -129,7 +137,7 @@ int BraggPredictionRot::Calc(const DiffractionExperiment &experiment, const Crys
|
||||
};
|
||||
|
||||
// No solution for Laue equations
|
||||
if ((rho_sq < p_m3 * p_m3) || (p0_sq > 4 * S0 * S0))
|
||||
if ((rho_sq < p_m3 * p_m3) || (p0_sq > four_S0_sq))
|
||||
continue;
|
||||
|
||||
// Effective rocking width for this reflection: mosaicity broadened by the bandwidth
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
// Where one pixel falls in the background beam-centre fit (FindBeamCenterFromBackground): which
|
||||
// radial-bin x sector cell of the fitted band it lands in about a trial centre, and the derivative
|
||||
// of its 2theta with respect to that centre. Written once and compiled both for the host fit and for
|
||||
// its device twin (BeamCenterBackgroundGPU), so the two read the same formula.
|
||||
//
|
||||
// The same formula is the same number only when both sides evaluate it the same way. The library
|
||||
// atan2f differs between glibc and CUDA in the last bit, and the compilers fuse multiply-adds each
|
||||
// in their own way; either moves a pixel within a rounding of a cell edge into the neighbouring cell
|
||||
// (measured on 16 Mpx sweeps: ~30 of 6.5 million pixels, enough to move the fitted centre by up to
|
||||
// 0.05 px). So the angles come from BackgroundAtan2 below and both translation units are compiled
|
||||
// without contraction (geom_refinement/CMakeLists.txt), and the host and the device then agree to
|
||||
// the bit.
|
||||
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
|
||||
#include "../../common/JFJochMath.h"
|
||||
|
||||
#ifdef __CUDACC__
|
||||
#define BACKGROUND_BAND_HD __host__ __device__ inline
|
||||
#else
|
||||
#define BACKGROUND_BAND_HD inline
|
||||
#endif
|
||||
|
||||
struct BackgroundBand {
|
||||
static constexpr int SECTORS = 36;
|
||||
static constexpr int RADIAL_BINS = 120;
|
||||
static constexpr int CELLS = RADIAL_BINS * SECTORS;
|
||||
|
||||
float rot[9]; // detector matrix, row major
|
||||
float pixel_size; // mm
|
||||
float distance; // mm
|
||||
float tt_lo, tt_hi; // the band in 2theta
|
||||
float d_tt; // width of one radial bin
|
||||
double tan_lo, tan_hi; // the band in tan(2theta), widened for the quick rejection
|
||||
};
|
||||
|
||||
// atan2(y, x) from IEEE operations alone - add, multiply, divide, square root, all correctly rounded on
|
||||
// the host and on the device - so that, compiled without contraction (see CMakeLists.txt), the host
|
||||
// and the device return the same bits. The library atan2f does not: glibc's and CUDA's differ in the
|
||||
// last place. Two half-angle reductions take the argument below tan(pi/16), where eleven terms of the
|
||||
// series leave an error under 1e-17.
|
||||
BACKGROUND_BAND_HD double BackgroundAtan2(double y, double x) {
|
||||
const double ax = x < 0 ? -x : x, ay = y < 0 ? -y : y;
|
||||
if (ax == 0.0 && ay == 0.0)
|
||||
return 0.0;
|
||||
const bool swap = ay > ax;
|
||||
double t = swap ? ax / ay : ay / ax;
|
||||
t = t / (1.0 + sqrt(1.0 + t * t));
|
||||
t = t / (1.0 + sqrt(1.0 + t * t));
|
||||
const double t2 = t * t;
|
||||
double series = 1.0 / 23.0;
|
||||
for (int k = 21; k >= 1; k -= 2)
|
||||
series = 1.0 / k - t2 * series;
|
||||
double a = 4.0 * t * series;
|
||||
if (swap) a = PI / 2 - a;
|
||||
if (x < 0) a = PI - a;
|
||||
return y < 0 ? -a : a;
|
||||
}
|
||||
|
||||
// The cell of pixel (x, y) about (beam_x, beam_y), or -1 when it is outside the band; for a pixel in
|
||||
// it, also the two components of the derivative of its 2theta with respect to the centre.
|
||||
BACKGROUND_BAND_HD int BackgroundBandCell(const BackgroundBand &b, int x, int y, float beam_x, float beam_y,
|
||||
float &jac_x, float &jac_y) {
|
||||
const float *rot = b.rot;
|
||||
const float u = (x - beam_x) * b.pixel_size;
|
||||
const float v = (y - beam_y) * b.pixel_size;
|
||||
const float lx = rot[0] * u + rot[1] * v + rot[2] * b.distance;
|
||||
const float ly = rot[3] * u + rot[4] * v + rot[5] * b.distance;
|
||||
const float lz = rot[6] * u + rot[7] * v + rot[8] * b.distance;
|
||||
const float rho_sq = lx * lx + ly * ly;
|
||||
// Most pixels lie outside the band. Those clearly outside it in tan(2theta) = rho / lz - by a
|
||||
// margin far above float rounding - skip the square root and both atan2 below; every pixel the
|
||||
// exact test would keep still reaches it.
|
||||
if (lz > 0.0f) {
|
||||
const double lz_sq = static_cast<double>(lz) * lz;
|
||||
if (rho_sq < b.tan_lo * b.tan_lo * lz_sq || rho_sq > b.tan_hi * b.tan_hi * lz_sq)
|
||||
return -1;
|
||||
}
|
||||
const float rho = sqrtf(rho_sq);
|
||||
const float two_theta = static_cast<float>(BackgroundAtan2(rho, lz));
|
||||
if (two_theta < b.tt_lo || two_theta >= b.tt_hi || rho == 0.0f)
|
||||
return -1;
|
||||
|
||||
const float phi = static_cast<float>(BackgroundAtan2(ly, lx));
|
||||
// Both bins are clamped: a pixel one float ulp below the top of the band divides to exactly
|
||||
// RADIAL_BINS, which is one cell past the end of every accumulator.
|
||||
int r_bin = static_cast<int>((two_theta - b.tt_lo) / b.d_tt);
|
||||
r_bin = r_bin < 0 ? 0 : (r_bin > BackgroundBand::RADIAL_BINS - 1 ? BackgroundBand::RADIAL_BINS - 1 : r_bin);
|
||||
int s_bin = static_cast<int>((phi + PI) / (2 * PI) * BackgroundBand::SECTORS);
|
||||
s_bin = s_bin < 0 ? 0 : (s_bin > BackgroundBand::SECTORS - 1 ? BackgroundBand::SECTORS - 1 : s_bin);
|
||||
|
||||
// d(2theta)/d(beam), through the lab coordinate: the detector coordinate depends on the centre
|
||||
// only as (x - beam_x), so moving the centre is moving the pixel.
|
||||
const float denominator = rho * rho + lz * lz;
|
||||
const float g_x = lz * lx / (rho * denominator);
|
||||
const float g_y = lz * ly / (rho * denominator);
|
||||
const float g_z = -rho / denominator;
|
||||
jac_x = -b.pixel_size * (g_x * rot[0] + g_y * rot[3] + g_z * rot[6]);
|
||||
jac_y = -b.pixel_size * (g_x * rot[1] + g_y * rot[4] + g_z * rot[7]);
|
||||
return r_bin * BackgroundBand::SECTORS + s_bin;
|
||||
}
|
||||
@@ -0,0 +1,210 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#include "BeamCenterBackgroundGPU.h"
|
||||
|
||||
#include <cub/device/device_radix_sort.cuh>
|
||||
|
||||
#include "../indexing/CUDAMemHelpers.h"
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr int THREADS = 256;
|
||||
|
||||
void check(cudaError_t err, const char *what) {
|
||||
if (err != cudaSuccess)
|
||||
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
|
||||
std::string("Beam centre from background: ") + what + ": " + cudaGetErrorString(err));
|
||||
}
|
||||
|
||||
// Every pixel's cell (BackgroundBand::CELLS where it is outside the band or unusable), its index,
|
||||
// and its derivatives.
|
||||
__global__ void bin_kernel(BackgroundBand band, int width, size_t npixels, float beam_x, float beam_y,
|
||||
const char *__restrict__ usable, int32_t *__restrict__ key,
|
||||
int32_t *__restrict__ index, float *__restrict__ jac_x,
|
||||
float *__restrict__ jac_y) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= npixels)
|
||||
return;
|
||||
int cell = BackgroundBand::CELLS;
|
||||
float jx = 0.0f, jy = 0.0f;
|
||||
if (usable[i]) {
|
||||
const int c = BackgroundBandCell(band, static_cast<int>(i % width), static_cast<int>(i / width),
|
||||
beam_x, beam_y, jx, jy);
|
||||
if (c >= 0)
|
||||
cell = c;
|
||||
}
|
||||
key[i] = cell;
|
||||
index[i] = static_cast<int32_t>(i);
|
||||
jac_x[i] = jx;
|
||||
jac_y[i] = jy;
|
||||
}
|
||||
|
||||
// Where each cell's pixels start in the sorted keys; offset[CELLS] is the number of binned pixels.
|
||||
__global__ void offset_kernel(const int32_t *__restrict__ sorted_key, size_t n, int32_t *__restrict__ offset) {
|
||||
const int c = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (c > BackgroundBand::CELLS)
|
||||
return;
|
||||
size_t lo = 0, hi = n;
|
||||
while (lo < hi) {
|
||||
const size_t mid = (lo + hi) / 2;
|
||||
if (sorted_key[mid] < c) lo = mid + 1;
|
||||
else hi = mid;
|
||||
}
|
||||
offset[c] = static_cast<int32_t>(lo);
|
||||
}
|
||||
|
||||
// One thread per cell, walking its pixels in pixel order: a partial sum per host row block, added to
|
||||
// the cell's total when the block changes - the host's arithmetic, step for step. A cell whose pixels
|
||||
// are clipped (clip_limit != nullptr) skips the ones above its limit, as the host's clip rounds do.
|
||||
__global__ void sum_kernel(int width, const int32_t *__restrict__ offset, const int32_t *__restrict__ index,
|
||||
const int32_t *__restrict__ row_block, const float *__restrict__ mean,
|
||||
const float *__restrict__ jac_x, const float *__restrict__ jac_y,
|
||||
const float *__restrict__ clip_limit,
|
||||
double *__restrict__ sum, double *__restrict__ sum_sq,
|
||||
double *__restrict__ sum_jx, double *__restrict__ sum_jy,
|
||||
int32_t *__restrict__ count) {
|
||||
const int c = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (c >= BackgroundBand::CELLS)
|
||||
return;
|
||||
const bool clipped = clip_limit != nullptr;
|
||||
const float limit = clipped ? clip_limit[c] : 0.0f;
|
||||
|
||||
double s = 0, ss = 0, jx = 0, jy = 0;
|
||||
double bs = 0, bss = 0, bjx = 0, bjy = 0;
|
||||
int32_t n = 0;
|
||||
int block = -1;
|
||||
for (int32_t p = offset[c]; p < offset[c + 1]; p++) {
|
||||
const int32_t i = index[p];
|
||||
const float value = mean[i];
|
||||
if (clipped && (limit < 0.0f || value > limit))
|
||||
continue;
|
||||
const int b = row_block[i / width];
|
||||
if (b != block) {
|
||||
s += bs; ss += bss; jx += bjx; jy += bjy;
|
||||
bs = bss = bjx = bjy = 0;
|
||||
block = b;
|
||||
}
|
||||
n++;
|
||||
bs += value;
|
||||
bss += static_cast<double>(value) * value;
|
||||
if (!clipped) {
|
||||
bjx += jac_x[i];
|
||||
bjy += jac_y[i];
|
||||
}
|
||||
}
|
||||
s += bs; ss += bss; jx += bjx; jy += bjy;
|
||||
sum[c] = s;
|
||||
sum_sq[c] = ss;
|
||||
count[c] = n;
|
||||
if (!clipped) {
|
||||
sum_jx[c] = jx;
|
||||
sum_jy[c] = jy;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
struct BeamCenterBackgroundGPU::Impl {
|
||||
int width, height;
|
||||
size_t npixels;
|
||||
CudaStream stream;
|
||||
CudaDevicePtr<char> usable;
|
||||
CudaDevicePtr<float> mean;
|
||||
CudaDevicePtr<int32_t> row_block;
|
||||
CudaDevicePtr<int32_t> key, sorted_key, index, sorted_index;
|
||||
CudaDevicePtr<float> jac_x, jac_y;
|
||||
CudaDevicePtr<int32_t> offset;
|
||||
CudaDevicePtr<float> clip_limit;
|
||||
CudaDevicePtr<double> sum, sum_sq, sum_jx, sum_jy;
|
||||
CudaDevicePtr<int32_t> count;
|
||||
CudaDevicePtr<uint8_t> sort_scratch;
|
||||
size_t sort_scratch_bytes = 0;
|
||||
|
||||
Impl(int w, int h)
|
||||
: width(w), height(h), npixels(static_cast<size_t>(w) * h),
|
||||
usable(npixels), mean(npixels), row_block(h),
|
||||
key(npixels), sorted_key(npixels), index(npixels), sorted_index(npixels),
|
||||
jac_x(npixels), jac_y(npixels), offset(BackgroundBand::CELLS + 1),
|
||||
clip_limit(BackgroundBand::CELLS),
|
||||
sum(BackgroundBand::CELLS), sum_sq(BackgroundBand::CELLS),
|
||||
sum_jx(BackgroundBand::CELLS), sum_jy(BackgroundBand::CELLS),
|
||||
count(BackgroundBand::CELLS) {
|
||||
// Keys run to CELLS inclusive, the bin of everything outside the band.
|
||||
check(cub::DeviceRadixSort::SortPairs(nullptr, sort_scratch_bytes, key.get(), sorted_key.get(),
|
||||
index.get(), sorted_index.get(), npixels, 0, end_bit(), stream),
|
||||
"sort size");
|
||||
sort_scratch = CudaDevicePtr<uint8_t>(sort_scratch_bytes);
|
||||
}
|
||||
|
||||
static int end_bit() {
|
||||
int bits = 0;
|
||||
while ((1 << bits) <= BackgroundBand::CELLS) bits++;
|
||||
return bits;
|
||||
}
|
||||
|
||||
void Download(std::vector<double> &s, std::vector<double> &ss, std::vector<double> *jx,
|
||||
std::vector<double> *jy, std::vector<int32_t> &n) {
|
||||
const size_t cells = BackgroundBand::CELLS;
|
||||
check(cudaMemcpyAsync(s.data(), sum.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy");
|
||||
check(cudaMemcpyAsync(ss.data(), sum_sq.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy");
|
||||
if (jx) check(cudaMemcpyAsync(jx->data(), sum_jx.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy");
|
||||
if (jy) check(cudaMemcpyAsync(jy->data(), sum_jy.get(), cells * sizeof(double), cudaMemcpyDeviceToHost, stream), "copy");
|
||||
check(cudaMemcpyAsync(n.data(), count.get(), cells * sizeof(int32_t), cudaMemcpyDeviceToHost, stream), "copy");
|
||||
check(cudaStreamSynchronize(stream), "sums");
|
||||
}
|
||||
};
|
||||
|
||||
BeamCenterBackgroundGPU::BeamCenterBackgroundGPU(int width, int height, const std::vector<int> &block_row,
|
||||
const char *usable, const float *mean)
|
||||
: impl(std::make_unique<Impl>(width, height)) {
|
||||
std::vector<int32_t> row_block(height);
|
||||
for (size_t b = 0; b + 1 < block_row.size(); b++)
|
||||
for (int y = block_row[b]; y < block_row[b + 1]; y++)
|
||||
row_block[y] = static_cast<int32_t>(b);
|
||||
check(cudaMemcpyAsync(impl->row_block.get(), row_block.data(), height * sizeof(int32_t),
|
||||
cudaMemcpyHostToDevice, impl->stream), "upload");
|
||||
check(cudaMemcpyAsync(impl->usable.get(), usable, impl->npixels, cudaMemcpyHostToDevice, impl->stream), "upload");
|
||||
check(cudaMemcpyAsync(impl->mean.get(), mean, impl->npixels * sizeof(float), cudaMemcpyHostToDevice,
|
||||
impl->stream), "upload");
|
||||
check(cudaStreamSynchronize(impl->stream), "upload");
|
||||
}
|
||||
|
||||
BeamCenterBackgroundGPU::~BeamCenterBackgroundGPU() = default;
|
||||
|
||||
void BeamCenterBackgroundGPU::Bin(const BackgroundBand &band, float beam_x, float beam_y,
|
||||
std::vector<double> &sum, std::vector<double> &sum_sq,
|
||||
std::vector<double> &sum_jx, std::vector<double> &sum_jy,
|
||||
std::vector<int32_t> &count) {
|
||||
Impl &d = *impl;
|
||||
const auto blocks = static_cast<unsigned>((d.npixels + THREADS - 1) / THREADS);
|
||||
bin_kernel<<<blocks, THREADS, 0, d.stream>>>(band, d.width, d.npixels, beam_x, beam_y, d.usable,
|
||||
d.key, d.index, d.jac_x, d.jac_y);
|
||||
check(cudaGetLastError(), "bin");
|
||||
// A radix sort is stable, so each cell's pixels come out in pixel order.
|
||||
check(cub::DeviceRadixSort::SortPairs(d.sort_scratch.get(), d.sort_scratch_bytes, d.key.get(),
|
||||
d.sorted_key.get(), d.index.get(), d.sorted_index.get(),
|
||||
d.npixels, 0, Impl::end_bit(), d.stream), "sort");
|
||||
constexpr int cell_blocks = (BackgroundBand::CELLS + 1 + THREADS - 1) / THREADS;
|
||||
offset_kernel<<<cell_blocks, THREADS, 0, d.stream>>>(d.sorted_key, d.npixels, d.offset);
|
||||
check(cudaGetLastError(), "offsets");
|
||||
sum_kernel<<<cell_blocks, THREADS, 0, d.stream>>>(d.width, d.offset, d.sorted_index, d.row_block, d.mean,
|
||||
d.jac_x, d.jac_y, nullptr, d.sum, d.sum_sq,
|
||||
d.sum_jx, d.sum_jy, d.count);
|
||||
check(cudaGetLastError(), "sums");
|
||||
d.Download(sum, sum_sq, &sum_jx, &sum_jy, count);
|
||||
}
|
||||
|
||||
void BeamCenterBackgroundGPU::Clip(const std::vector<float> &clip_limit,
|
||||
std::vector<double> &sum, std::vector<double> &sum_sq,
|
||||
std::vector<int32_t> &count) {
|
||||
Impl &d = *impl;
|
||||
check(cudaMemcpyAsync(d.clip_limit.get(), clip_limit.data(), BackgroundBand::CELLS * sizeof(float),
|
||||
cudaMemcpyHostToDevice, d.stream), "upload");
|
||||
constexpr int cell_blocks = (BackgroundBand::CELLS + THREADS - 1) / THREADS;
|
||||
sum_kernel<<<cell_blocks, THREADS, 0, d.stream>>>(d.width, d.offset, d.sorted_index, d.row_block, d.mean,
|
||||
d.jac_x, d.jac_y, d.clip_limit, d.sum, d.sum_sq,
|
||||
d.sum_jx, d.sum_jy, d.count);
|
||||
check(cudaGetLastError(), "clip");
|
||||
d.Download(sum, sum_sq, nullptr, nullptr, count);
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
// Included only under JFJOCH_USE_CUDA. Free of CUDA headers, so the host fit can hold one.
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <vector>
|
||||
|
||||
#include "BackgroundBand.h"
|
||||
|
||||
// The two passes over the pixels of the background beam-centre fit (FindBeamCenterFromBackground),
|
||||
// on the device: binning every usable pixel into its cell about a trial centre, and summing the
|
||||
// binned pixels again under a clip. They are all of the fit's cost; the fit itself stays on the host.
|
||||
//
|
||||
// Each cell is summed in the order the host sums it - pixel order within each of the host's row
|
||||
// blocks, the blocks then added in block order - so a pixel that lands in the same cell on both
|
||||
// sides adds the same rounding on both. Whatever differs comes from BackgroundBandCell (see there).
|
||||
class BeamCenterBackgroundGPU {
|
||||
struct Impl;
|
||||
std::unique_ptr<Impl> impl;
|
||||
public:
|
||||
// block_row: the first row of each of the host's row blocks, and one past the last row at the end.
|
||||
BeamCenterBackgroundGPU(int width, int height, const std::vector<int> &block_row,
|
||||
const char *usable, const float *mean);
|
||||
~BeamCenterBackgroundGPU();
|
||||
|
||||
// Bin the band about (beam_x, beam_y) and sum each cell: the values, their squares, the two
|
||||
// derivatives, and the count.
|
||||
void Bin(const BackgroundBand &band, float beam_x, float beam_y,
|
||||
std::vector<double> &sum, std::vector<double> &sum_sq,
|
||||
std::vector<double> &sum_jx, std::vector<double> &sum_jy, std::vector<int32_t> &count);
|
||||
|
||||
// Sum the pixels the last Bin put in each cell again, leaving out those above the cell's
|
||||
// clip_limit and every pixel of a cell whose limit is negative.
|
||||
void Clip(const std::vector<float> &clip_limit,
|
||||
std::vector<double> &sum, std::vector<double> &sum_sq, std::vector<int32_t> &count);
|
||||
};
|
||||
@@ -2,13 +2,19 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#include "BeamCenterFromBackground.h"
|
||||
#include "BackgroundBand.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <thread>
|
||||
|
||||
#include "../../common/CompressedImage.h"
|
||||
#include "../../common/JFJochMath.h"
|
||||
#include "../../common/ParallelFor.h"
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
#include "../../common/CUDAWrapper.h"
|
||||
#include "BeamCenterBackgroundGPU.h"
|
||||
#endif
|
||||
|
||||
namespace {
|
||||
|
||||
@@ -17,8 +23,8 @@ namespace {
|
||||
constexpr float BAND_LOW_RES_A = 12.0f;
|
||||
constexpr float BAND_HIGH_RES_A = 2.2f;
|
||||
|
||||
constexpr int SECTORS = 36;
|
||||
constexpr int RADIAL_BINS = 120;
|
||||
constexpr int SECTORS = BackgroundBand::SECTORS;
|
||||
constexpr int RADIAL_BINS = BackgroundBand::RADIAL_BINS;
|
||||
|
||||
// A cell with fewer pixels than this has no usable mean.
|
||||
constexpr int MIN_PIXELS_PER_CELL = 20;
|
||||
@@ -83,7 +89,7 @@ float median_of(std::vector<float> &v) {
|
||||
std::optional<BeamCenterEstimate>
|
||||
FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const PixelMask &mask,
|
||||
const std::vector<float> &mean, size_t nthreads,
|
||||
std::optional<std::pair<float, float>> start) {
|
||||
std::optional<std::pair<float, float>> start, bool allow_device) {
|
||||
if (nthreads == 0)
|
||||
nthreads = std::max(1u, std::thread::hardware_concurrency());
|
||||
const auto W = static_cast<int>(experiment.GetXPixelsNumConv());
|
||||
@@ -111,7 +117,6 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe
|
||||
float beam_y = start ? start->second : geom.GetBeamY_pxl();
|
||||
|
||||
constexpr int n_cells = RADIAL_BINS * SECTORS;
|
||||
std::vector<int32_t> cell_of(n_pixels);
|
||||
std::vector<double> sum(n_cells), sum_sq(n_cells), sum_jx(n_cells), sum_jy(n_cells);
|
||||
std::vector<int32_t> count(n_cells), count_all(n_cells);
|
||||
std::vector<float> profile(RADIAL_BINS), d_profile(RADIAL_BINS), clip_limit(n_cells);
|
||||
@@ -127,16 +132,69 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe
|
||||
std::vector<double> block_jy(static_cast<size_t>(BLOCKS) * n_cells);
|
||||
std::vector<int32_t> block_count(static_cast<size_t>(BLOCKS) * n_cells);
|
||||
|
||||
// Whether a pixel can take part at all, which does not depend on the centre.
|
||||
std::vector<char, NoInitAllocator<char>> usable(n_pixels);
|
||||
ParallelFor(BLOCKS, nthreads, [&](int b) {
|
||||
for (size_t i = static_cast<size_t>(block_row[b]) * W; i < static_cast<size_t>(block_row[b + 1]) * W; i++)
|
||||
usable[i] = pixel_mask[i] == 0 && std::isfinite(mean[i]);
|
||||
});
|
||||
// The pixels of each block that fall in the band at the current centre, with their cell and value,
|
||||
// in pixel order from the block's first pixel on. The clipping rounds read these instead of the
|
||||
// whole detector, in the same order.
|
||||
std::vector<int32_t, NoInitAllocator<int32_t>> band_cell(n_pixels);
|
||||
std::vector<float, NoInitAllocator<float>> band_value(n_pixels);
|
||||
std::vector<size_t> band_pixels(BLOCKS);
|
||||
|
||||
// The blocks' cells folded in block order. Each cell is folded on its own, so the cells are split
|
||||
// over the threads and every cell is still summed in the same order.
|
||||
const auto fold = [&](bool with_jacobian) {
|
||||
ParallelChunks(n_cells, nthreads, [&](int c0, int c1) {
|
||||
for (int c = c0; c < c1; c++) {
|
||||
double s = 0, ss = 0, jx = 0, jy = 0;
|
||||
int32_t n = 0;
|
||||
for (int b = 0; b < BLOCKS; b++) {
|
||||
const size_t k = static_cast<size_t>(b) * n_cells + c;
|
||||
s += block_sum[k]; ss += block_sum_sq[k];
|
||||
if (with_jacobian) { jx += block_jx[k]; jy += block_jy[k]; }
|
||||
n += block_count[k];
|
||||
}
|
||||
sum[c] = s; sum_sq[c] = ss; count[c] = n;
|
||||
if (with_jacobian) { sum_jx[c] = jx; sum_jy[c] = jy; }
|
||||
}
|
||||
});
|
||||
};
|
||||
|
||||
// Most pixels lie outside the band. Those clearly outside it in tan(2theta) = rho / lz - by a
|
||||
// margin far above float rounding - skip the square root and both atan2 below; every pixel the exact
|
||||
// test would keep still reaches it.
|
||||
const double tan_lo = std::tan(static_cast<double>(tt_lo)) * (1.0 - 1e-3);
|
||||
const double tan_hi = tt_hi < PI / 2 ? std::tan(static_cast<double>(tt_hi)) * (1.0 + 1e-3) : INFINITY;
|
||||
|
||||
BackgroundBand band{};
|
||||
for (int k = 0; k < 9; k++)
|
||||
band.rot[k] = rot[k];
|
||||
band.pixel_size = pixel_size;
|
||||
band.distance = distance;
|
||||
band.tt_lo = tt_lo;
|
||||
band.tt_hi = tt_hi;
|
||||
band.d_tt = d_tt;
|
||||
band.tan_lo = tan_lo;
|
||||
band.tan_hi = tan_hi;
|
||||
|
||||
#ifndef JFJOCH_USE_CUDA
|
||||
(void) allow_device;
|
||||
#else
|
||||
// With a GPU the two passes over the pixels run there, and only the cells come back.
|
||||
std::unique_ptr<BeamCenterBackgroundGPU> gpu;
|
||||
if (allow_device && get_gpu_count() > 0)
|
||||
gpu = std::make_unique<BeamCenterBackgroundGPU>(W, H, block_row, usable.data(), mean.data());
|
||||
#endif
|
||||
|
||||
float step_x = 0.0f, step_y = 0.0f, sigma_x = 0.0f, sigma_y = 0.0f;
|
||||
float previous_x = 0.0f, previous_y = 0.0f;
|
||||
int reversals = 0;
|
||||
for (int iteration = 0; iteration < MAX_ITERATIONS; iteration++) {
|
||||
// The binning pass about the current centre, then the clip rounds over the pixels it binned.
|
||||
const auto bin_cpu = [&] {
|
||||
ParallelFor(BLOCKS, nthreads, [&](int b) {
|
||||
double *b_sum = block_sum.data() + static_cast<size_t>(b) * n_cells;
|
||||
double *b_sum_sq = block_sum_sq.data() + static_cast<size_t>(b) * n_cells;
|
||||
@@ -148,61 +206,62 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe
|
||||
std::fill(b_jx, b_jx + n_cells, 0.0);
|
||||
std::fill(b_jy, b_jy + n_cells, 0.0);
|
||||
std::fill(b_count, b_count + n_cells, 0);
|
||||
int32_t *cells = band_cell.data() + static_cast<size_t>(block_row[b]) * W;
|
||||
float *values = band_value.data() + static_cast<size_t>(block_row[b]) * W;
|
||||
size_t n_band = 0;
|
||||
for (int y = block_row[b]; y < block_row[b + 1]; y++) {
|
||||
for (int x = 0; x < W; x++) {
|
||||
const size_t i = static_cast<size_t>(y) * W + x;
|
||||
cell_of[i] = -1;
|
||||
if (pixel_mask[i] != 0 || !std::isfinite(mean[i]))
|
||||
if (!usable[i])
|
||||
continue;
|
||||
const float u = (x - beam_x) * pixel_size;
|
||||
const float v = (y - beam_y) * pixel_size;
|
||||
const float lx = rot[0] * u + rot[1] * v + rot[2] * distance;
|
||||
const float ly = rot[3] * u + rot[4] * v + rot[5] * distance;
|
||||
const float lz = rot[6] * u + rot[7] * v + rot[8] * distance;
|
||||
const float rho_sq = lx * lx + ly * ly;
|
||||
if (lz > 0.0f) {
|
||||
const double lz_sq = static_cast<double>(lz) * lz;
|
||||
if (rho_sq < tan_lo * tan_lo * lz_sq || rho_sq > tan_hi * tan_hi * lz_sq)
|
||||
continue;
|
||||
}
|
||||
const float rho = std::sqrt(rho_sq);
|
||||
const float two_theta = std::atan2(rho, lz);
|
||||
if (two_theta < tt_lo || two_theta >= tt_hi || rho == 0.0f)
|
||||
float jac_x, jac_y;
|
||||
const int cell = BackgroundBandCell(band, x, y, beam_x, beam_y, jac_x, jac_y);
|
||||
if (cell < 0)
|
||||
continue;
|
||||
|
||||
const float phi = std::atan2(ly, lx);
|
||||
// Both bins are clamped: a pixel one float ulp below the top of the band divides
|
||||
// to exactly RADIAL_BINS, which is one cell past the end of every accumulator.
|
||||
const int r_bin = std::clamp(static_cast<int>((two_theta - tt_lo) / d_tt), 0, RADIAL_BINS - 1);
|
||||
const int s_bin = std::clamp(static_cast<int>((phi + PI) / (2 * PI) * SECTORS), 0, SECTORS - 1);
|
||||
const int cell = r_bin * SECTORS + s_bin;
|
||||
|
||||
// d(2theta)/d(beam), through the lab coordinate: the detector coordinate depends
|
||||
// on the centre only as (x - beam_x), so moving the centre is moving the pixel.
|
||||
const float denominator = rho * rho + lz * lz;
|
||||
const float g_x = lz * lx / (rho * denominator);
|
||||
const float g_y = lz * ly / (rho * denominator);
|
||||
const float g_z = -rho / denominator;
|
||||
cell_of[i] = cell;
|
||||
cells[n_band] = cell;
|
||||
values[n_band] = mean[i];
|
||||
n_band++;
|
||||
b_count[cell]++;
|
||||
b_sum[cell] += mean[i];
|
||||
b_sum_sq[cell] += static_cast<double>(mean[i]) * mean[i];
|
||||
b_jx[cell] += -pixel_size * (g_x * rot[0] + g_y * rot[3] + g_z * rot[6]);
|
||||
b_jy[cell] += -pixel_size * (g_x * rot[1] + g_y * rot[4] + g_z * rot[7]);
|
||||
b_jx[cell] += jac_x;
|
||||
b_jy[cell] += jac_y;
|
||||
}
|
||||
}
|
||||
band_pixels[b] = n_band;
|
||||
});
|
||||
for (int c = 0; c < n_cells; c++) {
|
||||
double s = 0, ss = 0, jx = 0, jy = 0;
|
||||
int32_t n = 0;
|
||||
for (int b = 0; b < BLOCKS; b++) {
|
||||
const size_t k = static_cast<size_t>(b) * n_cells + c;
|
||||
s += block_sum[k]; ss += block_sum_sq[k];
|
||||
jx += block_jx[k]; jy += block_jy[k];
|
||||
n += block_count[k];
|
||||
fold(true);
|
||||
};
|
||||
const auto clip_cpu = [&] {
|
||||
ParallelFor(BLOCKS, nthreads, [&](int b) {
|
||||
double *b_sum = block_sum.data() + static_cast<size_t>(b) * n_cells;
|
||||
double *b_sum_sq = block_sum_sq.data() + static_cast<size_t>(b) * n_cells;
|
||||
int32_t *b_count = block_count.data() + static_cast<size_t>(b) * n_cells;
|
||||
std::fill(b_sum, b_sum + n_cells, 0.0);
|
||||
std::fill(b_sum_sq, b_sum_sq + n_cells, 0.0);
|
||||
std::fill(b_count, b_count + n_cells, 0);
|
||||
const int32_t *cells = band_cell.data() + static_cast<size_t>(block_row[b]) * W;
|
||||
const float *values = band_value.data() + static_cast<size_t>(block_row[b]) * W;
|
||||
for (size_t j = 0; j < band_pixels[b]; j++) {
|
||||
const int32_t c = cells[j];
|
||||
const float value = values[j];
|
||||
if (clip_limit[c] < 0.0f || value > clip_limit[c])
|
||||
continue;
|
||||
b_count[c]++;
|
||||
b_sum[c] += value;
|
||||
b_sum_sq[c] += static_cast<double>(value) * value;
|
||||
}
|
||||
sum[c] = s; sum_sq[c] = ss; sum_jx[c] = jx; sum_jy[c] = jy; count[c] = n;
|
||||
}
|
||||
});
|
||||
fold(false);
|
||||
};
|
||||
|
||||
for (int iteration = 0; iteration < MAX_ITERATIONS; iteration++) {
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (gpu)
|
||||
gpu->Bin(band, beam_x, beam_y, sum, sum_sq, sum_jx, sum_jy, count);
|
||||
else
|
||||
#endif
|
||||
bin_cpu();
|
||||
|
||||
count_all = count; // the Jacobian sums belong to the unclipped pixel set
|
||||
|
||||
@@ -213,33 +272,12 @@ FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const Pixe
|
||||
const double variance = std::max(sum_sq[c] / count[c] - m * m, 0.0);
|
||||
clip_limit[c] = static_cast<float>(m + CLIP_SIGMA * std::sqrt(variance));
|
||||
}
|
||||
ParallelFor(BLOCKS, nthreads, [&](int b) {
|
||||
double *b_sum = block_sum.data() + static_cast<size_t>(b) * n_cells;
|
||||
double *b_sum_sq = block_sum_sq.data() + static_cast<size_t>(b) * n_cells;
|
||||
int32_t *b_count = block_count.data() + static_cast<size_t>(b) * n_cells;
|
||||
std::fill(b_sum, b_sum + n_cells, 0.0);
|
||||
std::fill(b_sum_sq, b_sum_sq + n_cells, 0.0);
|
||||
std::fill(b_count, b_count + n_cells, 0);
|
||||
const size_t lo = static_cast<size_t>(block_row[b]) * W;
|
||||
const size_t hi = static_cast<size_t>(block_row[b + 1]) * W;
|
||||
for (size_t i = lo; i < hi; i++) {
|
||||
const int32_t c = cell_of[i];
|
||||
if (c < 0 || clip_limit[c] < 0.0f || mean[i] > clip_limit[c])
|
||||
continue;
|
||||
b_count[c]++;
|
||||
b_sum[c] += mean[i];
|
||||
b_sum_sq[c] += static_cast<double>(mean[i]) * mean[i];
|
||||
}
|
||||
});
|
||||
for (int c = 0; c < n_cells; c++) {
|
||||
double s = 0, ss = 0;
|
||||
int32_t n = 0;
|
||||
for (int b = 0; b < BLOCKS; b++) {
|
||||
const size_t k = static_cast<size_t>(b) * n_cells + c;
|
||||
s += block_sum[k]; ss += block_sum_sq[k]; n += block_count[k];
|
||||
}
|
||||
sum[c] = s; sum_sq[c] = ss; count[c] = n;
|
||||
}
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (gpu)
|
||||
gpu->Clip(clip_limit, sum, sum_sq, count);
|
||||
else
|
||||
#endif
|
||||
clip_cpu();
|
||||
}
|
||||
|
||||
// Radial profile: the median over the sectors that have a mean, on rings that are
|
||||
|
||||
@@ -40,10 +40,14 @@ struct BeamCenterEstimate {
|
||||
// `start` is where the walk begins; the centre in the file when it is not given. The walk advances
|
||||
// by a bounded distance per iteration, so where it starts decides how much of its budget is spent
|
||||
// travelling and - on a surface with more than one basin - which fixed point it can reach at all.
|
||||
//
|
||||
// With a GPU the passes over the pixels run on it (BeamCenterBackgroundGPU); allow_device = false
|
||||
// keeps them on the host, which is what the parity test compares against.
|
||||
std::optional<BeamCenterEstimate>
|
||||
FindBeamCenterFromBackground(const DiffractionExperiment &experiment, const PixelMask &mask,
|
||||
const std::vector<float> &mean, size_t nthreads = 0,
|
||||
std::optional<std::pair<float, float>> start = {});
|
||||
std::optional<std::pair<float, float>> start = {},
|
||||
bool allow_device = true);
|
||||
|
||||
// The precision of a centre that is the FFT capture alone, with no walk behind it. The capture is
|
||||
// a half-pixel grid position read off a surface, measured over 75 rotation datasets at a median
|
||||
|
||||
@@ -24,6 +24,12 @@ ADD_LIBRARY(JFJochGeomRefinement STATIC
|
||||
XtalOptimizer.cpp
|
||||
XtalOptimizer.h
|
||||
XtalResidual.h
|
||||
XtalRefine.cpp
|
||||
XtalRefine.h
|
||||
LMSolver.cpp
|
||||
LMSolver.h
|
||||
Dual.h
|
||||
BackgroundBand.h
|
||||
PostRefine.cpp
|
||||
PostRefine.h
|
||||
GeometryRefiner.cpp
|
||||
@@ -37,7 +43,8 @@ ADD_LIBRARY(JFJochGeomRefinement STATIC
|
||||
TARGET_LINK_LIBRARIES(JFJochGeomRefinement Ceres::ceres Eigen3::Eigen JFJochCommon fftw3f)
|
||||
|
||||
IF (JFJOCH_CUDA_AVAILABLE)
|
||||
TARGET_SOURCES(JFJochGeomRefinement PRIVATE BeamCenterFFTGPU.cu BeamCenterFFTGPU.h)
|
||||
TARGET_SOURCES(JFJochGeomRefinement PRIVATE BeamCenterFFTGPU.cu BeamCenterFFTGPU.h
|
||||
BeamCenterBackgroundGPU.cu BeamCenterBackgroundGPU.h)
|
||||
# Same static/dynamic cuFFT choice as the FFT indexer, and for the same reasons - see the long
|
||||
# note in image_analysis/indexing/CMakeLists.txt.
|
||||
IF (JFJOCH_PORTABLE_ONLY AND TARGET CUDA::cufft_static)
|
||||
@@ -46,3 +53,13 @@ IF (JFJOCH_CUDA_AVAILABLE)
|
||||
TARGET_LINK_LIBRARIES(JFJochGeomRefinement CUDA::cufft)
|
||||
ENDIF()
|
||||
ENDIF()
|
||||
|
||||
# The background beam-centre fit bins every pixel with BackgroundBand.h on the host and on the device
|
||||
# and must get the same bits on both (see there). That needs no multiply-add contracted on either side:
|
||||
# GCC and Clang contract by default and nvcc does too, while MSVC does not without /fp:contract.
|
||||
IF (JFJOCH_CUDA_AVAILABLE)
|
||||
SET_SOURCE_FILES_PROPERTIES(BeamCenterBackgroundGPU.cu PROPERTIES COMPILE_OPTIONS "--fmad=false")
|
||||
ENDIF()
|
||||
IF (CMAKE_CXX_COMPILER_ID MATCHES "GNU|Clang")
|
||||
SET_SOURCE_FILES_PROPERTIES(BeamCenterFromBackground.cpp PROPERTIES COMPILE_OPTIONS "-ffp-contract=off")
|
||||
ENDIF()
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
// A forward-mode dual number with N derivative lanes: a value and its gradient with respect to N
|
||||
// parameters. The residuals of the crystal refinement are written as templates over their scalar type,
|
||||
// so the same code runs on a plain double and on this. The value part of every operation is the plain
|
||||
// double arithmetic of the same expression - written the way ceres::Jet writes it, division through the
|
||||
// reciprocal - so a residual evaluated on a Dual has the same value as on a Jet.
|
||||
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
|
||||
#include <Eigen/Core>
|
||||
|
||||
template<int N>
|
||||
struct Dual {
|
||||
double a = 0.0;
|
||||
double v[N] = {};
|
||||
|
||||
Dual() = default;
|
||||
Dual(double value) : a(value) {} // NOLINT: implicit, a constant is a dual with zero derivatives
|
||||
|
||||
static Dual Variable(double value, int lane) {
|
||||
Dual d(value);
|
||||
d.v[lane] = 1.0;
|
||||
return d;
|
||||
}
|
||||
|
||||
Dual &operator+=(const Dual &o) { a += o.a; for (int i = 0; i < N; i++) v[i] += o.v[i]; return *this; }
|
||||
Dual &operator-=(const Dual &o) { a -= o.a; for (int i = 0; i < N; i++) v[i] -= o.v[i]; return *this; }
|
||||
Dual &operator*=(const Dual &o) { *this = *this * o; return *this; }
|
||||
Dual &operator/=(const Dual &o) { *this = *this / o; return *this; }
|
||||
|
||||
friend Dual operator+(const Dual &x) { return x; }
|
||||
friend Dual operator-(const Dual &x) {
|
||||
Dual r(-x.a);
|
||||
for (int i = 0; i < N; i++) r.v[i] = -x.v[i];
|
||||
return r;
|
||||
}
|
||||
|
||||
friend Dual operator+(const Dual &x, const Dual &y) {
|
||||
Dual r(x.a + y.a);
|
||||
for (int i = 0; i < N; i++) r.v[i] = x.v[i] + y.v[i];
|
||||
return r;
|
||||
}
|
||||
friend Dual operator+(const Dual &x, double s) { Dual r = x; r.a += s; return r; }
|
||||
friend Dual operator+(double s, const Dual &x) { Dual r = x; r.a += s; return r; }
|
||||
|
||||
friend Dual operator-(const Dual &x, const Dual &y) {
|
||||
Dual r(x.a - y.a);
|
||||
for (int i = 0; i < N; i++) r.v[i] = x.v[i] - y.v[i];
|
||||
return r;
|
||||
}
|
||||
friend Dual operator-(const Dual &x, double s) { Dual r = x; r.a -= s; return r; }
|
||||
friend Dual operator-(double s, const Dual &x) {
|
||||
Dual r(s - x.a);
|
||||
for (int i = 0; i < N; i++) r.v[i] = -x.v[i];
|
||||
return r;
|
||||
}
|
||||
|
||||
friend Dual operator*(const Dual &x, const Dual &y) {
|
||||
Dual r(x.a * y.a);
|
||||
for (int i = 0; i < N; i++) r.v[i] = x.a * y.v[i] + x.v[i] * y.a;
|
||||
return r;
|
||||
}
|
||||
friend Dual operator*(const Dual &x, double s) {
|
||||
Dual r(x.a * s);
|
||||
for (int i = 0; i < N; i++) r.v[i] = x.v[i] * s;
|
||||
return r;
|
||||
}
|
||||
friend Dual operator*(double s, const Dual &x) { return x * s; }
|
||||
|
||||
friend Dual operator/(const Dual &x, const Dual &y) {
|
||||
const double y_inv = 1.0 / y.a;
|
||||
const double q = x.a * y_inv;
|
||||
Dual r(q);
|
||||
for (int i = 0; i < N; i++) r.v[i] = (x.v[i] - q * y.v[i]) * y_inv;
|
||||
return r;
|
||||
}
|
||||
friend Dual operator/(const Dual &x, double s) {
|
||||
const double s_inv = 1.0 / s;
|
||||
return x * s_inv;
|
||||
}
|
||||
friend Dual operator/(double s, const Dual &y) {
|
||||
const double y_inv = 1.0 / y.a;
|
||||
const double d = -s * y_inv * y_inv;
|
||||
Dual r(s * y_inv);
|
||||
for (int i = 0; i < N; i++) r.v[i] = d * y.v[i];
|
||||
return r;
|
||||
}
|
||||
|
||||
friend bool operator<(const Dual &x, const Dual &y) { return x.a < y.a; }
|
||||
friend bool operator>(const Dual &x, const Dual &y) { return x.a > y.a; }
|
||||
friend bool operator<=(const Dual &x, const Dual &y) { return x.a <= y.a; }
|
||||
friend bool operator>=(const Dual &x, const Dual &y) { return x.a >= y.a; }
|
||||
friend bool operator==(const Dual &x, const Dual &y) { return x.a == y.a; }
|
||||
friend bool operator!=(const Dual &x, const Dual &y) { return x.a != y.a; }
|
||||
|
||||
// The chain rule for a function of one argument: value f, derivative df.
|
||||
Dual Chain(double f, double df) const {
|
||||
Dual r(f);
|
||||
for (int i = 0; i < N; i++) r.v[i] = df * v[i];
|
||||
return r;
|
||||
}
|
||||
|
||||
friend Dual sqrt(const Dual &x) {
|
||||
const double s = std::sqrt(x.a);
|
||||
return x.Chain(s, 0.5 / s);
|
||||
}
|
||||
friend Dual cos(const Dual &x) { return x.Chain(std::cos(x.a), -std::sin(x.a)); }
|
||||
friend Dual sin(const Dual &x) { return x.Chain(std::sin(x.a), std::cos(x.a)); }
|
||||
friend Dual hypot(const Dual &x, const Dual &y, const Dual &z) {
|
||||
// As ceres::hypot(Jet, Jet, Jet): the value is std::hypot, the derivative x/h dx + y/h dy + z/h dz.
|
||||
const double h = std::hypot(x.a, y.a, z.a);
|
||||
Dual r(h);
|
||||
for (int i = 0; i < N; i++) r.v[i] = x.a / h * x.v[i] + y.a / h * y.v[i] + z.a / h * z.v[i];
|
||||
return r;
|
||||
}
|
||||
friend int fpclassify(const Dual &x) { return std::fpclassify(x.a); }
|
||||
};
|
||||
|
||||
// What Eigen needs to hold a Dual in a fixed-size matrix (the reciprocal basis is built in one).
|
||||
namespace Eigen {
|
||||
template<int N>
|
||||
struct NumTraits<Dual<N>> : GenericNumTraits<double> {
|
||||
typedef Dual<N> Real;
|
||||
typedef Dual<N> NonInteger;
|
||||
typedef Dual<N> Nested;
|
||||
typedef Dual<N> Literal;
|
||||
enum {
|
||||
IsComplex = 0, IsInteger = 0, IsSigned = 1, RequireInitialization = 1,
|
||||
ReadCost = 1, AddCost = 1, MulCost = 1
|
||||
};
|
||||
static inline Real epsilon() { return Real(std::numeric_limits<double>::epsilon()); }
|
||||
static inline Real dummy_precision() { return Real(1e-12); }
|
||||
static inline Real highest() { return Real(std::numeric_limits<double>::max()); }
|
||||
static inline Real lowest() { return Real(-std::numeric_limits<double>::max()); }
|
||||
static inline int digits10() { return NumTraits<double>::digits10(); }
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,241 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
// Adapted from https://github.com/ceres-solver/ceres-solver (internal/ceres/polynomial.cc,
|
||||
// include/ceres/internal/sphere_manifold_functions.h, householder_vector.h)
|
||||
// Copyright 2023 Google Inc. All rights reserved.
|
||||
// BSD-3-Clause, see licenses/ceres-solver.txt
|
||||
|
||||
#include "LMSolver.h"
|
||||
|
||||
#include <Eigen/Eigenvalues>
|
||||
|
||||
namespace {
|
||||
void HouseholderVector3(const double x[3], double v[3], double &beta) {
|
||||
const double sigma = x[0] * x[0] + x[1] * x[1];
|
||||
v[0] = x[0];
|
||||
v[1] = x[1];
|
||||
v[2] = 1.0;
|
||||
beta = 0.0;
|
||||
const double x_pivot = x[2];
|
||||
if (sigma <= std::numeric_limits<double>::epsilon()) {
|
||||
if (x_pivot < 0.0)
|
||||
beta = 2.0;
|
||||
return;
|
||||
}
|
||||
const double mu = std::sqrt(x_pivot * x_pivot + sigma);
|
||||
const double v_pivot = (x_pivot <= 0.0) ? x_pivot - mu : -sigma / (x_pivot + mu);
|
||||
beta = 2.0 * v_pivot * v_pivot / (sigma + v_pivot * v_pivot);
|
||||
v[0] /= v_pivot;
|
||||
v[1] /= v_pivot;
|
||||
}
|
||||
|
||||
double Norm3(const double x[3]) {
|
||||
return std::sqrt(x[0] * x[0] + x[1] * x[1] + x[2] * x[2]);
|
||||
}
|
||||
|
||||
using Vector = Eigen::VectorXd;
|
||||
using Matrix = Eigen::MatrixXd;
|
||||
|
||||
double EvaluatePolynomial(const Vector &polynomial, double x) {
|
||||
double v = 0.0;
|
||||
for (int i = 0; i < polynomial.size(); ++i)
|
||||
v = v * x + polynomial(i);
|
||||
return v;
|
||||
}
|
||||
|
||||
void BalanceCompanionMatrix(Matrix &companion_matrix) {
|
||||
Matrix offdiagonal = companion_matrix;
|
||||
offdiagonal.diagonal().setZero();
|
||||
const int degree = static_cast<int>(companion_matrix.rows());
|
||||
const double gamma = 0.9;
|
||||
bool scaling_has_changed;
|
||||
do {
|
||||
scaling_has_changed = false;
|
||||
for (int i = 0; i < degree; ++i) {
|
||||
const double col_norm = offdiagonal.col(i).lpNorm<1>();
|
||||
if (std::fpclassify(col_norm) != FP_ZERO) {
|
||||
const double row_norm = offdiagonal.row(i).lpNorm<1>();
|
||||
int exponent = 0;
|
||||
std::frexp(row_norm / col_norm, &exponent);
|
||||
exponent /= 2;
|
||||
if (exponent != 0) {
|
||||
const double scaled_col_norm = std::ldexp(col_norm, exponent);
|
||||
const double scaled_row_norm = std::ldexp(row_norm, -exponent);
|
||||
if (scaled_col_norm + scaled_row_norm < gamma * (col_norm + row_norm)) {
|
||||
scaling_has_changed = true;
|
||||
offdiagonal.row(i) *= std::ldexp(1.0, -exponent);
|
||||
offdiagonal.col(i) *= std::ldexp(1.0, exponent);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} while (scaling_has_changed);
|
||||
offdiagonal.diagonal() = companion_matrix.diagonal();
|
||||
companion_matrix = offdiagonal;
|
||||
}
|
||||
|
||||
// Real parts of the roots, as Ceres' FindPolynomialRoots (the imaginary parts are not used here).
|
||||
bool FindPolynomialRoots(const Vector &polynomial_in, Vector &real) {
|
||||
if (polynomial_in.size() == 0)
|
||||
return false;
|
||||
int lead = 0;
|
||||
while (lead < polynomial_in.size() - 1 && polynomial_in(lead) == 0.0)
|
||||
++lead;
|
||||
Vector polynomial = polynomial_in.tail(polynomial_in.size() - lead);
|
||||
const int degree = static_cast<int>(polynomial.size()) - 1;
|
||||
if (degree == 0) {
|
||||
real.resize(0);
|
||||
return true;
|
||||
}
|
||||
if (degree == 1) {
|
||||
real.resize(1);
|
||||
real(0) = -polynomial(1) / polynomial(0);
|
||||
return true;
|
||||
}
|
||||
if (degree == 2) {
|
||||
const double a = polynomial(0);
|
||||
const double b = polynomial(1);
|
||||
const double c = polynomial(2);
|
||||
const double D = b * b - 4 * a * c;
|
||||
const double sqrt_D = std::sqrt(std::fabs(D));
|
||||
real.setZero(2);
|
||||
if (D >= 0) {
|
||||
if (b >= 0) {
|
||||
real(0) = (-b - sqrt_D) / (2.0 * a);
|
||||
real(1) = (2.0 * c) / (-b - sqrt_D);
|
||||
} else {
|
||||
real(0) = (2.0 * c) / (-b + sqrt_D);
|
||||
real(1) = (-b + sqrt_D) / (2.0 * a);
|
||||
}
|
||||
} else {
|
||||
real(0) = -b / (2.0 * a);
|
||||
real(1) = -b / (2.0 * a);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
polynomial /= polynomial(0);
|
||||
Matrix companion = Matrix::Zero(degree, degree);
|
||||
companion.diagonal(-1).setOnes();
|
||||
companion.col(degree - 1) = -polynomial.reverse().head(degree);
|
||||
BalanceCompanionMatrix(companion);
|
||||
Eigen::EigenSolver<Matrix> solver(companion, false);
|
||||
if (solver.info() != Eigen::Success)
|
||||
return false;
|
||||
real = solver.eigenvalues().real();
|
||||
return true;
|
||||
}
|
||||
|
||||
void MinimizePolynomial(const Vector &polynomial, double x_min, double x_max,
|
||||
double &optimal_x, double &optimal_value) {
|
||||
optimal_x = (x_min + x_max) / 2.0;
|
||||
optimal_value = EvaluatePolynomial(polynomial, optimal_x);
|
||||
const double x_min_value = EvaluatePolynomial(polynomial, x_min);
|
||||
if (x_min_value < optimal_value) {
|
||||
optimal_value = x_min_value;
|
||||
optimal_x = x_min;
|
||||
}
|
||||
const double x_max_value = EvaluatePolynomial(polynomial, x_max);
|
||||
if (x_max_value < optimal_value) {
|
||||
optimal_value = x_max_value;
|
||||
optimal_x = x_max;
|
||||
}
|
||||
if (polynomial.rows() <= 2)
|
||||
return;
|
||||
|
||||
const int degree = static_cast<int>(polynomial.rows()) - 1;
|
||||
Vector derivative(degree);
|
||||
for (int i = 0; i < degree; ++i)
|
||||
derivative(i) = (degree - i) * polynomial(i);
|
||||
Vector roots_real;
|
||||
if (!FindPolynomialRoots(derivative, roots_real))
|
||||
return;
|
||||
for (int i = 0; i < roots_real.rows(); ++i) {
|
||||
const double root = roots_real(i);
|
||||
if (root < x_min || root > x_max)
|
||||
continue;
|
||||
const double value = EvaluatePolynomial(polynomial, root);
|
||||
if (value < optimal_value) {
|
||||
optimal_value = value;
|
||||
optimal_x = root;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vector FindInterpolatingPolynomial(const std::vector<LMLineSample> &samples) {
|
||||
int num_constraints = 0;
|
||||
for (const auto &s: samples)
|
||||
num_constraints += (s.value_is_valid ? 1 : 0) + (s.gradient_is_valid ? 1 : 0);
|
||||
const int degree = num_constraints - 1;
|
||||
Matrix lhs = Matrix::Zero(num_constraints, num_constraints);
|
||||
Vector rhs = Vector::Zero(num_constraints);
|
||||
int row = 0;
|
||||
for (const auto &s: samples) {
|
||||
if (s.value_is_valid) {
|
||||
for (int j = 0; j <= degree; ++j)
|
||||
lhs(row, j) = std::pow(s.x, degree - j);
|
||||
rhs(row) = s.value;
|
||||
++row;
|
||||
}
|
||||
if (s.gradient_is_valid) {
|
||||
for (int j = 0; j < degree; ++j)
|
||||
lhs(row, j) = (degree - j) * std::pow(s.x, degree - j - 1);
|
||||
rhs(row) = s.gradient;
|
||||
++row;
|
||||
}
|
||||
}
|
||||
Eigen::FullPivLU<Matrix> lu(lhs);
|
||||
return lu.setThreshold(0.0).solve(rhs);
|
||||
}
|
||||
}
|
||||
|
||||
void SpherePlus3(const double x[3], const double delta[2], double out[3]) {
|
||||
const double norm_delta = std::sqrt(delta[0] * delta[0] + delta[1] * delta[1]);
|
||||
if (norm_delta == 0.0) {
|
||||
out[0] = x[0];
|
||||
out[1] = x[1];
|
||||
out[2] = x[2];
|
||||
return;
|
||||
}
|
||||
double v[3], beta;
|
||||
HouseholderVector3(x, v, beta);
|
||||
const double sin_delta_by_delta = std::sin(norm_delta) / norm_delta;
|
||||
const double y[3] = {sin_delta_by_delta * delta[0], sin_delta_by_delta * delta[1], std::cos(norm_delta)};
|
||||
const double vy = v[0] * y[0] + v[1] * y[1] + v[2] * y[2];
|
||||
const double x_norm = Norm3(x);
|
||||
for (int i = 0; i < 3; i++)
|
||||
out[i] = x_norm * (y[i] - v[i] * (beta * vy));
|
||||
}
|
||||
|
||||
void SpherePlusJacobian3(const double x[3], double jacobian[3][2]) {
|
||||
double v[3], beta;
|
||||
HouseholderVector3(x, v, beta);
|
||||
const double x_norm = Norm3(x);
|
||||
for (int i = 0; i < 2; ++i)
|
||||
for (int r = 0; r < 3; ++r)
|
||||
jacobian[r][i] = (-beta * v[i] * v[r] + (r == i ? 1.0 : 0.0)) * x_norm;
|
||||
}
|
||||
|
||||
double LMInterpolatedStepSize(const LMLineSample &lowerbound, const LMLineSample &previous,
|
||||
const LMLineSample ¤t, double min_step, double max_step) {
|
||||
if (!current.value_is_valid)
|
||||
return std::min(std::max(current.x * 0.5, min_step), max_step);
|
||||
|
||||
std::vector<LMLineSample> samples{lowerbound, current};
|
||||
if (previous.value_is_valid)
|
||||
samples.push_back(previous);
|
||||
|
||||
const Vector polynomial = FindInterpolatingPolynomial(samples);
|
||||
double step = 0.0, value = 0.0;
|
||||
MinimizePolynomial(polynomial, min_step, max_step, step, value);
|
||||
for (const auto &s: samples) {
|
||||
if (s.x < min_step || s.x > max_step)
|
||||
continue;
|
||||
const double v = EvaluatePolynomial(polynomial, s.x);
|
||||
if (v < value) {
|
||||
step = s.x;
|
||||
value = v;
|
||||
}
|
||||
}
|
||||
return step;
|
||||
}
|
||||
@@ -0,0 +1,365 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
// The minimiser below follows Ceres Solver's trust-region Levenberg-Marquardt step for step - its
|
||||
// options, its Jacobi scaling, its damping and radius updates, its stopping rules, its box projection,
|
||||
// its projected Armijo line search on bounded problems and its SphereManifold - so that a problem moved
|
||||
// off Ceres takes the same path to the same answer. Adapted from
|
||||
// https://github.com/ceres-solver/ceres-solver (internal/ceres/trust_region_minimizer.cc,
|
||||
// levenberg_marquardt_strategy.cc, trust_region_step_evaluator.cc, line_search.cc, polynomial.cc,
|
||||
// include/ceres/internal/sphere_manifold_functions.h, householder_vector.h)
|
||||
// Copyright 2023 Google Inc. All rights reserved.
|
||||
// BSD-3-Clause, see licenses/ceres-solver.txt
|
||||
//
|
||||
// What it does NOT take from Ceres is the Jacobian: the caller hands over J^T J and J^T r directly,
|
||||
// accumulated however it likes, and the minimiser never sees a row of J. Everything Ceres computes from
|
||||
// the Jacobian - the column norms, the normal equations, the model cost change - is a function of those
|
||||
// two alone. Only the options the crystal refinements use are reproduced: monotonic steps, no inner
|
||||
// iterations, a dense Cholesky of the normal equations.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <chrono>
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
#include <vector>
|
||||
|
||||
#include <Eigen/Dense>
|
||||
|
||||
struct LMBlock {
|
||||
int offset = 0; // into the ambient parameter vector
|
||||
int size = 0; // ambient size, at most 3
|
||||
bool constant = false;
|
||||
bool sphere = false; // Ceres' SphereManifold: the norm is kept, the tangent has size - 1 coordinates
|
||||
double lower[3] = {-std::numeric_limits<double>::max(), -std::numeric_limits<double>::max(),
|
||||
-std::numeric_limits<double>::max()};
|
||||
double upper[3] = {std::numeric_limits<double>::max(), std::numeric_limits<double>::max(),
|
||||
std::numeric_limits<double>::max()};
|
||||
|
||||
int TangentSize() const { return constant ? 0 : (sphere ? size - 1 : size); }
|
||||
};
|
||||
|
||||
struct LMOptions {
|
||||
int max_iterations = 50;
|
||||
double max_time_s = 1e9;
|
||||
};
|
||||
|
||||
enum class LMTermination { Convergence, NoConvergence, Failure };
|
||||
|
||||
struct LMSummary {
|
||||
LMTermination termination = LMTermination::Failure;
|
||||
// Iterations as Ceres counts them in Summary::iterations: iteration 0 included.
|
||||
int iterations = 0;
|
||||
int evaluations = 0;
|
||||
int line_search_steps = 0;
|
||||
double initial_cost = 0.0;
|
||||
double final_cost = 0.0;
|
||||
|
||||
bool IsSolutionUsable() const { return termination != LMTermination::Failure; }
|
||||
};
|
||||
|
||||
// Ceres' SphereManifold<3> for a three-vector: the Householder reflection that takes x to the pole, the
|
||||
// Plus that walks a tangent step along the sphere, and the 3x2 Jacobian of that Plus at zero step.
|
||||
void SpherePlus3(const double x[3], const double delta[2], double out[3]);
|
||||
void SpherePlusJacobian3(const double x[3], double jacobian[3][2]);
|
||||
|
||||
// The polynomial step-size choice of Ceres' Armijo line search, cubic interpolation.
|
||||
struct LMLineSample {
|
||||
double x = 0.0;
|
||||
double value = 0.0;
|
||||
double gradient = 0.0;
|
||||
bool value_is_valid = false;
|
||||
bool gradient_is_valid = false;
|
||||
};
|
||||
double LMInterpolatedStepSize(const LMLineSample &lowerbound, const LMLineSample &previous,
|
||||
const LMLineSample ¤t, double min_step, double max_step);
|
||||
|
||||
// Evaluate is called as eval(x, cost, g, H): x the ambient parameters, cost 1/2 sum of squared
|
||||
// residuals, and - where g and H are not null - the gradient J^T r and J^T J in the TANGENT coordinates
|
||||
// of the non-constant blocks, in block order. It returns false where anything came out non-finite.
|
||||
// x is updated in place on success; it is left untouched on failure.
|
||||
template<class Evaluate>
|
||||
LMSummary SolveLM(std::vector<double> &x_io, const std::vector<LMBlock> &blocks, const LMOptions &options,
|
||||
Evaluate &&eval) {
|
||||
using Vec = Eigen::VectorXd;
|
||||
using Mat = Eigen::MatrixXd;
|
||||
constexpr double kMax = std::numeric_limits<double>::max();
|
||||
|
||||
// Ceres' defaults, which is what the callers always ran with.
|
||||
constexpr double initial_radius = 1e4;
|
||||
constexpr double max_radius = 1e16;
|
||||
constexpr double min_radius = 1e-32;
|
||||
constexpr double min_relative_decrease = 1e-3;
|
||||
constexpr double min_lm_diagonal = 1e-6;
|
||||
constexpr double max_lm_diagonal = 1e32;
|
||||
constexpr int max_consecutive_invalid_steps = 5;
|
||||
constexpr double function_tolerance = 1e-6;
|
||||
constexpr double gradient_tolerance = 1e-10;
|
||||
constexpr double parameter_tolerance = 1e-8;
|
||||
constexpr double sufficient_decrease = 1e-4;
|
||||
constexpr double max_step_contraction = 1e-3;
|
||||
constexpr double min_step_contraction = 0.6;
|
||||
constexpr double min_line_search_step = 1e-9;
|
||||
constexpr int max_line_search_iterations = 20;
|
||||
|
||||
const auto start = std::chrono::steady_clock::now();
|
||||
LMSummary summary;
|
||||
|
||||
int n = 0;
|
||||
bool constrained = false;
|
||||
for (const auto &b: blocks) {
|
||||
for (int j = 0; j < b.size; j++)
|
||||
if (!std::isfinite(x_io[b.offset + j]))
|
||||
return summary;
|
||||
n += b.TangentSize();
|
||||
for (int j = 0; j < b.size; j++) {
|
||||
if (b.constant) {
|
||||
if (x_io[b.offset + j] < b.lower[j] || x_io[b.offset + j] > b.upper[j])
|
||||
return summary;
|
||||
} else {
|
||||
if (b.lower[j] >= b.upper[j])
|
||||
return summary;
|
||||
if (b.lower[j] > -kMax || b.upper[j] < kMax)
|
||||
constrained = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// x (+) delta, block by block, projected onto the bounds - Ceres' ParameterBlock::Plus.
|
||||
const auto plus = [&](const Vec &x, const Vec &delta, Vec &out) {
|
||||
out = x;
|
||||
int t = 0;
|
||||
for (const auto &b: blocks) {
|
||||
if (b.constant)
|
||||
continue;
|
||||
if (b.sphere) {
|
||||
SpherePlus3(x.data() + b.offset, delta.data() + t, out.data() + b.offset);
|
||||
} else {
|
||||
for (int j = 0; j < b.size; j++)
|
||||
out[b.offset + j] = x[b.offset + j] + delta[t + j];
|
||||
}
|
||||
for (int j = 0; j < b.size; j++) {
|
||||
out[b.offset + j] = std::max(out[b.offset + j], b.lower[j]);
|
||||
out[b.offset + j] = std::min(out[b.offset + j], b.upper[j]);
|
||||
}
|
||||
t += b.TangentSize();
|
||||
}
|
||||
};
|
||||
// Norms over the parameters Ceres keeps in its state: the non-constant blocks only.
|
||||
const auto free_norm = [&](const Vec &v) {
|
||||
double s = 0.0;
|
||||
for (const auto &b: blocks)
|
||||
if (!b.constant)
|
||||
for (int j = 0; j < b.size; j++)
|
||||
s += v[b.offset + j] * v[b.offset + j];
|
||||
return std::sqrt(s);
|
||||
};
|
||||
const auto free_max_norm = [&](const Vec &v) {
|
||||
double m = 0.0;
|
||||
for (const auto &b: blocks)
|
||||
if (!b.constant)
|
||||
for (int j = 0; j < b.size; j++)
|
||||
m = std::max(m, std::fabs(v[b.offset + j]));
|
||||
return m;
|
||||
};
|
||||
|
||||
Vec x = Eigen::Map<const Vec>(x_io.data(), static_cast<Eigen::Index>(x_io.size()));
|
||||
if (constrained) {
|
||||
Vec projected;
|
||||
plus(x, Vec::Zero(n), projected);
|
||||
x = projected;
|
||||
}
|
||||
|
||||
// One evaluation point with everything Ceres computes there.
|
||||
struct Point {
|
||||
Vec x;
|
||||
double cost = kMax;
|
||||
Vec g;
|
||||
Mat H;
|
||||
bool valid = false;
|
||||
};
|
||||
const auto evaluate = [&](const Vec &at, Point &p) {
|
||||
p.x = at;
|
||||
p.g.setZero(n);
|
||||
p.H.setZero(n, n);
|
||||
summary.evaluations++;
|
||||
p.valid = eval(at.data(), p.cost, &p.g, &p.H) && std::isfinite(p.cost);
|
||||
if (!p.valid)
|
||||
p.cost = kMax;
|
||||
};
|
||||
|
||||
Point cur;
|
||||
evaluate(x, cur);
|
||||
if (!cur.valid)
|
||||
return summary;
|
||||
summary.initial_cost = cur.cost;
|
||||
|
||||
// Jacobi scaling, fixed from the Jacobian at the starting point.
|
||||
Vec scale(n);
|
||||
for (int i = 0; i < n; i++)
|
||||
scale[i] = 1.0 / (1.0 + std::sqrt(cur.H(i, i)));
|
||||
|
||||
Vec gs, neg_g, projected;
|
||||
Mat Hs;
|
||||
double gradient_max_norm = 0.0;
|
||||
const auto take_point = [&]() {
|
||||
gs = scale.cwiseProduct(cur.g);
|
||||
Hs = scale.asDiagonal() * cur.H * scale.asDiagonal();
|
||||
neg_g = -cur.g;
|
||||
plus(cur.x, neg_g, projected);
|
||||
gradient_max_norm = free_max_norm(cur.x - projected);
|
||||
};
|
||||
take_point();
|
||||
|
||||
double radius = initial_radius;
|
||||
double decrease_factor = 2.0;
|
||||
bool reuse_diagonal = false;
|
||||
Vec diagonal(n);
|
||||
int consecutive_invalid = 0;
|
||||
bool any_successful_step = false;
|
||||
bool step_successful = true; // iteration 0
|
||||
int iteration = 0;
|
||||
|
||||
Point trial; // the last point the line search evaluated, reused as the candidate when it is one
|
||||
|
||||
const auto step_rejected = [&]() {
|
||||
radius = radius / decrease_factor;
|
||||
decrease_factor *= 2.0;
|
||||
reuse_diagonal = true;
|
||||
};
|
||||
|
||||
const auto finish = [&](LMTermination t) {
|
||||
summary.termination = t;
|
||||
summary.final_cost = cur.cost;
|
||||
if (t != LMTermination::Failure)
|
||||
for (int i = 0; i < x.size(); i++)
|
||||
x_io[i] = cur.x[i];
|
||||
return summary;
|
||||
};
|
||||
|
||||
for (;;) {
|
||||
// FinalizeIterationAndCheckIfMinimizerCanContinue
|
||||
summary.iterations++;
|
||||
if (std::chrono::duration<double>(std::chrono::steady_clock::now() - start).count()
|
||||
>= options.max_time_s)
|
||||
return finish(LMTermination::NoConvergence);
|
||||
if (iteration >= options.max_iterations)
|
||||
return finish(LMTermination::NoConvergence);
|
||||
if (step_successful && gradient_max_norm <= gradient_tolerance)
|
||||
return finish(LMTermination::Convergence);
|
||||
if (radius <= min_radius)
|
||||
return finish(LMTermination::Convergence);
|
||||
|
||||
iteration++;
|
||||
step_successful = false;
|
||||
|
||||
// ComputeTrustRegionStep: the damped normal equations of the scaled Jacobian.
|
||||
if (!reuse_diagonal)
|
||||
for (int i = 0; i < n; i++)
|
||||
diagonal[i] = std::min(std::max(Hs(i, i), min_lm_diagonal), max_lm_diagonal);
|
||||
Mat lhs = Hs;
|
||||
for (int i = 0; i < n; i++) {
|
||||
const double d = std::sqrt(diagonal[i] / radius);
|
||||
lhs(i, i) += d * d;
|
||||
}
|
||||
reuse_diagonal = true;
|
||||
Eigen::LLT<Mat, Eigen::Upper> llt(lhs);
|
||||
bool step_valid = false;
|
||||
Vec step;
|
||||
double model_cost_change = 0.0;
|
||||
if (llt.info() == Eigen::Success) {
|
||||
step = -llt.solve(gs);
|
||||
if (step.allFinite()) {
|
||||
model_cost_change = -(step.dot(gs) + 0.5 * step.dot(Hs * step));
|
||||
step_valid = model_cost_change > 0.0;
|
||||
}
|
||||
}
|
||||
if (!step_valid) {
|
||||
if (++consecutive_invalid >= max_consecutive_invalid_steps)
|
||||
return finish(LMTermination::Failure);
|
||||
step_rejected();
|
||||
continue;
|
||||
}
|
||||
consecutive_invalid = 0;
|
||||
Vec delta = step.cwiseProduct(scale);
|
||||
|
||||
bool have_trial = false;
|
||||
if (constrained) {
|
||||
// Projected Armijo line search along delta, cubic interpolation.
|
||||
const double initial_gradient = cur.g.dot(delta);
|
||||
const double direction_max_norm = delta.lpNorm<Eigen::Infinity>();
|
||||
LMLineSample initial{0.0, cur.cost, initial_gradient, true, true};
|
||||
LMLineSample previous, current;
|
||||
const auto line_eval = [&](double alpha, LMLineSample &s) {
|
||||
s = LMLineSample{};
|
||||
s.x = alpha;
|
||||
Vec moved;
|
||||
plus(cur.x, Vec(alpha * delta), moved);
|
||||
evaluate(moved, trial);
|
||||
have_trial = true;
|
||||
if (!trial.valid)
|
||||
return;
|
||||
s.value = trial.cost;
|
||||
s.value_is_valid = true;
|
||||
s.gradient = delta.dot(trial.g);
|
||||
s.gradient_is_valid = std::isfinite(s.gradient);
|
||||
};
|
||||
line_eval(1.0, current);
|
||||
bool success = true;
|
||||
int ls_iterations = 0;
|
||||
while (!current.value_is_valid
|
||||
|| current.value > initial.value + sufficient_decrease * initial_gradient * current.x) {
|
||||
++ls_iterations;
|
||||
if (ls_iterations >= max_line_search_iterations) {
|
||||
success = false;
|
||||
break;
|
||||
}
|
||||
const double alpha = LMInterpolatedStepSize(initial, previous, current,
|
||||
max_step_contraction * current.x,
|
||||
min_step_contraction * current.x);
|
||||
if (alpha * direction_max_norm < min_line_search_step) {
|
||||
success = false;
|
||||
break;
|
||||
}
|
||||
previous = current;
|
||||
line_eval(alpha, current);
|
||||
}
|
||||
summary.line_search_steps += ls_iterations;
|
||||
if (success)
|
||||
delta *= current.x;
|
||||
}
|
||||
|
||||
// ComputeCandidatePointAndEvaluateCost
|
||||
Vec candidate_x;
|
||||
plus(cur.x, delta, candidate_x);
|
||||
Point cand;
|
||||
if (have_trial && trial.x == candidate_x)
|
||||
cand = std::move(trial);
|
||||
else
|
||||
evaluate(candidate_x, cand);
|
||||
|
||||
if (any_successful_step) {
|
||||
const double step_norm = free_norm(cur.x - cand.x);
|
||||
if (step_norm <= parameter_tolerance * (free_norm(cur.x) + parameter_tolerance))
|
||||
return finish(LMTermination::Convergence);
|
||||
}
|
||||
if (std::fabs(cur.cost - cand.cost) <= function_tolerance * cur.cost)
|
||||
return finish(LMTermination::Convergence);
|
||||
|
||||
const double relative_decrease = (cand.cost >= kMax)
|
||||
? std::numeric_limits<double>::lowest()
|
||||
: (cur.cost - cand.cost) / model_cost_change;
|
||||
if (relative_decrease > min_relative_decrease) {
|
||||
any_successful_step = true;
|
||||
step_successful = true;
|
||||
cur = std::move(cand);
|
||||
take_point();
|
||||
radius = radius / std::max(1.0 / 3.0, 1.0 - std::pow(2.0 * relative_decrease - 1.0, 3));
|
||||
radius = std::min(max_radius, radius);
|
||||
decrease_factor = 2.0;
|
||||
reuse_diagonal = false;
|
||||
} else {
|
||||
step_rejected();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -7,84 +7,11 @@
|
||||
|
||||
#include "XtalOptimizer.h"
|
||||
#include "XtalResidual.h"
|
||||
#include "ceres/ceres.h"
|
||||
#include "XtalRefine.h"
|
||||
#include "ceres/rotation.h"
|
||||
#include "Dual.h"
|
||||
#include "LatticeReduction.h"
|
||||
|
||||
// Soft header prior on ONE beam-centre component (the spindle-parallel, gauge-weak one). Residual = w*(b - b0);
|
||||
// the caller sets w so the prior behaves like a sigma-pixel restraint that competes with the (unit-weight)
|
||||
// positional residuals - strong enough to pin the gauge direction, negligible in the well-constrained one.
|
||||
// Soft restraint on one direction of a two-component block: g.(p - p0), weighted. Used for the beam
|
||||
// centre and for the detector tilt, which are the same gauge seen twice (see the gauge block below),
|
||||
// so they take the same direction g and cannot disagree about it.
|
||||
struct GaugeDirectionPrior {
|
||||
GaugeDirectionPrior(double gx, double gy, double p0, double weight)
|
||||
: gx(gx), gy(gy), p0(p0), weight(weight) {}
|
||||
template<typename T>
|
||||
bool operator()(const T *const p, T *residual) const {
|
||||
residual[0] = T(weight) * (T(gx) * p[0] + T(gy) * p[1] - T(p0));
|
||||
return true;
|
||||
}
|
||||
double gx, gy, p0, weight;
|
||||
};
|
||||
|
||||
struct XtalResidualRotationOnlyPrecomp {
|
||||
XtalResidualRotationOnlyPrecomp(const Coord &recip_obs,
|
||||
const CrystalLattice &latt,
|
||||
double h, double k, double l)
|
||||
: s_obs(recip_obs),
|
||||
astar(latt.Astar()), bstar(latt.Bstar()), cstar(latt.Cstar()),
|
||||
h(h), k(k), l(l) {
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
bool operator()(const T *const rot_aa, T *residual) const {
|
||||
const T astar_unrot[3] = {T(astar.x), T(astar.y), T(astar.z)};
|
||||
const T bstar_unrot[3] = {T(bstar.x), T(bstar.y), T(bstar.z)};
|
||||
const T cstar_unrot[3] = {T(cstar.x), T(cstar.y), T(cstar.z)};
|
||||
|
||||
T astar_rot[3], bstar_rot[3], cstar_rot[3];
|
||||
|
||||
const AngleAxisRotator<T> rot(rot_aa);
|
||||
rot.Rotate(astar_unrot, astar_rot);
|
||||
rot.Rotate(bstar_unrot, bstar_rot);
|
||||
rot.Rotate(cstar_unrot, cstar_rot);
|
||||
|
||||
const Eigen::Matrix<T, 3, 1> s_pred(T(h) * astar_rot[0] + T(k) * bstar_rot[0] + T(l) * cstar_rot[0],
|
||||
T(h) * astar_rot[1] + T(k) * bstar_rot[1] + T(l) * cstar_rot[1],
|
||||
T(h) * astar_rot[2] + T(k) * bstar_rot[2] + T(l) * cstar_rot[2]
|
||||
);
|
||||
|
||||
// Residual in reciprocal space
|
||||
residual[0] = T(s_obs.x) - s_pred[0];
|
||||
residual[1] = T(s_obs.y) - s_pred[1];
|
||||
residual[2] = T(s_obs.z) - s_pred[2];
|
||||
return true;
|
||||
}
|
||||
|
||||
const Coord s_obs;
|
||||
const Coord astar, bstar, cstar;
|
||||
const double h, k, l;
|
||||
};
|
||||
|
||||
// Regularizer: penalises ||rot_aa|| to prefer the smallest rotation that
|
||||
// explains the data. Weight should be chosen in the same units as the
|
||||
// reciprocal-space residuals (Å⁻¹ per radian). A value of ~0.01–0.1 is
|
||||
// typically enough to break degeneracy without biasing the solution.
|
||||
struct RotationNormRegularizer {
|
||||
explicit RotationNormRegularizer(double weight) : weight(weight) {}
|
||||
|
||||
template<typename T>
|
||||
bool operator()(const T *const rot_aa, T *residual) const {
|
||||
residual[0] = T(weight) * rot_aa[0];
|
||||
residual[1] = T(weight) * rot_aa[1];
|
||||
residual[2] = T(weight) * rot_aa[2];
|
||||
return true;
|
||||
}
|
||||
|
||||
const double weight;
|
||||
};
|
||||
|
||||
// Prior confidence weight per spot: how strong the spot is FOR ITS RESOLUTION. The frame's spots are
|
||||
// ordered by resolution and cut into equal-count shells, and each intensity is divided by its shell
|
||||
// median. Refinement needs the high-resolution spots (they carry the cell and distance information) and
|
||||
@@ -153,12 +80,11 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
const int num_threads) {
|
||||
try {
|
||||
// A coplanar basis has no reciprocal cell: 1/V is infinite, every predicted reciprocal vector
|
||||
// comes out NaN, and Ceres fails on the very first evaluation - after dumping the offending
|
||||
// block to stderr. There is nothing for the refinement to recover here, so refuse the lattice
|
||||
// before the problem is built rather than let the solver discover it. The check has to be on
|
||||
// the vectors: this close to flat, float cell angles no longer carry even the SIGN of the
|
||||
// metric determinant, and the triclinic branch of XtalResidual then clamps c into the a-b
|
||||
// plane and divides by the zero volume that makes.
|
||||
// comes out NaN, and the solver fails on the very first evaluation. There is nothing for the
|
||||
// refinement to recover here, so refuse the lattice before the problem is built rather than let
|
||||
// the solver discover it. The check has to be on the vectors: this close to flat, float cell
|
||||
// angles no longer carry even the SIGN of the metric determinant, and the triclinic branch of
|
||||
// XtalResidual then clamps c into the a-b plane and divides by the zero volume that makes.
|
||||
if (data.latt.VolumeFraction() < MIN_BASIS_VOLUME_FRACTION)
|
||||
return false;
|
||||
|
||||
@@ -168,24 +94,21 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
double beta = data.latt.GetUnitCell().beta;
|
||||
|
||||
// Initial guess for the parameters
|
||||
double beam[2] = {data.geom.GetBeamX_pxl(), data.geom.GetBeamY_pxl()};
|
||||
double distance_mm = data.geom.GetDetectorDistance_mm();
|
||||
const double distance_mm = data.geom.GetDetectorDistance_mm();
|
||||
|
||||
double detector_rot[2] = {data.geom.GetPoniRot1_rad(), data.geom.GetPoniRot2_rad()};
|
||||
|
||||
// The per-frame constants of the reduced residual (see XtalFrameConstants), one entry per frame
|
||||
// that contributes. Reserved up front and never grown past that, so the residual blocks' pointers
|
||||
// into it stay valid, and declared before the problem so that it outlives it.
|
||||
std::vector<XtalFrameConstants> frame_const;
|
||||
frame_const.reserve(spots.size());
|
||||
|
||||
ceres::Problem problem;
|
||||
|
||||
double latt_vec0[3] = {0.0, 0.0, 0.0};
|
||||
double latt_vec1[3] = {0.0, 0.0, 0.0};
|
||||
double latt_vec2[3] = {0.0, 0.0, 0.0};
|
||||
|
||||
double rot_vec[3] = {1, 0, 0};
|
||||
XtalRefineProblem problem;
|
||||
problem.crystal_system = data.crystal_system;
|
||||
problem.distance_mm = distance_mm;
|
||||
double *beam = problem.beam;
|
||||
beam[0] = data.geom.GetBeamX_pxl();
|
||||
beam[1] = data.geom.GetBeamY_pxl();
|
||||
double *detector_rot = problem.detector_rot;
|
||||
detector_rot[0] = data.geom.GetPoniRot1_rad();
|
||||
detector_rot[1] = data.geom.GetPoniRot2_rad();
|
||||
double *latt_vec0 = problem.latt_vec0;
|
||||
double *latt_vec1 = problem.latt_vec1;
|
||||
double *latt_vec2 = problem.latt_vec2;
|
||||
double *rot_vec = problem.rot_vec;
|
||||
|
||||
switch (data.crystal_system) {
|
||||
case gemmi::CrystalSystem::Orthorhombic:
|
||||
@@ -252,14 +175,15 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
const double sin_rot3 = std::sin(data.geom.GetPoniRot3_rad());
|
||||
|
||||
// Per-image rotation refinement frees only the beam and the orientation and holds the other five
|
||||
// blocks constant, so the seven-block residual makes Ceres differentiate 17 parameters to use 5.
|
||||
// Where that is the configuration, use the reduced residual instead - identical fit, Jet<5>
|
||||
// autodiff. Any other combination (stills also free the cell, the offline refiner frees distance
|
||||
// and detector angles) keeps the general form below.
|
||||
// blocks constant. Where that is the configuration, the solver uses the reduced residual - the
|
||||
// identical fit, with the crystal half worked out once (see XtalResidualBeamOrientation). Any
|
||||
// other combination (stills also free the cell, the rotation indexer frees detector angles and
|
||||
// spindle) keeps the general form.
|
||||
const bool beam_and_orientation_only = data.refine_beam_center
|
||||
&& !data.refine_detector_angles
|
||||
&& !data.refine_rotation_axis
|
||||
&& !data.refine_unit_cell;
|
||||
problem.beam_and_orientation_only = beam_and_orientation_only;
|
||||
|
||||
// Sum of w^2 over the spots that entered - the beam prior below is scaled by it so that its
|
||||
// strength relative to the data is the same weighted or not. Equals the residual block count
|
||||
@@ -281,9 +205,8 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
rot_matr = data.axis->GetTransformationAngle(angle_deg);
|
||||
}
|
||||
|
||||
if (beam_and_orientation_only)
|
||||
frame_const.emplace_back(detector_rot, rot_vec, angle_rad, latt_vec1, latt_vec2,
|
||||
data.crystal_system);
|
||||
const int frame_index = static_cast<int>(problem.frame_angle_rad.size());
|
||||
problem.frame_angle_rad.push_back(angle_rad);
|
||||
|
||||
// Add residuals for each point
|
||||
for (size_t j = 0; j < spots[i].size(); j++) {
|
||||
@@ -333,7 +256,7 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
const double weight_sq = weight.empty() ? 1.0 : weight[j] * weight[j];
|
||||
effective_spots += weight_sq;
|
||||
|
||||
const XtalResidual residual(pt.x, pt.y,
|
||||
problem.residuals.emplace_back(pt.x, pt.y,
|
||||
data.geom.GetWavelength_A(),
|
||||
data.geom.GetPixelSize_mm(),
|
||||
cos_rot3, sin_rot3,
|
||||
@@ -341,38 +264,14 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
h, k, l,
|
||||
data.crystal_system,
|
||||
data.geom.GetOrientation());
|
||||
|
||||
// Ceres has no per-residual weight; ScaledLoss(nullptr, a) multiplies the squared
|
||||
// residual by the constant a, i.e. it applies a weight of sqrt(a) to the residual.
|
||||
ceres::LossFunction *loss = weight.empty()
|
||||
? nullptr
|
||||
: new ceres::ScaledLoss(nullptr, weight_sq,
|
||||
ceres::TAKE_OWNERSHIP);
|
||||
|
||||
if (beam_and_orientation_only)
|
||||
problem.AddResidualBlock(
|
||||
new ceres::AutoDiffCostFunction<XtalResidualBeamOrientation, 3, 2, 3>(
|
||||
new XtalResidualBeamOrientation(residual, distance_mm, frame_const.back())),
|
||||
loss,
|
||||
beam,
|
||||
latt_vec0
|
||||
);
|
||||
else
|
||||
problem.AddResidualBlock(
|
||||
new ceres::AutoDiffCostFunction<XtalResidualFixedDistance, 3, 2, 2, 3, 3, 3, 3>(
|
||||
new XtalResidualFixedDistance(residual, distance_mm)),
|
||||
loss,
|
||||
beam,
|
||||
detector_rot,
|
||||
rot_vec,
|
||||
latt_vec0,
|
||||
latt_vec1,
|
||||
latt_vec2
|
||||
);
|
||||
problem.frame.push_back(frame_index);
|
||||
// A per-residual weight w enters the squared residual as w^2.
|
||||
if (!weight.empty())
|
||||
problem.weight_sq.push_back(weight_sq);
|
||||
}
|
||||
}
|
||||
|
||||
if (problem.NumResidualBlocks() < data.min_spots)
|
||||
if (static_cast<int64_t>(problem.residuals.size()) < data.min_spots)
|
||||
return false;
|
||||
|
||||
// The gauge direction of a single-axis rotation experiment - parallel to the spindle - written
|
||||
@@ -440,9 +339,8 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
const double gauge_w = data.geom.GetPixelSize_mm() / (distance_mm * data.geom.GetWavelength_A())
|
||||
* std::sqrt(effective_spots) / sigma_px;
|
||||
|
||||
if (!data.refine_beam_center)
|
||||
problem.SetParameterBlockConstant(beam);
|
||||
else if (data.axis) {
|
||||
problem.beam_constant = !data.refine_beam_center;
|
||||
if (data.refine_beam_center && data.axis) {
|
||||
// Gauge handling (single-axis rotation): rotating the whole experiment about the spindle leaves every
|
||||
// spot position unchanged, so the beam-centre component PARALLEL to the spindle is a null/gauge-weak
|
||||
// direction. Refining it freely lets it wander (~+3 px) and absorb centroid systematics into a wrong
|
||||
@@ -450,23 +348,19 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
// does drift - it is only LaB6-monitored to ~a few px), RESTRAIN it toward the header with a soft
|
||||
// prior: the gauge direction has ~zero data sensitivity so the prior pins it near the header, while a
|
||||
// real, well-supported drift can still overcome it.
|
||||
problem.AddResidualBlock(
|
||||
new ceres::AutoDiffCostFunction<GaugeDirectionPrior, 1, 2>(
|
||||
new GaugeDirectionPrior(gauge_beam_x, gauge_beam_y,
|
||||
gauge_beam_x * beam[0] + gauge_beam_y * beam[1], gauge_w)),
|
||||
nullptr, beam);
|
||||
problem.priors.push_back({XtalRefinePrior::Block::Beam, gauge_beam_x, gauge_beam_y,
|
||||
gauge_beam_x * beam[0] + gauge_beam_y * beam[1], gauge_w});
|
||||
}
|
||||
|
||||
// Distance, detector angles, rotation axis and cell are parameter blocks only in the general
|
||||
// seven-block residual; the reduced one bakes them in, so there is nothing left to configure.
|
||||
if (!beam_and_orientation_only) {
|
||||
if (!data.refine_detector_angles) {
|
||||
problem.SetParameterBlockConstant(detector_rot);
|
||||
} else {
|
||||
problem.detector_rot_constant = !data.refine_detector_angles;
|
||||
if (data.refine_detector_angles) {
|
||||
const double rot_range = 3.0 / 180.0 * PI;
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
problem.SetParameterLowerBound(detector_rot, i, detector_rot[i] - rot_range);
|
||||
problem.SetParameterUpperBound(detector_rot, i, detector_rot[i] + rot_range);
|
||||
problem.detector_rot_lower[i] = detector_rot[i] - rot_range;
|
||||
problem.detector_rot_upper[i] = detector_rot[i] + rot_range;
|
||||
}
|
||||
// The same gauge as the beam prior above, described a second time: the tilt moves the
|
||||
// direct beam exactly as the beam centre does, at D/pixel px per radian, so leaving
|
||||
@@ -497,20 +391,15 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
if (budget[i] <= 0.0)
|
||||
continue;
|
||||
problem.AddResidualBlock(
|
||||
new ceres::AutoDiffCostFunction<GaugeDirectionPrior, 1, 2>(
|
||||
new GaugeDirectionPrior(dirs[i][0], dirs[i][1],
|
||||
dirs[i][0] * detector_rot[0]
|
||||
+ dirs[i][1] * detector_rot[1],
|
||||
gauge_w * (sigma_px / budget[i]) * lever)),
|
||||
nullptr, detector_rot);
|
||||
problem.priors.push_back({XtalRefinePrior::Block::DetectorRot, dirs[i][0], dirs[i][1],
|
||||
dirs[i][0] * detector_rot[0] + dirs[i][1] * detector_rot[1],
|
||||
gauge_w * (sigma_px / budget[i]) * lever});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!data.refine_rotation_axis) {
|
||||
problem.SetParameterBlockConstant(rot_vec);
|
||||
} else {
|
||||
problem.rot_vec_constant = !data.refine_rotation_axis;
|
||||
if (data.refine_rotation_axis) {
|
||||
// Only the DIRECTION of the goniometer axis is a parameter. The residual applies
|
||||
// angle_rad * |rot_vec|, so a free three-vector also fits a rotation SCALE - which
|
||||
// GoniometerAxis::Axis() then normalises away, leaving the candidate scored by
|
||||
@@ -520,63 +409,46 @@ bool XtalOptimizerInternal(XtalOptimizerData &data,
|
||||
// data it recovers 54 % of a known scale error, repeated first passes on one dataset
|
||||
// disagree with each other in SIGN, and on the one dataset with a real 1.3 % stage
|
||||
// fault it comes out negative. The rotation scale is measured properly, once, with
|
||||
// four gates and a jackknife, in PostRefine.
|
||||
problem.SetManifold(rot_vec, new ceres::SphereManifold<3>);
|
||||
// four gates and a jackknife, in PostRefine. Refined on the sphere (see SolveXtalRefine).
|
||||
}
|
||||
|
||||
if (!data.refine_unit_cell) {
|
||||
problem.SetParameterBlockConstant(latt_vec1);
|
||||
problem.SetParameterBlockConstant(latt_vec2);
|
||||
} else {
|
||||
problem.latt_vec1_constant = !data.refine_unit_cell;
|
||||
problem.latt_vec2_constant = !data.refine_unit_cell;
|
||||
if (data.refine_unit_cell) {
|
||||
// Parameter bounds
|
||||
// Lengths
|
||||
for (int i = 0; i < 3; ++i) {
|
||||
problem.SetParameterLowerBound(latt_vec1, i, data.min_length_A);
|
||||
problem.SetParameterUpperBound(latt_vec1, i, data.max_length_A);
|
||||
problem.latt_vec1_lower[i] = data.min_length_A;
|
||||
problem.latt_vec1_upper[i] = data.max_length_A;
|
||||
}
|
||||
|
||||
if (data.crystal_system == gemmi::CrystalSystem::Monoclinic) {
|
||||
const double beta_lo = std::max(1e-6, PI * (data.min_angle_deg / 180.0));
|
||||
const double beta_hi = std::min(PI - 1e-6, PI * (data.max_angle_deg / 180.0));
|
||||
problem.SetParameterLowerBound(latt_vec2, 0, beta_lo);
|
||||
problem.SetParameterUpperBound(latt_vec2, 0, beta_hi);
|
||||
problem.latt_vec2_constant = false;
|
||||
problem.latt_vec2_lower[0] = std::max(1e-6, PI * (data.min_angle_deg / 180.0));
|
||||
problem.latt_vec2_upper[0] = std::min(PI - 1e-6, PI * (data.max_angle_deg / 180.0));
|
||||
} else if (data.crystal_system == gemmi::CrystalSystem::Triclinic) {
|
||||
// α, β, γ bounds (radians)
|
||||
const double alo = PI * (data.min_angle_deg / 180.0);
|
||||
const double ahi = PI * (data.max_angle_deg / 180.0);
|
||||
for (int i = 0; i < 3; ++i) {
|
||||
problem.SetParameterLowerBound(latt_vec2, i, alo);
|
||||
problem.SetParameterUpperBound(latt_vec2, i, ahi);
|
||||
problem.latt_vec2_lower[i] = alo;
|
||||
problem.latt_vec2_upper[i] = ahi;
|
||||
}
|
||||
} else {
|
||||
// Orthorhombic / Tetragonal / Cubic / Hexagonal:
|
||||
// latt_vec2 has no meaning for these systems — always freeze it.
|
||||
problem.SetParameterBlockConstant(latt_vec2);
|
||||
problem.latt_vec2_constant = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Configure solver
|
||||
ceres::Solver::Options options;
|
||||
// Normal equations, not QR. The problem is very tall and thin - thousands of spots against at
|
||||
// most 17 parameters - and that is the shape DENSE_QR handles worst: it copies the Jacobian out
|
||||
// of Ceres' row-major storage into a column-major buffer on every solve, and Eigen's blocked
|
||||
// Householder then degenerates to the unblocked path because its block size is min(48, columns).
|
||||
// Accumulating J^T J reads the Jacobian once instead. Both solve the same damped system, so the
|
||||
// step is the same to round-off; the column scaling Ceres applies by default and the LM diagonal
|
||||
// keep the squared condition number in hand.
|
||||
options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY;
|
||||
options.minimizer_progress_to_stdout = false;
|
||||
// Stopping rule: a bound on iterations is reproducible, a bound on wall-clock time is not (see
|
||||
// XtalOptimizerData::max_iterations).
|
||||
if (data.max_iterations > 0)
|
||||
options.max_num_iterations = data.max_iterations;
|
||||
problem.options.max_iterations = data.max_iterations;
|
||||
else
|
||||
options.max_solver_time_in_seconds = data.max_time;
|
||||
options.logging_type = ceres::LoggingType::SILENT;
|
||||
options.num_threads = num_threads; // usually 1 (called from many threads); caller may raise it
|
||||
ceres::Solver::Summary summary;
|
||||
|
||||
// Run optimization
|
||||
ceres::Solve(options, &problem, &summary);
|
||||
problem.options.max_time_s = data.max_time;
|
||||
const LMSummary summary = SolveXtalRefine(problem, num_threads);
|
||||
|
||||
// Only a genuine numerical failure is rejected here: a solve that ran out of iterations or
|
||||
// out of time but still descended counts as usable, which is what the real-time caller
|
||||
@@ -653,7 +525,7 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data,
|
||||
return false;
|
||||
|
||||
// Parameter: angle-axis for the extra rotation. Identity == {0,0,0}.
|
||||
double rot_aa[3] = {0.0, 0.0, 0.0};
|
||||
std::vector<double> rot_aa = {0.0, 0.0, 0.0};
|
||||
|
||||
// Spot selection by current indexing (same approach as XtalOptimizerInternal)
|
||||
const Coord a0 = data.latt.Vec0();
|
||||
@@ -662,7 +534,12 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data,
|
||||
|
||||
const float tol_sq = tolerance * tolerance;
|
||||
|
||||
ceres::Problem problem;
|
||||
// Each selected spot: its observed reciprocal vector and the indices it is fitted to.
|
||||
struct Observation {
|
||||
Coord s_obs;
|
||||
double h, k, l;
|
||||
};
|
||||
std::vector<Observation> observations;
|
||||
|
||||
for (const auto &pt : spots) {
|
||||
if (!data.index_ice_rings && pt.ice_ring)
|
||||
@@ -697,42 +574,68 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data,
|
||||
if (data.axis.has_value())
|
||||
s_obs = data.axis->GetTransformationAngle(pt.phi) * s_obs;
|
||||
|
||||
auto *cost =
|
||||
new ceres::AutoDiffCostFunction<XtalResidualRotationOnlyPrecomp, 3, 3>(
|
||||
new XtalResidualRotationOnlyPrecomp(s_obs, data.latt, h, k, l)
|
||||
);
|
||||
|
||||
problem.AddResidualBlock(cost, nullptr, rot_aa);
|
||||
observations.push_back({s_obs, h, k, l});
|
||||
}
|
||||
|
||||
if (problem.NumResidualBlocks() < data.min_spots)
|
||||
if (static_cast<int64_t>(observations.size()) < data.min_spots)
|
||||
return false;
|
||||
|
||||
// Regularization: prefer the smallest rotation correction that fits the
|
||||
// data. This is essential when spots are nearly coplanar in reciprocal
|
||||
// space (e.g. still images), where the rotation component perpendicular
|
||||
// to the scattering plane is otherwise underdetermined.
|
||||
// The weight is in Å⁻¹ rad⁻¹; tune relative to your typical residual.
|
||||
{
|
||||
const double reg_weight = 0.05; // e.g. 0.05
|
||||
problem.AddResidualBlock(
|
||||
new ceres::AutoDiffCostFunction<RotationNormRegularizer, 3, 3>(
|
||||
new RotationNormRegularizer(reg_weight)),
|
||||
nullptr, rot_aa);
|
||||
}
|
||||
// Residual: s_obs - R(rot_aa) (h a* + k b* + l c*), the reciprocal basis rotated once per
|
||||
// evaluation and shared by every spot.
|
||||
//
|
||||
// Regularization: prefer the smallest rotation correction that fits the data, w * rot_aa. This is
|
||||
// essential when spots are nearly coplanar in reciprocal space (e.g. still images), where the
|
||||
// rotation component perpendicular to the scattering plane is otherwise underdetermined. The
|
||||
// weight is in A^-1 rad^-1, relative to the typical residual.
|
||||
const double reg_weight = 0.05;
|
||||
const Coord astar = data.latt.Astar(), bstar = data.latt.Bstar(), cstar = data.latt.Cstar();
|
||||
const auto evaluate = [&](const double *aa, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) {
|
||||
using D = Dual<3>;
|
||||
const D aa_d[3] = {D::Variable(aa[0], 0), D::Variable(aa[1], 1), D::Variable(aa[2], 2)};
|
||||
const AngleAxisRotator<D> rot(aa_d);
|
||||
const double astar_unrot[3] = {astar.x, astar.y, astar.z};
|
||||
const double bstar_unrot[3] = {bstar.x, bstar.y, bstar.z};
|
||||
const double cstar_unrot[3] = {cstar.x, cstar.y, cstar.z};
|
||||
D astar_rot[3], bstar_rot[3], cstar_rot[3];
|
||||
rot.Rotate(astar_unrot, astar_rot);
|
||||
rot.Rotate(bstar_unrot, bstar_rot);
|
||||
rot.Rotate(cstar_unrot, cstar_rot);
|
||||
|
||||
ceres::Solver::Options options;
|
||||
options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY; // tall and thin, as above
|
||||
options.minimizer_progress_to_stdout = false;
|
||||
cost = 0.0;
|
||||
const auto add = [&](double r, const double *J) {
|
||||
cost += 0.5 * r * r;
|
||||
if (!g)
|
||||
return;
|
||||
for (int i = 0; i < 3; i++) {
|
||||
(*g)[i] += J[i] * r;
|
||||
for (int j = 0; j < 3; j++)
|
||||
(*H)(i, j) += J[i] * J[j];
|
||||
}
|
||||
};
|
||||
for (const auto &o: observations) {
|
||||
const double s_obs[3] = {o.s_obs.x, o.s_obs.y, o.s_obs.z};
|
||||
for (int c = 0; c < 3; c++) {
|
||||
const D pred = o.h * astar_rot[c] + o.k * bstar_rot[c] + o.l * cstar_rot[c];
|
||||
const double J[3] = {-pred.v[0], -pred.v[1], -pred.v[2]};
|
||||
add(s_obs[c] - pred.a, J);
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < 3; c++) {
|
||||
double J[3] = {0.0, 0.0, 0.0};
|
||||
J[c] = reg_weight;
|
||||
add(reg_weight * aa[c], J);
|
||||
}
|
||||
return std::isfinite(cost) && (!g || (g->allFinite() && H->allFinite()));
|
||||
};
|
||||
|
||||
std::vector<LMBlock> blocks(1);
|
||||
blocks[0].size = 3;
|
||||
LMOptions options;
|
||||
if (data.max_iterations > 0)
|
||||
options.max_num_iterations = data.max_iterations;
|
||||
options.max_iterations = data.max_iterations;
|
||||
else
|
||||
options.max_solver_time_in_seconds = data.max_time;
|
||||
options.logging_type = ceres::LoggingType::SILENT;
|
||||
options.num_threads = 1;
|
||||
|
||||
ceres::Solver::Summary summary;
|
||||
ceres::Solve(options, &problem, &summary);
|
||||
options.max_time_s = data.max_time;
|
||||
const LMSummary summary = SolveLM(rot_aa, blocks, options, evaluate);
|
||||
|
||||
if (!summary.IsSolutionUsable())
|
||||
return false;
|
||||
@@ -747,7 +650,7 @@ bool XtalOptimizerRotationOnly(XtalOptimizerData &data,
|
||||
// rotating the reciprocal vectors (a*, b*, c*) by the same R. No
|
||||
// transpose or inversion of R is needed here.
|
||||
double R_raw[9];
|
||||
ceres::AngleAxisToRotationMatrix(rot_aa, R_raw); // row-major 3x3
|
||||
ceres::AngleAxisToRotationMatrix(rot_aa.data(), R_raw); // row-major 3x3
|
||||
|
||||
Eigen::Matrix3d R;
|
||||
R << R_raw[0], R_raw[3], R_raw[6],
|
||||
|
||||
@@ -62,7 +62,7 @@ struct XtalOptimizerData {
|
||||
std::optional<Coord> angle_axis;
|
||||
};
|
||||
|
||||
// num_threads sets the Ceres solver thread count for the internal least-squares refine. It defaults
|
||||
// num_threads sets the thread count of the internal least-squares refine (the answer does not depend on it). It defaults
|
||||
// to 1 because XtalOptimizer is usually called from many threads at once; raise it only when a caller
|
||||
// runs a small number of refinements concurrently and wants each to use several cores.
|
||||
bool XtalOptimizer(XtalOptimizerData &data, std::span<const std::vector<SpotToSave>> spots,
|
||||
|
||||
@@ -0,0 +1,334 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#include "XtalRefine.h"
|
||||
|
||||
#include <array>
|
||||
|
||||
#include "Dual.h"
|
||||
#include "../../common/ParallelFor.h"
|
||||
|
||||
namespace {
|
||||
// Ambient layout of the parameter vector and the order of the blocks in it.
|
||||
constexpr int OFF_BEAM = 0, OFF_ROT = 2, OFF_AXIS = 4, OFF_P0 = 7, OFF_LEN = 10, OFF_ANG = 13, N_AMBIENT = 16;
|
||||
|
||||
// Derivative lanes. The observed half of a residual depends on beam, detector angles and spindle
|
||||
// (OBS lanes), the predicted half on orientation and cell (PRED lanes); each half is carried on a
|
||||
// dual number of its own width and the two are put side by side only in the sums.
|
||||
constexpr int OBS = 6, PRED = 9, LANES = OBS + PRED;
|
||||
using DO = Dual<OBS>;
|
||||
using DP = Dual<PRED>;
|
||||
|
||||
// The reduced problem: beam (2) and orientation (3) only.
|
||||
constexpr int R_OBS = 2, R_PRED = 3, R_LANES = R_OBS + R_PRED;
|
||||
|
||||
// Sums over one block of residuals, in lane coordinates.
|
||||
template<int L>
|
||||
struct Sums {
|
||||
double cost = 0.0;
|
||||
double g[L] = {};
|
||||
double H[L][L] = {}; // upper triangle
|
||||
|
||||
void Add(double r, const double *J, double w2) {
|
||||
cost += 0.5 * w2 * r * r;
|
||||
for (int i = 0; i < L; i++) {
|
||||
const double wj = w2 * J[i];
|
||||
g[i] += wj * r;
|
||||
for (int j = i; j < L; j++)
|
||||
H[i][j] += wj * J[j];
|
||||
}
|
||||
}
|
||||
|
||||
void Add(const Sums &o) {
|
||||
cost += o.cost;
|
||||
for (int i = 0; i < L; i++) {
|
||||
g[i] += o.g[i];
|
||||
for (int j = i; j < L; j++)
|
||||
H[i][j] += o.H[i][j];
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
std::vector<LMBlock> MakeBlocks(const XtalRefineProblem &p) {
|
||||
std::vector<LMBlock> blocks(6);
|
||||
blocks[0].offset = OFF_BEAM;
|
||||
blocks[0].size = 2;
|
||||
blocks[0].constant = p.beam_constant;
|
||||
|
||||
blocks[1].offset = OFF_ROT;
|
||||
blocks[1].size = 2;
|
||||
blocks[1].constant = p.beam_and_orientation_only || p.detector_rot_constant;
|
||||
for (int j = 0; j < 2; j++) {
|
||||
blocks[1].lower[j] = p.detector_rot_lower[j];
|
||||
blocks[1].upper[j] = p.detector_rot_upper[j];
|
||||
}
|
||||
|
||||
blocks[2].offset = OFF_AXIS;
|
||||
blocks[2].size = 3;
|
||||
blocks[2].constant = p.beam_and_orientation_only || p.rot_vec_constant;
|
||||
blocks[2].sphere = true;
|
||||
|
||||
blocks[3].offset = OFF_P0;
|
||||
blocks[3].size = 3;
|
||||
|
||||
blocks[4].offset = OFF_LEN;
|
||||
blocks[4].size = 3;
|
||||
blocks[4].constant = p.beam_and_orientation_only || p.latt_vec1_constant;
|
||||
|
||||
blocks[5].offset = OFF_ANG;
|
||||
blocks[5].size = 3;
|
||||
blocks[5].constant = p.beam_and_orientation_only || p.latt_vec2_constant;
|
||||
|
||||
for (int j = 0; j < 3; j++) {
|
||||
blocks[4].lower[j] = p.latt_vec1_lower[j];
|
||||
blocks[4].upper[j] = p.latt_vec1_upper[j];
|
||||
blocks[5].lower[j] = p.latt_vec2_lower[j];
|
||||
blocks[5].upper[j] = p.latt_vec2_upper[j];
|
||||
}
|
||||
return blocks;
|
||||
}
|
||||
|
||||
// Where each lane lands among the tangent coordinates of the free blocks; -1 for a held one.
|
||||
template<size_t L>
|
||||
std::array<int, L> LaneToTangent(const std::vector<LMBlock> &blocks, const std::array<int, L> &lane_block,
|
||||
const std::array<int, L> &lane_index) {
|
||||
std::array<int, L> map{};
|
||||
std::array<int, 6> tangent_offset{};
|
||||
int t = 0;
|
||||
for (size_t b = 0; b < blocks.size(); b++) {
|
||||
tangent_offset[b] = t;
|
||||
t += blocks[b].TangentSize();
|
||||
}
|
||||
for (size_t l = 0; l < L; l++)
|
||||
map[l] = blocks[lane_block[l]].constant ? -1 : tangent_offset[lane_block[l]] + lane_index[l];
|
||||
return map;
|
||||
}
|
||||
|
||||
template<int L>
|
||||
void ToTangent(const Sums<L> &s, const std::array<int, L> &map, Eigen::VectorXd &g, Eigen::MatrixXd &H) {
|
||||
for (int i = 0; i < L; i++) {
|
||||
if (map[i] < 0)
|
||||
continue;
|
||||
g[map[i]] += s.g[i];
|
||||
for (int j = i; j < L; j++) {
|
||||
if (map[j] < 0)
|
||||
continue;
|
||||
H(map[i], map[j]) += s.H[i][j];
|
||||
if (map[i] != map[j])
|
||||
H(map[j], map[i]) += s.H[i][j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void AddPriors(const XtalRefineProblem &p, const double *x, double &cost, Eigen::VectorXd *g,
|
||||
Eigen::MatrixXd *H, int beam_tangent, int rot_tangent) {
|
||||
for (const auto &prior: p.priors) {
|
||||
const bool on_beam = prior.block == XtalRefinePrior::Block::Beam;
|
||||
const double *v = x + (on_beam ? OFF_BEAM : OFF_ROT);
|
||||
const double r = prior.weight * (prior.gx * v[0] + prior.gy * v[1] - prior.p0);
|
||||
cost += 0.5 * r * r;
|
||||
const int t = on_beam ? beam_tangent : rot_tangent;
|
||||
if (!g || t < 0)
|
||||
continue;
|
||||
const double J[2] = {prior.weight * prior.gx, prior.weight * prior.gy};
|
||||
for (int i = 0; i < 2; i++) {
|
||||
(*g)[t + i] += J[i] * r;
|
||||
for (int j = 0; j < 2; j++)
|
||||
(*H)(t + i, t + j) += J[i] * J[j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<class D>
|
||||
D Seed(double value, int lane, bool free) {
|
||||
return free ? D::Variable(value, lane) : D(value);
|
||||
}
|
||||
|
||||
// Residual blocks of at least this many residuals; the cut depends on the count alone.
|
||||
constexpr int MIN_RESIDUALS_PER_BLOCK = 256;
|
||||
|
||||
template<int L, class Fn>
|
||||
Sums<L> SumResiduals(const XtalRefineProblem &p, int num_threads, Fn &&residual) {
|
||||
const int n = static_cast<int>(p.residuals.size());
|
||||
std::vector<Sums<L>> partial(ReductionBlocks(n, MIN_RESIDUALS_PER_BLOCK));
|
||||
ParallelBlocks(n, std::max(1, num_threads), [&](int b, int lo, int hi) {
|
||||
for (int i = lo; i < hi; i++)
|
||||
residual(i, partial[b]);
|
||||
}, MIN_RESIDUALS_PER_BLOCK);
|
||||
Sums<L> total;
|
||||
for (const auto &s: partial)
|
||||
total.Add(s);
|
||||
return total;
|
||||
}
|
||||
|
||||
double WeightSq(const XtalRefineProblem &p, int i) {
|
||||
return p.weight_sq.empty() ? 1.0 : p.weight_sq[i];
|
||||
}
|
||||
|
||||
bool AllFinite(double cost, const Eigen::VectorXd *g, const Eigen::MatrixXd *H) {
|
||||
return std::isfinite(cost) && (!g || (g->allFinite() && H->allFinite()));
|
||||
}
|
||||
|
||||
// Seven-block residual (XtalResidualFixedDistance), with every block that depends on parameters
|
||||
// alone - detector-angle sines and cosines, the spindle back-rotation of each frame, the reciprocal
|
||||
// basis of the cell, the orientation's rotation - worked out once per evaluation.
|
||||
bool EvaluateGeneral(const XtalRefineProblem &p, const std::array<int, LANES> &map, int num_threads,
|
||||
const double *x, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) {
|
||||
const bool beam_free = map[0] >= 0, rot_free = map[2] >= 0, axis_free = map[4] >= 0;
|
||||
const bool p0_free = map[OBS] >= 0, len_free = map[OBS + 3] >= 0, ang_free = map[OBS + 6] >= 0;
|
||||
|
||||
const DO beam[2] = {Seed<DO>(x[OFF_BEAM], 0, beam_free),
|
||||
Seed<DO>(x[OFF_BEAM + 1], 1, beam_free)};
|
||||
const DO rot1 = Seed<DO>(x[OFF_ROT], 2, rot_free);
|
||||
const DO rot2 = Seed<DO>(x[OFF_ROT + 1], 3, rot_free);
|
||||
const DO c1 = cos(rot1), s1 = sin(rot1), c2 = cos(rot2), s2 = sin(rot2);
|
||||
|
||||
DO axis[3] = {x[OFF_AXIS], x[OFF_AXIS + 1], x[OFF_AXIS + 2]};
|
||||
if (axis_free) {
|
||||
double J[3][2];
|
||||
SpherePlusJacobian3(x + OFF_AXIS, J);
|
||||
for (int k = 0; k < 3; k++) {
|
||||
axis[k].v[4] = J[k][0];
|
||||
axis[k].v[5] = J[k][1];
|
||||
}
|
||||
}
|
||||
std::vector<AngleAxisRotator<DO>> rot_back;
|
||||
rot_back.reserve(p.frame_angle_rad.size());
|
||||
for (const double angle: p.frame_angle_rad) {
|
||||
const DO aa_back[3] = {angle * axis[0], angle * axis[1], angle * axis[2]};
|
||||
rot_back.emplace_back(aa_back);
|
||||
}
|
||||
|
||||
DP p0[3], len[3], ang[3];
|
||||
for (int k = 0; k < 3; k++) {
|
||||
p0[k] = Seed<DP>(x[OFF_P0 + k], k, p0_free);
|
||||
len[k] = Seed<DP>(x[OFF_LEN + k], 3 + k, len_free);
|
||||
ang[k] = Seed<DP>(x[OFF_ANG + k], 6 + k, ang_free);
|
||||
}
|
||||
Eigen::Matrix<DP, 3, 1> bxc, cxa, axb;
|
||||
DP invV;
|
||||
XtalResidual::ReciprocalBasis(len, ang, p.crystal_system, bxc, cxa, axb, invV);
|
||||
const AngleAxisRotator<DP> rot_p0(p0);
|
||||
|
||||
const Sums<LANES> s = SumResiduals<LANES>(p, num_threads, [&](int i, Sums<LANES> &acc) {
|
||||
const XtalResidual &res = p.residuals[i];
|
||||
DO obs[3];
|
||||
res.ObservedRecipCore(beam, p.distance_mm, c1, s1, c2, s2, rot_back[p.frame[i]], obs);
|
||||
DP unrot[3], pred[3];
|
||||
res.CombineRecipUnrot(bxc, cxa, axb, invV, unrot);
|
||||
rot_p0.Rotate(unrot, pred);
|
||||
const double w2 = WeightSq(p, i);
|
||||
for (int k = 0; k < 3; k++) {
|
||||
double J[LANES];
|
||||
for (int l = 0; l < OBS; l++)
|
||||
J[l] = obs[k].v[l];
|
||||
for (int l = 0; l < PRED; l++)
|
||||
J[OBS + l] = -pred[k].v[l];
|
||||
acc.Add(obs[k].a - pred[k].a, J, w2);
|
||||
}
|
||||
});
|
||||
|
||||
cost = s.cost;
|
||||
if (g)
|
||||
ToTangent<LANES>(s, map, *g, *H);
|
||||
AddPriors(p, x, cost, g, H, map[0], map[2]);
|
||||
return AllFinite(cost, g, H);
|
||||
}
|
||||
|
||||
// The reduced residual (XtalResidualBeamOrientation): detector, spindle and cell held, so the
|
||||
// back-rotation of each frame and the unrotated prediction of each residual are constants.
|
||||
bool EvaluateBeamOrientation(const XtalRefineProblem &p, const std::array<int, R_LANES> &map,
|
||||
const std::vector<AngleAxisRotator<double>> &rot_back,
|
||||
const std::vector<std::array<double, 3>> &unrot, double c1, double s1,
|
||||
double c2, double s2, int num_threads,
|
||||
const double *x, double &cost, Eigen::VectorXd *g, Eigen::MatrixXd *H) {
|
||||
using DB = Dual<R_OBS>;
|
||||
using DR = Dual<R_PRED>;
|
||||
const bool beam_free = map[0] >= 0;
|
||||
const DB beam[2] = {beam_free ? DB::Variable(x[OFF_BEAM], 0) : DB(x[OFF_BEAM]),
|
||||
beam_free ? DB::Variable(x[OFF_BEAM + 1], 1) : DB(x[OFF_BEAM + 1])};
|
||||
const DR p0[3] = {DR::Variable(x[OFF_P0], 0), DR::Variable(x[OFF_P0 + 1], 1),
|
||||
DR::Variable(x[OFF_P0 + 2], 2)};
|
||||
const AngleAxisRotator<DR> rot_p0(p0);
|
||||
|
||||
const Sums<R_LANES> s = SumResiduals<R_LANES>(p, num_threads, [&](int i, Sums<R_LANES> &acc) {
|
||||
const XtalResidual &res = p.residuals[i];
|
||||
DB obs[3];
|
||||
res.ObservedRecipCore(beam, p.distance_mm, c1, s1, c2, s2, rot_back[p.frame[i]], obs);
|
||||
DR pred[3];
|
||||
rot_p0.Rotate(unrot[i].data(), pred);
|
||||
const double w2 = WeightSq(p, i);
|
||||
for (int k = 0; k < 3; k++) {
|
||||
double J[R_LANES];
|
||||
for (int l = 0; l < R_OBS; l++)
|
||||
J[l] = obs[k].v[l];
|
||||
for (int l = 0; l < R_PRED; l++)
|
||||
J[R_OBS + l] = -pred[k].v[l];
|
||||
acc.Add(obs[k].a - pred[k].a, J, w2);
|
||||
}
|
||||
});
|
||||
|
||||
cost = s.cost;
|
||||
if (g)
|
||||
ToTangent<R_LANES>(s, map, *g, *H);
|
||||
AddPriors(p, x, cost, g, H, map[0], -1);
|
||||
return AllFinite(cost, g, H);
|
||||
}
|
||||
}
|
||||
|
||||
LMSummary SolveXtalRefine(XtalRefineProblem &p, int num_threads) {
|
||||
const std::vector<LMBlock> blocks = MakeBlocks(p);
|
||||
|
||||
std::vector<double> x(N_AMBIENT);
|
||||
const auto put = [&](int off, const double *v, int n) { for (int i = 0; i < n; i++) x[off + i] = v[i]; };
|
||||
put(OFF_BEAM, p.beam, 2);
|
||||
put(OFF_ROT, p.detector_rot, 2);
|
||||
put(OFF_AXIS, p.rot_vec, 3);
|
||||
put(OFF_P0, p.latt_vec0, 3);
|
||||
put(OFF_LEN, p.latt_vec1, 3);
|
||||
put(OFF_ANG, p.latt_vec2, 3);
|
||||
|
||||
LMSummary summary;
|
||||
if (p.beam_and_orientation_only) {
|
||||
const std::array<int, R_LANES> lane_block = {0, 0, 3, 3, 3};
|
||||
const std::array<int, R_LANES> lane_index = {0, 1, 0, 1, 2};
|
||||
const auto map = LaneToTangent<R_LANES>(blocks, lane_block, lane_index);
|
||||
|
||||
std::vector<AngleAxisRotator<double>> rot_back;
|
||||
rot_back.reserve(p.frame_angle_rad.size());
|
||||
for (const double angle: p.frame_angle_rad)
|
||||
rot_back.push_back(XtalFrameConstants::BackRotator(angle, p.rot_vec));
|
||||
Eigen::Matrix<double, 3, 1> bxc, cxa, axb;
|
||||
double invV;
|
||||
XtalResidual::ReciprocalBasis(p.latt_vec1, p.latt_vec2, p.crystal_system, bxc, cxa, axb, invV);
|
||||
std::vector<std::array<double, 3>> unrot(p.residuals.size());
|
||||
for (size_t i = 0; i < p.residuals.size(); i++)
|
||||
p.residuals[i].CombineRecipUnrot(bxc, cxa, axb, invV, unrot[i].data());
|
||||
const double c1 = std::cos(p.detector_rot[0]), s1 = std::sin(p.detector_rot[0]);
|
||||
const double c2 = std::cos(p.detector_rot[1]), s2 = std::sin(p.detector_rot[1]);
|
||||
|
||||
summary = SolveLM(x, blocks, p.options, [&](const double *at, double &cost, Eigen::VectorXd *g,
|
||||
Eigen::MatrixXd *H) {
|
||||
return EvaluateBeamOrientation(p, map, rot_back, unrot, c1, s1, c2, s2, num_threads, at, cost, g, H);
|
||||
});
|
||||
} else {
|
||||
const std::array<int, LANES> lane_block = {0, 0, 1, 1, 2, 2, 3, 3, 3, 4, 4, 4, 5, 5, 5};
|
||||
const std::array<int, LANES> lane_index = {0, 1, 0, 1, 0, 1, 0, 1, 2, 0, 1, 2, 0, 1, 2};
|
||||
const auto map = LaneToTangent<LANES>(blocks, lane_block, lane_index);
|
||||
summary = SolveLM(x, blocks, p.options, [&](const double *at, double &cost, Eigen::VectorXd *g,
|
||||
Eigen::MatrixXd *H) {
|
||||
return EvaluateGeneral(p, map, num_threads, at, cost, g, H);
|
||||
});
|
||||
}
|
||||
|
||||
if (summary.IsSolutionUsable()) {
|
||||
const auto get = [&](int off, double *v, int n) { for (int i = 0; i < n; i++) v[i] = x[off + i]; };
|
||||
get(OFF_BEAM, p.beam, 2);
|
||||
get(OFF_ROT, p.detector_rot, 2);
|
||||
get(OFF_AXIS, p.rot_vec, 3);
|
||||
get(OFF_P0, p.latt_vec0, 3);
|
||||
get(OFF_LEN, p.latt_vec1, 3);
|
||||
get(OFF_ANG, p.latt_vec2, 3);
|
||||
}
|
||||
return summary;
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <limits>
|
||||
#include <vector>
|
||||
|
||||
#include "XtalResidual.h"
|
||||
#include "LMSolver.h"
|
||||
|
||||
// A soft restraint w * (g . p - p0) on one direction g of a two-component block - the beam centre or
|
||||
// the detector tilt, which are the same gauge seen twice and so take the same direction (see
|
||||
// XtalOptimizer).
|
||||
struct XtalRefinePrior {
|
||||
enum class Block { Beam, DetectorRot } block = Block::Beam;
|
||||
double gx = 0.0, gy = 0.0, p0 = 0.0, weight = 0.0;
|
||||
};
|
||||
|
||||
// The least-squares problem XtalOptimizer solves, as data: the residuals with the frame each belongs
|
||||
// to, the parameter blocks with what is held and what is bounded, and the priors. The parameter arrays
|
||||
// are in/out. Parameter blocks: beam(2), detector_rot(2), rot_vec(3, the spindle, refined on the sphere),
|
||||
// latt_vec0(3, orientation angle-axis), latt_vec1(3, cell lengths), latt_vec2(3, cell angles) - see
|
||||
// XtalResidual. The distance is a constant of the problem.
|
||||
struct XtalRefineProblem {
|
||||
static constexpr double kNoBound = std::numeric_limits<double>::max();
|
||||
|
||||
gemmi::CrystalSystem crystal_system = gemmi::CrystalSystem::Triclinic;
|
||||
// Only beam and orientation free, everything else held: the reduced residual (see
|
||||
// XtalResidualBeamOrientation), whose crystal half is a constant of the problem.
|
||||
bool beam_and_orientation_only = false;
|
||||
double distance_mm = 0.0;
|
||||
|
||||
std::vector<XtalResidual> residuals;
|
||||
std::vector<int> frame; // per residual, an index into frame_angle_rad
|
||||
std::vector<double> frame_angle_rad;
|
||||
std::vector<double> weight_sq; // per residual; empty = unweighted
|
||||
|
||||
double beam[2] = {0, 0};
|
||||
double detector_rot[2] = {0, 0};
|
||||
double rot_vec[3] = {1, 0, 0};
|
||||
double latt_vec0[3] = {0, 0, 0};
|
||||
double latt_vec1[3] = {0, 0, 0};
|
||||
double latt_vec2[3] = {0, 0, 0};
|
||||
|
||||
bool beam_constant = false;
|
||||
bool detector_rot_constant = true;
|
||||
bool rot_vec_constant = true;
|
||||
bool latt_vec1_constant = true;
|
||||
bool latt_vec2_constant = true;
|
||||
|
||||
double detector_rot_lower[2] = {-kNoBound, -kNoBound}, detector_rot_upper[2] = {kNoBound, kNoBound};
|
||||
double latt_vec1_lower[3] = {-kNoBound, -kNoBound, -kNoBound}, latt_vec1_upper[3] = {kNoBound, kNoBound, kNoBound};
|
||||
double latt_vec2_lower[3] = {-kNoBound, -kNoBound, -kNoBound}, latt_vec2_upper[3] = {kNoBound, kNoBound, kNoBound};
|
||||
|
||||
std::vector<XtalRefinePrior> priors;
|
||||
|
||||
LMOptions options;
|
||||
};
|
||||
|
||||
// Solves the problem in place. The residual sums are cut into blocks that depend on the problem
|
||||
// alone, so the answer is the same at any thread count.
|
||||
LMSummary SolveXtalRefine(XtalRefineProblem &problem, int num_threads);
|
||||
@@ -7,8 +7,6 @@
|
||||
|
||||
#include <Eigen/Dense>
|
||||
|
||||
#include "ceres/ceres.h"
|
||||
#include "ceres/rotation.h"
|
||||
#include "gemmi/symmetry.hpp"
|
||||
|
||||
#include "../../common/JFJochException.h"
|
||||
@@ -110,6 +108,9 @@ inline void EffectiveCellFromParams(gemmi::CrystalSystem symmetry, const double
|
||||
}
|
||||
}
|
||||
|
||||
// The scalar type is a template parameter throughout: a double, a ceres::Jet or a Dual (Dual.h). The
|
||||
// mathematical functions are called unqualified, so each type's own overload is found by lookup.
|
||||
//
|
||||
// Detector -> reciprocal geometry residual, shared by the per-image XtalOptimizer (one lattice, one
|
||||
// frame) and the offline GeometryRefiner (shared beam/distance/cell blocks, one orientation block per
|
||||
// frame). Parameter blocks: beam(2), distance_mm(1), detector_rot(2 = rot1,rot2), rotation_axis(3),
|
||||
@@ -164,16 +165,18 @@ struct XtalResidual {
|
||||
// detector_rot[0] = rot1, detector_rot[1] = rot2 are refined; rot3 is fixed
|
||||
// (e.g. from a PONI import) and baked in here as a constant so that a non-zero
|
||||
// rot3 is not silently dropped during refinement.
|
||||
using std::cos;
|
||||
using std::sin;
|
||||
const C rot1 = detector_rot[0];
|
||||
const C rot2 = detector_rot[1];
|
||||
|
||||
// Ry(+rot1): rotation around Y-axis
|
||||
const C c1 = ceres::cos(rot1);
|
||||
const C s1 = ceres::sin(rot1);
|
||||
const C c1 = cos(rot1);
|
||||
const C s1 = sin(rot1);
|
||||
|
||||
// Rx(-rot2): rotation around X-axis with inverted sign (PyFAI left-handed)
|
||||
const C c2 = ceres::cos(rot2);
|
||||
const C s2 = ceres::sin(rot2);
|
||||
const C c2 = cos(rot2);
|
||||
const C s2 = sin(rot2);
|
||||
|
||||
// Apply the goniometer "back-to-start" rotation of this frame's angle.
|
||||
const C aa_back[3] = {
|
||||
@@ -227,7 +230,8 @@ struct XtalResidual {
|
||||
const T z = t2_z;
|
||||
|
||||
// convert to recip space
|
||||
const T lab_norm = ceres::sqrt(x * x + y * y + z * z);
|
||||
using std::sqrt;
|
||||
const T lab_norm = sqrt(x * x + y * y + z * z);
|
||||
const T inv_norm = T(1) / lab_norm;
|
||||
|
||||
T recip_raw[3];
|
||||
@@ -258,6 +262,9 @@ struct XtalResidual {
|
||||
static void ReciprocalBasis(const C *const p1, const C *const p2, gemmi::CrystalSystem symmetry,
|
||||
Eigen::Matrix<C, 3, 1> &bxc, Eigen::Matrix<C, 3, 1> &cxa,
|
||||
Eigen::Matrix<C, 3, 1> &axb, C &invV) {
|
||||
using std::cos;
|
||||
using std::sin;
|
||||
using std::sqrt;
|
||||
// Build unit cell lengths and B (convention: columns are a, b, c prior to global rotation)
|
||||
Eigen::Matrix<C, 3, 1> e_uc_len = Eigen::Matrix<C, 3, 1>::Zero();
|
||||
Eigen::Matrix<C, 3, 3> B = Eigen::Matrix<C, 3, 3>::Identity();
|
||||
@@ -275,14 +282,14 @@ struct XtalResidual {
|
||||
} else if (symmetry == gemmi::CrystalSystem::Monoclinic) {
|
||||
// Unique axis b: alpha = gamma = 90°, beta free (angle between a and c)
|
||||
e_uc_len << p1[0], p1[1], p1[2];
|
||||
B(0, 2) = ceres::cos(p2[0]);
|
||||
B(2, 2) = ceres::sin(p2[0]);
|
||||
B(0, 2) = cos(p2[0]);
|
||||
B(2, 2) = sin(p2[0]);
|
||||
} else {
|
||||
// Triclinic: p1 = (a,b,c), p2 = (alpha, beta, gamma) in radians
|
||||
const C ca = ceres::cos(p2[0]);
|
||||
const C cb = ceres::cos(p2[1]);
|
||||
const C cg = ceres::cos(p2[2]);
|
||||
const C sg = ceres::sin(p2[2]);
|
||||
const C ca = cos(p2[0]);
|
||||
const C cb = cos(p2[1]);
|
||||
const C cg = cos(p2[2]);
|
||||
const C sg = sin(p2[2]);
|
||||
|
||||
e_uc_len << p1[0], p1[1], p1[2];
|
||||
|
||||
@@ -297,7 +304,7 @@ struct XtalResidual {
|
||||
const C cx = cb;
|
||||
const C cy = (ca - cb * cg) / sg;
|
||||
const C v = C(1) - cx * cx - cy * cy;
|
||||
const C cz = (v >= C(0)) ? ceres::sqrt(v) : C(0);
|
||||
const C cz = (v >= C(0)) ? sqrt(v) : C(0);
|
||||
|
||||
B(0, 2) = cx;
|
||||
B(1, 2) = cy;
|
||||
@@ -399,12 +406,12 @@ struct XtalResidual {
|
||||
// reciprocal basis of the fixed cell. They are constants of the whole problem there - one frame per
|
||||
// image, one cell - and deriving them inside each residual costs six trigonometric calls, a hypot and a
|
||||
// division per evaluation for numbers that never change. Built by the caller, which has to keep it alive
|
||||
// as long as the ceres::Problem that points at it.
|
||||
// as long as the residuals that point at it.
|
||||
struct XtalFrameConstants {
|
||||
XtalFrameConstants(const double *detector_rot, const double *rotation_axis, double angle_rad,
|
||||
const double *uc_len, const double *uc_angle, gemmi::CrystalSystem symmetry)
|
||||
: c1(ceres::cos(detector_rot[0])), s1(ceres::sin(detector_rot[0])),
|
||||
c2(ceres::cos(detector_rot[1])), s2(ceres::sin(detector_rot[1])),
|
||||
: c1(cos(detector_rot[0])), s1(sin(detector_rot[0])),
|
||||
c2(cos(detector_rot[1])), s2(sin(detector_rot[1])),
|
||||
rot_back(BackRotator(angle_rad, rotation_axis)) {
|
||||
XtalResidual::ReciprocalBasis(uc_len, uc_angle, symmetry, bxc, cxa, axb, invV);
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
|
||||
#include "../../common/ParallelFor.h"
|
||||
#include "RotationScaleMerge.h"
|
||||
#include "RotationScaleMergeGPU.h" // SurfaceTerm, the surface fit's term on both paths
|
||||
|
||||
#include <algorithm>
|
||||
#include <atomic>
|
||||
@@ -3334,7 +3335,7 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
|
||||
// traffic of the data it touched, over hundreds of megabytes. Copying once turns them into
|
||||
// sequential walks of a compact array. The copy keeps fulls order, so each sum below is formed
|
||||
// from exactly the same terms in exactly the same order.
|
||||
struct Term { float I, sigma, corr, d; int32_t cell, group; };
|
||||
using Term = RotationScaleMergeGPU::SurfaceTerm; // float I, sigma, corr, d; int32_t cell, group
|
||||
std::vector<Term> term;
|
||||
std::vector<uint8_t> term_parity; // frame parity, read only by the group-ordered copy below
|
||||
// Which terms each pass walks, as positions in `term`. The cross-validated halves then cost half a
|
||||
@@ -3437,22 +3438,27 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
|
||||
std::vector<double> shw_shell(nshell, 0.0), shw_cell(nshell > 0 ? ncell : 0, 0.0);
|
||||
std::vector<int32_t> group_shell(n_groups, 0); // the shell each ASU group sits in, for the gate
|
||||
if (nshell > 0) {
|
||||
std::vector<float> s2;
|
||||
s2.reserve(term.size());
|
||||
for (const Term &t : term)
|
||||
s2.push_back(t.d > 0.0f ? 1.0f / (t.d * t.d) : 0.0f);
|
||||
const int n_term = static_cast<int>(term.size());
|
||||
std::vector<float> s2(n_term);
|
||||
ParallelChunks(n_term, nt, [&](int lo, int hi) {
|
||||
for (int k = lo; k < hi; ++k)
|
||||
s2[k] = term[k].d > 0.0f ? 1.0f / (term[k].d * term[k].d) : 0.0f;
|
||||
});
|
||||
// The edges are order statistics of s2, so they are the same values however the sort orders
|
||||
// equal elements among themselves.
|
||||
std::vector<float> sorted = s2;
|
||||
ParallelSort(sorted.begin(), sorted.end(), nt, std::less<float>());
|
||||
std::vector<float> edge(nshell - 1);
|
||||
size_t prev = 0;
|
||||
for (int i = 1; i < nshell; ++i) {
|
||||
const size_t pos = s2.size() * static_cast<size_t>(i) / static_cast<size_t>(nshell);
|
||||
std::nth_element(s2.begin() + prev, s2.begin() + pos, s2.end());
|
||||
edge[i - 1] = s2[pos];
|
||||
prev = pos;
|
||||
}
|
||||
for (int i = 1; i < nshell; ++i)
|
||||
edge[i - 1] = sorted[sorted.size() * static_cast<size_t>(i) / static_cast<size_t>(nshell)];
|
||||
std::vector<uint8_t> term_shell(n_term);
|
||||
ParallelChunks(n_term, nt, [&](int lo, int hi) {
|
||||
for (int k = lo; k < hi; ++k)
|
||||
term_shell[k] = static_cast<uint8_t>(std::upper_bound(edge.begin(), edge.end(), s2[k]) - edge.begin());
|
||||
});
|
||||
for (size_t k = 0; k < term.size(); ++k) {
|
||||
const Term &t = term[k];
|
||||
const float v = t.d > 0.0f ? 1.0f / (t.d * t.d) : 0.0f;
|
||||
const int s = static_cast<int>(std::upper_bound(edge.begin(), edge.end(), v) - edge.begin());
|
||||
const int s = term_shell[k];
|
||||
group_shell[t.group] = s;
|
||||
const double sc = static_cast<double>(t.sigma) * t.corr;
|
||||
if (!(sc > 0.0)) continue;
|
||||
@@ -3475,8 +3481,41 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
|
||||
std::vector<int32_t> g_start(n_groups + 1, 0);
|
||||
for (const Term &t : term) ++g_start[t.group + 1];
|
||||
for (int g = 0; g < n_groups; ++g) g_start[g + 1] += g_start[g];
|
||||
std::vector<RefTerm> gterm(term.size());
|
||||
{
|
||||
std::vector<RefTerm> gterm;
|
||||
// With a GPU the two passes of every round run there (RotationScaleMergeGPU::Surface*), on the same
|
||||
// terms in the same order: the device gets the terms with the group order as a permutation, and each
|
||||
// subset cut into this fit's reduction blocks with every block's terms ordered by cell, so that one
|
||||
// device thread forms what one block of the host loop sums into one cell.
|
||||
bool on_gpu = false;
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
on_gpu = gpu_active_;
|
||||
if (on_gpu) {
|
||||
std::vector<int32_t> gperm(term.size());
|
||||
std::vector<int32_t> fill(g_start.begin(), g_start.end() - 1);
|
||||
for (size_t k = 0; k < term.size(); ++k)
|
||||
gperm[fill[term[k].group]++] = static_cast<int32_t>(k);
|
||||
gpu_->SurfaceSetTerms(static_cast<int>(term.size()), term.data(), term_parity.data(),
|
||||
n_groups, gperm.data(), g_start.data(), ncell);
|
||||
for (int parity : {0, 1, -1}) {
|
||||
const std::vector<int32_t> &sel = subset(parity);
|
||||
const int n = static_cast<int>(sel.size());
|
||||
const int nb = ReductionBlocks(n, SURFACE_BLOCK);
|
||||
std::vector<int32_t> perm(n), seg_start(static_cast<size_t>(nb) * ncell + 1, n);
|
||||
ParallelFor(nb, nt, [&](int b) {
|
||||
const int lo = static_cast<int>(static_cast<int64_t>(n) * b / nb);
|
||||
const int hi = static_cast<int>(static_cast<int64_t>(n) * (b + 1) / nb);
|
||||
std::vector<int32_t> pos(ncell + 1, 0);
|
||||
for (int k = lo; k < hi; ++k) ++pos[term[sel[k]].cell + 1];
|
||||
for (int c = 0; c < ncell; ++c) pos[c + 1] += pos[c];
|
||||
for (int c = 0; c < ncell; ++c) seg_start[static_cast<size_t>(b) * ncell + c] = lo + pos[c];
|
||||
for (int k = lo; k < hi; ++k) perm[lo + pos[term[sel[k]].cell]++] = sel[k];
|
||||
});
|
||||
gpu_->SurfaceSetSubset(parity < 0 ? 2 : parity, nb, perm.data(), seg_start.data());
|
||||
}
|
||||
}
|
||||
#endif
|
||||
if (!on_gpu) {
|
||||
gterm.resize(term.size());
|
||||
std::vector<int32_t> fill(g_start.begin(), g_start.end() - 1);
|
||||
for (size_t k = 0; k < term.size(); ++k) {
|
||||
const Term &t = term[k];
|
||||
@@ -3488,6 +3527,12 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
|
||||
// and the score.
|
||||
std::vector<double> sw(n_groups), swI(n_groups);
|
||||
auto reference = [&](int parity, const std::vector<double> &A) {
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (on_gpu) {
|
||||
gpu_->SurfaceReference(parity, A.data());
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
ParallelChunks(n_groups, nt, [&](int glo, int ghi) {
|
||||
for (int g = glo; g < ghi; ++g) {
|
||||
double s_w = 0.0, s_wI = 0.0;
|
||||
@@ -3520,7 +3565,7 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
|
||||
const std::vector<int32_t> &sel = subset(parity);
|
||||
std::vector<double> A(ncell, 1.0);
|
||||
// Per-block cell accumulators, allocated once for the whole fit rather than per round.
|
||||
const int nb = ReductionBlocks(static_cast<int>(sel.size()), SURFACE_BLOCK);
|
||||
const int nb = on_gpu ? 0 : ReductionBlocks(static_cast<int>(sel.size()), SURFACE_BLOCK);
|
||||
std::vector<std::vector<double>> tcross(nb, std::vector<double>(ncell)), tref2(nb, std::vector<double>(ncell));
|
||||
settled = false;
|
||||
n_clamped = 0;
|
||||
@@ -3555,22 +3600,28 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
|
||||
// there is no ordering that keeps threads off each other's bins, and there are only ncell
|
||||
// of them, so per-block copies are cheap and the fixed blocks keep it the same at any -N.
|
||||
std::vector<double> cross(ncell, 0.0), ref2(ncell, 0.0);
|
||||
ParallelBlocks(static_cast<int>(sel.size()), nt, [&](int b, int lo, int hi) {
|
||||
std::vector<double> &xcross = tcross[b], &xref2 = tref2[b];
|
||||
std::fill(xcross.begin(), xcross.end(), 0.0);
|
||||
std::fill(xref2.begin(), xref2.end(), 0.0);
|
||||
for (int k = lo; k < hi; ++k) {
|
||||
const Term &o = term[sel[k]];
|
||||
if (sw[o.group] <= 0.0) continue;
|
||||
const double Iref = swI[o.group] / sw[o.group], a = A[o.cell];
|
||||
const double Is = static_cast<double>(o.I) * o.corr * a, sc = static_cast<double>(o.sigma) * o.corr * a;
|
||||
if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue;
|
||||
const double w = 1.0 / (sc * sc);
|
||||
xcross[o.cell] += w * Is * Iref; xref2[o.cell] += w * Iref * Iref;
|
||||
}
|
||||
}, SURFACE_BLOCK);
|
||||
for (int b = 0; b < nb; ++b)
|
||||
for (int c = 0; c < ncell; ++c) { cross[c] += tcross[b][c]; ref2[c] += tref2[b][c]; }
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (on_gpu)
|
||||
gpu_->SurfaceFitSums(parity < 0 ? 2 : parity, cross.data(), ref2.data());
|
||||
#endif
|
||||
if (!on_gpu) {
|
||||
ParallelBlocks(static_cast<int>(sel.size()), nt, [&](int b, int lo, int hi) {
|
||||
std::vector<double> &xcross = tcross[b], &xref2 = tref2[b];
|
||||
std::fill(xcross.begin(), xcross.end(), 0.0);
|
||||
std::fill(xref2.begin(), xref2.end(), 0.0);
|
||||
for (int k = lo; k < hi; ++k) {
|
||||
const Term &o = term[sel[k]];
|
||||
if (sw[o.group] <= 0.0) continue;
|
||||
const double Iref = swI[o.group] / sw[o.group], a = A[o.cell];
|
||||
const double Is = static_cast<double>(o.I) * o.corr * a, sc = static_cast<double>(o.sigma) * o.corr * a;
|
||||
if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue;
|
||||
const double w = 1.0 / (sc * sc);
|
||||
xcross[o.cell] += w * Is * Iref; xref2[o.cell] += w * Iref * Iref;
|
||||
}
|
||||
}, SURFACE_BLOCK);
|
||||
for (int b = 0; b < nb; ++b)
|
||||
for (int c = 0; c < ncell; ++c) { cross[c] += tcross[b][c]; ref2[c] += tref2[b][c]; }
|
||||
}
|
||||
std::vector<double> dsorted = cross;
|
||||
std::nth_element(dsorted.begin(), dsorted.begin() + dsorted.size() / 2, dsorted.end());
|
||||
const double lambda = 0.1 * std::max(1e-30, dsorted[dsorted.size() / 2]);
|
||||
@@ -3672,20 +3723,33 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
|
||||
const int nsh_cc = nshell > 0 ? nshell : 1;
|
||||
auto half_means = [&](int parity, const std::vector<double> &A) {
|
||||
reference(parity, A);
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (on_gpu)
|
||||
gpu_->SurfaceGetReference(sw.data(), swI.data());
|
||||
#endif
|
||||
std::vector<double> m(n_groups, std::numeric_limits<double>::quiet_NaN());
|
||||
for (int g = 0; g < n_groups; ++g)
|
||||
if (sw[g] > 0.0) m[g] = swI[g] / sw[g];
|
||||
return m;
|
||||
};
|
||||
auto shell_cc = [&](const std::vector<double> &x, const std::vector<double> &y, int s) {
|
||||
double n = 0, sx = 0, sy = 0, sxx = 0, syy = 0, sxy = 0;
|
||||
// The correlation within every shell, in one walk over the groups: each shell still sums its own
|
||||
// groups in group order.
|
||||
auto shell_cc = [&](const std::vector<double> &x, const std::vector<double> &y) {
|
||||
struct Sums { double n = 0, sx = 0, sy = 0, sxx = 0, syy = 0, sxy = 0; };
|
||||
std::vector<Sums> sum(nsh_cc);
|
||||
for (int g = 0; g < n_groups; ++g) {
|
||||
if (group_shell[g] != s || !std::isfinite(x[g]) || !std::isfinite(y[g])) continue;
|
||||
n += 1; sx += x[g]; sy += y[g]; sxx += x[g] * x[g]; syy += y[g] * y[g]; sxy += x[g] * y[g];
|
||||
if (!std::isfinite(x[g]) || !std::isfinite(y[g])) continue;
|
||||
Sums &u = sum[group_shell[g]];
|
||||
u.n += 1; u.sx += x[g]; u.sy += y[g]; u.sxx += x[g] * x[g]; u.syy += y[g] * y[g]; u.sxy += x[g] * y[g];
|
||||
}
|
||||
const double vx = sxx - sx * sx / n, vy = syy - sy * sy / n;
|
||||
return (n >= 50 && vx > 0.0 && vy > 0.0) ? (sxy - sx * sy / n) / std::sqrt(vx * vy)
|
||||
: std::numeric_limits<double>::quiet_NaN();
|
||||
std::vector<double> cc(nsh_cc, std::numeric_limits<double>::quiet_NaN());
|
||||
for (int s = 0; s < nsh_cc; ++s) {
|
||||
const Sums &u = sum[s];
|
||||
const double vx = u.sxx - u.sx * u.sx / u.n, vy = u.syy - u.sy * u.sy / u.n;
|
||||
if (u.n >= 50 && vx > 0.0 && vy > 0.0)
|
||||
cc[s] = (u.sxy - u.sx * u.sy / u.n) / std::sqrt(vx * vy);
|
||||
}
|
||||
return cc;
|
||||
};
|
||||
const std::vector<double> ident(ncell, 1.0);
|
||||
const std::vector<double> A_even = fit_surface(0), A_odd = fit_surface(1);
|
||||
@@ -3702,8 +3766,9 @@ void RotationScaleMerge::ApplyCellSurface(const std::vector<int32_t> &cell, int
|
||||
// Following Fisher (1915) Biometrika 10, 507-521
|
||||
double gain = 0.0;
|
||||
int n_cc = 0;
|
||||
const std::vector<double> cc0 = shell_cc(odd0, even0), cc1 = shell_cc(odd1, even1);
|
||||
for (int sh = 0; sh < nsh_cc; ++sh) {
|
||||
const double c0 = shell_cc(odd0, even0, sh), c1 = shell_cc(odd1, even1, sh);
|
||||
const double c0 = cc0[sh], c1 = cc1[sh];
|
||||
if (std::isfinite(c0) && std::isfinite(c1)) { gain += std::atanh(c1) - std::atanh(c0); ++n_cc; }
|
||||
}
|
||||
if (n_cc > 0) gain /= n_cc;
|
||||
@@ -5699,7 +5764,8 @@ RotationScaleMerge::Result RotationScaleMerge::Run(bool for_search, bool full_st
|
||||
ReducePartialGroupMeans(n_groups, partial_mean);
|
||||
ComputePerFrameCC(partial_mean, cc, cc_n);
|
||||
}
|
||||
FinalizePerFrameScale(cc, cc_n, partial_scaled);
|
||||
if (write_back_per_frame_scale)
|
||||
FinalizePerFrameScale(cc, cc_n, partial_scaled);
|
||||
|
||||
// The filters below remove observations by zeroing corr, which is what takes an observation out of
|
||||
// the 3D combine, the merge and the error model alike (excluding them from the ASU grouping is NOT
|
||||
|
||||
@@ -119,6 +119,11 @@ public:
|
||||
// compares nothing, asks for it to be left out.
|
||||
Result Run(bool for_search, bool full_stats, bool measure_cc_before_corrections);
|
||||
|
||||
// Whether Run() writes the per-frame G / CC / mosaicity back onto the outcomes (on by default). Off
|
||||
// for a merge that is not the run's answer - the P1 cross-check - so the per-image table and the
|
||||
// unmerged MTZ describe the merge that was written.
|
||||
void SetWriteBackPerFrameScale(bool on) { write_back_per_frame_scale = on; }
|
||||
|
||||
// Override the high-resolution cut for the next Run() - used to gate the de-novo P1 search pass at
|
||||
// <I/sigma> >= 1 without cutting the final in-symmetry merge. Reset to the manual limit afterwards.
|
||||
void SetDMinLimit(std::optional<double> d_min_A) { d_min_limit = d_min_A; }
|
||||
@@ -413,6 +418,7 @@ private:
|
||||
std::unique_ptr<RotationScaleMergeGPU> gpu_;
|
||||
bool gpu_active_ = false;
|
||||
#endif
|
||||
bool write_back_per_frame_scale = true; // see SetWriteBackPerFrameScale
|
||||
|
||||
// --- helpers (each a flat pass; see the .cpp) ---
|
||||
// Turn the per-frame mean background under the reflections (accumulated by the ingest fill loop) into
|
||||
|
||||
@@ -18,13 +18,15 @@ namespace {
|
||||
constexpr int BLK = 256;
|
||||
constexpr int MIN_REFLECTIONS = 20;
|
||||
|
||||
// Every kernel and copy here is queued on the legacy NULL stream, so - as in BeamCenterFFTGPU -
|
||||
// no buffer comes from the pool. A pooled buffer is freed with cudaFreeAsync on the thread's
|
||||
// non-blocking allocation stream, which is not ordered after the NULL stream. Each entry point
|
||||
// below waits for its own work before it returns, so no free here has yet overtaken a read, but
|
||||
// that holds only by that convention, and not at all for a free while unwinding from a failed
|
||||
// call; nor can compute-sanitizer --track-stream-ordered-races see it, and it reported every
|
||||
// reassigned merge buffer as a use-after-free. cudaFree synchronises the device first.
|
||||
// Every kernel and copy here is queued on the instance's own stream (Impl::stream), not on the
|
||||
// legacy NULL stream: the merge and the image analysis of a probe pass beside it would otherwise
|
||||
// wait for each other's work at every launch and every synchronisation. As in BeamCenterFFTGPU no buffer comes from the pool. A
|
||||
// pooled buffer is freed with cudaFreeAsync on the thread's allocation stream, which is not
|
||||
// ordered after this one. Each entry point below waits for its own work before it returns, so no
|
||||
// free here has yet overtaken a read, but that holds only by that convention, and not at all for a
|
||||
// free while unwinding from a failed call; nor can compute-sanitizer --track-stream-ordered-races
|
||||
// see it, and it reported every reassigned merge buffer as a use-after-free. cudaFree
|
||||
// synchronises the device first.
|
||||
constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous;
|
||||
|
||||
__device__ __forceinline__ double SafeInvD(double x, double fallback) {
|
||||
@@ -636,6 +638,110 @@ namespace {
|
||||
}
|
||||
}
|
||||
|
||||
// --- correction-surface fit (ApplyCellSurface) ---
|
||||
// The host fit is the reference here: these kernels form the same sums from the same terms in the
|
||||
// same order, and every rounding is spelled out (__dmul_rn / __dadd_rn round each step on its own,
|
||||
// fma rounds once) to be the one the host build makes - nvcc would otherwise contract a multiply
|
||||
// and an add wherever it sees them, and GCC at -march=x86-64-v3 does so only in some of them. The
|
||||
// products are the host's (I * corr) * a and (sigma * corr) * a.
|
||||
using SurfaceTerm = RotationScaleMergeGPU::SurfaceTerm;
|
||||
|
||||
__device__ __forceinline__ double SurfaceIs(const SurfaceTerm &t, double a) {
|
||||
return __dmul_rn(__dmul_rn(double(t.I), double(t.corr)), a);
|
||||
}
|
||||
|
||||
__device__ __forceinline__ double SurfaceSigma(const SurfaceTerm &t, double a) {
|
||||
return __dmul_rn(__dmul_rn(double(t.sigma), double(t.corr)), a);
|
||||
}
|
||||
|
||||
// One thread per ASU group: the group's inverse-variance sums over its terms of frame parity `parity`
|
||||
// (< 0 = all), in fulls order - the host's `reference`. The host builds that loop twice, and only
|
||||
// the parity-filtered copy fuses the multiply-add of swI; the unfiltered one rounds the product.
|
||||
__global__ void SurfaceReferenceKernel(int n_groups, int parity, const int32_t *__restrict__ gperm,
|
||||
const int32_t *__restrict__ gstart,
|
||||
const SurfaceTerm *__restrict__ term,
|
||||
const uint8_t *__restrict__ term_parity,
|
||||
const double *__restrict__ A,
|
||||
double *__restrict__ sw, double *__restrict__ swI) {
|
||||
for (int g = blockIdx.x * blockDim.x + threadIdx.x; g < n_groups; g += gridDim.x * blockDim.x) {
|
||||
double s_w = 0.0, s_wI = 0.0;
|
||||
for (int k = gstart[g]; k < gstart[g + 1]; ++k) {
|
||||
const int i = gperm[k];
|
||||
if (parity >= 0 && term_parity[i] != parity) continue;
|
||||
const SurfaceTerm t = term[i];
|
||||
const double a = A[t.cell];
|
||||
const double Is = SurfaceIs(t, a), sc = SurfaceSigma(t, a);
|
||||
const double w = 1.0 / __dmul_rn(sc, sc);
|
||||
s_w = __dadd_rn(s_w, w);
|
||||
s_wI = parity >= 0 ? fma(Is, w, s_wI) : __dadd_rn(s_wI, __dmul_rn(Is, w));
|
||||
}
|
||||
sw[g] = s_w; swI[g] = s_wI;
|
||||
}
|
||||
}
|
||||
|
||||
// The fit's sums of one round. Each term's contribution is formed on its own thread (one per term,
|
||||
// in segment order), and only the two multiply-adds that sum them are left to the thread of each
|
||||
// (reduction block, cell) - the per-block accumulators of the host fit, one slot at a time, walking
|
||||
// its segment in term order. A term the host skips is marked by a NaN Iref (a kept one is finite).
|
||||
__global__ void SurfaceFitTermKernel(int n, const int32_t *__restrict__ perm,
|
||||
const SurfaceTerm *__restrict__ term,
|
||||
const double *__restrict__ A,
|
||||
const double *__restrict__ sw, const double *__restrict__ swI,
|
||||
double *__restrict__ w_Is, double *__restrict__ w_Iref,
|
||||
double *__restrict__ Iref_out) {
|
||||
for (int k = blockIdx.x * blockDim.x + threadIdx.x; k < n; k += gridDim.x * blockDim.x) {
|
||||
const SurfaceTerm t = term[perm[k]];
|
||||
Iref_out[k] = NAN;
|
||||
if (sw[t.group] <= 0.0) continue;
|
||||
const double Iref = swI[t.group] / sw[t.group], a = A[t.cell];
|
||||
const double Is = SurfaceIs(t, a), sc = SurfaceSigma(t, a);
|
||||
if (!isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue;
|
||||
const double w = 1.0 / __dmul_rn(sc, sc);
|
||||
w_Is[k] = __dmul_rn(w, Is);
|
||||
w_Iref[k] = __dmul_rn(w, Iref);
|
||||
Iref_out[k] = Iref;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void SurfaceFitSegmentKernel(int n_seg, const int32_t *__restrict__ seg_start,
|
||||
const double *__restrict__ w_Is, const double *__restrict__ w_Iref,
|
||||
const double *__restrict__ Iref,
|
||||
double *__restrict__ tcross, double *__restrict__ tref2) {
|
||||
for (int s = blockIdx.x * blockDim.x + threadIdx.x; s < n_seg; s += gridDim.x * blockDim.x) {
|
||||
double xcross = 0.0, xref2 = 0.0;
|
||||
for (int k = seg_start[s]; k < seg_start[s + 1]; ++k) {
|
||||
if (isnan(Iref[k])) continue;
|
||||
xcross = fma(w_Is[k], Iref[k], xcross);
|
||||
xref2 = fma(w_Iref[k], Iref[k], xref2);
|
||||
}
|
||||
tcross[s] = xcross; tref2[s] = xref2;
|
||||
}
|
||||
}
|
||||
|
||||
// One thread per cell: the blocks' sums added up in block order, as the host adds its slots.
|
||||
__global__ void SurfaceFitCellKernel(int n_blocks, int ncell, const double *__restrict__ tcross,
|
||||
const double *__restrict__ tref2,
|
||||
double *__restrict__ cross, double *__restrict__ ref2) {
|
||||
for (int c = blockIdx.x * blockDim.x + threadIdx.x; c < ncell; c += gridDim.x * blockDim.x) {
|
||||
double sc = 0.0, sr = 0.0;
|
||||
for (int b = 0; b < n_blocks; ++b) {
|
||||
sc = __dadd_rn(sc, tcross[b * ncell + c]);
|
||||
sr = __dadd_rn(sr, tref2[b * ncell + c]);
|
||||
}
|
||||
cross[c] = sc; ref2[c] = sr;
|
||||
}
|
||||
}
|
||||
|
||||
void CudaCheck(cudaError_t e, const char *what);
|
||||
|
||||
// A copy on the instance's stream, waited for - what cudaMemcpy on the NULL stream was, without also
|
||||
// waiting for every other stream on the card.
|
||||
void CopyAndWait(void *dst, const void *src, size_t bytes, cudaMemcpyKind kind, cudaStream_t s,
|
||||
const char *what) {
|
||||
CudaCheck(cudaMemcpyAsync(dst, src, bytes, kind, s), what);
|
||||
CudaCheck(cudaStreamSynchronize(s), what);
|
||||
}
|
||||
|
||||
void CudaCheck(cudaError_t e, const char *what) {
|
||||
if (e != cudaSuccess)
|
||||
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
|
||||
@@ -703,6 +809,9 @@ namespace {
|
||||
}
|
||||
|
||||
struct RotationScaleMergeGPU::Impl {
|
||||
// First, so it goes last: the buffers below are freed before the stream their work ran on.
|
||||
std::unique_ptr<CudaStream> stream;
|
||||
cudaStream_t s() const { return stream->get(); }
|
||||
int device = 0; // the GPU this instance's buffers live on
|
||||
bool available = false;
|
||||
int n_obs = 0, n_frames = 0, n_groups = 0;
|
||||
@@ -736,7 +845,7 @@ struct RotationScaleMergeGPU::Impl {
|
||||
void Upload(CudaDevicePtr<T> &dst, const T *src, int n) const {
|
||||
dst = Alloc<T>(std::max(1, n));
|
||||
if (n > 0)
|
||||
CudaCheck(cudaMemcpy(dst.get(), src, size_t(n) * sizeof(T), cudaMemcpyHostToDevice), "upload");
|
||||
CopyAndWait(dst.get(), src, size_t(n) * sizeof(T), cudaMemcpyHostToDevice, s(), "upload");
|
||||
}
|
||||
|
||||
// immutable per-obs
|
||||
@@ -801,6 +910,18 @@ struct RotationScaleMergeGPU::Impl {
|
||||
CudaDevicePtr<uint8_t> f_sco_ok;
|
||||
CudaDevicePtr<int32_t> f_frame_perm, f_frame_start, f_frame_count;
|
||||
CudaDevicePtr<int32_t> f_gperm, f_gstart, f_gcount;
|
||||
// correction-surface fit (one ApplyCellSurface call at a time): its terms and their group CSR, the
|
||||
// three subsets' per-(block, cell) segments, the surface and the sums of the round
|
||||
int s_ncell = 0, s_n_groups = 0;
|
||||
int s_n_blocks[3] = {0, 0, 0};
|
||||
int s_n_sel[3] = {0, 0, 0};
|
||||
size_t s_slots = 0; // capacity of s_tcross / s_tref2
|
||||
CudaDevicePtr<RotationScaleMergeGPU::SurfaceTerm> s_term;
|
||||
CudaDevicePtr<uint8_t> s_parity;
|
||||
CudaDevicePtr<int32_t> s_gperm, s_gstart;
|
||||
CudaDevicePtr<int32_t> s_perm[3], s_seg_start[3];
|
||||
CudaDevicePtr<double> s_A, s_sw, s_swI, s_tcross, s_tref2, s_cross, s_ref2;
|
||||
CudaDevicePtr<double> s_w_Is, s_w_Iref, s_Iref; // per term of a subset, in segment order
|
||||
};
|
||||
|
||||
// Set the device this instance's memory lives on for the duration of a call, and put the caller's
|
||||
@@ -833,6 +954,7 @@ RotationScaleMergeGPU::RotationScaleMergeGPU() : impl_(std::make_unique<Impl>())
|
||||
// can put the caller's device back instead of leaving the thread moved.
|
||||
impl_->device = 0;
|
||||
DeviceGuard guard(impl_->device, true);
|
||||
impl_->stream = std::make_unique<CudaStream>();
|
||||
impl_->available = true;
|
||||
}
|
||||
}
|
||||
@@ -878,10 +1000,10 @@ void RotationScaleMergeGPU::SetPartialsLayout(int n_obs, int n_frames,
|
||||
|
||||
namespace {
|
||||
template <typename T>
|
||||
void UploadChunk(CudaDevicePtr<T> &dst, int offset, int count, const T *v) {
|
||||
void UploadChunk(CudaDevicePtr<T> &dst, int offset, int count, const T *v, cudaStream_t s) {
|
||||
if (count > 0)
|
||||
CudaCheck(cudaMemcpy(dst.get() + offset, v, size_t(count) * sizeof(T),
|
||||
cudaMemcpyHostToDevice), "upload chunk");
|
||||
CopyAndWait(dst.get() + offset, v, size_t(count) * sizeof(T), cudaMemcpyHostToDevice, s,
|
||||
"upload chunk");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -889,34 +1011,34 @@ void RotationScaleMergeGPU::SetObsField(ObsField f, int offset, int count, const
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
switch (f) {
|
||||
case ObsField::I: UploadChunk(d.I, offset, count, v); break;
|
||||
case ObsField::Sigma: UploadChunk(d.sigma, offset, count, v); break;
|
||||
case ObsField::PrescalingCorr: UploadChunk(d.prescaling_corr, offset, count, v); break;
|
||||
case ObsField::Partiality: UploadChunk(d.partiality, offset, count, v); break;
|
||||
case ObsField::Zeta: UploadChunk(d.zeta, offset, count, v); break;
|
||||
case ObsField::Corr0: UploadChunk(d.corr, offset, count, v); break;
|
||||
case ObsField::Bkg: UploadChunk(d.bkg, offset, count, v); break;
|
||||
case ObsField::VarBkg: UploadChunk(d.var_bkg, offset, count, v); break;
|
||||
case ObsField::ImageNumber: UploadChunk(d.image_number, offset, count, v); break;
|
||||
case ObsField::D: UploadChunk(d.d_obs, offset, count, v); break;
|
||||
case ObsField::Px: UploadChunk(d.px_obs, offset, count, v); break;
|
||||
case ObsField::Py: UploadChunk(d.py_obs, offset, count, v); break;
|
||||
case ObsField::I: UploadChunk(d.I, offset, count, v, impl_->s()); break;
|
||||
case ObsField::Sigma: UploadChunk(d.sigma, offset, count, v, impl_->s()); break;
|
||||
case ObsField::PrescalingCorr: UploadChunk(d.prescaling_corr, offset, count, v, impl_->s()); break;
|
||||
case ObsField::Partiality: UploadChunk(d.partiality, offset, count, v, impl_->s()); break;
|
||||
case ObsField::Zeta: UploadChunk(d.zeta, offset, count, v, impl_->s()); break;
|
||||
case ObsField::Corr0: UploadChunk(d.corr, offset, count, v, impl_->s()); break;
|
||||
case ObsField::Bkg: UploadChunk(d.bkg, offset, count, v, impl_->s()); break;
|
||||
case ObsField::VarBkg: UploadChunk(d.var_bkg, offset, count, v, impl_->s()); break;
|
||||
case ObsField::ImageNumber: UploadChunk(d.image_number, offset, count, v, impl_->s()); break;
|
||||
case ObsField::D: UploadChunk(d.d_obs, offset, count, v, impl_->s()); break;
|
||||
case ObsField::Px: UploadChunk(d.px_obs, offset, count, v, impl_->s()); break;
|
||||
case ObsField::Py: UploadChunk(d.py_obs, offset, count, v, impl_->s()); break;
|
||||
}
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SetObsFrame(int offset, int count, const int32_t *frame) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
UploadChunk(impl_->frame, offset, count, frame);
|
||||
UploadChunk(impl_->frame, offset, count, frame, impl_->s());
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SetObsOnIce(int offset, int count, const uint8_t *on_ice) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
UploadChunk(impl_->on_ice, offset, count, on_ice);
|
||||
UploadChunk(impl_->on_ice, offset, count, on_ice, impl_->s());
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SetObsClipped(int offset, int count, const uint8_t *clipped) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
UploadChunk(impl_->clipped, offset, count, clipped);
|
||||
UploadChunk(impl_->clipped, offset, count, clipped, impl_->s());
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SetGroups(int n_groups, const int32_t *group, const int32_t *group_perm,
|
||||
@@ -934,8 +1056,8 @@ void RotationScaleMergeGPU::SetGroups(int n_groups, const int32_t *group, const
|
||||
|
||||
void RotationScaleMergeGPU::SetCorr(const float *corr) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
CudaCheck(cudaMemcpy(impl_->corr.get(), corr, size_t(impl_->n_obs) * sizeof(float),
|
||||
cudaMemcpyHostToDevice), "upload corr");
|
||||
CopyAndWait(impl_->corr.get(), corr, size_t(impl_->n_obs) * sizeof(float),
|
||||
cudaMemcpyHostToDevice, impl_->s(), "upload corr");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::ScalePartials(int iters, double min_partiality, bool /*has_d_min*/) {
|
||||
@@ -943,42 +1065,42 @@ void RotationScaleMergeGPU::ScalePartials(int iters, double min_partiality, bool
|
||||
auto &d = *impl_;
|
||||
// Reset per call: the host keeps the G of a frame across calls (RunScalingLoop), so a frame this
|
||||
// call did not fit must read as unfitted, not as fitted with the value of the call before.
|
||||
CudaCheck(cudaMemset(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t)), "memset scaled");
|
||||
CudaCheck(cudaMemset(d.g.get(), 0, size_t(d.n_frames) * sizeof(double)), "memset g"); // unscaled g unused
|
||||
CudaCheck(cudaMemsetAsync(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t), impl_->s()), "memset scaled");
|
||||
CudaCheck(cudaMemsetAsync(d.g.get(), 0, size_t(d.n_frames) * sizeof(double), impl_->s()), "memset g"); // unscaled g unused
|
||||
const int obs_blocks = (d.n_obs + BLK - 1) / BLK;
|
||||
const int upd_blocks = std::min(65535, obs_blocks);
|
||||
const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK);
|
||||
for (int it = 0; it < iters; ++it) {
|
||||
ReduceGroupMeansKernel<<<grp_blocks, BLK>>>(d.n_groups, min_partiality,
|
||||
ReduceGroupMeansKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(d.n_groups, min_partiality,
|
||||
d.group_perm.get(), d.group_start.get(), d.group_count.get(),
|
||||
d.I.get(), d.sigma.get(), d.partiality.get(), d.corr.get(), d.group_mean.get());
|
||||
CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch");
|
||||
PrepScaleObsKernel<<<obs_blocks, BLK>>>(d.n_obs, min_partiality, d.group.get(), d.partiality.get(), d.prescaling_corr.get(),
|
||||
PrepScaleObsKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.n_obs, min_partiality, d.group.get(), d.partiality.get(), d.prescaling_corr.get(),
|
||||
d.zeta.get(), d.on_ice.get(), d.group_mean.get(), d.sigma.get(), d.inv_sigma.get(),
|
||||
d.sco_coeff.get(), d.sco_ok.get());
|
||||
CudaCheck(cudaGetLastError(), "PrepScaleObsKernel launch");
|
||||
FitPerFrameGKernel<<<d.n_frames, BLK>>>(d.n_frames, d.frame_start.get(), d.frame_count.get(),
|
||||
FitPerFrameGKernel<<<d.n_frames, BLK, 0, impl_->s()>>>(d.n_frames, d.frame_start.get(), d.frame_count.get(),
|
||||
d.I.get(), d.inv_sigma.get(), d.sco_coeff.get(), d.sco_ok.get(), nullptr, d.g.get(), d.scaled.get());
|
||||
CudaCheck(cudaGetLastError(), "FitPerFrameGKernel launch");
|
||||
UpdateCorrKernel<<<upd_blocks, BLK>>>(d.n_obs, d.frame.get(), d.prescaling_corr.get(), d.partiality.get(),
|
||||
UpdateCorrKernel<<<upd_blocks, BLK, 0, impl_->s()>>>(d.n_obs, d.frame.get(), d.prescaling_corr.get(), d.partiality.get(),
|
||||
d.g.get(), d.scaled.get(), d.corr.get());
|
||||
}
|
||||
CudaCheck(cudaGetLastError(), "kernel launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "scale sync");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "scale sync");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::GetCorr(float *corr_out) const {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
CudaCheck(cudaMemcpy(corr_out, impl_->corr.get(), size_t(impl_->n_obs) * sizeof(float),
|
||||
cudaMemcpyDeviceToHost), "download corr");
|
||||
CopyAndWait(corr_out, impl_->corr.get(), size_t(impl_->n_obs) * sizeof(float),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "download corr");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::GetG(double *g_out, uint8_t *scaled_out) const {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
CudaCheck(cudaMemcpy(g_out, impl_->g.get(), size_t(impl_->n_frames) * sizeof(double),
|
||||
cudaMemcpyDeviceToHost), "download g");
|
||||
CudaCheck(cudaMemcpy(scaled_out, impl_->scaled.get(), size_t(impl_->n_frames) * sizeof(uint8_t),
|
||||
cudaMemcpyDeviceToHost), "download scaled");
|
||||
CopyAndWait(g_out, impl_->g.get(), size_t(impl_->n_frames) * sizeof(double),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "download g");
|
||||
CopyAndWait(scaled_out, impl_->scaled.get(), size_t(impl_->n_frames) * sizeof(uint8_t),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "download scaled");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SetFrameCellOk(const uint8_t *frame_cell_ok) {
|
||||
@@ -1030,19 +1152,19 @@ void RotationScaleMergeGPU::MergeEmSamples(bool for_search, double min_partialit
|
||||
|
||||
const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK);
|
||||
const int obs_blocks = std::min(65535, (nf + BLK - 1) / BLK);
|
||||
MergeEmStatsKernel<<<grp_blocks, BLK>>>(p);
|
||||
MergeSamplesKernel<<<obs_blocks, BLK>>>(nf, p);
|
||||
MergeEmStatsKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(p);
|
||||
MergeSamplesKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(nf, p);
|
||||
CudaCheck(cudaGetLastError(), "merge em/samples launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "merge em/samples sync");
|
||||
CudaCheck(cudaMemcpy(em_mean_out, d.m_em_mean.get(), size_t(ng) * sizeof(double),
|
||||
cudaMemcpyDeviceToHost), "dl em_mean");
|
||||
CudaCheck(cudaMemcpy(cnt_out, d.m_cnt.get(), size_t(ng) * sizeof(int32_t),
|
||||
cudaMemcpyDeviceToHost), "dl cnt");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "merge em/samples sync");
|
||||
CopyAndWait(em_mean_out, d.m_em_mean.get(), size_t(ng) * sizeof(double),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "dl em_mean");
|
||||
CopyAndWait(cnt_out, d.m_cnt.get(), size_t(ng) * sizeof(int32_t),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "dl cnt");
|
||||
if (nf > 0) {
|
||||
CudaCheck(cudaMemcpy(s2_out, d.m_s2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl s2");
|
||||
CudaCheck(cudaMemcpy(I2_out, d.m_I2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl I2");
|
||||
CudaCheck(cudaMemcpy(dev2_out, d.m_dev2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost), "dl dev2");
|
||||
CudaCheck(cudaMemcpy(valid_out, d.m_valid.get(), size_t(nf) * sizeof(uint8_t), cudaMemcpyDeviceToHost), "dl valid");
|
||||
CopyAndWait(s2_out, d.m_s2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl s2");
|
||||
CopyAndWait(I2_out, d.m_I2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl I2");
|
||||
CopyAndWait(dev2_out, d.m_dev2.get(), size_t(nf) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl dev2");
|
||||
CopyAndWait(valid_out, d.m_valid.get(), size_t(nf) * sizeof(uint8_t), cudaMemcpyDeviceToHost, impl_->s(), "dl valid");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1091,10 +1213,10 @@ void RotationScaleMergeGPU::MergeAccum(double error_model_a, double error_model_
|
||||
p.rejected_obs = d.m_rejected.get();
|
||||
|
||||
const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK);
|
||||
MergeAccumKernel<<<grp_blocks, BLK>>>(p);
|
||||
MergeAccumKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(p);
|
||||
CudaCheck(cudaGetLastError(), "merge accum launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "merge accum sync");
|
||||
CudaCheck(cudaMemcpy(rejected_obs, d.m_rejected.get(), size_t(d.n_fulls) * sizeof(uint8_t), cudaMemcpyDeviceToHost),
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "merge accum sync");
|
||||
CopyAndWait(rejected_obs, d.m_rejected.get(), size_t(d.n_fulls) * sizeof(uint8_t), cudaMemcpyDeviceToHost, impl_->s(),
|
||||
"dl rejected_obs");
|
||||
}
|
||||
|
||||
@@ -1105,7 +1227,7 @@ void RotationScaleMergeGPU::MergeAccumRange(int g0, int n, double *swI, double *
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
auto dl = [&](void *h, const auto &s) {
|
||||
CudaCheck(cudaMemcpy(h, s.get() + g0, size_t(n) * sizeof(*s.get()), cudaMemcpyDeviceToHost),
|
||||
CopyAndWait(h, s.get() + g0, size_t(n) * sizeof(*s.get()), cudaMemcpyDeviceToHost, impl_->s(),
|
||||
"dl accum"); };
|
||||
dl(swI, d.a_swI); dl(sw, d.a_sw); dl(swIh0, d.a_swIh0); dl(swIh1, d.a_swIh1);
|
||||
dl(swh0, d.a_swh0); dl(swh1, d.a_swh1); dl(swh_typ0, d.a_swht0); dl(swh_typ1, d.a_swht1);
|
||||
@@ -1139,17 +1261,17 @@ void RotationScaleMergeGPU::MergeRmeas(const double *merged_I, double *absdev, d
|
||||
p.rejected_obs = d.m_rejected.get();
|
||||
|
||||
const int grp_blocks = std::min(65535, (ng + BLK - 1) / BLK);
|
||||
MergeRmeasKernel<<<grp_blocks, BLK>>>(p);
|
||||
MergeRmeasKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(p);
|
||||
CudaCheck(cudaGetLastError(), "merge rmeas launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "merge rmeas sync");
|
||||
CudaCheck(cudaMemcpy(absdev, d.r_absdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl absdev");
|
||||
CudaCheck(cudaMemcpy(sumI, d.r_sumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumI");
|
||||
CudaCheck(cudaMemcpy(wabsdev, d.r_wabsdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl wabsdev");
|
||||
CudaCheck(cudaMemcpy(wsumI, d.r_wsumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl wsumI");
|
||||
CudaCheck(cudaMemcpy(sumv, d.r_sumv.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumv");
|
||||
CudaCheck(cudaMemcpy(sumv2, d.r_sumv2.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost), "dl sumv2");
|
||||
CudaCheck(cudaMemcpy(n, d.r_n.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost), "dl rn");
|
||||
CudaCheck(cudaMemcpy(nusable, d.r_nusable.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost), "dl rnusable");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "merge rmeas sync");
|
||||
CopyAndWait(absdev, d.r_absdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl absdev");
|
||||
CopyAndWait(sumI, d.r_sumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumI");
|
||||
CopyAndWait(wabsdev, d.r_wabsdev.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl wabsdev");
|
||||
CopyAndWait(wsumI, d.r_wsumI.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl wsumI");
|
||||
CopyAndWait(sumv, d.r_sumv.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumv");
|
||||
CopyAndWait(sumv2, d.r_sumv2.get(), size_t(ng) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(), "dl sumv2");
|
||||
CopyAndWait(n, d.r_n.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost, impl_->s(), "dl rn");
|
||||
CopyAndWait(nusable, d.r_nusable.get(), size_t(ng) * sizeof(int32_t), cudaMemcpyDeviceToHost, impl_->s(), "dl rnusable");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SmoothCorr(const uint8_t *apply, const double *ratio) {
|
||||
@@ -1158,10 +1280,10 @@ void RotationScaleMergeGPU::SmoothCorr(const uint8_t *apply, const double *ratio
|
||||
d.Upload(d.smooth_apply, apply, d.n_frames);
|
||||
d.Upload(d.smooth_ratio, ratio, d.n_frames);
|
||||
const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK);
|
||||
SmoothCorrKernel<<<blocks, BLK>>>(d.n_obs, d.frame.get(), d.smooth_apply.get(),
|
||||
SmoothCorrKernel<<<blocks, BLK, 0, impl_->s()>>>(d.n_obs, d.frame.get(), d.smooth_apply.get(),
|
||||
d.smooth_ratio.get(), d.corr.get());
|
||||
CudaCheck(cudaGetLastError(), "smooth corr launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "smooth corr sync");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "smooth corr sync");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SmoothFullsCorr(const uint8_t *apply, const double *ratio) {
|
||||
@@ -1171,23 +1293,23 @@ void RotationScaleMergeGPU::SmoothFullsCorr(const uint8_t *apply, const double *
|
||||
d.Upload(d.smooth_apply, apply, d.n_frames);
|
||||
d.Upload(d.smooth_ratio, ratio, d.n_frames);
|
||||
const int blocks = std::min(65535, (d.n_fulls + BLK - 1) / BLK);
|
||||
SmoothCorrKernel<<<blocks, BLK>>>(d.n_fulls, d.f_frame.get(), d.smooth_apply.get(),
|
||||
SmoothCorrKernel<<<blocks, BLK, 0, impl_->s()>>>(d.n_fulls, d.f_frame.get(), d.smooth_apply.get(),
|
||||
d.smooth_ratio.get(), d.f_corr.get());
|
||||
CudaCheck(cudaGetLastError(), "smooth fulls corr launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "smooth fulls corr sync");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "smooth fulls corr sync");
|
||||
}
|
||||
|
||||
int64_t RotationScaleMergeGPU::FilterCorrByZeta(double min_zeta) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
CudaDevicePtr<unsigned long long> dropped = d.Alloc<unsigned long long>(1);
|
||||
CudaCheck(cudaMemset(dropped.get(), 0, sizeof(unsigned long long)), "zero zeta drop count");
|
||||
CudaCheck(cudaMemsetAsync(dropped.get(), 0, sizeof(unsigned long long), impl_->s()), "zero zeta drop count");
|
||||
const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK);
|
||||
FilterZetaKernel<<<blocks, BLK>>>(d.n_obs, min_zeta, d.zeta.get(), d.corr.get(), dropped.get());
|
||||
FilterZetaKernel<<<blocks, BLK, 0, impl_->s()>>>(d.n_obs, min_zeta, d.zeta.get(), d.corr.get(), dropped.get());
|
||||
CudaCheck(cudaGetLastError(), "zeta filter launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "zeta filter sync");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "zeta filter sync");
|
||||
unsigned long long n = 0;
|
||||
CudaCheck(cudaMemcpy(&n, dropped.get(), sizeof(unsigned long long), cudaMemcpyDeviceToHost),
|
||||
CopyAndWait(&n, dropped.get(), sizeof(unsigned long long), cudaMemcpyDeviceToHost, impl_->s(),
|
||||
"dl zeta drop count");
|
||||
return static_cast<int64_t>(n);
|
||||
}
|
||||
@@ -1197,9 +1319,9 @@ void RotationScaleMergeGPU::FilterCorrByFrame(const uint8_t *reject) {
|
||||
auto &d = *impl_;
|
||||
d.Upload(d.filter_reject, reject, d.n_frames);
|
||||
const int blocks = std::min(65535, (d.n_obs + BLK - 1) / BLK);
|
||||
FilterFrameKernel<<<blocks, BLK>>>(d.n_obs, d.frame.get(), d.filter_reject.get(), d.corr.get());
|
||||
FilterFrameKernel<<<blocks, BLK, 0, impl_->s()>>>(d.n_obs, d.frame.get(), d.filter_reject.get(), d.corr.get());
|
||||
CudaCheck(cudaGetLastError(), "frame filter launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "frame filter sync");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "frame filter sync");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::ComputePartialCC(double min_partiality, double *cc_out, int64_t *cc_n_out) {
|
||||
@@ -1208,19 +1330,19 @@ void RotationScaleMergeGPU::ComputePartialCC(double min_partiality, double *cc_o
|
||||
const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK);
|
||||
// Post-smooth group means (reuse the scaling reduce; reads the resident, smoothed corr), then the
|
||||
// per-frame CC over the resident partials. Only the tiny per-frame cc/cc_n come back to the host.
|
||||
ReduceGroupMeansKernel<<<grp_blocks, BLK>>>(d.n_groups, min_partiality,
|
||||
ReduceGroupMeansKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(d.n_groups, min_partiality,
|
||||
d.group_perm.get(), d.group_start.get(), d.group_count.get(),
|
||||
d.I.get(), d.sigma.get(), d.partiality.get(), d.corr.get(), d.group_mean.get());
|
||||
CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch");
|
||||
PerFrameCCKernel<<<d.n_frames, BLK>>>(d.n_frames, min_partiality,
|
||||
PerFrameCCKernel<<<d.n_frames, BLK, 0, impl_->s()>>>(d.n_frames, min_partiality,
|
||||
d.frame_start.get(), d.frame_count.get(), d.I.get(), d.sigma.get(), d.partiality.get(),
|
||||
d.corr.get(), d.on_ice.get(), d.group.get(), d.group_mean.get(), d.cc.get(), d.cc_n.get());
|
||||
CudaCheck(cudaGetLastError(), "partial CC launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "partial CC sync");
|
||||
CudaCheck(cudaMemcpy(cc_out, d.cc.get(), size_t(d.n_frames) * sizeof(double),
|
||||
cudaMemcpyDeviceToHost), "download cc");
|
||||
CudaCheck(cudaMemcpy(cc_n_out, d.cc_n.get(), size_t(d.n_frames) * sizeof(int64_t),
|
||||
cudaMemcpyDeviceToHost), "download cc_n");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "partial CC sync");
|
||||
CopyAndWait(cc_out, d.cc.get(), size_t(d.n_frames) * sizeof(double),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "download cc");
|
||||
CopyAndWait(cc_n_out, d.cc_n.get(), size_t(d.n_frames) * sizeof(int64_t),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "download cc_n");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SetRawRuns(int n_runs, int n_perm, const int32_t *perm,
|
||||
@@ -1247,8 +1369,8 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti
|
||||
float max_frame_gap) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
CudaCheck(cudaMemcpy(d.rr_group.get(), rawrun_group, size_t(d.n_runs) * sizeof(int32_t),
|
||||
cudaMemcpyHostToDevice), "upload rr_group");
|
||||
CopyAndWait(d.rr_group.get(), rawrun_group, size_t(d.n_runs) * sizeof(int32_t),
|
||||
cudaMemcpyHostToDevice, impl_->s(), "upload rr_group");
|
||||
|
||||
CombineParams p{};
|
||||
p.n_runs = d.n_runs;
|
||||
@@ -1267,13 +1389,13 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti
|
||||
const int blocks = std::min(65535, (d.n_runs + BLK - 1) / BLK);
|
||||
|
||||
// Count pass: how many fulls each run emits.
|
||||
CombineKernel<false><<<blocks, BLK>>>(p);
|
||||
CombineKernel<false><<<blocks, BLK, 0, impl_->s()>>>(p);
|
||||
CudaCheck(cudaGetLastError(), "combine count launch");
|
||||
|
||||
// Exclusive prefix sum on the host (deterministic) -> per-run output offset + total fulls.
|
||||
std::vector<int32_t> nevents(d.n_runs);
|
||||
CudaCheck(cudaMemcpy(nevents.data(), d.rr_nevents.get(), size_t(d.n_runs) * sizeof(int32_t),
|
||||
cudaMemcpyDeviceToHost), "download nevents");
|
||||
CopyAndWait(nevents.data(), d.rr_nevents.get(), size_t(d.n_runs) * sizeof(int32_t),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "download nevents");
|
||||
std::vector<int32_t> offset(d.n_runs);
|
||||
int64_t acc = 0;
|
||||
for (int r = 0; r < d.n_runs; ++r) { offset[r] = static_cast<int32_t>(acc); acc += nevents[r]; }
|
||||
@@ -1292,8 +1414,8 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti
|
||||
d.f_rlp = d.Alloc<float>(nf); d.f_zeta = d.Alloc<float>(nf);
|
||||
d.f_inv_sigma = d.Alloc<double>(nf);
|
||||
d.f_sco_coeff = d.Alloc<float>(nf); d.f_sco_ok = d.Alloc<uint8_t>(nf);
|
||||
CudaCheck(cudaMemcpy(d.rr_offset.get(), offset.data(), size_t(d.n_runs) * sizeof(int32_t),
|
||||
cudaMemcpyHostToDevice), "upload offset");
|
||||
CopyAndWait(d.rr_offset.get(), offset.data(), size_t(d.n_runs) * sizeof(int32_t),
|
||||
cudaMemcpyHostToDevice, impl_->s(), "upload offset");
|
||||
|
||||
p.rr_offset = d.rr_offset.get();
|
||||
p.f_h = d.f_h.get(); p.f_k = d.f_k.get(); p.f_l = d.f_l.get();
|
||||
@@ -1303,10 +1425,10 @@ int RotationScaleMergeGPU::Combine(const int32_t *rawrun_group, double min_parti
|
||||
p.f_var_bkg = d.f_var_bkg.get(); p.f_var_per_I = d.f_var_per_I.get();
|
||||
p.f_on_ice = d.f_on_ice.get(); p.f_clipped = d.f_clipped.get();
|
||||
if (d.n_fulls > 0) {
|
||||
CombineKernel<true><<<blocks, BLK>>>(p);
|
||||
CombineKernel<true><<<blocks, BLK, 0, impl_->s()>>>(p);
|
||||
CudaCheck(cudaGetLastError(), "combine emit launch");
|
||||
}
|
||||
CudaCheck(cudaDeviceSynchronize(), "combine sync");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "combine sync");
|
||||
return d.n_fulls;
|
||||
}
|
||||
|
||||
@@ -1318,7 +1440,7 @@ void RotationScaleMergeGPU::GetFulls(int32_t *h, int32_t *k, int32_t *l, float *
|
||||
const size_t n = static_cast<size_t>(dd.n_fulls);
|
||||
if (n == 0) return;
|
||||
auto dl = [&](void *dst, const void *src, size_t bytes) {
|
||||
CudaCheck(cudaMemcpy(dst, src, bytes, cudaMemcpyDeviceToHost), "download fulls");
|
||||
CopyAndWait(dst, src, bytes, cudaMemcpyDeviceToHost, impl_->s(), "download fulls");
|
||||
};
|
||||
dl(h, dd.f_h.get(), n * sizeof(int32_t)); dl(k, dd.f_k.get(), n * sizeof(int32_t));
|
||||
dl(l, dd.f_l.get(), n * sizeof(int32_t)); dl(frame, dd.f_frame.get(), n * sizeof(int32_t));
|
||||
@@ -1334,8 +1456,8 @@ void RotationScaleMergeGPU::GetFullsKeys(int32_t *frame, int32_t *group) const {
|
||||
const auto &d = *impl_;
|
||||
if (d.n_fulls == 0) return;
|
||||
const size_t bytes = size_t(d.n_fulls) * sizeof(int32_t);
|
||||
CudaCheck(cudaMemcpy(frame, d.f_frame.get(), bytes, cudaMemcpyDeviceToHost), "download f_frame");
|
||||
CudaCheck(cudaMemcpy(group, d.f_group.get(), bytes, cudaMemcpyDeviceToHost), "download f_group");
|
||||
CopyAndWait(frame, d.f_frame.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_frame");
|
||||
CopyAndWait(group, d.f_group.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_group");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SetFullsFrameCSR(const int32_t *frame_perm, int n_perm,
|
||||
@@ -1363,15 +1485,15 @@ void RotationScaleMergeGPU::ResetFullsScale() {
|
||||
if (nf == 0) return;
|
||||
const int obs_blocks = std::min(65535, (nf + BLK - 1) / BLK);
|
||||
// Unity model: partiality/prescaling_corr/zeta = 1 so coeff = mean; corr starts at 1.
|
||||
FillKernel<<<obs_blocks, BLK>>>(d.f_corr.get(), nf, 1.0f);
|
||||
FillKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.f_corr.get(), nf, 1.0f);
|
||||
CudaCheck(cudaGetLastError(), "FillKernel launch");
|
||||
FillKernel<<<obs_blocks, BLK>>>(d.f_partiality.get(), nf, 1.0f);
|
||||
FillKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.f_partiality.get(), nf, 1.0f);
|
||||
CudaCheck(cudaGetLastError(), "FillKernel launch");
|
||||
FillKernel<<<obs_blocks, BLK>>>(d.f_rlp.get(), nf, 1.0f);
|
||||
FillKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.f_rlp.get(), nf, 1.0f);
|
||||
CudaCheck(cudaGetLastError(), "FillKernel launch");
|
||||
FillKernel<<<obs_blocks, BLK>>>(d.f_zeta.get(), nf, 1.0f);
|
||||
FillKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(d.f_zeta.get(), nf, 1.0f);
|
||||
CudaCheck(cudaGetLastError(), "FillKernel launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "reset fulls scale sync");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "reset fulls scale sync");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::ScaleFulls(int iters, double min_partiality) {
|
||||
@@ -1383,38 +1505,38 @@ void RotationScaleMergeGPU::ScaleFulls(int iters, double min_partiality) {
|
||||
const int grp_blocks = std::min(65535, (d.n_groups + BLK - 1) / BLK);
|
||||
|
||||
// Reset per call, as ScalePartials: the host keeps the G of a frame across calls.
|
||||
CudaCheck(cudaMemset(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t)), "memset f scaled");
|
||||
CudaCheck(cudaMemset(d.g.get(), 0, size_t(d.n_frames) * sizeof(double)), "memset f g");
|
||||
CudaCheck(cudaMemsetAsync(d.scaled.get(), 0, size_t(d.n_frames) * sizeof(uint8_t), impl_->s()), "memset f scaled");
|
||||
CudaCheck(cudaMemsetAsync(d.g.get(), 0, size_t(d.n_frames) * sizeof(double), impl_->s()), "memset f g");
|
||||
|
||||
for (int it = 0; it < iters; ++it) {
|
||||
ReduceGroupMeansKernel<<<grp_blocks, BLK>>>(d.n_groups, min_partiality,
|
||||
ReduceGroupMeansKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(d.n_groups, min_partiality,
|
||||
d.f_gperm.get(), d.f_gstart.get(), d.f_gcount.get(),
|
||||
d.f_I.get(), d.f_sigma.get(), d.f_partiality.get(), d.f_corr.get(), d.group_mean.get());
|
||||
CudaCheck(cudaGetLastError(), "ReduceGroupMeansKernel launch");
|
||||
// Not grid-stride, so its grid has to cover every full - unlike the grid-stride kernels
|
||||
// below, which the 65535 cap is there for. Capped, it would silently leave the tail of
|
||||
// sco_coeff/sco_ok stale above 16.8M fulls.
|
||||
PrepScaleObsKernel<<<(nf + BLK - 1) / BLK, BLK>>>(nf, min_partiality, d.f_group.get(), d.f_partiality.get(),
|
||||
PrepScaleObsKernel<<<(nf + BLK - 1) / BLK, BLK, 0, impl_->s()>>>(nf, min_partiality, d.f_group.get(), d.f_partiality.get(),
|
||||
d.f_rlp.get(), d.f_zeta.get(), d.f_on_ice.get(), d.group_mean.get(),
|
||||
d.f_sigma.get(), d.f_inv_sigma.get(), d.f_sco_coeff.get(), d.f_sco_ok.get());
|
||||
CudaCheck(cudaGetLastError(), "PrepScaleObsKernel launch");
|
||||
FitPerFrameGKernel<<<d.n_frames, BLK>>>(d.n_frames,
|
||||
FitPerFrameGKernel<<<d.n_frames, BLK, 0, impl_->s()>>>(d.n_frames,
|
||||
d.f_frame_start.get(), d.f_frame_count.get(), d.f_I.get(), d.f_inv_sigma.get(),
|
||||
d.f_sco_coeff.get(), d.f_sco_ok.get(), d.f_frame_perm.get(), d.g.get(), d.scaled.get());
|
||||
CudaCheck(cudaGetLastError(), "FitPerFrameGKernel launch");
|
||||
UpdateCorrKernel<<<obs_blocks, BLK>>>(nf, d.f_frame.get(), d.f_rlp.get(), d.f_partiality.get(),
|
||||
UpdateCorrKernel<<<obs_blocks, BLK, 0, impl_->s()>>>(nf, d.f_frame.get(), d.f_rlp.get(), d.f_partiality.get(),
|
||||
d.g.get(), d.scaled.get(), d.f_corr.get());
|
||||
}
|
||||
CudaCheck(cudaGetLastError(), "scale fulls launch");
|
||||
CudaCheck(cudaDeviceSynchronize(), "scale fulls sync");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "scale fulls sync");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::GetFullsCorr(float *corr) const {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
const auto &d = *impl_;
|
||||
if (d.n_fulls == 0) return;
|
||||
CudaCheck(cudaMemcpy(corr, d.f_corr.get(), size_t(d.n_fulls) * sizeof(float),
|
||||
cudaMemcpyDeviceToHost), "download f_corr");
|
||||
CopyAndWait(corr, d.f_corr.get(), size_t(d.n_fulls) * sizeof(float),
|
||||
cudaMemcpyDeviceToHost, impl_->s(), "download f_corr");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::GetFullsPxPy(float *px, float *py) const {
|
||||
@@ -1422,8 +1544,8 @@ void RotationScaleMergeGPU::GetFullsPxPy(float *px, float *py) const {
|
||||
const auto &d = *impl_;
|
||||
if (d.n_fulls == 0) return;
|
||||
const size_t bytes = size_t(d.n_fulls) * sizeof(float);
|
||||
CudaCheck(cudaMemcpy(px, d.f_px.get(), bytes, cudaMemcpyDeviceToHost), "download f_px");
|
||||
CudaCheck(cudaMemcpy(py, d.f_py.get(), bytes, cudaMemcpyDeviceToHost), "download f_py");
|
||||
CopyAndWait(px, d.f_px.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_px");
|
||||
CopyAndWait(py, d.f_py.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_py");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::GetFullsVariance(float *var_bkg, float *var_per_I) const {
|
||||
@@ -1431,8 +1553,8 @@ void RotationScaleMergeGPU::GetFullsVariance(float *var_bkg, float *var_per_I) c
|
||||
const auto &d = *impl_;
|
||||
if (d.n_fulls == 0) return;
|
||||
const size_t bytes = size_t(d.n_fulls) * sizeof(float);
|
||||
CudaCheck(cudaMemcpy(var_bkg, d.f_var_bkg.get(), bytes, cudaMemcpyDeviceToHost), "download f_var_bkg");
|
||||
CudaCheck(cudaMemcpy(var_per_I, d.f_var_per_I.get(), bytes, cudaMemcpyDeviceToHost),
|
||||
CopyAndWait(var_bkg, d.f_var_bkg.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "download f_var_bkg");
|
||||
CopyAndWait(var_per_I, d.f_var_per_I.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(),
|
||||
"download f_var_per_I");
|
||||
}
|
||||
|
||||
@@ -1440,6 +1562,91 @@ void RotationScaleMergeGPU::SetFullsCorr(const float *corr) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
if (d.n_fulls == 0) return;
|
||||
CudaCheck(cudaMemcpy(d.f_corr.get(), corr, size_t(d.n_fulls) * sizeof(float),
|
||||
cudaMemcpyHostToDevice), "upload f_corr");
|
||||
CopyAndWait(d.f_corr.get(), corr, size_t(d.n_fulls) * sizeof(float),
|
||||
cudaMemcpyHostToDevice, impl_->s(), "upload f_corr");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SurfaceSetTerms(int n_terms, const SurfaceTerm *term, const uint8_t *parity,
|
||||
int n_groups, const int32_t *gperm, const int32_t *gstart,
|
||||
int ncell) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
d.s_ncell = ncell;
|
||||
d.s_n_groups = n_groups;
|
||||
d.Upload(d.s_term, term, n_terms);
|
||||
d.Upload(d.s_parity, parity, n_terms);
|
||||
d.Upload(d.s_gperm, gperm, n_terms);
|
||||
d.Upload(d.s_gstart, gstart, n_groups + 1);
|
||||
d.s_A = d.Alloc<double>(std::max(1, ncell));
|
||||
d.s_cross = d.Alloc<double>(std::max(1, ncell));
|
||||
d.s_ref2 = d.Alloc<double>(std::max(1, ncell));
|
||||
d.s_sw = d.Alloc<double>(std::max(1, n_groups));
|
||||
d.s_swI = d.Alloc<double>(std::max(1, n_groups));
|
||||
d.s_w_Is = d.Alloc<double>(std::max(1, n_terms));
|
||||
d.s_w_Iref = d.Alloc<double>(std::max(1, n_terms));
|
||||
d.s_Iref = d.Alloc<double>(std::max(1, n_terms));
|
||||
for (int &nb : d.s_n_blocks) nb = 0;
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SurfaceSetSubset(int subset, int n_blocks, const int32_t *perm,
|
||||
const int32_t *seg_start) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
const int n_seg = n_blocks * d.s_ncell;
|
||||
d.s_n_sel[subset] = n_blocks > 0 ? seg_start[n_seg] : 0;
|
||||
d.Upload(d.s_perm[subset], perm, d.s_n_sel[subset]);
|
||||
d.Upload(d.s_seg_start[subset], seg_start, n_seg + 1);
|
||||
d.s_n_blocks[subset] = n_blocks;
|
||||
if (size_t(n_seg) > d.s_slots) {
|
||||
d.s_tcross = d.Alloc<double>(n_seg);
|
||||
d.s_tref2 = d.Alloc<double>(n_seg);
|
||||
d.s_slots = n_seg;
|
||||
}
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SurfaceReference(int parity, const double *A) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
CopyAndWait(d.s_A.get(), A, size_t(d.s_ncell) * sizeof(double), cudaMemcpyHostToDevice, impl_->s(),
|
||||
"upload surface");
|
||||
const int grp_blocks = std::min(65535, (d.s_n_groups + BLK - 1) / BLK);
|
||||
if (grp_blocks > 0)
|
||||
SurfaceReferenceKernel<<<grp_blocks, BLK, 0, impl_->s()>>>(d.s_n_groups, parity, d.s_gperm.get(),
|
||||
d.s_gstart.get(), d.s_term.get(), d.s_parity.get(), d.s_A.get(), d.s_sw.get(), d.s_swI.get());
|
||||
CudaCheck(cudaGetLastError(), "surface reference launch");
|
||||
CudaCheck(cudaStreamSynchronize(impl_->s()), "surface reference sync");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SurfaceGetReference(double *sw, double *swI) const {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
const size_t bytes = size_t(d.s_n_groups) * sizeof(double);
|
||||
CopyAndWait(sw, d.s_sw.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "dl surface sw");
|
||||
CopyAndWait(swI, d.s_swI.get(), bytes, cudaMemcpyDeviceToHost, impl_->s(), "dl surface swI");
|
||||
}
|
||||
|
||||
void RotationScaleMergeGPU::SurfaceFitSums(int subset, double *cross, double *ref2) {
|
||||
DeviceGuard guard(impl_->device, impl_->available);
|
||||
auto &d = *impl_;
|
||||
const int ncell = d.s_ncell, nb = d.s_n_blocks[subset], n_seg = nb * ncell;
|
||||
if (nb == 0) {
|
||||
std::fill(cross, cross + ncell, 0.0);
|
||||
std::fill(ref2, ref2 + ncell, 0.0);
|
||||
return;
|
||||
}
|
||||
SurfaceFitTermKernel<<<std::min(65535, (d.s_n_sel[subset] + BLK - 1) / BLK), BLK, 0, impl_->s()>>>(
|
||||
d.s_n_sel[subset], d.s_perm[subset].get(), d.s_term.get(), d.s_A.get(), d.s_sw.get(), d.s_swI.get(),
|
||||
d.s_w_Is.get(), d.s_w_Iref.get(), d.s_Iref.get());
|
||||
CudaCheck(cudaGetLastError(), "surface fit term launch");
|
||||
SurfaceFitSegmentKernel<<<std::min(65535, (n_seg + BLK - 1) / BLK), BLK, 0, impl_->s()>>>(n_seg,
|
||||
d.s_seg_start[subset].get(), d.s_w_Is.get(), d.s_w_Iref.get(), d.s_Iref.get(),
|
||||
d.s_tcross.get(), d.s_tref2.get());
|
||||
CudaCheck(cudaGetLastError(), "surface fit segment launch");
|
||||
SurfaceFitCellKernel<<<std::min(65535, (ncell + BLK - 1) / BLK), BLK, 0, impl_->s()>>>(nb, ncell,
|
||||
d.s_tcross.get(), d.s_tref2.get(), d.s_cross.get(), d.s_ref2.get());
|
||||
CudaCheck(cudaGetLastError(), "surface cell sum launch");
|
||||
CopyAndWait(cross, d.s_cross.get(), size_t(ncell) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(),
|
||||
"dl surface cross");
|
||||
CopyAndWait(ref2, d.s_ref2.get(), size_t(ncell) * sizeof(double), cudaMemcpyDeviceToHost, impl_->s(),
|
||||
"dl surface ref2");
|
||||
}
|
||||
|
||||
@@ -192,6 +192,34 @@ public:
|
||||
// Download the fulls' working corr (length = n_fulls), valid after ScaleFulls.
|
||||
void GetFullsCorr(float *corr) const;
|
||||
|
||||
// --- correction-surface fit (RotationScaleMerge::ApplyCellSurface) ---
|
||||
// The two passes every round of the fit makes over its terms, on the device; the host keeps the
|
||||
// per-cell step, the gauge and the cross-validation. Both sums are formed in exactly the order the
|
||||
// host forms them, so the fitted surface is the host's to the last bit (see the kernels).
|
||||
|
||||
// One observation as the surface fit sees it: the host's term, uploaded as it stands.
|
||||
struct SurfaceTerm { float I, sigma, corr, d; int32_t cell, group; };
|
||||
|
||||
// The terms (in fulls order) with each one's frame parity, the ASU-group CSR over them (gperm lists
|
||||
// the terms of group g at [gstart[g], gstart[g+1]), in fulls order) and the cell count.
|
||||
void SurfaceSetTerms(int n_terms, const SurfaceTerm *term, const uint8_t *parity,
|
||||
int n_groups, const int32_t *gperm, const int32_t *gstart, int ncell);
|
||||
|
||||
// One subset of the terms (0 = even frames, 1 = odd, 2 = all), cut into the host's n_blocks
|
||||
// reduction blocks and ordered within each block by cell, keeping term order inside a cell:
|
||||
// block b, cell c is perm[seg_start[b * ncell + c], seg_start[b * ncell + c + 1]).
|
||||
void SurfaceSetSubset(int subset, int n_blocks, const int32_t *perm, const int32_t *seg_start);
|
||||
|
||||
// The per-group reference sums sw / swI over the terms of frame parity `parity` (< 0 = all) with
|
||||
// the surface A (length ncell) applied. They stay on the device for SurfaceFitSums;
|
||||
// SurfaceGetReference downloads them (length n_groups each).
|
||||
void SurfaceReference(int parity, const double *A);
|
||||
void SurfaceGetReference(double *sw, double *swI) const;
|
||||
|
||||
// The fit's per-cell sums over one subset against the last SurfaceReference and its A:
|
||||
// cross = sum w Is Iref and ref2 = sum w Iref^2 (length ncell each).
|
||||
void SurfaceFitSums(int subset, double *cross, double *ref2);
|
||||
|
||||
private:
|
||||
struct Impl;
|
||||
std::unique_ptr<Impl> impl_;
|
||||
|
||||
@@ -17,7 +17,7 @@ AdaptiveSpotFinderCPU::AdaptiveSpotFinderCPU(const AzimuthalIntegrationMapping &
|
||||
ring_cnt.assign(nbins, 0);
|
||||
ring_mean.assign(nbins, 0.0f);
|
||||
ring_sigma.assign(nbins, 0.0f);
|
||||
ring_thr.assign(nbins, 0.0f);
|
||||
ring_thr.assign(nbins + 1, INFINITY); // the last entry is for pixels outside every ring
|
||||
ring_bkg.assign(nbins, NAN);
|
||||
ring_bits.assign(OutputSize(), 0);
|
||||
ring_hist.assign(nbins * HIST_VALUES, 0);
|
||||
@@ -54,84 +54,54 @@ void AdaptiveSpotFinderCPU::BeginRings() {
|
||||
rings_from_blocks = true;
|
||||
}
|
||||
|
||||
// The plain pass over pixels [first, first + n).
|
||||
// The plain pass over pixels [first, first + n): the histogram of each ring's values, from which
|
||||
// PlainRings() takes the integer sums, and the fused profile when asked for. One loop reads each pixel
|
||||
// once for both.
|
||||
void AdaptiveSpotFinderCPU::AccumulateRingsBlock(const ImagePreprocessorBuffer &image, size_t first, size_t n) {
|
||||
const auto &pixel_to_bin = mapping.GetPixelToBin();
|
||||
const size_t nbins = ring_sum.size();
|
||||
const float *corrections = mapping.Corrections().data();
|
||||
|
||||
// Consecutive pixels mostly share a ring, so a ring's sums are held in locals while they do and
|
||||
// written back when the ring changes: the same additions in the same order, without a store and a
|
||||
// reload of the same address on every pixel. The azimuthal-integration sums get a loop of their own
|
||||
// over the block, so that neither loop runs out of registers for its sums.
|
||||
if (fuse_azint) {
|
||||
size_t cur = nbins; // the ring held in the locals below; nbins = none
|
||||
float az_sum = 0.0f, az_sum2 = 0.0f;
|
||||
uint32_t az_cnt = 0;
|
||||
for (size_t pxl = first; pxl < first + n; ++pxl) {
|
||||
const int32_t v = image[pxl];
|
||||
if (v == INT32_MIN || v == INT32_MAX) continue; // bad / saturated
|
||||
const uint16_t b = pixel_to_bin[pxl];
|
||||
if (b >= nbins) continue; // masked / out of range (UINT16_MAX)
|
||||
if (b != cur) {
|
||||
if (cur != nbins) {
|
||||
azint_sum[cur] = az_sum;
|
||||
azint_sum2[cur] = az_sum2;
|
||||
azint_count[cur] = az_cnt;
|
||||
}
|
||||
cur = b;
|
||||
az_sum = azint_sum[b];
|
||||
az_sum2 = azint_sum2[b];
|
||||
az_cnt = azint_count[b];
|
||||
}
|
||||
const float val = static_cast<float>(v) * corrections[pxl];
|
||||
const float val_sq = val * val;
|
||||
az_sum += val;
|
||||
az_sum2 += val_sq;
|
||||
++az_cnt;
|
||||
}
|
||||
if (cur != nbins) {
|
||||
azint_sum[cur] = az_sum;
|
||||
azint_sum2[cur] = az_sum2;
|
||||
azint_count[cur] = az_cnt;
|
||||
}
|
||||
}
|
||||
|
||||
// Values outside the histogram are listed by a second loop, run only when the block has any: a
|
||||
// call in this loop would leave the sums in memory again.
|
||||
uint32_t *hist = ring_hist.data();
|
||||
|
||||
// Consecutive pixels mostly share a ring, so the ring's profile sums are held in locals while they
|
||||
// do and written back when the ring changes: the same additions in the same order, without a store
|
||||
// and a reload of the same address on every pixel. Values outside the histogram are listed by a
|
||||
// second loop, run only when the block has any.
|
||||
bool overflow = false;
|
||||
size_t cur = nbins;
|
||||
int64_t sum = 0, cnt = 0;
|
||||
uint64_t sum2 = 0;
|
||||
size_t cur = nbins; // the ring held in the locals below; nbins = none
|
||||
float az_sum = 0.0f, az_sum2 = 0.0f;
|
||||
uint32_t az_cnt = 0;
|
||||
for (size_t pxl = first; pxl < first + n; ++pxl) {
|
||||
const int32_t v = image[pxl];
|
||||
if (v == INT32_MIN || v == INT32_MAX) continue; // bad / saturated
|
||||
const uint16_t b = pixel_to_bin[pxl];
|
||||
if (b >= nbins) continue; // masked / out of range (UINT16_MAX)
|
||||
if (b != cur) {
|
||||
if (cur != nbins) {
|
||||
ring_sum[cur] = sum;
|
||||
ring_sum2[cur] = sum2;
|
||||
ring_cnt[cur] = cnt;
|
||||
}
|
||||
cur = b;
|
||||
sum = ring_sum[b];
|
||||
sum2 = ring_sum2[b];
|
||||
cnt = ring_cnt[b];
|
||||
}
|
||||
sum += v;
|
||||
sum2 += static_cast<uint64_t>(static_cast<int64_t>(v) * v);
|
||||
cnt += 1;
|
||||
if (v >= 0 && v < HIST_VALUES)
|
||||
if (static_cast<uint32_t>(v) < HIST_VALUES)
|
||||
hist[b * HIST_VALUES + v] += 1;
|
||||
else
|
||||
overflow = true;
|
||||
if (!fuse_azint) continue;
|
||||
if (b != cur) {
|
||||
if (cur != nbins) {
|
||||
azint_sum[cur] = az_sum;
|
||||
azint_sum2[cur] = az_sum2;
|
||||
azint_count[cur] = az_cnt;
|
||||
}
|
||||
cur = b;
|
||||
az_sum = azint_sum[b];
|
||||
az_sum2 = azint_sum2[b];
|
||||
az_cnt = azint_count[b];
|
||||
}
|
||||
const float val = static_cast<float>(v) * corrections[pxl];
|
||||
const float val_sq = val * val;
|
||||
az_sum += val;
|
||||
az_sum2 += val_sq;
|
||||
++az_cnt;
|
||||
}
|
||||
if (cur != nbins) {
|
||||
ring_sum[cur] = sum;
|
||||
ring_sum2[cur] = sum2;
|
||||
ring_cnt[cur] = cnt;
|
||||
azint_sum[cur] = az_sum;
|
||||
azint_sum2[cur] = az_sum2;
|
||||
azint_count[cur] = az_cnt;
|
||||
}
|
||||
|
||||
if (overflow)
|
||||
@@ -146,7 +116,8 @@ void AdaptiveSpotFinderCPU::AccumulateRingsBlock(const ImagePreprocessorBuffer &
|
||||
}
|
||||
|
||||
// A sigma-clip pass over the plain pass's values: each distinct value of a ring meets the same test the
|
||||
// pixels holding it would, and its pixels are added as a count.
|
||||
// pixels holding it would, and its pixels are added as a count. Integer sums, so with nothing clipped
|
||||
// (clip_k = INFINITY) they are the plain sums over the pixels themselves.
|
||||
void AdaptiveSpotFinderCPU::ClipRings(float clip_k) {
|
||||
const size_t nbins = ring_sum.size();
|
||||
|
||||
@@ -155,6 +126,7 @@ void AdaptiveSpotFinderCPU::ClipRings(float clip_k) {
|
||||
std::fill(ring_cnt.begin(), ring_cnt.end(), 0);
|
||||
|
||||
const auto keep = [&](uint16_t b, int32_t v) {
|
||||
if (std::isinf(clip_k)) return true;
|
||||
const float lo = ring_mean[b] - clip_k * ring_sigma[b];
|
||||
const float hi = ring_mean[b] + clip_k * ring_sigma[b];
|
||||
return !(v < lo || v > hi); // exclude peaks / outliers
|
||||
@@ -197,6 +169,7 @@ void AdaptiveSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image,
|
||||
AccumulateRingsBlock(image, 0, static_cast<size_t>(width) * height);
|
||||
}
|
||||
rings_from_blocks = false;
|
||||
ClipRings(INFINITY);
|
||||
UpdateRingStatistics();
|
||||
ClipRings(3.0f);
|
||||
UpdateRingStatistics();
|
||||
@@ -259,19 +232,31 @@ void AdaptiveSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image,
|
||||
|
||||
void AdaptiveSpotFinderCPU::FlagRow(const ImagePreprocessorBuffer &image, int32_t row) {
|
||||
const auto &pixel_to_bin = mapping.GetPixelToBin();
|
||||
const size_t nbins = ring_thr.size();
|
||||
const auto nbins = static_cast<uint32_t>(ring_thr.size() - 1); // ring_thr[nbins] is +inf
|
||||
const float *thr = ring_thr.data();
|
||||
const int32_t *img = image.data();
|
||||
const uint16_t *bin = pixel_to_bin.data();
|
||||
const size_t first = static_cast<size_t>(row) * width;
|
||||
const size_t end = first + width;
|
||||
|
||||
for (size_t pxl = first; pxl < first + width; ++pxl) {
|
||||
const int32_t v = image[pxl];
|
||||
const uint16_t b = pixel_to_bin[pxl];
|
||||
bool strong = false;
|
||||
if (v == INT32_MAX)
|
||||
strong = true;
|
||||
else if (v != INT32_MIN && b < nbins && v >= ring_thr[b])
|
||||
strong = true;
|
||||
|
||||
if (strong)
|
||||
ring_bits[pxl / 32] |= 1U << (pxl % 32);
|
||||
// Saturated is strong, bad is not, and a pixel outside every ring (bin >= nbins) meets the +inf
|
||||
// threshold. Written without branches and a word of 32 pixels at a time, so that the loop vectorises.
|
||||
const auto strong = [&](size_t pxl) -> uint32_t {
|
||||
const int32_t v = img[pxl];
|
||||
const uint32_t b = std::min<uint32_t>(bin[pxl], nbins);
|
||||
return (v == INT32_MAX) | ((v != INT32_MIN) & (v >= thr[b]));
|
||||
};
|
||||
size_t pxl = first;
|
||||
while (pxl < end && pxl % 32 != 0) {
|
||||
ring_bits[pxl / 32] |= strong(pxl) << (pxl % 32);
|
||||
++pxl;
|
||||
}
|
||||
for (; pxl + 32 <= end; pxl += 32) {
|
||||
uint32_t word = 0;
|
||||
for (uint32_t j = 0; j < 32; ++j)
|
||||
word |= strong(pxl + j) << j;
|
||||
ring_bits[pxl / 32] |= word;
|
||||
}
|
||||
for (; pxl < end; ++pxl)
|
||||
ring_bits[pxl / 32] |= strong(pxl) << (pxl % 32);
|
||||
}
|
||||
|
||||
@@ -57,7 +57,7 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU {
|
||||
std::vector<int64_t> ring_cnt;
|
||||
std::vector<float> ring_mean;
|
||||
std::vector<float> ring_sigma;
|
||||
std::vector<float> ring_thr;
|
||||
std::vector<float> ring_thr; // nbins + 1: the last is +inf, the threshold of a pixel outside every ring
|
||||
// ring_mean of the last Detect(), NaN where the ring holds too few pixels to be its own background.
|
||||
// Kept separately because ring_mean carries the previous frame's value for an empty ring.
|
||||
std::vector<float> ring_bkg;
|
||||
@@ -65,8 +65,8 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU {
|
||||
// local-box mask that ImageSpotFinderCPU::Detect leaves in output_buffer.
|
||||
std::vector<uint32_t> ring_bits;
|
||||
// The plain pass's valid pixels as a per-ring histogram of their values (HIST_VALUES bins per
|
||||
// ring) plus a list of the values outside it, so the two sigma-clip passes sum over distinct
|
||||
// values instead of over the image again. Integer sums, so the same totals.
|
||||
// ring) plus a list of the values outside it. The plain sums and the two sigma-clip passes are all
|
||||
// taken from it, over distinct values instead of over the image. Integer sums, so the same totals.
|
||||
static constexpr int32_t HIST_VALUES = 1024;
|
||||
std::vector<uint32_t> ring_hist;
|
||||
std::vector<std::pair<uint16_t, int32_t>> ring_overflow; // (ring, value)
|
||||
@@ -84,7 +84,7 @@ class AdaptiveSpotFinderCPU : public ImageSpotFinderCPU {
|
||||
|
||||
// Zero the sums of the plain ring pass (and of the fused profile).
|
||||
void ResetRings();
|
||||
// One sigma-clip pass over the plain pass's values.
|
||||
// One sigma-clip pass over the plain pass's values; clip_k = INFINITY gives the plain sums.
|
||||
void ClipRings(float clip_k);
|
||||
// ring_mean / ring_sigma from the current sums.
|
||||
void UpdateRingStatistics();
|
||||
|
||||
+65
-43
@@ -216,7 +216,9 @@ void HotPixelFinder::AddLevels(const std::vector<int32_t> §or_level, const s
|
||||
const int r = static_cast<int>(k / SECTORS);
|
||||
level[k] = std::max(ring_level[r], sector_level[k]);
|
||||
const float noise = std::max(std::sqrt(static_cast<float>(std::max(level[k], 0))), ring_spread[r]);
|
||||
threshold[k] = static_cast<float>(level[k]) + LIT_NSIGMA * noise + LIT_OFFSET;
|
||||
// One rounding for level + nsigma * noise and one for the offset, written out so that every
|
||||
// compiler takes the same two - the device takes them too (HotPixelsGPU.cu).
|
||||
threshold[k] = std::fma(LIT_NSIGMA, noise, static_cast<float>(level[k])) + LIT_OFFSET;
|
||||
}
|
||||
|
||||
// Every sum is an integer, so the result does not depend on the order the frames arrive in.
|
||||
@@ -230,50 +232,24 @@ void HotPixelFinder::AddLevels(const std::vector<int32_t> §or_level, const s
|
||||
}
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
void HotPixelFinder::PrepareDevice() {
|
||||
std::lock_guard lock(m);
|
||||
if (!gpu)
|
||||
gpu = std::make_unique<HotPixelFinderGPU>(key.get(), width * height, key_begin, nrings, SECTORS,
|
||||
HotPixelLevelRules{MIN_SECTOR_PIXELS, MIN_RING_PIXELS,
|
||||
LIT_NSIGMA, LIT_OFFSET});
|
||||
}
|
||||
|
||||
void HotPixelFinder::AddDeviceImage(const int32_t *device_image, HotPixelFinderGPU::Frame &frame) {
|
||||
HotPixelFinderGPU *device;
|
||||
{
|
||||
std::lock_guard lock(m);
|
||||
if (!gpu)
|
||||
gpu = std::make_unique<HotPixelFinderGPU>(key.get(), width * height, key_begin, nrings, SECTORS);
|
||||
device = gpu.get();
|
||||
}
|
||||
std::vector<uint32_t> count;
|
||||
std::vector<int32_t> sector_median, ring_median, ring_mad;
|
||||
device->Statistics(device_image, frame, count, sector_median, ring_median, ring_mad);
|
||||
|
||||
// The same levels AddImage takes off its scratch buffer, and under the same pixel minima.
|
||||
const size_t nkeys = static_cast<size_t>(nrings) * SECTORS;
|
||||
std::vector<int32_t> sector_level(nkeys, 0);
|
||||
for (size_t k = 0; k < nkeys; k++)
|
||||
if (count[k] >= MIN_SECTOR_PIXELS)
|
||||
sector_level[k] = sector_median[k];
|
||||
std::vector<int32_t> ring_level(nrings, 0);
|
||||
std::vector<float> ring_spread(nrings, 0.0f);
|
||||
std::vector<char> ring_ok(nrings, 0);
|
||||
for (int r = 0; r < nrings; r++) {
|
||||
size_t n = 0;
|
||||
for (int s = 0; s < SECTORS; s++)
|
||||
n += count[r * SECTORS + s];
|
||||
if (n < MIN_RING_PIXELS) continue;
|
||||
ring_ok[r] = 1;
|
||||
ring_level[r] = ring_median[r];
|
||||
ring_spread[r] = 1.4826f * static_cast<float>(ring_mad[r]);
|
||||
}
|
||||
|
||||
std::vector<int32_t> level;
|
||||
std::vector<float> threshold;
|
||||
AddLevels(sector_level, ring_level, ring_spread, ring_ok, level, threshold);
|
||||
device->Accumulate(device_image, frame, level, threshold, ring_ok);
|
||||
PrepareDevice();
|
||||
gpu->Add(device_image, frame);
|
||||
std::lock_guard lock(m);
|
||||
frames++;
|
||||
}
|
||||
#endif
|
||||
|
||||
HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double spacing_deg, size_t nthreads) {
|
||||
std::lock_guard lock(m);
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (gpu)
|
||||
gpu->Download(n_lit.get(), n_error.get(), sum_value.get(), n_error_ring_ok.get(), error_level_sum.get());
|
||||
#endif
|
||||
Result ret;
|
||||
ret.frames = frames;
|
||||
ret.mask.assign(width * height, 0);
|
||||
@@ -281,12 +257,38 @@ HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double sp
|
||||
|
||||
// The chance rate per ring, from the pixels lit on no more than half of their frames: whatever
|
||||
// lights those - reflections, zingers, noise above the bound - lights a defect-free pixel too.
|
||||
// Counted in integers by blocks of rows in parallel, so the totals do not depend on the split.
|
||||
std::vector<double> lit(nrings, 0.0), seen(nrings, 0.0);
|
||||
for (size_t i = 0; i < width * height; i++)
|
||||
if (key[i] >= 0 && n_valid(i) > 0 && 2 * n_lit[i] <= n_valid(i)) {
|
||||
lit[key[i] / SECTORS] += n_lit[i];
|
||||
seen[key[i] / SECTORS] += n_valid(i);
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
if (gpu) {
|
||||
std::vector<int64_t> device_lit, device_seen;
|
||||
gpu->ChanceCounts(device_lit, device_seen);
|
||||
for (int r = 0; r < nrings; r++) {
|
||||
lit[r] = static_cast<double>(device_lit[r]);
|
||||
seen[r] = static_cast<double>(device_seen[r]);
|
||||
}
|
||||
} else
|
||||
#endif
|
||||
{
|
||||
std::vector<std::vector<int64_t>> block_lit(BANDS), block_seen(BANDS);
|
||||
const size_t rows_per_band = (height + BANDS - 1) / BANDS;
|
||||
ParallelFor(static_cast<int>(BANDS), nthreads, [&](int b) {
|
||||
block_lit[b].assign(nrings, 0);
|
||||
block_seen[b].assign(nrings, 0);
|
||||
const size_t begin = std::min(width * height, b * rows_per_band * width);
|
||||
const size_t end = std::min(width * height, (b + 1) * rows_per_band * width);
|
||||
for (size_t i = begin; i < end; i++)
|
||||
if (key[i] >= 0 && n_valid(i) > 0 && 2 * n_lit[i] <= n_valid(i)) {
|
||||
block_lit[b][key[i] / SECTORS] += n_lit[i];
|
||||
block_seen[b][key[i] / SECTORS] += n_valid(i);
|
||||
}
|
||||
});
|
||||
for (int r = 0; r < nrings; r++)
|
||||
for (size_t b = 0; b < BANDS; b++) {
|
||||
lit[r] += static_cast<double>(block_lit[b][r]);
|
||||
seen[r] += static_cast<double>(block_seen[b][r]);
|
||||
}
|
||||
}
|
||||
std::vector<int> k_chance(nrings, n + 1);
|
||||
for (int r = 0; r < nrings; r++)
|
||||
if (seen[r] > 0.0)
|
||||
@@ -295,6 +297,26 @@ HotPixelFinder::Result HotPixelFinder::GetMask(double oscillation_deg, double sp
|
||||
|
||||
// Persistent: lit on more frames than one reflection or chance explains.
|
||||
const int min_valid = std::max(10, n / 2);
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
// With a GPU the per-pixel sums stay there. Only the pixels that can be masked come back - those the
|
||||
// tests below could pass (see GetCandidates) - into the host arrays, which are zero everywhere else,
|
||||
// so the tests below run on them unchanged.
|
||||
if (gpu) {
|
||||
const auto c = gpu->GetCandidates(frames, min_valid, spacing_deg > 0.0, k_chance);
|
||||
for (size_t j = 0; j < c.index.size(); j++) {
|
||||
const size_t i = c.index[j];
|
||||
n_lit[i] = c.n_lit[j];
|
||||
n_error[i] = c.n_error[j];
|
||||
n_error_ring_ok[i] = c.n_error_ring_ok[j];
|
||||
sum_value[i] = c.sum_value[j];
|
||||
error_level_sum[i] = c.error_level_sum[j];
|
||||
}
|
||||
for (size_t k = 0; k < key_frames.size(); k++) {
|
||||
key_frames[k] = static_cast<uint16_t>(c.key_frames[k]);
|
||||
key_level_sum[k] = c.key_level_sum[k];
|
||||
}
|
||||
}
|
||||
#endif
|
||||
std::vector<uint8_t> persistent(width * height, 0);
|
||||
ParallelChunks(static_cast<int>(height), nthreads, [&](int y0, int y1) {
|
||||
for (size_t y = y0; y < static_cast<size_t>(y1); y++)
|
||||
|
||||
+6
-2
@@ -94,6 +94,10 @@ public:
|
||||
// `scratch` above. The per-pixel sums are then kept on the device and replace the host's when the
|
||||
// mask is read, so a finder is fed one way or the other, not both. Thread safe.
|
||||
void AddDeviceImage(const int32_t *device_image, HotPixelFinderGPU::Frame &frame);
|
||||
|
||||
// Build the device half now rather than with the first device frame, so the workers do not wait on
|
||||
// it one behind the other.
|
||||
void PrepareDevice();
|
||||
#endif
|
||||
|
||||
// The mask, from the frames added so far. oscillation_deg is the rotation per image and
|
||||
@@ -132,11 +136,11 @@ private:
|
||||
std::unique_ptr<uint16_t[]> n_error_ring_ok;
|
||||
std::unique_ptr<int64_t[]> error_level_sum;
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
std::unique_ptr<HotPixelFinderGPU> gpu; // built by the first device frame
|
||||
std::unique_ptr<HotPixelFinderGPU> gpu; // built by PrepareDevice or the first device frame
|
||||
#endif
|
||||
|
||||
// Each ring-sector's level and lit threshold, from the frame's order statistics, and the frame's
|
||||
// share of the per-key sums. The host and the device path both come through here.
|
||||
// share of the per-key sums. The device does the same in HotPixelsGPU.cu, with the same roundings.
|
||||
void AddLevels(const std::vector<int32_t> §or_level, const std::vector<int32_t> &ring_level,
|
||||
const std::vector<float> &ring_spread, const std::vector<char> &ring_ok,
|
||||
std::vector<int32_t> &level, std::vector<float> &threshold);
|
||||
|
||||
+198
-47
@@ -3,6 +3,8 @@
|
||||
|
||||
#include "HotPixelsGPU.h"
|
||||
|
||||
#include <cub/device/device_radix_sort.cuh>
|
||||
|
||||
#include "../common/JFJochException.h"
|
||||
|
||||
namespace {
|
||||
@@ -145,6 +147,84 @@ __global__ void accumulate_kernel(const int32_t *__restrict__ image, const int32
|
||||
}
|
||||
}
|
||||
|
||||
// Each key's level and lit threshold from the frame's order statistics, and the frame's share of the
|
||||
// per-key sums - HotPixelFinder::AddImage and AddLevels, step for step. The threshold is
|
||||
// fma(nsigma, noise, level) + offset, the two roundings the host takes (see AddLevels).
|
||||
__global__ void levels_kernel(size_t nkeys, int sectors, HotPixelLevelRules rules, const uint32_t *__restrict__ count,
|
||||
const int32_t *__restrict__ sector_median, const int32_t *__restrict__ ring_median,
|
||||
const int32_t *__restrict__ ring_mad, int32_t *__restrict__ level,
|
||||
float *__restrict__ threshold, char *__restrict__ ring_ok,
|
||||
uint32_t *__restrict__ key_frames, int64_t *__restrict__ key_level_sum) {
|
||||
const size_t k = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (k >= nkeys) return;
|
||||
const size_t r = k / sectors;
|
||||
uint32_t n = 0;
|
||||
for (int s = 0; s < sectors; s++)
|
||||
n += count[r * sectors + s];
|
||||
const bool ok = n >= static_cast<uint32_t>(rules.min_ring_pixels);
|
||||
const int32_t ring_level = ok ? ring_median[r] : 0;
|
||||
const float ring_spread = ok ? 1.4826f * static_cast<float>(ring_mad[r]) : 0.0f;
|
||||
const int32_t sector_level = count[k] >= static_cast<uint32_t>(rules.min_sector_pixels) ? sector_median[k] : 0;
|
||||
const int32_t lv = max(ring_level, sector_level);
|
||||
const float root = sqrtf(static_cast<float>(max(lv, 0)));
|
||||
const float noise = root < ring_spread ? ring_spread : root;
|
||||
level[k] = lv;
|
||||
threshold[k] = __fadd_rn(__fmaf_rn(rules.lit_nsigma, noise, static_cast<float>(lv)), rules.lit_offset);
|
||||
if (k % sectors == 0)
|
||||
ring_ok[r] = ok;
|
||||
if (ok) {
|
||||
atomicAdd(&key_frames[k], 1u);
|
||||
atomicAdd(reinterpret_cast<unsigned long long *>(&key_level_sum[k]),
|
||||
static_cast<unsigned long long>(static_cast<int64_t>(lv)));
|
||||
}
|
||||
}
|
||||
|
||||
__device__ int valid_frames(const int32_t *key, const uint32_t *key_frames, const uint16_t *n_error_ring_ok, size_t i) {
|
||||
return static_cast<int>(key_frames[key[i]]) - static_cast<int>(n_error_ring_ok[i]);
|
||||
}
|
||||
|
||||
__global__ void chance_kernel(size_t npixels, int sectors, const int32_t *__restrict__ key,
|
||||
const uint32_t *__restrict__ key_frames, const uint16_t *__restrict__ n_lit,
|
||||
const uint16_t *__restrict__ n_error_ring_ok, unsigned long long *__restrict__ lit,
|
||||
unsigned long long *__restrict__ seen) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= npixels || key[i] < 0) return;
|
||||
const int nv = valid_frames(key, key_frames, n_error_ring_ok, i);
|
||||
if (nv > 0 && 2 * static_cast<int>(n_lit[i]) <= nv) {
|
||||
atomicAdd(&lit[key[i] / sectors], static_cast<unsigned long long>(n_lit[i]));
|
||||
atomicAdd(&seen[key[i] / sectors], static_cast<unsigned long long>(nv));
|
||||
}
|
||||
}
|
||||
|
||||
// Writes the candidates' indices from `out` on when `out` is given, and counts them either way.
|
||||
__global__ void candidate_kernel(size_t npixels, int sectors, uint32_t frames, int min_valid, bool spacing_ok,
|
||||
const int32_t *__restrict__ key, const uint32_t *__restrict__ key_frames,
|
||||
const uint16_t *__restrict__ n_lit, const uint16_t *__restrict__ n_error,
|
||||
const uint16_t *__restrict__ n_error_ring_ok, const int *__restrict__ k_chance,
|
||||
uint32_t *__restrict__ count, uint32_t *__restrict__ out) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i >= npixels || key[i] < 0) return;
|
||||
const int nv = valid_frames(key, key_frames, n_error_ring_ok, i);
|
||||
const bool error = 2 * static_cast<uint32_t>(n_error[i]) > frames;
|
||||
const int lit = n_lit[i];
|
||||
const bool persistent = spacing_ok && nv >= min_valid && lit > 0
|
||||
&& lit >= min(max(2, k_chance[key[i] / sectors]), nv);
|
||||
if (!error && !persistent) return;
|
||||
const uint32_t slot = atomicAdd(count, 1u);
|
||||
if (out) out[slot] = static_cast<uint32_t>(i);
|
||||
}
|
||||
|
||||
__global__ void iota_kernel(size_t n, uint32_t *__restrict__ out) {
|
||||
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (i < n) out[i] = static_cast<uint32_t>(i);
|
||||
}
|
||||
|
||||
template <class T>
|
||||
__global__ void gather_kernel(size_t n, const uint32_t *__restrict__ index, const T *__restrict__ in, T *__restrict__ out) {
|
||||
const size_t j = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
||||
if (j < n) out[j] = in[index[j]];
|
||||
}
|
||||
|
||||
// The shared tables and sums are filled on a stream of their own, added to on the workers' streams
|
||||
// and downloaded on the NULL stream, so they are allocated synchronously rather than from the pool:
|
||||
// a pooled buffer is freed on the thread's allocation stream, which none of those is ordered before.
|
||||
@@ -156,22 +236,18 @@ constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous;
|
||||
} // namespace
|
||||
|
||||
HotPixelFinderGPU::HotPixelFinderGPU(const int32_t *host_key, size_t npixels,
|
||||
const std::vector<uint32_t> &host_key_begin, int nrings, int sectors)
|
||||
const std::vector<uint32_t> &host_key_begin, int nrings, int sectors,
|
||||
const HotPixelLevelRules &rules)
|
||||
: npixels(npixels), nkeys(static_cast<size_t>(nrings) * sectors), nrings(nrings),
|
||||
sectors(sectors),
|
||||
key(npixels, ALLOC), pixels_by_key(host_key_begin.back(), ALLOC), key_begin(host_key_begin.size(), ALLOC),
|
||||
sectors(sectors), rules(rules),
|
||||
key(npixels, ALLOC), pixels_by_key(std::max<size_t>(host_key_begin.back(), 1), ALLOC),
|
||||
key_begin(host_key_begin.size(), ALLOC),
|
||||
n_lit(npixels, ALLOC), n_error(npixels, ALLOC), n_error_ring_ok(npixels, ALLOC),
|
||||
sum_value(npixels, ALLOC), error_level_sum(npixels, ALLOC) {
|
||||
std::vector<uint32_t> pixels(host_key_begin.back());
|
||||
std::vector<uint32_t> filled(host_key_begin.begin(), host_key_begin.end() - 1);
|
||||
for (size_t i = 0; i < npixels; i++)
|
||||
if (host_key[i] >= 0)
|
||||
pixels[filled[host_key[i]]++] = static_cast<uint32_t>(i);
|
||||
|
||||
sum_value(npixels, ALLOC), error_level_sum(npixels, ALLOC),
|
||||
key_frames(std::max<size_t>(nkeys, 1), ALLOC), key_level_sum(std::max<size_t>(nkeys, 1), ALLOC) {
|
||||
cuda_err(cudaEventCreateWithFlags(&last_accumulate, cudaEventDisableTiming));
|
||||
CudaStream stream;
|
||||
cuda_err(cudaMemcpyAsync(key, host_key, npixels * sizeof(int32_t), cudaMemcpyHostToDevice, stream));
|
||||
cuda_err(cudaMemcpyAsync(pixels_by_key, pixels.data(), pixels.size() * sizeof(uint32_t),
|
||||
cudaMemcpyHostToDevice, stream));
|
||||
cuda_err(cudaMemcpyAsync(key_begin, host_key_begin.data(), host_key_begin.size() * sizeof(uint32_t),
|
||||
cudaMemcpyHostToDevice, stream));
|
||||
cuda_err(cudaMemsetAsync(n_lit, 0, npixels * sizeof(uint16_t), stream));
|
||||
@@ -179,16 +255,34 @@ HotPixelFinderGPU::HotPixelFinderGPU(const int32_t *host_key, size_t npixels,
|
||||
cuda_err(cudaMemsetAsync(n_error_ring_ok, 0, npixels * sizeof(uint16_t), stream));
|
||||
cuda_err(cudaMemsetAsync(sum_value, 0, npixels * sizeof(int64_t), stream));
|
||||
cuda_err(cudaMemsetAsync(error_level_sum, 0, npixels * sizeof(int64_t), stream));
|
||||
cuda_err(cudaStreamSynchronize(stream));
|
||||
cuda_err(cudaMemsetAsync(key_frames, 0, std::max<size_t>(nkeys, 1) * sizeof(uint32_t), stream));
|
||||
cuda_err(cudaMemsetAsync(key_level_sum, 0, std::max<size_t>(nkeys, 1) * sizeof(int64_t), stream));
|
||||
|
||||
// The unmasked pixels grouped by key: a stable sort of the pixel indices by key, so each key's
|
||||
// pixels stay in pixel order. A masked pixel's key, -1, is the largest as unsigned and sorts last.
|
||||
{
|
||||
CudaDevicePtr<uint32_t> index(npixels, ALLOC), sorted_key(npixels, ALLOC), sorted_index(npixels, ALLOC);
|
||||
iota_kernel<<<static_cast<unsigned>((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>(npixels, index);
|
||||
cuda_err(cudaGetLastError());
|
||||
const auto *keys_in = reinterpret_cast<const uint32_t *>(key.get());
|
||||
size_t bytes = 0;
|
||||
cuda_err(cub::DeviceRadixSort::SortPairs(nullptr, bytes, keys_in, sorted_key.get(), index.get(),
|
||||
sorted_index.get(), npixels, 0, 32, stream));
|
||||
CudaDevicePtr<uint8_t> scratch(bytes, ALLOC);
|
||||
cuda_err(cub::DeviceRadixSort::SortPairs(scratch.get(), bytes, keys_in, sorted_key.get(), index.get(),
|
||||
sorted_index.get(), npixels, 0, 32, stream));
|
||||
if (host_key_begin.back() > 0)
|
||||
cuda_err(cudaMemcpyAsync(pixels_by_key, sorted_index, host_key_begin.back() * sizeof(uint32_t),
|
||||
cudaMemcpyDeviceToDevice, stream));
|
||||
cuda_err(cudaStreamSynchronize(stream));
|
||||
}
|
||||
}
|
||||
|
||||
void HotPixelFinderGPU::Statistics(const int32_t *device_image, Frame &frame, std::vector<uint32_t> &count,
|
||||
std::vector<int32_t> §or_median, std::vector<int32_t> &ring_median,
|
||||
std::vector<int32_t> &ring_mad) {
|
||||
count.resize(nkeys);
|
||||
sector_median.resize(nkeys);
|
||||
ring_median.resize(nrings);
|
||||
ring_mad.resize(nrings);
|
||||
HotPixelFinderGPU::~HotPixelFinderGPU() {
|
||||
if (last_accumulate) cudaEventDestroy(last_accumulate);
|
||||
}
|
||||
|
||||
void HotPixelFinderGPU::Add(const int32_t *device_image, Frame &frame) {
|
||||
if (nrings == 0)
|
||||
return;
|
||||
if (!frame.count.get()) {
|
||||
@@ -201,44 +295,101 @@ void HotPixelFinderGPU::Statistics(const int32_t *device_image, Frame &frame, st
|
||||
frame.ring_ok = CudaDevicePtr<char>(nrings);
|
||||
}
|
||||
const cudaStream_t stream = *frame.stream;
|
||||
sector_kernel<<<static_cast<unsigned>(nkeys), THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin, frame.count,
|
||||
frame.sector_median);
|
||||
sector_kernel<<<static_cast<unsigned>(nkeys), THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin,
|
||||
frame.count, frame.sector_median);
|
||||
cuda_err(cudaGetLastError());
|
||||
ring_kernel<<<nrings, THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin, frame.count, sectors,
|
||||
frame.ring_median, frame.ring_mad);
|
||||
cuda_err(cudaGetLastError());
|
||||
|
||||
cuda_err(cudaMemcpyAsync(count.data(), frame.count, nkeys * sizeof(uint32_t), cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(sector_median.data(), frame.sector_median, nkeys * sizeof(int32_t),
|
||||
cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(ring_median.data(), frame.ring_median, nrings * sizeof(int32_t),
|
||||
cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(ring_mad.data(), frame.ring_mad, nrings * sizeof(int32_t),
|
||||
cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaStreamSynchronize(stream));
|
||||
}
|
||||
|
||||
void HotPixelFinderGPU::Accumulate(const int32_t *device_image, Frame &frame, const std::vector<int32_t> &level,
|
||||
const std::vector<float> &threshold, const std::vector<char> &ring_ok) {
|
||||
const cudaStream_t stream = *frame.stream;
|
||||
cuda_err(cudaMemcpyAsync(frame.level, level.data(), nkeys * sizeof(int32_t), cudaMemcpyHostToDevice, stream));
|
||||
cuda_err(cudaMemcpyAsync(frame.threshold, threshold.data(), nkeys * sizeof(float), cudaMemcpyHostToDevice,
|
||||
stream));
|
||||
cuda_err(cudaMemcpyAsync(frame.ring_ok, ring_ok.data(), nrings * sizeof(char), cudaMemcpyHostToDevice, stream));
|
||||
levels_kernel<<<static_cast<unsigned>((nkeys + THREADS - 1) / THREADS), THREADS, 0, stream>>>(
|
||||
nkeys, sectors, rules, frame.count, frame.sector_median, frame.ring_median, frame.ring_mad,
|
||||
frame.level, frame.threshold, frame.ring_ok, key_frames, key_level_sum);
|
||||
cuda_err(cudaGetLastError());
|
||||
|
||||
std::lock_guard lock(accumulate_mutex);
|
||||
cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0));
|
||||
accumulate_kernel<<<static_cast<unsigned>((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>(
|
||||
device_image, key, npixels, sectors, frame.level, frame.threshold, frame.ring_ok,
|
||||
n_lit, n_error, sum_value, n_error_ring_ok, error_level_sum);
|
||||
cuda_err(cudaGetLastError());
|
||||
cuda_err(cudaEventRecord(last_accumulate, stream));
|
||||
}
|
||||
|
||||
void HotPixelFinderGPU::ChanceCounts(std::vector<int64_t> &lit, std::vector<int64_t> &seen) {
|
||||
CudaStream stream;
|
||||
cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0));
|
||||
const size_t rings = std::max(nrings, 1);
|
||||
CudaDevicePtr<unsigned long long> d_lit(rings, ALLOC), d_seen(rings, ALLOC);
|
||||
cuda_err(cudaMemsetAsync(d_lit, 0, rings * sizeof(unsigned long long), stream));
|
||||
cuda_err(cudaMemsetAsync(d_seen, 0, rings * sizeof(unsigned long long), stream));
|
||||
chance_kernel<<<static_cast<unsigned>((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>(
|
||||
npixels, sectors, key, key_frames, n_lit, n_error_ring_ok, d_lit, d_seen);
|
||||
cuda_err(cudaGetLastError());
|
||||
lit.resize(nrings);
|
||||
seen.resize(nrings);
|
||||
cuda_err(cudaMemcpyAsync(lit.data(), d_lit, nrings * sizeof(int64_t), cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(seen.data(), d_seen, nrings * sizeof(int64_t), cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaStreamSynchronize(stream));
|
||||
}
|
||||
|
||||
void HotPixelFinderGPU::Download(uint16_t *host_n_lit, uint16_t *host_n_error, int64_t *host_sum_value,
|
||||
uint16_t *host_n_error_ring_ok, int64_t *host_error_level_sum) {
|
||||
cuda_err(cudaMemcpy(host_n_lit, n_lit, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
|
||||
cuda_err(cudaMemcpy(host_n_error, n_error, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
|
||||
cuda_err(cudaMemcpy(host_sum_value, sum_value, npixels * sizeof(int64_t), cudaMemcpyDeviceToHost));
|
||||
cuda_err(cudaMemcpy(host_n_error_ring_ok, n_error_ring_ok, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
|
||||
cuda_err(cudaMemcpy(host_error_level_sum, error_level_sum, npixels * sizeof(int64_t), cudaMemcpyDeviceToHost));
|
||||
HotPixelFinderGPU::Candidates HotPixelFinderGPU::GetCandidates(uint32_t frames, int min_valid, bool spacing_ok,
|
||||
const std::vector<int> &k_chance) {
|
||||
CudaStream stream;
|
||||
cuda_err(cudaStreamWaitEvent(stream, last_accumulate, 0));
|
||||
const unsigned blocks = static_cast<unsigned>((npixels + THREADS - 1) / THREADS);
|
||||
CudaDevicePtr<int> d_k_chance(std::max<size_t>(k_chance.size(), 1), ALLOC);
|
||||
CudaDevicePtr<uint32_t> d_count(1, ALLOC);
|
||||
if (!k_chance.empty())
|
||||
cuda_err(cudaMemcpyAsync(d_k_chance, k_chance.data(), k_chance.size() * sizeof(int), cudaMemcpyHostToDevice,
|
||||
stream));
|
||||
cuda_err(cudaMemsetAsync(d_count, 0, sizeof(uint32_t), stream));
|
||||
candidate_kernel<<<blocks, THREADS, 0, stream>>>(npixels, sectors, frames, min_valid, spacing_ok, key, key_frames,
|
||||
n_lit, n_error, n_error_ring_ok, d_k_chance, d_count, nullptr);
|
||||
cuda_err(cudaGetLastError());
|
||||
uint32_t n = 0;
|
||||
cuda_err(cudaMemcpyAsync(&n, d_count, sizeof(uint32_t), cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaStreamSynchronize(stream));
|
||||
|
||||
Candidates c;
|
||||
c.key_frames.resize(nkeys);
|
||||
c.key_level_sum.resize(nkeys);
|
||||
if (nkeys > 0) {
|
||||
cuda_err(cudaMemcpyAsync(c.key_frames.data(), key_frames, nkeys * sizeof(uint32_t), cudaMemcpyDeviceToHost,
|
||||
stream));
|
||||
cuda_err(cudaMemcpyAsync(c.key_level_sum.data(), key_level_sum, nkeys * sizeof(int64_t),
|
||||
cudaMemcpyDeviceToHost, stream));
|
||||
}
|
||||
if (n > 0) {
|
||||
CudaDevicePtr<uint32_t> index(n, ALLOC);
|
||||
CudaDevicePtr<uint16_t> g_lit(n, ALLOC), g_error(n, ALLOC), g_error_ring_ok(n, ALLOC);
|
||||
CudaDevicePtr<int64_t> g_sum(n, ALLOC), g_error_level(n, ALLOC);
|
||||
cuda_err(cudaMemsetAsync(d_count, 0, sizeof(uint32_t), stream));
|
||||
candidate_kernel<<<blocks, THREADS, 0, stream>>>(npixels, sectors, frames, min_valid, spacing_ok, key,
|
||||
key_frames, n_lit, n_error, n_error_ring_ok, d_k_chance,
|
||||
d_count, index);
|
||||
cuda_err(cudaGetLastError());
|
||||
const unsigned gb = (n + THREADS - 1) / THREADS;
|
||||
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, n_lit.get(), g_lit.get());
|
||||
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, n_error.get(), g_error.get());
|
||||
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, n_error_ring_ok.get(), g_error_ring_ok.get());
|
||||
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, sum_value.get(), g_sum.get());
|
||||
gather_kernel<<<gb, THREADS, 0, stream>>>(n, index, error_level_sum.get(), g_error_level.get());
|
||||
cuda_err(cudaGetLastError());
|
||||
c.index.resize(n);
|
||||
c.n_lit.resize(n);
|
||||
c.n_error.resize(n);
|
||||
c.n_error_ring_ok.resize(n);
|
||||
c.sum_value.resize(n);
|
||||
c.error_level_sum.resize(n);
|
||||
cuda_err(cudaMemcpyAsync(c.index.data(), index, n * sizeof(uint32_t), cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(c.n_lit.data(), g_lit, n * sizeof(uint16_t), cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(c.n_error.data(), g_error, n * sizeof(uint16_t), cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(c.n_error_ring_ok.data(), g_error_ring_ok, n * sizeof(uint16_t),
|
||||
cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(c.sum_value.data(), g_sum, n * sizeof(int64_t), cudaMemcpyDeviceToHost, stream));
|
||||
cuda_err(cudaMemcpyAsync(c.error_level_sum.data(), g_error_level, n * sizeof(int64_t),
|
||||
cudaMemcpyDeviceToHost, stream));
|
||||
}
|
||||
cuda_err(cudaStreamSynchronize(stream));
|
||||
return c;
|
||||
}
|
||||
|
||||
+54
-23
@@ -10,33 +10,53 @@
|
||||
|
||||
#include "../image_analysis/indexing/CUDAMemHelpers.h"
|
||||
|
||||
// The device half of HotPixelFinder, for frames already preprocessed on the GPU: the per-frame order
|
||||
// statistics (each ring-sector's median, each ring's median and median absolute deviation) and the
|
||||
// per-pixel sums run where the image already is, and only the per-key statistics - tens of thousands
|
||||
// of numbers - come to the host, which turns them into levels and thresholds with the very code the
|
||||
// host path uses. Every statistic is an exact order statistic of integers and every sum an integer,
|
||||
// so the sums, and with them the mask, are identical to what HotPixelFinder::AddImage produces.
|
||||
// What a frame's levels and lit thresholds are made of (HotPixelFinder's constants), handed to the
|
||||
// device so the two halves cannot drift apart.
|
||||
struct HotPixelLevelRules {
|
||||
int min_sector_pixels;
|
||||
int min_ring_pixels;
|
||||
float lit_nsigma;
|
||||
float lit_offset;
|
||||
};
|
||||
|
||||
// The device half of HotPixelFinder, for frames already preprocessed on the GPU. Everything a frame adds
|
||||
// stays on the device: the per-frame order statistics (each ring-sector's median, each ring's median
|
||||
// and median absolute deviation), the levels and thresholds made from them, and the per-pixel and
|
||||
// per-key sums. Every statistic is an exact order statistic of integers, the threshold is computed
|
||||
// with the same rounding steps as the host's (see HotPixelFinder::AddLevels) and every sum is an
|
||||
// integer, so the sums are identical to what HotPixelFinder::AddImage produces. When the mask is read
|
||||
// only what can decide it comes back: per-ring counts for the chance rate, then the few pixels that can
|
||||
// be persistent or carry the error value.
|
||||
class HotPixelFinderGPU {
|
||||
const size_t npixels;
|
||||
const size_t nkeys;
|
||||
const int nrings;
|
||||
const int sectors;
|
||||
const HotPixelLevelRules rules;
|
||||
|
||||
CudaDevicePtr<int32_t> key; // ring * sectors + sector of each pixel, -1 masked
|
||||
CudaDevicePtr<uint32_t> pixels_by_key; // the unmasked pixels, grouped by key
|
||||
CudaDevicePtr<uint32_t> pixels_by_key; // the unmasked pixels, grouped by key, in pixel order
|
||||
CudaDevicePtr<uint32_t> key_begin; // where each key's pixels start in pixels_by_key
|
||||
|
||||
// The per-pixel sums, exactly those of HotPixelFinder.
|
||||
// The per-pixel and per-key sums, exactly those of HotPixelFinder.
|
||||
CudaDevicePtr<uint16_t> n_lit, n_error, n_error_ring_ok;
|
||||
CudaDevicePtr<int64_t> sum_value, error_level_sum;
|
||||
// Frames are selected on their workers' streams in parallel, but each pixel's sums are plain
|
||||
// read-modify-writes, so one frame at a time adds to them.
|
||||
CudaDevicePtr<uint32_t> key_frames;
|
||||
CudaDevicePtr<int64_t> key_level_sum;
|
||||
|
||||
// Each pixel's sums are plain read-modify-writes, so the frames add to them one after another:
|
||||
// every accumulation waits on the device for the one before it (an event, not the host).
|
||||
std::mutex accumulate_mutex;
|
||||
cudaEvent_t last_accumulate = nullptr;
|
||||
|
||||
public:
|
||||
// One worker's buffers, on the stream its frames are preprocessed on.
|
||||
struct Frame {
|
||||
explicit Frame(std::shared_ptr<CudaStream> stream) : stream(std::move(stream)) {}
|
||||
// Its frames are queued, not waited for (Add), so the buffers below must not be freed before
|
||||
// the stream has used them.
|
||||
~Frame() { if (stream) cudaStreamSynchronize(*stream); }
|
||||
Frame(Frame &&) = default;
|
||||
std::shared_ptr<CudaStream> stream;
|
||||
CudaDevicePtr<uint32_t> count; // valid pixels per key
|
||||
CudaDevicePtr<int32_t> sector_median; // per key
|
||||
@@ -46,21 +66,32 @@ public:
|
||||
CudaDevicePtr<char> ring_ok;
|
||||
};
|
||||
|
||||
// The keys of HotPixelFinder: one per pixel, and where each key's pixels start among the unmasked
|
||||
// pixels sorted by key.
|
||||
HotPixelFinderGPU(const int32_t *key, size_t npixels, const std::vector<uint32_t> &key_begin, int nrings,
|
||||
int sectors);
|
||||
int sectors, const HotPixelLevelRules &rules);
|
||||
~HotPixelFinderGPU();
|
||||
HotPixelFinderGPU(const HotPixelFinderGPU &) = delete;
|
||||
HotPixelFinderGPU &operator=(const HotPixelFinderGPU &) = delete;
|
||||
|
||||
// The lower median of the valid values of each key (count[k] of them, 0 where there are none), and
|
||||
// of each ring the median and the lower median of the absolute deviations from it.
|
||||
void Statistics(const int32_t *device_image, Frame &frame, std::vector<uint32_t> &count,
|
||||
std::vector<int32_t> §or_median, std::vector<int32_t> &ring_median,
|
||||
std::vector<int32_t> &ring_mad);
|
||||
// Add one frame, preprocessed on frame.stream, as HotPixelFinder::AddImage does. Queued on that
|
||||
// stream; nothing is waited for on the host.
|
||||
void Add(const int32_t *device_image, Frame &frame);
|
||||
|
||||
// Add the frame to the per-pixel sums, with each key's level and lit threshold and each ring's
|
||||
// verdict on whether it has a level at all - as HotPixelFinder::AddImage does.
|
||||
void Accumulate(const int32_t *device_image, Frame &frame, const std::vector<int32_t> &level,
|
||||
const std::vector<float> &threshold, const std::vector<char> &ring_ok);
|
||||
// Per ring, over the pixels lit on no more than half of their valid frames: the lit frames and the
|
||||
// valid frames, summed (HotPixelFinder::GetMask's chance rate). Waits for every frame added.
|
||||
void ChanceCounts(std::vector<int64_t> &lit, std::vector<int64_t> &seen);
|
||||
|
||||
// The per-pixel sums, npixels each.
|
||||
void Download(uint16_t *n_lit, uint16_t *n_error, int64_t *sum_value, uint16_t *n_error_ring_ok,
|
||||
int64_t *error_level_sum);
|
||||
// The pixels that can be masked: those holding the error value on more than half of `frames`, and,
|
||||
// where spacing_ok, those lit on at least min(max(2, k_chance[ring]), valid frames) of at least
|
||||
// min_valid valid frames - a persistent pixel is lit on at least that many, because one reflection
|
||||
// explains at least two. For each, its index and per-pixel sums; and every key's sums.
|
||||
struct Candidates {
|
||||
std::vector<uint32_t> index;
|
||||
std::vector<uint16_t> n_lit, n_error, n_error_ring_ok;
|
||||
std::vector<int64_t> sum_value, error_level_sum;
|
||||
std::vector<uint32_t> key_frames;
|
||||
std::vector<int64_t> key_level_sum;
|
||||
};
|
||||
Candidates GetCandidates(uint32_t frames, int min_valid, bool spacing_ok, const std::vector<int> &k_chance);
|
||||
};
|
||||
|
||||
+84
-28
@@ -57,6 +57,7 @@
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
#include "../image_analysis/image_preprocessing/ImagePreprocessorGPU.h"
|
||||
#include "../image_analysis/image_preprocessing/ImagePreprocessorBufferGPU.h"
|
||||
#include "../image_analysis/spot_finding/AdaptiveSpotFinderGPU.h"
|
||||
#endif
|
||||
#include "../image_analysis/scale_merge/Merge.h"
|
||||
#include "../image_analysis/scale_merge/RfreeFlags.h"
|
||||
@@ -244,10 +245,10 @@ namespace {
|
||||
// as signal. The margin is capped at a tenth of the sweep so a short run still has a sample.
|
||||
constexpr int PRESCAN_END_MARGIN_IMAGES = 5;
|
||||
|
||||
// Workers reading the pre-scan sample. Each owns a shard of the beam-stop projection so no two
|
||||
// threads touch the same accumulator, and a shard costs 20 bytes per pixel - 362 MB on a 16M
|
||||
// detector - so this is capped well below the worker count of the run proper. The accumulation
|
||||
// is memory-bound rather than compute-bound, so a handful of workers already saturates it.
|
||||
// Workers reading the pre-scan sample. Each holds detector-sized buffers of its own, and pages of
|
||||
// fresh memory are slow to fault in when many threads do it at once, so this is capped well below
|
||||
// the worker count of the run proper. The beam-stop projection is memory-bound rather than
|
||||
// compute-bound, so a handful of workers already saturates it.
|
||||
constexpr size_t PRESCAN_MAX_WORKERS = 8;
|
||||
|
||||
// The spot width is measured on a GROWING share of the pre-scan sample: every eighth frame of
|
||||
@@ -929,12 +930,32 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
|
||||
struct PreScanWorker {
|
||||
std::vector<uint8_t> decompression_buffer;
|
||||
JFJochReaderRawImage raw_image;
|
||||
std::unique_ptr<ImagePreprocessorCPU> preprocessor;
|
||||
std::unique_ptr<ImagePreprocessor> preprocessor;
|
||||
std::unique_ptr<ImagePreprocessorBuffer> preprocessed;
|
||||
std::unique_ptr<ImageSpotFinder> spot_finder;
|
||||
};
|
||||
const auto make_worker = [&] {
|
||||
PreScanWorker w;
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
// On the card where there is one, as in the image loops: the frame is decoded and preprocessed
|
||||
// there and the spots are found and extracted there. The device finders give the host's spot
|
||||
// list to the bit - integer ring sums, the same connected components in the same order
|
||||
// (AdaptiveSpotFinderGPU, SpotExtractorGPU) - so this is a choice of where, not of what. The
|
||||
// preprocessed image comes back only for the spot width, which reads pixels around the spots.
|
||||
if (want_spots && get_gpu_count() > 0) {
|
||||
auto stream = std::make_shared<CudaStream>();
|
||||
w.preprocessor = std::make_unique<ImagePreprocessorGPU>(prescan_x, prescan_mask, stream,
|
||||
/*copy_image_to_host=*/want_width);
|
||||
w.preprocessed = std::make_unique<ImagePreprocessorBufferGPU>(prescan_x.GetPixelsNum(),
|
||||
/*host_mirror=*/want_width);
|
||||
if (config_.spot_finding.adaptive_threshold)
|
||||
w.spot_finder = std::make_unique<AdaptiveSpotFinderGPU>(*prescan_mapping, stream);
|
||||
else
|
||||
w.spot_finder = std::make_unique<ImageSpotFinderGPU>(prescan_x.GetXPixelsNumConv(),
|
||||
prescan_x.GetYPixelsNumConv(), stream);
|
||||
return w;
|
||||
}
|
||||
#endif
|
||||
if (want_spots) {
|
||||
w.preprocessor = std::make_unique<ImagePreprocessorCPU>(prescan_x, prescan_mask);
|
||||
w.preprocessed = std::make_unique<ImagePreprocessorBuffer>(prescan_x.GetPixelsNum());
|
||||
@@ -956,8 +977,20 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
|
||||
bool for_width, std::vector<spot_width::FluxCurve> &curves,
|
||||
std::vector<float> &spot_q) {
|
||||
try {
|
||||
w.preprocessor->Analyze(*w.preprocessed,
|
||||
image.GetUncompressedPtr(w.decompression_buffer), image.GetMode());
|
||||
// As in the image loops: a frame the device cannot decode goes to the host decoder.
|
||||
ImageStatistics stats;
|
||||
bool decoded_on_device = false;
|
||||
try {
|
||||
decoded_on_device = w.preprocessor->AnalyzeCompressed(*w.preprocessed, image, stats);
|
||||
} catch (const JFJochException &e) {
|
||||
cuda_throw_if_context_lost();
|
||||
logger.Warning("Pre-scan: device decoding of image {} failed ({}), decompressing it on "
|
||||
"the host", image_idx, e.what());
|
||||
cuda_clear_error();
|
||||
}
|
||||
if (!decoded_on_device)
|
||||
w.preprocessor->Analyze(*w.preprocessed,
|
||||
image.GetUncompressedPtr(w.decompression_buffer), image.GetMode());
|
||||
} catch (const std::exception &e) {
|
||||
if (IsFatalResourceError(e)) throw;
|
||||
logger.Warning("Pre-scan: failed to preprocess image {}: {}", image_idx, e.what());
|
||||
@@ -1014,9 +1047,9 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
|
||||
|
||||
// Read the sample on several workers. The reader serialises on the HDF5 lock, but the
|
||||
// decompression, the projection and the spot finding - which is all of the cost on a large
|
||||
// detector - run in parallel. Each worker accumulates into a shard of its own, so nothing is
|
||||
// locked while an image is added, and the per-frame results are stitched together in sample
|
||||
// order below so the beam centre sees the same input however the workers interleaved.
|
||||
// detector - run in parallel. The projection is integer sums, so the order the workers add their
|
||||
// frames in does not reach it, and the per-frame results are stitched together in sample order
|
||||
// below so the beam centre sees the same input however the workers interleaved.
|
||||
//
|
||||
// Two passes over the sample. The first builds the projection, which is all the shadow, the
|
||||
// defective pixels and the beam-centre capture below read; the second finds the spots, for the
|
||||
@@ -1026,13 +1059,12 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
|
||||
const std::vector<int> ordinals(sample.begin(), sample.end());
|
||||
const size_t nworkers = std::min<size_t>(std::max<size_t>(config_.nthreads, 1),
|
||||
std::min(PRESCAN_MAX_WORKERS, ordinals.size()));
|
||||
finder.SetShardCount(nworkers);
|
||||
{
|
||||
std::atomic<size_t> next{0};
|
||||
std::vector<std::future<void>> futures;
|
||||
futures.reserve(nworkers);
|
||||
for (size_t t = 0; t < nworkers; t++)
|
||||
futures.emplace_back(std::async(std::launch::async, [&, t] {
|
||||
futures.emplace_back(std::async(std::launch::async, [&] {
|
||||
std::vector<uint8_t> shadow_buffer;
|
||||
JFJochReaderRawImage raw_image;
|
||||
for (size_t i = next.fetch_add(1); i < ordinals.size(); i = next.fetch_add(1)) {
|
||||
@@ -1057,7 +1089,7 @@ void Rugnux::PreScan(int start_image, int images_to_process, int frame_count, Ru
|
||||
msg.image = raw_image.image;
|
||||
msg.number = ordinal;
|
||||
msg.original_number = image_idx;
|
||||
finder.AddImage(msg, shadow_buffer, t);
|
||||
finder.AddImage(msg, shadow_buffer);
|
||||
}
|
||||
}
|
||||
}));
|
||||
@@ -1670,6 +1702,11 @@ void Rugnux::MaskDefectivePixels(int start_image, const std::vector<int> &sample
|
||||
const size_t nthreads = static_cast<size_t>(std::max(config_.nthreads, 1));
|
||||
HotPixelFinder finder(about, pixel_mask_, nthreads);
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
// Built here, once, so the workers' first frames do not queue behind it.
|
||||
if (get_gpu_count() > 0)
|
||||
finder.PrepareDevice();
|
||||
#endif
|
||||
std::atomic<size_t> next{0};
|
||||
std::vector<std::future<void>> futures;
|
||||
const size_t nworkers = std::min<size_t>(nthreads, std::min(PRESCAN_MAX_WORKERS, sample.size()));
|
||||
@@ -8389,6 +8426,27 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
|
||||
|
||||
const auto &twin_sg_opt = experiment_.GetGemmiSpaceGroup();
|
||||
const gemmi::SpaceGroup *twin_sg = twin_sg_opt ? &*twin_sg_opt : nullptr;
|
||||
|
||||
// Diffraction anisotropy (see where it is reported, below), made beside the analyses that come
|
||||
// before it: it reads the merge, the integrated observations and the per-frame scales the merge
|
||||
// wrote back, none of which they change, and most of it is gathering the observations.
|
||||
std::future<decltype(sm.statistics.anisotropy)> anisotropy;
|
||||
if (!geometry_prepass && !superseded && result.consensus_cell) {
|
||||
AnisotropyRunInfo aniso_run;
|
||||
if (sm.statistics.sweep_quality.measured && sm.statistics.sweep_quality.sweep_deg > 0.0f)
|
||||
aniso_run.observed_rotation_deg = sm.statistics.sweep_quality.sweep_deg;
|
||||
aniso_run.dose_term_in_scale_model = experiment_.GetScalingSettings().GetCorrectionSurfaces();
|
||||
aniso_run.radiation_damage_relative_b = sm.statistics.radiation_damage_delta_b;
|
||||
const float wedge_deg = experiment_.GetGoniometer() ? experiment_.GetGoniometer()->GetWedge_deg() : 0.0f;
|
||||
anisotropy = std::async(std::launch::async,
|
||||
[&, aniso_run, wedge_deg, cell = *result.consensus_cell,
|
||||
rotation = experiment_.IsRotationIndexing()] {
|
||||
return AnalyzeAnisotropy(sm.merged,
|
||||
ScaledObservations(indexer->GetIntegrationOutcome(), rotation, twin_sg,
|
||||
wedge_deg, 0.5, config_.nthreads),
|
||||
cell, twin_sg, aniso_run);
|
||||
});
|
||||
}
|
||||
// Not on the geometry pre-pass, nor on a superseded one: the analysis goes into that pass's
|
||||
// statistics text and its written reflections, and neither survives the run. The promotion flag
|
||||
// below is a different thing - it is what the SEARCH did, the second pass reads it, and it is
|
||||
@@ -8619,21 +8677,8 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
|
||||
// unmerged observations, because a merge has exact Laue symmetry by construction and the
|
||||
// tensor directions the symmetry forbids - the only place a dataset measures its own
|
||||
// systematic error - are identically zero in it.
|
||||
if (result.consensus_cell) {
|
||||
AnisotropyRunInfo aniso_run;
|
||||
if (sm.statistics.sweep_quality.measured && sm.statistics.sweep_quality.sweep_deg > 0.0f)
|
||||
aniso_run.observed_rotation_deg = sm.statistics.sweep_quality.sweep_deg;
|
||||
aniso_run.dose_term_in_scale_model =
|
||||
experiment_.GetScalingSettings().GetCorrectionSurfaces();
|
||||
aniso_run.radiation_damage_relative_b = sm.statistics.radiation_damage_delta_b;
|
||||
sm.statistics.anisotropy = AnalyzeAnisotropy(
|
||||
sm.merged,
|
||||
ScaledObservations(indexer->GetIntegrationOutcome(),
|
||||
experiment_.IsRotationIndexing(), twin_sg,
|
||||
experiment_.GetGoniometer()
|
||||
? experiment_.GetGoniometer()->GetWedge_deg() : 0.0f,
|
||||
0.5, config_.nthreads),
|
||||
*result.consensus_cell, twin_sg, aniso_run);
|
||||
if (anisotropy.valid()) {
|
||||
sm.statistics.anisotropy = anisotropy.get();
|
||||
stats_text << AnisotropyToText(sm.statistics.anisotropy) << "\n";
|
||||
}
|
||||
}
|
||||
@@ -8869,8 +8914,10 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
|
||||
Logger held = Logger::Buffered();
|
||||
auto pending = std::async(std::launch::async, validate, std::ref(held));
|
||||
experiment_.SpaceGroupNumber(1);
|
||||
rsm->SetWriteBackPerFrameScale(false);
|
||||
p1_merged_early = rsm->Run(/*for_search=*/false, /*full_stats=*/true,
|
||||
/*measure_cc_before_corrections=*/false);
|
||||
rsm->SetWriteBackPerFrameScale(true);
|
||||
experiment_.SetSpaceGroup(data_sg);
|
||||
validation = pending.get();
|
||||
held.ReplayInto(logger);
|
||||
@@ -9046,6 +9093,9 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
|
||||
const double em_a = result.error_model_a;
|
||||
const double em_b = result.error_model_b;
|
||||
const auto res_fit = result.resolution_fit_A;
|
||||
const int iter_partials = result.scaling_iterations_partials;
|
||||
const int iter_fulls = result.scaling_iterations_fulls;
|
||||
const bool converged = result.scaling_converged;
|
||||
// The unmerged MTZ below does not depend on this merge, so it is built meanwhile, from
|
||||
// the experiment as it stands in the determined group. The merge rewrites each image's
|
||||
// mosaicity, which the file's batch headers carry, so that is filled in only after it.
|
||||
@@ -9062,7 +9112,10 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
|
||||
// Both the merge and the MTZ read the group from the experiment, so it is set for
|
||||
// the whole of it and restored after.
|
||||
experiment_.SpaceGroupNumber(1);
|
||||
if (!p1_merged_early)
|
||||
rsm->SetWriteBackPerFrameScale(false);
|
||||
auto p1 = scale_and_merge("P1 cross-check", false, false, std::move(p1_merged_early));
|
||||
rsm->SetWriteBackPerFrameScale(true);
|
||||
// The scaler still holds the observations in the indexing they were merged in, so every
|
||||
// relabelling since (the written setting, the model's indexing) is applied to this merge
|
||||
// too - or it would describe the dataset on other axes than the merged output beside it,
|
||||
@@ -9135,6 +9188,9 @@ ProcessResult Rugnux::RunPipeline(RugnuxObserver *observer, bool write_output, b
|
||||
result.error_model_a = em_a;
|
||||
result.error_model_b = em_b;
|
||||
result.resolution_fit_A = res_fit;
|
||||
result.scaling_iterations_partials = iter_partials;
|
||||
result.scaling_iterations_fulls = iter_fulls;
|
||||
result.scaling_converged = converged;
|
||||
if (determined != nullptr && determined->number > 1)
|
||||
logger.Info("P1 cross-check dataset written to {} ({} unique reflections): the "
|
||||
"same observations merged in P1 instead of {}, so a wrong space group "
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
|
||||
#include "../common/AzimuthalIntegrationMapping.h"
|
||||
#include "../common/AzimuthalIntegrationProfile.h"
|
||||
#include "../image_analysis/azint/AzIntEngineCPU.h"
|
||||
#include "../image_analysis/azint/AzIntEngineGPU.h"
|
||||
#include "../image_analysis/spot_finding/AdaptiveSpotFinderCPU.h"
|
||||
#include "../image_analysis/spot_finding/AdaptiveSpotFinderGPU.h"
|
||||
@@ -161,6 +162,46 @@ TEST_CASE("AdaptiveSpotFinderGPU_AzimuthalIntegration", "[AdaptiveSpotFinderGPU]
|
||||
}
|
||||
}
|
||||
|
||||
// The GPU azimuthal integration against the CPU one, on a pixel count that is not a multiple of four (the
|
||||
// kernel reads four pixels at a time and does the rest one by one) and with masked and saturated pixels
|
||||
// in it. The per-ring pixel counts are integers and must agree exactly; the float sums only to rounding.
|
||||
TEST_CASE("AzIntEngineGPU_MatchesCPU", "[AdaptiveSpotFinderGPU]") {
|
||||
if (get_gpu_count() == 0) {
|
||||
WARN("No CUDA GPU present. Skipping AzIntEngineGPU_MatchesCPU");
|
||||
return;
|
||||
}
|
||||
|
||||
DiffractionExperiment x(DetDECTRIS(1031, 1063, "Test", {}));
|
||||
x.DetectorDistance_mm(80).BeamX_pxl(515).BeamY_pxl(530);
|
||||
x.QSpacingForAzimInt_recipA(0.05).QRangeForAzimInt_recipA(0.05, 5.0);
|
||||
REQUIRE(x.GetPixelsNum() % 4 != 0);
|
||||
PixelMask pixel_mask(x);
|
||||
AzimuthalIntegrationMapping mapping(x, pixel_mask);
|
||||
|
||||
ImagePreprocessorBufferGPU buffer(x.GetPixelsNum());
|
||||
for (size_t i = 0; i < x.GetPixelsNum(); i++)
|
||||
buffer[i] = (i % 997 == 0) ? INT32_MIN : (i % 1009 == 0) ? INT32_MAX
|
||||
: 8 + static_cast<int32_t>((i * 7919) % 23);
|
||||
REQUIRE(cudaMemcpy(buffer.getGPUBuffer(), buffer.getBuffer().data(),
|
||||
x.GetPixelsNum() * sizeof(int32_t), cudaMemcpyHostToDevice) == cudaSuccess);
|
||||
REQUIRE(cudaDeviceSynchronize() == cudaSuccess);
|
||||
|
||||
AzimuthalIntegrationProfile cpu_profile(mapping), gpu_profile(mapping);
|
||||
AzIntEngineCPU(mapping).Run(buffer, cpu_profile);
|
||||
AzIntEngineGPU(mapping, std::make_shared<CudaStream>()).Run(buffer, gpu_profile);
|
||||
|
||||
REQUIRE(gpu_profile.GetPixelCount() == cpu_profile.GetPixelCount());
|
||||
const auto ref = cpu_profile.GetResult();
|
||||
const auto got = gpu_profile.GetResult();
|
||||
REQUIRE(ref.size() == got.size());
|
||||
for (size_t b = 0; b < ref.size(); b++) {
|
||||
if (std::isnan(ref[b]))
|
||||
CHECK(std::isnan(got[b]));
|
||||
else
|
||||
CHECK(got[b] == Catch::Approx(ref[b]).epsilon(1e-5));
|
||||
}
|
||||
}
|
||||
|
||||
// The ring sums are built by atomics, which arrive in an arbitrary order, so the same frame has to be
|
||||
// re-run to show the engine agrees with itself: detection is a hard "value >= threshold" on integer
|
||||
// counts, and a threshold that wobbles between runs flips pixels on the boundary and with them the size
|
||||
|
||||
@@ -307,3 +307,29 @@ TEST_CASE("FindBeamCenter_BlanksTheBeamStopOutOfTheCapture", "[BeamCenter]") {
|
||||
CHECK(masked_error < 12.0f);
|
||||
CHECK(masked_error <= unmasked_error);
|
||||
}
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
#include "../common/CUDAWrapper.h"
|
||||
|
||||
// With a GPU the passes over the pixels run on it. The cell a pixel lands in is computed from IEEE
|
||||
// operations alone and without contraction on both sides, and each cell is summed in the host's order,
|
||||
// so the walk ends on the same bits - a tilted detector and an offset centre included.
|
||||
TEST_CASE("BeamCenterFromBackground_DeviceMatchesHost", "[BeamCenter]") {
|
||||
if (get_gpu_count() == 0)
|
||||
SKIP("no GPU");
|
||||
DiffractionExperiment x = TestExperiment();
|
||||
x.PoniRot1_rad(0.005f).PoniRot2_rad(-0.003f);
|
||||
PixelMask pixel_mask(x);
|
||||
|
||||
const DiffractionGeometry geom_true = OffsetBy(x.GetDiffractionGeometry(), 7.0f, -4.5f);
|
||||
const auto projection = SynthesiseProjection(x, pixel_mask, geom_true, 60.0f, 0.5f);
|
||||
const auto host = FindBeamCenterFromBackground(x, pixel_mask, projection, 0, {}, /*allow_device=*/false);
|
||||
const auto device = FindBeamCenterFromBackground(x, pixel_mask, projection, 0, {}, /*allow_device=*/true);
|
||||
|
||||
REQUIRE(host.has_value());
|
||||
REQUIRE(device.has_value());
|
||||
CHECK(device->beam_x_pxl == host->beam_x_pxl);
|
||||
CHECK(device->beam_y_pxl == host->beam_y_pxl);
|
||||
CHECK(device->sigma_pxl == host->sigma_pxl);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -112,6 +112,8 @@ ADD_EXECUTABLE(jfjoch_test
|
||||
RingsFromProfileTest.cpp
|
||||
CalibrationTest.cpp
|
||||
XtalOptimizerTest.cpp
|
||||
XtalRefineTest.cpp
|
||||
XtalRefineCeres.h
|
||||
CrystalLatticeTest.cpp
|
||||
FPGAPTPTest.cpp
|
||||
ResolutionShellsTest.cpp
|
||||
@@ -146,6 +148,7 @@ ADD_EXECUTABLE(jfjoch_test
|
||||
AnisotropyAnalysisTest.cpp
|
||||
ModelScalingTest.cpp
|
||||
ModelScaleGPUTest.cpp
|
||||
CorrectionSurfaceGPUTest.cpp
|
||||
TwinningAnalysisTest.cpp
|
||||
TranslationalNCSTest.cpp
|
||||
RfreeFlagsTest.cpp
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#include <catch2/catch_all.hpp>
|
||||
#include "../common/CUDAWrapper.h"
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <random>
|
||||
#include <vector>
|
||||
|
||||
#include "../common/ParallelFor.h"
|
||||
#include "../image_analysis/scale_merge/RotationScaleMergeGPU.h"
|
||||
|
||||
namespace {
|
||||
|
||||
using Term = RotationScaleMergeGPU::SurfaceTerm;
|
||||
|
||||
// The roundings RotationScaleMerge::ApplyCellSurface makes on the host (x86-64-v3 build), spelled out:
|
||||
// a volatile result is rounded on its own and never fused into the next operation, std::fma is fused.
|
||||
double Mul(double a, double b) { volatile double r = a * b; return r; }
|
||||
double Add(double a, double b) { volatile double r = a + b; return r; }
|
||||
|
||||
constexpr int SURFACE_BLOCK = 32768; // ApplyCellSurface's reduction block
|
||||
|
||||
struct Surface {
|
||||
int n_groups = 0, ncell = 0;
|
||||
std::vector<Term> term;
|
||||
std::vector<uint8_t> parity;
|
||||
std::vector<int32_t> gperm, gstart;
|
||||
std::vector<int32_t> sel[3]; // even, odd, all - in term order
|
||||
};
|
||||
|
||||
Surface MakeSurface(int n_terms, int n_groups, int ncell, uint32_t seed) {
|
||||
std::mt19937 rng(seed);
|
||||
std::uniform_int_distribution<int> group(0, n_groups - 1), cell(0, ncell - 1), bit(0, 1);
|
||||
std::uniform_real_distribution<float> u(0.0f, 1.0f);
|
||||
Surface s;
|
||||
s.n_groups = n_groups;
|
||||
s.ncell = ncell;
|
||||
for (int i = 0; i < n_terms; ++i) {
|
||||
// Negative intensities, and now and then a sigma of zero: both reach the host sums as they are.
|
||||
const float I = 1000.0f * u(rng) - 100.0f;
|
||||
const float sigma = (i % 997 == 0) ? 0.0f : 1.0f + 30.0f * u(rng);
|
||||
// Every 50th group gets no terms at all, so its reference is empty.
|
||||
int g = group(rng);
|
||||
if (g % 50 == 0) g = (g + 1) % n_groups;
|
||||
s.term.push_back({I, sigma, 0.5f + u(rng), 1.0f + 3.0f * u(rng), cell(rng), g});
|
||||
s.parity.push_back(static_cast<uint8_t>(bit(rng)));
|
||||
s.sel[s.parity.back()].push_back(i);
|
||||
s.sel[2].push_back(i);
|
||||
}
|
||||
s.gstart.assign(n_groups + 1, 0);
|
||||
for (const Term &t : s.term) ++s.gstart[t.group + 1];
|
||||
for (int g = 0; g < n_groups; ++g) s.gstart[g + 1] += s.gstart[g];
|
||||
s.gperm.resize(n_terms);
|
||||
std::vector<int32_t> fill(s.gstart.begin(), s.gstart.end() - 1);
|
||||
for (int i = 0; i < n_terms; ++i) s.gperm[fill[s.term[i].group]++] = i;
|
||||
return s;
|
||||
}
|
||||
|
||||
void HostReference(const Surface &s, int parity, const std::vector<double> &A,
|
||||
std::vector<double> &sw, std::vector<double> &swI) {
|
||||
sw.assign(s.n_groups, 0.0);
|
||||
swI.assign(s.n_groups, 0.0);
|
||||
for (int g = 0; g < s.n_groups; ++g) {
|
||||
double s_w = 0.0, s_wI = 0.0;
|
||||
for (int k = s.gstart[g]; k < s.gstart[g + 1]; ++k) {
|
||||
const int i = s.gperm[k];
|
||||
if (parity >= 0 && s.parity[i] != parity) continue;
|
||||
const Term &t = s.term[i];
|
||||
const double a = A[t.cell];
|
||||
const double Is = Mul(Mul(t.I, t.corr), a), sc = Mul(Mul(t.sigma, t.corr), a);
|
||||
const double w = 1.0 / Mul(sc, sc);
|
||||
s_w = Add(s_w, w);
|
||||
s_wI = parity >= 0 ? std::fma(Is, w, s_wI) : Add(s_wI, Mul(Is, w));
|
||||
}
|
||||
sw[g] = s_w;
|
||||
swI[g] = s_wI;
|
||||
}
|
||||
}
|
||||
|
||||
void HostFitSums(const Surface &s, const std::vector<int32_t> &sel, const std::vector<double> &A,
|
||||
const std::vector<double> &sw, const std::vector<double> &swI,
|
||||
std::vector<double> &cross, std::vector<double> &ref2) {
|
||||
cross.assign(s.ncell, 0.0);
|
||||
ref2.assign(s.ncell, 0.0);
|
||||
const int n = static_cast<int>(sel.size()), nb = ReductionBlocks(n, SURFACE_BLOCK);
|
||||
for (int b = 0; b < nb; ++b) {
|
||||
std::vector<double> xcross(s.ncell, 0.0), xref2(s.ncell, 0.0);
|
||||
const int lo = static_cast<int>(int64_t(n) * b / nb), hi = static_cast<int>(int64_t(n) * (b + 1) / nb);
|
||||
for (int k = lo; k < hi; ++k) {
|
||||
const Term &t = s.term[sel[k]];
|
||||
if (sw[t.group] <= 0.0) continue;
|
||||
const double Iref = swI[t.group] / sw[t.group], a = A[t.cell];
|
||||
const double Is = Mul(Mul(t.I, t.corr), a), sc = Mul(Mul(t.sigma, t.corr), a);
|
||||
if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue;
|
||||
const double w = 1.0 / Mul(sc, sc);
|
||||
xcross[t.cell] = std::fma(Mul(w, Is), Iref, xcross[t.cell]);
|
||||
xref2[t.cell] = std::fma(Mul(w, Iref), Iref, xref2[t.cell]);
|
||||
}
|
||||
for (int c = 0; c < s.ncell; ++c) {
|
||||
cross[c] = Add(cross[c], xcross[c]);
|
||||
ref2[c] = Add(ref2[c], xref2[c]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The device side of one subset: ApplyCellSurface's per-block counting sort by cell.
|
||||
void UploadSubset(RotationScaleMergeGPU &gpu, const Surface &s, int id, const std::vector<int32_t> &sel) {
|
||||
const int n = static_cast<int>(sel.size()), nb = ReductionBlocks(n, SURFACE_BLOCK);
|
||||
std::vector<int32_t> perm(n), seg_start(static_cast<size_t>(nb) * s.ncell + 1, n);
|
||||
for (int b = 0; b < nb; ++b) {
|
||||
const int lo = static_cast<int>(int64_t(n) * b / nb), hi = static_cast<int>(int64_t(n) * (b + 1) / nb);
|
||||
std::vector<int32_t> pos(s.ncell + 1, 0);
|
||||
for (int k = lo; k < hi; ++k) ++pos[s.term[sel[k]].cell + 1];
|
||||
for (int c = 0; c < s.ncell; ++c) pos[c + 1] += pos[c];
|
||||
for (int c = 0; c < s.ncell; ++c) seg_start[size_t(b) * s.ncell + c] = lo + pos[c];
|
||||
for (int k = lo; k < hi; ++k) perm[lo + pos[s.term[sel[k]].cell]++] = sel[k];
|
||||
}
|
||||
gpu.SurfaceSetSubset(id, nb, perm.data(), seg_start.data());
|
||||
}
|
||||
|
||||
bool SameBits(const std::vector<double> &a, const std::vector<double> &b) {
|
||||
return a.size() == b.size() && std::memcmp(a.data(), b.data(), a.size() * sizeof(double)) == 0;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("CorrectionSurfaceGPU_SumsMatchHostBitForBit", "[RotationScale][gpu]") {
|
||||
if (get_gpu_count() == 0)
|
||||
SKIP("No GPU");
|
||||
// Enough terms for several reduction blocks in every subset.
|
||||
const Surface s = MakeSurface(250000, 4000, 144, 7);
|
||||
RotationScaleMergeGPU gpu;
|
||||
REQUIRE(gpu.Available());
|
||||
gpu.SurfaceSetTerms(static_cast<int>(s.term.size()), s.term.data(), s.parity.data(), s.n_groups,
|
||||
s.gperm.data(), s.gstart.data(), s.ncell);
|
||||
for (int id = 0; id < 3; ++id)
|
||||
UploadSubset(gpu, s, id, s.sel[id]);
|
||||
|
||||
std::mt19937 rng(11);
|
||||
std::uniform_real_distribution<double> u(0.7, 1.4);
|
||||
std::vector<double> A(s.ncell);
|
||||
for (double &a : A) a = u(rng);
|
||||
|
||||
for (int parity : {0, 1, -1}) {
|
||||
const int id = parity < 0 ? 2 : parity;
|
||||
std::vector<double> sw, swI, cross, ref2;
|
||||
HostReference(s, parity, A, sw, swI);
|
||||
HostFitSums(s, s.sel[id], A, sw, swI, cross, ref2);
|
||||
REQUIRE(ReductionBlocks(static_cast<int>(s.sel[id].size()), SURFACE_BLOCK) > 1);
|
||||
|
||||
gpu.SurfaceReference(parity, A.data());
|
||||
std::vector<double> dsw(s.n_groups), dswI(s.n_groups), dcross(s.ncell), dref2(s.ncell);
|
||||
gpu.SurfaceGetReference(dsw.data(), dswI.data());
|
||||
gpu.SurfaceFitSums(id, dcross.data(), dref2.data());
|
||||
CHECK(SameBits(sw, dsw));
|
||||
CHECK(SameBits(swI, dswI));
|
||||
CHECK(SameBits(cross, dcross));
|
||||
CHECK(SameBits(ref2, dref2));
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
+112
-14
@@ -8,6 +8,7 @@
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <optional>
|
||||
#include <thread>
|
||||
#include <vector>
|
||||
|
||||
#include "../common/DetectorSetup.h"
|
||||
@@ -168,16 +169,15 @@ TEST_CASE("ShadowFinder_MaskDoesNotDependOnTheThreadCount", "[ShadowFinder]") {
|
||||
CHECK(finder.GetMask(8) == one);
|
||||
}
|
||||
|
||||
// Workers accumulate into shards of their own and the shards are summed when the projection is read,
|
||||
// so which worker saw which frame must not reach the answer - including the maximum, which only one
|
||||
// shard holds when the reflection is on a single frame.
|
||||
TEST_CASE("ShadowFinder_ShardingDoesNotChangeTheProjection", "[ShadowFinder]") {
|
||||
// Workers add their frames concurrently, each starting at a different band of the projection, so
|
||||
// which worker added which frame, and in what order, must not reach the answer - including the
|
||||
// maximum, which one frame alone holds when the reflection is on a single frame.
|
||||
TEST_CASE("ShadowFinder_ConcurrentWorkersDoNotChangeTheProjection", "[ShadowFinder]") {
|
||||
const DiffractionExperiment x = TestExperiment();
|
||||
const PixelMask pixel_mask(x);
|
||||
|
||||
ShadowFinder serial(x, pixel_mask);
|
||||
ShadowFinder sharded(x, pixel_mask);
|
||||
sharded.SetShardCount(4);
|
||||
ShadowFinder concurrent(x, pixel_mask);
|
||||
|
||||
std::vector<std::vector<int32_t>> frames;
|
||||
std::vector<uint8_t> buffer;
|
||||
@@ -185,22 +185,32 @@ TEST_CASE("ShadowFinder_ShardingDoesNotChangeTheProjection", "[ShadowFinder]") {
|
||||
frames.push_back(Scene(/*cross=*/false, /*reflection=*/f == 0));
|
||||
DataMessage msg{};
|
||||
msg.image = CompressedImage(frames.back(), W, H);
|
||||
serial.AddImage(msg, buffer, 0);
|
||||
sharded.AddImage(msg, buffer, static_cast<size_t>(f) % 4);
|
||||
serial.AddImage(msg, buffer);
|
||||
}
|
||||
std::vector<std::thread> workers;
|
||||
for (int t = 0; t < 4; t++)
|
||||
workers.emplace_back([&, t] {
|
||||
std::vector<uint8_t> worker_buffer;
|
||||
for (int f = NFRAMES - 1 - t; f >= 0; f -= 4) {
|
||||
DataMessage msg{};
|
||||
msg.image = CompressedImage(frames[f], W, H);
|
||||
concurrent.AddImage(msg, worker_buffer);
|
||||
}
|
||||
});
|
||||
for (auto &w : workers) w.join();
|
||||
|
||||
CHECK(serial.GetFrameCount() == sharded.GetFrameCount());
|
||||
CHECK(serial.GetFrameCount() == concurrent.GetFrameCount());
|
||||
|
||||
const auto a = serial.GetMeanProjection();
|
||||
const auto b = sharded.GetMeanProjection();
|
||||
const auto b = concurrent.GetMeanProjection();
|
||||
REQUIRE(a.size() == b.size());
|
||||
// NAN marks a pixel nothing counted, and NAN != NAN, so compare the bits rather than the values.
|
||||
CHECK(memcmp(a.data(), b.data(), a.size() * sizeof(float)) == 0);
|
||||
|
||||
// The reflection is on one frame, so its maximum lives in a single shard. If the fold lost it,
|
||||
// the mask would swallow the reflection instead of giving it back.
|
||||
CHECK(serial.GetMask(1) == sharded.GetMask(1));
|
||||
CHECK(sharded.GetMask(1)[I(C - 14, C - 2)] == 0);
|
||||
// The reflection is on one frame, so only that frame's maximum sees it. If it were lost, the mask
|
||||
// would swallow the reflection instead of giving it back.
|
||||
CHECK(serial.GetMask(1) == concurrent.GetMask(1));
|
||||
CHECK(concurrent.GetMask(1)[I(C - 14, C - 2)] == 0);
|
||||
}
|
||||
|
||||
// Four opaque arms and a centred disk: the scene is invariant under a quarter turn, so the mask must
|
||||
@@ -461,3 +471,91 @@ TEST_CASE("ShadowFinder_ABrightRingIsNotABeamStop", "[ShadowFinder]") {
|
||||
// Anything much beyond the stop, its arm and their penumbra means the walk ran away.
|
||||
CHECK(std::count(mask.begin(), mask.end(), 1u) < 6000);
|
||||
}
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
#include "../common/CUDAWrapper.h"
|
||||
#include "../compression/JFJochCompressor.h"
|
||||
|
||||
namespace {
|
||||
// The mask and the mean projection of the same frames, once from the host projection (the frames
|
||||
// handed over uncompressed) and once from the device's (handed over as bitshuffle+LZ4, which is
|
||||
// what sends them to the GPU).
|
||||
struct HostAndDevice {
|
||||
std::vector<uint32_t> host_mask, device_mask;
|
||||
std::vector<float> host_mean, device_mean;
|
||||
};
|
||||
|
||||
HostAndDevice MaskBothWays(const DiffractionExperiment &x, const std::vector<std::vector<int32_t>> &frames) {
|
||||
const PixelMask pixel_mask(x);
|
||||
HostAndDevice out;
|
||||
std::vector<uint8_t> buffer;
|
||||
{
|
||||
ShadowFinder finder(x, pixel_mask);
|
||||
for (const auto &frame : frames) {
|
||||
DataMessage msg{};
|
||||
msg.image = CompressedImage(frame, W, H);
|
||||
finder.AddImage(msg, buffer);
|
||||
}
|
||||
out.host_mask = finder.GetMask();
|
||||
out.host_mean = finder.GetMeanProjection();
|
||||
}
|
||||
{
|
||||
ShadowFinder finder(x, pixel_mask);
|
||||
JFJochBitShuffleCompressor compressor(CompressionAlgorithm::BSHUF_LZ4);
|
||||
std::vector<std::vector<uint8_t>> compressed;
|
||||
for (const auto &frame : frames) {
|
||||
compressed.push_back(compressor.Compress(frame));
|
||||
DataMessage msg{};
|
||||
msg.image = CompressedImage(compressed.back().data(), compressed.back().size(), W, H,
|
||||
CompressedImageMode::Int32, CompressionAlgorithm::BSHUF_LZ4);
|
||||
finder.AddImage(msg, buffer);
|
||||
}
|
||||
out.device_mask = finder.GetMask();
|
||||
out.device_mean = finder.GetMeanProjection();
|
||||
}
|
||||
return out;
|
||||
}
|
||||
}
|
||||
|
||||
// The mask is made on the GPU wherever the projection is there. On scenes where no pixel sits within a
|
||||
// rounding of a threshold the two must agree to the pixel: every step but the polarization factor,
|
||||
// the Poisson test's logarithm and the arm search's azimuth is exact on both, and these scenes are
|
||||
// built so that none of the three decides anything at an edge.
|
||||
TEST_CASE("ShadowFinder_DeviceMaskMatchesHost", "[ShadowFinder]") {
|
||||
if (get_gpu_count() == 0)
|
||||
SKIP("no GPU");
|
||||
|
||||
SECTION("a beam stop with a reflection behind it") {
|
||||
std::vector<std::vector<int32_t>> frames;
|
||||
for (int f = 0; f < NFRAMES; f++)
|
||||
frames.push_back(Scene(false, f == 0));
|
||||
const auto r = MaskBothWays(TestExperiment(), frames);
|
||||
CHECK(std::count(r.host_mask.begin(), r.host_mask.end(), 1u) == 2612);
|
||||
CHECK(r.device_mask == r.host_mask);
|
||||
CHECK(std::memcmp(r.device_mean.data(), r.host_mean.data(), r.host_mean.size() * sizeof(float)) == 0);
|
||||
}
|
||||
|
||||
SECTION("an arm that lets part of the beam through, across a module gap") {
|
||||
constexpr int32_t BRIGHT = 50;
|
||||
constexpr int ARM_HALF_WIDE = 15, GAP_X0 = 200, GAP_X1 = 216, OPAQUE_FROM_X = 232;
|
||||
std::vector<std::vector<int32_t>> frames;
|
||||
for (int f = 0; f < NFRAMES; f++) {
|
||||
frames.emplace_back(static_cast<size_t>(W) * H, BRIGHT);
|
||||
auto &frame = frames.back();
|
||||
for (int y = 0; y < H; y++)
|
||||
for (int xi = 0; xi < W; xi++) {
|
||||
const int dx = xi - C, dy = y - C;
|
||||
if (dx * dx + dy * dy <= STOP_R * STOP_R)
|
||||
frame[I(xi, y)] = 0;
|
||||
else if (dx >= 0 && std::abs(dy) <= ARM_HALF_WIDE)
|
||||
frame[I(xi, y)] = xi >= OPAQUE_FROM_X ? 0 : BRIGHT * 6 / 10;
|
||||
if (xi >= GAP_X0 && xi <= GAP_X1)
|
||||
frame[I(xi, y)] = INT32_MIN;
|
||||
}
|
||||
}
|
||||
const auto r = MaskBothWays(TestExperiment(), frames);
|
||||
CHECK(std::count(r.host_mask.begin(), r.host_mask.end(), ShadowFinder::TRANSMITTING) > 0);
|
||||
CHECK(r.device_mask == r.host_mask);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
#pragma once
|
||||
|
||||
// The reference the LM solver of XtalRefine is checked against: the same XtalRefineProblem handed to
|
||||
// Ceres exactly as XtalOptimizer used to build it - the same residual functors, losses, priors, bounds,
|
||||
// manifold and options.
|
||||
|
||||
#include "ceres/ceres.h"
|
||||
#include "../image_analysis/geom_refinement/XtalRefine.h"
|
||||
|
||||
struct XtalRefineCeresPrior {
|
||||
XtalRefineCeresPrior(double gx, double gy, double p0, double weight)
|
||||
: gx(gx), gy(gy), p0(p0), weight(weight) {}
|
||||
template<typename T>
|
||||
bool operator()(const T *const p, T *residual) const {
|
||||
residual[0] = T(weight) * (T(gx) * p[0] + T(gy) * p[1] - T(p0));
|
||||
return true;
|
||||
}
|
||||
double gx, gy, p0, weight;
|
||||
};
|
||||
|
||||
inline ceres::Solver::Summary SolveXtalRefineCeres(XtalRefineProblem &p, int num_threads) {
|
||||
std::vector<XtalFrameConstants> frame_const;
|
||||
frame_const.reserve(p.frame_angle_rad.size());
|
||||
if (p.beam_and_orientation_only)
|
||||
for (const double angle: p.frame_angle_rad)
|
||||
frame_const.emplace_back(p.detector_rot, p.rot_vec, angle, p.latt_vec1, p.latt_vec2, p.crystal_system);
|
||||
|
||||
ceres::Problem problem;
|
||||
for (size_t i = 0; i < p.residuals.size(); i++) {
|
||||
ceres::LossFunction *loss = p.weight_sq.empty()
|
||||
? nullptr
|
||||
: new ceres::ScaledLoss(nullptr, p.weight_sq[i], ceres::TAKE_OWNERSHIP);
|
||||
if (p.beam_and_orientation_only)
|
||||
problem.AddResidualBlock(
|
||||
new ceres::AutoDiffCostFunction<XtalResidualBeamOrientation, 3, 2, 3>(
|
||||
new XtalResidualBeamOrientation(p.residuals[i], p.distance_mm, frame_const[p.frame[i]])),
|
||||
loss, p.beam, p.latt_vec0);
|
||||
else
|
||||
problem.AddResidualBlock(
|
||||
new ceres::AutoDiffCostFunction<XtalResidualFixedDistance, 3, 2, 2, 3, 3, 3, 3>(
|
||||
new XtalResidualFixedDistance(p.residuals[i], p.distance_mm)),
|
||||
loss, p.beam, p.detector_rot, p.rot_vec, p.latt_vec0, p.latt_vec1, p.latt_vec2);
|
||||
}
|
||||
for (const auto &prior: p.priors)
|
||||
problem.AddResidualBlock(
|
||||
new ceres::AutoDiffCostFunction<XtalRefineCeresPrior, 1, 2>(
|
||||
new XtalRefineCeresPrior(prior.gx, prior.gy, prior.p0, prior.weight)),
|
||||
nullptr, prior.block == XtalRefinePrior::Block::Beam ? p.beam : p.detector_rot);
|
||||
|
||||
const auto bounds = [&](double *block, const double *lo, const double *hi, int n) {
|
||||
for (int i = 0; i < n; i++) {
|
||||
if (lo[i] > -XtalRefineProblem::kNoBound)
|
||||
problem.SetParameterLowerBound(block, i, lo[i]);
|
||||
if (hi[i] < XtalRefineProblem::kNoBound)
|
||||
problem.SetParameterUpperBound(block, i, hi[i]);
|
||||
}
|
||||
};
|
||||
if (p.beam_constant)
|
||||
problem.SetParameterBlockConstant(p.beam);
|
||||
if (!p.beam_and_orientation_only) {
|
||||
if (p.detector_rot_constant)
|
||||
problem.SetParameterBlockConstant(p.detector_rot);
|
||||
else
|
||||
bounds(p.detector_rot, p.detector_rot_lower, p.detector_rot_upper, 2);
|
||||
if (p.rot_vec_constant)
|
||||
problem.SetParameterBlockConstant(p.rot_vec);
|
||||
else
|
||||
problem.SetManifold(p.rot_vec, new ceres::SphereManifold<3>);
|
||||
if (p.latt_vec1_constant)
|
||||
problem.SetParameterBlockConstant(p.latt_vec1);
|
||||
else
|
||||
bounds(p.latt_vec1, p.latt_vec1_lower, p.latt_vec1_upper, 3);
|
||||
if (p.latt_vec2_constant)
|
||||
problem.SetParameterBlockConstant(p.latt_vec2);
|
||||
else
|
||||
bounds(p.latt_vec2, p.latt_vec2_lower, p.latt_vec2_upper, 3);
|
||||
}
|
||||
|
||||
ceres::Solver::Options options;
|
||||
options.linear_solver_type = ceres::DENSE_NORMAL_CHOLESKY;
|
||||
options.minimizer_progress_to_stdout = false;
|
||||
options.max_num_iterations = p.options.max_iterations;
|
||||
options.max_solver_time_in_seconds = p.options.max_time_s;
|
||||
options.logging_type = ceres::LoggingType::SILENT;
|
||||
options.num_threads = num_threads;
|
||||
ceres::Solver::Summary summary;
|
||||
ceres::Solve(options, &problem, &summary);
|
||||
return summary;
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
||||
// SPDX-License-Identifier: GPL-3.0-only
|
||||
|
||||
// Ceres first: its logging header defines a CHECK macro of its own, which Catch's must replace here.
|
||||
#include "XtalRefineCeres.h"
|
||||
#undef CHECK
|
||||
#include <catch2/catch_all.hpp>
|
||||
|
||||
#include "../image_analysis/geom_refinement/LatticeReduction.h"
|
||||
#include "../image_analysis/bragg_prediction/BraggPrediction.h"
|
||||
|
||||
// The LM solver of XtalRefine is meant to take Ceres' path to Ceres' answer: same steps accepted, same
|
||||
// stopping rule, same point. These cases hand one problem to both and compare.
|
||||
namespace {
|
||||
// A monoclinic crystal rotated about X over ten 3-degree frames, its predicted spots as observations;
|
||||
// the refinement starts from a perturbed beam, tilt and cell.
|
||||
XtalRefineProblem RotationProblem(bool reduced, bool weighted) {
|
||||
DiffractionExperiment exp;
|
||||
exp.IncidentEnergy_keV(WVL_1A_IN_KEV).BeamX_pxl(1000).BeamY_pxl(1000)
|
||||
.PoniRot1_rad(0.01).PoniRot2_rad(0.02).DetectorDistance_mm(200);
|
||||
const auto geom = exp.GetDiffractionGeometry();
|
||||
const CrystalLattice latt(40, 50, 80, 90, 95, 90);
|
||||
const GoniometerAxis axis("omega", 0.0f, 3.0f, Coord(1, 0, 0), std::nullopt);
|
||||
const gemmi::CrystalSystem sys = gemmi::CrystalSystem::Monoclinic;
|
||||
|
||||
XtalRefineProblem p;
|
||||
p.crystal_system = sys;
|
||||
p.beam_and_orientation_only = reduced;
|
||||
p.distance_mm = 200;
|
||||
|
||||
BraggPrediction prediction;
|
||||
const BraggPredictionSettings settings{.high_res_A = 1.5, .ewald_dist_cutoff = 0.002};
|
||||
for (int img = 0; img < 10; img++) {
|
||||
const float angle_deg = axis.GetAngle_deg(img) + axis.GetWedge_deg() / 2.0f;
|
||||
const auto n = prediction.Calc(exp, latt.Multiply(axis.GetTransformationAngle(angle_deg).transpose()),
|
||||
settings);
|
||||
p.frame_angle_rad.push_back(angle_deg * PI / 180.0);
|
||||
for (int i = 0; i < n; i++) {
|
||||
const auto &r = prediction.GetReflections().at(i);
|
||||
p.residuals.emplace_back(r.predicted_x + 0.3 * std::sin(i), r.predicted_y + 0.3 * std::cos(i),
|
||||
geom.GetWavelength_A(), geom.GetPixelSize_mm(), 1.0, 0.0,
|
||||
angle_deg * PI / 180.0, r.h, r.k, r.l, sys);
|
||||
p.frame.push_back(img);
|
||||
if (weighted)
|
||||
p.weight_sq.push_back(0.2 + 0.6 * (i % 5) / 4.0);
|
||||
}
|
||||
}
|
||||
|
||||
p.beam[0] = 1000.0;
|
||||
p.beam[1] = 997.0;
|
||||
p.detector_rot[0] = 0.012;
|
||||
p.detector_rot[1] = 0.018;
|
||||
p.rot_vec[0] = 1.0;
|
||||
p.rot_vec[1] = 0.0;
|
||||
p.rot_vec[2] = 0.0;
|
||||
double beta = 0;
|
||||
LatticeToRodriguesLengthsBeta_Mono(CrystalLattice(39.7f, 50.6f, 79.6f, 90.0f, 94.5f, 90.0f),
|
||||
p.latt_vec0, p.latt_vec1, beta);
|
||||
p.latt_vec2[0] = beta;
|
||||
|
||||
if (!reduced) {
|
||||
p.detector_rot_constant = false;
|
||||
p.rot_vec_constant = false;
|
||||
p.latt_vec1_constant = false;
|
||||
p.latt_vec2_constant = false;
|
||||
for (int i = 0; i < 2; i++) {
|
||||
p.detector_rot_lower[i] = p.detector_rot[i] - 0.05;
|
||||
p.detector_rot_upper[i] = p.detector_rot[i] + 0.05;
|
||||
}
|
||||
for (int i = 0; i < 3; i++) {
|
||||
p.latt_vec1_lower[i] = 5.0;
|
||||
p.latt_vec1_upper[i] = 100.0;
|
||||
}
|
||||
p.latt_vec2_lower[0] = PI / 3;
|
||||
p.latt_vec2_upper[0] = 2 * PI / 3;
|
||||
p.priors.push_back({XtalRefinePrior::Block::Beam, 1.0, 0.0, p.beam[0], 0.5});
|
||||
p.priors.push_back({XtalRefinePrior::Block::DetectorRot, 0.0, 1.0, p.detector_rot[1], 50.0});
|
||||
}
|
||||
p.options.max_iterations = 50;
|
||||
return p;
|
||||
}
|
||||
|
||||
void CompareWithCeres(XtalRefineProblem p) {
|
||||
XtalRefineProblem q = p;
|
||||
const LMSummary lm = SolveXtalRefine(p, 4);
|
||||
const ceres::Solver::Summary ref = SolveXtalRefineCeres(q, 4);
|
||||
|
||||
REQUIRE(lm.IsSolutionUsable() == ref.IsSolutionUsable());
|
||||
CHECK(lm.iterations == static_cast<int>(ref.iterations.size()));
|
||||
CHECK(lm.final_cost == Catch::Approx(ref.final_cost).epsilon(1e-9));
|
||||
const auto same = [](const double *a, const double *b, int n, double tol) {
|
||||
for (int i = 0; i < n; i++)
|
||||
CHECK(a[i] == Catch::Approx(b[i]).margin(tol));
|
||||
};
|
||||
same(p.beam, q.beam, 2, 1e-7);
|
||||
same(p.detector_rot, q.detector_rot, 2, 1e-10);
|
||||
same(p.rot_vec, q.rot_vec, 3, 1e-10);
|
||||
same(p.latt_vec0, q.latt_vec0, 3, 1e-10);
|
||||
same(p.latt_vec1, q.latt_vec1, 3, 1e-8);
|
||||
same(p.latt_vec2, q.latt_vec2, 3, 1e-10);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("XtalRefine_matches_Ceres_full", "[XtalOptimizer]") {
|
||||
CompareWithCeres(RotationProblem(false, false));
|
||||
}
|
||||
|
||||
TEST_CASE("XtalRefine_matches_Ceres_full_weighted", "[XtalOptimizer]") {
|
||||
CompareWithCeres(RotationProblem(false, true));
|
||||
}
|
||||
|
||||
TEST_CASE("XtalRefine_matches_Ceres_beam_orientation", "[XtalOptimizer]") {
|
||||
CompareWithCeres(RotationProblem(true, false));
|
||||
}
|
||||
|
||||
TEST_CASE("XtalRefine_same_answer_at_any_thread_count", "[XtalOptimizer]") {
|
||||
XtalRefineProblem a = RotationProblem(false, false);
|
||||
XtalRefineProblem b = a;
|
||||
SolveXtalRefine(a, 1);
|
||||
SolveXtalRefine(b, 7);
|
||||
for (int i = 0; i < 3; i++) {
|
||||
CHECK(a.latt_vec0[i] == b.latt_vec0[i]);
|
||||
CHECK(a.latt_vec1[i] == b.latt_vec1[i]);
|
||||
}
|
||||
CHECK(a.beam[0] == b.beam[0]);
|
||||
CHECK(a.beam[1] == b.beam[1]);
|
||||
}
|
||||
Reference in New Issue
Block a user