Build Packages / Create release (push) Successful in 17s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m22s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m37s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 9m33s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 10m39s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 11m4s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 13m19s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 17m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 18m49s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 19m10s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m26s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m31s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 18m54s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m45s
Build Packages / Generate python client (push) Successful in 37s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 20m20s
Build Packages / Build documentation (push) Successful in 1m32s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m37s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m6s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 19m49s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 20m29s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 17m2s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 14m27s
Build Packages / Unit tests (push) Successful in 1h18m12s
* Rugnux: Performance improvements on GPU and CPU (more of the pre-scan and of scaling on the GPU, faster CPU spot finding and crystal refinement), with unchanged results. * Rugnux: More robust processing - patches of persistently hot pixels are masked, an inconsistent merge triggers a retry at the measured beam centre, and builds targeting different CPU levels give the same results. * Rugnux: Improved scaling and merging - reflections with an overloaded pixel are dropped, as in XDS, sparse rotation sweeps are scaled more reliably, and French-Wilson amplitudes use an anisotropic Wilson prior. * Rugnux: Improved space-group determination - glide planes in groups without a centre of symmetry, screw axes from short or weak axial rows kept when a higher group is adopted, and more reliable decisions on twinned and pseudo-symmetric crystals. * Rugnux: Improved small-molecule processing - spots that grow wider than the integration disk and split spots are integrated over their measured footprint, sparse lattices are integrated on every frame, and the `.hkl` file holds unmerged scaled reflections (SHELX HKLF 4). * Rugnux: Reads Rigaku d*TREK SMV images (Saturn CCD), including detector 2theta and encoded pixel overflows; home-source (rotating-anode) datasets were added to the validation battery. * jfjoch_viewer: Fixed processing failing at the end with "Wrong JPEG library version" on Linux; the merge window shows the space group with proper subscripts and a checklist of crystal pathologies. Reviewed-on: #84 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
1069 lines
51 KiB
C++
1069 lines
51 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "ShadowFinder.h"
|
|
#include "ShadowFinderInternal.h"
|
|
|
|
#include <algorithm>
|
|
#include <cmath>
|
|
#include <limits>
|
|
#include <atomic>
|
|
#include <future>
|
|
#include <thread>
|
|
#include <numbers>
|
|
#include <type_traits>
|
|
|
|
#include <spdlog/spdlog.h>
|
|
|
|
#include "../../common/CUDAWrapper.h"
|
|
#include "../../common/ParallelFor.h"
|
|
#include "../../common/JFJochException.h"
|
|
|
|
using namespace shadow_finder;
|
|
|
|
static_assert(ShadowFinder::SHADOW == MASK_SHADOW && ShadowFinder::TRANSMITTING == MASK_TRANSMITTING);
|
|
|
|
// Binary-image helpers on a width*height frame stored row-major as char (0/1). All run once,
|
|
// at GetMask() time, and all are O(pixels) rather than O(pixels * radius).
|
|
namespace {
|
|
|
|
// A per-pixel array of GetMask(). A std::vector zeroes what it allocates on the thread that makes it,
|
|
// and on a 16 Mpx detector the thirty-odd arrays below come to 1.6 GB, every page of which then
|
|
// faulted in on the calling thread, one after the other. A Plane is not initialised when it is made:
|
|
// the ParallelChunks that first writes it touches its pages, spread over the workers. Every Plane
|
|
// below is therefore written in full before it is read.
|
|
template <typename T>
|
|
using Plane = std::vector<T, NoInitAllocator<T>>;
|
|
|
|
// A Plane that starts at `value` everywhere, for the ones that are filled piecemeal afterwards.
|
|
template <typename T>
|
|
Plane<T> filled_plane(int n, T value, size_t nthreads) {
|
|
Plane<T> out(n);
|
|
ParallelChunks(n, nthreads, [&](int lo, int hi) {
|
|
std::fill(out.begin() + lo, out.begin() + hi, value);
|
|
});
|
|
return out;
|
|
}
|
|
|
|
// Deficit of an observed count against its expectation, in standard deviations, from the Poisson
|
|
// likelihood ratio. Zero when the observation is not below the expectation.
|
|
double poisson_deficit_sigma(double observed, double expected) {
|
|
if (expected <= 0.0 || observed >= expected)
|
|
return 0.0;
|
|
const double ll = 2.0 * (expected - observed
|
|
+ (observed > 0.0 ? observed * std::log(observed / expected) : 0.0));
|
|
return ll > 0.0 ? std::sqrt(ll) : 0.0;
|
|
}
|
|
|
|
// 8-connected dilation by `r` pixels, i.e. every pixel within Chebyshev distance r of a set one.
|
|
//
|
|
// This was a multi-source BFS, which is what the distance is defined by - but on a full rectangle
|
|
// with no obstacles the 8-connected graph distance IS the Chebyshev distance (a path stepping
|
|
// towards the target never has to leave the frame), so the result is a dilation by the (2r+1) square
|
|
// clipped to the frame. A square is the product of a horizontal and a vertical segment and max is
|
|
// associative, so it separates into one pass along x and one along y - O(1) per pixel whatever r is,
|
|
// no queue, and no 4-bytes-per-pixel distance array. The values are 0/1, so the running "is any set"
|
|
// is just a count of set pixels in the window.
|
|
Plane<char> dilate(const Plane<char> &in, int W, int H, int r, size_t nthreads) {
|
|
if (r <= 0)
|
|
return in;
|
|
Plane<char> tmp(in.size()), out(in.size());
|
|
|
|
ParallelChunks(H, nthreads, [&](int ylo, int yhi) {
|
|
for (int y = ylo; y < yhi; y++) {
|
|
const char *src = in.data() + static_cast<size_t>(y) * W;
|
|
char *dst = tmp.data() + static_cast<size_t>(y) * W;
|
|
int count = 0;
|
|
for (int x = 0; x <= std::min(r, W - 1); x++)
|
|
count += src[x];
|
|
for (int x = 0; x < W; x++) {
|
|
dst[x] = count > 0;
|
|
if (x + r + 1 < W) count += src[x + r + 1];
|
|
if (x - r >= 0) count -= src[x - r];
|
|
}
|
|
}
|
|
});
|
|
|
|
// The vertical pass walks a column, which strides by a whole row. Take a strip of columns at a
|
|
// time so each row it touches is read contiguously instead of one byte per cache line.
|
|
constexpr int STRIP = 64;
|
|
ParallelChunks((W + STRIP - 1) / STRIP, nthreads, [&](int slo, int shi) {
|
|
std::vector<int> count(STRIP);
|
|
for (int st = slo; st < shi; st++) {
|
|
const int x0 = st * STRIP, xn = std::min(STRIP, W - x0);
|
|
std::fill(count.begin(), count.begin() + xn, 0);
|
|
for (int y = 0; y <= std::min(r, H - 1); y++)
|
|
for (int i = 0; i < xn; i++)
|
|
count[i] += tmp[static_cast<size_t>(y) * W + x0 + i];
|
|
for (int y = 0; y < H; y++) {
|
|
for (int i = 0; i < xn; i++)
|
|
out[static_cast<size_t>(y) * W + x0 + i] = count[i] > 0;
|
|
if (y + r + 1 < H)
|
|
for (int i = 0; i < xn; i++)
|
|
count[i] += tmp[static_cast<size_t>(y + r + 1) * W + x0 + i];
|
|
if (y - r >= 0)
|
|
for (int i = 0; i < xn; i++)
|
|
count[i] -= tmp[static_cast<size_t>(y - r) * W + x0 + i];
|
|
}
|
|
}
|
|
});
|
|
return out;
|
|
}
|
|
|
|
// Erosion by `r` = dilation of the complement. The dilation is clipped to the frame and cannot seed
|
|
// outside it, so outside the frame contributes nothing - a pixel within r of the edge is eroded only
|
|
// by what the frame actually holds.
|
|
Plane<char> erode(const Plane<char> &in, int W, int H, int r, size_t nthreads) {
|
|
Plane<char> comp(in.size());
|
|
ParallelChunks(static_cast<int>(in.size()), nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++) comp[i] = !in[i];
|
|
});
|
|
const auto grown = dilate(comp, W, H, r, nthreads);
|
|
Plane<char> out(in.size());
|
|
ParallelChunks(static_cast<int>(out.size()), nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++) out[i] = !grown[i];
|
|
});
|
|
return out;
|
|
}
|
|
|
|
// Join a region across the module gaps it crosses: every run of gap pixels along a row or a column
|
|
// whose two ends both touch the region becomes region. A gap carries no data, so a shadow that
|
|
// continues on both sides of it is one shadow - but a gap can be wider than BRIDGE_PX reaches (17 px
|
|
// between the rows of PILATUS modules), and an arm crossing one fell apart into pieces each too small
|
|
// to be believed.
|
|
Plane<char> bridge_gaps(const Plane<char> ®ion, const Plane<char> &valid, int W, int H, size_t nthreads) {
|
|
Plane<char> out = region;
|
|
// The lines of one direction are independent: each reads `region` and only ever sets its own pixels.
|
|
auto walk = [&](int n_lines, int len, auto index) {
|
|
ParallelChunks(n_lines, nthreads, [&](int lo, int hi) {
|
|
for (int line = lo; line < hi; line++) {
|
|
int k = 0;
|
|
while (k < len) {
|
|
if (valid[index(line, k)]) { k++; continue; }
|
|
const int start = k;
|
|
while (k < len && !valid[index(line, k)]) k++;
|
|
if (start > 0 && k < len && region[index(line, start - 1)] && region[index(line, k)])
|
|
for (int j = start; j < k; j++) out[index(line, j)] = 1;
|
|
}
|
|
}
|
|
});
|
|
};
|
|
walk(H, W, [W](int y, int x) { return static_cast<size_t>(y) * W + x; });
|
|
walk(W, H, [W](int x, int y) { return static_cast<size_t>(y) * W + x; });
|
|
return out;
|
|
}
|
|
|
|
// The 8-connected components of `member`: each member pixel gets the index of its component, dense
|
|
// from 0 and in no particular order, and every other pixel -1.
|
|
//
|
|
// Labelled in parallel. Each band of rows is flooded on its own, then the pieces that touch across a
|
|
// band boundary are joined. Which pixels share a component is all a caller reads, and that does not
|
|
// depend on how the rows were split.
|
|
struct Components {
|
|
Plane<int> id;
|
|
int count = 0;
|
|
};
|
|
|
|
Components label_components(const Plane<char> &member, int W, int H, size_t nthreads) {
|
|
const int bands = std::min(64, H); // never more bands than rows, so none is empty
|
|
std::vector<int> band_row(bands + 1);
|
|
for (int b = 0; b <= bands; b++)
|
|
band_row[b] = static_cast<int>(static_cast<int64_t>(b) * H / bands);
|
|
|
|
Components out;
|
|
out.id = Plane<int>(member.size());
|
|
std::vector<int> band_pieces(bands, 0);
|
|
ParallelFor(bands, nthreads, [&](int b) {
|
|
const size_t lo = static_cast<size_t>(band_row[b]) * W, hi = static_cast<size_t>(band_row[b + 1]) * W;
|
|
std::fill(out.id.begin() + lo, out.id.begin() + hi, -1);
|
|
std::vector<size_t> stack;
|
|
int pieces = 0;
|
|
for (size_t start = lo; start < hi; start++) {
|
|
if (!member[start] || out.id[start] >= 0)
|
|
continue;
|
|
out.id[start] = pieces;
|
|
stack.push_back(start);
|
|
while (!stack.empty()) {
|
|
const size_t i = stack.back(); stack.pop_back();
|
|
const int y = static_cast<int>(i / W), x = static_cast<int>(i % W);
|
|
for (int dy = -1; dy <= 1; dy++)
|
|
for (int dx = -1; dx <= 1; dx++) {
|
|
const int yy = y + dy, xx = x + dx;
|
|
if (yy < band_row[b] || yy >= band_row[b + 1] || xx < 0 || xx >= W)
|
|
continue;
|
|
const size_t j = static_cast<size_t>(yy) * W + xx;
|
|
if (member[j] && out.id[j] < 0) { out.id[j] = pieces; stack.push_back(j); }
|
|
}
|
|
}
|
|
pieces++;
|
|
}
|
|
band_pieces[b] = pieces;
|
|
});
|
|
|
|
// A piece is named by its band's first index plus its number in the band, and the pieces are
|
|
// joined across each boundary row by union-find.
|
|
std::vector<int> first(bands + 1, 0);
|
|
for (int b = 0; b < bands; b++)
|
|
first[b + 1] = first[b] + band_pieces[b];
|
|
std::vector<int> parent(first[bands]);
|
|
for (size_t k = 0; k < parent.size(); k++)
|
|
parent[k] = static_cast<int>(k);
|
|
const auto find = [&](int k) {
|
|
while (parent[k] != k) { parent[k] = parent[parent[k]]; k = parent[k]; }
|
|
return k;
|
|
};
|
|
for (int b = 0; b + 1 < bands; b++) {
|
|
const size_t above = static_cast<size_t>(band_row[b + 1] - 1) * W, below = above + W;
|
|
for (int x = 0; x < W; x++) {
|
|
if (!member[above + x])
|
|
continue;
|
|
for (int xx = std::max(0, x - 1); xx <= std::min(W - 1, x + 1); xx++)
|
|
if (member[below + xx]) {
|
|
const int ra = find(first[b] + out.id[above + x]);
|
|
const int rb = find(first[b + 1] + out.id[below + xx]);
|
|
if (ra != rb) parent[std::max(ra, rb)] = std::min(ra, rb);
|
|
}
|
|
}
|
|
}
|
|
std::vector<int> dense(parent.size(), -1), component(parent.size());
|
|
for (size_t k = 0; k < parent.size(); k++) {
|
|
const int root = find(static_cast<int>(k));
|
|
if (dense[root] < 0) dense[root] = out.count++;
|
|
component[k] = dense[root];
|
|
}
|
|
ParallelFor(bands, nthreads, [&](int b) {
|
|
const size_t lo = static_cast<size_t>(band_row[b]) * W, hi = static_cast<size_t>(band_row[b + 1]) * W;
|
|
for (size_t i = lo; i < hi; i++)
|
|
if (out.id[i] >= 0) out.id[i] = component[first[b] + out.id[i]];
|
|
});
|
|
return out;
|
|
}
|
|
|
|
// Fill holes: background not reachable from the image border becomes region.
|
|
//
|
|
// The flood is run over the bounding box of `region` grown by one, not the whole detector. Outside
|
|
// that box every pixel is background and the box's surrounding ring is background too, so the whole
|
|
// outside is one border-connected component: a background pixel inside the box is border-connected
|
|
// exactly when it reaches the ring. The beam stop occupies a small part of a detector, so this is
|
|
// the same answer over a fraction of the pixels.
|
|
Plane<char> fill_holes(const Plane<char> ®ion, int W, int H, size_t nthreads) {
|
|
std::vector<int> row_x0(H, W), row_x1(H, -1);
|
|
ParallelChunks(H, nthreads, [&](int ylo, int yhi) {
|
|
for (int y = ylo; y < yhi; y++)
|
|
for (int x = 0; x < W; x++)
|
|
if (region[static_cast<size_t>(y) * W + x]) {
|
|
row_x0[y] = std::min(row_x0[y], x);
|
|
row_x1[y] = x;
|
|
}
|
|
});
|
|
int x0 = W, x1 = -1, y0 = H, y1 = -1;
|
|
for (int y = 0; y < H; y++)
|
|
if (row_x1[y] >= 0) {
|
|
x0 = std::min(x0, row_x0[y]); x1 = std::max(x1, row_x1[y]);
|
|
y0 = std::min(y0, y); y1 = y;
|
|
}
|
|
if (x1 < 0)
|
|
return region; // nothing to enclose
|
|
x0 = std::max(0, x0 - 1); x1 = std::min(W - 1, x1 + 1);
|
|
y0 = std::max(0, y0 - 1); y1 = std::min(H - 1, y1 + 1);
|
|
|
|
// The background of the box, in components; one that reaches the box's edge is outside.
|
|
const int BW = x1 - x0 + 1, BH = y1 - y0 + 1;
|
|
Plane<char> background(static_cast<size_t>(BW) * BH);
|
|
ParallelChunks(BH, nthreads, [&](int lo, int hi) {
|
|
for (int by = lo; by < hi; by++)
|
|
for (int bx = 0; bx < BW; bx++)
|
|
background[static_cast<size_t>(by) * BW + bx] = !region[static_cast<size_t>(by + y0) * W + bx + x0];
|
|
});
|
|
const auto pieces = label_components(background, BW, BH, nthreads);
|
|
std::vector<char> outside(pieces.count, 0);
|
|
const auto edge = [&](int bx, int by) {
|
|
const int c = pieces.id[static_cast<size_t>(by) * BW + bx];
|
|
if (c >= 0) outside[c] = 1;
|
|
};
|
|
for (int bx = 0; bx < BW; bx++) { edge(bx, 0); edge(bx, BH - 1); }
|
|
for (int by = 0; by < BH; by++) { edge(0, by); edge(BW - 1, by); }
|
|
|
|
Plane<char> out = region;
|
|
ParallelChunks(BH, nthreads, [&](int lo, int hi) {
|
|
for (int by = lo; by < hi; by++)
|
|
for (int bx = 0; bx < BW; bx++) {
|
|
const int c = pieces.id[static_cast<size_t>(by) * BW + bx];
|
|
if (c >= 0 && !outside[c])
|
|
out[static_cast<size_t>(by + y0) * W + bx + x0] = 1;
|
|
}
|
|
});
|
|
return out;
|
|
}
|
|
|
|
// Sum of `in` over the k x k box centred on each pixel, zero outside the frame.
|
|
//
|
|
// Each row's and each column's running sum keeps exactly the terms it had, in the order it had them,
|
|
// so the floating-point form rounds identically - only the traversal changes. Walking one column at
|
|
// a time strides a whole row per step and misses on every access, so the vertical pass takes a strip
|
|
// of columns together and reads each row it touches contiguously. Rows, and strips, are independent.
|
|
template <typename T>
|
|
Plane<T> box_sum(const Plane<T> &in, int W, int H, int k, size_t nthreads) {
|
|
const int half = k / 2;
|
|
Plane<T> row(in.size()), out(in.size());
|
|
|
|
ParallelChunks(H, nthreads, [&](int ylo, int yhi) {
|
|
for (int y = ylo; y < yhi; y++) {
|
|
const T *src = in.data() + static_cast<size_t>(y) * W;
|
|
T *dst = row.data() + static_cast<size_t>(y) * W;
|
|
T s = 0;
|
|
for (int x = 0; x <= std::min(half, W - 1); x++)
|
|
s += src[x];
|
|
for (int x = 0; x < W; x++) {
|
|
dst[x] = s;
|
|
if (x + half + 1 < W) s += src[x + half + 1];
|
|
if (x - half >= 0) s -= src[x - half];
|
|
}
|
|
}
|
|
});
|
|
|
|
constexpr int STRIP = 64;
|
|
ParallelChunks((W + STRIP - 1) / STRIP, nthreads, [&](int slo, int shi) {
|
|
std::vector<T> s(STRIP);
|
|
for (int st = slo; st < shi; st++) {
|
|
const int x0 = st * STRIP, xn = std::min(STRIP, W - x0);
|
|
std::fill(s.begin(), s.begin() + xn, T{0});
|
|
for (int y = 0; y <= std::min(half, H - 1); y++)
|
|
for (int i = 0; i < xn; i++)
|
|
s[i] += row[static_cast<size_t>(y) * W + x0 + i];
|
|
for (int y = 0; y < H; y++) {
|
|
for (int i = 0; i < xn; i++)
|
|
out[static_cast<size_t>(y) * W + x0 + i] = s[i];
|
|
if (y + half + 1 < H)
|
|
for (int i = 0; i < xn; i++)
|
|
s[i] += row[static_cast<size_t>(y + half + 1) * W + x0 + i];
|
|
if (y - half >= 0)
|
|
for (int i = 0; i < xn; i++)
|
|
s[i] -= row[static_cast<size_t>(y - half) * W + x0 + i];
|
|
}
|
|
}
|
|
});
|
|
return out;
|
|
}
|
|
|
|
// The values of each ring, laid end to end, with the ring's slice given by offset[r]..offset[r+1].
|
|
// radius and pooled do not change over the three baseline iterations, so this is built once and
|
|
// every iteration is an order statistic of the same, already-sorted, ring.
|
|
struct RingValues {
|
|
Plane<float> values;
|
|
std::vector<int> offset;
|
|
};
|
|
|
|
RingValues bin_by_ring(const Plane<float> &values, const Plane<char> &valid,
|
|
const Plane<int> &radius, int max_radius, size_t nthreads) {
|
|
// Counted and scattered by blocks of pixels in parallel: each block writes its values of a ring
|
|
// after those of the blocks before it, so every ring holds its values in pixel order, as a single
|
|
// pass would leave them - and they are sorted below in any case.
|
|
constexpr int BLOCKS = 64;
|
|
const size_t n = values.size();
|
|
const auto block_begin = [n](int b) { return n * b / BLOCKS; };
|
|
const size_t rings = static_cast<size_t>(max_radius) + 1;
|
|
std::vector<int> cursor(BLOCKS * rings, 0);
|
|
ParallelFor(BLOCKS, nthreads, [&](int b) {
|
|
int *count = cursor.data() + b * rings;
|
|
for (size_t i = block_begin(b); i < block_begin(b + 1); i++)
|
|
if (valid[i])
|
|
count[radius[i]]++;
|
|
});
|
|
|
|
RingValues rv;
|
|
rv.offset.assign(max_radius + 2, 0);
|
|
for (size_t r = 0; r < rings; r++) {
|
|
int at = rv.offset[r];
|
|
for (int b = 0; b < BLOCKS; b++) {
|
|
const int count = cursor[b * rings + r];
|
|
cursor[b * rings + r] = at;
|
|
at += count;
|
|
}
|
|
rv.offset[r + 1] = at;
|
|
}
|
|
|
|
rv.values.resize(rv.offset[max_radius + 1]);
|
|
ParallelFor(BLOCKS, nthreads, [&](int b) {
|
|
int *next = cursor.data() + b * rings;
|
|
for (size_t i = block_begin(b); i < block_begin(b + 1); i++)
|
|
if (valid[i])
|
|
rv.values[next[radius[i]]++] = values[i];
|
|
});
|
|
|
|
// Sorted once; the three iterations then only pick a rank and count a prefix.
|
|
ParallelFor(max_radius + 1, nthreads, [&](int r) {
|
|
std::sort(rv.values.begin() + rv.offset[r], rv.values.begin() + rv.offset[r + 1]);
|
|
});
|
|
return rv;
|
|
}
|
|
|
|
} // namespace
|
|
|
|
ShadowFinder::ShadowFinder(const DiffractionExperiment &experiment, const PixelMask &mask)
|
|
: width(static_cast<int>(experiment.GetXPixelsNumConv())),
|
|
height(static_cast<int>(experiment.GetYPixelsNumConv())),
|
|
beam_x(experiment.GetBeamX_pxl()),
|
|
beam_y(experiment.GetBeamY_pxl()),
|
|
geometry(experiment.GetDiffractionGeometry()),
|
|
polarization(experiment.GetPolarizationFactor()),
|
|
pixel_mask(mask.GetMask(experiment)) {
|
|
if (pixel_mask.size() != static_cast<size_t>(width) * height)
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
|
|
"ShadowFinder: pixel mask does not match the detector");
|
|
#ifdef JFJOCH_USE_CUDA
|
|
if (get_gpu_count() > 0) {
|
|
const size_t npixels = static_cast<size_t>(width) * height;
|
|
gpu_pending = std::async(std::launch::async, [npixels] {
|
|
return std::make_unique<ShadowAccumulatorGPU>(npixels);
|
|
});
|
|
}
|
|
#endif
|
|
}
|
|
|
|
void ShadowFinder::BeamCenter(float x, float y) {
|
|
std::unique_lock ul(m);
|
|
beam_x = x;
|
|
beam_y = y;
|
|
}
|
|
|
|
#ifdef JFJOCH_USE_CUDA
|
|
ShadowFinder::Projection ShadowFinder::Reduce() const {
|
|
// The device holds its own projection. Bring it back; when every frame went to the GPU it is the
|
|
// whole answer.
|
|
Projection out;
|
|
if (Gpu() && gpu->GetFrameCount() > 0) {
|
|
gpu->Download(out.max_value, out.sum_value, out.valid_count);
|
|
out.frames = gpu->GetFrameCount();
|
|
}
|
|
if (host.frames == 0)
|
|
return out;
|
|
|
|
// A pixel is touched by one worker only and the sums and counts are integers, so the result is
|
|
// the same as folding on one thread.
|
|
out.frames += host.frames;
|
|
ParallelChunks(static_cast<int>(out.max_value.size()), std::thread::hardware_concurrency(), [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++) {
|
|
if (host.valid_count[i] == 0)
|
|
continue;
|
|
if (out.valid_count[i] == 0 || host.max_value[i] > out.max_value[i])
|
|
out.max_value[i] = host.max_value[i];
|
|
out.sum_value[i] += host.sum_value[i];
|
|
out.valid_count[i] += host.valid_count[i];
|
|
}
|
|
});
|
|
return out;
|
|
}
|
|
#endif
|
|
|
|
template<class T>
|
|
void ShadowFinder::Add(const T *ptr, size_t begin, size_t end) {
|
|
// The pixel type's sentinel extreme marks "no data" (module gap / masked): the
|
|
// preprocessor/writer stores INT*_MIN for signed and UINT*_MAX for unsigned. For signed
|
|
// types the opposite extreme is a genuine saturated value and is kept, so a saturated
|
|
// reflection still registers as bright.
|
|
T masked;
|
|
if constexpr (std::is_signed_v<T>)
|
|
masked = std::numeric_limits<T>::min();
|
|
else
|
|
masked = std::numeric_limits<T>::max();
|
|
|
|
for (size_t i = begin; i < end; i++) {
|
|
const T v = ptr[i];
|
|
if (v == masked)
|
|
continue;
|
|
const int64_t vi = static_cast<int64_t>(v);
|
|
if (host.valid_count[i] == 0 || vi > host.max_value[i])
|
|
host.max_value[i] = vi;
|
|
host.sum_value[i] += vi;
|
|
host.valid_count[i]++;
|
|
}
|
|
}
|
|
|
|
void ShadowFinder::AddImage(const DataMessage &data, std::vector<uint8_t> &buffer) {
|
|
if (static_cast<size_t>(data.image.GetWidth()) * data.image.GetHeight()
|
|
!= static_cast<size_t>(width) * height)
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
|
|
"ShadowFinder: image size does not match the detector");
|
|
|
|
#ifdef JFJOCH_USE_CUDA
|
|
// One device, so the frames queue here - but each is only a chunk upload plus two kernels, and
|
|
// nothing was decompressed on the host to get this far.
|
|
if (ShadowAccumulatorGPU *acc = ShadowAccumulatorGPU::Supports(data.image) ? Gpu() : nullptr) {
|
|
std::unique_lock ul(gpu_mutex);
|
|
try {
|
|
acc->Add(data.image);
|
|
return;
|
|
} catch (const std::exception &e) {
|
|
spdlog::warn("Beam stop: GPU accumulate failed ({}), falling back to the host", e.what());
|
|
cuda_clear_error(); // handled - see cuda_clear_error()
|
|
}
|
|
}
|
|
#endif
|
|
|
|
const size_t npixels = static_cast<size_t>(width) * height;
|
|
{
|
|
std::unique_lock ul(host_mutex);
|
|
if (host.max_value.empty()) {
|
|
host.max_value.resize(npixels);
|
|
host.sum_value.resize(npixels);
|
|
host.valid_count.resize(npixels);
|
|
}
|
|
}
|
|
const auto ptr = data.image.GetUncompressedPtr(buffer);
|
|
const size_t rows_per_band = (static_cast<size_t>(height) + BANDS - 1) / BANDS;
|
|
const size_t first = next_band.fetch_add(1);
|
|
for (size_t b = 0; b < BANDS; b++) {
|
|
const size_t band = (first + b) % BANDS;
|
|
const size_t begin = std::min(npixels, band * rows_per_band * width);
|
|
const size_t end = std::min(npixels, (band + 1) * rows_per_band * width);
|
|
std::lock_guard lock(band_mutex[band]);
|
|
switch (data.image.GetMode()) {
|
|
case CompressedImageMode::Int8: Add(reinterpret_cast<const int8_t *>(ptr), begin, end); break;
|
|
case CompressedImageMode::Uint8: Add(reinterpret_cast<const uint8_t *>(ptr), begin, end); break;
|
|
case CompressedImageMode::Int16: Add(reinterpret_cast<const int16_t *>(ptr), begin, end); break;
|
|
case CompressedImageMode::Uint16: Add(reinterpret_cast<const uint16_t *>(ptr), begin, end); break;
|
|
case CompressedImageMode::Int32: Add(reinterpret_cast<const int32_t *>(ptr), begin, end); break;
|
|
case CompressedImageMode::Uint32: Add(reinterpret_cast<const uint32_t *>(ptr), begin, end); break;
|
|
default:
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
|
|
"ShadowFinder: unsupported image mode");
|
|
}
|
|
}
|
|
std::unique_lock ul(host_mutex);
|
|
host.frames++;
|
|
}
|
|
|
|
#ifdef JFJOCH_USE_CUDA
|
|
ShadowAccumulatorGPU *ShadowFinder::Gpu() const {
|
|
std::unique_lock ul(gpu_mutex);
|
|
if (gpu_pending.valid()) {
|
|
try {
|
|
gpu = gpu_pending.get();
|
|
} catch (const std::exception &e) {
|
|
// A GPU that cannot hold the projection is not a reason to fail: the host path gives
|
|
// the same answer, only slower.
|
|
spdlog::warn("Beam stop: GPU projection unavailable ({}), accumulating on the host",
|
|
e.what());
|
|
gpu.reset();
|
|
}
|
|
}
|
|
return gpu.get();
|
|
}
|
|
#endif
|
|
|
|
uint32_t ShadowFinder::GetFrameCount() const {
|
|
std::unique_lock ul(m);
|
|
uint32_t frames = host.frames;
|
|
#ifdef JFJOCH_USE_CUDA
|
|
if (gpu) frames += gpu->GetFrameCount();
|
|
#endif
|
|
return frames;
|
|
}
|
|
|
|
const ShadowFinder::Projection &ShadowFinder::Reduced() const {
|
|
#ifdef JFJOCH_USE_CUDA
|
|
if (gpu && gpu->GetFrameCount() > 0) {
|
|
if (!reduced)
|
|
reduced = Reduce();
|
|
return *reduced;
|
|
}
|
|
#endif
|
|
return host;
|
|
}
|
|
|
|
void ShadowFinder::ReleaseProjection() {
|
|
#ifdef JFJOCH_USE_CUDA
|
|
std::unique_lock ul(m);
|
|
reduced.reset();
|
|
#endif
|
|
}
|
|
|
|
std::vector<float> ShadowFinder::GetMeanProjection() const {
|
|
std::unique_lock ul(m);
|
|
#ifdef JFJOCH_USE_CUDA
|
|
if (Gpu() && gpu->GetFrameCount() > 0 && host.frames == 0)
|
|
return gpu->MeanProjection(pixel_mask);
|
|
#endif
|
|
const Projection &p = Reduced();
|
|
const auto &sum_value = p.sum_value;
|
|
const auto &valid_count = p.valid_count;
|
|
|
|
std::vector<float> mean(static_cast<size_t>(width) * height);
|
|
ParallelChunks(static_cast<int>(mean.size()), std::thread::hardware_concurrency(), [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0
|
|
? static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]) : NAN;
|
|
});
|
|
return mean;
|
|
}
|
|
|
|
std::vector<uint32_t> ShadowFinder::GetMask(size_t nthreads) const {
|
|
std::unique_lock ul(m);
|
|
if (nthreads == 0)
|
|
nthreads = std::max(1u, std::thread::hardware_concurrency());
|
|
#ifdef JFJOCH_USE_CUDA
|
|
// Where every frame went to the device the mask is made there too, from the projection as it lies.
|
|
if (Gpu() && gpu->GetFrameCount() > 0 && host.frames == 0) {
|
|
const float diag = std::hypot(static_cast<float>(width), static_cast<float>(height));
|
|
if (!std::isfinite(beam_x) || !std::isfinite(beam_y)
|
|
|| std::fabs(beam_x - width * 0.5f) > 4.0f * diag || std::fabs(beam_y - height * 0.5f) > 4.0f * diag)
|
|
return std::vector<uint32_t>(static_cast<size_t>(width) * height, 0);
|
|
ShadowMaskSetup setup;
|
|
setup.width = width;
|
|
setup.height = height;
|
|
setup.beam_x = beam_x;
|
|
setup.beam_y = beam_y;
|
|
const auto rot = geometry.GetDetectorMatrix().arr();
|
|
for (int k = 0; k < 9; k++)
|
|
setup.det_matrix[k] = rot[k];
|
|
setup.pixel_size_mm = geometry.GetPixelSize_mm();
|
|
setup.distance_mm = geometry.GetDetectorDistance_mm();
|
|
setup.has_polarization = polarization.has_value();
|
|
setup.polarization = polarization.value_or(0.0f);
|
|
return gpu->Mask(setup, pixel_mask);
|
|
}
|
|
#endif
|
|
const Projection &p = Reduced();
|
|
const auto &max_value = p.max_value;
|
|
const auto &sum_value = p.sum_value;
|
|
const auto &valid_count = p.valid_count;
|
|
const uint32_t frames = p.frames;
|
|
|
|
const int W = width, H = height;
|
|
const int n_pixels = W * H;
|
|
|
|
std::vector<uint32_t> mask(n_pixels, 0);
|
|
if (frames == 0)
|
|
return mask;
|
|
|
|
// The rings are meaningless about a centre that is not a real position - a NaN centre would
|
|
// even turn lround below into a negative array index - and a centre many detector sizes away
|
|
// makes max_radius, and every per-ring table sized by it, arbitrarily large. No centre, no
|
|
// shadow.
|
|
const float diag = std::hypot(static_cast<float>(W), static_cast<float>(H));
|
|
if (!std::isfinite(beam_x) || !std::isfinite(beam_y)
|
|
|| std::fabs(beam_x - W * 0.5f) > 4.0f * diag || std::fabs(beam_y - H * 0.5f) > 4.0f * diag)
|
|
return mask;
|
|
|
|
// mean projection, usable pixels and radius from the beam centre. The mean is divided by the
|
|
// polarization factor, so that what is left varies around a ring only where something is in the
|
|
// way; `pol` is kept because the counts the Poisson test is made of are the ones that were
|
|
// recorded, not these. The factor is read off the experiment's own geometry - which follows the
|
|
// detector's tilt, quarter turns and in-plane rotation - about the centre the rings use.
|
|
// Following Kahn, Fourme, Gadet, Janin, Dumas & Andre (1982) J. Appl. Cryst. 15, 330-337
|
|
DiffractionGeometry pol_geometry = geometry;
|
|
pol_geometry.BeamX_pxl(beam_x).BeamY_pxl(beam_y);
|
|
|
|
Plane<float> mean(n_pixels);
|
|
Plane<float> pol(n_pixels);
|
|
Plane<char> valid(n_pixels);
|
|
Plane<int> radius(n_pixels);
|
|
std::atomic<int> max_radius_atomic{0};
|
|
ParallelChunks(H, nthreads, [&](int ylo, int yhi) {
|
|
int local_max = 0;
|
|
for (int y = ylo; y < yhi; y++)
|
|
for (int x = 0; x < W; x++) {
|
|
const int i = y * W + x;
|
|
const float dx = x - beam_x, dy = y - beam_y;
|
|
pol[i] = polarization.has_value()
|
|
? pol_geometry.CalcAzIntPolarizationCorr(static_cast<float>(x),
|
|
static_cast<float>(y),
|
|
polarization.value())
|
|
: 1.0f;
|
|
// pol > 0 also rejects a NaN factor. The correction is exactly zero only where
|
|
// the polarization fully suppresses the scattering, so such a pixel carries no
|
|
// usable ring information anyway - and divided by zero it would put an inf into
|
|
// the pooled means, which the running box sums then turn into NaN for a whole row.
|
|
mean[i] = 0.0f;
|
|
valid[i] = 0;
|
|
if (valid_count[i] > 0 && pixel_mask[i] == 0 && pol[i] > 0.0f) {
|
|
mean[i] = static_cast<float>(static_cast<double>(sum_value[i])
|
|
/ valid_count[i] / pol[i]);
|
|
valid[i] = 1;
|
|
}
|
|
radius[i] = static_cast<int>(std::lround(std::sqrt(dx * dx + dy * dy)));
|
|
local_max = std::max(local_max, radius[i]);
|
|
}
|
|
// max is associative, so folding the per-worker maxima gives the same answer whatever
|
|
// order they finish in.
|
|
int prev = max_radius_atomic.load();
|
|
while (prev < local_max && !max_radius_atomic.compare_exchange_weak(prev, local_max)) {}
|
|
});
|
|
const int max_radius = max_radius_atomic.load();
|
|
|
|
|
|
// Pool the background over a small box before testing it. A background of a fraction of
|
|
// a count per pixel per frame gives no single pixel enough counts to tell a shadow from
|
|
// a Poisson hole; the stop and its arm are wider than the box, so pooling costs no
|
|
// resolution that matters and multiplies the statistics by the pixels in the box.
|
|
// The count is a count: at most 25 pixels, so an integer box sum is exact and costs half the
|
|
// memory of the floating-point one it replaces. The background itself stays in double - its
|
|
// running sum adds and subtracts across a whole row, and in float the rounding of the two would
|
|
// not cancel, which moves pixels across the shadow threshold below.
|
|
Plane<double> num(n_pixels);
|
|
Plane<int32_t> den(n_pixels);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++) {
|
|
num[i] = valid[i] ? mean[i] : 0.0;
|
|
den[i] = valid[i] ? 1 : 0;
|
|
}
|
|
});
|
|
const auto pooled_sum = box_sum(num, W, H, POOL_PX, nthreads);
|
|
const auto pooled_count = box_sum(den, W, H, POOL_PX, nthreads);
|
|
Plane<float> pooled(n_pixels);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
pooled[i] = pooled_count[i] > 0 ? static_cast<float>(pooled_sum[i] / pooled_count[i]) : 0.0f;
|
|
});
|
|
|
|
|
|
// Azimuthal comparison: the median of the ring, iterated so the shadow stays out of the
|
|
// baseline it is measured against. Comparing a pixel's background against the background at its
|
|
// own resolution is how DEFPIX recognises a shaded detector region.
|
|
// Following Kabsch (2010) Acta Cryst. D66, 125-132
|
|
//
|
|
// The iteration only ever excludes pixels whose pooled background is below a cut, and dividing by
|
|
// a positive baseline is monotone - so the excluded pixels of a ring are exactly the lowest ones,
|
|
// and the pixels the next median is taken over are exactly the rest. Each iteration's median is
|
|
// therefore an order statistic of the ring's values, which do not change: bin and sort the rings
|
|
// once, then each iteration picks a rank and counts a prefix. That replaces nine full-image
|
|
// passes (use / ratio / excluded, three times) with one, and three re-binnings with none.
|
|
const RingValues rings = bin_by_ring(pooled, valid, radius, max_radius, nthreads);
|
|
std::vector<float> baseline(max_radius + 1, 0.0f);
|
|
{
|
|
std::vector<int> excluded_in_ring(max_radius + 1, 0);
|
|
for (int iter = 0; iter < 3; iter++)
|
|
ParallelFor(max_radius + 1, nthreads, [&](int r) {
|
|
const int lo = rings.offset[r], hi = rings.offset[r + 1];
|
|
const int n = hi - lo;
|
|
const int m = excluded_in_ring[r];
|
|
const int avail = n - m;
|
|
baseline[r] = (avail <= 0) ? 0.0f : rings.values[lo + m + avail / 2];
|
|
// Counted the same way the per-pixel test below is written, so the two agree bit for
|
|
// bit; the ring is sorted, so this is the length of a prefix.
|
|
const float d = std::max(baseline[r], 1e-6f);
|
|
int excl = 0;
|
|
while (excl < n && rings.values[lo + excl] / d < SHADOW_RATIO)
|
|
excl++;
|
|
excluded_in_ring[r] = excl;
|
|
});
|
|
}
|
|
Plane<float> ratio(n_pixels);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
ratio[i] = valid[i] ? pooled[i] / std::max(baseline[radius[i]], 1e-6f) : 1.0f;
|
|
});
|
|
|
|
|
|
// A ring whose background was never counted carries no information to test a pixel against.
|
|
// Walking outward, every ring before the first countable one lies wholly inside the stop - a
|
|
// ring fully within the disk has no unshadowed pixel for the median to find, which is exactly
|
|
// where an azimuthal comparison must fail. Those rings are shadow in their entirety.
|
|
// Innermost rings hold only a handful of pixels, too few to judge, so they are stepped over
|
|
// rather than allowed to end the walk.
|
|
std::vector<int> ring_pixels(max_radius + 1, 0);
|
|
for (int rad = 0; rad <= max_radius; rad++)
|
|
ring_pixels[rad] = rings.offset[rad + 1] - rings.offset[rad];
|
|
|
|
// A ring lies inside the stop when its background is a fraction of what this detector's
|
|
// background typically is. Counting statistics cannot decide this: on a bright dataset the
|
|
// shadow is still well counted. The comparison used to be against the LARGEST background of any
|
|
// ring further out, and that reads a sample whose background peaks in a strong ring away from
|
|
// the beam - a powder standard, a strong solvent ring - as a beam stop the size of that ring:
|
|
// the ordinary background inside it is legitimately below a third of the peak. On one corpus
|
|
// dataset it declared 16 % of the detector to be stop, with diffraction rings visible inside the
|
|
// disk it masked. The median over the rings this walk is willing to judge is what "typically"
|
|
// means, and it is robust from both sides - to a bright ring, and to a corner ring of a handful
|
|
// of pixels whose median is one pixel's mean.
|
|
const int blocked_out_to = BlockedOutTo(baseline, ring_pixels);
|
|
|
|
// The counts a pixel's pooled background is made of, and the counts the ring says it should
|
|
// have had. The test is on the deficit between them, in units of its own Poisson scatter.
|
|
Plane<float> deficit(n_pixels);
|
|
Plane<char> low(n_pixels);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++) {
|
|
deficit[i] = 0.0f;
|
|
low[i] = 0;
|
|
if (!valid[i])
|
|
continue;
|
|
// The deficit is significant or not in the counts that were recorded, so the pooled
|
|
// background and what the ring expects of it are both put back on the detector's own
|
|
// scale before they are compared - the same factor on each, so it is the units of the
|
|
// comparison that change and not its answer.
|
|
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
|
|
deficit[i] = static_cast<float>(poisson_deficit_sigma(pooled[i] * counted,
|
|
baseline[radius[i]] * counted));
|
|
low[i] = ratio[i] < SHADOW_RATIO && deficit[i] > MIN_DEFICIT_SIGMA;
|
|
}
|
|
});
|
|
|
|
|
|
// The shadow is a low region large enough to have been cast by something. Nothing anchors it to
|
|
// the beam centre: hardware that shadows the detector need not touch the direct beam, and a pin
|
|
// or a loop typically does not - it begins some way out in radius, with lit detector between it
|
|
// and the stop. What keeps the test specific instead is size, since the background wanders by a
|
|
// pixel or two at a time and hardware does not.
|
|
const Plane<char> bridged = dilate(low, W, H, BRIDGE_PX, nthreads);
|
|
Plane<char> region(n_pixels);
|
|
{
|
|
const auto pieces = label_components(bridged, W, H, nthreads);
|
|
std::vector<std::atomic<int>> n_low(pieces.count);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
if (pieces.id[i] >= 0 && low[i])
|
|
n_low[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
|
|
});
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
region[i] = pieces.id[i] >= 0 && n_low[pieces.id[i]].load(std::memory_order_relaxed) >= MIN_SHADOW_PIXELS
|
|
? low[i] : 0;
|
|
});
|
|
}
|
|
|
|
// The rings that lie wholly inside the stop are decided by the ring walk above rather than by
|
|
// the per-pixel test, so they join the region after it and are not asked to be large.
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
if (valid[i] && radius[i] <= blocked_out_to)
|
|
region[i] = 1;
|
|
});
|
|
|
|
|
|
// Recorded reflections. A small cluster is required so a single-frame zinger does not count.
|
|
Plane<char> lit(n_pixels);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
lit[i] = (valid_count[i] > 0) && (max_value[i] >= MIN_REFLECTION);
|
|
});
|
|
Plane<char> reflection(n_pixels);
|
|
ParallelChunks(H, nthreads, [&](int ylo, int yhi) {
|
|
for (int y = ylo; y < yhi; y++)
|
|
for (int x = 0; x < W; x++) {
|
|
const int i = y * W + x;
|
|
reflection[i] = 0;
|
|
if (!lit[i]) continue;
|
|
int neighbours = 0;
|
|
for (int dy = -1; dy <= 1; dy++)
|
|
for (int dx = -1; dx <= 1; dx++) {
|
|
const int yy = y + dy, xx = x + dx;
|
|
if ((dx || dy) && yy >= 0 && yy < H && xx >= 0 && xx < W && lit[yy * W + xx])
|
|
neighbours++;
|
|
}
|
|
reflection[i] = (neighbours >= 2);
|
|
}
|
|
});
|
|
|
|
|
|
// Grow the soft boundary, round it and fill the disk interior.
|
|
const Plane<char> penumbra = dilate(region, W, H, PENUMBRA_MAX_PX, nthreads);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
if (penumbra[i] && valid[i] && ratio[i] < PENUMBRA_RATIO && deficit[i] > MIN_DEFICIT_SIGMA)
|
|
region[i] = 1;
|
|
});
|
|
|
|
|
|
region = erode(dilate(region, W, H, 2, nthreads), W, H, 2, nthreads);
|
|
|
|
region = fill_holes(region, W, H, nthreads);
|
|
|
|
|
|
// Expose recorded reflections - done last, with no fill afterwards, so a spot the shadow
|
|
// still covered is given back rather than re-enclosed.
|
|
const Plane<char> reflection_grown = dilate(reflection, W, H, 1, nthreads);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++) {
|
|
if (reflection_grown[i])
|
|
region[i] = 0;
|
|
mask[i] = region[i] ? SHADOW : 0;
|
|
}
|
|
});
|
|
|
|
// Hardware that lets part of the beam through - a thin holder arm, or one whose shadow is blurred
|
|
// by its distance from the detector - sits between PENUMBRA_RATIO and SHADOW_RATIO along most of
|
|
// its length and crosses SHADOW_RATIO only in places, so the region test above keeps only the
|
|
// fragments of it that are deep, each too small to be believed; and where the background is
|
|
// bright, the attenuated pixels it does keep reach MIN_REFLECTION and are given back as
|
|
// reflections. Such an arm is found here as what the mask above left out: the significantly
|
|
// DIM pixels, joined across small breaks and module gaps, in a piece as large as a shadow must be
|
|
// and deep in enough places. Nothing is given back inside it - the reflections behind it are
|
|
// recorded attenuated, and giving them back is what integrates them low.
|
|
//
|
|
// It only adds whole pieces. The dim pixels along the edge of an ordinary stop are left as the
|
|
// region above drew them: each such edge feeds decisions downstream that can sit at a margin, so
|
|
// a sweep with no transmitting hardware keeps its mask bit for bit.
|
|
//
|
|
// Dim is judged against the ring's own azimuthal variation, not against its median alone. A
|
|
// polarization factor that is not the beam's leaves every ring brighter along one axis and dimmer
|
|
// along the other, smoothly and symmetrically about the beam; at high angle the dim lobes sit
|
|
// below PENUMBRA_RATIO over a quarter of the detector, deep in places from counting noise, and
|
|
// joined they made one piece that covered it. Hardware is not a second harmonic of the ring: it
|
|
// stands in a few sectors, and those sectors are left out of the fit. The harmonic is only ever
|
|
// allowed to explain a dim sector away - where it would ask for more than the median, the median
|
|
// stands - so the step can only drop what it found before, never find something new.
|
|
const int n_bands = max_radius / HARMONIC_BAND_PX + 1;
|
|
const size_t n_sectors = static_cast<size_t>(n_bands) * HARMONIC_SECTORS;
|
|
// Gathered by blocks of rows in parallel and joined in block order. Only the median of each
|
|
// sector is read, and that is the same whatever order its values were gathered in.
|
|
constexpr int SECTOR_BLOCKS = 64;
|
|
std::vector<std::vector<std::vector<float>>> block_values(SECTOR_BLOCKS);
|
|
ParallelFor(SECTOR_BLOCKS, nthreads, [&](int b) {
|
|
auto &values = block_values[b];
|
|
values.resize(n_sectors);
|
|
const int lo = static_cast<int>(static_cast<int64_t>(n_pixels) * b / SECTOR_BLOCKS);
|
|
const int hi = static_cast<int>(static_cast<int64_t>(n_pixels) * (b + 1) / SECTOR_BLOCKS);
|
|
for (int i = lo; i < hi; i++) {
|
|
if (!valid[i] || region[i])
|
|
continue;
|
|
const float dx = static_cast<float>(i % W) - beam_x, dy = static_cast<float>(i / W) - beam_y;
|
|
const double phi = std::atan2(dy, dx) + std::numbers::pi;
|
|
const int sector = std::min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * std::numbers::pi) * HARMONIC_SECTORS));
|
|
values[static_cast<size_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector].push_back(ratio[i]);
|
|
}
|
|
});
|
|
std::vector<std::vector<float>> sector_values(n_sectors);
|
|
ParallelFor(static_cast<int>(n_sectors), nthreads, [&](int k) {
|
|
for (const auto &values : block_values)
|
|
sector_values[k].insert(sector_values[k].end(), values[k].begin(), values[k].end());
|
|
});
|
|
block_values.clear();
|
|
// The median of each sector with enough pixels to have one.
|
|
std::vector<double> sector_median(n_sectors, -1.0);
|
|
ParallelFor(static_cast<int>(n_sectors), nthreads, [&](int k) {
|
|
auto &v = sector_values[k];
|
|
if (v.size() >= MIN_SECTOR_PIXELS) {
|
|
std::nth_element(v.begin(), v.begin() + v.size() / 2, v.end());
|
|
sector_median[k] = v[v.size() / 2];
|
|
}
|
|
});
|
|
std::vector<float> harm_c, harm_s;
|
|
HarmonicFit(sector_median, n_bands, harm_c, harm_s);
|
|
|
|
Plane<char> dim = filled_plane<char>(n_pixels, 0, nthreads);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++) {
|
|
if (!valid[i] || region[i] || ratio[i] >= PENUMBRA_RATIO)
|
|
continue;
|
|
// cos 2phi and sin 2phi from the offset to the beam.
|
|
const float dx = static_cast<float>(i % W) - beam_x, dy = static_cast<float>(i / W) - beam_y;
|
|
const float r2 = std::max(dx * dx + dy * dy, 1e-6f);
|
|
const int band = radius[i] / HARMONIC_BAND_PX;
|
|
const float model = std::min(1.0f, 1.0f + harm_c[band] * (dx * dx - dy * dy) / r2
|
|
+ harm_s[band] * 2.0f * dx * dy / r2);
|
|
if (model >= 1.0f) {
|
|
dim[i] = deficit[i] > MIN_DEFICIT_SIGMA;
|
|
continue;
|
|
}
|
|
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
|
|
dim[i] = ratio[i] < PENUMBRA_RATIO * model
|
|
&& poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA;
|
|
}
|
|
});
|
|
const Plane<char> joined = bridge_gaps(dilate(dim, W, H, BRIDGE_PX, nthreads), valid, W, H, nthreads);
|
|
const auto pieces = label_components(joined, W, H, nthreads);
|
|
std::vector<std::atomic<int>> n_dim(pieces.count), n_low(pieces.count);
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++)
|
|
if (pieces.id[i] >= 0 && dim[i]) {
|
|
n_dim[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
|
|
if (low[i])
|
|
n_low[pieces.id[i]].fetch_add(1, std::memory_order_relaxed);
|
|
}
|
|
});
|
|
ParallelChunks(n_pixels, nthreads, [&](int lo, int hi) {
|
|
for (int i = lo; i < hi; i++) {
|
|
const int c = pieces.id[i];
|
|
if (c >= 0 && dim[i] && n_dim[c].load(std::memory_order_relaxed) >= MIN_SHADOW_PIXELS
|
|
&& n_low[c].load(std::memory_order_relaxed) >= MIN_CORE_PIXELS)
|
|
mask[i] = TRANSMITTING;
|
|
}
|
|
});
|
|
return mask;
|
|
}
|
|
|
|
namespace shadow_finder {
|
|
|
|
int BlockedOutTo(const std::vector<float> &baseline, const std::vector<int> &ring_pixels) {
|
|
const int max_radius = static_cast<int>(baseline.size()) - 1;
|
|
// A ring lies inside the stop when its background is a fraction of what this detector's
|
|
// background typically is. Counting statistics cannot decide this: on a bright dataset the
|
|
// shadow is still well counted. The comparison used to be against the LARGEST background of any
|
|
// ring further out, and that reads a sample whose background peaks in a strong ring away from
|
|
// the beam - a powder standard, a strong solvent ring - as a beam stop the size of that ring:
|
|
// the ordinary background inside it is legitimately below a third of the peak. On one corpus
|
|
// dataset it declared 16 % of the detector to be stop, with diffraction rings visible inside the
|
|
// disk it masked. The median over the rings this walk is willing to judge is what "typically"
|
|
// means, and it is robust from both sides - to a bright ring, and to a corner ring of a handful
|
|
// of pixels whose median is one pixel's mean.
|
|
std::vector<float> judgeable;
|
|
for (int rad = 0; rad <= max_radius; rad++)
|
|
if (ring_pixels[rad] >= MIN_RING_PIXELS)
|
|
judgeable.push_back(baseline[rad]);
|
|
float typical_background = 0.0f;
|
|
if (!judgeable.empty()) {
|
|
const auto middle = judgeable.begin() + judgeable.size() / 2;
|
|
std::nth_element(judgeable.begin(), middle, judgeable.end());
|
|
typical_background = *middle;
|
|
}
|
|
|
|
int blocked_out_to = -1;
|
|
for (int rad = 0; rad <= max_radius; rad++) {
|
|
if (ring_pixels[rad] < MIN_RING_PIXELS)
|
|
continue;
|
|
if (baseline[rad] >= BLOCKED_RING_RATIO * typical_background)
|
|
break;
|
|
blocked_out_to = rad;
|
|
}
|
|
return blocked_out_to;
|
|
}
|
|
|
|
void HarmonicFit(const std::vector<double> §or_median, int n_bands,
|
|
std::vector<float> &harm_c, std::vector<float> &harm_s) {
|
|
harm_c.assign(n_bands, 0.0f);
|
|
harm_s.assign(n_bands, 0.0f);
|
|
for (int band = 0; band < n_bands; band++) {
|
|
std::vector<double> med(HARMONIC_SECTORS), c(HARMONIC_SECTORS), s(HARMONIC_SECTORS);
|
|
for (int k = 0; k < HARMONIC_SECTORS; k++) {
|
|
med[k] = sector_median[static_cast<size_t>(band) * HARMONIC_SECTORS + k];
|
|
const double phi = (k + 0.5) * 2.0 * std::numbers::pi / HARMONIC_SECTORS - std::numbers::pi;
|
|
c[k] = std::cos(2 * phi);
|
|
s[k] = std::sin(2 * phi);
|
|
}
|
|
// Least squares of med = m + p cos 2phi + q sin 2phi over the sectors that are not themselves
|
|
// dim against the fit, three times over as the baseline is; the model is relative, 1 + (p cos + q sin)/m.
|
|
std::vector<char> use(HARMONIC_SECTORS);
|
|
for (int k = 0; k < HARMONIC_SECTORS; k++)
|
|
use[k] = med[k] >= 0;
|
|
for (int iter = 0; iter < 3; iter++) {
|
|
double n = 0, sc = 0, ss = 0, scc = 0, sss = 0, scs = 0, y = 0, yc = 0, ys = 0;
|
|
for (int k = 0; k < HARMONIC_SECTORS; k++) {
|
|
if (!use[k]) continue;
|
|
n++; sc += c[k]; ss += s[k]; scc += c[k] * c[k]; sss += s[k] * s[k]; scs += c[k] * s[k];
|
|
y += med[k]; yc += med[k] * c[k]; ys += med[k] * s[k];
|
|
}
|
|
// Sectors crowded into a narrow arc cannot tell a harmonic from a level; the determinant of
|
|
// the normal matrix, per sector cubed, is 1/4 on a full ring.
|
|
const double det = n * (scc * sss - scs * scs) - sc * (sc * sss - scs * ss) + ss * (sc * scs - scc * ss);
|
|
if (n < 6 || det < 0.01 * n * n * n) {
|
|
harm_c[band] = harm_s[band] = 0.0f;
|
|
break;
|
|
}
|
|
const double m = (y * (scc * sss - scs * scs) - sc * (yc * sss - scs * ys) + ss * (yc * scs - scc * ys)) / det;
|
|
const double p = (n * (yc * sss - ys * scs) - y * (sc * sss - scs * ss) + ss * (sc * ys - yc * ss)) / det;
|
|
const double q = (n * (scc * ys - scs * yc) - sc * (sc * ys - yc * ss) + y * (sc * scs - scc * ss)) / det;
|
|
if (m <= 0) {
|
|
harm_c[band] = harm_s[band] = 0.0f;
|
|
break;
|
|
}
|
|
harm_c[band] = static_cast<float>(p / m);
|
|
harm_s[band] = static_cast<float>(q / m);
|
|
for (int k = 0; k < HARMONIC_SECTORS; k++)
|
|
use[k] = med[k] >= 0 && med[k] >= PENUMBRA_RATIO * (1.0 + harm_c[band] * c[k] + harm_s[band] * s[k]);
|
|
}
|
|
}
|
|
}
|
|
|
|
} // namespace shadow_finder
|