Files
Jungfraujoch/image_analysis/beam_stop/ShadowFinder.cpp
T
jungfrauandClaude Opus 5 f12e3b6252 Parallelize the beam-stop pre-scan
The pre-scan read its sample of frames in a plain serial loop: one thread
did the HDF5 read, the decompression and the full-detector accumulation for
every frame. The cost is fixed per frame rather than per dataset, so it grew
straight with detector area - measured at 0.9 s on a 2M-pixel detector and
7.9 s on a 16M-pixel one, where it was 11% of the whole run with 47 of 48
cores idle.

Frames are now read on several workers. ShadowFinder keeps one projection per
worker so nothing is locked while an image is added, and the projections are
summed when the mask is read; the sums and counts are integers, so the result
does not depend on how the frames were spread over the workers. A shard that
never counted a pixel is skipped when the maxima are merged - it holds 0,
which would otherwise beat a genuinely negative maximum.

Worker count is capped (PRESCAN_MAX_WORKERS): a shard costs 20 bytes per
pixel, and the accumulation is memory-bound, so a handful of workers already
saturates it.

The beam-centre spot pool is stitched together in sample order after the
workers join, so frame numbering and the spot list are what the serial read
produced regardless of how the workers interleaved. A frame still joins the
pool only if it could be read.

ShadowFinder::AddImage took its decompression scratch buffer BY VALUE, so the
caller's buffer stayed empty and every frame allocated and zero-filled a fresh
full-size uncompressed image (72 MB on a 16M-pixel detector) and freed it
again. It takes a reference now, and each worker reuses one buffer.

Measured on a 16M-pixel rotation dataset: pre-scan 7.9 s -> 4.8 s, whole run
69.2 s -> 65.1 s. Results are unchanged - same shadow pixel count, same space
group, same merged reflection count and merging statistics on both a 16M and a
2M-pixel dataset, and the beam-centre path still commits the same centre.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-15 17:13:53 -04:00

458 lines
19 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "ShadowFinder.h"
#include <algorithm>
#include <cmath>
#include <limits>
#include <queue>
#include <type_traits>
#include "../../common/JFJochException.h"
// A pixel is shadow when its background is below this fraction of the background it is
// compared against.
constexpr float SHADOW_RATIO = 0.35f;
// The boundary grows outward into partially shadowed pixels down to this fraction, but no
// further than PENUMBRA_MAX_PX from the core.
constexpr float PENUMBRA_RATIO = 0.72f;
constexpr int PENUMBRA_MAX_PX = 14;
// Bridge module gaps and small breaks that the holder arm crosses.
constexpr int BRIDGE_PX = 6;
// A pixel whose maximum reaches this recorded a real reflection and is never masked - a
// beam stop cannot block a reflection that was measured.
constexpr int64_t MIN_REFLECTION = 25;
// Counts the background must have accumulated over the frames and the pooled pixels before
// a dip in it is believable. Below this a Poisson hole is indistinguishable from a shadow,
// and testing anyway masks whole detectors on low-background data.
constexpr double MIN_EXPECTED_COUNTS = 60;
// Side of the box the background is pooled over before testing. Its area is how many pixels back
// a ring's countability test, which decides where an azimuthal comparison is possible at all.
constexpr int POOL_PX = 5;
constexpr double MEAN_POOLED_PIXELS = POOL_PX * POOL_PX;
// A ring with fewer valid pixels than this says nothing about whether it was counted.
constexpr int MIN_RING_PIXELS = 32;
// Binary-image helpers on a width*height frame stored row-major as char (0/1). All run once,
// at GetMask() time; the BFS forms keep them O(pixels) rather than O(pixels * radius).
namespace {
// 8-connected dilation by `r` pixels (Chebyshev), via a multi-source BFS.
std::vector<char> dilate(const std::vector<char> &in, int W, int H, int r) {
if (r <= 0)
return in;
std::vector<int> dist(in.size(), -1);
std::queue<int> q;
for (size_t i = 0; i < in.size(); i++)
if (in[i]) { dist[i] = 0; q.push(static_cast<int>(i)); }
while (!q.empty()) {
const int i = q.front(); q.pop();
if (dist[i] >= r)
continue;
const int y = i / W, x = i % W;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if (yy < 0 || yy >= H || xx < 0 || xx >= W)
continue;
const int j = yy * W + xx;
if (dist[j] < 0) { dist[j] = dist[i] + 1; q.push(j); }
}
}
std::vector<char> out(in.size());
for (size_t i = 0; i < out.size(); i++)
out[i] = (dist[i] >= 0) ? 1 : 0;
return out;
}
// Erosion by `r` = dilation of the complement; outside the frame counts as complement.
std::vector<char> erode(const std::vector<char> &in, int W, int H, int r) {
std::vector<char> comp(in.size());
for (size_t i = 0; i < in.size(); i++)
comp[i] = !in[i];
const auto grown = dilate(comp, W, H, r);
std::vector<char> out(in.size());
for (size_t i = 0; i < out.size(); i++)
out[i] = !grown[i];
return out;
}
// Pixels of `passable` reachable from any of `seeds` (8-connected flood).
std::vector<char> flood(const std::vector<char> &passable, int W, int H, const std::vector<int> &seeds) {
std::vector<char> visited(passable.size(), 0);
std::queue<int> q;
for (const int s : seeds)
if (passable[s] && !visited[s]) { visited[s] = 1; q.push(s); }
while (!q.empty()) {
const int i = q.front(); q.pop();
const int y = i / W, x = i % W;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if (yy < 0 || yy >= H || xx < 0 || xx >= W)
continue;
const int j = yy * W + xx;
if (passable[j] && !visited[j]) { visited[j] = 1; q.push(j); }
}
}
return visited;
}
// Fill holes: background not reachable from the image border becomes region.
std::vector<char> fill_holes(const std::vector<char> &region, int W, int H) {
std::vector<char> bg_visited(region.size(), 0);
std::queue<int> q;
auto push = [&](int i) { if (!region[i] && !bg_visited[i]) { bg_visited[i] = 1; q.push(i); } };
for (int x = 0; x < W; x++) { push(x); push((H - 1) * W + x); }
for (int y = 0; y < H; y++) { push(y * W); push(y * W + W - 1); }
while (!q.empty()) {
const int i = q.front(); q.pop();
const int y = i / W, x = i % W;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if (yy < 0 || yy >= H || xx < 0 || xx >= W)
continue;
const int j = yy * W + xx;
if (!region[j] && !bg_visited[j]) { bg_visited[j] = 1; q.push(j); }
}
}
std::vector<char> out = region;
for (size_t i = 0; i < out.size(); i++)
if (!region[i] && !bg_visited[i])
out[i] = 1;
return out;
}
// Sum of `in` over the k x k box centred on each pixel, zero outside the frame.
std::vector<double> box_sum(const std::vector<double> &in, int W, int H, int k) {
const int half = k / 2;
std::vector<double> row(in.size(), 0.0), out(in.size(), 0.0);
for (int y = 0; y < H; y++) {
double s = 0;
for (int x = 0; x <= std::min(half, W - 1); x++)
s += in[y * W + x];
for (int x = 0; x < W; x++) {
row[y * W + x] = s;
if (x + half + 1 < W) s += in[y * W + x + half + 1];
if (x - half >= 0) s -= in[y * W + x - half];
}
}
for (int x = 0; x < W; x++) {
double s = 0;
for (int y = 0; y <= std::min(half, H - 1); y++)
s += row[y * W + x];
for (int y = 0; y < H; y++) {
out[y * W + x] = s;
if (y + half + 1 < H) s += row[(y + half + 1) * W + x];
if (y - half >= 0) s -= row[(y - half) * W + x];
}
}
return out;
}
// Median of `values` per integer radius, over the pixels flagged in `use`.
std::vector<float> ring_median(const std::vector<float> &values, const std::vector<char> &use,
const std::vector<int> &radius, int max_radius) {
std::vector<std::vector<float>> bins(max_radius + 1);
for (size_t i = 0; i < values.size(); i++)
if (use[i])
bins[radius[i]].push_back(values[i]);
std::vector<float> out(max_radius + 1, 0.0f);
for (int r = 0; r <= max_radius; r++) {
auto &b = bins[r];
if (!b.empty()) {
const size_t k = b.size() / 2;
std::nth_element(b.begin(), b.begin() + k, b.end());
out[r] = b[k];
}
}
return out;
}
} // namespace
ShadowFinder::ShadowFinder(const DiffractionExperiment &experiment, const PixelMask &mask)
: width(static_cast<int>(experiment.GetXPixelsNumConv())),
height(static_cast<int>(experiment.GetYPixelsNumConv())),
beam_x(experiment.GetBeamX_pxl()),
beam_y(experiment.GetBeamY_pxl()),
pixel_mask(mask.GetMask(experiment)) {
if (pixel_mask.size() != static_cast<size_t>(width) * height)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: pixel mask does not match the detector");
SetShardCount(1);
}
void ShadowFinder::SetShardCount(size_t n) {
const size_t npixels = static_cast<size_t>(width) * height;
shards.clear();
shards.resize(std::max<size_t>(1, n));
for (auto &p : shards) {
p.max_value.assign(npixels, 0);
p.sum_value.assign(npixels, 0);
p.valid_count.assign(npixels, 0);
p.frames = 0;
}
}
ShadowFinder::Projection ShadowFinder::Reduce() const {
if (shards.size() == 1)
return shards[0];
Projection out;
const size_t npixels = static_cast<size_t>(width) * height;
out.max_value.assign(npixels, 0);
out.sum_value.assign(npixels, 0);
out.valid_count.assign(npixels, 0);
for (const auto &p : shards) {
out.frames += p.frames;
for (size_t i = 0; i < npixels; i++) {
if (p.valid_count[i] == 0)
continue;
if (out.valid_count[i] == 0 || p.max_value[i] > out.max_value[i])
out.max_value[i] = p.max_value[i];
out.sum_value[i] += p.sum_value[i];
out.valid_count[i] += p.valid_count[i];
}
}
return out;
}
template<class T>
void ShadowFinder::Add(const T *ptr, Projection &p) {
// The pixel type's sentinel extreme marks "no data" (module gap / masked): the
// preprocessor/writer stores INT*_MIN for signed and UINT*_MAX for unsigned. For signed
// types the opposite extreme is a genuine saturated value and is kept, so a saturated
// reflection still registers as bright.
T masked;
if constexpr (std::is_signed_v<T>)
masked = std::numeric_limits<T>::min();
else
masked = std::numeric_limits<T>::max();
for (size_t i = 0; i < p.max_value.size(); i++) {
const T v = ptr[i];
if (v == masked)
continue;
const int64_t vi = static_cast<int64_t>(v);
if (p.valid_count[i] == 0 || vi > p.max_value[i])
p.max_value[i] = vi;
p.sum_value[i] += vi;
p.valid_count[i]++;
}
p.frames++;
}
void ShadowFinder::AddImage(const DataMessage &data, std::vector<uint8_t> &buffer, size_t shard) {
if (static_cast<size_t>(data.image.GetWidth()) * data.image.GetHeight()
!= static_cast<size_t>(width) * height)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: image size does not match the detector");
if (shard >= shards.size())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: shard out of range");
Projection &p = shards[shard];
const auto ptr = data.image.GetUncompressedPtr(buffer);
switch (data.image.GetMode()) {
case CompressedImageMode::Int8: Add(reinterpret_cast<const int8_t *>(ptr), p); break;
case CompressedImageMode::Uint8: Add(reinterpret_cast<const uint8_t *>(ptr), p); break;
case CompressedImageMode::Int16: Add(reinterpret_cast<const int16_t *>(ptr), p); break;
case CompressedImageMode::Uint16: Add(reinterpret_cast<const uint16_t *>(ptr), p); break;
case CompressedImageMode::Int32: Add(reinterpret_cast<const int32_t *>(ptr), p); break;
case CompressedImageMode::Uint32: Add(reinterpret_cast<const uint32_t *>(ptr), p); break;
default:
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"ShadowFinder: unsupported image mode");
}
}
uint32_t ShadowFinder::GetFrameCount() const {
std::unique_lock ul(m);
uint32_t frames = 0;
for (const auto &p : shards) frames += p.frames;
return frames;
}
std::vector<float> ShadowFinder::GetMeanProjection() const {
std::unique_lock ul(m);
const Projection p = Reduce();
const auto &sum_value = p.sum_value;
const auto &valid_count = p.valid_count;
std::vector<float> mean(static_cast<size_t>(width) * height, NAN);
for (size_t i = 0; i < mean.size(); i++)
if (valid_count[i] > 0 && pixel_mask[i] == 0)
mean[i] = static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]);
return mean;
}
std::vector<uint32_t> ShadowFinder::GetMask() const {
std::unique_lock ul(m);
const Projection p = Reduce();
const auto &max_value = p.max_value;
const auto &sum_value = p.sum_value;
const auto &valid_count = p.valid_count;
const uint32_t frames = p.frames;
const int W = width, H = height;
const int n_pixels = W * H;
std::vector<uint32_t> mask(n_pixels, 0);
if (frames == 0)
return mask;
// mean projection, usable pixels and radius from the beam centre
std::vector<float> mean(n_pixels, 0.0f);
std::vector<char> valid(n_pixels, 0);
std::vector<int> radius(n_pixels, 0);
int max_radius = 0;
for (int y = 0; y < H; y++)
for (int x = 0; x < W; x++) {
const int i = y * W + x;
if (valid_count[i] > 0 && pixel_mask[i] == 0) {
mean[i] = static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]);
valid[i] = 1;
}
const float dx = x - beam_x, dy = y - beam_y;
radius[i] = static_cast<int>(std::lround(std::sqrt(dx * dx + dy * dy)));
max_radius = std::max(max_radius, radius[i]);
}
// Pool the background over a small box before testing it. A background of a fraction of
// a count per pixel per frame gives no single pixel enough counts to tell a shadow from
// a Poisson hole; the stop and its arm are wider than the box, so pooling costs no
// resolution that matters and multiplies the statistics by the pixels in the box.
std::vector<double> num(n_pixels), den(n_pixels);
for (int i = 0; i < n_pixels; i++) {
num[i] = valid[i] ? mean[i] : 0.0;
den[i] = valid[i] ? 1.0 : 0.0;
}
const auto pooled_sum = box_sum(num, W, H, POOL_PX);
const auto pooled_count = box_sum(den, W, H, POOL_PX);
std::vector<float> pooled(n_pixels, 0.0f);
for (int i = 0; i < n_pixels; i++)
if (pooled_count[i] > 0)
pooled[i] = static_cast<float>(pooled_sum[i] / pooled_count[i]);
// Azimuthal comparison: the median of the ring, iterated so the shadow stays out of the
// baseline it is measured against.
std::vector<float> ratio(n_pixels, 1.0f);
std::vector<char> excluded(n_pixels, 0);
std::vector<float> baseline;
for (int iter = 0; iter < 3; iter++) {
std::vector<char> use(n_pixels);
for (int i = 0; i < n_pixels; i++)
use[i] = valid[i] && !excluded[i];
baseline = ring_median(pooled, use, radius, max_radius);
for (int i = 0; i < n_pixels; i++)
if (valid[i])
ratio[i] = pooled[i] / std::max(baseline[radius[i]], 1e-6f);
for (int i = 0; i < n_pixels; i++)
excluded[i] = valid[i] && ratio[i] < SHADOW_RATIO;
}
// A ring whose background was never counted carries no information to test a pixel against.
// Walking outward, every ring before the first countable one lies wholly inside the stop - a
// ring fully within the disk has no unshadowed pixel for the median to find, which is exactly
// where an azimuthal comparison must fail. Those rings are shadow in their entirety.
// Innermost rings hold only a handful of pixels, too few to judge, so they are stepped over
// rather than allowed to end the walk.
std::vector<int> ring_pixels(max_radius + 1, 0);
for (int i = 0; i < n_pixels; i++)
if (valid[i])
ring_pixels[radius[i]]++;
// A ring lies inside the stop when its background is a fraction of the background further out.
// Counting statistics cannot decide this: on a bright dataset the shadow is still well counted.
// The comparison is only ever used to answer "is this whole ring inside the stop", never to
// judge an individual pixel, so taking the largest background over an outward window is safe
// here in a way it would not be per pixel. It is taken only over the rings this same walk is
// willing to judge, though: at the corner of the detector a ring holds a handful of pixels and
// its median is one pixel's mean, so one recorded reflection out there would otherwise become
// the background every ring inside it is compared against.
std::vector<float> outward_max(max_radius + 2, 0.0f);
for (int rad = max_radius; rad >= 0; rad--)
outward_max[rad] = std::max(ring_pixels[rad] >= MIN_RING_PIXELS ? baseline[rad] : 0.0f,
outward_max[rad + 1]);
int blocked_out_to = -1;
for (int rad = 0; rad <= max_radius; rad++) {
if (ring_pixels[rad] < MIN_RING_PIXELS)
continue;
if (baseline[rad] >= SHADOW_RATIO * outward_max[rad])
break;
blocked_out_to = rad;
}
std::vector<char> low(n_pixels, 0);
for (int i = 0; i < n_pixels; i++) {
if (!valid[i])
continue;
if (radius[i] <= blocked_out_to) {
low[i] = 1;
continue;
}
const double counted = frames * pooled_count[i];
low[i] = ratio[i] < SHADOW_RATIO && baseline[radius[i]] * counted >= MIN_EXPECTED_COUNTS;
}
// The shadow is the low region connected to the beam centre, bridging the gaps it crosses.
const std::vector<char> bridged = dilate(low, W, H, BRIDGE_PX);
std::vector<int> seeds;
for (int i = 0; i < n_pixels; i++)
if (radius[i] < 4)
seeds.push_back(i);
const std::vector<char> connected = flood(bridged, W, H, seeds);
std::vector<char> region(n_pixels);
for (int i = 0; i < n_pixels; i++)
region[i] = low[i] && connected[i];
// Recorded reflections. A small cluster is required so a single-frame zinger does not count.
std::vector<char> lit(n_pixels, 0);
for (int i = 0; i < n_pixels; i++)
lit[i] = (valid_count[i] > 0) && (max_value[i] >= MIN_REFLECTION);
std::vector<char> reflection(n_pixels, 0);
for (int y = 0; y < H; y++)
for (int x = 0; x < W; x++) {
const int i = y * W + x;
if (!lit[i]) continue;
int neighbours = 0;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if ((dx || dy) && yy >= 0 && yy < H && xx >= 0 && xx < W && lit[yy * W + xx])
neighbours++;
}
reflection[i] = (neighbours >= 2);
}
// Grow the soft boundary, round it and fill the disk interior.
const std::vector<char> penumbra = dilate(region, W, H, PENUMBRA_MAX_PX);
for (int i = 0; i < n_pixels; i++)
if (penumbra[i] && valid[i] && ratio[i] < PENUMBRA_RATIO)
region[i] = 1;
region = erode(dilate(region, W, H, 2), W, H, 2);
region = fill_holes(region, W, H);
// Expose recorded reflections - done last, with no fill afterwards, so a spot the shadow
// still covered is given back rather than re-enclosed.
const std::vector<char> reflection_grown = dilate(reflection, W, H, 1);
for (int i = 0; i < n_pixels; i++)
if (reflection_grown[i])
region[i] = 0;
for (int i = 0; i < n_pixels; i++)
mask[i] = region[i] ? 1 : 0;
return mask;
}