The CPU-only image loop is DRAM-bound on 16M frames: decode, preprocess, the adaptive finder's plain ring pass and FlagRings each streamed the whole frame through memory. - JFJochDecompressHperfBlocks hands each decoded bitshuffle block to a callback; with no output buffer the block is unshuffled into a reused block-sized scratch (JFJochDecompressBlocks). - MXAnalysisWithoutFPGA::PreprocessCPU preprocesses each block into the int32 buffer (ImagePreprocessorCPU::AnalyzeBlock) and, when the fused CPU finder runs, puts it through the plain ring pass + fused azint (AdaptiveSpotFinderCPU::AccumulateRingsBlock) while it is in cache. Detect() then starts from those sums. The per-worker decompression buffer is no longer allocated for bitshuffled data. - FlagRings becomes FlagRow, called by DetectAt's first pass for row y+NBX just before that row enters the vertical sums; first_pass_needed is marked from each row's candidates at the same point. Exact: blocks arrive in pixel order, so the float azint sums see the same pixels in the same order; the per-pixel expressions are unchanged; everything else is integer. p.hkl, p.mtz, p_P1.mtz and p_unmerged.mtz byte-identical to rc173 on myob, cytc, lyso, sparse (CPU-only build), GPU myob identical (the GPU path does not take this route). CPU-only, 32 workers, under the gpulock on a shared (loaded) machine, base -> fused, two rounds (second in reversed order): myob 155.9 -> 97.7 s, 139.9 -> 88.1 s (loop 46.1 -> 26.8 s/pass; user 3690 -> 2221 s) cytc 220.8 -> 156.6 s, 216.6 -> 155.1 s (user 5160 -> 4128 s) lyso 134.5 -> 123.3 s, 69.5 -> 64.8 s peak RSS myob 15.2 -> 10.8 GB, cytc 12.6 -> 10.4 GB, lyso 7.0 -> 6.7 GB Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
215 lines
8.7 KiB
C++
215 lines
8.7 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include <algorithm>
|
|
#include <cmath>
|
|
|
|
#include "AdaptiveSpotFinderCPU.h"
|
|
#include "AdaptiveThreshold.h"
|
|
|
|
AdaptiveSpotFinderCPU::AdaptiveSpotFinderCPU(const AzimuthalIntegrationMapping &in_mapping)
|
|
: ImageSpotFinderCPU(static_cast<int32_t>(in_mapping.GetWidth()),
|
|
static_cast<int32_t>(in_mapping.GetHeight())),
|
|
mapping(in_mapping) {
|
|
const size_t nbins = mapping.GetBinNumber();
|
|
ring_sum.assign(nbins, 0);
|
|
ring_sum2.assign(nbins, 0);
|
|
ring_cnt.assign(nbins, 0);
|
|
ring_mean.assign(nbins, 0.0f);
|
|
ring_sigma.assign(nbins, 0.0f);
|
|
ring_thr.assign(nbins, 0.0f);
|
|
ring_bkg.assign(nbins, NAN);
|
|
ring_bits.assign(OutputSize(), 0);
|
|
ring_hist.assign(nbins * HIST_VALUES, 0);
|
|
azint_sum.assign(nbins, 0.0f);
|
|
azint_sum2.assign(nbins, 0.0f);
|
|
azint_count.assign(nbins, 0);
|
|
}
|
|
|
|
void AdaptiveSpotFinderCPU::GetProfile(AzimuthalIntegrationProfile &profile) const {
|
|
profile.Clear(mapping);
|
|
profile.Add(azint_sum, azint_sum2, azint_count);
|
|
}
|
|
|
|
// Per-ring background statistics with iterated peak exclusion following peakfinder8:
|
|
// Barty et al. (2014) J. Appl. Cryst. 47, 1118-1131
|
|
// Per-ring mean/variance from the raw (photon) image: a plain pass over every valid pixel, then
|
|
// sigma-clip passes keeping only pixels within clip_k sigma of the current ring mean, which removes
|
|
// the Bragg peaks from the background estimate.
|
|
void AdaptiveSpotFinderCPU::ResetRings() {
|
|
std::fill(ring_sum.begin(), ring_sum.end(), 0);
|
|
std::fill(ring_sum2.begin(), ring_sum2.end(), 0);
|
|
std::fill(ring_cnt.begin(), ring_cnt.end(), 0);
|
|
std::fill(ring_hist.begin(), ring_hist.end(), 0);
|
|
ring_overflow.clear();
|
|
if (fuse_azint) {
|
|
std::fill(azint_sum.begin(), azint_sum.end(), 0.0f);
|
|
std::fill(azint_sum2.begin(), azint_sum2.end(), 0.0f);
|
|
std::fill(azint_count.begin(), azint_count.end(), 0);
|
|
}
|
|
}
|
|
|
|
void AdaptiveSpotFinderCPU::BeginRings() {
|
|
ResetRings();
|
|
rings_from_blocks = true;
|
|
}
|
|
|
|
// The plain pass over pixels [first, first + n).
|
|
void AdaptiveSpotFinderCPU::AccumulateRingsBlock(const ImagePreprocessorBuffer &image, size_t first, size_t n) {
|
|
const auto &pixel_to_bin = mapping.GetPixelToBin();
|
|
const size_t nbins = ring_sum.size();
|
|
const float *corrections = mapping.Corrections().data();
|
|
|
|
for (size_t pxl = first; pxl < first + n; ++pxl) {
|
|
const int32_t v = image[pxl];
|
|
if (v == INT32_MIN || v == INT32_MAX) continue; // bad / saturated
|
|
const uint16_t b = pixel_to_bin[pxl];
|
|
if (b >= nbins) continue; // masked / out of range (UINT16_MAX)
|
|
if (fuse_azint) {
|
|
const float val = static_cast<float>(v) * corrections[pxl];
|
|
const float val_sq = val * val;
|
|
azint_sum[b] += val;
|
|
azint_sum2[b] += val_sq;
|
|
++azint_count[b];
|
|
}
|
|
ring_sum[b] += v;
|
|
ring_sum2[b] += static_cast<uint64_t>(static_cast<int64_t>(v) * v);
|
|
ring_cnt[b] += 1;
|
|
if (v >= 0 && v < HIST_VALUES)
|
|
ring_hist[b * HIST_VALUES + v] += 1;
|
|
else
|
|
ring_overflow.emplace_back(b, v);
|
|
}
|
|
}
|
|
|
|
// A sigma-clip pass over the plain pass's values: each distinct value of a ring meets the same test the
|
|
// pixels holding it would, and its pixels are added as a count.
|
|
void AdaptiveSpotFinderCPU::ClipRings(float clip_k) {
|
|
const size_t nbins = ring_sum.size();
|
|
|
|
std::fill(ring_sum.begin(), ring_sum.end(), 0);
|
|
std::fill(ring_sum2.begin(), ring_sum2.end(), 0);
|
|
std::fill(ring_cnt.begin(), ring_cnt.end(), 0);
|
|
|
|
const auto keep = [&](uint16_t b, int32_t v) {
|
|
const float lo = ring_mean[b] - clip_k * ring_sigma[b];
|
|
const float hi = ring_mean[b] + clip_k * ring_sigma[b];
|
|
return !(v < lo || v > hi); // exclude peaks / outliers
|
|
};
|
|
for (size_t b = 0; b < nbins; ++b)
|
|
for (int32_t v = 0; v < HIST_VALUES; ++v) {
|
|
const uint32_t n = ring_hist[b * HIST_VALUES + v];
|
|
if (n == 0 || !keep(static_cast<uint16_t>(b), v)) continue;
|
|
ring_sum[b] += static_cast<int64_t>(n) * v;
|
|
ring_sum2[b] += static_cast<uint64_t>(n) * static_cast<uint64_t>(static_cast<int64_t>(v) * v);
|
|
ring_cnt[b] += n;
|
|
}
|
|
for (const auto &[b, v] : ring_overflow) {
|
|
if (!keep(b, v)) continue;
|
|
ring_sum[b] += v;
|
|
ring_sum2[b] += static_cast<uint64_t>(static_cast<int64_t>(v) * v);
|
|
ring_cnt[b] += 1;
|
|
}
|
|
}
|
|
|
|
void AdaptiveSpotFinderCPU::UpdateRingStatistics() {
|
|
const size_t nbins = ring_sum.size();
|
|
for (size_t b = 0; b < nbins; ++b) {
|
|
if (ring_cnt[b] > 0) {
|
|
const double m = static_cast<double>(ring_sum[b]) / ring_cnt[b];
|
|
const double var = std::max(0.0, static_cast<double>(ring_sum2[b]) / ring_cnt[b] - m * m);
|
|
ring_mean[b] = static_cast<float>(m);
|
|
ring_sigma[b] = static_cast<float>(std::sqrt(var));
|
|
}
|
|
}
|
|
}
|
|
|
|
void AdaptiveSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image,
|
|
const SpotFindingSettings &settings) {
|
|
const size_t nbins = ring_sum.size();
|
|
|
|
// --- Stage A: robust per-ring background (one plain pass + two sigma-clip passes) ---
|
|
if (!rings_from_blocks) {
|
|
ResetRings();
|
|
AccumulateRingsBlock(image, 0, static_cast<size_t>(width) * height);
|
|
}
|
|
rings_from_blocks = false;
|
|
UpdateRingStatistics();
|
|
ClipRings(3.0f);
|
|
UpdateRingStatistics();
|
|
ClipRings(3.0f);
|
|
UpdateRingStatistics();
|
|
|
|
// --- Stage B: per-ring threshold from the single portable knob E (false pixels / frame) ---
|
|
int64_t n_total = 0;
|
|
double g_sum = 0.0, g_sum2 = 0.0;
|
|
for (size_t b = 0; b < nbins; ++b) {
|
|
n_total += ring_cnt[b];
|
|
g_sum += static_cast<double>(ring_sum[b]);
|
|
g_sum2 += static_cast<double>(ring_sum2[b]);
|
|
}
|
|
if (n_total == 0) {
|
|
// Nothing valid to threshold against: leave no strong pixels for ExtractSpots to build on.
|
|
std::fill(output_buffer.begin(), output_buffer.end(), 0);
|
|
std::fill(ring_bkg.begin(), ring_bkg.end(), NAN);
|
|
return;
|
|
}
|
|
|
|
for (size_t b = 0; b < nbins; ++b)
|
|
ring_bkg[b] = (ring_cnt[b] < adaptive_threshold::MIN_RING_PIXELS) ? NAN : ring_mean[b];
|
|
|
|
const double E = std::max(1.0f, settings.false_pixels_per_frame);
|
|
double p = E / static_cast<double>(n_total);
|
|
p = std::min(std::max(p, 1e-9), 0.1);
|
|
const float z = static_cast<float>(adaptive_threshold::NormalQuantile(1.0 - p));
|
|
|
|
// whole-frame fallback background for rings too sparse to trust on their own
|
|
const double g_mean = g_sum / n_total;
|
|
const double g_sigma = std::sqrt(std::max(0.0, g_sum2 / n_total - g_mean * g_mean));
|
|
const float g_thr = adaptive_threshold::RingThreshold(static_cast<float>(g_mean),
|
|
static_cast<float>(g_sigma), p, z);
|
|
|
|
for (size_t b = 0; b < nbins; ++b)
|
|
ring_thr[b] = (ring_cnt[b] < adaptive_threshold::MIN_RING_PIXELS)
|
|
? g_thr
|
|
: adaptive_threshold::RingThreshold(ring_mean[b], ring_sigma[b], p, z);
|
|
|
|
// --- Stage C: the ring threshold, intersected with the classic local-box SNR test ---
|
|
std::fill(ring_bits.begin(), ring_bits.end(), 0);
|
|
|
|
if (settings.signal_to_noise_threshold <= 0.0f) {
|
|
// No local test asked for: the ring threshold alone decides, as the fixed photon floor
|
|
// alone would in the classic finder.
|
|
for (int32_t row = 0; row < height; row++)
|
|
FlagRow(image, row);
|
|
output_buffer = ring_bits;
|
|
return;
|
|
}
|
|
|
|
// The ring threshold IS the photon floor here, so the local pass must not apply another one.
|
|
SpotFindingSettings local = settings;
|
|
local.photon_count_threshold = 0;
|
|
// Only the ring pixels survive the intersection, so the local test is asked of those alone. They
|
|
// are flagged row by row as the local pass reaches each row, which saves reading the image for it.
|
|
DetectAt(image, local, ring_bits, [&](int32_t row) { FlagRow(image, row); });
|
|
}
|
|
|
|
void AdaptiveSpotFinderCPU::FlagRow(const ImagePreprocessorBuffer &image, int32_t row) {
|
|
const auto &pixel_to_bin = mapping.GetPixelToBin();
|
|
const size_t nbins = ring_thr.size();
|
|
const size_t first = static_cast<size_t>(row) * width;
|
|
|
|
for (size_t pxl = first; pxl < first + width; ++pxl) {
|
|
const int32_t v = image[pxl];
|
|
const uint16_t b = pixel_to_bin[pxl];
|
|
bool strong = false;
|
|
if (v == INT32_MAX)
|
|
strong = true;
|
|
else if (v != INT32_MIN && b < nbins && v >= ring_thr[b])
|
|
strong = true;
|
|
|
|
if (strong)
|
|
ring_bits[pxl / 32] |= 1U << (pxl % 32);
|
|
}
|
|
}
|