Add fused GPU adaptive spot finder (azint + spot finding in one pass)

AdaptiveSpotFinderGPU does the per-resolution-ring reduction once on the GPU and
drives both products from it: the azimuthal-integration profile (corrected space)
and the self-calibrating adaptive spot-detection threshold (raw counts). This
replaces the separate GPU azint pass and the host-side adaptive spot finder that
runs on the GPU path today. On a ~4.5 MP detector it does both jobs in ~1 ms/frame
versus ~40 ms for the CPU adaptive finder (~42x), with an identical spot list and
azimuthal profile.

The per-ring threshold math (Poisson tail + read-floored Gaussian, operating point
from the false-pixels-per-frame knob) is factored into AdaptiveThreshold.h so the
CPU and GPU finders share one source of truth and cannot drift.

Wired opt-in via a MXAnalysisWithoutFPGA constructor flag, default on for the rugnux
offline path and the interactive viewer, off for the online receiver (so the broker
path is unchanged). When on, Analyze() skips the separate azint pass and lifts the
profile from the fused engine. The viewer gains an "Adaptive threshold" checkbox that
greys out the signal/noise and photon-count sliders (the adaptive finder uses neither).

Dedicated tests exercise both products (spot-finding parity vs the CPU finder,
azimuthal profile vs a standalone GPU azint) plus a speed benchmark. Validated
end-to-end on lysozyme serial stills: fused == CPU-adaptive index rate and merge stats.

Docs: new section 3.2 in docs/CPU_DATA_ANALYSIS.md.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-07-25 20:10:45 +02:00
co-authored by Claude Opus 4.8
parent a43f9bda13
commit b65691f312
13 changed files with 812 additions and 96 deletions
@@ -7,64 +7,7 @@
#include <unordered_map>
#include "AdaptiveSpotFinderCPU.h"
namespace {
// Number of background pixels a ring needs before its own statistics are trusted; sparser rings
// (detector corners, heavily masked, innermost) fall back to the whole-frame background.
constexpr int64_t MIN_RING_PIXELS = 40;
// Inverse standard-normal CDF (Acklam's rational approximation, ~1e-9 accuracy). Only called once
// per frame, so accuracy over speed.
double NormalQuantile(double p) {
if (p <= 0.0) return -40.0;
if (p >= 1.0) return 40.0;
static const double a[] = {-3.969683028665376e+01, 2.209460984245205e+02, -2.759285104469687e+02,
1.383577518672690e+02, -3.066479806614716e+01, 2.506628277459239e+00};
static const double b[] = {-5.447609879822406e+01, 1.615858368580409e+02, -1.556989798598866e+02,
6.680131188771972e+01, -1.328068155288572e+01};
static const double c[] = {-7.784894002430293e-03, -3.223964580411365e-01, -2.400758277161838e+00,
-2.549732539343734e+00, 4.374664141464968e+00, 2.938163982698783e+00};
static const double d[] = {7.784695709041462e-03, 3.224671290700398e-01, 2.445134137142996e+00,
3.754408661907416e+00};
const double plow = 0.02425, phigh = 1.0 - 0.02425;
if (p < plow) {
double q = std::sqrt(-2.0 * std::log(p));
return (((((c[0]*q+c[1])*q+c[2])*q+c[3])*q+c[4])*q+c[5]) /
((((d[0]*q+d[1])*q+d[2])*q+d[3])*q+1.0);
} else if (p <= phigh) {
double q = p - 0.5, r = q*q;
return (((((a[0]*r+a[1])*r+a[2])*r+a[3])*r+a[4])*r+a[5])*q /
(((((b[0]*r+b[1])*r+b[2])*r+b[3])*r+b[4])*r+1.0);
} else {
double q = std::sqrt(-2.0 * std::log(1.0 - p));
return -(((((c[0]*q+c[1])*q+c[2])*q+c[3])*q+c[4])*q+c[5]) /
((((d[0]*q+d[1])*q+d[2])*q+d[3])*q+1.0);
}
}
// Smallest integer count whose Poisson(mu) upper tail P(X >= k) <= p. This is the correct
// significance floor while the background is countable (it carries the sqrt(mu) shot-noise
// implicitly, so a bright low-resolution ring gets a high threshold). It DEGENERATES at mu -> 0
// (a single photon on a zero background is "significant"), which is why it is max'd with a
// read-noise-floored Gaussian arm by the caller. Short-circuits to Gaussian for large mu.
float PoissonThreshold(double mu, double p, double z) {
if (mu > 50.0)
return static_cast<float>(mu + z * std::sqrt(mu));
if (mu < 1e-6) mu = 1e-6;
const double target = 1.0 - p;
double pmf = std::exp(-mu);
double cdf = pmf;
int k = 0;
while (cdf < target && k < 1000) {
++k;
pmf *= mu / k;
cdf += pmf;
}
return static_cast<float>(k + 1);
}
} // namespace
#include "AdaptiveThreshold.h"
AdaptiveSpotFinderCPU::AdaptiveSpotFinderCPU(const AzimuthalIntegrationMapping &in_mapping)
: ImageSpotFinder(static_cast<int32_t>(in_mapping.GetWidth()),
@@ -142,32 +85,18 @@ std::vector<DiffractionSpot> AdaptiveSpotFinderCPU::Run(const ImagePreprocessorB
const double E = std::max(1.0f, settings.false_pixels_per_frame);
double p = E / static_cast<double>(n_total);
p = std::min(std::max(p, 1e-9), 0.1);
const float z = static_cast<float>(NormalQuantile(1.0 - p));
// A ring's threshold is background mean + z sigmas. sigma combines the ring's own (peak-excluded)
// scatter with an excess-noise floor READ: near-zero-background rings scatter MORE than pure
// Poisson (charge sharing / read noise / occasional spurious low counts), so a per-ring sigma
// alone collapses toward zero on empty high-resolution rings and the threshold would flood. READ
// is a detector-level photon-scale constant (the same for every dataset -- it is NOT the
// per-dataset knob), so the operating point still self-calibrates through mean and sigma while
// staying physical where the background vanishes.
const float READ = 1.0f;
auto ring_threshold = [&](float mean, float sigma) {
// Poisson significance (correct where the background is countable) floored by a
// read-noise-aware Gaussian arm (which alone survives mean -> 0, where Poisson degenerates
// to "one photon is significant" and would flood the empty high-resolution rings).
const float gauss = mean + z * std::sqrt(sigma * sigma + READ * READ);
const float poisson = PoissonThreshold(mean, static_cast<double>(p), static_cast<double>(z));
return std::max(gauss, poisson);
};
const float z = static_cast<float>(adaptive_threshold::NormalQuantile(1.0 - p));
// whole-frame fallback background for rings too sparse to trust on their own
const double g_mean = g_sum / n_total;
const double g_sigma = std::sqrt(std::max(0.0, g_sum2 / n_total - g_mean * g_mean));
const float g_thr = ring_threshold(static_cast<float>(g_mean), static_cast<float>(g_sigma));
const float g_thr = adaptive_threshold::RingThreshold(static_cast<float>(g_mean),
static_cast<float>(g_sigma), p, z);
for (size_t b = 0; b < nbins; ++b)
ring_thr[b] = (ring_cnt[b] < MIN_RING_PIXELS) ? g_thr : ring_threshold(ring_mean[b], ring_sigma[b]);
ring_thr[b] = (ring_cnt[b] < adaptive_threshold::MIN_RING_PIXELS)
? g_thr
: adaptive_threshold::RingThreshold(ring_mean[b], ring_sigma[b], p, z);
// --- Stage C: flag strong pixels into the bit buffer (value >= ring threshold) ---
for (size_t i = 0; i < OutputSize(); ++i)