Files
Jungfraujoch/image_analysis/spot_finding/ImageSpotFinderCPU.cpp
T
leonarski_fandClaude Opus 5 1a1e05ad14
Build Packages / build:viewer-tgz:cpu (push) Successful in 7m50s
Build Packages / build:viewer-tgz:cuda (push) Successful in 8m38s
Build Packages / build:rpm (ubuntu2404_nocuda) (push) Successful in 13m32s
Build Packages / build:rpm (ubuntu2204_nocuda) (push) Successful in 14m17s
Build Packages / build:rpm (rocky8_nocuda) (push) Successful in 14m21s
Build Packages / build:rpm (rocky8_sls9) (push) Successful in 14m27s
Build Packages / build:rpm (rocky9_nocuda) (push) Successful in 14m39s
Build Packages / build:rpm (rocky8) (push) Successful in 11m59s
Build Packages / build:rpm (rocky9_sls9) (push) Successful in 13m8s
Build Packages / XDS test (durin plugin) (push) Successful in 7m15s
Build Packages / Generate python client (push) Successful in 24s
Build Packages / Build documentation (push) Successful in 1m5s
Build Packages / Create release (push) Skipped
Build Packages / build:rpm (rocky9) (push) Successful in 12m28s
Build Packages / build:rpm (ubuntu2404) (push) Successful in 12m50s
Build Packages / build:rpm (ubuntu2204) (push) Successful in 13m18s
Build Packages / DIALS test (push) Successful in 14m17s
Build Packages / XDS test (neggia plugin) (push) Successful in 8m9s
Build Packages / XDS test (JFJoch plugin) (push) Successful in 8m52s
Build Packages / Unit tests (push) Successful in 59m1s
Build Packages / build:windows:nocuda (push) Canceled after 0s
Build Packages / build:windows:cuda (push) Canceled after 0s
spot_finding: run the same two passes on the CPU as on the GPU
ImageSpotFinderGPU::Detect launches its kernel twice, feeding the first
pass's strong-pixel bitmap back in so the second recomputes each local
background with those pixels excluded and keeps them strong. The CPU
finder ran a single pass, so the two returned different spot lists for the
same frame and a dataset processed without a GPU did not match one
processed with it.

It matters for any spot wide enough to reach into its own 31x31 background
box: the spot inflates the mean and variance it is then tested against, so
its outer pixels fail the SNR test. On the test image added here - a 5x5
core at 300 counts with a one-pixel ring at 25 - a single pass returns the
25-pixel core and 7500 counts where two passes return the full 49 pixels
and 8100.

pxl_val also becomes int64_t, matching the GPU's pixel_result signature.
It was int32_t, so pxl_val * pxl_val overflowed above 46341 counts even
though the surrounding sums were already 64-bit.

The new parity test compares PixelCount and Count, not just the centroid,
which does not move for a symmetric spot whether or not the ring was
picked up; it was confirmed to fail against the old single-pass CPU.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-07-31 15:37:22 +02:00

157 lines
6.4 KiB
C++

// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <bitset>
#include "ImageSpotFinderCPU.h"
#include "StrongPixelSet.h"
ImageSpotFinderCPU::ImageSpotFinderCPU(int32_t in_width, int32_t in_height)
: ImageSpotFinder(in_width, in_height), first_pass_buffer(OutputSize(), 0) {}
void ImageSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image,
const SpotFindingSettings &settings) {
// Two passes, as ImageSpotFinderGPU::Detect does. The second recomputes every local background
// with the pixels the first found strong taken out of it, and keeps those pixels strong. It
// matters because a spot wide enough to reach into its own background window inflates the mean
// and variance it is then tested against, so its outer pixels fail the SNR test on a single
// pass. The GPU has always done this; running one pass here made the two finders return
// different spot lists for the same frame.
DetectPass(image, settings, nullptr, first_pass_buffer);
DetectPass(image, settings, first_pass_buffer.data(), output_buffer);
}
void ImageSpotFinderCPU::DetectPass(const ImagePreprocessorBuffer &image,
const SpotFindingSettings &settings,
const uint32_t *prev_strong,
std::vector<uint32_t> &out_buffer) {
for (int i = 0; i < OutputSize(); i++)
out_buffer[i] = 0;
// A pixel found strong by the previous pass reads as INT32_MAX, which the accumulation below
// already skips and the acceptance test below already takes as strong - the same substitution
// the GPU kernel makes when it reads prev_out.
auto value_at = [&](int32_t pxl) -> int32_t {
if (prev_strong && (prev_strong[pxl / 32] & (1U << (pxl % 32))))
return INT32_MAX;
return image[pxl];
};
std::bitset<32> out = 0;
if (settings.signal_to_noise_threshold <= 0.0) {
if (settings.photon_count_threshold > 0) {
for (int pxl = 0; pxl < height * width; pxl++) {
int32_t bit = pxl % 32;
int32_t pxl_val = value_at(pxl);
if (pxl_val == INT32_MAX || (pxl_val > settings.photon_count_threshold && pxl_val != INT32_MIN))
out.set(bit);
if (bit == 31) {
out_buffer[pxl / 32] = out.to_ulong();
out.reset();
}
}
}
} else {
float strong2 = settings.signal_to_noise_threshold * settings.signal_to_noise_threshold;
// Sum and sum of squares of (2*NBY+1) vertical elements
// These are updated after each line is finished
// 64-bit integer guarantees calculations are made without rounding errors
std::vector<int64_t> sum_vert(width, 0);
std::vector<int64_t> sum2_vert(width, 0);
std::vector<uint16_t> valid_vert(width, 0);
for (int line = 0; line < NBX; line++) {
for (int col = 0; col < width; col++) {
auto pxl = line * width + col;
int64_t tmp = value_at(pxl);
if (tmp != INT32_MAX && tmp != INT32_MIN) {
sum_vert[col] += tmp;
sum2_vert[col] += tmp * tmp;
valid_vert[col] += 1;
}
}
}
for (int line = 0; line < height; line++) {
for (int col = 0; col < width; col++) {
if (line < height - NBX) {
auto pxl = (line + NBX) * width + col;
int64_t tmp = value_at(pxl);
if (tmp != INT32_MAX && tmp != INT32_MIN) {
sum_vert[col] += tmp;
sum2_vert[col] += tmp * tmp;
valid_vert[col] += 1;
}
}
if (line >= NBX + 1) {
auto pxl = (line - (NBX + 1)) * width + col;
int64_t tmp = value_at(pxl);
if (tmp != INT32_MAX && tmp != INT32_MIN) {
sum_vert[col] -= tmp;
sum2_vert[col] -= tmp * tmp;
valid_vert[col] -= 1;
}
}
}
int64_t sum = 0;
int64_t sum2 = 0;
int64_t valid = 0;
for (int col = 0; col < NBX; col++) {
sum += sum_vert[col];
sum2 += sum2_vert[col];
valid += valid_vert[col];
}
for (int col = 0; col < width; col++) {
if (col < width - NBX) {
sum += sum_vert[col + NBX];
sum2 += sum2_vert[col + NBX];
valid += valid_vert[col + NBX];
}
if (col >= NBX + 1) {
sum -= sum_vert[col - NBX - 1];
sum2 -= sum2_vert[col - NBX - 1];
valid -= valid_vert[col - NBX - 1];
}
const int32_t pxl = line * width + col;
const int64_t pxl_val = value_at(pxl);
int64_t sum_local = sum - pxl_val;
int64_t sum2_local = sum2 - pxl_val * pxl_val;
int64_t valid_local = valid - 1;
int64_t var = valid_local * sum2_local - (sum_local * sum_local);
int64_t in_minus_mean = pxl_val * valid_local - sum_local;
const int32_t bit = pxl % 32;
if ((pxl_val == INT32_MAX) // saturated pixel, or strong in the previous pass, is accepted always
|| ((pxl_val != INT32_MIN && // pixel is not bad pixel
valid_local > MIN_VALID_PIXELS && // too many bad pixels around will give poor statistics
(pxl_val > settings.photon_count_threshold) && // pixel is above count threshold
(in_minus_mean > 0) && // pixel value is larger than mean
(in_minus_mean * in_minus_mean > static_cast<int64_t>(std::ceil(var * strong2))))))
// pixel is above SNR threshold
out.set(bit);
if (bit == 31) {
out_buffer[pxl / 32] = out.to_ulong();
out.reset() ;
}
}
}
}
if (height * width % 32 != 0)
out_buffer[OutputSize() - 1] = out.to_ulong();
}