Two fixes found in the post-merge profile, where the previous commit's loops did not do what they were written to do: - AccumulateRingsBlock: the ring_overflow.emplace_back call inside the loop left no callee-saved registers for the locals, so GCC kept sum/sum2/count on the stack and the store-reload chain was still there (29% of the function's samples on one stack add). The azint sums get a loop of their own over the block, and the out-of-histogram values are listed by a second loop run only when the block has any, in the same pixel order; now the sums live in registers. Same additions in the same order. - DetectPass: inlined into DetectPass's main loop the vertical slide was not vectorised (the vectorised copies GCC made were for the other call sites). It is now a free function, SlideVertical, which GCC vectorises on its own. Measured (perf, 75 s of myob, CPU-only, relative to the unchanged FlagRow/AnalyzeBlock): AccumulateRingsBlock -4..-11%, DetectPass + SlideVertical -7..-14%. Byte-identical p.hkl, p.mtz, p_P1.mtz, p_unmerged.mtz on myob, cytc, lyso, sparse (CPU) and myob (GPU); ImageSpotFinderCPU*, AdaptiveSpotFinder, SpotFinding, AzimuthalIntegration and portable tests pass. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
313 lines
15 KiB
C++
313 lines
15 KiB
C++
// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include <algorithm>
|
|
#include <bit>
|
|
#include <bitset>
|
|
#include <cmath>
|
|
|
|
#include "ImageSpotFinderCPU.h"
|
|
#include "StrongPixelSet.h"
|
|
|
|
ImageSpotFinderCPU::ImageSpotFinderCPU(int32_t in_width, int32_t in_height)
|
|
: ImageSpotFinder(in_width, in_height), first_pass_buffer(OutputSize(), 0) {}
|
|
|
|
void ImageSpotFinderCPU::Detect(const ImagePreprocessorBuffer &image,
|
|
const SpotFindingSettings &settings) {
|
|
// Two passes, as ImageSpotFinderGPU::Detect does. The second recomputes every local background
|
|
// with the pixels the first found strong taken out of it, and keeps those pixels strong. It
|
|
// matters because a spot wide enough to reach into its own background window inflates the mean
|
|
// and variance it is then tested against, so its outer pixels fail the SNR test on a single
|
|
// pass. The GPU has always done this; running one pass here made the two finders return
|
|
// different spot lists for the same frame.
|
|
DetectPass(image, settings, nullptr, first_pass_buffer);
|
|
DetectPass(image, settings, first_pass_buffer.data(), output_buffer);
|
|
}
|
|
|
|
namespace {
|
|
|
|
// The local-box SNR test of one pixel. sum/sum2/valid are its window's, centre pixel included.
|
|
bool StrongInWindow(int64_t pxl_val, int64_t sum, int64_t sum2, int64_t valid,
|
|
const SpotFindingSettings &settings, float strong2) {
|
|
const int64_t sum_local = sum - pxl_val;
|
|
const int64_t sum2_local = sum2 - pxl_val * pxl_val;
|
|
const int64_t valid_local = valid - 1;
|
|
|
|
const int64_t var = valid_local * sum2_local - (sum_local * sum_local);
|
|
const int64_t in_minus_mean = pxl_val * valid_local - sum_local;
|
|
|
|
return (pxl_val == INT32_MAX) // saturated pixel, or strong in the previous pass, is accepted always
|
|
|| ((pxl_val != INT32_MIN && // pixel is not bad pixel
|
|
valid_local > ImageSpotFinder::MIN_VALID_PIXELS && // too many bad pixels around will give poor statistics
|
|
(pxl_val > settings.photon_count_threshold) && // pixel is above count threshold
|
|
(in_minus_mean > 0) && // pixel value is larger than mean
|
|
(in_minus_mean * in_minus_mean > static_cast<int64_t>(std::ceil(var * strong2)))));
|
|
// pixel is above SNR threshold
|
|
}
|
|
|
|
// Adds row `in` to the vertical sums and takes row `out` out of them; nullptr is no row. A bad or
|
|
// saturated pixel adds 0. Written without branches, and as a function of its own so that the loop
|
|
// vectorises whatever it is inlined into; integer sums, so the same totals as pixel by pixel.
|
|
void SlideVertical(const int32_t *in, const int32_t *out, int32_t width,
|
|
int64_t *sum, int64_t *sum2, uint16_t *valid) {
|
|
for (int32_t col = 0; col < width; col++) {
|
|
const int64_t a = in ? in[col] : INT32_MIN;
|
|
const int64_t r = out ? out[col] : INT32_MIN;
|
|
const bool a_ok = a != INT32_MAX && a != INT32_MIN;
|
|
const bool r_ok = r != INT32_MAX && r != INT32_MIN;
|
|
const int64_t a_v = a_ok ? a : 0;
|
|
const int64_t r_v = r_ok ? r : 0;
|
|
sum[col] += a_v - r_v;
|
|
sum2[col] += a_v * a_v - r_v * r_v;
|
|
valid[col] += static_cast<uint16_t>(a_ok - r_ok);
|
|
}
|
|
}
|
|
|
|
} // namespace
|
|
|
|
void ImageSpotFinderCPU::DetectAt(const ImagePreprocessorBuffer &image, const SpotFindingSettings &settings,
|
|
const std::vector<uint32_t> &candidates,
|
|
const std::function<void(int32_t)> &fill_row) {
|
|
candidate_windows.clear();
|
|
// Filled in by DetectPass as the candidates of each row become known.
|
|
first_pass_needed.assign(static_cast<size_t>(height) * ((width + 31) / 32), 0);
|
|
DetectPass(image, settings, nullptr, first_pass_buffer, candidates.data(), fill_row);
|
|
|
|
std::fill(output_buffer.begin(), output_buffer.end(), 0);
|
|
const float strong2 = settings.signal_to_noise_threshold * settings.signal_to_noise_threshold;
|
|
const auto first_pass = [&](int32_t pxl) { return (first_pass_buffer[pxl / 32] >> (pxl % 32)) & 1U; };
|
|
|
|
for (const auto &c : candidate_windows) {
|
|
const int32_t line = c.pxl / width;
|
|
const int32_t col = c.pxl % width;
|
|
int64_t sum = c.sum, sum2 = c.sum2, valid = c.valid;
|
|
// Take out of the window what the second pass does not count: the first pass's strong pixels.
|
|
for (int32_t y = std::max(line - NBX, 0); y <= std::min(line + NBX, height - 1); y++) {
|
|
const int32_t first = y * width + std::max(col - NBX, 0);
|
|
const int32_t last = y * width + std::min(col + NBX, width - 1);
|
|
for (int32_t w = first / 32; w <= last / 32; w++) {
|
|
uint32_t bits = first_pass_buffer[w];
|
|
if (w == first / 32)
|
|
bits &= UINT32_MAX << (first % 32);
|
|
if (w == last / 32 && last % 32 != 31)
|
|
bits &= (1U << (last % 32 + 1)) - 1;
|
|
while (bits) {
|
|
const int32_t q = w * 32 + std::countr_zero(bits);
|
|
bits &= bits - 1;
|
|
const int64_t v = image[q];
|
|
if (v != INT32_MAX && v != INT32_MIN) {
|
|
sum -= v;
|
|
sum2 -= v * v;
|
|
valid -= 1;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
const int64_t pxl_val = first_pass(c.pxl) ? INT32_MAX : image[c.pxl];
|
|
if (StrongInWindow(pxl_val, sum, sum2, valid, settings, strong2))
|
|
output_buffer[c.pxl / 32] |= 1U << (c.pxl % 32);
|
|
}
|
|
}
|
|
|
|
void ImageSpotFinderCPU::DetectPass(const ImagePreprocessorBuffer &image,
|
|
const SpotFindingSettings &settings,
|
|
const uint32_t *prev_strong,
|
|
std::vector<uint32_t> &out_buffer,
|
|
const uint32_t *candidates,
|
|
const std::function<void(int32_t)> &fill_row) {
|
|
for (int i = 0; i < OutputSize(); i++)
|
|
out_buffer[i] = 0;
|
|
|
|
// The first pass's bits are read only inside the window of a candidate (and at the candidate), so
|
|
// with candidates it tests only the pixels of the row/column blocks such a window reaches; the bits
|
|
// it leaves unset there are never read. Those blocks are marked from each row's candidates as the
|
|
// row enters the vertical sums below, which is before any row within NBX of it is tested - so the
|
|
// candidates themselves can be filled in a row at a time (fill_row), while the row is in cache.
|
|
const int32_t nblocks = (width + 31) / 32;
|
|
const auto new_row = [&](int32_t r) {
|
|
if (!candidates)
|
|
return;
|
|
if (fill_row)
|
|
fill_row(r);
|
|
const int32_t first = r * width;
|
|
const int32_t last = first + width - 1;
|
|
for (int32_t w = first / 32; w <= last / 32; w++) {
|
|
uint32_t bits = candidates[w];
|
|
if (w == first / 32)
|
|
bits &= UINT32_MAX << (first % 32);
|
|
if (w == last / 32 && last % 32 != 31)
|
|
bits &= (1U << (last % 32 + 1)) - 1;
|
|
for (; bits; bits &= bits - 1) {
|
|
const int32_t col = w * 32 + std::countr_zero(bits) - first;
|
|
for (int32_t y = std::max(r - NBX, 0); y <= std::min(r + NBX, height - 1); y++)
|
|
for (int32_t b = std::max(col - NBX, 0) / 32; b <= std::min(col + NBX, width - 1) / 32; b++)
|
|
first_pass_needed[static_cast<size_t>(y) * nblocks + b] = 1;
|
|
}
|
|
}
|
|
};
|
|
|
|
// A pixel found strong by the previous pass reads as INT32_MAX, which the accumulation below
|
|
// already skips and the acceptance test below already takes as strong - the same substitution
|
|
// the GPU kernel makes when it reads prev_out.
|
|
auto value_at = [&](int32_t pxl) -> int32_t {
|
|
if (prev_strong && (prev_strong[pxl / 32] & (1U << (pxl % 32))))
|
|
return INT32_MAX;
|
|
return image[pxl];
|
|
};
|
|
|
|
std::bitset<32> out = 0;
|
|
|
|
if (settings.signal_to_noise_threshold <= 0.0) {
|
|
if (settings.photon_count_threshold > 0) {
|
|
for (int pxl = 0; pxl < height * width; pxl++) {
|
|
int32_t bit = pxl % 32;
|
|
int32_t pxl_val = value_at(pxl);
|
|
if (pxl_val == INT32_MAX || (pxl_val > settings.photon_count_threshold && pxl_val != INT32_MIN))
|
|
out.set(bit);
|
|
|
|
if (bit == 31) {
|
|
out_buffer[pxl / 32] = out.to_ulong();
|
|
out.reset();
|
|
}
|
|
}
|
|
}
|
|
} else {
|
|
float strong2 = settings.signal_to_noise_threshold * settings.signal_to_noise_threshold;
|
|
|
|
// Sum and sum of squares of (2*NBY+1) vertical elements
|
|
// These are updated after each line is finished
|
|
// 64-bit integer guarantees calculations are made without rounding errors
|
|
std::vector<int64_t> sum_vert(width, 0);
|
|
std::vector<int64_t> sum2_vert(width, 0);
|
|
std::vector<uint16_t> valid_vert(width, 0);
|
|
|
|
// Add the row entering the window (line_in) and take out the one leaving it (line_out); -1 is
|
|
// no row. A pixel strong in the previous pass is not counted either; SlideVertical adds it with
|
|
// the rest of its row and it is taken out again below - there are few, and the sums are
|
|
// integers, so the totals are the same as never adding it.
|
|
const int32_t *img = image.data();
|
|
const int32_t w = width;
|
|
int64_t *sv = sum_vert.data();
|
|
int64_t *sv2 = sum2_vert.data();
|
|
uint16_t *vv = valid_vert.data();
|
|
auto slide_vert = [&](int line_in, int line_out) {
|
|
SlideVertical(line_in >= 0 ? img + static_cast<size_t>(line_in) * w : nullptr,
|
|
line_out >= 0 ? img + static_cast<size_t>(line_out) * w : nullptr, w, sv, sv2, vv);
|
|
if (!prev_strong)
|
|
return;
|
|
// sign = +1 takes out what line_in added, -1 puts back what line_out took out.
|
|
const auto uncount_strong = [&](int line, int64_t sign) {
|
|
const int32_t first = line * w, last = first + w - 1;
|
|
for (int32_t word = first / 32; word <= last / 32; word++) {
|
|
uint32_t bits = prev_strong[word];
|
|
if (word == first / 32)
|
|
bits &= UINT32_MAX << (first % 32);
|
|
if (word == last / 32 && last % 32 != 31)
|
|
bits &= (1U << (last % 32 + 1)) - 1;
|
|
for (; bits; bits &= bits - 1) {
|
|
const int32_t pxl = word * 32 + std::countr_zero(bits);
|
|
const int64_t v = img[pxl];
|
|
if (v == INT32_MAX || v == INT32_MIN)
|
|
continue;
|
|
sv[pxl - first] -= sign * v;
|
|
sv2[pxl - first] -= sign * v * v;
|
|
vv[pxl - first] -= static_cast<uint16_t>(sign);
|
|
}
|
|
}
|
|
};
|
|
if (line_in >= 0)
|
|
uncount_strong(line_in, 1);
|
|
if (line_out >= 0)
|
|
uncount_strong(line_out, -1);
|
|
};
|
|
|
|
for (int line = 0; line < NBX; line++) {
|
|
new_row(line);
|
|
slide_vert(line, -1);
|
|
}
|
|
|
|
for (int line = 0; line < height; line++) {
|
|
if (line < height - NBX)
|
|
new_row(line + NBX);
|
|
slide_vert(line < height - NBX ? line + NBX : -1, line >= NBX + 1 ? line - (NBX + 1) : -1);
|
|
|
|
if (candidates) {
|
|
// Only the runs of 32-column blocks whose result is read (first_pass_needed; every
|
|
// candidate lies in one). Each run starts its window from the vertical sums over the
|
|
// columns it covers there - integers, so the same sums the running window holds at
|
|
// that column - and slides it as below; the bits outside the runs stay 0, as below.
|
|
const uint8_t *needed = first_pass_needed.data() + static_cast<size_t>(line) * nblocks;
|
|
for (int32_t b = 0; b < nblocks;) {
|
|
if (!needed[b]) { b++; continue; }
|
|
const int32_t c0 = b * 32;
|
|
while (b < nblocks && needed[b]) b++;
|
|
const int32_t c1 = std::min(b * 32, width);
|
|
int64_t sum = 0, sum2 = 0, valid = 0;
|
|
for (int32_t c = std::max(c0 - NBX, 0); c <= std::min(c0 + NBX, width - 1); c++) {
|
|
sum += sum_vert[c];
|
|
sum2 += sum2_vert[c];
|
|
valid += valid_vert[c];
|
|
}
|
|
for (int32_t col = c0; col < c1; col++) {
|
|
if (col > c0) {
|
|
if (col < width - NBX) {
|
|
sum += sum_vert[col + NBX];
|
|
sum2 += sum2_vert[col + NBX];
|
|
valid += valid_vert[col + NBX];
|
|
}
|
|
if (col >= NBX + 1) {
|
|
sum -= sum_vert[col - NBX - 1];
|
|
sum2 -= sum2_vert[col - NBX - 1];
|
|
valid -= valid_vert[col - NBX - 1];
|
|
}
|
|
}
|
|
const int32_t pxl = line * width + col;
|
|
if (candidates[pxl / 32] & (1U << (pxl % 32)))
|
|
candidate_windows.push_back({pxl, sum, sum2, valid});
|
|
if (StrongInWindow(value_at(pxl), sum, sum2, valid, settings, strong2))
|
|
out_buffer[pxl / 32] |= 1U << (pxl % 32);
|
|
}
|
|
}
|
|
continue;
|
|
}
|
|
|
|
int64_t sum = 0;
|
|
int64_t sum2 = 0;
|
|
int64_t valid = 0;
|
|
|
|
for (int col = 0; col < NBX; col++) {
|
|
sum += sum_vert[col];
|
|
sum2 += sum2_vert[col];
|
|
valid += valid_vert[col];
|
|
}
|
|
|
|
for (int col = 0; col < width; col++) {
|
|
if (col < width - NBX) {
|
|
sum += sum_vert[col + NBX];
|
|
sum2 += sum2_vert[col + NBX];
|
|
valid += valid_vert[col + NBX];
|
|
}
|
|
|
|
if (col >= NBX + 1) {
|
|
sum -= sum_vert[col - NBX - 1];
|
|
sum2 -= sum2_vert[col - NBX - 1];
|
|
valid -= valid_vert[col - NBX - 1];
|
|
}
|
|
const int32_t pxl = line * width + col;
|
|
const int32_t bit = pxl % 32;
|
|
|
|
if (StrongInWindow(value_at(pxl), sum, sum2, valid, settings, strong2))
|
|
out.set(bit);
|
|
|
|
if (bit == 31) {
|
|
out_buffer[pxl / 32] = out.to_ulong();
|
|
out.reset() ;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
if (height * width % 32 != 0)
|
|
out_buffer[OutputSize() - 1] |= out.to_ulong(); // zeroed above; the candidates path sets its bits directly
|
|
}
|