AdaptiveSpotFinderCPU::AccumulateRings: consecutive pixels mostly share a ring, so the ring's sum/sum2/count and the fused azint sums are held in locals while they do and stored when the ring changes - the same additions in the same order (az_sum2 is still contracted to the same FMA), without a store-and-reload chain through memory on every pixel. ImageSpotFinderCPU::DetectPass: the vertical-sum update (add the entering row, take out the leaving one) is one branch-free loop over the raw image that GCC vectorises (int64 lanes). A pixel strong in the previous pass used to be substituted per pixel through a bit test, which kept the loop scalar; it is now added with its row and taken out again from the few set bits of prev_strong. Integer sums, so the same totals. (A first, fully branch-free version that kept the per-pixel bit test did not vectorise on the prev_strong path and was measured slower; this is its replacement.) ImagePreprocessorCPU: the per-engine std::vector<bool> built bit by bit from the 32-bit mask (~10 core-s per cytc run, one per worker per pass) is replaced by 32-pixel mask words that PixelMask derives once beside its binary mask; each engine copies 2 MB. A branch-free rewrite of the Analyze loop was measured and dropped: the loop is bound by reading the decompressed image (330 vs 328 core-s on cytc), so only the mask test changed. Measured (perf, 499 Hz, CPU-only build, cytc, first version of this change): AccumulateRings 591 -> 539 core-s. Byte-identical p.hkl, p.mtz, p_P1.mtz, p_unmerged.mtz on myob, cytc, lyso, sparse (CPU) and myob, lyso (GPU). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
65 lines
2.9 KiB
C++
65 lines
2.9 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "ImagePreprocessorCPU.h"
|
|
|
|
ImagePreprocessorCPU::ImagePreprocessorCPU(const DiffractionExperiment &experiment, const PixelMask &mask)
|
|
: ImagePreprocessor(experiment),
|
|
mask_bits(mask.GetPackedMask()) {}
|
|
|
|
ImageStatistics ImagePreprocessorCPU::Analyze(ImagePreprocessorBuffer &processed_image, const uint8_t *image_ptr, CompressedImageMode image_mode) {
|
|
switch (image_mode) {
|
|
case CompressedImageMode::Int8:
|
|
return Analyze<int8_t>(processed_image, image_ptr, INT8_MIN, INT8_MAX);
|
|
case CompressedImageMode::Int16:
|
|
return Analyze<int16_t>(processed_image, image_ptr, INT16_MIN, INT16_MAX);
|
|
case CompressedImageMode::Int32:
|
|
return Analyze<int32_t>(processed_image, image_ptr, INT32_MIN, INT32_MAX);
|
|
case CompressedImageMode::Uint8:
|
|
return Analyze<uint8_t>(processed_image, image_ptr, UINT8_MAX, UINT8_MAX);
|
|
case CompressedImageMode::Uint16:
|
|
return Analyze<uint16_t>(processed_image, image_ptr, UINT16_MAX, UINT16_MAX);
|
|
case CompressedImageMode::Uint32:
|
|
return Analyze<uint32_t>(processed_image, image_ptr, UINT32_MAX, UINT32_MAX);
|
|
default:
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "RGB/float mode not supported");
|
|
}
|
|
}
|
|
|
|
template<class T>
|
|
ImageStatistics ImagePreprocessorCPU::Analyze(ImagePreprocessorBuffer &processed_image, const uint8_t *input, T err_pixel_val, T sat_pixel_val) {
|
|
|
|
if (processed_image.size() != npixels)
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Processed image size mismatch");
|
|
|
|
auto image = reinterpret_cast<const T *>(input);
|
|
|
|
ImageStatistics ret{};
|
|
|
|
if (sat_pixel_val > saturation_limit)
|
|
sat_pixel_val = static_cast<T>(saturation_limit);
|
|
|
|
for (int i = 0; i < npixels; i++) {
|
|
if ((mask_bits[i / 32] >> (i % 32)) & 1U) {
|
|
processed_image[i] = INT32_MIN;
|
|
++ret.masked_pixel_count;
|
|
} else if (image[i] == err_pixel_val) {
|
|
// Error/invalid marker = the pixel type's extreme value (0xFFFFFFFF for EIGER uint32).
|
|
// Tested before saturation, since for unsigned types the marker also exceeds sat_pixel_val
|
|
// (which is clipped above to the HDF5 saturation_value).
|
|
processed_image[i] = INT32_MIN;
|
|
++ret.error_pixel_count;
|
|
} else if (image[i] >= sat_pixel_val) {
|
|
processed_image[i] = INT32_MAX;
|
|
++ret.saturated_pixel_count;
|
|
} else {
|
|
processed_image[i] = static_cast<int32_t>(image[i]);
|
|
if (image[i] > ret.max_value)
|
|
ret.max_value = image[i];
|
|
if (image[i] < ret.min_value)
|
|
ret.min_value = image[i];
|
|
}
|
|
}
|
|
return ret;
|
|
}
|