The CPU-only image loop is DRAM-bound on 16M frames: decode, preprocess, the adaptive finder's plain ring pass and FlagRings each streamed the whole frame through memory. - JFJochDecompressHperfBlocks hands each decoded bitshuffle block to a callback; with no output buffer the block is unshuffled into a reused block-sized scratch (JFJochDecompressBlocks). - MXAnalysisWithoutFPGA::PreprocessCPU preprocesses each block into the int32 buffer (ImagePreprocessorCPU::AnalyzeBlock) and, when the fused CPU finder runs, puts it through the plain ring pass + fused azint (AdaptiveSpotFinderCPU::AccumulateRingsBlock) while it is in cache. Detect() then starts from those sums. The per-worker decompression buffer is no longer allocated for bitshuffled data. - FlagRings becomes FlagRow, called by DetectAt's first pass for row y+NBX just before that row enters the vertical sums; first_pass_needed is marked from each row's candidates at the same point. Exact: blocks arrive in pixel order, so the float azint sums see the same pixels in the same order; the per-pixel expressions are unchanged; everything else is integer. p.hkl, p.mtz, p_P1.mtz and p_unmerged.mtz byte-identical to rc173 on myob, cytc, lyso, sparse (CPU-only build), GPU myob identical (the GPU path does not take this route). CPU-only, 32 workers, under the gpulock on a shared (loaded) machine, base -> fused, two rounds (second in reversed order): myob 155.9 -> 97.7 s, 139.9 -> 88.1 s (loop 46.1 -> 26.8 s/pass; user 3690 -> 2221 s) cytc 220.8 -> 156.6 s, 216.6 -> 155.1 s (user 5160 -> 4128 s) lyso 134.5 -> 123.3 s, 69.5 -> 64.8 s peak RSS myob 15.2 -> 10.8 GB, cytc 12.6 -> 10.4 GB, lyso 7.0 -> 6.7 GB Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
73 lines
3.4 KiB
C++
73 lines
3.4 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "ImagePreprocessorCPU.h"
|
|
|
|
ImagePreprocessorCPU::ImagePreprocessorCPU(const DiffractionExperiment &experiment, const PixelMask &mask)
|
|
: ImagePreprocessor(experiment),
|
|
mask_1bit(npixels, false) {
|
|
for (int i = 0; i < npixels; i++)
|
|
mask_1bit[i] = (mask.GetMask().at(i) != 0);
|
|
}
|
|
|
|
ImageStatistics ImagePreprocessorCPU::Analyze(ImagePreprocessorBuffer &processed_image, const uint8_t *image_ptr, CompressedImageMode image_mode) {
|
|
if (processed_image.size() != npixels)
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Processed image size mismatch");
|
|
|
|
ImageStatistics ret{};
|
|
AnalyzeBlock(processed_image, 0, image_ptr, npixels, image_mode, ret);
|
|
return ret;
|
|
}
|
|
|
|
void ImagePreprocessorCPU::AnalyzeBlock(ImagePreprocessorBuffer &processed_image, size_t first, const uint8_t *input,
|
|
size_t n, CompressedImageMode image_mode, ImageStatistics &stats) {
|
|
switch (image_mode) {
|
|
case CompressedImageMode::Int8:
|
|
return AnalyzeBlock<int8_t>(processed_image, first, input, n, INT8_MIN, INT8_MAX, stats);
|
|
case CompressedImageMode::Int16:
|
|
return AnalyzeBlock<int16_t>(processed_image, first, input, n, INT16_MIN, INT16_MAX, stats);
|
|
case CompressedImageMode::Int32:
|
|
return AnalyzeBlock<int32_t>(processed_image, first, input, n, INT32_MIN, INT32_MAX, stats);
|
|
case CompressedImageMode::Uint8:
|
|
return AnalyzeBlock<uint8_t>(processed_image, first, input, n, UINT8_MAX, UINT8_MAX, stats);
|
|
case CompressedImageMode::Uint16:
|
|
return AnalyzeBlock<uint16_t>(processed_image, first, input, n, UINT16_MAX, UINT16_MAX, stats);
|
|
case CompressedImageMode::Uint32:
|
|
return AnalyzeBlock<uint32_t>(processed_image, first, input, n, UINT32_MAX, UINT32_MAX, stats);
|
|
default:
|
|
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "RGB/float mode not supported");
|
|
}
|
|
}
|
|
|
|
template<class T>
|
|
void ImagePreprocessorCPU::AnalyzeBlock(ImagePreprocessorBuffer &processed_image, size_t first, const uint8_t *input,
|
|
size_t n, T err_pixel_val, T sat_pixel_val, ImageStatistics &ret) {
|
|
auto image = reinterpret_cast<const T *>(input);
|
|
int32_t *out = processed_image.data() + first;
|
|
|
|
if (sat_pixel_val > saturation_limit)
|
|
sat_pixel_val = static_cast<T>(saturation_limit);
|
|
|
|
for (size_t i = 0; i < n; i++) {
|
|
if (mask_1bit[first + i] != 0) {
|
|
out[i] = INT32_MIN;
|
|
++ret.masked_pixel_count;
|
|
} else if (image[i] == err_pixel_val) {
|
|
// Error/invalid marker = the pixel type's extreme value (0xFFFFFFFF for EIGER uint32).
|
|
// Tested before saturation, since for unsigned types the marker also exceeds sat_pixel_val
|
|
// (which is clipped above to the HDF5 saturation_value).
|
|
out[i] = INT32_MIN;
|
|
++ret.error_pixel_count;
|
|
} else if (image[i] >= sat_pixel_val) {
|
|
out[i] = INT32_MAX;
|
|
++ret.saturated_pixel_count;
|
|
} else {
|
|
out[i] = static_cast<int32_t>(image[i]);
|
|
if (image[i] > ret.max_value)
|
|
ret.max_value = image[i];
|
|
if (image[i] < ret.min_value)
|
|
ret.min_value = image[i];
|
|
}
|
|
}
|
|
}
|