// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #pragma once #include "BraggIntegrationEngine.h" class CompressedImage; // Plain-C++ reference/fallback engine: a faithful serial re-expression of BraggIntegrate2D (box // sum) and ProfileIntegrate2D (Kabsch profile fit) reading the preprocessed int32 image. Also the // numeric oracle the CUDA engine is checked against. class BraggIntegrationEngineCPU : public BraggIntegrationEngine { // Core integrator, templated on a pixel sampler so it reads either the preprocessed int32 buffer // or a raw CompressedImage of any pixel type - both presented per-pixel in the INT32_MIN(masked)/ // INT32_MAX(saturated) convention - without ever materialising a second full-image copy. // Full-frame scratch of RunImpl: the reflection mask and the signal-region owner map. A call writes // only around its reflections, so these keep only the 16x16-pixel tiles written since the last // Clear(): a frame-sized array was mostly never read, yet over a sweep every page of it got // touched, in every worker. Reading a tile nothing wrote gives `empty`. template class TiledFrame { static constexpr int TILE = 16; int tiles_x; T empty; std::vector tile_start; // per tile: where it starts in `pixels`, -1 = not written std::vector written; // the tiles written, for Clear() std::vector pixels; public: TiledFrame(int width, int height, T empty) : tiles_x((width + TILE - 1) / TILE), empty(empty), tile_start(static_cast(tiles_x) * ((height + TILE - 1) / TILE), -1) {} T Get(int x, int y) const { const int32_t start = tile_start[(y / TILE) * tiles_x + x / TILE]; return start < 0 ? empty : pixels[start + (y % TILE) * TILE + x % TILE]; } T &At(int x, int y) { const int t = (y / TILE) * tiles_x + x / TILE; if (tile_start[t] < 0) { tile_start[t] = static_cast(pixels.size()); pixels.resize(pixels.size() + TILE * TILE, empty); written.push_back(t); } return pixels[tile_start[t] + (y % TILE) * TILE + x % TILE]; } void Clear() { for (int t : written) tile_start[t] = -1; written.clear(); pixels.clear(); } }; TiledFrame refl_mask; TiledFrame owner; template std::vector RunImpl(const Sampler &img, const std::vector &predicted, size_t npredicted, int64_t image_number); public: explicit BraggIntegrationEngineCPU(const DiffractionExperiment &experiment); using BraggIntegrationEngine::Run; // keep the preprocessed-buffer overload visible std::vector Run(const ImagePreprocessorBuffer &image, const std::vector &predicted, size_t npredicted, int64_t image_number) override; // FPGA workflow: integrate straight off the assembled detector image, reading only the pixels // inside each reflection disk (no whole-image conversion - the FPGA host cannot afford one at its // frame rate). Masked pixels carry the type minimum and saturated the type maximum. std::vector Run(const CompressedImage &image, const std::vector &predicted, size_t npredicted, int64_t image_number); };