refl_mask (uint8) and owner (uint32) were full-frame arrays per engine, ~90 MB at 18 Mpixel, of which a call writes only the neighbourhoods of its reflections; over a sweep every page of both got touched in every worker, and the dirty-rectangle clears were row-strided memsets. They are now a TiledFrame: a per-tile directory plus a pool of 16x16 tiles allocated on first write and dropped by Clear(); reading a tile nothing wrote gives the empty value (0 / BRAGG_OWNER_NONE). Writes and reads are exactly those of the flat arrays, so the same ownership decisions. Measured (perf, 499 Hz, CPU-only build, cytc): RunImpl 1633 -> 1557 core-s, memset 88 -> 32 core-s. Peak RSS and page faults did not move measurably (myob 15.3 GB; the peak is set elsewhere). Byte-identical p.hkl, p.mtz, p_P1.mtz, p_unmerged.mtz on myob, cytc, lyso, sparse (CPU) and myob, lyso (GPU); Bragg* tests pass. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
76 lines
3.6 KiB
C++
76 lines
3.6 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include "BraggIntegrationEngine.h"
|
|
|
|
class CompressedImage;
|
|
|
|
// Plain-C++ reference/fallback engine: a faithful serial re-expression of BraggIntegrate2D (box
|
|
// sum) and ProfileIntegrate2D (Kabsch profile fit) reading the preprocessed int32 image. Also the
|
|
// numeric oracle the CUDA engine is checked against.
|
|
class BraggIntegrationEngineCPU : public BraggIntegrationEngine {
|
|
// Core integrator, templated on a pixel sampler so it reads either the preprocessed int32 buffer
|
|
// or a raw CompressedImage of any pixel type - both presented per-pixel in the INT32_MIN(masked)/
|
|
// INT32_MAX(saturated) convention - without ever materialising a second full-image copy.
|
|
// Full-frame scratch of RunImpl: the reflection mask and the signal-region owner map. A call writes
|
|
// only around its reflections, so these keep only the 16x16-pixel tiles written since the last
|
|
// Clear(): a frame-sized array was mostly never read, yet over a sweep every page of it got
|
|
// touched, in every worker. Reading a tile nothing wrote gives `empty`.
|
|
template <class T>
|
|
class TiledFrame {
|
|
static constexpr int TILE = 16;
|
|
int tiles_x;
|
|
T empty;
|
|
std::vector<int32_t> tile_start; // per tile: where it starts in `pixels`, -1 = not written
|
|
std::vector<int32_t> written; // the tiles written, for Clear()
|
|
std::vector<T> pixels;
|
|
public:
|
|
TiledFrame(int width, int height, T empty)
|
|
: tiles_x((width + TILE - 1) / TILE), empty(empty),
|
|
tile_start(static_cast<size_t>(tiles_x) * ((height + TILE - 1) / TILE), -1) {}
|
|
T Get(int x, int y) const {
|
|
const int32_t start = tile_start[(y / TILE) * tiles_x + x / TILE];
|
|
return start < 0 ? empty : pixels[start + (y % TILE) * TILE + x % TILE];
|
|
}
|
|
T &At(int x, int y) {
|
|
const int t = (y / TILE) * tiles_x + x / TILE;
|
|
if (tile_start[t] < 0) {
|
|
tile_start[t] = static_cast<int32_t>(pixels.size());
|
|
pixels.resize(pixels.size() + TILE * TILE, empty);
|
|
written.push_back(t);
|
|
}
|
|
return pixels[tile_start[t] + (y % TILE) * TILE + x % TILE];
|
|
}
|
|
void Clear() {
|
|
for (int t : written)
|
|
tile_start[t] = -1;
|
|
written.clear();
|
|
pixels.clear();
|
|
}
|
|
};
|
|
TiledFrame<uint8_t> refl_mask;
|
|
TiledFrame<uint32_t> owner;
|
|
|
|
template <class Sampler>
|
|
std::vector<Reflection> RunImpl(const Sampler &img, const std::vector<Reflection> &predicted,
|
|
size_t npredicted, int64_t image_number);
|
|
|
|
public:
|
|
explicit BraggIntegrationEngineCPU(const DiffractionExperiment &experiment);
|
|
|
|
using BraggIntegrationEngine::Run; // keep the preprocessed-buffer overload visible
|
|
|
|
std::vector<Reflection> Run(const ImagePreprocessorBuffer &image,
|
|
const std::vector<Reflection> &predicted, size_t npredicted,
|
|
int64_t image_number) override;
|
|
|
|
// FPGA workflow: integrate straight off the assembled detector image, reading only the pixels
|
|
// inside each reflection disk (no whole-image conversion - the FPGA host cannot afford one at its
|
|
// frame rate). Masked pixels carry the type minimum and saturated the type maximum.
|
|
std::vector<Reflection> Run(const CompressedImage &image,
|
|
const std::vector<Reflection> &predicted, size_t npredicted,
|
|
int64_t image_number);
|
|
};
|