The CPU-only image loop is DRAM-bound on 16M frames: decode, preprocess, the adaptive finder's plain ring pass and FlagRings each streamed the whole frame through memory. - JFJochDecompressHperfBlocks hands each decoded bitshuffle block to a callback; with no output buffer the block is unshuffled into a reused block-sized scratch (JFJochDecompressBlocks). - MXAnalysisWithoutFPGA::PreprocessCPU preprocesses each block into the int32 buffer (ImagePreprocessorCPU::AnalyzeBlock) and, when the fused CPU finder runs, puts it through the plain ring pass + fused azint (AdaptiveSpotFinderCPU::AccumulateRingsBlock) while it is in cache. Detect() then starts from those sums. The per-worker decompression buffer is no longer allocated for bitshuffled data. - FlagRings becomes FlagRow, called by DetectAt's first pass for row y+NBX just before that row enters the vertical sums; first_pass_needed is marked from each row's candidates at the same point. Exact: blocks arrive in pixel order, so the float azint sums see the same pixels in the same order; the per-pixel expressions are unchanged; everything else is integer. p.hkl, p.mtz, p_P1.mtz and p_unmerged.mtz byte-identical to rc173 on myob, cytc, lyso, sparse (CPU-only build), GPU myob identical (the GPU path does not take this route). CPU-only, 32 workers, under the gpulock on a shared (loaded) machine, base -> fused, two rounds (second in reversed order): myob 155.9 -> 97.7 s, 139.9 -> 88.1 s (loop 46.1 -> 26.8 s/pass; user 3690 -> 2221 s) cytc 220.8 -> 156.6 s, 216.6 -> 155.1 s (user 5160 -> 4128 s) lyso 134.5 -> 123.3 s, 69.5 -> 64.8 s peak RSS myob 15.2 -> 10.8 GB, cytc 12.6 -> 10.4 GB, lyso 7.0 -> 6.7 GB Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
118 lines
6.2 KiB
C++
118 lines
6.2 KiB
C++
// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <mutex>
|
|
|
|
#include "../common/JFJochMessages.h"
|
|
#include "../common/DiffractionExperiment.h"
|
|
#include "../common/AzimuthalIntegrationMapping.h"
|
|
#include "../common/PixelMask.h"
|
|
#include "../common/AzimuthalIntegrationProfile.h"
|
|
#include "bragg_prediction/BraggPrediction.h"
|
|
#include "bragg_integration/BraggIntegrationEngine.h"
|
|
#include "spot_finding/ImageSpotFinder.h"
|
|
#include "spot_finding/AdaptiveSpotFinderCPU.h"
|
|
#include "indexing/IndexerThreadPool.h"
|
|
#include "azint/AzIntEngine.h"
|
|
#include "roi/ROIIntegration.h"
|
|
#include "IndexAndRefine.h"
|
|
#include "image_preprocessing/ImagePreprocessor.h"
|
|
#include "image_preprocessing/ImagePreprocessorBuffer.h"
|
|
#include "image_preprocessing/ImagePreprocessorCPU.h"
|
|
|
|
class CudaStream;
|
|
class AdaptiveSpotFinderGPU;
|
|
|
|
// MXAnalysisWithoutFPGA is not thread safe - it has to owned by a single thread
|
|
class MXAnalysisWithoutFPGA {
|
|
const DiffractionExperiment &experiment;
|
|
const AzimuthalIntegrationMapping &integration;
|
|
|
|
std::vector<uint8_t> decompression_buffer;
|
|
|
|
std::unique_ptr<ImagePreprocessor> preprocessor;
|
|
// The preprocessor, where it is the CPU one.
|
|
ImagePreprocessorCPU *preprocessor_cpu = nullptr;
|
|
|
|
size_t npixels;
|
|
size_t xpixels;
|
|
|
|
// Built on first use: the fused adaptive finder produces the azimuthal profile as a by-product,
|
|
// so on the rugnux path this engine is constructed and then never run.
|
|
std::unique_ptr<AzIntEngine> azint;
|
|
AzIntEngine &AzInt();
|
|
std::unique_ptr<ROIIntegration> roi;
|
|
// Built on first use. Which finder an image takes arrives with its SpotFindingSettings, and
|
|
// with adaptive detection on - the default everywhere but the broker - this one is never asked
|
|
// for; on the GPU it is ~14 MB and 15 device allocations per worker.
|
|
std::unique_ptr<ImageSpotFinder> spotFinder;
|
|
ImageSpotFinder &FixedThresholdFinder();
|
|
// Self-calibrating finder, used when spot settings request adaptive detection. Kept alongside the
|
|
// default finder because the choice arrives with the per-image settings, not at construction. It is
|
|
// an AdaptiveSpotFinderCPU by default; on the GPU path, when the fused engine is enabled (rugnux
|
|
// offline only), it is instead an AdaptiveSpotFinderGPU that also computes the azimuthal profile,
|
|
// aliased through fused_adaptive so Analyze() can take that profile and skip the separate azint pass.
|
|
std::unique_ptr<ImageSpotFinder> adaptiveSpotFinder;
|
|
AdaptiveSpotFinderGPU *fused_adaptive = nullptr;
|
|
// The CPU finder, where it gives the profile too (the azimuthal integration being on the CPU).
|
|
AdaptiveSpotFinderCPU *fused_adaptive_cpu = nullptr;
|
|
const bool enable_fused_adaptive_gpu;
|
|
IndexAndRefine &indexer;
|
|
std::unique_ptr<BraggPrediction> prediction;
|
|
std::unique_ptr<BraggIntegrationEngine> bragg_engine;
|
|
// What the supercell probe's integrations added to bragg_engine's counts (see BraggCounts).
|
|
BraggIntegrationCounts probe_counts;
|
|
std::unique_ptr<ImagePreprocessorBuffer> preprocessor_buffer;
|
|
const PixelMask &mask;
|
|
|
|
// Decompress the image into decompression_buffer (or read it straight from the message, when it is
|
|
// not compressed) and return where it landed.
|
|
const uint8_t *Decompress(const CompressedImage &image);
|
|
|
|
// The CPU preprocessing. A bitshuffled image is decoded a block (~32 KB) at a time and each block
|
|
// is preprocessed while it is in cache, then - with ring_pass - put through the adaptive finder's
|
|
// plain ring pass as well, so the image is never written out whole before it is preprocessed and
|
|
// the preprocessed pixels are read back from cache, not from memory. The blocks come in pixel
|
|
// order, so every sum is taken in the same order as by separate passes.
|
|
ImageStatistics PreprocessCPU(const CompressedImage &image, bool ring_pass);
|
|
|
|
// Pixels outside the resolution limits, bit-packed. Built by the integration mapping, which is
|
|
// shared by every worker's engine and hands out the same mask to all of them.
|
|
std::shared_ptr<const std::vector<uint32_t>> mask_resolution;
|
|
// The limits mask_resolution was built for. Kept as the OPTIONAL the caller passed, so an unset
|
|
// high-resolution limit compares equal to itself and the mask is not rebuilt on every image.
|
|
std::optional<float> mask_high_res;
|
|
std::optional<float> mask_low_res;
|
|
void UpdateMaskResolution(const SpotFindingSettings& settings);
|
|
#ifdef JFJOCH_USE_CUDA
|
|
std::shared_ptr<CudaStream> stream; // kept so RebuildROI() can recreate the GPU ROI engine
|
|
#endif
|
|
public:
|
|
// enable_fused_adaptive_gpu turns on the fused GPU azint+adaptive spot finder (only takes effect on
|
|
// the GPU path with adaptive detection). The rugnux offline path and the interactive viewer enable
|
|
// it by default, as does the online receiver. It only changes performance - the fused engine
|
|
// reproduces the CPU finder's spots. Note it also decides whether the preprocessed image is copied
|
|
// back to the host each frame: that copy exists only for a CPU engine to read, and with the flag on
|
|
// no CPU engine is built, so the copy is skipped.
|
|
MXAnalysisWithoutFPGA(const DiffractionExperiment &experiment, const AzimuthalIntegrationMapping &integration,
|
|
const PixelMask &mask, IndexAndRefine &indexer, bool enable_fused_adaptive_gpu = false);
|
|
void Analyze(DataMessage &output, AzimuthalIntegrationProfile &profile, const SpotFindingSettings &spot_finding_settings);
|
|
|
|
// Surgical ROI-only paths used when a full re-analysis is not wanted: rebuild the
|
|
// ROI engine after the ROI set changes, recompute ROIs after preprocessing a new
|
|
// image (reanalyze off), or just rerun ROIs on the current preprocessed image (an
|
|
// interactive ROI move). A full Analyze() already computes ROIs, so needs nothing.
|
|
void RebuildROI();
|
|
void AnalyzeROIOnly(DataMessage &output);
|
|
void RunROIOnly(DataMessage &output);
|
|
|
|
// What this worker's Bragg integrator counted (BraggIntegrationCounts). Each worker builds its own
|
|
// analysis, so a caller that wants the run's totals sums this over the workers it started.
|
|
[[nodiscard]] BraggIntegrationCounts BraggCounts() const;
|
|
};
|
|
|
|
|
|
|