AdaptiveSpotFinderCPU::AccumulateRings: consecutive pixels mostly share a ring, so the ring's sum/sum2/count and the fused azint sums are held in locals while they do and stored when the ring changes - the same additions in the same order (az_sum2 is still contracted to the same FMA), without a store-and-reload chain through memory on every pixel. ImageSpotFinderCPU::DetectPass: the vertical-sum update (add the entering row, take out the leaving one) is one branch-free loop over the raw image that GCC vectorises (int64 lanes). A pixel strong in the previous pass used to be substituted per pixel through a bit test, which kept the loop scalar; it is now added with its row and taken out again from the few set bits of prev_strong. Integer sums, so the same totals. (A first, fully branch-free version that kept the per-pixel bit test did not vectorise on the prev_strong path and was measured slower; this is its replacement.) ImagePreprocessorCPU: the per-engine std::vector<bool> built bit by bit from the 32-bit mask (~10 core-s per cytc run, one per worker per pass) is replaced by 32-pixel mask words that PixelMask derives once beside its binary mask; each engine copies 2 MB. A branch-free rewrite of the Analyze loop was measured and dropped: the loop is bound by reading the decompressed image (330 vs 328 core-s on cytc), so only the mask test changed. Measured (perf, 499 Hz, CPU-only build, cytc, first version of this change): AccumulateRings 591 -> 539 core-s. Byte-identical p.hkl, p.mtz, p_P1.mtz, p_unmerged.mtz on myob, cytc, lyso, sparse (CPU) and myob, lyso (GPU). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
81 lines
4.0 KiB
C++
81 lines
4.0 KiB
C++
// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include "CompressedImage.h"
|
|
#include "DetectorSetup.h"
|
|
#include "DiffractionExperiment.h"
|
|
#include "../jungfrau/JFCalibration.h"
|
|
|
|
struct PixelMaskStatistics {
|
|
uint32_t user_mask;
|
|
uint32_t noisy_pixel;
|
|
uint32_t error_pixel;
|
|
uint32_t chip_gap_pixel;
|
|
uint32_t total_masked;
|
|
uint32_t module_gap_pixel;
|
|
};
|
|
|
|
class PixelMask {
|
|
std::vector<uint32_t> mask;
|
|
std::vector<uint32_t> raw_mask;
|
|
// One byte per pixel, 1 where the pixel is masked at all - the form the GPU image preprocessor
|
|
// uploads - and the checksum the shared device-table cache keys on (CudaSharedTables.h). Both are
|
|
// pure functions of `mask`, and an analysis engine is built per worker per pass, so they are
|
|
// derived here once instead of in every one of those engines.
|
|
std::vector<uint8_t> binary_mask;
|
|
uint64_t binary_mask_checksum = 0;
|
|
// The same flag packed 32 pixels to a word (pixel i is bit i % 32 of word i / 32) - the form the CPU
|
|
// image preprocessor reads, derived here for the same reason.
|
|
std::vector<uint32_t> packed_mask;
|
|
|
|
uint32_t LoadMask(const std::vector<uint32_t>& mask, uint8_t bit);
|
|
// Everything that follows from the mask, recomputed wherever the mask changes.
|
|
void UpdateDerived(const DiffractionExperiment& experiment);
|
|
void UpdateBinaryMask();
|
|
|
|
void CalcEdgePixels_i(const DiffractionExperiment& experiment);
|
|
|
|
public:
|
|
// NXmx bits
|
|
constexpr static const uint8_t ModuleGapPixelBit = 0;
|
|
constexpr static const uint8_t ErrorPixelBit = 1;
|
|
constexpr static const uint8_t NoisyPixelBit = 4;
|
|
constexpr static const uint8_t UserMaskedPixelBit = 8;
|
|
constexpr static const uint8_t BeamStopPixelBit = 9;
|
|
// Found bad on the run's own frames (rugnux pre-scan, HotPixelFinder): lit above its ring on too
|
|
// many frames, or holding the error value on most of them.
|
|
constexpr static const uint8_t HotPixelBit = 10;
|
|
constexpr static const uint8_t ChipGapPixelBit = 31;
|
|
constexpr static const uint8_t ModuleEdgePixelBit = 30;
|
|
|
|
PixelMask();
|
|
explicit PixelMask(size_t width, size_t height);
|
|
explicit PixelMask(const DiffractionExperiment& experiment);
|
|
explicit PixelMask(const std::vector<uint32_t> &mask);
|
|
|
|
void CalcEdgePixels(const DiffractionExperiment& experiment);
|
|
void LoadUserMask(const DiffractionExperiment& experiment, const std::vector<uint32_t>& mask);
|
|
void LoadUserMask(const DiffractionExperiment& experiment, const CompressedImage& image);
|
|
void LoadBeamStopMask(const DiffractionExperiment& experiment, const std::vector<uint32_t>& mask);
|
|
// The beam-stop shadow belongs to the run that found it, not to the dataset, so a mask read back
|
|
// from a file that carries one starts clear. The user mask (bit 8) is deliberately left alone.
|
|
void ClearBeamStopMask(const DiffractionExperiment& experiment);
|
|
void LoadHotPixelMask(const DiffractionExperiment& experiment, const std::vector<uint32_t>& mask);
|
|
void LoadDECTRISBadPixelMask(const std::vector<uint32_t>& mask);
|
|
void LoadDarkBadPixelMask(const DiffractionExperiment& experiment, const std::vector<uint32_t>& mask);
|
|
void LoadDetectorBadPixelMask(const DiffractionExperiment& experiment, const JFCalibration *calib);
|
|
|
|
[[nodiscard]] const std::vector<uint32_t> &GetMaskRaw() const;
|
|
[[nodiscard]] const std::vector<uint32_t> &GetMask(const DiffractionExperiment& experiment) const;
|
|
[[nodiscard]] const std::vector<uint32_t> &GetMask() const;
|
|
[[nodiscard]] const std::vector<uint8_t> &GetBinaryMask() const;
|
|
[[nodiscard]] const std::vector<uint32_t> &GetPackedMask() const;
|
|
[[nodiscard]] uint64_t GetBinaryMaskChecksum() const;
|
|
|
|
[[nodiscard]] std::vector<uint32_t> GetUserMask(const DiffractionExperiment& experiment) const;
|
|
[[nodiscard]] std::vector<uint32_t> GetUserMask() const;
|
|
[[nodiscard]] PixelMaskStatistics GetStatistics() const;
|
|
};
|