Files
Jungfraujoch/common/PixelMask.cpp
T
leonarski_fandClaude Opus 5.5 c8692e320c CPU pixel loops: ring sums held in registers, vectorised vertical window, shared packed mask
AdaptiveSpotFinderCPU::AccumulateRings: consecutive pixels mostly share a ring, so the ring's
sum/sum2/count and the fused azint sums are held in locals while they do and stored when the ring
changes - the same additions in the same order (az_sum2 is still contracted to the same FMA), without
a store-and-reload chain through memory on every pixel.

ImageSpotFinderCPU::DetectPass: the vertical-sum update (add the entering row, take out the leaving
one) is one branch-free loop over the raw image that GCC vectorises (int64 lanes). A pixel strong in
the previous pass used to be substituted per pixel through a bit test, which kept the loop scalar;
it is now added with its row and taken out again from the few set bits of prev_strong. Integer sums,
so the same totals. (A first, fully branch-free version that kept the per-pixel bit test did not
vectorise on the prev_strong path and was measured slower; this is its replacement.)

ImagePreprocessorCPU: the per-engine std::vector<bool> built bit by bit from the 32-bit mask
(~10 core-s per cytc run, one per worker per pass) is replaced by 32-pixel mask words that PixelMask
derives once beside its binary mask; each engine copies 2 MB. A branch-free rewrite of the Analyze
loop was measured and dropped: the loop is bound by reading the decompressed image (330 vs 328
core-s on cytc), so only the mask test changed.

Measured (perf, 499 Hz, CPU-only build, cytc, first version of this change): AccumulateRings
591 -> 539 core-s. Byte-identical p.hkl, p.mtz, p_P1.mtz, p_unmerged.mtz on myob, cytc, lyso,
sparse (CPU) and myob, lyso (GPU).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
2026-09-28 02:19:31 +02:00

367 lines
14 KiB
C++

// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "PixelMask.h"
#include "RawToConvertedGeometry.h"
#include "TableChecksum.h"
#include "JFJochException.h"
#include "JFJochCompressor.h"
PixelMask::PixelMask() = default;
PixelMask::PixelMask(size_t width, size_t height)
: mask(width*height, 0) {
UpdateBinaryMask();
}
PixelMask::PixelMask(const DiffractionExperiment &experiment)
: PixelMask(experiment.GetXPixelsNumConv(),
experiment.GetYPixelsNumConv()) {
CalcEdgePixels(experiment);
}
PixelMask::PixelMask(const std::vector<uint32_t> &in_mask) : mask(in_mask) {
UpdateBinaryMask();
}
uint32_t PixelMask::LoadMask(const std::vector<uint32_t> &input_mask, uint8_t bit) {
uint32_t ret = 0;
if (input_mask.size() != mask.size())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Input match doesn't fit the detector ");
for (int i = 0; i < mask.size(); i++) {
if (input_mask[i] != 0) {
mask[i] |= (1 << bit);
ret++;
} else
mask[i] &= ~(1 << bit);
}
return ret;
}
void PixelMask::UpdateDerived(const DiffractionExperiment &experiment) {
switch (experiment.GetDetectorType()) {
case DetectorType::JUNGFRAU:
case DetectorType::EIGER:
raw_mask.resize(experiment.GetModulesNum() * RAW_MODULE_SIZE, 0);
ConvertedToRawGeometry(experiment, raw_mask.data(), mask.data());
break;
default:
raw_mask.clear();
break;
}
UpdateBinaryMask();
}
void PixelMask::UpdateBinaryMask() {
binary_mask.resize(mask.size());
for (size_t i = 0; i < mask.size(); i++)
binary_mask[i] = (mask[i] != 0);
binary_mask_checksum = TableChecksum(binary_mask.data(), binary_mask.size());
packed_mask.assign((mask.size() + 31) / 32, 0);
for (size_t i = 0; i < mask.size(); i++)
packed_mask[i / 32] |= static_cast<uint32_t>(binary_mask[i]) << (i % 32);
}
void PixelMask::CalcEdgePixels_i(const DiffractionExperiment &experiment) {
if (experiment.GetDetectorType() == DetectorType::DECTRIS)
return;
size_t nmodules = experiment.GetModulesNum();
auto settings = experiment.GetImageFormatSettings();
// Set module gaps to 1
std::vector<uint32_t> module_gaps(nmodules * RAW_MODULE_SIZE, 0);
std::vector<uint32_t> module_gaps_conv(experiment.GetPixelsNumConv(), 1);
RawToConvertedGeometry(experiment, module_gaps_conv.data(), module_gaps.data());
LoadMask(module_gaps_conv, ModuleGapPixelBit);
// Calculate module edges and chip edges
std::vector<uint32_t> module_edge(nmodules * RAW_MODULE_SIZE, 0);
std::vector<uint32_t> chip_edge(nmodules * RAW_MODULE_SIZE, 0);
for (int64_t module = 0; module < nmodules; module++) {
for (int64_t line = 0; line < RAW_MODULE_LINES; line++) {
for (int64_t col = 0; col < RAW_MODULE_COLS; col++) {
int64_t pixel = module * RAW_MODULE_SIZE + line * RAW_MODULE_COLS + col;
if ((line == 0)
|| (line == RAW_MODULE_LINES - 1)
|| (col == 0)
|| (col == RAW_MODULE_COLS - 1))
module_edge[pixel] = 1;
if ((col == 255) || (col == 256)
|| (col == 511) || (col == 512)
|| (col == 767) || (col == 768)
|| (line == 255) || (line == 256))
chip_edge[pixel] = 1;
}
}
}
std::vector<uint32_t> module_edge_conv(experiment.GetPixelsNumConv(), 0);
if (experiment.GetMaskModuleEdges())
RawToConvertedGeometry(experiment, module_edge_conv.data(), module_edge.data());
LoadMask(module_edge_conv, ModuleEdgePixelBit);
std::vector<uint32_t> chip_edge_conv(experiment.GetPixelsNumConv(), 0);
if (experiment.GetMaskChipEdges())
RawToConvertedGeometry(experiment, chip_edge_conv.data(), chip_edge.data());
LoadMask(chip_edge_conv, ChipGapPixelBit);
}
void PixelMask::CalcEdgePixels(const DiffractionExperiment &experiment) {
CalcEdgePixels_i(experiment);
UpdateDerived(experiment);
}
const std::vector<uint32_t> &PixelMask::GetMaskRaw() const {
if (raw_mask.empty())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Raw format not available for this detector");
return raw_mask;
}
const std::vector<uint32_t> &PixelMask::GetMask() const {
return mask;
}
const std::vector<uint8_t> &PixelMask::GetBinaryMask() const {
return binary_mask;
}
const std::vector<uint32_t> &PixelMask::GetPackedMask() const {
return packed_mask;
}
uint64_t PixelMask::GetBinaryMaskChecksum() const {
return binary_mask_checksum;
}
const std::vector<uint32_t> &PixelMask::GetMask(const DiffractionExperiment& experiment) const {
if (experiment.IsGeometryTransformed())
return GetMask();
else
return GetMaskRaw();
}
std::vector<uint32_t> PixelMask::GetUserMask() const {
std::vector<uint32_t> ret = GetMask();
for (auto &i: ret)
i = ((i & (1 << UserMaskedPixelBit)) != 0) ? 1 : 0;
return ret;
}
std::vector<uint32_t> PixelMask::GetUserMask(const DiffractionExperiment& experiment) const {
if (experiment.IsGeometryTransformed())
return GetUserMask();
else {
std::vector<uint32_t> tmp = GetUserMask();
std::vector<uint32_t> ret(experiment.GetModulesNum() * RAW_MODULE_SIZE, 0);
ConvertedToRawGeometry(experiment, ret.data(), tmp.data());
return ret;
}
}
void PixelMask::LoadDetectorBadPixelMask(const DiffractionExperiment &experiment, const JFCalibration *calib) {
if (experiment.GetDetectorType() == DetectorType::DECTRIS)
return;
std::vector<uint32_t> input_mask(experiment.GetModulesNum() * RAW_MODULE_SIZE, 0);
std::vector<uint32_t> input_mask_rms(experiment.GetModulesNum() * RAW_MODULE_SIZE, 0);
if (calib != nullptr) {
for (int sc = 0; sc < experiment.GetStorageCellNumber(); sc++) {
// For multiple SC PixelMask is logical sum of all image masks
// (this can be too much, but better than too little)
auto pedestal_g0 = calib->GetPedestal(0, sc);
auto pedestal_g0_rms = calib->GetPedestalRMS(0, sc);
auto pedestal_g1 = calib->GetPedestal(1, sc);
auto pedestal_g2 = calib->GetPedestal(2, sc);
for (int i = 0; i < experiment.GetModulesNum() * RAW_MODULE_SIZE; i++) {
if (pedestal_g1[i] > 16383)
input_mask[i] = 1;
if (!experiment.IsFixedGainG1()) {
if (pedestal_g0[i] >= 16383) {
if (experiment.IsMaskPixelsWithoutG0())
input_mask[i] = 1;
} else if (pedestal_g0_rms[i] > experiment.GetImageFormatSettings().GetPedestalG0RMSLimit())
input_mask_rms[i] = 1;
if (pedestal_g2[i] >= 16383)
input_mask[i] = 1;
}
}
}
}
std::vector<uint32_t> input_mask_conv(experiment.GetPixelsNumConv(), 0);
RawToConvertedGeometry(experiment, input_mask_conv.data(), input_mask.data());
std::vector<uint32_t> input_mask_rms_conv(experiment.GetPixelsNumConv(), 0);
RawToConvertedGeometry(experiment, input_mask_rms_conv.data(), input_mask_rms.data());
LoadMask(input_mask_conv, ErrorPixelBit);
LoadMask(input_mask_rms_conv, NoisyPixelBit);
CalcEdgePixels_i(experiment);
UpdateDerived(experiment);
}
PixelMaskStatistics PixelMask::GetStatistics() const {
PixelMaskStatistics ret{};
for (const auto &i: mask) {
if (i & (1 << ModuleGapPixelBit))
ret.module_gap_pixel++;
else {
if (i != 0)
ret.total_masked++;
if (i & (1 << ErrorPixelBit))
ret.error_pixel++;
if (i & (1 << NoisyPixelBit))
ret.noisy_pixel++;
if (i & (1 << UserMaskedPixelBit))
ret.user_mask++;
if (i & ((1 << ChipGapPixelBit) | (1 << ModuleEdgePixelBit)))
ret.chip_gap_pixel++;
}
}
return ret;
}
void PixelMask::LoadUserMask(const DiffractionExperiment& experiment, const std::vector<uint32_t> &in_mask) {
if (in_mask.size() == mask.size()) {
LoadMask(in_mask, UserMaskedPixelBit);
UpdateDerived(experiment);
} else if (in_mask.size() == experiment.GetModulesNum() * RAW_MODULE_SIZE) {
std::vector<uint32_t> tmp(experiment.GetPixelsNumConv(), 0);
RawToConvertedGeometry(experiment, tmp.data(), in_mask. data());
LoadMask(tmp, UserMaskedPixelBit);
UpdateDerived(experiment);
} else
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Size of input user mask invalid");
}
void PixelMask::LoadBeamStopMask(const DiffractionExperiment& experiment, const std::vector<uint32_t> &in_mask) {
if (in_mask.size() != mask.size())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Size of input beam stop mask invalid");
LoadMask(in_mask, BeamStopPixelBit);
UpdateDerived(experiment);
}
void PixelMask::LoadHotPixelMask(const DiffractionExperiment& experiment, const std::vector<uint32_t> &in_mask) {
if (in_mask.size() != mask.size())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Size of input hot pixel mask invalid");
LoadMask(in_mask, HotPixelBit);
UpdateDerived(experiment);
}
void PixelMask::ClearBeamStopMask(const DiffractionExperiment& experiment) {
for (auto &i: mask)
i &= ~(1u << BeamStopPixelBit);
UpdateDerived(experiment);
}
void PixelMask::LoadUserMask(const DiffractionExperiment& experiment, const CompressedImage& image) {
const size_t width = image.GetWidth();
const size_t height = image.GetHeight();
// The image has to match one of the two layouts handled by the vector
// overload below: converted geometry, or raw stacked modules.
const bool converted = (width == static_cast<size_t>(experiment.GetXPixelsNumConv()))
&& (height == static_cast<size_t>(experiment.GetYPixelsNumConv()));
const bool raw = (width == static_cast<size_t>(RAW_MODULE_COLS))
&& (height == static_cast<size_t>(RAW_MODULE_LINES * experiment.GetModulesNum()));
if (!converted && !raw)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"User mask image size doesn't match the detector");
std::vector<uint8_t> buffer;
const uint8_t *bytes = image.GetUncompressedPtr(buffer);
// A pixel is masked when its value is non-zero. Read each pixel as an
// unsigned integer of the matching width - the sign is irrelevant when
// comparing against zero.
std::vector<uint32_t> mask(width * height);
auto binarize = [&](auto sample) {
using sample_t = decltype(sample);
const auto *typed = reinterpret_cast<const sample_t *>(bytes);
for (size_t i = 0; i < mask.size(); i++)
mask[i] = (typed[i] != 0) ? 1 : 0;
};
switch (image.GetMode()) {
case CompressedImageMode::Uint8:
case CompressedImageMode::Int8:
binarize(uint8_t{});
break;
case CompressedImageMode::Uint16:
case CompressedImageMode::Int16:
binarize(uint16_t{});
break;
case CompressedImageMode::Uint32:
case CompressedImageMode::Int32:
binarize(uint32_t{});
break;
default:
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"User mask must be an 8-, 16- or 32-bit integer image");
}
LoadUserMask(experiment, mask);
}
void PixelMask::LoadDECTRISBadPixelMask(const std::vector<uint32_t> &input_mask) {
if (input_mask.size() != mask.size())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Input match doesn't fit the detector ");
uint32_t user_bitmask = (1 << UserMaskedPixelBit);
uint32_t bad_pixel_bitmask = ~((1 << UserMaskedPixelBit) | (1 << ModuleGapPixelBit) | (1 << ChipGapPixelBit));
for (int i = 0; i < mask.size(); i++) {
if ((input_mask[i] & (1 << ModuleGapPixelBit)) != 0) {
mask[i] = (1 << ModuleGapPixelBit);
} else {
mask[i] = 0;
if (input_mask[i] & bad_pixel_bitmask) {
mask[i] |= (1 << ErrorPixelBit);
}
// User and chip gap are just transferred
if ((input_mask[i] & (1 << UserMaskedPixelBit)) != 0) {
mask[i] |= (1 << UserMaskedPixelBit);
}
if ((input_mask[i] & (1 << ChipGapPixelBit)) != 0) {
mask[i] |= (1 << ChipGapPixelBit);
}
}
}
raw_mask = {}; // For DECTRIS - there is no raw mask
UpdateBinaryMask();
}
void PixelMask::LoadDarkBadPixelMask(const DiffractionExperiment& experiment, const std::vector<uint32_t> &input_mask) {
if (input_mask.size() != mask.size())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Input match doesn't fit the detector ");
for (int i = 0; i < mask.size(); i++) {
// Ignore module gap (doesn't matter) or bad pixels
if ((mask[i] & (1 << ModuleGapPixelBit | 1 << ErrorPixelBit)) != 0)
continue;
if (input_mask[i] != 0) {
mask[i] |= (1 << NoisyPixelBit);
} else {
mask[i] &= ~(1 << NoisyPixelBit);
}
}
UpdateDerived(experiment);
}