Files
Jungfraujoch/common/AzimuthalIntegrationMapping.cpp
T
jungfrauandClaude Opus 5 5ee0f22a61 Build the detector's lookup tables once, not once per worker
The image loop gives every worker its own analysis engine, so a run builds ninety-six of them. Each
one derived, from scratch, tables that are the same in all of them: the byte-per-pixel mask, the
resolution mask, the radial kernel, and the checksum that names the shared device tables.

The checksum was the worst of it, because it is part of the cache KEY and so is computed before the
lookup - a hit still hashed the whole table. On a 16 Mpx detector that is the bin table, the
corrections and the mask, 126 MB an engine, about twelve gigabytes over a run, to answer a question
whose answer had not changed. The header said it cost nothing measurable; a profile says otherwise,
and says it is worst exactly during the ramp when the machine has nothing else to do.

It cannot simply be remembered against the address, which is what it exists to catch: a buffer can
be freed and another allocated where it was, and the cache would then hand back a device copy of
something else. So the owner of the bytes computes it instead. The azimuthal mapping writes its two
tables in its constructor and never again. The pixel mask re-derives its binary form and its
checksum on every path that changes the mask, and all of those paths are now private to the class.
The key therefore still describes the bytes as they are at the moment of the lookup.

The resolution mask was two passes over every pixel - a float comparison into a vector<bool>, then a
bit-by-bit repack - in each of the ninety-six. It is one pass now, writing the packed form directly,
built once for the limits asked for and handed out as a shared pointer so a worker keeps the mask it
was given. The radial kernel is cached on the six numbers it is derived from.

Nothing computes a different value; only who computes it changes. Byte-identical merged output on a
16 Mpx set and on a small one.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_011n8riB6X59oRjkrSHzNPAU
2026-08-23 12:59:58 -04:00

270 lines
10 KiB
C++

// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "JFJochMath.h"
#include <cmath>
#include <thread>
#include <future>
#include "AzimuthalIntegrationMapping.h"
#include "JFJochException.h"
#include "DiffractionGeometry.h"
#include "RawToConvertedGeometry.h"
#include "TableChecksum.h"
AzimuthalIntegrationMapping::AzimuthalIntegrationMapping(const DiffractionExperiment &experiment,
const PixelMask& mask,
size_t in_nthreads)
: settings(experiment.GetAzimuthalIntegrationSettings()),
wavelength(experiment.GetWavelength_A()),
// The dimensions of the image this mapping is built for: converted when the geometry is
// transformed, raw when it is not. They have to match pixel_to_bin, which is sized per mode
// below - the adaptive spot finders walk the image with these and index pixel_to_bin with it.
width(experiment.GetXPixelsNum()),
height(experiment.GetYPixelsNum()) {
if (width <= 0)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Detector width must be above 0");
if (height <= 0)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Detector height must be above 0");
if (settings.GetBinCount() >= UINT16_MAX)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid,
"Cannot handle more than 65534 az. int. bins");
polarization_factor = experiment.GetPolarizationFactor();
if (in_nthreads == 0)
nthreads = std::thread::hardware_concurrency();
else
nthreads = in_nthreads;
nthreads = std::clamp<size_t>(nthreads, 1, 64);
if (!experiment.IsGeometryTransformed())
SetupRawGeom(experiment, mask.GetMaskRaw());
else
SetupConvGeom(experiment.GetDiffractionGeometry(),mask.GetMask());
UpdateMaxBinNumber();
pixel_to_bin_checksum = TableChecksum(pixel_to_bin.data(), pixel_to_bin.size() * sizeof(uint16_t));
corrections_checksum = TableChecksum(corrections.data(), corrections.size() * sizeof(float));
}
void AzimuthalIntegrationMapping::SetupConvGeomRows(const DiffractionGeometry &geom, const std::vector<uint32_t> &mask,
size_t row0, size_t row_end) {
for (size_t row = row0; row < row_end && row < height; row++) {
for (size_t col = 0; col < width; col++)
SetupPixel(geom, mask, row * width + col, col, row);
}
}
void AzimuthalIntegrationMapping::SetupConvGeom(const DiffractionGeometry &geom, const std::vector<uint32_t> &mask) {
pixel_to_bin.resize(width * height, UINT16_MAX);
pixel_resolution.resize(width * height, 0);
corrections.resize(width * height, 0);
if (mask.size() != width * height)
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Mask size invalid");
if (nthreads <= 1) {
SetupConvGeomRows(geom, mask, 0, height);
} else {
auto local_nthreads = std::min(nthreads, height);
std::vector<std::future<void>> futures;
futures.reserve(local_nthreads);
for (size_t t = 0; t < local_nthreads; ++t)
futures.emplace_back(std::async(std::launch::async,
&AzimuthalIntegrationMapping::SetupConvGeomRows,
this, std::cref(geom), std::cref(mask),
t * height / local_nthreads,
(t + 1) * height / local_nthreads));
for (auto &f: futures)
f.get();
}
}
void AzimuthalIntegrationMapping::SetupRawGeom(const DiffractionExperiment &experiment,
const std::vector<uint32_t> &mask) {
if (mask.size() != RAW_MODULE_SIZE * experiment.GetModulesNum())
throw JFJochException(JFJochExceptionCategory::InputParameterInvalid, "Mask size invalid");
pixel_to_bin.resize(RAW_MODULE_SIZE * experiment.GetModulesNum(), UINT16_MAX);
pixel_resolution.resize(RAW_MODULE_SIZE * experiment.GetModulesNum(), 0);
corrections.resize(RAW_MODULE_SIZE * experiment.GetModulesNum(), 0);
auto geom = experiment.GetDiffractionGeometry();
if (nthreads <= 1) {
for (int m = 0; m < experiment.GetModulesNum(); m++) {
for (int pxl = 0; pxl < RAW_MODULE_SIZE; pxl++) {
auto [x,y] = RawToConvertedCoordinate(experiment, m, pxl);
SetupPixel(geom, mask, m * RAW_MODULE_SIZE + pxl, x, y);
}
}
} else {
auto local_nthreads = std::min<size_t>(nthreads, experiment.GetModulesNum());
std::vector<std::future<void>> futures;
futures.reserve(local_nthreads);
for (size_t t = 0; t < local_nthreads; ++t) {
const size_t module_begin = t * experiment.GetModulesNum() / local_nthreads;
const size_t module_end = (t + 1) * experiment.GetModulesNum() / local_nthreads;
futures.emplace_back(std::async(std::launch::async, [&, module_begin, module_end] {
for (size_t m = module_begin; m < module_end; ++m) {
for (int pxl = 0; pxl < RAW_MODULE_SIZE; ++pxl) {
auto [x, y] = RawToConvertedCoordinate(experiment, m, pxl);
SetupPixel(geom, mask, m * RAW_MODULE_SIZE + pxl, x, y);
}
}
}));
}
for (auto &f: futures)
f.get();
}
}
void AzimuthalIntegrationMapping::SetupPixel(const DiffractionGeometry &geom,
const std::vector<uint32_t> &mask,
uint32_t pxl, uint32_t col, uint32_t row) {
if (mask[pxl] != 0)
return;
auto x = static_cast<float>(col);
auto y = static_cast<float>(row);
float d = geom.PxlToRes(x, y);
float phi_rad = geom.Phi_rad(x, y);
pixel_resolution[pxl] = d;
float corr = 1.0;
if (settings.IsSolidAngleCorrection())
corr /= geom.CalcAzIntSolidAngleCorr(x, y);
if (settings.IsPolarizationCorrection() && polarization_factor)
corr /= geom.CalcAzIntPolarizationCorr(x, y, polarization_factor.value());
corrections[pxl] = corr;
if (d > 0) {
float q = 2.0f * static_cast<float>(PI) / d;
pixel_to_bin[pxl] = settings.GetBin(q, phi_rad * 180.0 / PI);
}
}
uint16_t AzimuthalIntegrationMapping::GetBinNumber() const {
return settings.GetBinCount();
}
const std::vector<uint16_t> &AzimuthalIntegrationMapping::GetPixelToBin() const {
return pixel_to_bin;
}
const std::vector<float> &AzimuthalIntegrationMapping::GetBinToQ() const {
return bin_to_q;
}
const std::vector<float> &AzimuthalIntegrationMapping::GetBinToD() const {
return bin_to_d;
}
const std::vector<float> &AzimuthalIntegrationMapping::GetBinToTwoTheta() const {
return bin_to_2theta;
}
const std::vector<float> &AzimuthalIntegrationMapping::GetBinToPhi() const {
return bin_to_phi;
}
uint16_t AzimuthalIntegrationMapping::QToBin(float q) const {
return settings.QToBin(q);
}
void AzimuthalIntegrationMapping::UpdateMaxBinNumber() {
bin_to_q.resize(settings.GetBinCount());
bin_to_d.resize(settings.GetBinCount());
bin_to_2theta.resize(settings.GetBinCount());
bin_to_phi.resize(settings.GetBinCount());
for (int j = 0; j < settings.GetAzimuthalBinCount(); j++) {
for (int i = 0; i < settings.GetQBinCount(); i++) {
bin_to_q[j * settings.GetQBinCount() + i] = static_cast<float>(settings.GetQSpacing_recipA() * (i + 0.5) + settings.GetLowQ_recipA());
bin_to_d[j * settings.GetQBinCount() + i] = 2.0f * static_cast<float>(PI) / bin_to_q[j * settings.GetQBinCount() + i];
bin_to_2theta[j * settings.GetQBinCount() + i] = 2.0f * asinf(bin_to_q[i] * wavelength / (4.0f * static_cast<float>(PI))) * 180.0f /
static_cast<float>(PI);
bin_to_phi[j * settings.GetQBinCount() + i] = static_cast<float>(j) * 360.0f / static_cast<float>(settings.GetAzimuthalBinCount());
}
}
}
const std::vector<float> &AzimuthalIntegrationMapping::Corrections() const {
return corrections;
}
const std::vector<float> &AzimuthalIntegrationMapping::Resolution() const {
return pixel_resolution;
}
uint64_t AzimuthalIntegrationMapping::GetPixelToBinChecksum() const {
return pixel_to_bin_checksum;
}
uint64_t AzimuthalIntegrationMapping::GetCorrectionsChecksum() const {
return corrections_checksum;
}
std::shared_ptr<const std::vector<uint32_t>>
AzimuthalIntegrationMapping::ResolutionMaskBits(std::optional<float> high_res,
std::optional<float> low_res) const {
const std::lock_guard lock(res_mask_mutex);
if (res_mask_bits && res_mask_high == high_res && res_mask_low == low_res)
return res_mask_bits;
// An unset limit masks nothing at that end. At the high-resolution end 0 does that on its own - no
// pixel has d < 0, and the detector's own edge is where the pixels stop anyway; at the
// low-resolution end every pixel lies above any finite stand-in, so it takes an infinite one.
const float high = high_res.value_or(0.0f);
const float low = low_res.value_or(INFINITY);
const size_t npixel = pixel_resolution.size();
auto bits = std::make_shared<std::vector<uint32_t>>(npixel / 32 + (npixel % 32 != 0 ? 1 : 0), 0);
for (size_t i = 0; i < npixel; i++)
if (pixel_resolution[i] > low || pixel_resolution[i] < high)
(*bits)[i / 32] |= 1u << (i % 32);
res_mask_high = high_res;
res_mask_low = low_res;
res_mask_bits = bits;
return bits;
}
const AzimuthalIntegrationSettings &AzimuthalIntegrationMapping::Settings() const {
return settings;
}
size_t AzimuthalIntegrationMapping::GetWidth() const {
return width;
}
size_t AzimuthalIntegrationMapping::GetHeight() const {
return height;
}
int32_t AzimuthalIntegrationMapping::GetAzimuthalBinCount() const {
return settings.GetAzimuthalBinCount();
}
int32_t AzimuthalIntegrationMapping::GetQBinCount() const {
return settings.GetQBinCount();
}
size_t AzimuthalIntegrationMapping::GetNThreads() const {
return nthreads;
}