Files
Jungfraujoch/tests/BraggIntegrationEngineGPUTest.cpp
T
leonarski_fandClaude Opus 5 09fb8e0306 Bragg integration: clip the background ring high side instead of trimming it
The r2..r3 background ring was averaged with a 10% SYMMETRIC trimmed mean. A
symmetric trim is not a consistent estimator of the mean of a right-skewed
(Poisson) sample: on a clean Poisson ring it sits ~0.1 ct/px BELOW the true
mean at every level, and with ~50 signal pixels in the r1 disk that
under-subtraction adds ~5 counts to every partial on every frame. Measured two
independent ways on four rotation datasets - stored background_mean against a
plain ring mean over the same pixels on reflection-free frames, and directly on
apertures that provably hold no reflection. Empty-aperture pedestal, counts:
plain mean -0.03..-0.20, 10% symmetric trim +5.05..+6.34, 4 sigma clip
+0.02..+0.54.

Replace it with a high-side-only sigma clip at mean + n*sqrt(mean), n = 4 for
monochromatic data. It rejects the same one-sided contamination the trim was
there for - better, in fact: a 40 px neighbour core at +100 ct shifts the trim
by +10.1 ct/px, because a symmetric trim collapses once contamination exceeds
~10% of the ring, versus +0.009 ct/px at 4 sigma. False rejection on a clean
ring is 0.04-0.39%. Broadband data keep their tuned 3 sigma clip unchanged. The
trim stays reachable with --background-trim for back compatibility; setting
either estimator clears the other, so they can never stack. --integrator boxsum
does not take the clip (matching what the shipped clip already did), so it now
uses the plain ring mean unless --background-trim is given.

The intensities get measurably more accurate: per-shell agreement with an
independent processing of the same images improves on 14 of 16 crystals
(weighted -0.0347, outermost shell 12/4), the outermost-shell R_meas NUMERATOR
- absolute scatter, not a denominator effect - falls 13.5% median on 16/5, and
CC1/2 in the outer shell improves on 14/7.

EXPECT <I/sigma> TO FALL AND EDGE R_meas TO RISE. Both are inflated by
information-free counts, so both get worse when the bias is removed; neither is
evidence against this change. That fingerprint is exactly how the trimmed mean
was accepted in the first place.

Known cost: over the 37-crystal rotation battery the de-novo space-group count
goes 34 OK / 3 DIFF to 33 / 4. The single regression is a two-lattice crystal
whose merge fails the absolute-sanity gate under either background (R_meas
63.5%, CC1/2 72.2%) and which carries an unresolved indexing ambiguity on the
very operator being scored, so its operator CC is diluted by construction. No
other crystal changes space group, and twin protection is not weakened - the
H-ratio veto that refuses genuinely twinned crystals gets MORE decisive
(1.63 -> 1.84, 2.83 -> 3.99).

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-05 18:03:11 +02:00

210 lines
9.7 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <catch2/catch_all.hpp>
#include "../common/CUDAWrapper.h"
#ifdef JFJOCH_USE_CUDA
#include <chrono>
#include <cmath>
#include <vector>
#include "../common/BraggIntegrationSettings.h"
#include "../common/DetectorSetup.h"
#include "../common/DiffractionExperiment.h"
#include "../common/Reflection.h"
#include "../image_analysis/bragg_integration/BraggIntegrationEngineCPU.h"
#include "../image_analysis/bragg_integration/BraggIntegrationEngineGPU.h"
#include "../image_analysis/image_preprocessing/ImagePreprocessorBufferGPU.h"
namespace {
// A grid of clean Gaussian spots on a flat background, each seeding one predicted reflection.
struct Scene {
std::vector<int32_t> image;
std::vector<Reflection> predicted;
size_t width = 0, height = 0;
};
Reflection MakeReflection(float x, float y, float d, int hkl) {
Reflection r{};
r.h = hkl; r.k = hkl; r.l = hkl;
r.predicted_x = x;
r.predicted_y = y;
r.d = d;
r.rlp = 1.0f;
r.partiality = 1.0f;
return r;
}
Scene BuildScene(size_t width, size_t height, int spacing = 60) {
Scene s;
s.width = width;
s.height = height;
s.image.assign(width * height, 12); // flat background
// A grid of spots, well separated so background rings do not overlap the neighbours' disks.
// A spread of intensities (some weak, some very strong) and a spread of d (so several resolution
// shells are populated) exercises the strong-spot selection, shell learning and the fit.
const int margin = 45;
int hkl = 1;
for (int gy = 0; margin + gy * spacing < static_cast<int>(height) - margin; ++gy) {
for (int gx = 0; margin + gx * spacing < static_cast<int>(width) - margin; ++gx) {
const float cx = static_cast<float>(margin + gx * spacing) + 0.3f; // sub-pixel offset
const float cy = static_cast<float>(margin + gy * spacing) - 0.2f;
const double amp = 150.0 + 60.0 * ((gx * 7 + gy * 13) % 30); // 150..1890
const double sigma = 1.3;
for (int dy = -6; dy <= 6; ++dy)
for (int dx = -6; dx <= 6; ++dx) {
const int x = static_cast<int>(std::lround(cx)) + dx;
const int y = static_cast<int>(std::lround(cy)) + dy;
if (x < 0 || y < 0 || x >= static_cast<int>(width) || y >= static_cast<int>(height)) continue;
const double ex = x - cx, ey = y - cy;
const double g = amp * std::exp(-(ex * ex + ey * ey) / (2.0 * sigma * sigma));
s.image[y * width + x] += static_cast<int32_t>(std::lround(g));
}
const float d = 1.4f + 0.12f * static_cast<float>((gx + gy) % 12); // 1.4..2.72 A
s.predicted.push_back(MakeReflection(cx, cy, d, hkl++));
}
}
// A few masked (INT32_MIN) and saturated (INT32_MAX) pixels in background gaps to exercise the
// validity rejection in both engines identically.
for (int k = 0; k < 20; ++k) {
const size_t idx = (static_cast<size_t>(k) * 2654435761u) % s.image.size();
s.image[idx] = (k % 2) ? INT32_MIN : INT32_MAX;
}
return s;
}
// clip_nsigma 0 selects the OTHER background-ring estimator, the symmetric trim, so the two branches
// the CPU and GPU each implement separately are both covered.
DiffractionExperiment MakeExperiment(IntegratorMode mode, std::optional<float> bandwidth_fwhm,
float clip_nsigma = 4.0f,
const DetectorSetup &det = DetJF(2)) {
DiffractionExperiment experiment(det); // DetJF(2) (small) keeps the correctness test fast
experiment.DetectorDistance_mm(100.0f).IncidentEnergy_keV(WVL_1A_IN_KEV)
.BeamX_pxl(400.0f).BeamY_pxl(400.0f);
experiment.BandwidthFWHM(bandwidth_fwhm);
BraggIntegrationSettings settings;
settings.Integrator(mode);
if (clip_nsigma > 0.0f)
settings.BackgroundClipNSigma(clip_nsigma);
else
settings.BackgroundTrimFraction(0.10f);
experiment.ImportBraggIntegrationSettings(settings);
return experiment;
}
void CompareCpuVsGpu(IntegratorMode mode, std::optional<float> bandwidth_fwhm,
float clip_nsigma = 4.0f) {
const DiffractionExperiment experiment = MakeExperiment(mode, bandwidth_fwhm, clip_nsigma);
const size_t width = experiment.GetXPixelsNum();
const size_t height = experiment.GetYPixelsNum();
const size_t npixel = experiment.GetPixelsNum();
REQUIRE(npixel == width * height);
const Scene scene = BuildScene(width, height);
REQUIRE(scene.image.size() == npixel);
REQUIRE(scene.predicted.size() > 60);
// CPU reference
ImagePreprocessorBuffer cpu_image(npixel);
for (size_t i = 0; i < npixel; ++i)
cpu_image[i] = scene.image[i];
BraggIntegrationEngineCPU cpu(experiment);
const auto out_cpu = cpu.Run(cpu_image, scene.predicted, scene.predicted.size(), 5);
// GPU under test, identical input uploaded to the device
auto stream = std::make_shared<CudaStream>();
ImagePreprocessorBufferGPU gpu_image(npixel);
for (size_t i = 0; i < npixel; ++i)
gpu_image[i] = scene.image[i];
REQUIRE(cudaMemcpyAsync(gpu_image.getGPUBuffer(), gpu_image.getBuffer().data(),
npixel * sizeof(int32_t), cudaMemcpyHostToDevice, *stream) == cudaSuccess);
BraggIntegrationEngineGPU gpu(experiment, stream);
const auto out_gpu = gpu.Run(gpu_image, scene.predicted, scene.predicted.size(), 5);
// The ok/observed decisions are deterministic geometry, so both engines return the same set in
// the same (predicted-index) order. Intensities differ only by float rounding and the unordered
// atomic summation of the learned profile, so compare up to a small tolerance.
REQUIRE(out_gpu.size() == out_cpu.size());
REQUIRE(out_cpu.size() > 40);
for (size_t i = 0; i < out_cpu.size(); ++i) {
INFO("mode " << static_cast<int>(mode) << " reflection " << i << " hkl " << out_cpu[i].h);
CHECK(out_gpu[i].h == out_cpu[i].h);
CHECK(out_gpu[i].image_number == out_cpu[i].image_number);
CHECK(out_gpu[i].bkg == Catch::Approx(out_cpu[i].bkg).epsilon(0.02).margin(0.5));
CHECK(out_gpu[i].I == Catch::Approx(out_cpu[i].I).epsilon(0.03).margin(2.0));
CHECK(out_gpu[i].sigma == Catch::Approx(out_cpu[i].sigma).epsilon(0.03).margin(0.5));
}
}
} // namespace
TEST_CASE("BraggIntegrationEngineGPU_MatchesCPU") {
if (get_gpu_count() == 0) {
WARN("No CUDA GPU present. Skipping BraggIntegrationEngineGPU_MatchesCPU");
return;
}
SECTION("BoxSum") { CompareCpuVsGpu(IntegratorMode::BoxSum, std::nullopt); }
SECTION("ProfileGaussian mono") { CompareCpuVsGpu(IntegratorMode::ProfileGaussian, std::nullopt); }
SECTION("ProfileGaussian broadband") { CompareCpuVsGpu(IntegratorMode::ProfileGaussian, 0.03f); }
SECTION("ProfileEmpirical") { CompareCpuVsGpu(IntegratorMode::ProfileEmpirical, std::nullopt); }
SECTION("ProfileGaussian mono trim") { CompareCpuVsGpu(IntegratorMode::ProfileGaussian, std::nullopt, 0.0f); }
}
// Hidden ([.]) benchmark: the raison d'etre of the GPU port is < 2 ms/frame (vs ~142 ms on the CPU
// for ProfileIntegrate2D). Run explicitly with: ./jfjoch_test "[bragg_bench]"
TEST_CASE("BraggIntegrationEngineGPU_Benchmark", "[.][bragg_bench]") {
if (get_gpu_count() == 0) {
WARN("No CUDA GPU present. Skipping benchmark");
return;
}
const DiffractionExperiment experiment = MakeExperiment(IntegratorMode::ProfileGaussian, std::nullopt,
4.0f, DetJF4M());
const size_t width = experiment.GetXPixelsNum();
const size_t height = experiment.GetYPixelsNum();
const size_t npixel = experiment.GetPixelsNum();
REQUIRE(npixel == width * height);
auto stream = std::make_shared<CudaStream>();
BraggIntegrationEngineGPU gpu(experiment, stream);
for (int spacing : {28, 40, 60, 90}) {
const Scene scene = BuildScene(width, height, spacing);
const size_t nrefl = scene.predicted.size();
ImagePreprocessorBufferGPU gpu_image(npixel);
for (size_t i = 0; i < npixel; ++i) gpu_image[i] = scene.image[i];
REQUIRE(cudaMemcpyAsync(gpu_image.getGPUBuffer(), gpu_image.getBuffer().data(),
npixel * sizeof(int32_t), cudaMemcpyHostToDevice, *stream) == cudaSuccess);
cudaStreamSynchronize(*stream);
auto run = [&] { return gpu.Run(gpu_image, scene.predicted, nrefl, 0); };
for (int i = 0; i < 5; ++i) run(); // warm-up (allocations, JIT)
const int iters = 100;
const auto t0 = std::chrono::steady_clock::now();
size_t observed = 0;
for (int i = 0; i < iters; ++i) observed += run().size();
const auto t1 = std::chrono::steady_clock::now();
const double ms = std::chrono::duration<double, std::milli>(t1 - t0).count() / iters;
BraggIntegrationEngineCPU cpu(experiment);
ImagePreprocessorBuffer cpu_image(npixel);
for (size_t i = 0; i < npixel; ++i) cpu_image[i] = scene.image[i];
const auto c0 = std::chrono::steady_clock::now();
const size_t cpu_observed = cpu.Run(cpu_image, scene.predicted, nrefl, 0).size();
const auto c1 = std::chrono::steady_clock::now();
const double cpu_ms = std::chrono::duration<double, std::milli>(c1 - c0).count();
WARN(width << "x" << height << " | " << nrefl << " refl (" << observed / iters
<< " obs) | GPU " << ms << " ms | CPU " << cpu_ms << " ms (" << cpu_observed
<< " obs) | speedup " << cpu_ms / ms << "x");
}
}
#endif