Files
Jungfraujoch/tests/FrenchWilsonTest.cpp
T
leonarski_fandClaude Opus 5.5 23f446acbb Rugnux: French-Wilson amplitudes as ctruncate gives them
Compared with ctruncate (CCP4 9, version 1.17.29) on rugnux's own merged
intensities of the open battery arm, three differences:

- ctruncate gives no amplitude to an intensity below -3.7 sigma (exactly
  that bound on every set checked; up to 2498 reflections on one set).
  Rugnux turned them into small, confident amplitudes (F/Fc ~0.3 on the
  sets whose background is over-subtracted on powder/ice rings). They now
  get F = NaN (missing in MTZ, '?' in mmCIF); IMEAN is kept. They are
  also left out of the shell mean that sets the Wilson prior.
- The switch to sqrt(I) at I/sigma = 4 left 4-6 sigma amplitudes 2-3%
  above ctruncate's on every set (1.019-1.028). The posterior now applies
  up to 20 sigma (emulated: 0.995-1.000).
- A shell whose mean intensity is not positive gave a prior at the 1e-10
  clamp and amplitudes of ~0 (one set's outer shell); it now takes the
  nearest lower-resolution shell's mean.

The remaining gap (weak amplitudes 2-5% below ctruncate's in the outer
shells) is ctruncate's anisotropy-corrected prior; not attempted. An
offline R-free ablation put the whole ctruncate conversion at -0.0006
median (15/18 sets better) and dropping its rejected negatives at -0.0009
mean (-0.009 on the worst set).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
2026-09-26 13:43:59 +02:00

170 lines
7.9 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <catch2/catch_all.hpp>
#include <cmath>
#include <vector>
#include "../image_analysis/scale_merge/FrenchWilson.h"
namespace {
const gemmi::SpaceGroup &SG(int number) { return *gemmi::find_spacegroup_by_number(number); }
MergedReflection Refl(int h, int k, int l, float d, float I, float sigma) {
MergedReflection r;
r.h = h; r.k = k; r.l = l; r.d = d; r.I = I; r.sigma = sigma;
return r;
}
// A resolution-spread of ordinary reflections so a Wilson mean can be formed per shell.
std::vector<MergedReflection> Background() {
std::vector<MergedReflection> v;
for (int h = 1; h <= 12; ++h)
for (int k = 0; k <= 12; ++k)
for (int l = 0; l <= 12; ++l)
v.push_back(Refl(h, k, l, 40.0f / (1 + h * h + k * k + l * l), 800.0f, 20.0f));
return v;
}
}
TEST_CASE("French-Wilson: strong reflections reduce to sqrt(I)", "[french_wilson]") {
auto v = Background();
v.push_back(Refl(1, 0, 0, 25.0f, 40000.0f, 50.0f)); // I/sigma = 800, clearly strong
ApplyFrenchWilson(v, SG(1));
CHECK(v.back().F == Catch::Approx(std::sqrt(40000.0)).epsilon(0.02)); // ~200
CHECK(v.back().sigmaF >= 0.0f);
CHECK(std::isfinite(v.back().sigmaF));
}
TEST_CASE("French-Wilson: weak and negative intensities get a positive amplitude", "[french_wilson]") {
auto v = Background();
v.push_back(Refl(2, 0, 0, 20.0f, -40.0f, 50.0f)); // negative measured intensity
v.push_back(Refl(3, 0, 0, 15.0f, 10.0f, 50.0f)); // weak, I < sigma
ApplyFrenchWilson(v, SG(1));
const auto& neg = v[v.size() - 2];
const auto& weak = v.back();
CHECK(std::isfinite(neg.F));
CHECK(neg.F > 0.0f); // Bayesian estimate is positive (naive sqrt would give 0)
CHECK(std::isfinite(weak.F));
CHECK(weak.F > 0.0f);
}
TEST_CASE("French-Wilson: amplitudes are always finite and non-negative", "[french_wilson]") {
std::vector<MergedReflection> v;
// A deliberate mix: strong, weak, negative, tiny sigma, across resolution.
for (int i = 0; i < 300; ++i) {
const float d = 20.0f / (1 + 0.05f * i);
const float I = (i % 7 == 0) ? -30.0f : static_cast<float>((i % 50) * 40);
v.push_back(Refl(1 + i, 2, 3, d, I, 25.0f));
}
ApplyFrenchWilson(v, SG(96)); // P4(3)2(1)2 (has centric reflections + epsilon>1 axes)
for (const auto& r : v) {
CHECK(std::isfinite(r.F));
CHECK(r.F >= 0.0f);
CHECK(std::isfinite(r.sigmaF));
CHECK(r.sigmaF >= 0.0f);
}
}
TEST_CASE("French-Wilson: centric weak reflection gets a smaller amplitude than acentric", "[french_wilson]") {
// At the same resolution (same Wilson mean Sigma) and the same near-zero intensity, the centric
// prior puts more weight near |F|=0, so the posterior <|F|> is smaller than for an acentric
// reflection (0.80*sqrt(Sigma) vs 0.89*sqrt(Sigma) at I=0). This pins the centric/acentric prior
// the right way round: if the two priors were swapped the inequality below would flip.
std::vector<MergedReflection> v;
// Uniform strong background across resolution, so every shell has ~the same Wilson mean.
for (int h = 1; h <= 10; ++h)
for (int k = 1; k <= 10; ++k)
for (int l = 1; l <= 6; ++l)
v.push_back(Refl(h, k, l, 20.0f / (0.5f + 0.1f * (h + k + l)), 1000.0f, 30.0f));
// Two weak (I=0) probes at the SAME resolution: (3,1,0) is centric in P4 (l=0 zone), (3,1,4) is
// acentric. d is set directly, so both share a shell (hence Sigma) regardless of the cell.
v.push_back(Refl(3, 1, 0, 5.0f, 0.0f, 10.0f));
v.push_back(Refl(3, 1, 4, 5.0f, 0.0f, 10.0f));
ApplyFrenchWilson(v, SG(75)); // P4
const auto& centric = v[v.size() - 2];
const auto& acentric = v.back();
CHECK(centric.F > 0.0f);
CHECK(acentric.F > 0.0f);
CHECK(centric.F < acentric.F);
}
TEST_CASE("French-Wilson: unusable sigma falls back to sqrt(max(I,0))", "[french_wilson]") {
auto v = Background();
v.push_back(Refl(4, 0, 0, 12.0f, 144.0f, NAN)); // no sigma
v.push_back(Refl(5, 0, 0, 11.0f, -5.0f, NAN)); // no sigma, negative I
ApplyFrenchWilson(v, SG(1));
CHECK(v[v.size() - 2].F == Catch::Approx(12.0f)); // sqrt(144)
CHECK(v.back().F == Catch::Approx(0.0f)); // sqrt(max(-5,0))
}
TEST_CASE("French-Wilson: an intensity far below zero gets no amplitude, as in ctruncate", "[french_wilson]") {
auto v = Background();
v.push_back(Refl(2, 0, 0, 20.0f, -3.6f * 50.0f, 50.0f)); // just above the -3.7 sigma bound
v.push_back(Refl(3, 0, 0, 15.0f, -3.8f * 50.0f, 50.0f)); // just below it
ApplyFrenchWilson(v, SG(1));
const auto &kept = v[v.size() - 2];
const auto &rejected = v.back();
CHECK(std::isfinite(kept.F));
CHECK(kept.F > 0.0f);
CHECK(std::isnan(rejected.F));
CHECK(std::isnan(rejected.sigmaF));
CHECK(rejected.I == Catch::Approx(-190.0f)); // the intensity itself is kept
}
TEST_CASE("French-Wilson: rejected intensities stay out of the Wilson prior", "[french_wilson]") {
// One shell of ordinary reflections at <I> = 800, plus a probe. Adding many strongly negative
// intensities to the shell must not change the probe's amplitude: they get no amplitude and do
// not enter the shell mean.
auto make = [](bool with_negatives) {
std::vector<MergedReflection> v;
for (int h = 1; h <= 40; ++h)
v.push_back(Refl(h, 1, 1, 3.0f + 0.01f * h, 800.0f, 20.0f));
if (with_negatives)
for (int h = 1; h <= 40; ++h)
v.push_back(Refl(h, 2, 1, 3.0f + 0.01f * h, -400.0f, 20.0f)); // -20 sigma
v.push_back(Refl(1, 3, 1, 3.0f, 10.0f, 20.0f));
FrenchWilsonOptions opts;
opts.num_shells = 1;
ApplyFrenchWilson(v, SG(1), opts);
CHECK(v.back().F != Catch::Approx(std::sqrt(10.0f)).epsilon(0.01)); // the posterior, not the naive sqrt
return v.back().F;
};
CHECK(make(true) == Catch::Approx(make(false)).epsilon(1e-6));
}
TEST_CASE("French-Wilson: a shell with a non-positive mean borrows the lower-resolution shell's prior", "[french_wilson]") {
// Two shells in 1/d^2: the low-resolution one at <I> = 400, the high-resolution one averaging
// below zero. A weak probe in the outer shell must get the amplitude the inner shell's prior
// gives, not the ~0 a prior at the clamp would give.
std::vector<MergedReflection> v;
for (int h = 1; h <= 30; ++h) {
v.push_back(Refl(h, 1, 1, 4.0f, 400.0f, 20.0f));
v.push_back(Refl(h, 2, 1, 2.0f, (h % 2) ? 10.0f : -30.0f, 20.0f));
}
v.push_back(Refl(1, 3, 1, 2.0f, 5.0f, 20.0f)); // outer-shell probe
v.push_back(Refl(1, 4, 1, 4.0f, 5.0f, 20.0f)); // same measurement in the inner shell
FrenchWilsonOptions opts;
opts.num_shells = 2;
opts.min_reflections_per_shell = 10;
ApplyFrenchWilson(v, SG(1), opts);
const auto &outer = v[v.size() - 2];
const auto &inner = v.back();
CHECK(outer.F > 1.0f); // a prior at the clamp gives ~1e-5
CHECK(outer.F == Catch::Approx(inner.F).epsilon(1e-3));
}
TEST_CASE("French-Wilson: a 5 sigma intensity is still pulled toward the prior", "[french_wilson]") {
// Below the strong cutoff the posterior applies: on a weak shell's prior (<I> = sigma) a 5 sigma
// acentric measurement is pulled to roughly I - sigma^2/<I> = 4 sigma, not left at sqrt(I).
std::vector<MergedReflection> v;
for (int h = 1; h <= 60; ++h)
v.push_back(Refl(h, 1, 1, 3.0f + 0.01f * h, (h % 3) ? 0.0f : 60.0f, 20.0f)); // <I> = 20 = sigma
v.push_back(Refl(1, 5, 1, 3.0f, 100.0f, 20.0f));
FrenchWilsonOptions opts;
opts.num_shells = 1;
ApplyFrenchWilson(v, SG(1), opts);
CHECK(v.back().F < 0.95f * std::sqrt(100.0f));
CHECK(v.back().F > std::sqrt(60.0f));
}