Files
Jungfraujoch/tests/FrenchWilsonTest.cpp
T
leonarski_fandClaude Opus 5.5 4979f8aa68 French-Wilson: anisotropic Wilson prior from the fitted anisotropy tensor
The French-Wilson prior of each reflection is now epsilon * K_shell * a(h), with
a(h) = exp(-1/2 s^T B s) from the deviatoric tensor AnalyzeAnisotropy already fits
(the form it is fitted in) and K_shell = sum(I/eps) / sum(a), so a shell's priors still
average to its measured mean. The amplitudes are made isotropically at the merge as
before and made again once the tensor exists (full pipeline and --mode scale). Only
F/SIGF and F(+)/F(-) change; IMEAN/I(+)/I(-) are bit-identical. Applied whenever a
tensor was fitted, with no detection gate: a near-isotropic tensor gives a(h) ~ 1 and
the isotropic prior back, and the prior wants the best estimate of <I> along h whatever
its cause. Follows ctruncate's anisotropic prior (Ballard & Stein, CCP4); credit in
ACKNOWLEDGEMENT.md, CPU_DATA_ANALYSIS.md and at the algorithm.

--model scaling (ModelScaling.cpp) was checked: k_overall + symmetry-constrained
anisotropic B + flat bulk solvent fitted on the working set, against the same FW F
written to the MTZ - as REFMAC/phenix.refine do. No change needed.

Evidence (REFMAC 10-cycle restrained refinement of the deposited model, R-free on
the depositor's free reflections shared by both data sets; base = rc174 processing,
same IMEAN):
  set   base    new     d          set   base    new     d
  9rcs  0.3475  0.3494  +0.0019    8qq7  0.4452  0.4503  +0.0051
  9yzk  0.3192  0.3171  -0.0021    9hs7  0.2898  0.2465  -0.0433
  6yqf  0.4642  0.4543  -0.0099    5nw5  0.3256  0.3206  -0.0050
  7n2s  0.3126  0.2998  -0.0128    6z8o  0.2927  0.2892  -0.0035
  6qaj  0.3390  0.3038  -0.0352    6moj  0.2756  0.2651  -0.0105
  6r72  0.3826  0.3797  -0.0029    7qij  0.3239  0.3140  -0.0099
  anisotropic sets: median -0.0075, mean -0.0107, 10/12 better
  isotropic controls: 5reo -0.0014, 7kcn +0.0003, 6fid +0.0002, 11if 0.0000
rugnux's own --model R-free moves the same way (median about -0.019; 5nw5 +0.006),
R_model shell-scaled too; dep_cc_delta unchanged (intensity based). The adoption rule
(median gain >= 0.005 on the anisotropic sets, no control worse than +0.002) is met.
The two sets that lose are the one with a FLAT resolution signature (8qq7) and 9rcs,
where the exp form drives the dead direction's prior to ~0 beyond 3.7 A.
For scale: ctruncate's own anisotropic prior on the same merges moved the same
REFMAC R-free by a median of only -0.0008 (9hs7 +0.026).
Inhouse lyso_x06da_ref, thau_x10sa_0p1deg: every battery metric unchanged.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
2026-10-04 14:24:14 +02:00

219 lines
10 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <catch2/catch_all.hpp>
#include <cmath>
#include <vector>
#include "../image_analysis/scale_merge/AnisotropyAnalysis.h"
#include "../image_analysis/scale_merge/FrenchWilson.h"
namespace {
const gemmi::SpaceGroup &SG(int number) { return *gemmi::find_spacegroup_by_number(number); }
MergedReflection Refl(int h, int k, int l, float d, float I, float sigma) {
MergedReflection r;
r.h = h; r.k = k; r.l = l; r.d = d; r.I = I; r.sigma = sigma;
return r;
}
// A resolution-spread of ordinary reflections so a Wilson mean can be formed per shell.
std::vector<MergedReflection> Background() {
std::vector<MergedReflection> v;
for (int h = 1; h <= 12; ++h)
for (int k = 0; k <= 12; ++k)
for (int l = 0; l <= 12; ++l)
v.push_back(Refl(h, k, l, 40.0f / (1 + h * h + k * k + l * l), 800.0f, 20.0f));
return v;
}
}
TEST_CASE("French-Wilson: strong reflections reduce to sqrt(I)", "[french_wilson]") {
auto v = Background();
v.push_back(Refl(1, 0, 0, 25.0f, 40000.0f, 50.0f)); // I/sigma = 800, clearly strong
ApplyFrenchWilson(v, SG(1));
CHECK(v.back().F == Catch::Approx(std::sqrt(40000.0)).epsilon(0.02)); // ~200
CHECK(v.back().sigmaF >= 0.0f);
CHECK(std::isfinite(v.back().sigmaF));
}
TEST_CASE("French-Wilson: weak and negative intensities get a positive amplitude", "[french_wilson]") {
auto v = Background();
v.push_back(Refl(2, 0, 0, 20.0f, -40.0f, 50.0f)); // negative measured intensity
v.push_back(Refl(3, 0, 0, 15.0f, 10.0f, 50.0f)); // weak, I < sigma
ApplyFrenchWilson(v, SG(1));
const auto& neg = v[v.size() - 2];
const auto& weak = v.back();
CHECK(std::isfinite(neg.F));
CHECK(neg.F > 0.0f); // Bayesian estimate is positive (naive sqrt would give 0)
CHECK(std::isfinite(weak.F));
CHECK(weak.F > 0.0f);
}
TEST_CASE("French-Wilson: amplitudes are always finite and non-negative", "[french_wilson]") {
std::vector<MergedReflection> v;
// A deliberate mix: strong, weak, negative, tiny sigma, across resolution.
for (int i = 0; i < 300; ++i) {
const float d = 20.0f / (1 + 0.05f * i);
const float I = (i % 7 == 0) ? -30.0f : static_cast<float>((i % 50) * 40);
v.push_back(Refl(1 + i, 2, 3, d, I, 25.0f));
}
ApplyFrenchWilson(v, SG(96)); // P4(3)2(1)2 (has centric reflections + epsilon>1 axes)
for (const auto& r : v) {
CHECK(std::isfinite(r.F));
CHECK(r.F >= 0.0f);
CHECK(std::isfinite(r.sigmaF));
CHECK(r.sigmaF >= 0.0f);
}
}
TEST_CASE("French-Wilson: centric weak reflection gets a smaller amplitude than acentric", "[french_wilson]") {
// At the same resolution (same Wilson mean Sigma) and the same near-zero intensity, the centric
// prior puts more weight near |F|=0, so the posterior <|F|> is smaller than for an acentric
// reflection (0.80*sqrt(Sigma) vs 0.89*sqrt(Sigma) at I=0). This pins the centric/acentric prior
// the right way round: if the two priors were swapped the inequality below would flip.
std::vector<MergedReflection> v;
// Uniform strong background across resolution, so every shell has ~the same Wilson mean.
for (int h = 1; h <= 10; ++h)
for (int k = 1; k <= 10; ++k)
for (int l = 1; l <= 6; ++l)
v.push_back(Refl(h, k, l, 20.0f / (0.5f + 0.1f * (h + k + l)), 1000.0f, 30.0f));
// Two weak (I=0) probes at the SAME resolution: (3,1,0) is centric in P4 (l=0 zone), (3,1,4) is
// acentric. d is set directly, so both share a shell (hence Sigma) regardless of the cell.
v.push_back(Refl(3, 1, 0, 5.0f, 0.0f, 10.0f));
v.push_back(Refl(3, 1, 4, 5.0f, 0.0f, 10.0f));
ApplyFrenchWilson(v, SG(75)); // P4
const auto& centric = v[v.size() - 2];
const auto& acentric = v.back();
CHECK(centric.F > 0.0f);
CHECK(acentric.F > 0.0f);
CHECK(centric.F < acentric.F);
}
TEST_CASE("French-Wilson: unusable sigma falls back to sqrt(max(I,0))", "[french_wilson]") {
auto v = Background();
v.push_back(Refl(4, 0, 0, 12.0f, 144.0f, NAN)); // no sigma
v.push_back(Refl(5, 0, 0, 11.0f, -5.0f, NAN)); // no sigma, negative I
ApplyFrenchWilson(v, SG(1));
CHECK(v[v.size() - 2].F == Catch::Approx(12.0f)); // sqrt(144)
CHECK(v.back().F == Catch::Approx(0.0f)); // sqrt(max(-5,0))
}
TEST_CASE("French-Wilson: an intensity far below zero gets no amplitude, as in ctruncate", "[french_wilson]") {
auto v = Background();
v.push_back(Refl(2, 0, 0, 20.0f, -3.6f * 50.0f, 50.0f)); // just above the -3.7 sigma bound
v.push_back(Refl(3, 0, 0, 15.0f, -3.8f * 50.0f, 50.0f)); // just below it
ApplyFrenchWilson(v, SG(1));
const auto &kept = v[v.size() - 2];
const auto &rejected = v.back();
CHECK(std::isfinite(kept.F));
CHECK(kept.F > 0.0f);
CHECK(std::isnan(rejected.F));
CHECK(std::isnan(rejected.sigmaF));
CHECK(rejected.I == Catch::Approx(-190.0f)); // the intensity itself is kept
}
TEST_CASE("French-Wilson: rejected intensities stay out of the Wilson prior", "[french_wilson]") {
// One shell of ordinary reflections at <I> = 800, plus a probe. Adding many strongly negative
// intensities to the shell must not change the probe's amplitude: they get no amplitude and do
// not enter the shell mean.
auto make = [](bool with_negatives) {
std::vector<MergedReflection> v;
for (int h = 1; h <= 40; ++h)
v.push_back(Refl(h, 1, 1, 3.0f + 0.01f * h, 800.0f, 20.0f));
if (with_negatives)
for (int h = 1; h <= 40; ++h)
v.push_back(Refl(h, 2, 1, 3.0f + 0.01f * h, -400.0f, 20.0f)); // -20 sigma
v.push_back(Refl(1, 3, 1, 3.0f, 10.0f, 20.0f));
FrenchWilsonOptions opts;
opts.num_shells = 1;
ApplyFrenchWilson(v, SG(1), opts);
CHECK(v.back().F != Catch::Approx(std::sqrt(10.0f)).epsilon(0.01)); // the posterior, not the naive sqrt
return v.back().F;
};
CHECK(make(true) == Catch::Approx(make(false)).epsilon(1e-6));
}
TEST_CASE("French-Wilson: a shell with a non-positive mean borrows the lower-resolution shell's prior", "[french_wilson]") {
// Two shells in 1/d^2: the low-resolution one at <I> = 400, the high-resolution one averaging
// below zero. A weak probe in the outer shell must get the amplitude the inner shell's prior
// gives, not the ~0 a prior at the clamp would give.
std::vector<MergedReflection> v;
for (int h = 1; h <= 30; ++h) {
v.push_back(Refl(h, 1, 1, 4.0f, 400.0f, 20.0f));
v.push_back(Refl(h, 2, 1, 2.0f, (h % 2) ? 10.0f : -30.0f, 20.0f));
}
v.push_back(Refl(1, 3, 1, 2.0f, 5.0f, 20.0f)); // outer-shell probe
v.push_back(Refl(1, 4, 1, 4.0f, 5.0f, 20.0f)); // same measurement in the inner shell
FrenchWilsonOptions opts;
opts.num_shells = 2;
opts.min_reflections_per_shell = 10;
ApplyFrenchWilson(v, SG(1), opts);
const auto &outer = v[v.size() - 2];
const auto &inner = v.back();
CHECK(outer.F > 1.0f); // a prior at the clamp gives ~1e-5
CHECK(outer.F == Catch::Approx(inner.F).epsilon(1e-3));
}
TEST_CASE("French-Wilson: a 5 sigma intensity is still pulled toward the prior", "[french_wilson]") {
// Below the strong cutoff the posterior applies: on a weak shell's prior (<I> = sigma) a 5 sigma
// acentric measurement is pulled to roughly I - sigma^2/<I> = 4 sigma, not left at sqrt(I).
std::vector<MergedReflection> v;
for (int h = 1; h <= 60; ++h)
v.push_back(Refl(h, 1, 1, 3.0f + 0.01f * h, (h % 3) ? 0.0f : 60.0f, 20.0f)); // <I> = 20 = sigma
v.push_back(Refl(1, 5, 1, 3.0f, 100.0f, 20.0f));
FrenchWilsonOptions opts;
opts.num_shells = 1;
ApplyFrenchWilson(v, SG(1), opts);
CHECK(v.back().F < 0.95f * std::sqrt(100.0f));
CHECK(v.back().F > std::sqrt(60.0f));
}
TEST_CASE("French-Wilson: the anisotropic prior follows the tensor along each direction", "[french_wilson]") {
// A cubic 10 A cell with a deviatoric tensor of +20 A^2 along x and -10 A^2 along y and z: every
// ordinary reflection measures exactly 1000 * exp(-1/2 s^T B s), one shell. A weak probe along x
// must get the amplitude an isotropic prior at x's own <I> gives, and less than the same
// measurement along y.
const gemmi::UnitCell cell(10, 10, 10, 90, 90, 90);
AnisotropyResult tensor;
const double eigenvalue[3] = {20.0, -10.0, -10.0};
for (int n = 0; n < 3; ++n) {
tensor.eigenvalue[n] = eigenvalue[n];
for (int j = 0; j < 3; ++j)
tensor.eigenvector[n][j] = n == j ? 1.0 : 0.0;
}
const gemmi::Mat33 q = AnisotropyTensorHKL(tensor, cell);
auto expected = [&](int h, int k, int l) {
const gemmi::Vec3 v(h, k, l);
return 1000.0 * std::exp(-0.5 * v.dot(q.multiply(v)));
};
CHECK(expected(5, 0, 0) == Catch::Approx(1000.0 * std::exp(-0.5 * 20.0 * 0.25)));
auto with_probes = [](std::vector<MergedReflection> v) {
v.push_back(Refl(5, 0, 0, 2.0f, 30.0f, 40.0f));
v.push_back(Refl(0, 5, 0, 2.0f, 30.0f, 40.0f));
return v;
};
std::vector<MergedReflection> aniso_set;
for (int h = -8; h <= 8; ++h)
for (int k = -8; k <= 8; ++k)
for (int l = 1; l <= 8; ++l)
aniso_set.push_back(Refl(h, k, l, static_cast<float>(cell.calculate_d({{h, k, l}})),
static_cast<float>(expected(h, k, l)), 5.0f));
std::vector<MergedReflection> iso_set; // the same reflections, all at the x probe's <I>
for (const auto &r : aniso_set)
iso_set.push_back(Refl(r.h, r.k, r.l, r.d, static_cast<float>(expected(5, 0, 0)), 5.0f));
aniso_set = with_probes(aniso_set);
iso_set = with_probes(iso_set);
FrenchWilsonOptions opts;
opts.num_shells = 1;
ApplyFrenchWilson(iso_set, SG(1), opts);
opts.anisotropy_hkl = q;
ApplyFrenchWilson(aniso_set, SG(1), opts);
const auto &along_x = aniso_set[aniso_set.size() - 2];
const auto &along_y = aniso_set.back();
CHECK(along_x.F == Catch::Approx(iso_set[iso_set.size() - 2].F).epsilon(0.01));
CHECK(along_x.F < 0.9f * along_y.F);
}