The French-Wilson prior of each reflection is now epsilon * K_shell * a(h), with a(h) = exp(-1/2 s^T B s) from the deviatoric tensor AnalyzeAnisotropy already fits (the form it is fitted in) and K_shell = sum(I/eps) / sum(a), so a shell's priors still average to its measured mean. The amplitudes are made isotropically at the merge as before and made again once the tensor exists (full pipeline and --mode scale). Only F/SIGF and F(+)/F(-) change; IMEAN/I(+)/I(-) are bit-identical. Applied whenever a tensor was fitted, with no detection gate: a near-isotropic tensor gives a(h) ~ 1 and the isotropic prior back, and the prior wants the best estimate of <I> along h whatever its cause. Follows ctruncate's anisotropic prior (Ballard & Stein, CCP4); credit in ACKNOWLEDGEMENT.md, CPU_DATA_ANALYSIS.md and at the algorithm. --model scaling (ModelScaling.cpp) was checked: k_overall + symmetry-constrained anisotropic B + flat bulk solvent fitted on the working set, against the same FW F written to the MTZ - as REFMAC/phenix.refine do. No change needed. Evidence (REFMAC 10-cycle restrained refinement of the deposited model, R-free on the depositor's free reflections shared by both data sets; base = rc174 processing, same IMEAN): set base new d set base new d 9rcs 0.3475 0.3494 +0.0019 8qq7 0.4452 0.4503 +0.0051 9yzk 0.3192 0.3171 -0.0021 9hs7 0.2898 0.2465 -0.0433 6yqf 0.4642 0.4543 -0.0099 5nw5 0.3256 0.3206 -0.0050 7n2s 0.3126 0.2998 -0.0128 6z8o 0.2927 0.2892 -0.0035 6qaj 0.3390 0.3038 -0.0352 6moj 0.2756 0.2651 -0.0105 6r72 0.3826 0.3797 -0.0029 7qij 0.3239 0.3140 -0.0099 anisotropic sets: median -0.0075, mean -0.0107, 10/12 better isotropic controls: 5reo -0.0014, 7kcn +0.0003, 6fid +0.0002, 11if 0.0000 rugnux's own --model R-free moves the same way (median about -0.019; 5nw5 +0.006), R_model shell-scaled too; dep_cc_delta unchanged (intensity based). The adoption rule (median gain >= 0.005 on the anisotropic sets, no control worse than +0.002) is met. The two sets that lose are the one with a FLAT resolution signature (8qq7) and 9rcs, where the exp form drives the dead direction's prior to ~0 beyond 3.7 A. For scale: ctruncate's own anisotropic prior on the same merges moved the same REFMAC R-free by a median of only -0.0008 (9hs7 +0.026). Inhouse lyso_x06da_ref, thau_x10sa_0p1deg: every battery metric unchanged. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
219 lines
10 KiB
C++
219 lines
10 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include <catch2/catch_all.hpp>
|
|
|
|
#include <cmath>
|
|
#include <vector>
|
|
|
|
#include "../image_analysis/scale_merge/AnisotropyAnalysis.h"
|
|
#include "../image_analysis/scale_merge/FrenchWilson.h"
|
|
|
|
namespace {
|
|
const gemmi::SpaceGroup &SG(int number) { return *gemmi::find_spacegroup_by_number(number); }
|
|
|
|
MergedReflection Refl(int h, int k, int l, float d, float I, float sigma) {
|
|
MergedReflection r;
|
|
r.h = h; r.k = k; r.l = l; r.d = d; r.I = I; r.sigma = sigma;
|
|
return r;
|
|
}
|
|
// A resolution-spread of ordinary reflections so a Wilson mean can be formed per shell.
|
|
std::vector<MergedReflection> Background() {
|
|
std::vector<MergedReflection> v;
|
|
for (int h = 1; h <= 12; ++h)
|
|
for (int k = 0; k <= 12; ++k)
|
|
for (int l = 0; l <= 12; ++l)
|
|
v.push_back(Refl(h, k, l, 40.0f / (1 + h * h + k * k + l * l), 800.0f, 20.0f));
|
|
return v;
|
|
}
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: strong reflections reduce to sqrt(I)", "[french_wilson]") {
|
|
auto v = Background();
|
|
v.push_back(Refl(1, 0, 0, 25.0f, 40000.0f, 50.0f)); // I/sigma = 800, clearly strong
|
|
ApplyFrenchWilson(v, SG(1));
|
|
CHECK(v.back().F == Catch::Approx(std::sqrt(40000.0)).epsilon(0.02)); // ~200
|
|
CHECK(v.back().sigmaF >= 0.0f);
|
|
CHECK(std::isfinite(v.back().sigmaF));
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: weak and negative intensities get a positive amplitude", "[french_wilson]") {
|
|
auto v = Background();
|
|
v.push_back(Refl(2, 0, 0, 20.0f, -40.0f, 50.0f)); // negative measured intensity
|
|
v.push_back(Refl(3, 0, 0, 15.0f, 10.0f, 50.0f)); // weak, I < sigma
|
|
ApplyFrenchWilson(v, SG(1));
|
|
const auto& neg = v[v.size() - 2];
|
|
const auto& weak = v.back();
|
|
CHECK(std::isfinite(neg.F));
|
|
CHECK(neg.F > 0.0f); // Bayesian estimate is positive (naive sqrt would give 0)
|
|
CHECK(std::isfinite(weak.F));
|
|
CHECK(weak.F > 0.0f);
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: amplitudes are always finite and non-negative", "[french_wilson]") {
|
|
std::vector<MergedReflection> v;
|
|
// A deliberate mix: strong, weak, negative, tiny sigma, across resolution.
|
|
for (int i = 0; i < 300; ++i) {
|
|
const float d = 20.0f / (1 + 0.05f * i);
|
|
const float I = (i % 7 == 0) ? -30.0f : static_cast<float>((i % 50) * 40);
|
|
v.push_back(Refl(1 + i, 2, 3, d, I, 25.0f));
|
|
}
|
|
ApplyFrenchWilson(v, SG(96)); // P4(3)2(1)2 (has centric reflections + epsilon>1 axes)
|
|
for (const auto& r : v) {
|
|
CHECK(std::isfinite(r.F));
|
|
CHECK(r.F >= 0.0f);
|
|
CHECK(std::isfinite(r.sigmaF));
|
|
CHECK(r.sigmaF >= 0.0f);
|
|
}
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: centric weak reflection gets a smaller amplitude than acentric", "[french_wilson]") {
|
|
// At the same resolution (same Wilson mean Sigma) and the same near-zero intensity, the centric
|
|
// prior puts more weight near |F|=0, so the posterior <|F|> is smaller than for an acentric
|
|
// reflection (0.80*sqrt(Sigma) vs 0.89*sqrt(Sigma) at I=0). This pins the centric/acentric prior
|
|
// the right way round: if the two priors were swapped the inequality below would flip.
|
|
std::vector<MergedReflection> v;
|
|
// Uniform strong background across resolution, so every shell has ~the same Wilson mean.
|
|
for (int h = 1; h <= 10; ++h)
|
|
for (int k = 1; k <= 10; ++k)
|
|
for (int l = 1; l <= 6; ++l)
|
|
v.push_back(Refl(h, k, l, 20.0f / (0.5f + 0.1f * (h + k + l)), 1000.0f, 30.0f));
|
|
// Two weak (I=0) probes at the SAME resolution: (3,1,0) is centric in P4 (l=0 zone), (3,1,4) is
|
|
// acentric. d is set directly, so both share a shell (hence Sigma) regardless of the cell.
|
|
v.push_back(Refl(3, 1, 0, 5.0f, 0.0f, 10.0f));
|
|
v.push_back(Refl(3, 1, 4, 5.0f, 0.0f, 10.0f));
|
|
ApplyFrenchWilson(v, SG(75)); // P4
|
|
const auto& centric = v[v.size() - 2];
|
|
const auto& acentric = v.back();
|
|
CHECK(centric.F > 0.0f);
|
|
CHECK(acentric.F > 0.0f);
|
|
CHECK(centric.F < acentric.F);
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: unusable sigma falls back to sqrt(max(I,0))", "[french_wilson]") {
|
|
auto v = Background();
|
|
v.push_back(Refl(4, 0, 0, 12.0f, 144.0f, NAN)); // no sigma
|
|
v.push_back(Refl(5, 0, 0, 11.0f, -5.0f, NAN)); // no sigma, negative I
|
|
ApplyFrenchWilson(v, SG(1));
|
|
CHECK(v[v.size() - 2].F == Catch::Approx(12.0f)); // sqrt(144)
|
|
CHECK(v.back().F == Catch::Approx(0.0f)); // sqrt(max(-5,0))
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: an intensity far below zero gets no amplitude, as in ctruncate", "[french_wilson]") {
|
|
auto v = Background();
|
|
v.push_back(Refl(2, 0, 0, 20.0f, -3.6f * 50.0f, 50.0f)); // just above the -3.7 sigma bound
|
|
v.push_back(Refl(3, 0, 0, 15.0f, -3.8f * 50.0f, 50.0f)); // just below it
|
|
ApplyFrenchWilson(v, SG(1));
|
|
const auto &kept = v[v.size() - 2];
|
|
const auto &rejected = v.back();
|
|
CHECK(std::isfinite(kept.F));
|
|
CHECK(kept.F > 0.0f);
|
|
CHECK(std::isnan(rejected.F));
|
|
CHECK(std::isnan(rejected.sigmaF));
|
|
CHECK(rejected.I == Catch::Approx(-190.0f)); // the intensity itself is kept
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: rejected intensities stay out of the Wilson prior", "[french_wilson]") {
|
|
// One shell of ordinary reflections at <I> = 800, plus a probe. Adding many strongly negative
|
|
// intensities to the shell must not change the probe's amplitude: they get no amplitude and do
|
|
// not enter the shell mean.
|
|
auto make = [](bool with_negatives) {
|
|
std::vector<MergedReflection> v;
|
|
for (int h = 1; h <= 40; ++h)
|
|
v.push_back(Refl(h, 1, 1, 3.0f + 0.01f * h, 800.0f, 20.0f));
|
|
if (with_negatives)
|
|
for (int h = 1; h <= 40; ++h)
|
|
v.push_back(Refl(h, 2, 1, 3.0f + 0.01f * h, -400.0f, 20.0f)); // -20 sigma
|
|
v.push_back(Refl(1, 3, 1, 3.0f, 10.0f, 20.0f));
|
|
FrenchWilsonOptions opts;
|
|
opts.num_shells = 1;
|
|
ApplyFrenchWilson(v, SG(1), opts);
|
|
CHECK(v.back().F != Catch::Approx(std::sqrt(10.0f)).epsilon(0.01)); // the posterior, not the naive sqrt
|
|
return v.back().F;
|
|
};
|
|
CHECK(make(true) == Catch::Approx(make(false)).epsilon(1e-6));
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: a shell with a non-positive mean borrows the lower-resolution shell's prior", "[french_wilson]") {
|
|
// Two shells in 1/d^2: the low-resolution one at <I> = 400, the high-resolution one averaging
|
|
// below zero. A weak probe in the outer shell must get the amplitude the inner shell's prior
|
|
// gives, not the ~0 a prior at the clamp would give.
|
|
std::vector<MergedReflection> v;
|
|
for (int h = 1; h <= 30; ++h) {
|
|
v.push_back(Refl(h, 1, 1, 4.0f, 400.0f, 20.0f));
|
|
v.push_back(Refl(h, 2, 1, 2.0f, (h % 2) ? 10.0f : -30.0f, 20.0f));
|
|
}
|
|
v.push_back(Refl(1, 3, 1, 2.0f, 5.0f, 20.0f)); // outer-shell probe
|
|
v.push_back(Refl(1, 4, 1, 4.0f, 5.0f, 20.0f)); // same measurement in the inner shell
|
|
FrenchWilsonOptions opts;
|
|
opts.num_shells = 2;
|
|
opts.min_reflections_per_shell = 10;
|
|
ApplyFrenchWilson(v, SG(1), opts);
|
|
const auto &outer = v[v.size() - 2];
|
|
const auto &inner = v.back();
|
|
CHECK(outer.F > 1.0f); // a prior at the clamp gives ~1e-5
|
|
CHECK(outer.F == Catch::Approx(inner.F).epsilon(1e-3));
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: a 5 sigma intensity is still pulled toward the prior", "[french_wilson]") {
|
|
// Below the strong cutoff the posterior applies: on a weak shell's prior (<I> = sigma) a 5 sigma
|
|
// acentric measurement is pulled to roughly I - sigma^2/<I> = 4 sigma, not left at sqrt(I).
|
|
std::vector<MergedReflection> v;
|
|
for (int h = 1; h <= 60; ++h)
|
|
v.push_back(Refl(h, 1, 1, 3.0f + 0.01f * h, (h % 3) ? 0.0f : 60.0f, 20.0f)); // <I> = 20 = sigma
|
|
v.push_back(Refl(1, 5, 1, 3.0f, 100.0f, 20.0f));
|
|
FrenchWilsonOptions opts;
|
|
opts.num_shells = 1;
|
|
ApplyFrenchWilson(v, SG(1), opts);
|
|
CHECK(v.back().F < 0.95f * std::sqrt(100.0f));
|
|
CHECK(v.back().F > std::sqrt(60.0f));
|
|
}
|
|
|
|
TEST_CASE("French-Wilson: the anisotropic prior follows the tensor along each direction", "[french_wilson]") {
|
|
// A cubic 10 A cell with a deviatoric tensor of +20 A^2 along x and -10 A^2 along y and z: every
|
|
// ordinary reflection measures exactly 1000 * exp(-1/2 s^T B s), one shell. A weak probe along x
|
|
// must get the amplitude an isotropic prior at x's own <I> gives, and less than the same
|
|
// measurement along y.
|
|
const gemmi::UnitCell cell(10, 10, 10, 90, 90, 90);
|
|
AnisotropyResult tensor;
|
|
const double eigenvalue[3] = {20.0, -10.0, -10.0};
|
|
for (int n = 0; n < 3; ++n) {
|
|
tensor.eigenvalue[n] = eigenvalue[n];
|
|
for (int j = 0; j < 3; ++j)
|
|
tensor.eigenvector[n][j] = n == j ? 1.0 : 0.0;
|
|
}
|
|
const gemmi::Mat33 q = AnisotropyTensorHKL(tensor, cell);
|
|
auto expected = [&](int h, int k, int l) {
|
|
const gemmi::Vec3 v(h, k, l);
|
|
return 1000.0 * std::exp(-0.5 * v.dot(q.multiply(v)));
|
|
};
|
|
CHECK(expected(5, 0, 0) == Catch::Approx(1000.0 * std::exp(-0.5 * 20.0 * 0.25)));
|
|
|
|
auto with_probes = [](std::vector<MergedReflection> v) {
|
|
v.push_back(Refl(5, 0, 0, 2.0f, 30.0f, 40.0f));
|
|
v.push_back(Refl(0, 5, 0, 2.0f, 30.0f, 40.0f));
|
|
return v;
|
|
};
|
|
std::vector<MergedReflection> aniso_set;
|
|
for (int h = -8; h <= 8; ++h)
|
|
for (int k = -8; k <= 8; ++k)
|
|
for (int l = 1; l <= 8; ++l)
|
|
aniso_set.push_back(Refl(h, k, l, static_cast<float>(cell.calculate_d({{h, k, l}})),
|
|
static_cast<float>(expected(h, k, l)), 5.0f));
|
|
std::vector<MergedReflection> iso_set; // the same reflections, all at the x probe's <I>
|
|
for (const auto &r : aniso_set)
|
|
iso_set.push_back(Refl(r.h, r.k, r.l, r.d, static_cast<float>(expected(5, 0, 0)), 5.0f));
|
|
aniso_set = with_probes(aniso_set);
|
|
iso_set = with_probes(iso_set);
|
|
|
|
FrenchWilsonOptions opts;
|
|
opts.num_shells = 1;
|
|
ApplyFrenchWilson(iso_set, SG(1), opts);
|
|
opts.anisotropy_hkl = q;
|
|
ApplyFrenchWilson(aniso_set, SG(1), opts);
|
|
const auto &along_x = aniso_set[aniso_set.size() - 2];
|
|
const auto &along_y = aniso_set.back();
|
|
CHECK(along_x.F == Catch::Approx(iso_set[iso_set.size() - 2].F).epsilon(0.01));
|
|
CHECK(along_x.F < 0.9f * along_y.F);
|
|
}
|