Symmetry operators were scored by Pearson correlation on raw merged intensities. Both members of a symmetry pair sit at the same |s|, so the resolution fall-off is variance the two arms share exactly, and it inflates the correlation of true and false operators alike. The clearest demonstration is the test this commit adds: a synthetic data set with no symmetry at all - a radial fall-off times an independent per- reflection factor - is assigned point group 432 by the shipped code, with all 23 rotations confirmed at 0.632 to 0.656. The existing suite passes identically before and after, because nothing covered this. The new case fails 64 of its 100 assertions on the old scoring and passes on the new. On real data the same effect had the gate leaking: on one cubic crystal three pseudo-symmetric operators scored 0.506 to 0.517, above the 0.5 bound, so they were confirmed and 432 had to be refused further downstream by the twin-law and systematic-b guards. Scored on E-squared they read 0.283 to 0.298 and exactly the eleven genuine rotations of 23 are confirmed. The normalisation has to be over the reflections the correlation actually pairs. Reusing the existing normalised array is worse than doing nothing: it is normalised over the set the absence tests use, whose surviving fraction is itself resolution-dependent, and the coupling to the resolution cut rises from 0.086 to 0.262 against a raw baseline of 0.086. Normalised over the paired set it falls to 0.023. The twin-law H statistic keeps its own vectors on raw intensities. It shares the pair arrays with the correlation, and normalising in place moves it by up to 12% against a bound whose window is 5.5% wide. Verified rather than assumed: two instrumented binaries print the same H to twelve significant figures while the correlation differs. min_operator_cc goes 0.5 to 0.30. Normalised correlations run lower, and the observed window on rugnux's own search merges is 0.298 to 0.351; 0.35 is too high, because one crystal's weakest genuine operator reads 0.351. The headroom between a crystal's weakest true operator and its own measured false -operator floor widens on 14 of 14 crystals, median 0.430 to 0.619, and the worst operational margin goes from 0.031 to 0.051. Battery, twice, against a baseline reproducible to zero: the space group is identical on all 38 crystals, per-shell merging is 0 better and 0 worse across all 380 shells, and the merged mmCIF is byte-identical on 38 of 38 - on the 12 crystals whose integration radius now adapts as well as the 25 that do not. The arm is live rather than inert: all 313 operator correlations move while every pair count and every H value stays bit-identical, and on one cubic crystal three operators that raw intensities confirmed are rejected, with the space group unchanged. Following Padilla and Yeates (2003) Acta Cryst. D59, 1124-1130 for why a resolution-normalised statistic is the right one for a symmetry test. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CHMmeM1d489zvNFT7ZMN2P
370 lines
16 KiB
C++
370 lines
16 KiB
C++
#include <catch2/catch_all.hpp>
|
|
|
|
#include "../image_analysis/scale_merge/SearchSpaceGroup.h"
|
|
#include "gemmi/symmetry.hpp"
|
|
|
|
#include <algorithm>
|
|
#include <cmath>
|
|
#include <cstdint>
|
|
#include <string>
|
|
#include <tuple>
|
|
#include <unordered_set>
|
|
#include <vector>
|
|
|
|
namespace {
|
|
struct HKL {
|
|
int h = 0;
|
|
int k = 0;
|
|
int l = 0;
|
|
|
|
bool operator==(const HKL& o) const noexcept {
|
|
return h == o.h && k == o.k && l == o.l;
|
|
}
|
|
};
|
|
|
|
struct HKLHash {
|
|
size_t operator()(const HKL& x) const noexcept {
|
|
auto mix = [](uint64_t v) {
|
|
v ^= v >> 33;
|
|
v *= 0xff51afd7ed558ccdULL;
|
|
v ^= v >> 33;
|
|
v *= 0xc4ceb9fe1a85ec53ULL;
|
|
v ^= v >> 33;
|
|
return v;
|
|
};
|
|
return static_cast<size_t>(
|
|
mix(static_cast<uint64_t>(x.h)) ^
|
|
(mix(static_cast<uint64_t>(x.k)) << 1) ^
|
|
(mix(static_cast<uint64_t>(x.l)) << 2));
|
|
}
|
|
};
|
|
|
|
double CalcSyntheticD(int h, int k, int l) {
|
|
const double q2 = static_cast<double>(h * h + k * k + l * l);
|
|
return 40.0 / std::sqrt(q2 + 1.0);
|
|
}
|
|
|
|
double SyntheticIntensityFromAsu(const gemmi::Op::Miller& asu) {
|
|
uint64_t x = static_cast<uint64_t>((asu[0] + 31) * 73856093u) ^
|
|
static_cast<uint64_t>((asu[1] + 37) * 19349663u) ^
|
|
static_cast<uint64_t>((asu[2] + 41) * 83492791u);
|
|
x ^= x >> 13;
|
|
x *= 0x9e3779b97f4a7c15ULL;
|
|
x ^= x >> 17;
|
|
return 100.0 + static_cast<double>(x % 500);
|
|
}
|
|
|
|
std::vector<MergedReflection> GenerateMergedReflectionsForSpaceGroup(
|
|
const gemmi::SpaceGroup& sg,
|
|
int hmax = 8) {
|
|
|
|
std::vector<MergedReflection> merged;
|
|
std::unordered_set<HKL, HKLHash> added;
|
|
|
|
const gemmi::GroupOps gops = sg.operations();
|
|
const gemmi::ReciprocalAsu rasu(&sg);
|
|
|
|
for (int h = -hmax; h <= hmax; ++h) {
|
|
for (int k = -hmax; k <= hmax; ++k) {
|
|
for (int l = -hmax; l <= hmax; ++l) {
|
|
if (h == 0 && k == 0 && l == 0)
|
|
continue;
|
|
|
|
bool absent = false;
|
|
gemmi::Op::Miller hkl{{h, k, l}};
|
|
if (gops.is_systematically_absent(hkl))
|
|
absent = true;
|
|
|
|
const auto [asu, sign_plus] = rasu.to_asu_sign(hkl, gops);
|
|
if (!sign_plus)
|
|
continue;
|
|
|
|
const HKL key{h, k, l};
|
|
if (added.find(key) != added.end())
|
|
continue;
|
|
added.insert(key);
|
|
|
|
merged.push_back(MergedReflection{
|
|
.h = h,
|
|
.k = k,
|
|
.l = l,
|
|
.I = absent ? 0.0 : SyntheticIntensityFromAsu(asu),
|
|
.sigma = 1.0,
|
|
.d = CalcSyntheticD(h, k, l)
|
|
});
|
|
}
|
|
}
|
|
}
|
|
|
|
return merged;
|
|
}
|
|
}
|
|
|
|
TEST_CASE("SearchSpaceGroup detects synthetic space groups") {
|
|
struct Case {
|
|
std::string input_name;
|
|
std::string expected_short_name;
|
|
};
|
|
|
|
const std::vector<Case> cases = {
|
|
{"P 1", "P1"},
|
|
{"P 1 2 1", "P2"},
|
|
{"P 3 2 1", "P321"},
|
|
{"P 4 2 2", "P422"},
|
|
{"P 4 3 2", "P432"},
|
|
{"P 43 21 2", "P43212"},
|
|
{"P 6 2 2", "P622"},
|
|
{"C 1 2 1", "C2"},
|
|
{"C 2 2 2", "C222"},
|
|
{"I 4 3 2", "I432"},
|
|
{"I 21 21 21", "I212121"},
|
|
{"I 2 1 3", "I213"},
|
|
};
|
|
|
|
for (const auto& tc : cases) {
|
|
DYNAMIC_SECTION(tc.expected_short_name) {
|
|
const gemmi::SpaceGroup& sg = gemmi::get_spacegroup_by_name(tc.input_name);
|
|
const auto merged = GenerateMergedReflectionsForSpaceGroup(sg);
|
|
|
|
SearchSpaceGroupOptions opt;
|
|
opt.merge_friedel = true;
|
|
|
|
const auto result = SearchSpaceGroup(merged, opt);
|
|
|
|
// Several inputs cannot be told apart from intensities alone: enantiomorphic partners
|
|
// (P4_3 vs P4_1) and origin-ambiguous pairs (I2_12_12_1 vs I222, I2_13 vs I2_3) share
|
|
// the same systematic absences. The search reports those as alternatives, so the
|
|
// expected group must appear among the best group and its alternatives.
|
|
std::vector<std::string> accepted;
|
|
if (result.best_space_group.has_value())
|
|
accepted.push_back(result.best_space_group->short_name());
|
|
for (const auto& alt : result.alternatives)
|
|
accepted.push_back(alt.short_name());
|
|
|
|
INFO(SearchSpaceGroupResultToText(result));
|
|
REQUIRE(result.best_space_group.has_value());
|
|
CHECK(std::find(accepted.begin(), accepted.end(), tc.expected_short_name) != accepted.end());
|
|
}
|
|
}
|
|
}
|
|
|
|
// Regression: a real screw axis whose systematically-absent reflections carry a genuinely weak
|
|
// intensity but an UNDER-estimated sigma (so their I/sigma clears the "present" cut) must still be
|
|
// found. Reproduces a monoclinic 2_1 miss on weakly-diffracting monoclinic data, where the merged sigmas on
|
|
// the 0k0-odd reflections were ~2x too small and faked screw-axis violations. The E^2 intensity gate
|
|
// (present_e_squared) is what keeps those reflections classified absent.
|
|
TEST_CASE("SearchSpaceGroup finds a screw axis despite under-estimated sigmas on absent reflections") {
|
|
const gemmi::SpaceGroup& sg = gemmi::get_spacegroup_by_name("P 1 21 1");
|
|
auto merged = GenerateMergedReflectionsForSpaceGroup(sg, 18);
|
|
|
|
// Every systematically-absent (0k0, k odd) reflection: small-but-nonzero intensity (~2% of a
|
|
// normal reflection) with a far-too-small sigma, so I/sigma ~ 27 fakes a "present" reflection.
|
|
const gemmi::GroupOps gops = sg.operations();
|
|
int absent_count = 0;
|
|
for (auto& r : merged) {
|
|
const gemmi::Op::Miller hkl{{r.h, r.k, r.l}};
|
|
if (gops.is_systematically_absent(hkl)) {
|
|
r.I = 8.0f;
|
|
r.sigma = 0.3f;
|
|
++absent_count;
|
|
}
|
|
}
|
|
REQUIRE(absent_count >= 8); // enough predicted-absent reflections to be trusted
|
|
|
|
SearchSpaceGroupOptions opt;
|
|
opt.merge_friedel = true;
|
|
|
|
SECTION("intensity gate on (default): screw recovered") {
|
|
const auto result = SearchSpaceGroup(merged, opt);
|
|
INFO(SearchSpaceGroupResultToText(result));
|
|
REQUIRE(result.best_space_group.has_value());
|
|
CHECK(result.best_space_group->short_name() == "P21");
|
|
}
|
|
|
|
SECTION("intensity gate off (I/sigma only): the screw is missed") {
|
|
// Documents the failure the gate fixes: with I/sigma alone the too-small sigmas fake
|
|
// violations and the search falls back to the symmorphic group.
|
|
opt.present_e_squared = 0.0;
|
|
const auto result = SearchSpaceGroup(merged, opt);
|
|
INFO(SearchSpaceGroupResultToText(result));
|
|
REQUIRE(result.best_space_group.has_value());
|
|
CHECK(result.best_space_group->short_name() == "P2");
|
|
}
|
|
}
|
|
|
|
// Regression: the E^2 gate above compares a reflection to the mean of its RESOLUTION SHELL, which
|
|
// falls off with resolution, while a systematically-absent reflection keeps a small non-decaying
|
|
// residual (background / profile leakage). On a crystal whose axial rows are much stronger than an
|
|
// average reflection, that turns the high-resolution residuals into screw-axis violations and the
|
|
// screw is lost, although the reflections beside them in the same row are tens of times stronger.
|
|
// A tetragonal 42_12 case failed exactly this way (18 of 47 absent 00l over the cut, all beyond
|
|
// 3.7 A, at 1-2% of the l=4n reflections next to them). The threshold is therefore taken relative to
|
|
// the axial row the screw constrains, not to the shell.
|
|
TEST_CASE("SearchSpaceGroup finds a screw axis whose absent class is weak only within its own row") {
|
|
const gemmi::SpaceGroup& sg = gemmi::get_spacegroup_by_name("P 43 21 2");
|
|
auto merged = GenerateMergedReflectionsForSpaceGroup(sg, 12);
|
|
|
|
// Axial rows 40x stronger than a general reflection, and an absent class carrying ~2% of its own
|
|
// row - but half of a general reflection, so a threshold set against the shell calls every one of
|
|
// them a violation while a threshold set against the row calls none.
|
|
const gemmi::GroupOps gops = sg.operations();
|
|
int absent_on_axis = 0;
|
|
for (auto& r : merged) {
|
|
const gemmi::Op::Miller hkl{{r.h, r.k, r.l}};
|
|
if (gops.epsilon_factor_without_centering(hkl) <= 1)
|
|
continue;
|
|
if (gops.is_systematically_absent(hkl)) {
|
|
r.I = 300.0f;
|
|
r.sigma = 1.0f;
|
|
++absent_on_axis;
|
|
} else {
|
|
r.I *= 40.0f;
|
|
}
|
|
}
|
|
REQUIRE(absent_on_axis >= 8);
|
|
|
|
SearchSpaceGroupOptions opt;
|
|
opt.merge_friedel = true;
|
|
|
|
const auto result = SearchSpaceGroup(merged, opt);
|
|
INFO(SearchSpaceGroupResultToText(result));
|
|
REQUIRE(result.best_space_group.has_value());
|
|
// P4_1 2_1 2 and P4_3 2_1 2 are enantiomorphs and indistinguishable from intensities.
|
|
std::vector<std::string> accepted{result.best_space_group->short_name()};
|
|
for (const auto& alt : result.alternatives)
|
|
accepted.push_back(alt.short_name());
|
|
CHECK(std::find(accepted.begin(), accepted.end(), "P43212") != accepted.end());
|
|
}
|
|
// Regression: a screw's predicted-absent class is one row of reciprocal space, and that row is often
|
|
// the one a rotation sweep records least - it lies near the spindle, where the blind cusp maps onto
|
|
// itself and symmetry cannot fill it in. Counting the class therefore measures the geometry of the
|
|
// sweep, not the strength of the evidence, and a count gate refused a monoclinic crystal its 2_1 for
|
|
// having six 0k0-odd reflections rather than eight, every one of them measured at a thousandth of the
|
|
// row beside them. The class is judged by ScrewAbsenceEvidence instead, which reads the contrast
|
|
// against the row - so few-but-decisive is accepted and many-but-marginal is not.
|
|
TEST_CASE("SearchSpaceGroup weighs a screw's absences by evidence, not by how many were recorded") {
|
|
const gemmi::SpaceGroup& sg = gemmi::get_spacegroup_by_name("P 1 21 1");
|
|
const gemmi::GroupOps gops = sg.operations();
|
|
|
|
SearchSpaceGroupOptions opt;
|
|
opt.merge_friedel = true;
|
|
|
|
SECTION("five decisive absences, below min_absent_observed: the screw is still found") {
|
|
auto merged = GenerateMergedReflectionsForSpaceGroup(sg, 18);
|
|
// Keep five of the 0k0-odd reflections, at a thousandth of their row, and drop the rest - as a
|
|
// sweep along the 2-fold does, leaving too few to satisfy a count but plenty to decide.
|
|
int kept = 0;
|
|
std::erase_if(merged, [&](MergedReflection& r) {
|
|
if (!gops.is_systematically_absent(gemmi::Op::Miller{{r.h, r.k, r.l}}))
|
|
return false;
|
|
if (kept >= 5)
|
|
return true;
|
|
++kept;
|
|
r.I = 0.5;
|
|
return false;
|
|
});
|
|
REQUIRE(kept == 5);
|
|
REQUIRE(kept < opt.min_absent_observed);
|
|
|
|
const auto result = SearchSpaceGroup(merged, opt);
|
|
INFO(SearchSpaceGroupResultToText(result));
|
|
REQUIRE(result.best_space_group.has_value());
|
|
CHECK(result.best_space_group->short_name() == "P21");
|
|
}
|
|
|
|
SECTION("a uniformly weak axial row decides nothing, however many absences it holds") {
|
|
// The whole 0k0 row badly measured: the predicted-absent reflections are weak, but so is the
|
|
// rest of their row, so there is no contrast and no screw to claim. A violation count cannot
|
|
// see this - nothing on the row clears an absolute cut, so it reads zero violations and, with
|
|
// enough reflections to satisfy the count, would claim the 2_1 from no evidence at all.
|
|
auto merged = GenerateMergedReflectionsForSpaceGroup(sg, 18);
|
|
int absent_on_row = 0;
|
|
for (auto& r : merged) {
|
|
if (r.h != 0 || r.l != 0)
|
|
continue;
|
|
const bool absent = gops.is_systematically_absent(gemmi::Op::Miller{{r.h, r.k, r.l}});
|
|
r.I = absent ? 4.0 : 5.0;
|
|
absent_on_row += absent ? 1 : 0;
|
|
}
|
|
REQUIRE(absent_on_row >= opt.min_absent_observed);
|
|
|
|
const auto result = SearchSpaceGroup(merged, opt);
|
|
INFO(SearchSpaceGroupResultToText(result));
|
|
REQUIRE(result.best_space_group.has_value());
|
|
CHECK(result.best_space_group->short_name() == "P2");
|
|
}
|
|
}
|
|
|
|
// The operator correlation is on resolution-normalised E^2, not on raw I (see SearchSpaceGroup.cpp).
|
|
// Both members of a symmetry pair sit at the same |s|, so on raw intensities the resolution fall-off
|
|
// is variance shared perfectly between the two arms of every pair and reads as a correlation for ANY
|
|
// pairing at all. These two cases pin that down from both sides.
|
|
TEST_CASE("SearchSpaceGroup operator correlation reads symmetry, not the resolution fall-off",
|
|
"[SearchSpaceGroup]") {
|
|
// Intensities that are a smooth function of resolution times an INDEPENDENT per-reflection
|
|
// factor: a Wilson-like fall-off with no symmetry in it whatsoever.
|
|
auto radial_only = [](int hmax) {
|
|
std::vector<MergedReflection> merged;
|
|
for (int h = -hmax; h <= hmax; ++h)
|
|
for (int k = -hmax; k <= hmax; ++k)
|
|
for (int l = -hmax; l <= hmax; ++l) {
|
|
if ((h == 0 && k == 0 && l == 0) || std::make_tuple(-h, -k, -l) < std::make_tuple(h, k, l))
|
|
continue;
|
|
const double d = CalcSyntheticD(h, k, l);
|
|
const double falloff = std::exp(-30.0 / (d * d));
|
|
// Deterministic, independent of any symmetry mate: reuse the hash on the raw index.
|
|
const double jitter = SyntheticIntensityFromAsu(gemmi::Op::Miller{{h, k, l}}) / 350.0;
|
|
const double I = 1.0e5 * falloff * jitter;
|
|
merged.push_back(MergedReflection{
|
|
.h = h, .k = k, .l = l, .I = I, .sigma = I / 20.0, .d = d});
|
|
}
|
|
return merged;
|
|
};
|
|
|
|
SearchSpaceGroupOptions opt;
|
|
opt.merge_friedel = true;
|
|
|
|
SECTION("a fall-off with no symmetry in it confirms no operator") {
|
|
const auto result = SearchSpaceGroup(radial_only(8), opt);
|
|
INFO(SearchSpaceGroupResultToText(result));
|
|
REQUIRE(result.operator_scores.size() > 1);
|
|
for (const auto& s : result.operator_scores) {
|
|
INFO("operator " << s.op_triplet_hkl);
|
|
CHECK(s.n_pairs >= opt.min_pairs_per_operator);
|
|
CHECK(s.cc < opt.min_operator_cc);
|
|
CHECK_FALSE(s.present);
|
|
}
|
|
CHECK(result.point_group_hm == "1");
|
|
}
|
|
|
|
SECTION("a real operator under the same fall-off is confirmed, and does not move with the cut") {
|
|
// Same fall-off, but the intensities now carry a genuine monoclinic 2-fold.
|
|
const gemmi::SpaceGroup& sg = gemmi::get_spacegroup_by_name("P 1 2 1");
|
|
const gemmi::ReciprocalAsu rasu(&sg);
|
|
const gemmi::GroupOps gops = sg.operations();
|
|
auto merged = radial_only(8);
|
|
for (auto& r : merged) {
|
|
const auto [asu, plus] = rasu.to_asu_sign(gemmi::Op::Miller{{r.h, r.k, r.l}}, gops);
|
|
const double falloff = std::exp(-30.0 / (r.d * r.d));
|
|
r.I = 1.0e5 * falloff * SyntheticIntensityFromAsu(asu) / 350.0;
|
|
r.sigma = r.I / 20.0;
|
|
}
|
|
auto two_fold_cc = [&](double d_min) {
|
|
SearchSpaceGroupOptions o = opt;
|
|
o.d_min_limit_A = d_min;
|
|
const auto result = SearchSpaceGroup(merged, o);
|
|
INFO(SearchSpaceGroupResultToText(result));
|
|
REQUIRE(result.point_group_hm == "2");
|
|
double cc = -2.0;
|
|
for (const auto& s : result.operator_scores)
|
|
if (s.present)
|
|
cc = s.cc;
|
|
REQUIRE(cc > opt.min_operator_cc);
|
|
return cc;
|
|
};
|
|
// The whole point of normalising: how much of the fall-off is inside the merge no longer
|
|
// moves the operator's score, so the search resolution cut cannot decide the symmetry.
|
|
CHECK(std::fabs(two_fold_cc(0.0) - two_fold_cc(6.0)) < 0.05);
|
|
}
|
|
}
|