The pair rule drops the improbable (higher) member of a discordant pair. If the lower member lost its core to saturation or the mask, its profile estimate of what remained may be low, and a genuinely strong reflection would be replaced by the clipped value - the failure Aimless guards against. Integration now marks a reflection whose signal disk was not fully readable (Reflection::clipped, from the engines' existing full-disk test); the flag is carried through the partials to the fulls (OR over an event's partials, CPU and GPU combine alike), and the Wilson test never counts a clipped observation as a probable witness against a larger one. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
169 lines
7.9 KiB
C++
169 lines
7.9 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "WilsonOutliers.h"
|
|
|
|
#include <algorithm>
|
|
#include <cmath>
|
|
#include <numeric>
|
|
|
|
WilsonOutlierResult WilsonOutliers(const std::vector<WilsonObservation> &obs, double alpha) {
|
|
WilsonOutlierResult out;
|
|
out.rejected.assign(obs.size(), 0);
|
|
out.e2.assign(obs.size(), NAN);
|
|
|
|
std::vector<size_t> idx;
|
|
for (size_t i = 0; i < obs.size(); ++i) {
|
|
const auto &o = obs[i];
|
|
if (o.unit >= 0 && o.d > 0.0f && o.d < WILSON_OUTLIER_D_MAX && std::isfinite(o.I)
|
|
&& o.sigma > 0.0f && std::isfinite(o.sigma))
|
|
idx.push_back(i);
|
|
}
|
|
if (idx.empty())
|
|
return out;
|
|
|
|
// alpha / N per observation, half of it to each tail. Wilson's acentric law is P(E^2 > t) = exp(-t)
|
|
// and the centric one P(E^2 > t) = erfc(sqrt(t/2)) <= exp(-t/2), so t and 2t; the Gaussian tail is
|
|
// P(> z sigma) <= exp(-z^2/2) / 2. Both bounds are the conservative side of the exact quantile.
|
|
// Following Wilson (1949) Acta Cryst. 2, 318-321; the significance guard as in Aimless, Evans (2006)
|
|
// Acta Cryst. D62, 72-82.
|
|
const double p = alpha / (2.0 * static_cast<double>(idx.size()));
|
|
const double t = std::log(1.0 / p);
|
|
const double z = std::sqrt(2.0 * std::log(1.0 / (2.0 * p)));
|
|
|
|
// <I/epsilon> in shells of equal observation count, low to high resolution. Two thousand observations
|
|
// put the shell mean at 2% (an exponential's sd equals its mean); narrow shells keep the fall-off
|
|
// across one from reading as spread.
|
|
constexpr size_t OBS_PER_SHELL = 2000;
|
|
std::sort(idx.begin(), idx.end(), [&](size_t a, size_t b) {
|
|
return obs[a].d != obs[b].d ? obs[a].d > obs[b].d : a < b;
|
|
});
|
|
const size_t n_shells = std::max<size_t>(1, idx.size() / OBS_PER_SHELL);
|
|
const auto shell_of = [&](size_t j) { return j * n_shells / idx.size(); };
|
|
const auto lower_e2 = [&](const WilsonObservation &o, double mu) {
|
|
return (o.I - z * o.sigma) / (o.epsilon * mu);
|
|
};
|
|
const auto wilson_bound = [&](const WilsonObservation &o) { return o.centric ? 2.0 * t : t; };
|
|
// A shell is judged only where its mean is itself established at the same significance: with no
|
|
// measured <I>, nothing in the shell can be called improbable against it.
|
|
std::vector<double> mu(n_shells, 0.0);
|
|
std::vector<uint8_t> shell_valid(n_shells, 0);
|
|
{
|
|
std::vector<double> sum(n_shells, 0.0);
|
|
std::vector<size_t> cnt(n_shells, 0);
|
|
for (size_t j = 0; j < idx.size(); ++j) {
|
|
sum[shell_of(j)] += obs[idx[j]].I / obs[idx[j]].epsilon;
|
|
++cnt[shell_of(j)];
|
|
}
|
|
// Once more without the observations the plain Wilson bound already calls improbable: a single
|
|
// artefact thousands of times <I> would otherwise raise the mean of its own shell several-fold
|
|
// and hide the smaller artefacts sharing it.
|
|
std::vector<double> sum_kept(n_shells, 0.0), sum2_kept(n_shells, 0.0);
|
|
std::vector<size_t> cnt_kept(n_shells, 0);
|
|
for (size_t j = 0; j < idx.size(); ++j) {
|
|
const auto &o = obs[idx[j]];
|
|
const size_t s = shell_of(j);
|
|
if (sum[s] > 0.0 && lower_e2(o, sum[s] / cnt[s]) > wilson_bound(o))
|
|
continue;
|
|
const double x = o.I / o.epsilon;
|
|
sum_kept[s] += x; sum2_kept[s] += x * x; ++cnt_kept[s];
|
|
}
|
|
for (size_t s = 0; s < n_shells; ++s) {
|
|
if (cnt_kept[s] < 2) continue;
|
|
const double n = static_cast<double>(cnt_kept[s]);
|
|
mu[s] = sum_kept[s] / n;
|
|
const double var = std::max(0.0, (sum2_kept[s] - n * mu[s] * mu[s]) / (n - 1.0));
|
|
shell_valid[s] = mu[s] > 0.0 && mu[s] > z * std::sqrt(var / n);
|
|
}
|
|
}
|
|
|
|
// The tail's scale, peaks over threshold: above any threshold an exponential's excess is the same
|
|
// exponential, so the median excess of the acentric observations over a common threshold u is
|
|
// k ln 2. u is where Wilson leaves a hundred acentric observations above it. Only shells whose
|
|
// typical observation is significant at u take part: where sigma is comparable to u <I>, what
|
|
// exceeds u is the noise tail, and it would read as a heavy intensity tail.
|
|
size_t n_acentric = 0;
|
|
for (size_t j = 0; j < idx.size(); ++j)
|
|
if (!obs[idx[j]].centric && shell_valid[shell_of(j)]) ++n_acentric;
|
|
const double u = std::max(0.0, std::log(static_cast<double>(n_acentric) / 100.0));
|
|
std::vector<uint8_t> shell_measures_tail(n_shells, 0);
|
|
{
|
|
std::vector<std::vector<double>> noise(n_shells); // sigma / (epsilon <I/epsilon>)
|
|
for (size_t j = 0; j < idx.size(); ++j) {
|
|
const auto &o = obs[idx[j]];
|
|
if (shell_valid[shell_of(j)]) noise[shell_of(j)].push_back(o.sigma / (o.epsilon * mu[shell_of(j)]));
|
|
}
|
|
for (size_t s = 0; s < n_shells; ++s) {
|
|
auto &v = noise[s];
|
|
if (v.empty()) continue;
|
|
const size_t mid = v.size() / 2;
|
|
std::nth_element(v.begin(), v.begin() + mid, v.end());
|
|
shell_measures_tail[s] = z * v[mid] <= u;
|
|
}
|
|
}
|
|
std::vector<double> excess;
|
|
for (size_t j = 0; j < idx.size(); ++j) {
|
|
const auto &o = obs[idx[j]];
|
|
if (o.centric || !shell_measures_tail[shell_of(j)]) continue;
|
|
const double e2 = o.I / (o.epsilon * mu[shell_of(j)]);
|
|
if (e2 > u) excess.push_back(e2 - u);
|
|
}
|
|
if (!excess.empty()) {
|
|
const size_t mid = excess.size() / 2;
|
|
std::nth_element(excess.begin(), excess.begin() + mid, excess.end());
|
|
out.tail_scale = std::max(1.0, excess[mid] / std::log(2.0));
|
|
}
|
|
out.bound = out.tail_scale * t;
|
|
|
|
// Which observations are improbable, and the observations of each reflection (in valid shells).
|
|
std::vector<uint8_t> improbable(obs.size(), 0);
|
|
std::vector<size_t> tested;
|
|
for (size_t j = 0; j < idx.size(); ++j) {
|
|
const size_t i = idx[j];
|
|
if (!shell_valid[shell_of(j)]) continue;
|
|
const auto &o = obs[i];
|
|
const double m = mu[shell_of(j)];
|
|
out.e2[i] = static_cast<float>(o.I / (o.epsilon * m));
|
|
improbable[i] = lower_e2(o, m) > out.tail_scale * wilson_bound(o);
|
|
tested.push_back(i);
|
|
}
|
|
out.n_tested = tested.size();
|
|
std::sort(tested.begin(), tested.end(), [&](size_t a, size_t b) {
|
|
return obs[a].unit != obs[b].unit ? obs[a].unit < obs[b].unit : a < b;
|
|
});
|
|
|
|
// An improbable observation measured once is rejected. One with company is rejected when most of
|
|
// its reflection's other observations are probable, unclipped witnesses and it disagrees with their
|
|
// mean beyond the errors - for a pair, the other member. Where most are large, the reflection is
|
|
// (Aimless's "keep if most observations are large").
|
|
for (size_t lo = 0; lo < tested.size();) {
|
|
size_t hi = lo;
|
|
while (hi < tested.size() && obs[tested[hi]].unit == obs[tested[lo]].unit) ++hi;
|
|
for (size_t q = lo; q < hi; ++q) {
|
|
const size_t i = tested[q];
|
|
if (!improbable[i]) continue;
|
|
size_t n_probable = 0;
|
|
double sw = 0.0, swI = 0.0;
|
|
for (size_t r = lo; r < hi; ++r) {
|
|
const auto &o = obs[tested[r]];
|
|
if (r == q || improbable[tested[r]] || o.clipped) continue;
|
|
const double w = 1.0 / (static_cast<double>(o.sigma) * o.sigma);
|
|
sw += w; swI += w * o.I;
|
|
++n_probable;
|
|
}
|
|
const size_t n_others = hi - lo - 1;
|
|
bool reject = n_others == 0;
|
|
if (2 * n_probable > n_others) {
|
|
const auto &o = obs[i];
|
|
reject = std::fabs(o.I - swI / sw) > z * std::sqrt(static_cast<double>(o.sigma) * o.sigma + 1.0 / sw);
|
|
}
|
|
if (reject) {
|
|
out.rejected[i] = 1;
|
|
++out.n_rejected;
|
|
}
|
|
}
|
|
lo = hi;
|
|
}
|
|
return out;
|
|
}
|