// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #include "WilsonOutliers.h" #include "../../common/ParallelFor.h" #include #include #include WilsonOutlierResult WilsonOutliers(const std::vector &obs, double alpha, size_t nthreads) { WilsonOutlierResult out; out.rejected.assign(obs.size(), 0); out.e2.assign(obs.size(), NAN); std::vector idx; for (size_t i = 0; i < obs.size(); ++i) { const auto &o = obs[i]; if (o.unit >= 0 && o.d > 0.0f && o.d < WILSON_OUTLIER_D_MAX && std::isfinite(o.I) && o.sigma > 0.0f && std::isfinite(o.sigma)) idx.push_back(i); } if (idx.empty()) return out; // alpha / N per observation, half of it to each tail. Wilson's acentric law is P(E^2 > t) = exp(-t) // and the centric one P(E^2 > t) = erfc(sqrt(t/2)) <= exp(-t/2), so t and 2t; the Gaussian tail is // P(> z sigma) <= exp(-z^2/2) / 2. Both bounds are the conservative side of the exact quantile. // Following Wilson (1949) Acta Cryst. 2, 318-321; the significance guard as in Aimless, Evans (2006) // Acta Cryst. D62, 72-82. const double p = alpha / (2.0 * static_cast(idx.size())); const double t = std::log(1.0 / p); const double z = std::sqrt(2.0 * std::log(1.0 / (2.0 * p))); // in shells of equal observation count, low to high resolution. Two thousand observations // put the shell mean at 2% (an exponential's sd equals its mean); narrow shells keep the fall-off // across one from reading as spread. constexpr size_t OBS_PER_SHELL = 2000; ParallelSort(idx.begin(), idx.end(), nthreads, [&](size_t a, size_t b) { return obs[a].d != obs[b].d ? obs[a].d > obs[b].d : a < b; }); const size_t n_shells = std::max(1, idx.size() / OBS_PER_SHELL); // The observations in that order, so the walks below read them one after another instead of each // through idx, and where each shell starts: shell s is the j with j * n_shells / N == s, which begins // at the first j with j * n_shells >= s * N. Each shell is then one task, and every sum below is still // taken over its shell in the order of j, so the split changes no rounding. std::vector so(idx.size()); ParallelChunks(static_cast(idx.size()), nthreads, [&](int lo, int hi) { for (int j = lo; j < hi; ++j) so[j] = obs[idx[j]]; }); std::vector shell_start(n_shells + 1); for (size_t sh = 0; sh <= n_shells; ++sh) shell_start[sh] = (sh * idx.size() + n_shells - 1) / n_shells; const int n_shell_tasks = static_cast(n_shells); const auto lower_e2 = [&](const WilsonObservation &o, double mu) { return (o.I - z * o.sigma) / (o.epsilon * mu); }; const auto wilson_bound = [&](const WilsonObservation &o) { return o.centric ? 2.0 * t : t; }; // A shell is judged only where its mean is itself established at the same significance: with no // measured , nothing in the shell can be called improbable against it. std::vector mu(n_shells, 0.0); std::vector shell_valid(n_shells, 0); ParallelFor(n_shell_tasks, nthreads, [&](int sh) { double sum = 0.0; size_t cnt = 0; for (size_t j = shell_start[sh]; j < shell_start[sh + 1]; ++j) { sum += so[j].I / so[j].epsilon; ++cnt; } // Once more without the observations the plain Wilson bound already calls improbable: a single // artefact thousands of times would otherwise raise the mean of its own shell several-fold // and hide the smaller artefacts sharing it. double sum_kept = 0.0, sum2_kept = 0.0; size_t cnt_kept = 0; for (size_t j = shell_start[sh]; j < shell_start[sh + 1]; ++j) { const auto &o = so[j]; if (sum > 0.0 && lower_e2(o, sum / cnt) > wilson_bound(o)) continue; const double x = o.I / o.epsilon; sum_kept += x; sum2_kept += x * x; ++cnt_kept; } if (cnt_kept < 2) return; const double n = static_cast(cnt_kept); mu[sh] = sum_kept / n; const double var = std::max(0.0, (sum2_kept - n * mu[sh] * mu[sh]) / (n - 1.0)); shell_valid[sh] = mu[sh] > 0.0 && mu[sh] > z * std::sqrt(var / n); }); // The tail's scale, peaks over threshold: above any threshold an exponential's excess is the same // exponential, so the median excess of the acentric observations over a common threshold u is // k ln 2. u is where Wilson leaves a hundred acentric observations above it. Only shells whose // typical observation is significant at u take part: where sigma is comparable to u , what // exceeds u is the noise tail, and it would read as a heavy intensity tail. size_t n_acentric = 0; for (size_t sh = 0; sh < n_shells; ++sh) if (shell_valid[sh]) for (size_t j = shell_start[sh]; j < shell_start[sh + 1]; ++j) if (!so[j].centric) ++n_acentric; const double u = std::max(0.0, std::log(static_cast(n_acentric) / 100.0)); std::vector shell_measures_tail(n_shells, 0); ParallelFor(n_shell_tasks, nthreads, [&](int sh) { if (!shell_valid[sh]) return; std::vector v; // sigma / (epsilon ) v.reserve(shell_start[sh + 1] - shell_start[sh]); for (size_t j = shell_start[sh]; j < shell_start[sh + 1]; ++j) v.push_back(so[j].sigma / (so[j].epsilon * mu[sh])); if (v.empty()) return; const size_t mid = v.size() / 2; std::nth_element(v.begin(), v.begin() + mid, v.end()); shell_measures_tail[sh] = z * v[mid] <= u; }); std::vector excess; for (size_t sh = 0; sh < n_shells; ++sh) { if (!shell_measures_tail[sh]) continue; for (size_t j = shell_start[sh]; j < shell_start[sh + 1]; ++j) { const auto &o = so[j]; if (o.centric) continue; const double e2 = o.I / (o.epsilon * mu[sh]); if (e2 > u) excess.push_back(e2 - u); } } if (!excess.empty()) { const size_t mid = excess.size() / 2; std::nth_element(excess.begin(), excess.begin() + mid, excess.end()); out.tail_scale = std::max(1.0, excess[mid] / std::log(2.0)); } out.bound = out.tail_scale * t; // Which observations are improbable, and the observations of each reflection (in valid shells). // Each observation is written once, at its own index, so the shells can do it side by side; the // reflections are then formed by a sort on a total order, whatever order tested was filled in. std::vector improbable(obs.size(), 0); ParallelFor(n_shell_tasks, nthreads, [&](int sh) { if (!shell_valid[sh]) return; for (size_t j = shell_start[sh]; j < shell_start[sh + 1]; ++j) { const auto &o = so[j]; out.e2[idx[j]] = static_cast(o.I / (o.epsilon * mu[sh])); improbable[idx[j]] = lower_e2(o, mu[sh]) > out.tail_scale * wilson_bound(o); } }); std::vector tested; for (size_t sh = 0; sh < n_shells; ++sh) if (shell_valid[sh]) tested.insert(tested.end(), idx.begin() + shell_start[sh], idx.begin() + shell_start[sh + 1]); out.n_tested = tested.size(); ParallelSort(tested.begin(), tested.end(), nthreads, [&](size_t a, size_t b) { return obs[a].unit != obs[b].unit ? obs[a].unit < obs[b].unit : a < b; }); // An improbable observation measured once is rejected. One with company is rejected when most of // its reflection's other observations are probable, unclipped witnesses and it disagrees with their // mean beyond the errors - for a pair, the other member. Where most are large, the reflection is // (Aimless's "keep if most observations are large"). for (size_t lo = 0; lo < tested.size();) { size_t hi = lo; while (hi < tested.size() && obs[tested[hi]].unit == obs[tested[lo]].unit) ++hi; for (size_t q = lo; q < hi; ++q) { const size_t i = tested[q]; if (!improbable[i]) continue; size_t n_probable = 0; double sw = 0.0, swI = 0.0; for (size_t r = lo; r < hi; ++r) { const auto &o = obs[tested[r]]; if (r == q || improbable[tested[r]] || o.clipped) continue; const double w = 1.0 / (static_cast(o.sigma) * o.sigma); sw += w; swI += w * o.I; ++n_probable; } const size_t n_others = hi - lo - 1; bool reject = n_others == 0; if (2 * n_probable > n_others) { const auto &o = obs[i]; reject = std::fabs(o.I - swI / sw) > z * std::sqrt(static_cast(o.sigma) * o.sigma + 1.0 / sw); } if (reject) { out.rejected[i] = 1; ++out.n_rejected; } } lo = hi; } return out; }