// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #include "SearchSpaceGroup.h" #include "../../common/ParallelFor.h" #include #include #include #include #include #include #include #include #include #include #include #include #include namespace { // A merged reflection, folded onto the +/- Friedel-equivalent it represents, used as a // hash key to match symmetry-related reflections. struct HKLKey { int h = 0, k = 0, l = 0; bool operator==(const HKLKey& o) const noexcept { return h == o.h && k == o.k && l == o.l; } }; struct HKLKeyHash { size_t operator()(const HKLKey& key) const noexcept { auto mix = [](uint64_t x) { x ^= x >> 33; x *= 0xff51afd7ed558ccdULL; x ^= x >> 33; x *= 0xc4ceb9fe1a85ec53ULL; x ^= x >> 33; return x; }; return static_cast(mix(static_cast(key.h)) ^ (mix(static_cast(key.k)) << 1) ^ (mix(static_cast(key.l)) << 2)); } }; HKLKey Canonicalize(int h, int k, int l, bool merge_friedel) { if (merge_friedel && std::make_tuple(-h, -k, -l) < std::make_tuple(h, k, l)) return {-h, -k, -l}; return {h, k, l}; } double PearsonCC(const std::vector& x, const std::vector& y) { if (x.size() < 2) return std::numeric_limits::quiet_NaN(); double sx = 0, sy = 0, sxx = 0, syy = 0, sxy = 0; for (size_t i = 0; i < x.size(); ++i) { sx += x[i]; sy += y[i]; sxx += x[i] * x[i]; syy += y[i] * y[i]; sxy += x[i] * y[i]; } const double n = static_cast(x.size()); const double vx = sxx - sx * sx / n; const double vy = syy - sy * sy / n; if (vx <= 0 || vy <= 0) return std::numeric_limits::quiet_NaN(); return (sxy - sx * sy / n) / std::sqrt(vx * vy); } // A reflection is extinct from lattice centering alone (independent of any screw/glide) when a // centering translation makes its structure factor cancel. Mirrors the centering half of // gemmi::GroupOps::is_systematically_absent, so screw absences can be judged separately. bool CenteringAbsent(const gemmi::GroupOps& gops, const gemmi::Op::Miller& hkl) { for (size_t i = 1; i < gops.cen_ops.size(); ++i) { const auto& t = gops.cen_ops[i]; if ((t[0] * hkl[0] + t[1] * hkl[1] + t[2] * hkl[2]) % gemmi::Op::DEN != 0) return true; } return false; } // ---- Glide planes --------------------------------------------------------------------------- // // A glide is a mirror with a translation, so it extinguishes a whole ZONE of reciprocal space - // the plane of reflections the mirror maps to themselves - where a screw extinguishes one row. // The two are the same test on different sets, so they share ScrewZoneEvidence and differ only in // what the control class is: the rest of the ROW for a screw, the rest of the PLANE for a glide. // // NOT DETERMINABLE, AND NOT ATTEMPTED: the INVERSION CENTRE. Friedel's law makes the diffraction // pattern centrosymmetric whether or not the crystal is, so P2 and P2/m predict exactly the same // intensities and exactly the same absences and no measurement on this merge can separate them. // The enumeration below therefore refuses any non-Sohncke group whose absence signature a Sohncke // candidate already carries, which is precisely the set that differs by an inversion centre alone. // The one exception is not available here: with measurable anomalous signal Friedel's law breaks, // which is how the ENANTIOMORPH is decided elsewhere in this program - but that decides a hand, // not the presence of a centre, and it needs the unmerged Bijvoet pairs, not this merge. // The zone test itself is POINTLESS's, on a plane rather than a row (Evans, Acta Cryst. D67, // 282-292 (2011), App. A3) - the same credit the axial test above carries. // The distinct mirror directions of a group, as the rotation parts of its improper operators. // A glide and the centring partner sitting on the same plane (the c and the n of C 2/c) share a // rotation part and so are ONE zone; scoring them as two would count the same reflections twice. // EMPTY for every Sohncke group - which is what makes every line below inert on chiral data. std::vector MirrorDirections(const gemmi::GroupOps& gops) { std::vector out; for (const auto& op : gops.sym_ops) { // rot_type() is -2 exactly for a MIRROR (an improper operator of order two). Of the // improper operators only a mirror fixes a plane of reflections; an inversion or a -4 // fixes nothing but 000, so it would form an empty zone anyway. Filtered here rather // than left to fall out, so the intent is on the page. if (op.rot_type() != -2) continue; if (std::find(out.begin(), out.end(), op.rot) == out.end()) out.push_back(op.rot); } return out; } // Which mirror's zone this reflection lies in, or -1 for none. Reflections with two zero indices // are excluded: they are AXIAL, they are what the screw test is built on, and a c-glide normal to // b formally extinguishes 00l with l odd exactly as the 2_1 along c does. Leaving them in both // tests would score one reflection twice and let a glide take credit for a screw's row. int MirrorZoneOf(const std::vector& mirrors, const gemmi::Op::Miller& hkl) { if ((hkl[0] == 0 ? 1 : 0) + (hkl[1] == 0 ? 1 : 0) + (hkl[2] == 0 ? 1 : 0) >= 2) return -1; for (size_t m = 0; m < mirrors.size(); ++m) if (gemmi::Op{mirrors[m], {0, 0, 0}}.apply_to_hkl(hkl) == hkl) return static_cast(m); return -1; } // The zone a mirror fixes, named the way a crystallographer names it - "h0l" for the mirror // normal to b, "hhl" for a diagonal one. Built by asking the operator which of a handful of // representative reflections it fixes rather than by decoding the matrix, which keeps it correct // for settings nobody wrote a case for; falls back to the matrix's own H-M-ish description. std::string MirrorZoneLabel(const gemmi::Op::Rot& rot) { const gemmi::Op op{rot, {0, 0, 0}}; static const std::pair named[] = { {"0kl", {{0, 1, 2}}}, {"h0l", {{1, 0, 2}}}, {"hk0", {{1, 2, 0}}}, {"hhl", {{1, 1, 2}}}, {"hkh", {{1, 2, 1}}}, {"hkk", {{2, 1, 1}}}, }; for (const auto& [label, hkl] : named) if (op.apply_to_hkl(hkl) == hkl) return label; return "zone"; } // The reciprocal-space ROW a reflection lies on: its direction, reduced by the gcd and // sign-canonicalised, so 0,0,l and 0,0,-l are one row and h,0,0 is a different one. Screw // absences are judged against the other reflections of their own row (see below). using AxialRow = std::array; // constexpr, and std::gcd rather than std::gcd of std::abs - std::gcd is defined on the absolute // values already, so the abs calls were redundant, and dropping them lets the identities below be // checked at compile time on MSVC too (whose std::abs(int) is constexpr only from C++23). constexpr AxialRow RowOf(int h, int k, int l) { const int g = std::gcd(std::gcd(h, k), l); if (g > 0) { h /= g; k /= g; l /= g; } if (std::make_tuple(-h, -k, -l) < std::make_tuple(h, k, l)) return {-h, -k, -l}; return {h, k, l}; } // THE CANONICAL 00l ROW IS {0,0,-1}, NOT {0,0,1} - and likewise h00 and 0k0. Pinned here because // getting it wrong is silent: a row key written out by hand never equals the key a reflection of // that row looks up with, so every std::map lookup misses, nothing is reported, and the code just // quietly stops doing whatever it keyed on the row. Build row keys through RowOf, never by // writing the indices. static_assert(RowOf(0, 0, 1) == AxialRow{0, 0, -1}); static_assert(RowOf(1, 0, 0) == AxialRow{-1, 0, 0}); static_assert(RowOf(0, 1, 0) == AxialRow{0, -1, 0}); // Median of an unordered set (reordered in place); 0 for an empty set. double MedianOf(std::vector& v) { if (v.empty()) return 0.0; const size_t mid = v.size() / 2; std::nth_element(v.begin(), v.begin() + mid, v.end()); return v[mid]; } // How unlikely a predicted-absent class would be if the condition producing it did not exist, // in nats. Used for a SCREW against the rest of its own axial row, and for a CENTERING against // the present class. // // Scoring the absence against its own control class follows the POINTLESS zone test // (Evans, Acta Cryst D67, 282-292 (2011), App. A3); the Beta tail here is an analytic null in // place of its control transforms. // // Under "no condition" the absent class and its control are both Wilson-distributed with the SAME // mean, so with each absent intensity expressed in units of the control mean, the fraction // T = sum_u / (sum_u + n_control) follows Beta(n_absent, n_control) exactly. The control's own // strength cancels out of it - which is the property a count does not have, and the reason a // uniformly weak class decides nothing here instead of deciding "absent". Returns -log of that // Beta lower tail. // // Only the leading term of the regularized incomplete beta is kept. It is exact as T -> 0, which is // where a condition is claimed, and dropping the (1-T)^n_control factor only ever UNDER-states the // evidence, which is the safe direction for a test that has to clear a bound. // No measurement can place a merged intensity at exactly zero, so sum_u is floored at a // thousandth of the control mean per absent reflection. Without it, a zone whose absences all // merged non-positive gives sum_u = 0 exactly, T clamps to the epsilon below, and EACH absent // reflection is worth ~690 nats. That was harmless while the value only had to clear a bound of // 20, but it is now summed across zones and ranks the candidates, and it inverted the ranking // outright: a zone of 2 absences that were never measurable scored 1378 nats where a genuine zone // of 6 absences at 1% of its row scores 22, so the candidate claiming a screw on an UNMEASURED // row beat the one whose rows are actually dead, by 60x. // // The constant is deliberately an order of magnitude below the precision any real merge reaches - // a thousandth of the row mean needs I/sigma ~ 1000 against that row, where ISa tops out near 40 - // so it can only ever remove the singularity, never suppress evidence a measurement could have // produced. Measured over the realistic range it changes no genuine zone at all (22.0, 34.7 and // 65.4 nats to four figures) and takes the unmeasurable ones to 13 and 36. constexpr double MIN_U_PER_ABSENT_REFLECTION = 1e-3; // A pseudo-translation that imitates a screw axis must be HALF-INTEGER ALONG THE ROW. A screw // along c extinguishes 00l for l odd; a translation can only reproduce that alternation if its // component along c is 1/2, because cos(2*pi*l*u_c) alternates with l for no other value. The // other two components obey no such constraint - a real crystal's translation sits wherever its // molecules do - so they are refined on the merged intensities, each of the seven half-integer // vectors serving as a start with the row's own component pinned at 1/2. No Patterson, no FFT, // no RNG. // // Why the screw test needs this at all: a translation splits every reflection into an enhanced and a // suppressed class by the parity of 2u.h, and for a half-integer u on the screw's own axis those two // classes are exactly the zone's absent class and its control. The absence test then pays the // modulation twice, once in the numerator and once in the denominator, and reads a class that is // merely suppressed as extinct. // Following Read, Adams & McCoy (2013) Acta Cryst. D69, 176-183 // // The modulation is measured on GENERAL reflections, which no screw axis touches, and IN A CONE // AROUND THE ROW'S OWN DIRECTION. That last part is not a refinement, it is what makes the // correction work: the two copies are displaced anisotropically, so the modulation has a different // depth in every direction of reciprocal space, and one pooled number per shell is the wrong model // for a single axial row. Measured on an orthorhombic crystal with a pseudo-body-centring, the // modulation reads 0.03 along c* and 0.20 perpendicular to it in the same resolution shell. Using // the pooled value both UNDER-corrects the row that needs it (the deep direction) and OVER-corrects // the rows that do not, which is wrong in both directions at once; per-row it costs the crystal's // two genuine screws under 3 nats each while taking the false one from 64 nats to 15. // // The depth is read per resolution shell in the cone, as the mean E^2 of the reflections the // translation places near phase 1/2 - which is where the row's absent class sits, its on-row phase // being exactly pi - over the mean of those it places near phase 0, which are that class's // control. A PARITY SPLIT by h.t for the idealised half-integer t measures the same thing only // when the transverse components really are half-integer. Off that, a cone member's phase // scrambles with its transverse indices while the on-row phase stays exactly pi, the two parity // classes mix, and the split reads no modulation on a row that is suppressed eight-fold: measured // on a monoclinic crystal whose translation refines to (1/2, 0.46, 0.03), the parity ratio // against (1/2, 1/2, 0) reads 0.9-1.2 in every shell above the first while the phase windows read // 0.08-0.18, and the difference is a spurious screw on the row. Where the transverse components // ARE half-integer the two are the same measurement, so nothing moves on a crystal the parity // split already read correctly. // // Three constraints, all physical, keep the estimate from inventing a correction where none exists: // // - the ratio is clamped to [0,1]. A pseudo-translation can only SUPPRESS a class; a ratio above 1 // is noise, and admitting it would STRENGTHEN a screw claim, which no translation can justify. // So this can only ever weaken screw evidence, never manufacture it - refining the transverse // components keeps that direction, because a deeper depth only ever divides the absent class UP. // - it is non-decreasing with resolution (running maximum from low resolution outward): the two // copies decorrelate as resolution rises, so the modulation dies away. Without this, a cubic // metric alone produces a spurious correction - h^2+k^2+l^2 fixes the parity of h+k+l, so a // narrow high-resolution shell can hold one class almost alone and read a contrast of 0.16 or 5. // The refined vector does not lift the rule: an orthorhombic crystal carrying BOTH a genuine // order-two screw and a translation that dies by 2.4 A needs the correction to die with it, or // the screw is paid for at low resolution and again at high. // - a shell needs both phase windows properly populated inside the cone, or it measures nothing // and inherits the running value. // // And two bounds decide whether to correct at all: // // - the class must be at least twice suppressed somewhere. On a crystal with no pseudo-translation // every one of the seven ratios sits at 0.9-1.0; the crystals that carry one read 0.03-0.13 on // the affected row. The populations are nowhere near each other, so a wide bound costs nothing. // - a translation the data are invariant under in EVERY direction and at every resolution is a // LATTICE vector, not a pseudo-symmetry, and de-modulating by it divides a measured zero by a // measured zero. This is asked of the POOLED ratio - all directions, all shells - and not of the // cone, deliberately: in the cone the two populations OVERLAP, because a strong pseudo-symmetry // approaches extinction along its own direction exactly as a centring does. Measured, the cone // reads 0.025 for an undeclared C-centring and 0.029 for a genuine pseudo-body-centring - a // refusal read there refuses both or neither, and refusing neither costs a C2 crystal both of // its real screws (+20 and +24 nats to -62 and -14). Pooled, the same two read -0.015 and 0.60. // A lattice translation has nowhere to hide in the pooled number: it is extinct everywhere. // // LIMIT - ONLY ORDER-2 SCREWS ARE TREATED. The seven vectors are complete for a 2(1), and for // nothing else. A 3(1) extinguishes 00l for l not a multiple of 3, and a translation with a 1/3 // component along c modulates that row on the same period; a 4(1) and a 1/4 component likewise. // Nothing here looks for those, so on a higher-order screw this says NOTHING - the question is // UNANSWERED, not answered in the negative, and a reader must not read an untouched 3(1) or 4(1) // zone as one that was tested and cleared. // // A cell is required, because a cone in reciprocal space is a metric statement. Without one the // correction is simply not applied - the confounder cannot be separated from the row it sits on. struct RowModulation { std::array halves{}; // the nearest half-integer translation: {1,1,1} is (1/2,1/2,1/2) std::vector ratio; // per shell, mean E^2 at translation phase 1/2 over phase 0 double deepest = 1.0; // the ratio at the lowest resolution it could be measured at bool found = false; }; constexpr double PSEUDO_TRANSLATION_MAX_RATIO = 0.50; // must be twice suppressed along the row to be used constexpr double PSEUDO_TRANSLATION_MIN_POOLED = 0.20; // below this, pooled, the class is extinct: a lattice vector constexpr size_t PSEUDO_TRANSLATION_MIN_PER_CLASS = 30; // per shell, per class, inside the cone constexpr double PSEUDO_TRANSLATION_CONE_COS = 0.9; // ~26 degrees around the row's direction // cos of the angle between the scattering vectors of hkl and of the row, from the reciprocal metric. // gemmi gives the diagonal form h.G*.h as calculate_1_d2; the cross term follows by polarisation. double CosToRow(const gemmi::UnitCell& cell, int h, int k, int l, const std::array& row) { const double qh = cell.calculate_1_d2(gemmi::Op::Miller{{h, k, l}}); const double qr = cell.calculate_1_d2(gemmi::Op::Miller{{row[0], row[1], row[2]}}); if (qh <= 0.0 || qr <= 0.0) return 0.0; const double qs = cell.calculate_1_d2(gemmi::Op::Miller{{h + row[0], k + row[1], l + row[2]}}); return 0.5 * (qs - qh - qr) / std::sqrt(qh * qr); } RowModulation FindRowModulation(const gemmi::UnitCell& cell, const std::array& row, const std::vector& use, const std::vector& H, const std::vector& K, const std::vector& L, const std::vector& Esq, const std::vector& shell, int bins) { RowModulation best; if (bins <= 0) return best; // Only a translation with a 1/2 on this row's own axis modulates this row at all; the others // leave every reflection of it in the same class and cannot be the confounder. std::vector> candidates; for (int m = 1; m < 8; ++m) { const std::array t{m & 1, (m >> 1) & 1, (m >> 2) & 1}; if (t[0] * row[0] + t[1] * row[1] + t[2] * row[2] != 0) candidates.push_back(t); } // The general reflections lying near the row's direction, once for all seven. std::vector cone; for (size_t i = 0; i < use.size(); ++i) { if (!use[i] || shell[i] < 0) continue; if ((H[i] == 0 ? 1 : 0) + (K[i] == 0 ? 1 : 0) + (L[i] == 0 ? 1 : 0) >= 2) continue; if (std::fabs(CosToRow(cell, H[i], K[i], L[i], row)) > PSEUDO_TRANSLATION_CONE_COS) cone.push_back(i); } // WHERE the translation is is found by a least-squares fit of E^2 ~ A + B cos(2 pi h.u) over // the cone, shell by shell: the summed slope B peaks where the phases line up, which is at the // crystal's own translation, and scrambles away within a few thousandths. auto cos_fit = [&](const std::array& u) { std::vector n(bins, 0.0), sc(bins, 0.0), scc(bins, 0.0), se(bins, 0.0), sec(bins, 0.0); for (size_t i : cone) { const int b = shell[i]; const double c = std::cos(2.0 * gemmi::pi() * (H[i] * u[0] + K[i] * u[1] + L[i] * u[2])); const double e = std::max(0.0, Esq[i]); n[b] += 1.0; sc[b] += c; scc[b] += c * c; se[b] += e; sec[b] += e * c; } double objective = 0.0; for (int b = 0; b < bins; ++b) { const double det = n[b] * scc[b] - sc[b] * sc[b]; if (n[b] < 2.0 * PSEUDO_TRANSLATION_MIN_PER_CLASS || det <= 0.0) continue; objective += (n[b] * sec[b] - sc[b] * se[b]) / det; } return objective; }; // HOW DEEP it is is NOT read off that fit. The slope answers a different question from the one // the absence test asks, and answers it with a bias that runs the wrong way: for a real // modulation E^2 = A' - B' p (p = +/-1 by class) seen through a regressor c = p cos(2 pi u.k) // that is set slightly wrong, the least-squares slope is B' /, which is larger than // B' for ANY error in u - so a mis-set vector reports a DEEPER modulation than the truth, and // the refinement above, which maximises the summed slope, is pulled towards exactly that error. // Measured on an orthorhombic crystal whose translation really is the half-integer one, the // objective peaks not at the truth but symmetrically 0.012 to either side of it, where the // depth reads zero or negative against a true 0.03-0.13; a depth of zero then divides into the // absence test and takes a genuine screw's row with it. // // So the depth is read as the ratio of the two CLASS MEANS, at the two ends of the phase: the // reflections the translation places near phase 1/2, where the row's absent class sits, over // those it places near phase 0, which are that class's control. It is a mean over a population, // not a fitted coefficient, so no inverse variance can inflate it; at a half-integer u it is // the parity split exactly, which is what keeps a crystal whose translation IS half-integer // reading precisely what it read before this estimator existed. constexpr double PHASE_WINDOW_COS = 0.8; // |cos| above this: within ~1/16 of a cycle of an end auto window_depth = [&](const std::array& u, std::vector* per_shell) { std::vector lo(bins, 0.0), hi(bins, 0.0); std::vector nlo(bins, 0), nhi(bins, 0); for (size_t i : cone) { const int b = shell[i]; const double c = std::cos(2.0 * gemmi::pi() * (H[i] * u[0] + K[i] * u[1] + L[i] * u[2])); const double e = std::max(0.0, Esq[i]); if (c < -PHASE_WINDOW_COS) { lo[b] += e; ++nlo[b]; } else if (c > PHASE_WINDOW_COS) { hi[b] += e; ++nhi[b]; } } for (int b = 0; b < bins; ++b) { if (nlo[b] < PSEUDO_TRANSLATION_MIN_PER_CLASS || nhi[b] < PSEUDO_TRANSLATION_MIN_PER_CLASS || hi[b] <= 0.0) continue; (*per_shell)[b] = std::clamp((lo[b] / static_cast(nlo[b])) / (hi[b] / static_cast(nhi[b])), 0.0, 1.0); } }; for (const auto& t : candidates) { // The pooled ratio first: it is the only place a lattice translation can be told from a // pseudo-symmetry, because a lattice translation is extinct in every direction. double pooled0 = 0.0, pooled1 = 0.0; size_t pn0 = 0, pn1 = 0; for (size_t i = 0; i < use.size(); ++i) { if (!use[i] || shell[i] < 0) continue; if ((H[i] == 0 ? 1 : 0) + (K[i] == 0 ? 1 : 0) + (L[i] == 0 ? 1 : 0) >= 2) continue; const double e = std::max(0.0, Esq[i]); if (std::abs(H[i] * t[0] + K[i] * t[1] + L[i] * t[2]) % 2) { pooled1 += e; ++pn1; } else { pooled0 += e; ++pn0; } } if (pn0 < PSEUDO_TRANSLATION_MIN_PER_CLASS || pn1 < PSEUDO_TRANSLATION_MIN_PER_CLASS || pooled0 <= 0.0) continue; if ((pooled1 / static_cast(pn1)) / (pooled0 / static_cast(pn0)) <= PSEUDO_TRANSLATION_MIN_POOLED) continue; // The transverse components refined from the half-integer start, the row's own pinned at // 1/2 - the same coarse-to-fine local search TranslationalNCS runs from its Patterson // peak, over the components the row does not constrain. std::array u{0.5 * t[0], 0.5 * t[1], 0.5 * t[2]}; double best_obj = cos_fit(u); double s = 1.0 / 64.0; for (int round = 0; round < 3; ++round) { for (bool improved = true; improved;) { improved = false; for (int i = 0; i < 3; ++i) { if (row[i] != 0) continue; for (double g : {-1.0, -0.5, 0.5, 1.0}) { std::array v = u; v[i] += g * s; const double o = cos_fit(v); if (o > best_obj + 1e-6) { best_obj = o; u = v; improved = true; } } } } s /= 2.5; } std::vector measured(bins, -1.0); window_depth(u, &measured); std::vector ratio(bins, 1.0); bool have = false; double running = 1.0, deepest = 1.0; for (int b = 0; b < bins; ++b) { if (measured[b] >= 0.0) { if (!have) { running = measured[b]; deepest = measured[b]; } else running = std::max(running, measured[b]); have = true; } ratio[b] = have ? running : 1.0; } if (!have || deepest >= PSEUDO_TRANSLATION_MAX_RATIO) continue; if (!best.found || deepest < best.deepest) { best.halves = t; best.ratio = std::move(ratio); best.deepest = deepest; best.found = true; } } return best; } // The reciprocal-space row {1,0,0} written the way a crystallographer names the zone: h00. std::string RowLabel(std::array row) { static const char* letter[3] = {"h", "k", "l"}; // RowOf canonicalises the sign lexicographically, which can leave the leading component // negative; hkl and -h-k-l are one row, so name it by the positive one. for (int v : row) if (v != 0) { if (v < 0) for (int& c : row) c = -c; break; } std::string out; for (int i = 0; i < 3; ++i) out += row[i] == 0 ? "0" : (row[i] == 1 ? letter[i] : std::to_string(row[i]) + letter[i]); return out; } // The crystal axis a screw on this row would run along: h00 -> a, 0k0 -> b, 00l -> c. ' ' for a // row that is not a principal one, which no screw extinguishes. char AxisOf(std::array row) { if (row[1] == 0 && row[2] == 0) return 'a'; if (row[0] == 0 && row[2] == 0) return 'b'; if (row[0] == 0 && row[1] == 0) return 'c'; return ' '; } std::string FormatDouble(double v, int decimals) { std::ostringstream o; o << std::fixed << std::setprecision(decimals) << v; return o.str(); } std::array RotKey(const gemmi::Op& op) { std::array out{}; for (int i = 0; i < 3; ++i) for (int j = 0; j < 3; ++j) out[i * 3 + j] = op.rot[i][j]; return out; } // The group a set of rotations GENERATES: close the set under composition. Empty when the closure // is not a crystallographic point group, which has at most 24 proper rotations. That is not a // guard against nothing: where the caller gives no lattice system, candidates are enumerated in // every setting at once, so the confirmed set can hold a four-fold and a six-fold about the same // axis. Their product is an integer matrix of infinite order and the closure never terminates - // and a set with no finite group behind it has no point group to propose either way. // Sorted, so the result compares directly against a candidate's rotation set. constexpr size_t MAX_PROPER_ROTATIONS = 24; // 432, the largest crystallographic rotation group std::vector> GeneratedRotationSet(const std::vector& generators) { std::vector group{gemmi::Op::identity()}; for (size_t i = 0; i < group.size(); ++i) for (const gemmi::Op& g : generators) { gemmi::Op p = group[i] * g; p.tran = {0, 0, 0}; if (std::none_of(group.begin(), group.end(), [&](const gemmi::Op& q) { return q.rot == p.rot; })) group.push_back(p); if (group.size() > MAX_PROPER_ROTATIONS) return {}; } std::vector> out; for (const gemmi::Op& op : group) out.push_back(RotKey(op)); std::sort(out.begin(), out.end()); return out; } // Whether a CELL can host a group's rotations. A setting is a statement about DIRECTION - // P 1 1 2 puts the 2-fold on c and needs alpha = beta = 90, where P 1 2 1 puts it on b and needs // alpha = gamma = 90 - so a candidate offered in a setting the metric does not have would be // merged on axes the crystal does not have. Compared on the metric tensor, each element against // its own scale, which makes the bound a tolerance on a lattice ANGLE and on an axis-length ratio; // gemmi's own is_compatible_with_groupops takes one absolute eps in A^2 instead, which means // something different on a 30 A cell and on a 300 A one. constexpr double CELL_SETTING_TOLERANCE = 2e-3; // ~0.11 deg on an angle, ~0.1% on an axis ratio bool CellHostsRotations(const gemmi::UnitCell& cell, const gemmi::GroupOps& gops) { const auto g = cell.metric_tensor(); const double G[3][3] = {{g.u11, g.u12, g.u13}, {g.u12, g.u22, g.u23}, {g.u13, g.u23, g.u33}}; for (const gemmi::Op& op : gops.sym_ops) { for (int i = 0; i < 3; ++i) for (int j = 0; j < 3; ++j) { double v = 0; for (int p = 0; p < 3; ++p) for (int q = 0; q < 3; ++q) v += static_cast(op.rot[p][i]) * G[p][q] * op.rot[q][j]; v /= static_cast(gemmi::Op::DEN) * gemmi::Op::DEN; if (std::fabs(v - G[i][j]) > CELL_SETTING_TOLERANCE * std::sqrt(G[i][i] * G[j][j])) return false; } } return true; } // How a candidate is NAMED in the report. short_name() is what the reference settings have // always been printed as, and it stays that; it cannot express a setting, though - P 1 2 1 and // P 1 1 2 are both "P2" - so a non-reference setting is printed as its extended Hermann-Mauguin // name, which is the only faithful one. std::string SettingName(const gemmi::SpaceGroup& sg) { return sg.is_reference_setting() ? sg.short_name() : sg.xhm(); } // The absences a group predicts over a fixed probe of low-index reflections - everything Stage B // judges a candidate on. Two settings with the same signature are one hypothesis written twice // (an alternative-origin entry, or a screw whose absences the centering already accounts for), so // only the first of them is worth scoring. Four indices is enough for every Sohncke setting in // the table: taking the probe to six or eight admits exactly the same candidates. std::vector AbsenceSignature(const gemmi::SpaceGroup& sg) { const gemmi::GroupOps gops = sg.operations(); std::vector out; out.reserve(9 * 9 * 9); for (int h = -4; h <= 4; ++h) for (int k = -4; k <= 4; ++k) for (int l = -4; l <= 4; ++l) out.push_back(gops.is_systematically_absent({{h, k, l}}) ? 1 : 0); return out; } // The rotation part of a space group in the reference setting (identity included), as a // sorted list of matrices - the key that groups space groups into a candidate point group. // It must be the rotation SET, not gemmi's PointGroup enum: P321 and P312 are both "32" yet // have their 2-folds along different directions, and only the matrices tell them apart. using RotationSet = std::vector>; RotationSet RotationSetOf(const gemmi::SpaceGroup& sg) { RotationSet out; for (const auto& op : sg.operations().derive_symmorphic().sym_ops) out.push_back(RotKey(op)); std::sort(out.begin(), out.end()); return out; } // The same, keeping only the PROPER rotations. A non-Sohncke group's symmorphic derivation // carries its mirrors and its inversion as well, so RotationSetOf(P 2_1/c) has four matrices // where RotationSetOf(P 2_1) has two and the two would never be matched. What the intensities // determine is the Laue class, i.e. the proper rotations, and this is the key that groups the // Sohncke candidates and their non-Sohncke relatives into one family. RotationSet ProperRotationSetOf(const gemmi::SpaceGroup& sg) { RotationSet out; for (const auto& op : sg.operations().derive_symmorphic().sym_ops) if (op.det_rot() > 0) out.push_back(RotKey(op)); std::sort(out.begin(), out.end()); return out; } // Proper rotations of a crystal system's holohedry (the highest lattice symmetry it can host), // in the reference setting. Any candidate point group must be a subgroup of this. RotationSet HolohedryRotationSet(gemmi::CrystalSystem system) { int number = 0; switch (system) { case gemmi::CrystalSystem::Triclinic: number = 1; break; // P1 case gemmi::CrystalSystem::Monoclinic: number = 3; break; // P2 (unique axis b) case gemmi::CrystalSystem::Orthorhombic: number = 16; break; // P222 case gemmi::CrystalSystem::Tetragonal: number = 89; break; // P422 case gemmi::CrystalSystem::Trigonal: number = 155; break; // R32 case gemmi::CrystalSystem::Hexagonal: number = 177; break; // P622 case gemmi::CrystalSystem::Cubic: number = 207; break; // P432 } const auto* sg = gemmi::find_spacegroup_by_number(number); return sg ? RotationSetOf(*sg) : RotationSet{}; } // A candidate point group: its proper rotations (reference setting) and a representative // symmorphic space group (used when only the point group is wanted, or for display). struct PointGroupInfo { RotationSet rotation_set; std::vector rotations; // non-identity proper rotations const gemmi::SpaceGroup* representative = nullptr; // This rung exists only because no reference setting names it (the a- and c-unique monoclinic // 2-folds). It may be ADOPTED like any other, but it must not take part in judging a higher // promotion - see the two sites that read it below. bool widened = false; }; // Enumerate candidate point groups. When a holohedry is given (from the lattice metric), keep // only its subgroups - this both skips operators the lattice forbids and avoids accepting a // coincidental higher symmetry; all subgroups down to P1 are still candidates. std::vector EnumeratePointGroups(const std::optional& holohedry, const std::optional& cell, bool all_rotation_sets) { std::vector out; std::map index; // Two passes. The reference settings first, exactly as before; then, only if asked and only // for rotation sets NO reference setting carries, a setting that names them. The four such // sets are the a-unique and c-unique monoclinic 2-folds and the two rhombohedral-axes trigonal // groups, and without them a crystal whose only 2-fold lies on a or c has no rung to stand on // between P1 and 222, so it falls to P1. Restricting the second pass to sets the first did not // reach makes it a pure addition: every point group reachable before is still reached, by the // same group, in the same setting. for (int pass = 0; pass < 2; ++pass) { if (pass == 1 && (!all_rotation_sets || !cell.has_value())) break; for (const auto& sg : gemmi::spacegroup_tables::main) { if (!sg.is_sohncke() || sg.is_reference_setting() != (pass == 0)) continue; RotationSet rs = RotationSetOf(sg); if (holohedry.has_value() && !std::includes(holohedry->begin(), holohedry->end(), rs.begin(), rs.end())) continue; if (pass == 1 && (index.count(rs) > 0 || !CellHostsRotations(*cell, sg.operations()))) continue; auto it = index.find(rs); size_t pos; if (it == index.end()) { PointGroupInfo info; for (const auto& op : sg.operations().derive_symmorphic().sym_ops) { if (op.rot == gemmi::Op::identity().rot) continue; info.rotations.push_back(gemmi::Op{op.rot, {0, 0, 0}, op.notation}); } info.rotation_set = rs; info.widened = (pass == 1); pos = out.size(); index[rs] = pos; out.push_back(std::move(info)); } else { pos = it->second; } // Prefer a symmorphic representative (the plain point-group setting). auto& info = out[pos]; if (info.representative == nullptr || (!info.representative->is_symmorphic() && sg.is_symmorphic())) info.representative = &sg; } } return out; } // The merge flattened into the arrays the search reads: which reflections each stage may use, the // two resolution-shell normalisations, and the index that pairs a reflection with its symmetry // mate. Built once per merge and shared with OperatorCorrelation below, so a single operator asked // about on its own is scored on exactly the population Stage A scores its own operators on. struct NormalisedMerge { size_t n = 0; std::vector H, K, L; std::vector I, Sigma, IoverSigma; std::vector key; std::vector pass_absence, pass_cc; std::vector Esq; // normalised over pass_absence - for the absence tests std::vector Ecc; // normalised over pass_cc - for the operator correlation // The equal-count resolution shell each reflection of the absence set fell in, low resolution // first, and how many there were; -1 outside the set. The pseudo-translation test measures its // modulation per shell, so it needs the shells Esq was built on and not shells of its own. std::vector absence_shell; int absence_shell_count = 0; double present_cut = 0.0; std::unordered_map key_to_index; }; NormalisedMerge PrepareMerge(const std::vector& merged, const SearchSpaceGroupOptions& opt) { const size_t n = merged.size(); // Flatten the reflections and mark which ones each stage may use. The correlation stage drops // weak reflections; the absence stage must keep them - that is where the screw-axis signal is. std::vector H(n), K(n), L(n); std::vector I(n), Sigma(n), IoverSigma(n); std::vector key(n); std::vector pass_absence(n, 0), pass_cc(n, 0); for (size_t i = 0; i < n; ++i) { const auto& r = merged[i]; H[i] = r.h; K[i] = r.k; L[i] = r.l; I[i] = r.I; Sigma[i] = std::isfinite(r.sigma) && r.sigma > 0 ? r.sigma : 0.0; key[i] = Canonicalize(r.h, r.k, r.l, opt.merge_friedel); const bool finite = std::isfinite(r.I) && std::isfinite(r.sigma) && r.sigma > 0 && std::isfinite(r.d) && r.d > 0; const bool in_range = finite && (opt.d_min_limit_A <= 0 || r.d >= opt.d_min_limit_A); IoverSigma[i] = finite ? r.I / r.sigma : 0.0; pass_absence[i] = in_range; } // present_i_over_sigma is a cut on the reflection's own significance, and the merged I/sigma is not // that: it carries the error model's (b*I)^2 term, so it saturates at ISa = 1/b for a reflection // measured once and stops rising with the intensity above that knee. Convert the cut to the // quantity the merge exports, once, here - see SearchSpaceGroupOptions::merge_isa for the // derivation and for what it is worth. An unknown ISa (or b = 0, no systematic term) leaves the cut // exactly where the caller set it. double present_cut = opt.present_i_over_sigma; if (opt.merge_isa > 0.0) { const double r = opt.present_i_over_sigma / opt.merge_isa; present_cut = opt.present_i_over_sigma / std::sqrt(1.0 + r * r); } // The correlation stage uses only genuinely-present reflections. Near-zero (systematically // absent) reflections would otherwise form a second cluster at the origin and fake a high // correlation for false operators - fatal on centered lattices, where half the reflections // are extinct. for (size_t i = 0; i < n; ++i) pass_cc[i] = pass_absence[i] && IoverSigma[i] >= present_cut && (opt.min_i_over_sigma <= 0 || IoverSigma[i] >= opt.min_i_over_sigma); // Resolution-normalised intensity E^2 = I / (shell), from equal-count resolution shells over a // given subset of the merge. Which subset matters: E^2 is only free of the resolution fall-off on // the population it was normalised over, so a caller has to normalise over the reflections it will // actually use (see Ecc below). auto shell_normalised = [&](const std::vector& subset, std::vector* shell_out = nullptr, int* bins_out = nullptr) { std::vector E(n, 0.0); std::vector order; order.reserve(n); for (size_t i = 0; i < n; ++i) if (subset[i]) order.push_back(i); std::sort(order.begin(), order.end(), [&](size_t a, size_t b) { return merged[a].d > merged[b].d; }); // low res -> high res // A hundred reflections a shell, at most twenty-five shells - and integer division, so // below two hundred there is a single shell and E^2 is I over one global mean. That // leaves the operator correlation exactly where it was, Pearson being invariant to a // common scale, but not the E^2 overlap cap below: it then cuts on I against the whole // merge instead of against each reflection's own shell. const int bins = std::clamp(static_cast(order.size() / 100), 1, 25); const size_t per = (order.size() + bins - 1) / std::max(1, bins); for (size_t b = 0; b * per < order.size(); ++b) { const size_t lo = b * per, hi = std::min(order.size(), lo + per); double sum = 0.0; for (size_t j = lo; j < hi; ++j) sum += I[order[j]]; const double mean = (hi > lo) ? sum / static_cast(hi - lo) : 0.0; for (size_t j = lo; j < hi; ++j) { E[order[j]] = mean > 0.0 ? I[order[j]] / mean : 0.0; if (shell_out) (*shell_out)[order[j]] = static_cast(b); } if (bins_out) *bins_out = static_cast(b) + 1; } return E; }; // Over the reflections the absence test uses, so it can judge "present" by intensity magnitude // rather than by a possibly under-estimated sigma (see present_e_squared). std::vector absence_shell(n, -1); int absence_shell_count = 0; std::vector Esq = shell_normalised(pass_absence, &absence_shell, &absence_shell_count); // Overlap guard (Stage A / correlation only): drop the extreme resolution-normalised-E tail, which // on a two-lattice crystal is the one-sided overlap contamination that poisons the operator CC. // See SearchSpaceGroupOptions::max_e_squared_for_cc. Absences (pass_absence) keep the full range. if (opt.max_e_squared_for_cc > 0.0) for (size_t i = 0; i < n; ++i) if (pass_cc[i] && Esq[i] > opt.max_e_squared_for_cc) pass_cc[i] = false; // The operator correlation is scored on E^2, not on raw I. Both members of a symmetry pair sit at // the same |s|, so the resolution fall-off is variance shared perfectly between the two arms of // every pair: a Pearson CC on raw I measures the fall-off as well as the symmetry, and the fall-off // lifts a FALSE operator's CC as much as a true one's. Measured over the rotation battery with // shell-matched random pairing as the null for a metrically-allowed false operator, that raw-I noise // floor has a median of 0.31 and reaches 0.53 on one crystal - above the old bound of 0.5 outright - // and it varies more between crystals (spread 0.46) than the whole true/false gap is wide (0.38), so // an absolute bound on it was a different test on every crystal. It also moves with the search // resolution cut, by a median 0.09 and up to 0.23, which is what made that cut decide symmetries. // Normalised, that floor has a median of 0.015 and a maximum of 0.06, and moves by a median 0.03. // Following POINTLESS, which likewise scores each symmetry element on normalised intensities // (Evans, Acta Cryst. D62, 72-82 (2006)). // // Normalised over pass_cc - the reflections the correlation actually pairs - and NOT over the // pass_absence set Esq uses. pass_cc keeps only the stronger reflections and the fraction it keeps // itself falls with resolution, so an E^2 built on pass_absence still carries a resolution trend // inside the pass_cc subset; reusing Esq here makes the coupling WORSE than raw I (floor movement // 0.14). Esq stays as it is: the absence tests need their own set, and the E^2 cap above helps // DEFINE pass_cc, so normalising that over pass_cc would be circular. // // The price is a new dependence in place of the old one: the CC now moves with whatever defines // pass_cc, i.e. with present_i_over_sigma and the merge_isa conversion above. That conversion is // what keeps the dependence harmless - it holds the cut at one counting significance on every // crystal, so the population this is normalised over means the same thing on all of them. std::vector Ecc = shell_normalised(pass_cc); std::unordered_map key_to_index; key_to_index.reserve(n * 2); for (size_t i = 0; i < n; ++i) if (pass_absence[i]) key_to_index.emplace(key[i], static_cast(i)); return {.n = n, .H = std::move(H), .K = std::move(K), .L = std::move(L), .I = std::move(I), .Sigma = std::move(Sigma), .IoverSigma = std::move(IoverSigma), .key = std::move(key), .pass_absence = std::move(pass_absence), .pass_cc = std::move(pass_cc), .Esq = std::move(Esq), .Ecc = std::move(Ecc), .absence_shell = std::move(absence_shell), .absence_shell_count = absence_shell_count, .present_cut = present_cut, .key_to_index = std::move(key_to_index)}; } // Stage A's statistic for one rotation: the correlation of resolution-normalised E^2(h) with // E^2(Rh) over the pairs the operator makes, plus the sigma-free disagreement H and the R the // promotion gates read. `visited` and `epoch` are scratch, taken as arguments rather than owned so // that several operators can be scored at once - each worker keeps its own pair. SpaceGroupOperatorScore ScoreOperator(const NormalisedMerge& m, const SearchSpaceGroupOptions& opt, const gemmi::Op& op, std::vector& visited, uint32_t& epoch) { ++epoch; std::vector x, y; // raw merged I of each pair, for the H statistic std::vector ex, ey; // the same pairs as E^2, for the correlation for (size_t i = 0; i < m.n; ++i) { if (!m.pass_cc[i] || visited[i] == epoch) continue; const auto m2 = op.apply_to_hkl(gemmi::Op::Miller{{m.H[i], m.K[i], m.L[i]}}); const HKLKey k2 = Canonicalize(m2[0], m2[1], m2[2], opt.merge_friedel); if (k2 == m.key[i]) continue; // reflection lies on this rotation axis const auto it = m.key_to_index.find(k2); if (it == m.key_to_index.end()) continue; const int j = it->second; if (!m.pass_cc[j]) continue; x.push_back(m.I[i]); y.push_back(m.I[j]); ex.push_back(m.Ecc[i]); ey.push_back(m.Ecc[j]); visited[i] = epoch; visited[j] = epoch; } SpaceGroupOperatorScore s; s.op_triplet_hkl = op.as_hkl().triplet('h'); s.n_pairs = static_cast(x.size()); s.cc = PearsonCC(ex, ey); // Sigma-free disagreement over the same pairs (see SpaceGroupOptions::max_operator_h_ratio). // On RAW I, deliberately - which is why the pairs are collected twice. The shell divisor cancels // in |I1-I2|/(I1+I2) exactly for a pair whose members share a shell, but not for one that // straddles a shell boundary: measured over the battery's operators, normalising moves H by a // median 0.08% but by 3.7% at p95 and 12% at worst, and max_operator_h_ratio has been decided on // a margin of 1.8%. H is calibrated on raw I and stays there. std::vector hv; hv.reserve(x.size()); double r_num = 0.0, r_den = 0.0; // intensity-weighted R across the operator (see r_stat) for (size_t p = 0; p < x.size(); ++p) { const double denom = x[p] + y[p]; if (denom > 0.0) { hv.push_back(std::fabs(x[p] - y[p]) / denom); r_num += std::fabs(x[p] - y[p]); r_den += denom; } } if (!hv.empty()) { const size_t mid = hv.size() / 2; std::nth_element(hv.begin(), hv.begin() + mid, hv.end()); s.h_stat = hv[mid]; } if (r_den > 0.0) s.r_stat = r_num / r_den; s.present = s.n_pairs >= opt.min_pairs_per_operator && std::isfinite(s.cc) && s.cc >= opt.min_operator_cc; return s; } } double AbsenceEvidence(double sum_u, int n_absent, int n_control) { if (n_absent <= 0 || n_control <= 0) return 0.0; const double a = n_absent, b = n_control; const double u = std::max(sum_u, a * MIN_U_PER_ABSENT_REFLECTION); const double T = u / (u + b); return -(a * std::log(T) + std::lgamma(a + b) - std::lgamma(a + 1) - std::lgamma(b)); } // The same for a SCREW zone, with the control COUNT taken out of it - the b -> infinity limit of // AbsenceEvidence, i.e. -log P(Gamma(n_absent, 1) <= sum_u) to leading order. // // A screw's control class is the COMPLEMENT of its absent class on one axial row, so the two move // together: a candidate that predicts more of the row absent leaves fewer reflections to be judged // against. AbsenceEvidence grows with the control count, so that candidate is charged for the very // reflections it correctly called extinct, and a group whose absent class is a strict SUPERSET of // another's, with the extra reflections equally dead, could score LOWER. Measured on a tetragonal // 4_1/4_3 crystal: 29 dead 00l against 8 control read 47.7 nats where a subset of 19 of them against // 18 control read 55.5 - the wrong order, from the control count alone. // // The count belongs in a p-value for ONE candidate and not in a ranking of several: each candidate's // tail is computed against its own null, and -log p from different nulls is not one scale. What is // left is the likelihood ratio of the absent class - dead against Wilson at the row's own mean - // which is a sum over reflections and so comparable across candidates. Asymptotically it is // n_absent * (log(1/ubar) - 1), linear in the number of absences at fixed deadness, so an equally // dead superset can no longer score lower. It is also the same number wherever the control class // dwarfs the absent one, and it keeps the property the whole test is built on: a class as strong as // its control reads about -n_absent, so a uniformly weak axial row still decides nothing. // // NOT used for the CENTERING class (AbsenceEvidence above), whose control is the whole present // population rather than the complement of the claim on one row. // LIMIT - this is a SUM over the zone, so it is dominated by its strongest one or two members: one // absence at ten times its row mean moves -a*log(sum_u) further than the other forty put together. // A zone verdict can therefore hang on a single reflection, and moving the resolution cut by a few // hundredths of an angstrom - which drops or admits exactly that reflection - can move the zone by // tens of nats in either direction. Measured, not feared. That is a property of this statistic, // independent of anything upstream of it. double ScrewZoneEvidence(double sum_u, int n_absent) { if (n_absent <= 0) return 0.0; const double a = n_absent; const double u = std::max(sum_u, a * MIN_U_PER_ABSENT_REFLECTION); return -a * std::log(u) + std::lgamma(a + 1); } // The LIMIT above, answered where it arises: the zone's sum with its single largest member dropped, // rescaled so the result still estimates the same quantity. Under the null the u are Exp(1) and // E[max] = H_n = 1 + 1/2 + ... + 1/n, so the remaining n-1 of them sum to n - H_n in expectation and // dividing by that recovers the mean the whole sum would have given. Multiplying back by n leaves a // number ScrewZoneEvidence reads exactly as before - same statistic, same floor, same calibration, // with a robust estimate of the zone's deadness in place of a fragile one. // // A screw row is half a dozen to a few dozen reflections, so ONE of them at a third of its row moves // the zone by more than the other seven put together. Measured on a monoclinic crystal whose eight // absent 0k0 are dead in every run: between two scaling passes that differed only in which weak // frames were rejected, one of those eight moved from 14 +- 9 to 99 +- 10 while the other seven did // not move at all, and the zone went from 30.1 nats to 17.1 and lost the 2(1) under a bound of 20. // The two half-set merges of that one reflection read 198 and 2.5 in the second pass, i.e. it was // never measured to the precision its sigma claimed - but no sigma is read here, on purpose (see // min_screw_absence_evidence), so the defence has to be the estimator. Trimmed, that zone reads 27.6 // nats on the pass that had lost it, set by the seven reflections that did not move. // // ONE member, whatever the zone's size: a zone with two strong absences is not a zone with a bad // reflection in it, it is a zone that is not extinct, and trimming further would hide that. What // this costs a genuinely uniform zone is that the estimate is biased high by (n-1)/(n - H_n), i.e. // the evidence is UNDER-stated by n*log of that - 2.2 nats on a zone of eight, 3.9 on one of // sixty-four. It only ever refuses, never claims. double TrimmedZoneSum(double sum_u, double max_u, int n_absent) { if (n_absent < 2) return sum_u; double harmonic = 0.0; for (int i = 1; i <= n_absent; ++i) harmonic += 1.0 / i; const double expected_remaining = n_absent - harmonic; // > 0 for every n >= 2 return n_absent * std::max(0.0, sum_u - max_u) / expected_remaining; } SearchSpaceGroupResult SearchSpaceGroup( const std::vector& merged, const SearchSpaceGroupOptions& opt) { SearchSpaceGroupResult result; if (merged.empty()) return result; // The merge flattened and normalised once (see NormalisedMerge). Named locally, because every // stage below reads these arrays. const NormalisedMerge nm = PrepareMerge(merged, opt); const size_t n = nm.n; const auto& H = nm.H; const auto& K = nm.K; const auto& L = nm.L; const auto& I = nm.I; const auto& Sigma = nm.Sigma; const auto& IoverSigma = nm.IoverSigma; const auto& key = nm.key; const auto& pass_absence = nm.pass_absence; const auto& pass_cc = nm.pass_cc; const auto& Esq = nm.Esq; const double present_cut = nm.present_cut; const auto& absence_shell = nm.absence_shell; const int absence_shell_count = nm.absence_shell_count; // The pseudo-translation each axial row is judged against, and the factor its modulation puts on // the expected intensity of each reflection. See FindRowModulation. A reflection in the class the // translation suppresses is expected at `ratio` of its shell-mates in the other class, so the // absence test must score it against that and not against 1 - otherwise a class that is merely // suppressed reads as extinct and buys a screw axis. All ones when no translation was measured, // which is the common case and leaves every number below exactly as it was. // // The three principal rows are the ones screw axes extinguish in a conventional setting. A row // outside them simply gets no correction, which is the inert direction. std::map row_modulation; if (opt.cell.has_value()) // Through RowOf, not written out: it sign-canonicalises, so the key a reflection looks up // with is {0,0,-1} rather than {0,0,1} and a hand-written key silently never matches. for (const AxialRow& row : {RowOf(1, 0, 0), RowOf(0, 1, 0), RowOf(0, 0, 1)}) { auto m = FindRowModulation(*opt.cell, row, pass_absence, H, K, L, Esq, absence_shell, absence_shell_count); if (m.found) { result.pseudo_translations.push_back({row, m.halves, m.deepest}); row_modulation.emplace(row, std::move(m)); } } // How many reflections of each principal axial row the merge holds at all. A screw lives on one // row, so a row the sweep never recorded - the spindle's blind cone points down it, or its // reflections fall outside the resolution range - is a screw nobody could have tested, and the // difference between "tested and found absent" and "never looked at" is invisible in every other // number here. Counted once over the merge, because it is a property of the data and not of a // candidate. Only the three principal rows: a screw extinguishes no other. std::map axial_row_observed; for (size_t i = 0; i < n; ++i) if (pass_absence[i] && (H[i] == 0) + (K[i] == 0) + (L[i] == 0) == 2) ++axial_row_observed[RowOf(H[i], K[i], L[i])]; std::vector tncs_expected(n, 1.0); for (size_t i = 0; i < n && !row_modulation.empty(); ++i) { if (!pass_absence[i] || absence_shell[i] < 0) continue; const auto it = row_modulation.find(RowOf(H[i], K[i], L[i])); if (it == row_modulation.end()) continue; const auto& t = it->second.halves; if (std::abs(H[i] * t[0] + K[i] * t[1] + L[i] * t[2]) % 2) tncs_expected[i] = it->second.ratio[absence_shell[i]]; } // The same correction for a GLIDE zone, which needs it more. tncs_expected above reaches only // the three axial rows - the rows a screw extinguishes - and leaves every other reflection at 1. // A glide zone is a PLANE of general reflections, and a half-integer pseudo-translation // suppresses one parity class of the WHOLE crystal: a translation at (0,0,1/2) makes h0l with l // odd weak over hundreds of reflections, which is exactly the pattern a c-glide normal to b // makes. That is the one mechanism able to fake a glide across a whole zone, and it is not // hypothetical - two of the corpus's protein datasets carry a measurable one. So the absent // class is divided by the depth the modulation puts on it before it is judged, which is the same // measurement restated on the scale its control class is on. // // The DEEPEST of the modulations a reflection falls in the suppressed class of, not their // product: two rows reporting the same physical translation would otherwise correct for it // twice. All ones when no translation was measured, which is the common case. std::vector zone_tncs_expected(n, 1.0); for (size_t i = 0; i < n && !row_modulation.empty(); ++i) { if (!pass_absence[i] || absence_shell[i] < 0) continue; double f = 1.0; for (const auto& [row, m] : row_modulation) if (std::abs(H[i] * m.halves[0] + K[i] * m.halves[1] + L[i] * m.halves[2]) % 2) f = std::min(f, m.ratio[absence_shell[i]]); zone_tncs_expected[i] = f; } // The merge's own random-noise R floor from the two half-dataset merges (see // SearchSpaceGroupResult::merge_r_floor). Over the same present (pass_cc) reflections the operator // R is measured on, so the two are at the same intensity range and multiplicity and their ratio // means the same on every crystal - intensity-weighted, sigma-free, and built without applying any // candidate symmetry so a false operator cannot inflate it. { double r_num = 0.0, r_den = 0.0; for (size_t i = 0; i < n; ++i) { if (!pass_cc[i]) continue; const double a = merged[i].I_half[0], b = merged[i].I_half[1]; if (std::isfinite(a) && std::isfinite(b) && a + b > 0.0) { r_num += std::fabs(a - b); r_den += a + b; } } if (r_den > 0.0) result.merge_r_floor = r_num / r_den; } // --- Stage A: score each distinct rotation operator once --- auto score_operator = [&](const gemmi::Op& op, std::vector& visited, uint32_t& epoch) -> SpaceGroupOperatorScore { return ScoreOperator(nm, opt, op, visited, epoch); }; std::map, SpaceGroupOperatorScore> op_cache; std::vector visited(n, 0); uint32_t epoch = 0; auto operator_score = [&](const gemmi::Op& op) -> const SpaceGroupOperatorScore& { const auto rk = RotKey(op); auto it = op_cache.find(rk); if (it != op_cache.end()) return it->second; return op_cache.emplace(rk, score_operator(op, visited, epoch)).first->second; }; // Conjugate rotations (symmetry-equivalent within the point group) relate symmetry-equivalent // reflection sets, so on real data their CCs cluster; a noisy crystal can push one class member // below min_operator_cc while the class is unmistakably present (this was first seen on a cubic // crystal whose three 3-folds spread over 0.13 of CC with only the weakest below the bound). // Judge each conjugacy class by its mean CC, not its weakest // member, so a genuine high-symmetry point group is not lost to one marginal operator. chi2_under // (below) remains the safety net against a truly false promotion. Returns {all classes present, // worst class-mean CC}. auto point_group_present = [&](const std::vector& rots) -> std::pair { const size_t m = rots.size(); std::vector cls(m, -1); int n_cls = 0; for (size_t i = 0; i < m; ++i) { if (cls[i] >= 0) continue; cls[i] = n_cls; for (size_t j = i + 1; j < m; ++j) if (cls[j] < 0) for (const auto& p : rots) if ((p * rots[i] * p.inverse()).rot == rots[j].rot) { cls[j] = n_cls; break; } ++n_cls; } bool ok = true; double worst_mean = 1.0; for (int c = 0; c < n_cls; ++c) { // Average over the class members that actually have enough pairs to score; a single // low-multiplicity / degenerate (NaN) operator in an otherwise strong class is skipped, // not allowed to veto the class. The class must still have at least one scored member. double sum_cc = 0.0; int n_valid = 0; for (size_t i = 0; i < m; ++i) if (cls[i] == c) { const auto& s = operator_score(rots[i]); if (s.n_pairs >= opt.min_pairs_per_operator && std::isfinite(s.cc)) { sum_cc += s.cc; ++n_valid; } } const double mean_cc = n_valid > 0 ? sum_cc / n_valid : 0.0; worst_mean = std::min(worst_mean, mean_cc); if (n_valid == 0 || mean_cc < opt.min_operator_cc) ok = false; } return {ok, worst_mean}; }; std::optional holohedry; if (opt.lattice_system.has_value()) holohedry = HolohedryRotationSet(opt.lattice_system.value()); const auto point_groups = EnumeratePointGroups(holohedry, opt.cell, opt.enumerate_all_rotation_sets); // Every operator the search can ask about comes from this list, and scoring one is a pass over the // whole merge with a hash lookup per reflection - the most expensive thing in here. They do not // depend on each other, so score the distinct ones now and let the search below read the cache. // One `visited` per worker, not per operator: it is as long as the merge, so allocating it per // operator would cost more than the scoring. { std::vector distinct; std::vector> keys; for (const auto& pg : point_groups) for (const auto& rot : pg.rotations) { const auto rk = RotKey(rot); if (std::find(keys.begin(), keys.end(), rk) == keys.end()) { keys.push_back(rk); distinct.push_back(rot); } } std::vector scored(distinct.size()); ParallelChunks(static_cast(distinct.size()), std::min(opt.nthreads, distinct.size()), [&](int lo, int hi) { std::vector scratch(n, 0); uint32_t ep = 0; for (int i = lo; i < hi; ++i) scored[i] = score_operator(distinct[i], scratch, ep); }); for (size_t i = 0; i < distinct.size(); ++i) op_cache.emplace(keys[i], scored[i]); } // Mapping every observation onto its symmetry representative under a candidate's rotations - one // apply_to_hkl + Canonicalize per observation per operator - is the expensive half of BOTH // quantities below, and both need exactly the same mapping. Build it once per point group. struct Acc { double sw = 0.0, swI = 0.0; int n = 0; }; struct Orbits { std::vector acc; // one inverse-variance accumulator per orbit, in first-seen order std::vector orbit; // observation -> its orbit, -1 for one neither quantity below uses }; auto build_orbits = [&](const std::vector& rotations) -> Orbits { Orbits orb; orb.orbit.assign(n, -1); // The representative is interned to a dense index right here, so the two quantities below // index an array rather than hashing a key per observation - each of them is a pass over the // whole merge, and the lookup was the larger half of both. Same orbits, same order, same sums. std::unordered_map orbit_id; for (size_t i = 0; i < n; ++i) { if (!pass_cc[i] || !(Sigma[i] > 0.0)) continue; HKLKey best = key[i]; for (const auto& op : rotations) { const auto m = op.apply_to_hkl(gemmi::Op::Miller{{H[i], K[i], L[i]}}); const HKLKey k2 = Canonicalize(m[0], m[1], m[2], opt.merge_friedel); if (std::make_tuple(k2.h, k2.k, k2.l) < std::make_tuple(best.h, best.k, best.l)) best = k2; } const auto [it, fresh] = orbit_id.emplace(best, static_cast(orb.acc.size())); if (fresh) orb.acc.emplace_back(); orb.orbit[i] = it->second; Acc& g = orb.acc[it->second]; const double w = 1.0 / (Sigma[i] * Sigma[i]); g.sw += w; g.swI += w * I[i]; g.n += 1; } return orb; }; // Reduced chi^2 of the intensities merged under a point group's rotations - how well its symmetry // equivalents agree RELATIVE TO THEIR ERRORS. A real point group gives ~1; a false operator forces // non-equivalent reflections together, so they disagree by many sigma and chi^2 blows up. This is // more sensitive than R-meas to a strong pseudo-symmetry (where the intensities still correlate well // - high operator CC - but not within their errors). Inverse-variance weighted mean per orbit, over // the present (pass_cc) reflections. auto chi2_under = [&](const Orbits& orb) -> double { double chi2 = 0.0; long dof = 0; for (size_t i = 0; i < n; ++i) { if (orb.orbit[i] < 0) continue; const Acc& g = orb.acc[orb.orbit[i]]; if (g.n < 2) continue; const double mean = g.swI / g.sw, dev = I[i] - mean; chi2 += dev * dev / (Sigma[i] * Sigma[i]); } for (const Acc& g : orb.acc) if (g.n >= 2) dof += g.n - 1; return dof > 0 ? chi2 / static_cast(dof) : std::numeric_limits::quiet_NaN(); }; // Extra intensity-proportional systematic error a point group's merge has to invoke to reconcile // its symmetry equivalents: the smallest b for which sigma^2 + (b I)^2 brings the merged reduced // chi^2 down to 1. A genuine symmetry needs almost none - its equivalents already agree within // their errors, so the extra scatter is random and multiplicity absorbs it. A twin or pseudo- // symmetry forces non-equivalent reflections together, and that disagreement scales with I, so b // has to grow to swallow it (mirroring the merge error model's b / ISa collapse). This isolates // the systematic part of the scatter, which the fixed-sigma chi^2 ratio cannot: a genuine but // imperfectly-scaled high-symmetry merge and a twin can share a chi^2 ratio (~2) yet differ // sharply here (a genuine cubic step b x1.04 vs a merohedral twin b x1.6). // This `b` is a fraction of I fitted with the sigma^2 coefficient held at 1 - it is NOT the merge // error model's b, and NOT XDS's b either. The gate constants below are calibrated in this // convention; converting them to any other silently squares the ratios and makes the absolute // floor a-dependent, on a quantity that has no a. Leave it alone. auto merge_systematic_b = [&](const Orbits& orb) -> double { std::vector> obs; // I, sigma, deviation-from-orbit-mean long n_orbits = 0; // orbits contributing, for the degrees of freedom for (const Acc& g : orb.acc) if (g.n >= 2) ++n_orbits; for (size_t i = 0; i < n; ++i) { if (orb.orbit[i] < 0) continue; const Acc& g = orb.acc[orb.orbit[i]]; if (g.n < 2) continue; obs.push_back({I[i], Sigma[i], I[i] - g.swI / g.sw}); } // Each deviation is taken from a mean fitted on its own orbit, so one degree of freedom per // orbit is spent: dof = sum(n-1) = N - G, the same denominator chi2_under uses above. Dividing // by N instead understates the reduced chi^2 and so overstates the b that brings it to 1, and // it does so unevenly - a low-multiplicity parent loses a larger fraction of its dof than the // higher-multiplicity candidate, which inflates the b RATIO the gates read. const double dof = static_cast(obs.size()) - static_cast(n_orbits); if (obs.size() < 20 || dof <= 0.0) return 0.0; auto reduced_chi2 = [&](double b) { double s = 0.0; for (const auto& o : obs) s += o[2] * o[2] / (o[1] * o[1] + (b * o[0]) * (b * o[0])); return s / dof; }; if (reduced_chi2(0.0) <= 1.0) return 0.0; double lo = 0.0, hi = 2.0; // b is a fraction of I; 2.0 = 200% is far past any real error model for (int it = 0; it < 40; ++it) { const double mid = 0.5 * (lo + hi); (reduced_chi2(mid) > 1.0 ? lo : hi) = mid; } return 0.5 * (lo + hi); }; // Operator-CC-confirmed candidates, each with its merge chi^2 and systematic-error b; chi2_ref = // the most consistent. struct PGCand { const PointGroupInfo* pg; int order; double min_class_cc; double chi2; double b_extra; // Filled by the selection loop below and carried so the adopted candidate's H // ratio can be reported whether or not the bound had anything to say about it. double h_ratio = std::numeric_limits::quiet_NaN(); // The added operators' mean intensity-weighted R, and that over the merge's random // R floor - the full-resolution merge-degradation gate's numbers (see r_stat, // max_operator_r_over_best and max_operator_r_over_floor). Carried for the same // reason as h_ratio: to report the adopted candidate's whether or not it decided. double r_added = std::numeric_limits::quiet_NaN(); double r_over_floor = std::numeric_limits::quiet_NaN(); // Report-only, for the finalist ledger: the subgroup b the tests below compared // this candidate against, whether it survived every consistency test, and - when // it was the highest refusal - why it did not. The selection loop computes all // three already and drops them on the floor; the ledger is that table kept. double parent_b_used = -1.0; bool eligible = false; // Offered by the closure below rather than confirmed operator by operator. Like a // widened rung it may be ADOPTED but must not JUDGE - see the two sites that read it. bool closure = false; std::string why; }; int refused_order = 0; bool refused_twin_gate = false; double refused_fraction = std::numeric_limits::quiet_NaN(); std::string refused_pg_hm, refused_why; const gemmi::SpaceGroup *refused_pg_rep = nullptr; std::vector pg_cands; double chi2_ref = std::numeric_limits::infinity(); // Which point groups their own operators confirm. Serial: the operators were all scored in // parallel above, so this is a handful of cache lookups per group. std::vector confirmed; std::vector confirmed_cc; for (const auto& pg : point_groups) { const auto [present, min_class_cc] = point_group_present(pg.rotations); if (!present) continue; confirmed.push_back(&pg); confirmed_cc.push_back(min_class_cc); } // CLOSURE. The loop above admits a point group only when EVERY one of its operators clears the // bar, so admission is a conjunction over an enumerated list: one operator reading low deletes the // whole group, and the operators that DID clear the bar can then generate a group no admitted // candidate names. That is a self-contradiction in the operator table, not a low symmetry - a set // of five confirmed two-folds whose products include a refused four-fold cannot be a subgroup of // anything, and what it generates is the higher group. It happens whenever the merge the operators // were scored on cannot connect the parts of reciprocal space one operator compares while another // stays inside a well-connected part: the disconnected operator reads as broken and the group is // lost, though the rest of the table already implies it. // // So the generated group is offered as a candidate too. It is PROPOSED, not adopted: it goes // through the same chi^2, systematic-b, H and added-operator-R tests as every other candidate, // and like a widened rung it does not set chi2_ref and is not a parent, so offering it cannot // make any other promotion harder than it was. // // EXACTLY the generated group, never a supergroup of it. Offering the smallest candidate that // CONTAINS the generated set - to reach a rotation set the basis cannot name, such as a diagonal // orthorhombic one in a tetragonal metric - was tried and is an over-call: a crystal whose one // confirmed operator reads 0.889 while the other two read 0.05 then has 222 proposed on its // behalf, and where only one operator is confirmed the added-operator R is judged against the // loose first-step-out-of-P1 bound, which it cleared by 3%. That hole is real but it is an // enumeration hole, and it has to be closed by NAMING the group in a basis that carries it, not // by promoting past the evidence. std::vector confirmed_closure(confirmed.size(), 0); { std::vector generators; for (const auto& pg : point_groups) for (const auto& rot : pg.rotations) if (operator_score(rot).present && std::none_of(generators.begin(), generators.end(), [&](const gemmi::Op& g) { return g.rot == rot.rot; })) generators.push_back(rot); const auto generated = GeneratedRotationSet(generators); if (!generated.empty()) { const PointGroupInfo* proposal = nullptr; for (const auto& pg : point_groups) if (pg.rotation_set == generated) proposal = &pg; if (proposal != nullptr && std::find(confirmed.begin(), confirmed.end(), proposal) == confirmed.end()) { confirmed.push_back(proposal); confirmed_cc.push_back(point_group_present(proposal->rotations).second); confirmed_closure.push_back(1); if (proposal->representative) result.generated_point_group_hm = proposal->representative->point_group_hm(); } } } // Merging under a candidate costs three passes over the whole merge, and there are a handful of // candidates, each reading nothing but the shared reflection arrays - so give each one a thread. // Every candidate's sums are still formed on one thread, over the same reflections in the same // order, so the numbers this hands back do not depend on the split. std::vector cand_chi2(confirmed.size(), std::numeric_limits::quiet_NaN()); std::vector cand_b(confirmed.size(), 0.0); ParallelFor(static_cast(confirmed.size()), std::min(opt.nthreads, confirmed.size()), [&](int i) { if (confirmed[i]->rotations.empty()) return; const Orbits orb = build_orbits(confirmed[i]->rotations); cand_chi2[i] = chi2_under(orb); cand_b[i] = merge_systematic_b(orb); }); for (size_t i = 0; i < confirmed.size(); ++i) { const PointGroupInfo& pg = *confirmed[i]; pg_cands.push_back({&pg, static_cast(pg.rotations.size()) + 1, confirmed_cc[i], cand_chi2[i], cand_b[i]}); pg_cands.back().closure = confirmed_closure[i] != 0; // A rung the widened enumeration added does not set the reference chi^2, and neither does the // closure proposal. Otherwise adding it // would make every HIGHER promotion harder to reach - chi2_ref is a minimum - and the search // could answer LOWER because a candidate was offered, which is the opposite of what offering // it is for. Measured before this line existed: an F-cubic crystal's 432 and an orthorhombic // crystal's 222 were both refused once the a- and c-unique 2-folds joined the list. if (!pg.rotations.empty() && std::isfinite(cand_chi2[i]) && !pg.widened && !confirmed_closure[i]) chi2_ref = std::min(chi2_ref, cand_chi2[i]); } // The best-agreeing operator anywhere in the data: the smallest intensity-weighted R over every // CC-confirmed operator (see the merge-degradation gate below). A genuine symmetry operator // relates equal intensities, so its R is small; a low R is thus itself the mark of a genuine // operator, and the smallest one is the cleanest reference the data offers. It is GLOBAL, not // per-candidate, because a false point group can be built entirely from operators that agree // among THEMSELVES - a pseudo-tetragonal 4 whose 4-folds relate the same wrong intensities - and // only an operator outside that group (the one true 2-fold) exposes them. Confirmed operators // only: an unconfirmed (low-CC) one has, by construction, a high R and never sets this minimum. double global_best_r = std::numeric_limits::infinity(); int n_confirmed_ops = 0; for (const auto& [rk, s] : op_cache) if (s.present && s.n_pairs >= opt.min_pairs_per_operator && s.r_stat > 0.0) { global_best_r = std::min(global_best_r, s.r_stat); ++n_confirmed_ops; } // Choose the largest point group that is both operator-confirmed AND self-consistent (its merge // chi^2 is not inflated past the miscalibration-widened bound below; ties -> higher min class CC). // Identity (no operators) is always consistent, so it stays the P1 fallback. const PointGroupInfo* best_pg = nullptr; int best_pg_order = 0; double best_pg_min_cc = -2.0; for (auto& c : pg_cands) { // A genuine symmetry operator merges equivalent reflections, so it barely changes the reduced // chi^2 relative to the best subgroup - across the whole rotation-test battery every correct // point group stays within ~1.7x, even on weak or badly-integrated data (a cubic F432 chi2_ref // 8.3 -> 1.15; a tetragonal P41212 -> 1.71). A twin law or pseudo-symmetry forces non-equivalent // reflections together, so its ratio is markedly higher (a merohedral twin 2-fold: R3 3.02 -> // R32 6.07, ratio 2.01). max_merge_chi2_ratio sits between the two. (An earlier log10(chi2_ref) widening // compensated for an under-calibrated error model that inflated real-symmetry ratios with data // weakness; the variance-floor fix removed that inflation, and the widening now only let the // twin through, so it is gone.) bool consistent = c.pg->rotations.empty() || !std::isfinite(c.chi2) || !std::isfinite(chi2_ref) || c.chi2 <= chi2_ref * opt.max_merge_chi2_ratio; // Systematic-error test vs the largest confirmed subgroup (by rotation-set inclusion): merging // under a genuine operator gains multiplicity without intensity-proportional disagreement, so the // merge error model's b barely moves; a merohedral twin forces non-equivalent reflections together // and b balloons. It both RESCUES a genuine step whose chi^2 drifts just past the ratio bound // (imperfectly scaled data) and VETOES a twin whose chi^2 now looks self-consistent but whose b // balloons - the chi^2 ratio alone no longer separates them. double parent_b = -1.0; // Every confirmed subgroup of the largest order below this candidate. There can be more than one // - 422 has both 4 and 222 - and on a twinned crystal the rival is not a harmless alternative: a // P4 crystal twinned by 2[100] has its two twin 2-folds confirmed, so 222 is CC-confirmed too and // CONTAINS the twin laws. Normalising the H test against it hides the twin among the promotion's // own real operators (measured on the synthetic grid: ratio 8.19 against the true parent 4, 0.78 // against the rival 222). Which one is the true parent is exactly what is unknown here, so the // promotion must answer to all of them. std::vector parents; if (!c.pg->rotations.empty()) { int parent_order = 0; for (const auto& s : pg_cands) // Same reason as chi2_ref above: a rung only the widened enumeration offers, and the // closure proposal, are not parents. The promotion answers to the most damning of its // parents, so admitting two more order-2 subgroups of 222 makes the 222 step strictly // harder than it was before the rung was offered at all. if (!s.pg->widened && !s.closure && s.order < c.order && s.order >= parent_order && std::includes(c.pg->rotation_set.begin(), c.pg->rotation_set.end(), s.pg->rotation_set.begin(), s.pg->rotation_set.end())) { if (s.order > parent_order) { parent_order = s.order; parent_b = s.b_extra; parents.clear(); } else { // Tied parent. The b tests answer to the most damning of them, exactly as the H // test below does: the SMALLEST parent b is the one that makes the veto easiest to // trip and the rescue hardest to pass, and a rival subgroup that already contains // the twin laws has its own b ballooned - taking it would hide the twin. parent_b = std::min(parent_b, s.b_extra); } parents.push_back(s.pg); } } // Sigma-free twin test: compare the disagreement H of the operators this promotion ADDS with the // disagreement of the parent's own operators, measured on the same reflections. A real operator // relates equal intensities and matches the parent; a twin law relates different ones and reads // systematically higher. Skipped when either side has too few pairs to mean anything, and when // there is no parent group to normalise against (the first step out of P1). Where several parents // tie (see above), the promotion is judged on the most damning of them. double h_ratio = std::numeric_limits::quiet_NaN(); double h_added = 0.0; // the added operators' own H, for the twin fraction below double r_added = std::numeric_limits::quiet_NaN(); // added operators' mean R (full-res gate) for (const auto *parent : parents) { double h_new = 0.0, h_par = 0.0, r_new = 0.0; int n_new = 0, n_par = 0, pairs_new = 0, pairs_par = 0; for (const auto &rot : c.pg->rotations) { if (rot.rot == gemmi::Op::identity().rot) continue; const auto &os = operator_score(rot); if (os.n_pairs < opt.min_pairs_per_operator) continue; const bool in_parent = std::binary_search(parent->rotation_set.begin(), parent->rotation_set.end(), RotKey(rot)); if (in_parent) { h_par += os.h_stat; ++n_par; pairs_par += os.n_pairs; } else { h_new += os.h_stat; ++n_new; pairs_new += os.n_pairs; r_new += os.r_stat; } } if (n_new > 0 && n_par > 0 && pairs_new >= opt.min_pairs_for_h && pairs_par >= opt.min_pairs_for_h && h_par > 0.0) { const double r = (h_new / n_new) / (h_par / n_par); // Record the added operators' R at the same (max-H-ratio) parent the H ratio reports. if (!std::isfinite(h_ratio) || r > h_ratio) r_added = r_new / n_new; if (!std::isfinite(h_ratio) || r > h_ratio) { h_ratio = r; h_added = h_new / n_new; } } } // The first step out of P1 has no parent with operators, so the H loop above left r_added // unset: the added set is then ALL of this candidate's operators, and the only reference is // the merge's random-noise floor (there is no other operator to compare against). This is the // one guard that step has - the H ratio and the b veto both need a parent group. if (!std::isfinite(r_added) && !c.pg->rotations.empty()) { double r_new = 0.0; int n_new = 0, pairs_new = 0; for (const auto &rot : c.pg->rotations) { if (rot.rot == gemmi::Op::identity().rot) continue; const auto &os = operator_score(rot); if (os.n_pairs < opt.min_pairs_per_operator) continue; r_new += os.r_stat; ++n_new; pairs_new += os.n_pairs; } if (n_new > 0 && pairs_new >= opt.min_pairs_for_h) r_added = r_new / n_new; } c.parent_b_used = parent_b; c.r_added = r_added; c.r_over_floor = (std::isfinite(r_added) && std::isfinite(result.merge_r_floor) && result.merge_r_floor > 0.0) ? r_added / result.merge_r_floor : std::numeric_limits::quiet_NaN(); c.h_ratio = h_ratio; // The chi^2 ratio is only trustworthy when the error model is calibrated. When even the best // subgroup's reduced chi^2 (chi2_ref) is far above 1 - weak, low-resolution data whose merged // sigmas are badly under-estimated - the ratio grows with point-group order for genuine high // symmetry too and wrongly rejects it (a true weak F432 reaches ratio ~14). The systematic-b test // re-fits its own error, so it stays valid under a broken sigma model: a genuine step's b barely // moves (b-ratio ~1) while a twin's balloons. So once chi2_ref shows the error model is unreliable, // a promotion is rescued on the b-test alone (subject to the balloon veto below); otherwise the // rescue is confined to the narrow chi^2 band just past the ratio bound. const bool miscalibrated = std::isfinite(chi2_ref) && chi2_ref > opt.chi2_ref_reliable; if (!consistent && parent_b > 1e-4 && c.b_extra <= parent_b * opt.max_systematic_b_ratio && (miscalibrated || (std::isfinite(c.chi2) && std::isfinite(chi2_ref) && c.chi2 <= chi2_ref * opt.max_merge_chi2_rescue))) consistent = true; // The H test is a necessary condition for promotion where it can be computed: it is the only // statistic measured to separate genuine symmetry from a merohedral twin across data amounts. const bool h_refused = std::isfinite(h_ratio) && h_ratio > opt.max_operator_h_ratio; if (h_refused) consistent = false; // Full-resolution merge-degradation test. A necessary condition like H, but read against a // CLEAN reference rather than against a parent that can itself be contaminated - so it catches // a pseudo-symmetric cascade the parent-normalised gates wave through (a 2 -> 222 -> 422 built // by pooling one real 2-fold with false ones), and it is the only gate that can act on the // first step out of P1. Not rescuable: the b rescue above only lifts a chi^2-borderline case. // - When the data confirm more than one operator, judge the added operators' R against the // GLOBALLY best-agreeing operator (the smallest R anywhere): on a genuine group every // operator agrees about as well as that best one, so the ratio is ~1; a false operator - // even one whose group's members all agree among themselves, like a pseudo-tetragonal 4 - // sits far above it (see max_operator_r_over_best). // - The first step out of P1, where only one operator is confirmed at all, has no other // operator to be the best - fall back to the merge's own random-noise R floor // (max_operator_r_over_floor). bool r_refused = false; double r_gate_value = std::numeric_limits::quiet_NaN(); if (std::isfinite(c.r_added)) { if (n_confirmed_ops >= 2 && std::isfinite(global_best_r) && global_best_r > 0.0) { r_gate_value = c.r_added / global_best_r; r_refused = r_gate_value > opt.max_operator_r_over_best; } else if (std::isfinite(result.merge_r_floor) && result.merge_r_floor > 0.0) { r_gate_value = c.r_added / result.merge_r_floor; r_refused = r_gate_value > opt.max_operator_r_over_floor; } } if (r_refused) consistent = false; if (!consistent) { // Same three sentences the highest refusal gets below, but kept per candidate for the // ledger: short, because the long form is written once for the group the user is told about. c.why = h_refused ? "H ratio" : r_refused ? "added-operator R" : "merge chi^2"; // Record the highest-order refusal so the caller can say WHY it is processing lower. if (c.order > refused_order && c.pg->representative) { refused_order = c.order; refused_pg_hm = c.pg->representative->point_group_hm(); refused_pg_rep = c.pg->representative; refused_twin_gate = h_refused || r_refused; refused_fraction = h_refused ? std::max(0.0, 0.5 - h_added) : std::numeric_limits::quiet_NaN(); if (h_refused) // Twinning by a fraction a scales every twin-related difference by (1-2a), and // |I1-I2|/(I1+I2) is uniform on [0,1] for untwinned Wilson intensities, so the // added operator's median H reads (1-2a)/2 and implies a = 0.5 - H. Quoted because // it is the number a user acts on. It is a LOWER bound: measurement error only // adds to H, and so only subtracts from a. refused_why = "operator disagreement H is " + FormatDouble(h_ratio, 2) + "x the parent's (bound " + FormatDouble(opt.max_operator_h_ratio, 2) + ") - the added operator relates unequal intensities, as a twin law of " "fraction " + FormatDouble(std::max(0.0, 0.5 - h_added), 2) + " or more would"; else if (r_refused) refused_why = "its added operators' intensity-weighted R is " + FormatDouble(r_gate_value, 1) + "x " + (n_confirmed_ops >= 2 ? "the best-agreeing operator in the data (bound " + FormatDouble(opt.max_operator_r_over_best, 1) : "the merge's own random-noise floor (bound " + FormatDouble(opt.max_operator_r_over_floor, 1)) + "x) - at full resolution the added operator relates reflections that " "disagree far beyond measurement error, i.e. it is not a real symmetry"; else if (std::isfinite(c.chi2) && std::isfinite(chi2_ref)) refused_why = "merge chi^2 is " + FormatDouble(c.chi2 / chi2_ref, 2) + "x the subgroup's (bound " + FormatDouble(opt.max_merge_chi2_ratio, 2) + ")"; else refused_why = "the merge under it is not self-consistent"; } continue; } c.eligible = true; if (c.order > best_pg_order || (c.order == best_pg_order && c.min_class_cc > best_pg_min_cc)) { best_pg = c.pg; best_pg_order = c.order; best_pg_min_cc = c.min_class_cc; } } for (const auto& [rk, s] : op_cache) result.operator_scores.push_back(s); std::sort(result.operator_scores.begin(), result.operator_scores.end(), [](const auto& a, const auto& b) { // cc is NaN for an operator PearsonCC could not score; a NaN key breaks // std::sort's ordering (out-of-bounds UB), so rank those last explicitly. if (std::isfinite(a.cc) != std::isfinite(b.cc)) return std::isfinite(a.cc); return a.cc > b.cc; }); // A caller that already decided the point group elsewhere overrides the choice here, keeping the // operator scores and the refusal report Stage A just produced. Nothing else is bypassed: Stage B // below judges the absences on THIS merge's reflections, which is the whole point of pinning. if (opt.fixed_point_group.has_value()) { const RotationSet want = RotationSetOf(*opt.fixed_point_group); for (const auto& pg : point_groups) if (pg.rotation_set == want) { best_pg = &pg; best_pg_order = static_cast(pg.rotations.size()) + 1; break; } } if (best_pg == nullptr) // should not happen (C1 always qualifies) return result; if (best_pg->representative) { result.point_group_hm = best_pg->representative->point_group_hm(); result.point_group_representative = *best_pg->representative; } result.point_group_order = best_pg_order; // The H ratio of the promotion that was ADOPTED, reported whether the bound had anything to say // about it or not. Read after the choice is final, so a fixed_point_group override reports the // ratio of the group it forced rather than of the one Stage A would have taken. result.global_best_operator_r = std::isfinite(global_best_r) ? global_best_r : std::numeric_limits::quiet_NaN(); for (const auto& c : pg_cands) if (c.pg == best_pg) { result.generated_point_group_adopted = c.closure; result.h_ratio = c.h_ratio; result.r_added = c.r_added; result.r_over_floor = c.r_over_floor; if (std::isfinite(c.r_added) && std::isfinite(global_best_r) && global_best_r > 0.0) result.r_over_best = c.r_added / global_best_r; } result.h_ratio_bound = opt.max_operator_h_ratio; result.r_over_best_bound = opt.max_operator_r_over_best; result.r_over_floor_bound = opt.max_operator_r_over_floor; // The FINALIST LEDGER: every operator-confirmed hypothesis with its evidence vector, adopted and // refused alike. Report-only - nothing below reads it back, and the choice above is already final. // The numbers are the ones the selection loop formed anyway; what is new is that the losers keep // theirs instead of being reduced to a single "refused" line. for (const auto& c : pg_cands) { PointGroupLedgerEntry e; e.point_group_hm = c.pg->representative ? c.pg->representative->point_group_hm() : "1"; e.order = c.order; e.min_class_cc = c.min_class_cc; e.chi2 = c.chi2; if (std::isfinite(c.chi2) && std::isfinite(chi2_ref) && chi2_ref > 0.0) e.chi2_over_best = c.chi2 / chi2_ref; e.b_extra = c.b_extra; if (c.parent_b_used > 1e-4) e.b_over_parent = c.b_extra / c.parent_b_used; e.h_ratio = c.h_ratio; e.r_added = c.r_added; if (std::isfinite(c.r_added) && std::isfinite(global_best_r) && global_best_r > 0.0) e.r_over_best = c.r_added / global_best_r; e.r_over_floor = c.r_over_floor; e.adopted = (c.pg == best_pg); e.eligible = c.eligible; e.refused_reason = c.why; result.point_group_ledger.push_back(e); } std::sort(result.point_group_ledger.begin(), result.point_group_ledger.end(), [](const PointGroupLedgerEntry& a, const PointGroupLedgerEntry& b) { return a.order > b.order; }); // Only report a refusal that is actually ABOVE what was adopted. if (refused_order > best_pg_order) { result.refused_point_group_hm = refused_pg_hm; result.refused_reason = refused_why; result.refused_point_group_representative = *refused_pg_rep; result.refused_by_twin_gate = refused_twin_gate; result.refused_twin_fraction = refused_fraction; } // --- Stage B: pick the space group within the point group --- // Without screw/centering determination, return the symmorphic representative. if (!opt.determine_space_group || best_pg->rotations.empty()) { if (best_pg->representative) result.best_space_group = *best_pg->representative; return result; } // The candidate space groups of the chosen point group. Scoring one is a pass over the whole // merge with three absence tests per reflection, and there are up to a dozen of them; they read // nothing but the shared reflection arrays, so give each one a thread. Every candidate's own pass // is unchanged and they are appended in table order, so the ranking below sees what it saw before. std::vector sg_cands; for (const auto& sg : gemmi::spacegroup_tables::main) if (sg.is_sohncke() && sg.is_reference_setting() && RotationSetOf(sg) == best_pg->rotation_set) sg_cands.push_back(&sg); // The non-reference SETTINGS of the same point group, when asked for - and unconditionally when // the point group is one only a non-reference setting carries (Stage A's second pass), since // otherwise there is no candidate at all and the group would be lost after being found. These are // alternative namings at the same order, so nothing here can promote the point group; what they // add is a screw or a centering on the axis the data show it on. Two refusals bound them: a // setting whose axes the cell does not have is not offered, and one predicting exactly the // absences a candidate already offered predicts is the same hypothesis under another name. if (opt.cell.has_value() && (opt.enumerate_all_settings || (opt.enumerate_all_rotation_sets && sg_cands.empty()))) { std::vector> signatures; for (const auto* c : sg_cands) signatures.push_back(AbsenceSignature(*c)); for (const auto& sg : gemmi::spacegroup_tables::main) { if (!sg.is_sohncke() || sg.is_reference_setting() || RotationSetOf(sg) != best_pg->rotation_set || !CellHostsRotations(*opt.cell, sg.operations())) continue; auto sig = AbsenceSignature(sg); if (std::find(signatures.begin(), signatures.end(), sig) != signatures.end()) continue; signatures.push_back(std::move(sig)); sg_cands.push_back(&sg); } } // The NON-SOHNCKE groups built on the same proper rotations - the small-molecule answer, the one // that can name a GLIDE plane. A protein crystal is built from L-amino acids and is therefore // chiral, so its group is Sohncke and holds only rotations and screws: no glide, no mirror, no // inversion centre. Nothing here can be right for a protein, which is why these candidates have // to earn their place against a bar measured on protein data rather than a bar chosen for them // (min_glide_evidence_per_reflection, measured 0/140 false positives). // // A candidate whose absences a Sohncke candidate already predicts is NOT offered. That single // line is what keeps the inversion centre out: P2/m predicts exactly what P2 predicts, P-1 // exactly what P1 predicts, and Friedel's law says no measurement on merged intensities can // separate them. So only a group that predicts something MORE - a glide's zonal absences - is // ever a candidate, and a centre of symmetry is never claimed. if (opt.enumerate_non_sohncke && best_pg != nullptr) { std::vector> signatures; for (const auto* c : sg_cands) signatures.push_back(AbsenceSignature(*c)); for (const auto& sg : gemmi::spacegroup_tables::main) { if (sg.is_sohncke() || ProperRotationSetOf(sg) != best_pg->rotation_set) continue; // Same precondition as a non-reference Sohncke setting, and for the same reason: a // setting names the mirror by AXIS, so P 1 2_1/c 1 and P 1 2_1/a 1 are different claims // about the same crystal and a cell that cannot host one must not be offered it. if (!opt.cell.has_value() || !CellHostsRotations(*opt.cell, sg.operations())) continue; auto sig = AbsenceSignature(sg); if (std::find(signatures.begin(), signatures.end(), sig) != signatures.end()) continue; signatures.push_back(std::move(sig)); sg_cands.push_back(&sg); } } std::vector sg_scored(sg_cands.size()); ParallelFor(static_cast(sg_cands.size()), std::min(opt.nthreads, sg_cands.size()), [&](int ci) { const gemmi::SpaceGroup& sg = *sg_cands[ci]; const gemmi::GroupOps gops = sg.operations(); SpaceGroupCandidateScore s{.space_group = sg}; double absent_sum = 0, present_sum = 0; int present_n = 0; // Judge centering and screw/glide absences on separate reflection sets. Lumping them lets // a large, correct centering-absent set hide a few strong screw violations and over-claim // screw axes (e.g. I4_132 on I432 data). int centering_absent = 0, centering_violations = 0; double centering_absent_sum = 0; // The same two classes as E^2, which is what AbsenceEvidence is stated in: the centring-absent // class and the present class that is its control. double centering_absent_esq = 0, present_esq_sum = 0; int present_strong = 0; // A screw axis extinguishes only the reflections that lie ON it, so its absent class and the // rest of that same axial row are collected apart from the general reflections and judged // against each other, ROW BY ROW (see screw_e_squared below). // e_squared_expected is e_squared with the pseudo-translation's modulation divided out. BOTH // the zone's evidence and the violation test read it, and for the same reason: each compares // an absent reflection against a scale taken from the rest of its own row, and under a // half-integer pseudo-translation the rest of the row is the ENHANCED class while the absent // class is the suppressed one. The evidence divides by the row MEAN and the violation test // thresholds on the row MEDIAN, but the double-counting is the same in both. De-modulating is // not an invented measurement - it is the measurement restated on the scale it is judged // against. Measured on an orthorhombic crystal with a pseudo-body-centring: the false c zone // goes from 4 violations in 48 (under the 10% bound, so the screw passed) to 18 in 48, while // the crystal's two genuine screws gain NOT ONE violation, and neither does a genuine screw // on another crystal whose confirmed pseudo-translation sits on that very axis. struct ScrewAbsent { AxialRow row; double e_squared; double e_squared_expected; double i_over_sigma; }; std::vector screw_absent_refl; std::map> row_present_esq; // The candidate's mirror directions, and the zone of each: the reflections it fixes, split // into the ones the group extinguishes and the rest, which are their control. Empty for a // Sohncke candidate, so everything glide-related below costs one empty-vector test on the // path protein data takes. const std::vector mirrors = MirrorDirections(gops); struct GlideAbsent { int zone; double e_squared; double e_squared_expected; double i_over_sigma; }; std::vector glide_absent_refl; std::vector> zone_present_esq(mirrors.size()); for (size_t i = 0; i < n; ++i) { if (!pass_absence[i]) continue; const gemmi::Op::Miller hkl{{H[i], K[i], L[i]}}; // Present := statistically significant AND intensity-significant. The E^2 gate keeps a // weak axial reflection with an under-estimated sigma (fake high I/sigma) from faking a // screw-axis violation; it only relaxes "present", so it cannot over-call a screw whose // predicted-absent class carries real intensity. // // present_cut, not the fixed cut: on a merge weak enough that nothing clears the fixed cut, // screw_violations is identically zero, so every screw axis passes unchallenged, and // present_strong is zero, so the centering rescue below switches itself off on exactly the // weak data it exists for. present_cut is the same cut converted to the counting scale the // two stages share (see merge_isa); on a healthy merge it is the fixed cut to within 1%. const bool present = IoverSigma[i] > present_cut && (opt.present_e_squared <= 0.0 || Esq[i] > opt.present_e_squared); if (CenteringAbsent(gops, hkl)) { s.absent_observed += 1; absent_sum += IoverSigma[i]; centering_absent += 1; centering_absent_sum += IoverSigma[i]; centering_absent_esq += std::max(0.0, Esq[i]); if (present) { s.absent_violations += 1; centering_violations += 1; } } else if (gops.is_systematically_absent(hkl)) { s.absent_observed += 1; absent_sum += IoverSigma[i]; // A zonal absence belongs to the GLIDE that causes it and an axial one to the SCREW; // MirrorZoneOf returns -1 for the axial rows, so neither test ever sees the other's // reflections and no reflection is scored twice. const int mz = MirrorZoneOf(mirrors, hkl); if (mz >= 0) glide_absent_refl.push_back({mz, Esq[i], Esq[i] / zone_tncs_expected[i], IoverSigma[i]}); else screw_absent_refl.push_back({RowOf(H[i], K[i], L[i]), Esq[i], Esq[i] / tncs_expected[i], IoverSigma[i]}); } else { present_n += 1; present_sum += IoverSigma[i]; present_esq_sum += std::max(0.0, Esq[i]); if (present) present_strong += 1; // A non-identity rotation of the group maps this reflection to itself, i.e. it lies on // a rotation axis - the control class for that axis's screw absences. if (gops.epsilon_factor_without_centering(hkl) > 1) row_present_esq[RowOf(H[i], K[i], L[i])].push_back(Esq[i]); // ... and the control class of a GLIDE zone: the reflections of the same plane the // group does NOT extinguish. Raw E^2, as the screw row's control is: the modulation // correction divides the absent class by the scale it is judged against, and the // control class IS that scale. const int mz = MirrorZoneOf(mirrors, hkl); if (mz >= 0) zone_present_esq[mz].push_back(Esq[i]); } } // "Too strong to be systematically absent" is judged, for a SCREW, against the axial row the // screw constrains rather than against the shell mean over all reflections. An axial row can be // far stronger than an average reflection, and (shell) falls with resolution while a // systematically-absent reflection keeps a small non-decaying residual (background / profile // leakage) - so at high resolution the plain E^2 cut turns those residuals into violations even // though the reflections next to them in the same row are tens of times stronger. That cost a // real P4_1 2_1 2 crystal its 4_1: 18 of its 47 absent 00l crossed the cut, all beyond 3.7 A, // at 1-2% of the l=4n reflections beside them. Scaling by the row's own median E^2 removes the // resolution dependence; the scale is floored at 1 so a row weaker than average keeps the plain // cut, which makes this a rescue only - a screw can be recovered by it, never lost. // // Row by row, not pooled: a 4_1 along c and a 2_1 along a are separate conditions with separate // control rows, and on the same crystal one row can be 20x an average reflection while another // is half of one. Pooling them lets the weak rows set the threshold for the strong one and the // rescue never fires (that very crystal pooled to a row median of 0.7 and stayed at P42_12). // A row needs more than a reflection or two behind its median before that median may set the // threshold. The scale only ever RAISES the cut, so a row whose control class holds one strong // reflection would license a screw claim the row does not support - a genuine 4_2 whose 00l are // observed only at l=4n, strongly, would read its l=4n+2 reflections as absent and be reported as // 4_1/4_3. Below this count the row falls back to the plain cut, i.e. no rescue. constexpr size_t MIN_ROW_CONTROL_REFLECTIONS = 3; // That count guards the row MEDIAN, which sets a violation THRESHOLD. The row MEAN is a // different thing - it is the scale the zone's evidence is stated in, and a scale needs a // number to divide by, not a middle. Holding both to the same count made a candidate FORFEIT // a zone by being right: a screw predicts more of its own row absent, which leaves fewer // reflections in its control, and below three the zone was declared undetermined and its // evidence discarded. Measured on a tetragonal 4_1/4_3 wedge: thirteen 00l reflections at // 0.1% of the two l = 4n beside them scored ZERO, while the nine of them a 4_2 also predicts // absent, judged against a control that still held four dead l = 4n+2, scored 38.7 - so the // group that explains the row lost to one that explains part of it. // A mean of ONE number is still a scale to divide by, and on a crystal mounted along one of // its own axes the row can never hold more: an axial row near the spindle sits inside the // blind cone and is recordable only out to |q| = 2 sin(alpha)/lambda however long the sweep // runs. Holding the scale to two therefore measured the mounting, not the evidence - // measured on an orthorhombic wedge whose h00 held exactly 200 and 300, the 300 at 5e-5 of // the 200, and the zone scored ZERO. What a one-control zone can earn is capped by // MIN_U_PER_ABSENT_REFLECTION at 6.9 nats per absent reflection, well under // min_screw_absence_evidence, so it can add to a screw confirmed elsewhere but cannot // license one alone; the violation THRESHOLD still answers to MIN_ROW_CONTROL_REFLECTIONS. constexpr size_t MIN_ROW_CONTROL_FOR_SCALE = 1; std::map row_median, row_mean; for (auto& [row, esq] : row_present_esq) { if (esq.size() >= MIN_ROW_CONTROL_FOR_SCALE) row_mean[row] = std::accumulate(esq.begin(), esq.end(), 0.0) / esq.size(); if (esq.size() >= MIN_ROW_CONTROL_REFLECTIONS) row_median[row] = MedianOf(esq); } const int screw_absent = static_cast(screw_absent_refl.size()); int screw_violations = 0; // The absent intensities in units of their own row's control mean, kept ZONE BY ZONE - one // axial row is one screw condition, and the rows are measured to wildly different depths. A // row with no control class of its own cannot say whether its absences are weak or its whole // row is, so it contributes no evidence either way rather than being judged against the shell // mean; it is carried as an undetermined zone so the report can say so. struct ZoneSums { double sum_u = 0; double max_u = 0; int n_absent = 0; int n_control = 0; int n_violations = 0; }; std::map zones; for (const auto& a : screw_absent_refl) { const auto it = row_median.find(a.row); const double row_scale = it == row_median.end() ? 1.0 : std::max(1.0, it->second); // present_cut for the same reason as the general "present" above - with the fixed cut this // test cannot fire at all on a low-ISa merge. The row-relative E^2 gate is the other half of // the test and is unchanged, so an absence still has to look strong against its OWN axial row // to count as a violation. // LIMIT - the de-modulated E^2 is read here, but the bound this count is compared // against (opt.max_absent_violation_fraction, applied further down) was CALIBRATED ON // THE UNCORRECTED COUNT. Correcting a zone can therefore take it over a bound that was // never chosen with the correction in mind. On the crystal the correction was built for // that crossing IS the verdict: the corrected count crosses the bound while the corrected // evidence stays tens of nats positive - a residue of the correction's own under-measure, // see the deferral below - so there the count is the one instrument still standing. auto& z = zones[a.row]; z.n_absent += 1; if (a.i_over_sigma > present_cut && (opt.present_e_squared <= 0.0 || a.e_squared_expected > opt.present_e_squared * row_scale)) { ++screw_violations; // Per zone as well: the count-veto deferral below is licensed zone by zone. z.n_violations += 1; } const auto mit = row_mean.find(a.row); if (mit != row_mean.end() && mit->second > 0.0) { const double u = std::max(0.0, a.e_squared_expected) / mit->second; z.sum_u += u; z.max_u = std::max(z.max_u, u); z.n_control = static_cast(row_present_esq.at(a.row).size()); } } // Each zone is scored on its OWN evidence, and the strongest one speaks for the group. Pooling // the rows into one verdict lets a row the sweep barely sampled outweigh one that is decisively // dead - measured: a crystal whose h zone is confirmed on its own row and whose k zone is noise // pooled to a refusal and came out in the symmorphic group, with the confirmed screw lost and // nothing said about it. The pooled value is kept as a floor so several jointly-decisive zones // still count together, which is the case pooling gets right. double screw_sum_u = 0.0; int screw_scored = 0; double best_zone_evidence = 0.0; // Whether the violation-count veto may be DEFERRED to the evidence (the third arm of // screw_ok below). The evidence that clears the claim bar is a group-level number, but the // count it overrides indicts specific zones, so an overwhelming genuine zone can pay a // different zone's debts: measured, a genuine four-fold screw row at 194 nats lifted the // veto for two claimed two-fold rows whose own zones read 9 nats each with HALF their class // individually present - a pseudo-translation near (1/2,1/2,1/2), too weak for // FindRowModulation to see (cone ratio ~1.0 in every shell), held those rows at ~0.2 of // their controls without extinguishing them. So each zone that carries violations must hold // the deferral on its own, and it holds it only if // - it is dead PER REFLECTION: at least a nat per predicted-absent reflection, i.e. the // class at most e^-2 ~ 0.14 of its own row. The same sum-ranks/rate-licenses reasoning // as the glide bar below: the sum is linear in the count, so a long, merely-suppressed // row accumulates past any fixed bar (42 absences at a fifth of their row sum to 28 // nats), while the rate is bounded. Measured on the deferrals that were right against // the two that were wrong: 1.4-1.8 nats per reflection against 0.4-0.7. // // A measured pseudo-translation on the row does NOT revoke the deferral on its own, and used // to. It is already paid for, quantitatively: e_squared_expected divides the absent class by // the depth the modulation puts on it before either the evidence or the violation count reads // it, which is the same measurement restated on the scale it is judged against. Revoking the // deferral as well charges one measurement twice, and charges it categorically - the screw // claim then turns on which side of a fixed depth the estimator lands, and that estimate moves // with anything upstream that changes which reflections enter the cone. Measured: masking a // beam-stop arm - 2.3% of the detector, which IMPROVED the merge in every other respect - took // one row's depth from 0.68 to 0.37 across the bound and cost an orthorhombic crystal both of // its screws, on a half-integer vector its own Patterson does not carry, while the zone // evidence rose. Where a modulation really does hide a row, the de-modulated evidence per // reflection falls and the rate above refuses on its own. constexpr double SCREW_DEFERRAL_MIN_EVIDENCE_PER_ABSENT = 1.0; bool screw_deferral_ok = true; for (const auto& [row, z] : zones) { const double sum_u = TrimmedZoneSum(z.sum_u, z.max_u, z.n_absent); const double ev = z.n_control > 0 ? ScrewZoneEvidence(sum_u, z.n_absent) : 0.0; s.screw_zones.push_back({row, z.n_absent, z.n_control, ev}); if (z.n_control > 0) { screw_sum_u += sum_u; screw_scored += z.n_absent; best_zone_evidence = std::max(best_zone_evidence, ev); } if (z.n_violations > 0 && ev < SCREW_DEFERRAL_MIN_EVIDENCE_PER_ABSENT * z.n_absent) screw_deferral_ok = false; } // ---- Glide zones ------------------------------------------------------------------------ // The same statistic on a plane instead of a row: the absent class in units of its own // zone's control mean, through ScrewZoneEvidence. What differs is the SCALE it is read on. // ScrewZoneEvidence is a sum over reflections, so it is linear in the size of the class at // fixed deadness. A screw row holds tens of reflections; a glide zone holds hundreds to // thousands, so the sum reaches values on a merely-weak zone that no genuine screw row could // reach on a dead one, and the two are not on one scale. Measured over 140 protein datasets: // the largest false zone sums to 331 nats against 422 for the smallest genuine glide in the // corpus - a factor of 1.3, which is not a separation. Per reflection those same two read // 0.65 and 5.95. So the sum is what RANKS candidates (where every candidate is charged the // same way and the linearity cancels) and the RATE is what licenses the claim at all. { struct ZoneSums { double sum_u = 0; int n_absent = 0; int n_control = 0; }; std::vector zsum(mirrors.size()); std::vector zone_mean(mirrors.size(), 0.0), zone_median(mirrors.size(), 0.0); for (size_t m = 0; m < mirrors.size(); ++m) { auto& c = zone_present_esq[m]; if (c.empty()) continue; zone_mean[m] = std::accumulate(c.begin(), c.end(), 0.0) / c.size(); zone_median[m] = MedianOf(c); } int glide_violations = 0; for (const auto& a : glide_absent_refl) { auto& z = zsum[a.zone]; z.n_absent += 1; // n_control inside the guard, exactly as the screw row does it: a zone whose control // class has a non-positive mean E^2 - the whole plane is noise, half its intensities // negative - has no scale to judge its absences against, so it is UNDETERMINED and not // "measured at zero". Setting the count outside the guard was a real defect and it is // the count-linearity hazard in its worst form: sum_u stays 0, ScrewZoneEvidence // returns its floor value, and a zone of 1290 noise reflections is credited 7625 nats // of evidence for a glide nothing measured. Seen on a protein merge whose whole h0l // plane sits at ~ 1. if (zone_mean[a.zone] > 0.0) { z.sum_u += std::max(0.0, a.e_squared_expected) / zone_mean[a.zone]; z.n_control = static_cast(zone_present_esq[a.zone].size()); } // A violation is judged against the zone's OWN median, floored at 1, exactly as a // screw's is judged against its row's - a zone can be far stronger or far weaker than // an average reflection, and the floor makes the scaling a rescue only. const double scale = std::max(1.0, zone_median[a.zone]); if (a.i_over_sigma > present_cut && (opt.present_e_squared <= 0.0 || a.e_squared_expected > opt.present_e_squared * scale)) ++glide_violations; } constexpr int MIN_ZONE_CONTROL = 5; double weakest = std::numeric_limits::infinity(); for (size_t m = 0; m < mirrors.size(); ++m) { const auto& z = zsum[m]; if (z.n_absent == 0) continue; // a plain MIRROR: it extinguishes nothing, so there is nothing to judge const double ev = z.n_control > 0 ? ScrewZoneEvidence(z.sum_u, z.n_absent) : 0.0; const double per = ev / z.n_absent; const double ubar = z.n_control > 0 ? z.sum_u / z.n_absent : 0.0; s.glide_zones.push_back({MirrorZoneLabel(mirrors[m]), z.n_absent, z.n_control, ev, per, ubar}); s.glide_absent += z.n_absent; // A zone with no control class, too few absences, or too small a control refuses // the candidate. Note what the rate CANNOT do, which is why it is the gate: it is // bounded above by -log(MIN_U_PER_ABSENT_REFLECTION) - 1 = 5.91 nats however many // reflections the zone holds, so a count can inflate the SUM without limit but never // the rate. The bar at 2.0 sits well under that ceiling. if (z.n_control < MIN_ZONE_CONTROL || z.n_absent < opt.min_glide_absent) s.glide_unmeasurable = true; else weakest = std::min(weakest, per); } s.glide_absence_evidence = std::isfinite(weakest) ? weakest : 0.0; s.absent_violations += glide_violations; } s.absent_violations += screw_violations; s.centering_absent = centering_absent; // A centred group whose centering-absent class is EMPTY was not tested, it was skipped: the // reflections that would decide it are not in this merge (see centering_untested). s.centering_untested = sg.centring_type() != 'P' && centering_absent == 0; s.screw_absent = screw_absent; s.screw_absence_evidence = std::max(best_zone_evidence, ScrewZoneEvidence(screw_sum_u, screw_scored)); // Of the three axial rows, the ones this group's screws extinguish while the merge holds no // control class to judge them - claims the data can neither confirm nor refuse. Only a // SCREW claim counts: in a centred group the axial conditions are the centring's own // (I2_12_12_1 extinguishes h00 with h odd because h+k+l is odd, not because of the screw), // so counting those would let a candidate whose centring this merge never tested outrank // the primitive group it ties with - centred pairs like I222 / I2_12_12_1 rightly stay // tied, since no absence anywhere separates them. Counted only when at least one zone WAS // judged; consumed by the tie-break in the sort below. bool any_zone_judged = false; for (const auto& z : s.screw_zones) if (z.n_control > 0) any_zone_judged = true; if (any_zone_judged) { for (const AxialRow& row : {AxialRow{1, 0, 0}, AxialRow{0, 1, 0}, AxialRow{0, 0, 1}}) { if (!gops.is_systematically_absent(row) || CenteringAbsent(gops, row)) continue; bool judged = false; for (const auto& z : s.screw_zones) if (z.row == row && z.n_control > 0) judged = true; if (!judged) s.unjudged_screw_claims += 1; } } if (s.absent_observed > 0) s.absent_mean_i_over_sigma = absent_sum / s.absent_observed; if (present_n > 0) s.present_mean_i_over_sigma = present_sum / present_n; // The centering-absent class gets the same likelihood a screw zone gets, with the present class // as its control: how unlikely that class would be if the centering did not exist. A COUNT of // net absences cannot separate a true centering from a super-centering that swallows it, because // the super-centering predicts every true absence PLUS a block of present reflections and can // still net the larger count - measured, an F222 candidate with a third of its "absent" class as // strong as the present one out-counted the C222_1 whose class was at 0.4% of it. Under the // likelihood those two are orders of magnitude apart, on the same numbers, with no bound to tune. const double present_esq_mean = present_n > 0 ? present_esq_sum / present_n : 0.0; if (centering_absent > 0 && present_esq_mean > 0.0) s.centering_absence_evidence = AbsenceEvidence(centering_absent_esq / present_esq_mean, centering_absent, present_n); // Centering is judged by class STRENGTH, not a per-reflection violation count. A real centering // cancels structure factors, so its absent class is systematically weak - its mean signed // I/sigma sits well below the present class - regardless of noise or obverse/reverse twinning; // a false centering leaves the "absent" class as strong as the present one (mean ratio ~1). The // count-of-strong-violations gate is brittle on noisy/twinned data, where enough genuinely-absent // reflections randomly clear I/sigma>3 to trip the 10% bound though the class is 3-4x weaker (a // true R3 at 13.5% violations, absent 1.7 vs present 6.0). The mean is well-determined here // because a centering-absent class holds a third-to-half of all reflections. Screws keep the // count gate: their predicted-absent class is a handful of axial reflections, too few to average // - what they get instead is a row-relative threshold for counting a violation at all. const double present_mean = present_n > 0 ? present_sum / present_n : 0.0; const double centering_absent_mean = centering_absent > 0 ? centering_absent_sum / centering_absent : 0.0; // The centering-absent class proves itself weak in either of two floor-independent ways; a // FALSE centering (absent as strong as present) fails both: // (1) mean signed I/sigma well below the present class, OR // (2) its strong-reflection RATE well below the present class's own strong rate. // (2) is needed because weak / low-energy data carry a positive intensity floor (background / // profile leakage) that lifts abs to ~1.5-2.3 even for genuinely extinct reflections; when // the present class is itself weak (small present_mean) that additive floor inflates the mean // ratio past the bound and hides a real centering - e.g. an I-centred cubic crystal at low // energy, whose true I-centering sat at ratio ~0.57. Normalising the violation count by the // present class's own strong rate cancels the shared floor and stays reliable on weak data // (both rates shrink together). const double present_strong_rate = present_n > 0 ? static_cast(present_strong) / present_n : 0.0; const double centering_violation_rate = centering_absent > 0 ? static_cast(centering_violations) / centering_absent : 0.0; // The mean-ratio test only means anything while the present class carries signal: with a // present mean at or below zero the bound is non-positive and the comparison turns on the sign // of the absent mean rather than on its size, accepting or rejecting a centering by accident. // Leave that case to the rate test below, which counts violations and cannot change sign. const bool centering_ok = centering_absent == 0 || (present_n > 0 && present_mean > 0.0 && centering_absent_mean <= opt.max_absent_present_ratio * present_mean) || (present_strong_rate > 0.0 && centering_violation_rate <= opt.max_absent_present_ratio * present_strong_rate); const bool screw_ok = screw_absent == 0 || screw_violations <= opt.max_absent_violation_fraction * screw_absent // A class whose absence evidence already clears the claim bar cannot also be vetoed by // the raw violation count. The likelihood has priced those same reflections in - they // are inside sum_u - and still reads the class as extinct, so the count adds nothing it // has not already weighed, and adds it unweighted: a strong axial row near the spindle // carries rocking-curve tails that read as present at several sigma without being // structure factors, and a handful of them veto a class the evidence puts tens of nats // clear. Licensed zone by zone (screw_deferral_ok above): the deferral holds only where // the zones carrying the violations are themselves dead per reflection - on a zone // subsidised by another row's evidence the count is the only independent check left. || (s.screw_absence_evidence >= opt.min_screw_absence_evidence && screw_deferral_ok); s.consistent = centering_ok && screw_ok; sg_scored[ci] = std::move(s); }); for (auto& s : sg_scored) result.candidates.push_back(std::move(s)); // A candidate is eligible when its absences are confirmed and there are enough of them to // trust (the symmorphic group, with no absences, is always eligible as the fallback). Rank // eligible candidates by the EVIDENCE their predicted absences carry, centering and screws // together - both are -log Beta tails of an absent class against its own control, so they are in // the same units (nats) and add as independent evidence does. A count, even netted against the // violations, is the wrong scale: a false super-centering predicts every true absence PLUS a block // of present reflections, so it can net the larger count while a third of its class is as strong as // the present one. The likelihood collapses for exactly that class and grows for a genuine one, so // a real F222 still beats C222_1 on F-centred data and no bound has to be tuned to tell them apart. // The screw part is the SUM over the zones, not the group's single gating number: each axial row // is a separate condition tested on its own reflections, so the log-likelihoods add. Summing is // what makes an extra condition pay its own way - a group claiming one screw more gains that // zone's evidence when the row really is dead and gains little when the row is not, where a count // could only ever go up. (It gains LITTLE, not nothing: a zone whose absences were never // measurable still scores a bounded positive value, see MIN_U_PER_ABSENT_REFLECTION. Requiring a // minimum number of absences per zone before it may contribute, as the centring class does // through min_absent_observed, would close that too - it needs its own battery.) Pooling the rows into one Beta instead would make three genuine screws // read as weaker evidence than two whenever the third row is measured less deeply, which is a // property of the pooling, not of the crystal. auto absence_evidence = [](const SpaceGroupCandidateScore& s) { double total = s.centering_absence_evidence; for (const auto& z : s.screw_zones) if (z.n_control > 0) total += z.evidence; // Glide zones add on the same terms as screw rows: an extra claimed condition gains its // zone's evidence when the plane really is dead and is charged for it when it is not. A zone // as strong as its control reads about -n_absent, so a candidate that claims a glide the data // do not show loses hundreds of nats to the one that does not claim it - which is why a false // glide cannot out-rank the Sohncke group even before the eligibility bar is applied. for (const auto& z : s.glide_zones) if (z.n_control > 0) total += z.evidence; return total; }; // A candidate also needs enough EVIDENCE behind its absences before they may be claimed, and the two // kinds of absence need different measures of it. A centering class is a third to a half of every // reflection in the data set, so a count is a fair measure and min_absent_observed is never the // binding constraint. A screw class is a handful of axial reflections BY CONSTRUCTION - one row of // reciprocal space, often lying near the spindle where a rotation sweep records least - and there a // count measures the sweep's geometry, not the evidence: six axial reflections measured at zero // against a row that averages 1.4x the shell mean settle the question, while twenty uniformly weak // ones settle nothing. So the screw class is judged by its zones' AbsenceEvidence instead. // A candidate in a non-reference setting must also have had its CENTERING tested here. The // reference-setting path can adopt an untested centering because something else backs it - the // caller's centred-lattice re-test reindexes the data into the metric candidate's conventional // cell and only then commits - and a non-reference setting has no such backing: adopting one on // an absence class this merge does not contain would name a lattice the data never showed, which // is worse than the group it displaces. Refused, not warned about. auto eligible = [&](const SpaceGroupCandidateScore& s) { return s.consistent && !(s.centering_untested && !s.space_group.is_reference_setting()) && (s.centering_absent == 0 || s.centering_absent >= opt.min_absent_observed) && (s.screw_absent == 0 || s.screw_absence_evidence >= opt.min_screw_absence_evidence) // Every glide the candidate claims must be measurable here AND its zone must clear the // per-reflection bar. The WEAKEST zone decides (glide_absence_evidence is the minimum), // so a group claiming two glides needs both planes dead; and a zone this merge cannot // judge refuses the candidate rather than abstaining, because a glide is an extra claim // on top of a group that already fits without it. && (s.glide_absent == 0 || (!s.glide_unmeasurable && s.glide_absence_evidence >= opt.min_glide_evidence_per_reflection)); }; std::sort(result.candidates.begin(), result.candidates.end(), [&](const SpaceGroupCandidateScore& a, const SpaceGroupCandidateScore& b) { if (eligible(a) != eligible(b)) return eligible(a); if (absence_evidence(a) != absence_evidence(b)) return absence_evidence(a) > absence_evidence(b); if (a.absent_violations != b.absent_violations) return a.absent_violations < b.absent_violations; // prefer the honest, less over-claiming group // On a tie, the SOHNCKE group. Reaching here means the two predict the same // absences with the same violations, so what separates them is an element the // intensities cannot see, and the group that does not claim it is the one to // report. In practice the enumeration above has already dropped such a candidate; // this is the same rule stated where the ordering can be read. if (a.space_group.is_sohncke() != b.space_group.is_sohncke()) return a.space_group.is_sohncke(); // Still tied, and both have shown a screw on a row this merge DID measure: what // separates them is a further axial row with no control class here, where the // data say nothing either way. Prefer the group that claims it. This is a PRIOR, // not a measurement, and it is stated as one: a crystal that has already shown // one 2_1 is P2_12_12_1 an order of magnitude more often than P2_12_12 or P222_1, // and an orthorhombic sweep short of half a turn routinely records one axial row // and not the next (measured on consecutive 50 deg wedges of one crystal: the // first sees 00l and no 0k0, the second 0k0 and no 00l). The alternatives are // reported unchanged - only the representative moves. if (a.unjudged_screw_claims != b.unjudged_screw_claims) return a.unjudged_screw_claims > b.unjudged_screw_claims; // Genuinely indistinguishable (e.g. I23 vs I2_13, or an enantiomorphic pair): lower // space-group number is the representative. return a.space_group.number < b.space_group.number; }); if (!result.candidates.empty() && eligible(result.candidates.front())) { // Alternatives are the candidates with the same absence EVIDENCE and the same violation count // as the winner: the enantiomorphic / origin-ambiguous partners the data truly cannot separate. // A super-centering that nets the same count but over-claims scores different evidence, and is // therefore still not reported as an equal alternative. // // Keyed on the evidence and NOT on the absent COUNT. Two settings of one point group routinely // predict a different number of absences on a zone the sweep never measured - one extra // reflection on a row with no control class of its own, contributing nothing either way - and // excluding a candidate for that one reflection hides an answer these data support EXACTLY as // well. Measured on an orthorhombic wedge: four candidates at 100.7 nats with zero violations, // of which the two predicting one more (unmeasured) absence were dropped, and one of those was // the right answer. The winner itself is unaffected - it is candidates.front() either way, so // this only ever widens what the run REPORTS as indistinguishable. const double sel_evidence = absence_evidence(result.candidates.front()); const int sel_violations = result.candidates.front().absent_violations; for (auto& s : result.candidates) { if (!eligible(s) || s.absent_violations != sel_violations || std::fabs(absence_evidence(s) - sel_evidence) > 1e-9 * std::max(1.0, std::fabs(sel_evidence))) continue; s.selected = true; if (!result.best_space_group.has_value()) result.best_space_group = s.space_group; // representative (lowest number) else result.alternatives.push_back(s.space_group); } } // Which axis, if any, these data could not decide the screw on. The ambiguity itself is already // in `alternatives` - the selected set holds groups predicting different axial absences - but a // list of names does not say WHICH axis is open, nor why, and the scalar best_space_group reads // as a determination either way. So name it per row: a row is undetermined when two SELECTED // candidates disagree about whether it carries screw absences at all, which is exactly the case // where nothing in this merge separated them. An enantiomorphic or origin-ambiguous pair // (P4_1/P4_3, I222/I2_12_12_1) predicts the SAME absences on every row and is not named here - // that ambiguity is the hand, or the origin, and is reported as itself. if (result.best_space_group.has_value()) { for (const AxialRow& row : {RowOf(1, 0, 0), RowOf(0, 1, 0), RowOf(0, 0, 1)}) { const gemmi::Op::Miller hkl{{row[0], row[1], row[2]}}; bool claimed = false, unclaimed = false; int n_absent = 0, n_control = 0; for (const auto& s : result.candidates) { if (!s.selected) continue; const gemmi::GroupOps g = s.space_group.operations(); // The row's absences must come from a SCREW: in a centred group the axial condition // is the centring's own (I222 extinguishes h00 with h odd because h+k+l is odd), and // that is decided elsewhere, on the whole centring class. if (g.is_systematically_absent(hkl) && !CenteringAbsent(g, hkl)) { claimed = true; for (const auto& z : s.screw_zones) if (z.row == row) { n_absent = std::max(n_absent, z.n_absent); n_control = std::max(n_control, z.n_control); } } else { unclaimed = true; } } if (claimed && unclaimed) { const auto it = axial_row_observed.find(row); result.undetermined_screws.push_back( {row, RowLabel(row), AxisOf(row), it == axial_row_observed.end() ? 0 : it->second, n_absent, n_control}); } } } // Both answers, on every run. The winner above may be a non-Sohncke group, which is right for a // small molecule and impossible for a protein; a user who knows their sample is chiral has to be // able to read the Sohncke answer off the same run, and a user who is not sure has to be able to // see what the glide bought and how dead the zone behind it was. Neither is a decision - the // decision is best_space_group - they are the same ranking read at two places. for (const auto& s : result.candidates) if (eligible(s) && s.space_group.is_sohncke() && !result.sohncke_space_group.has_value()) result.sohncke_space_group = s.space_group; // ...and the glide answer only when it WON. Not "the best eligible non-Sohncke candidate": a // primitive candidate scored against CENTRED data can carry a perfectly dead glide zone that is // the centring seen edge-on - C-centred data extinguish h0l with h odd, which is exactly what // the a-glide of P 1 2_1/a 1 predicts, and that zone reads 5.9 nats/reflection because it really // is dead. The ranking settles it (measured on synthetic C 1 2 1 data: 13348 nats of centring // evidence for C2 against 91 for the glide candidate), so the field says what was ADOPTED and // the candidate table below still shows the hypothesis that lost. if (result.best_space_group.has_value() && !result.best_space_group->is_sohncke()) for (const auto& s : result.candidates) if (s.selected && !s.space_group.is_sohncke()) { result.glide_space_group = s.space_group; result.glide_zones = s.glide_zones; break; } return result; } static std::string LedgerCell(double v, int prec) { return std::isfinite(v) ? FormatDouble(v, prec) : std::string("-"); } std::string FinalistLedgerToText(const SearchSpaceGroupResult& result) { std::ostringstream os; os << "Finalist ledger - every point group whose operators the intensities confirmed on this " "merge,\n adopted and refused alike. REPORT-ONLY: the group this run processed in was " "chosen\n without any of it, and nothing below reads it back.\n"; // A search that confirmed no operator has no finalists, and an empty table under a header reads as // "nothing was wrong" when it means "nothing was asked". Say which it is, and stop. int confirmed_above_p1 = 0; for (const auto& e : result.point_group_ledger) if (e.order > 1) ++confirmed_above_p1; if (result.point_group_ledger.empty()) { os << " No finalists: the search confirmed no point-group operator on this merge, so there is " "no\n hypothesis to tabulate. This is the expected table for a crystal that is genuinely " "P1,\n and for one whose operators the correlation stage never confirmed.\n"; return os.str(); } if (confirmed_above_p1 == 0) { os << " No finalists above P1: the only row is the trivial group, so the ledger has nothing to " "weigh\n against anything. The table follows for completeness.\n"; } if (std::isfinite(result.merge_r_floor)) { os << " Reference R floor (random noise, half-set) " << FormatDouble(result.merge_r_floor, 4) << "; best-agreeing operator R " << LedgerCell(result.global_best_operator_r, 4); // On the FIRST step out of P1 there is no second operator to be the reference, so this ratio - // not any row below - carries the whole of the evidence. Spell it out rather than leaving the // reader to divide two numbers, because that is the case where the table itself is empty. if (std::isfinite(result.global_best_operator_r) && result.merge_r_floor > 0.0) os << " (" << FormatDouble(result.global_best_operator_r / result.merge_r_floor, 2) << "x the floor, bound " << FormatDouble(result.r_over_floor_bound, 2) << ")"; os << ".\n"; } else os << " Best-agreeing operator R " << LedgerCell(result.global_best_operator_r, 4) << " (no half-set intensities in this merge, so no random-noise floor;\n every R/floor " "reads '-' and the first step out of P1 is not measurable here).\n"; os << "\n PG ord minCC R_add R/best R/floor | H chi2/best b b/par verdict\n" " <-- calibrated evidence --> | <--- diagnostic only, NOT evidence --->\n"; for (const auto& e : result.point_group_ledger) { os << " " << std::left << std::setw(8) << e.point_group_hm << std::right << std::setw(3) << e.order << std::setw(8) << LedgerCell(e.min_class_cc, 3) << std::setw(8) << LedgerCell(e.r_added, 4) << std::setw(7) << LedgerCell(e.r_over_best, 2) << std::setw(8) << LedgerCell(e.r_over_floor, 2) << " |" << std::setw(6) << LedgerCell(e.h_ratio, 2) << std::setw(10) << LedgerCell(e.chi2_over_best, 2) << std::setw(9) << LedgerCell(e.b_extra, 4) << std::setw(7) << LedgerCell(e.b_over_parent, 2) << " " << (e.adopted ? "ADOPTED" : e.eligible ? "eligible" : "refused") << (e.refused_reason.empty() ? "" : " (" + e.refused_reason + ")"); // The margin, on the row that decided. The value alone does not say how close the run came to // being refused, and on this corpus that distance is small enough to matter (below). if (e.adopted && std::isfinite(e.r_over_best) && result.r_over_best_bound > 0.0) os << " [" << FormatDouble(result.r_over_best_bound - e.r_over_best, 2) << " below the " << FormatDouble(result.r_over_best_bound, 2) << " bound]"; else if (e.adopted && std::isfinite(e.r_over_floor) && result.r_over_floor_bound > 0.0) os << " [" << FormatDouble(result.r_over_floor_bound - e.r_over_floor, 2) << " below the " << FormatDouble(result.r_over_floor_bound, 2) << " floor bound]"; os << "\n"; } // How to read it. Every sentence here is something the numbers above have been measured to NOT // support, and each was a wrong reading somebody actually made. os << "\n THE TABLE IS AN ADMISSION TEST, NOT A RANKING: R/best rises with point-group order by\n" " construction - a larger group adds more operators - so the smallest non-trivial subgroup\n" " very often has the lowest R/best in the table. Sorting these rows and taking the minimum\n" " would demote nearly every genuinely high-symmetry crystal. The question the table answers\n" " is 'the largest order whose R/best stays near 1', not 'which row is smallest'.\n" "\n ONLY R/best AND R/floor ARE EVIDENCE. They are ratios to a reference a false hypothesis\n" " cannot move: the best-agreeing operator anywhere in the data, and the crystal's own\n" " half-set noise floor. H, chi2/best, b and b/par are printed because they are free, and are\n" " DIAGNOSTIC ONLY - on the calibration set chi2/best and b/par put their LARGEST value on a\n" " GENUINE crystal, with both known false cases inside the genuine range. A ratio to the\n" " PARENT is contaminated whenever the parent is itself false, which is exactly the case\n" " these columns would have to catch. They are neither a ranking nor a second opinion.\n" "\n BLIND SPOTS. This table is folded from merged intensities, so it can only see a\n" " hypothesis that DEGRADES A MERGE. It is structurally blind to the other two ways a group\n" " is over-called: a screw-axis over-call is in the same Laue class and folds a byte-identical\n" " merge, and a wrong LATTICE is not a hypothesis here at all - the cell is an input to this\n" " search, not something it weighs. A crystal whose only error is one of those has a\n" " clean-looking ledger, and a clean ledger is therefore not a clean bill of health.\n" "\n MARGIN. On the corpus this was calibrated on, the largest R/best ever seen on a GENUINE\n" " adopted group is 1.90, against a refusal bound of 2.00. The accept side has ~0.10 of\n" " headroom, not the wide gap the refused rows suggest, so a margin printed above as small is\n" " a real one - and any future tightening of this bound has almost none to spend.\n"; return os.str(); } std::optional OperatorCorrelation(const std::vector& merged, const gemmi::Mat33& hkl_matrix, const SearchSpaceGroupOptions& opt) { if (merged.empty()) return std::nullopt; // gemmi applies an Op to Miller indices through the TRANSPOSE of its rot (Op::apply_to_hkl), and // writes an hkl triplet from the transpose as well, so the operator whose Miller-index matrix is M // is the Op holding M^T - and its triplet then reads as the rows of M, as everywhere else here. gemmi::Op op{}; op.notation = 'h'; op.tran = {0, 0, 0}; for (int i = 0; i < 3; ++i) for (int j = 0; j < 3; ++j) op.rot[i][j] = static_cast(std::lround(hkl_matrix[j][i])) * gemmi::Op::DEN; const NormalisedMerge nm = PrepareMerge(merged, opt); std::vector visited(nm.n, 0); uint32_t epoch = 0; const SpaceGroupOperatorScore s = ScoreOperator(nm, opt, op, visited, epoch); // Too few pairs to judge - the same floor Stage A refuses to call an operator present on. if (s.n_pairs < opt.min_pairs_per_operator || !std::isfinite(s.cc)) return std::nullopt; return s; } std::string SearchSpaceGroupResultToText(const SearchSpaceGroupResult& result, size_t max_candidates_to_print) { std::ostringstream os; if (!result.refused_point_group_hm.empty()) os << "Higher symmetry " << result.refused_point_group_hm << " was confirmed by the operator " "correlations but REFUSED: " << result.refused_reason << ".\n" " Processing in the lower symmetry, which is the recoverable direction - if this is a " "twin, merging in the higher group would average non-equivalent reflections together and " "hide the twin law.\n"; os << "Point group: " << (result.point_group_hm.empty() ? "?" : result.point_group_hm) << " (from intensity correlations)\n"; os << " " << std::setw(14) << std::left << "operator" << std::right << std::setw(9) << "CC" << std::setw(10) << "pairs" << std::setw(9) << "symm" << std::setw(9) << "H" << std::setw(9) << "R" << "\n"; for (const auto& s : result.operator_scores) { os << " " << std::setw(14) << std::left << s.op_triplet_hkl << std::right << std::setw(9) << std::fixed << std::setprecision(3) << s.cc << std::setw(10) << s.n_pairs << std::setw(9) << (s.present ? "yes" : "no") << std::setw(9) << std::fixed << std::setprecision(3) << s.h_stat << std::setw(9) << std::fixed << std::setprecision(3) << s.r_stat << "\n"; } os << " H = median |I1-I2|/(I1+I2) over the operator's pairs - the disagreement it implies, with\n" " no sigma in it. The promotion gate is the RATIO of the mean H over the operators a\n" " promotion adds to the mean over the parent group's own, which is what separates a real\n" " symmetry (ratio near 1) from a merohedral twin law.\n"; if (!result.generated_point_group_hm.empty()) os << " The confirmed operators are NOT a group: what they generate is " << result.generated_point_group_hm << ", which the per-operator bar did not admit, so a set" " of operators\n the data confirmed has a product the data refused. " << result.generated_point_group_hm << (result.generated_point_group_adopted ? " was offered as a candidate and ADOPTED." : " was offered as a candidate and refused by the tests below.") << "\n"; if (std::isfinite(result.h_ratio)) os << " H ratio " << FormatDouble(result.h_ratio, 2) << " for the adopted point group (bound " << FormatDouble(result.h_ratio_bound, 2) << ").\n"; else os << " H ratio not available (no parent group to normalise against, or too few pairs).\n"; // R = intensity-weighted sum|I1-I2|/sum(I1+I2) across the operator's pairs (strong reflections // dominate, where the median H under-weights them). The merge-degradation gate scores the added // operators' R against a clean reference - the globally best-agreeing operator, or the merge's // own random-noise R floor (from the half-dataset merges) on the first step out of P1 - so it // catches false symmetry the parent-normalised H waves through. if (std::isfinite(result.r_added)) { os << " Added-operator R " << FormatDouble(result.r_added, 3); if (std::isfinite(result.r_over_best)) os << " = " << FormatDouble(result.r_over_best, 2) << "x the best operator (" << FormatDouble(result.global_best_operator_r, 3) << ")"; if (std::isfinite(result.r_over_floor)) os << ", " << FormatDouble(result.r_over_floor, 2) << "x the random-noise floor (" << FormatDouble(result.merge_r_floor, 3) << ")"; os << " for the adopted point group.\n"; } else if (std::isfinite(result.merge_r_floor)) { os << " Merge random-noise R floor " << FormatDouble(result.merge_r_floor, 3) << ".\n"; } os << "\nSpace-group candidates\n"; os << " " << std::setw(10) << std::left << "SG" << std::right << std::setw(9) << "absent" << std::setw(7) << "viol" << std::setw(11) << "abs" << std::setw(11) << "pres" << std::setw(9) << "screw" << std::setw(11) << "screw evid" << std::setw(11) << "cent evid" << std::setw(6) << "OK" << std::setw(11) << "centering" << "\n"; const size_t count = std::min(max_candidates_to_print, result.candidates.size()); for (size_t i = 0; i < count; ++i) { const auto& c = result.candidates[i]; os << (c.selected ? "* " : " ") << std::setw(10) << std::left << SettingName(c.space_group) << std::right << std::setw(9) << c.absent_observed << std::setw(7) << c.absent_violations << std::setw(11) << std::fixed << std::setprecision(2) << c.absent_mean_i_over_sigma << std::setw(11) << std::fixed << std::setprecision(2) << c.present_mean_i_over_sigma << std::setw(9) << c.screw_absent << std::setw(11) << std::fixed << std::setprecision(1) << c.screw_absence_evidence << std::setw(11) << std::fixed << std::setprecision(1) << c.centering_absence_evidence << std::setw(6) << (c.consistent ? "yes" : "no") << std::setw(11) << (c.centering_untested ? "UNTESTED" : (c.centering_absent > 0 ? "tested" : "-")) << "\n"; } os << " absent/viol = reflections the group predicts absent, and how many are nonetheless present.\n" " screw = how many of those lie on an axial row, i.e. are extinguished by a screw rather than\n" " by the centering; screw evid = how much likelier that class is if the screw exists than if it\n" " does not, judged against the rest of its own axial row (in nats - a real screw reads tens to\n" " hundreds, a false one at or below zero). cent evid is how unlikely the centering-absent class\n" " would be without the centering, judged against the present class; the two together rank the\n" " candidates. The columns say little about screws, because the merged sigma shrinks with\n" " I on absent and present alike.\n" " centering = whether the group's centering was decided here at all. UNTESTED means this\n" " merge holds none of the reflections that centering extinguishes - the data are indexed on\n" " the primitive sub-cell, so those reflections were never predicted or integrated - and the\n" " candidate scores zero absences for want of evidence, not because the centering is real.\n"; // Per-zone screw verdicts for the group that was chosen. One axial row is one screw condition, and // the rows are measured to very different depths, so a row this merge cannot judge is reported as // undetermined rather than being folded into a single yes/no for the whole group. for (const auto& c : result.candidates) { if (!c.selected || c.screw_zones.empty()) continue; os << "Screw conditions of " << SettingName(c.space_group) << ", zone by zone:\n"; for (const auto& pt : result.pseudo_translations) { os << " " << RowLabel(pt.row) << " is judged against a pseudo-translation at ("; for (int j = 0; j < 3; ++j) os << (j ? ", " : "") << (pt.halves[j] ? "1/2" : "0"); os << "), which holds the class it suppresses at " << std::fixed << std::setprecision(3) << pt.ratio << " of the other one along this row - so an absence there is expected to\n" << " be weak whether or not a screw exists, and the evidence below is scored against" << " that.\n"; } for (const auto& z : c.screw_zones) { os << " " << RowLabel(z.row) << " " << std::setw(3) << z.n_absent << " absent "; if (z.n_control == 0) os << "UNDETERMINED - this row carries no control reflections here\n"; else os << std::setw(4) << z.n_control << " control " << std::fixed << std::setprecision(1) << std::setw(8) << z.evidence << " nats\n"; } break; } // Per-zone GLIDE verdicts, on the same terms as the screw zones above. Printed only when a // glide was actually claimed - a Sohncke crystal (any protein) has none, and an empty heading // reads as "nothing was found" where it means "nothing was asked". if (result.glide_space_group.has_value() && !result.glide_zones.empty()) { os << "Glide planes of " << SettingName(*result.glide_space_group) << ", zone by zone:\n"; for (const auto& z : result.glide_zones) os << " " << std::setw(4) << std::left << z.label << std::right << std::setw(5) << z.n_absent << " absent " << std::setw(5) << z.n_control << " control at " << std::fixed << std::setprecision(4) << std::setw(9) << z.mean_u << " of the rest of their plane " << std::setprecision(2) << std::setw(7) << z.evidence_per_reflection << " nats/reflection\n"; os << " A glide plane is impossible in a crystal of chiral molecules, so this says the sample\n" " is not a protein. The claim is judged PER REFLECTION, not on the total: the statistic\n" " is linear in the size of the absent class and a zone is a plane, so a merely weak zone\n" " can sum to hundreds of nats while saying nothing about deadness.\n"; if (result.sohncke_space_group.has_value()) os << " Best Sohncke group for the same data: " << SettingName(*result.sohncke_space_group) << ".\n"; } // The axes the data left open, said in words. The zone table above is printed for the SELECTED // candidate only, and the selected candidate in exactly this case is the one with NO screw zones // - so without this the run's own account of an undecidable screw is a blank where the table // would have been. if (!result.undetermined_screws.empty()) { os << "Screw axes these data cannot decide\n"; for (const auto& u : result.undetermined_screws) { os << " " << u.axis << ": "; if (u.n_observed == 0) os << "the " << u.row_label << " row holds no reflections in this merge at all, so " "nothing here\n could test a screw along " << u.axis << " - the row was not recorded (it can lie in the spindle's\n blind cone, or " "outside the resolution range).\n"; else if (u.n_control == 0) os << "the " << u.row_label << " row holds " << u.n_observed << " reflection(s) but no " "control class for its absences,\n so how weak they are cannot be judged.\n"; else os << "the " << u.row_label << " row does not separate the candidates (" << u.n_absent << " absent, " << u.n_control << " control).\n"; } os << " A reflection file must carry ONE group, so the groups below are reported with the\n" " lowest-numbered - the one claiming no screw - standing for the set. That is a\n" " convention, not a measurement: on the axes named here the answer is all of them.\n"; } if (result.best_space_group.has_value()) { os << "Best space group: " << SettingName(*result.best_space_group); for (const auto& alt : result.alternatives) os << " or " << SettingName(alt); if (!result.alternatives.empty()) os << " (indistinguishable from these data)"; os << "\n"; // A group with a different CENTERING is a different lattice, not just a different group, and // a run reports one cell - the chosen group's. "C2 or P21 or P2" printed beside a single // C-centred cell is not something a user can act on: P2 and P21 live on the primitive // sub-cell, with their own cell constants and their own Miller indices. Name the ones the // reported cell does not describe, and say how far off it is for each. const char sel_centring = result.best_space_group->centring_type(); bool said_setting = false; for (const auto& alt : result.alternatives) { if (alt.centring_type() == sel_centring) continue; if (!said_setting) { os << " These are NOT all the same lattice, and one cell is reported for this run -\n" " " << SettingName(*result.best_space_group) << "'s (centering " << sel_centring << "). These need a cell of their own:\n"; said_setting = true; } os << " " << std::setw(10) << std::left << SettingName(alt) << std::right << "centering " << alt.centring_type() << ", cell volume " << std::fixed << std::setprecision(2) << static_cast(gemmi::centring_vectors(alt.centring_type()).size()) / static_cast(gemmi::centring_vectors(sel_centring).size()) << "x the reported one\n"; } if (said_setting) os << " Adopting one of those means reindexing to its cell; the cell constants reported\n" " and the Miller indices written are the chosen group's alone.\n"; } else { os << "Best space group: none determined\n"; } return os.str(); }