Files
Jungfraujoch/tools/battery/depdata_check.py
T
leonarski_fandClaude Opus 5.5 9b7a91d521 battery: depositor-data comparison column; REFMAC check reads partial-ANISOU models
depdata_check.py compares each open-arm merge with the depositor's own data, per resolution
shell, using the deposited model as the common yardstick: the rank correlation of our IMEAN with
|Fc|^2 minus that of the deposited intensities (or F^2), on the reflections both carry (d < 4 A,
eight equal-count shells), with the data carried into the model's setting by model_check's
change-of-basis search. New row keys dep_cc_delta_all, dep_cc_delta_outer (two outermost common
shells), dep_beyond_cc / dep_beyond_d (our correlation with |Fc|^2 past the deposited data's
limit), dep_kind, dep_d_min, dep_n_common, dep_status, dep_reason. gemmi only, no CCP4; runs on
every open-arm set with a model, and `report` fills it in for older runs (four processes).
Reported in its own report section and the per-set table, never scored.

Rank rather than Pearson correlation: on the phase-1 run a handful of gross outliers (I/sigma
above 1000 at 2.4 A, or ~1000x the shell median at 1.41 A) decided the outer-shell Pearson CC of
several merges on their own.

On the phase-1 run (38 of 40 open-arm sets compared; the two without are the lattice failures)
the outer-shell delta has median -0.012; 22 s for the whole report fill-in.

model_check.py: REFMAC's mmCIF reader stopped with "rdaniso_cif: Atom symbol mismatch" on 12 of 35
entries whose atoms carry ANISOU only in part. The model is now written in the PDB format where it
fits (under 100000 atoms, chain names of at most two characters), where each ANISOU follows its
own ATOM line; all 12 then score. R-factors of the entries that worked before move by < 0.001.
The runner also records model_check's baseline note (e.g. an intensity-only deposition) in
refmac_reason instead of leaving it empty.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-09-23 19:59:17 +02:00

221 lines
10 KiB
Python

#!/usr/bin/env python3
"""depdata_check.py -- Rugnux's merged intensities against the depositor's, shell by shell.
A second, REFMAC-free quality check for the open arm, run AFTER Rugnux has written its MTZ. The
deposited model is the common yardstick: per resolution shell, the rank correlation of Rugnux's
merged intensities with |Fc|^2 and that of the depositor's own data are computed on the SAME
reflections, and the difference is reported. |Fc| comes from the deposited model as it is (gemmi,
no bulk solvent, no refinement), so the model was refined against the depositor's data and the
comparison carries a home advantage for the deposition; a difference near zero means our merge is
as good as the one the model was built from.
* Rugnux's merge (IMEAN) is carried into the model's setting exactly as model_check.py does it
(every change of basis between the two lattices, the one whose data correlate best with |Fc|^2
at 3-6 A kept) and reduced to the deposited group's asymmetric unit;
* the depositor's data are the first merged reflection block of the entry's -sf.cif (intensities
where given, else amplitudes squared; with several wavelengths the one nearest ours), in the
model's cell and group;
* the reflections both carry, at d < 4 A (all of them where that leaves fewer than 400), are cut
into SHELLS shells of equal count in 1/d^2.
Columns (report-only, never scored):
dep_cc_delta_all CC(ours) - CC(deposited) over the common reflections
dep_cc_delta_outer the same, mean of the two outermost common shells
dep_beyond_cc CC(ours, |Fc|^2) in the outermost of up to three equal-count shells PAST the
deposited data's limit (null when we do not reach past it); the model was
never refined against these, so a clearly positive value is signal
dep_beyond_d that shell's high-resolution limit
depdata_check.py rugnux.mtz 1abc [--cache DIR]
"""
import argparse
import json
import os
import gemmi
import numpy as np
import model_check
import score
SHELLS = 8
D_COMMON = 4.0 # low resolution left out: unmodelled bulk solvent dominates there
KEYS = ("dep_status", "dep_reason", "dep_kind", "dep_d_min", "dep_n_common", "dep_cc_delta_all",
"dep_cc_delta_outer", "dep_beyond_cc", "dep_beyond_d")
def cc(a, b):
"""Rank correlation: a handful of gross outliers (seen in merges of both kinds) would decide a
shell's Pearson CC on their own."""
if len(a) < 20:
return None
return float(np.corrcoef(np.argsort(np.argsort(a)), np.argsort(np.argsort(b)))[0, 1])
def shells(d, n):
"""Shell index (0 = lowest resolution) of each reflection, n shells of equal count in 1/d^2."""
s2 = 1.0 / d ** 2
edges = np.quantile(s2, np.linspace(0, 1, n + 1))
return np.clip(np.searchsorted(edges, s2, side="right") - 1, 0, n - 1)
def rugnux_intensities(mtz, t, cell_m, sg_m):
"""IMEAN of Rugnux's merge reindexed by t and reduced to the model group's asymmetric unit."""
i = np.array(mtz.column_with_label("IMEAN"), copy=False).astype(float)
s = np.array(mtz.column_with_label("SIGIMEAN"), copy=False).astype(float)
good = np.isfinite(i) & np.isfinite(s) & (s > 0)
hkl, src = model_check.expand_p1(mtz)
hm = hkl @ t
integral = np.all(np.abs(hm - np.round(hm)) < 1e-3, axis=1) & good[src]
hm, src = np.round(hm[integral]).astype(int), src[integral]
hkl, i, _, _ = model_check.to_group(hm, i[src], s[src], np.zeros(len(src), int), cell_m, sg_m, src)
return hkl, i
def deposited_intensities(path, wavelength, cell_m, sg_m):
"""(hkl, I, kind) of the entry's merged data in the model's asymmetric unit, or None, why."""
blocks = []
for rb in gemmi.as_refln_blocks(gemmi.cif.read(path)):
labels = rb.column_labels()
if rb.is_unmerged():
continue
for kind, cols in (("I", ("intensity_meas", "intensity_sigma")),
("F", ("F_meas_au", "F_meas_sigma_au")),
("I", ("pdbx_I_plus", "pdbx_I_minus")),
("F", ("pdbx_F_plus", "pdbx_F_minus"))):
if all(c in labels for c in cols):
blocks.append((rb, kind, cols))
break
if not blocks:
return None, "no measured data in the deposited structure factors"
rb, kind, cols = blocks[0]
if len(blocks) > 1 and wavelength:
near = [abs(b[0].wavelength - wavelength) if b[0].wavelength else 9.0 for b in blocks]
if min(near) < 0.02:
rb, kind, cols = blocks[int(np.argmin(near))]
if not np.allclose(rb.cell.parameters, cell_m.parameters, rtol=0.02, atol=0.5):
return None, "deposited structure factors in another cell"
a = np.array(rb.make_float_array(cols[0]))
if cols[0].startswith("pdbx_"): # anomalous pair: the mean of the two hands
b = np.array(rb.make_float_array(cols[1]))
a = np.where(np.isfinite(a) & np.isfinite(b), (a + b) / 2, np.where(np.isfinite(a), a, b))
x = a * a if kind == "F" else a
ok = np.isfinite(x)
if "status" in rb.column_labels(): # '-', 'x', '<': not measured
st = [v.strip("'\"") for v in rb.block.find_values("_refln.status")]
if len(st) == len(x):
ok &= np.array([v not in ("-", "x", "<") for v in st])
hkl = np.array(rb.make_miller_array())[ok]
x = x[ok]
d_min = float(rb.cell.calculate_d_array(hkl).min())
hkl, x, _, _ = model_check.to_group(hkl, x, np.ones(len(x)), np.zeros(len(x), int), cell_m, sg_m,
np.arange(len(x)))
return (hkl, x, kind, d_min), None
def lookup(keys, hkl):
"""Index into `keys` (sorted) of each reflection of hkl, and which of them were found."""
k = model_check.hkl_key(hkl)
pos = np.clip(np.searchsorted(keys, k), 0, len(keys) - 1)
return pos, keys[pos] == k
def check(mtz_path, pdb_id, cache_dir=model_check.DEFAULT_CACHE, wavelength=None):
res = {k: None for k in KEYS}
res["dep_status"] = "failed"
pdb_id = pdb_id.lower()
if not model_check.PDB_ID.match(pdb_id):
return dict(res, dep_status="n/a", dep_reason="not a PDB entry")
try:
meta = model_check.deposition(pdb_id, cache_dir)
if meta is None:
return dict(res, dep_reason="entry could not be downloaded")
sf = model_check.fetch(model_check.RCSB + pdb_id + "-sf.cif.gz",
os.path.join(cache_dir, pdb_id + "-sf.cif.gz"))
if sf is None:
return dict(res, dep_status="n/a", dep_reason="no deposited structure factors")
st = gemmi.read_structure(meta["xyz"])
sg_m, cell_m = st.find_spacegroup(), st.cell
if sg_m is None:
return dict(res, dep_reason=f"model space group '{st.spacegroup_hm}' not recognised")
dep, why = deposited_intensities(sf, wavelength, cell_m, sg_m)
if dep is None:
return dict(res, dep_status="n/a", dep_reason=why)
hkl_d, i_d, res["dep_kind"], d_dep = dep
res["dep_d_min"] = round(d_dep, 3)
mtz = gemmi.read_mtz_file(mtz_path)
fc_hkl, fc = model_check.fcalc(st, mtz.resolution_high() - 0.005)
order = np.argsort(model_check.hkl_key(fc_hkl))
fc_keys, fc_hkl, fc2 = model_check.hkl_key(fc_hkl)[order], fc_hkl[order], fc[order] ** 2
fc_d = cell_m.calculate_d_array(fc_hkl)
best = None # the setting: best CC(I, |Fc|^2) at 3-6 A (everything if that is too few)
for t in model_check.basis_changes(mtz.cell, mtz.spacegroup, cell_m, sg_m):
hkl, i = rugnux_intensities(mtz, t, cell_m, sg_m)
pos, found = lookup(fc_keys, hkl)
sel = found.copy()
sel[found] &= (fc_d[pos[found]] >= 3) & (fc_d[pos[found]] <= 6)
if sel.sum() < 200:
sel = found
c = cc(i[sel], fc2[pos[sel]])
if c is not None and (best is None or c > best[0]):
best = (c, hkl, i)
if best is None or best[0] < 0.1: # a model-only |Fc| correlates weakly at 4-6 A
return dict(res, dep_reason="the merge does not match the model in any setting")
_, hkl_r, i_r = best
pos, found = lookup(fc_keys, hkl_r)
i_r, f_r, d_r = i_r[found], fc2[pos[found]], fc_d[pos[found]]
k_r = fc_keys[pos[found]]
_, a, b = np.intersect1d(k_r, model_check.hkl_key(hkl_d), return_indices=True)
ir, idp, fcc, d = i_r[a], i_d[b], f_r[a], d_r[a]
sel = d < D_COMMON
if sel.sum() < SHELLS * 50:
sel = np.ones(len(d), bool)
ir, idp, fcc, d = ir[sel], idp[sel], fcc[sel], d[sel]
res["dep_n_common"] = int(len(d))
if len(d) < SHELLS * 20:
return dict(res, dep_reason=f"only {len(d)} reflections in common")
res["dep_cc_delta_all"] = round(cc(ir, fcc) - cc(idp, fcc), 4)
sh = shells(d, SHELLS)
res["dep_cc_delta_outer"] = round(float(np.mean(
[cc(ir[sh == k], fcc[sh == k]) - cc(idp[sh == k], fcc[sh == k]) for k in (SHELLS - 1, SHELLS - 2)])), 4)
past = d_r < d_dep - 1e-3 # ours only, past the deposited data's limit
if past.sum() >= 60:
n = max(1, min(3, int(past.sum() // 300)))
sh = shells(d_r[past], n)
out = sh == n - 1
res["dep_beyond_cc"] = round(cc(i_r[past][out], f_r[past][out]), 4)
res["dep_beyond_d"] = round(float(d_r[past][out].min()), 3)
res["dep_status"] = "ok"
return res
except Exception as e: # one bad entry must not stop a battery
if os.environ.get("MODEL_CHECK_RAISE"):
raise
return dict(res, dep_reason=f"{type(e).__name__}: {e}"[:300])
def check_set(wd, set_id, cache_dir=model_check.DEFAULT_CACHE):
"""check() on a battery work directory (p.mtz, and the wavelength from p_report.txt)."""
try:
wl = float(score.read_report(os.path.join(wd, "p_report.txt")).get("WAVELENGTH"))
except (TypeError, ValueError):
wl = None
return check(os.path.join(wd, "p.mtz"), set_id.split("_")[0], cache_dir, wl)
def main():
ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
ap.add_argument("mtz", help="Rugnux merged MTZ (needs IMEAN, SIGIMEAN)")
ap.add_argument("pdb_id")
ap.add_argument("--cache", default=model_check.DEFAULT_CACHE)
ap.add_argument("--wavelength", type=float, help="picks the deposited block (multi-wavelength entries)")
a = ap.parse_args()
print(json.dumps(check(a.mtz, a.pdb_id, a.cache, a.wavelength), indent=1))
if __name__ == "__main__":
main()