depdata_check.py compares each open-arm merge with the depositor's own data, per resolution shell, using the deposited model as the common yardstick: the rank correlation of our IMEAN with |Fc|^2 minus that of the deposited intensities (or F^2), on the reflections both carry (d < 4 A, eight equal-count shells), with the data carried into the model's setting by model_check's change-of-basis search. New row keys dep_cc_delta_all, dep_cc_delta_outer (two outermost common shells), dep_beyond_cc / dep_beyond_d (our correlation with |Fc|^2 past the deposited data's limit), dep_kind, dep_d_min, dep_n_common, dep_status, dep_reason. gemmi only, no CCP4; runs on every open-arm set with a model, and `report` fills it in for older runs (four processes). Reported in its own report section and the per-set table, never scored. Rank rather than Pearson correlation: on the phase-1 run a handful of gross outliers (I/sigma above 1000 at 2.4 A, or ~1000x the shell median at 1.41 A) decided the outer-shell Pearson CC of several merges on their own. On the phase-1 run (38 of 40 open-arm sets compared; the two without are the lattice failures) the outer-shell delta has median -0.012; 22 s for the whole report fill-in. model_check.py: REFMAC's mmCIF reader stopped with "rdaniso_cif: Atom symbol mismatch" on 12 of 35 entries whose atoms carry ANISOU only in part. The model is now written in the PDB format where it fits (under 100000 atoms, chain names of at most two characters), where each ANISOU follows its own ATOM line; all 12 then score. R-factors of the entries that worked before move by < 0.001. The runner also records model_check's baseline note (e.g. an intensity-only deposition) in refmac_reason instead of leaving it empty. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
221 lines
10 KiB
Python
221 lines
10 KiB
Python
#!/usr/bin/env python3
|
|
"""depdata_check.py -- Rugnux's merged intensities against the depositor's, shell by shell.
|
|
|
|
A second, REFMAC-free quality check for the open arm, run AFTER Rugnux has written its MTZ. The
|
|
deposited model is the common yardstick: per resolution shell, the rank correlation of Rugnux's
|
|
merged intensities with |Fc|^2 and that of the depositor's own data are computed on the SAME
|
|
reflections, and the difference is reported. |Fc| comes from the deposited model as it is (gemmi,
|
|
no bulk solvent, no refinement), so the model was refined against the depositor's data and the
|
|
comparison carries a home advantage for the deposition; a difference near zero means our merge is
|
|
as good as the one the model was built from.
|
|
|
|
* Rugnux's merge (IMEAN) is carried into the model's setting exactly as model_check.py does it
|
|
(every change of basis between the two lattices, the one whose data correlate best with |Fc|^2
|
|
at 3-6 A kept) and reduced to the deposited group's asymmetric unit;
|
|
* the depositor's data are the first merged reflection block of the entry's -sf.cif (intensities
|
|
where given, else amplitudes squared; with several wavelengths the one nearest ours), in the
|
|
model's cell and group;
|
|
* the reflections both carry, at d < 4 A (all of them where that leaves fewer than 400), are cut
|
|
into SHELLS shells of equal count in 1/d^2.
|
|
|
|
Columns (report-only, never scored):
|
|
dep_cc_delta_all CC(ours) - CC(deposited) over the common reflections
|
|
dep_cc_delta_outer the same, mean of the two outermost common shells
|
|
dep_beyond_cc CC(ours, |Fc|^2) in the outermost of up to three equal-count shells PAST the
|
|
deposited data's limit (null when we do not reach past it); the model was
|
|
never refined against these, so a clearly positive value is signal
|
|
dep_beyond_d that shell's high-resolution limit
|
|
|
|
depdata_check.py rugnux.mtz 1abc [--cache DIR]
|
|
"""
|
|
import argparse
|
|
import json
|
|
import os
|
|
|
|
import gemmi
|
|
import numpy as np
|
|
|
|
import model_check
|
|
import score
|
|
|
|
SHELLS = 8
|
|
D_COMMON = 4.0 # low resolution left out: unmodelled bulk solvent dominates there
|
|
KEYS = ("dep_status", "dep_reason", "dep_kind", "dep_d_min", "dep_n_common", "dep_cc_delta_all",
|
|
"dep_cc_delta_outer", "dep_beyond_cc", "dep_beyond_d")
|
|
|
|
|
|
def cc(a, b):
|
|
"""Rank correlation: a handful of gross outliers (seen in merges of both kinds) would decide a
|
|
shell's Pearson CC on their own."""
|
|
if len(a) < 20:
|
|
return None
|
|
return float(np.corrcoef(np.argsort(np.argsort(a)), np.argsort(np.argsort(b)))[0, 1])
|
|
|
|
|
|
def shells(d, n):
|
|
"""Shell index (0 = lowest resolution) of each reflection, n shells of equal count in 1/d^2."""
|
|
s2 = 1.0 / d ** 2
|
|
edges = np.quantile(s2, np.linspace(0, 1, n + 1))
|
|
return np.clip(np.searchsorted(edges, s2, side="right") - 1, 0, n - 1)
|
|
|
|
|
|
def rugnux_intensities(mtz, t, cell_m, sg_m):
|
|
"""IMEAN of Rugnux's merge reindexed by t and reduced to the model group's asymmetric unit."""
|
|
i = np.array(mtz.column_with_label("IMEAN"), copy=False).astype(float)
|
|
s = np.array(mtz.column_with_label("SIGIMEAN"), copy=False).astype(float)
|
|
good = np.isfinite(i) & np.isfinite(s) & (s > 0)
|
|
hkl, src = model_check.expand_p1(mtz)
|
|
hm = hkl @ t
|
|
integral = np.all(np.abs(hm - np.round(hm)) < 1e-3, axis=1) & good[src]
|
|
hm, src = np.round(hm[integral]).astype(int), src[integral]
|
|
hkl, i, _, _ = model_check.to_group(hm, i[src], s[src], np.zeros(len(src), int), cell_m, sg_m, src)
|
|
return hkl, i
|
|
|
|
|
|
def deposited_intensities(path, wavelength, cell_m, sg_m):
|
|
"""(hkl, I, kind) of the entry's merged data in the model's asymmetric unit, or None, why."""
|
|
blocks = []
|
|
for rb in gemmi.as_refln_blocks(gemmi.cif.read(path)):
|
|
labels = rb.column_labels()
|
|
if rb.is_unmerged():
|
|
continue
|
|
for kind, cols in (("I", ("intensity_meas", "intensity_sigma")),
|
|
("F", ("F_meas_au", "F_meas_sigma_au")),
|
|
("I", ("pdbx_I_plus", "pdbx_I_minus")),
|
|
("F", ("pdbx_F_plus", "pdbx_F_minus"))):
|
|
if all(c in labels for c in cols):
|
|
blocks.append((rb, kind, cols))
|
|
break
|
|
if not blocks:
|
|
return None, "no measured data in the deposited structure factors"
|
|
rb, kind, cols = blocks[0]
|
|
if len(blocks) > 1 and wavelength:
|
|
near = [abs(b[0].wavelength - wavelength) if b[0].wavelength else 9.0 for b in blocks]
|
|
if min(near) < 0.02:
|
|
rb, kind, cols = blocks[int(np.argmin(near))]
|
|
if not np.allclose(rb.cell.parameters, cell_m.parameters, rtol=0.02, atol=0.5):
|
|
return None, "deposited structure factors in another cell"
|
|
a = np.array(rb.make_float_array(cols[0]))
|
|
if cols[0].startswith("pdbx_"): # anomalous pair: the mean of the two hands
|
|
b = np.array(rb.make_float_array(cols[1]))
|
|
a = np.where(np.isfinite(a) & np.isfinite(b), (a + b) / 2, np.where(np.isfinite(a), a, b))
|
|
x = a * a if kind == "F" else a
|
|
ok = np.isfinite(x)
|
|
if "status" in rb.column_labels(): # '-', 'x', '<': not measured
|
|
st = [v.strip("'\"") for v in rb.block.find_values("_refln.status")]
|
|
if len(st) == len(x):
|
|
ok &= np.array([v not in ("-", "x", "<") for v in st])
|
|
hkl = np.array(rb.make_miller_array())[ok]
|
|
x = x[ok]
|
|
d_min = float(rb.cell.calculate_d_array(hkl).min())
|
|
hkl, x, _, _ = model_check.to_group(hkl, x, np.ones(len(x)), np.zeros(len(x), int), cell_m, sg_m,
|
|
np.arange(len(x)))
|
|
return (hkl, x, kind, d_min), None
|
|
|
|
|
|
def lookup(keys, hkl):
|
|
"""Index into `keys` (sorted) of each reflection of hkl, and which of them were found."""
|
|
k = model_check.hkl_key(hkl)
|
|
pos = np.clip(np.searchsorted(keys, k), 0, len(keys) - 1)
|
|
return pos, keys[pos] == k
|
|
|
|
|
|
def check(mtz_path, pdb_id, cache_dir=model_check.DEFAULT_CACHE, wavelength=None):
|
|
res = {k: None for k in KEYS}
|
|
res["dep_status"] = "failed"
|
|
pdb_id = pdb_id.lower()
|
|
if not model_check.PDB_ID.match(pdb_id):
|
|
return dict(res, dep_status="n/a", dep_reason="not a PDB entry")
|
|
try:
|
|
meta = model_check.deposition(pdb_id, cache_dir)
|
|
if meta is None:
|
|
return dict(res, dep_reason="entry could not be downloaded")
|
|
sf = model_check.fetch(model_check.RCSB + pdb_id + "-sf.cif.gz",
|
|
os.path.join(cache_dir, pdb_id + "-sf.cif.gz"))
|
|
if sf is None:
|
|
return dict(res, dep_status="n/a", dep_reason="no deposited structure factors")
|
|
st = gemmi.read_structure(meta["xyz"])
|
|
sg_m, cell_m = st.find_spacegroup(), st.cell
|
|
if sg_m is None:
|
|
return dict(res, dep_reason=f"model space group '{st.spacegroup_hm}' not recognised")
|
|
dep, why = deposited_intensities(sf, wavelength, cell_m, sg_m)
|
|
if dep is None:
|
|
return dict(res, dep_status="n/a", dep_reason=why)
|
|
hkl_d, i_d, res["dep_kind"], d_dep = dep
|
|
res["dep_d_min"] = round(d_dep, 3)
|
|
|
|
mtz = gemmi.read_mtz_file(mtz_path)
|
|
fc_hkl, fc = model_check.fcalc(st, mtz.resolution_high() - 0.005)
|
|
order = np.argsort(model_check.hkl_key(fc_hkl))
|
|
fc_keys, fc_hkl, fc2 = model_check.hkl_key(fc_hkl)[order], fc_hkl[order], fc[order] ** 2
|
|
fc_d = cell_m.calculate_d_array(fc_hkl)
|
|
|
|
best = None # the setting: best CC(I, |Fc|^2) at 3-6 A (everything if that is too few)
|
|
for t in model_check.basis_changes(mtz.cell, mtz.spacegroup, cell_m, sg_m):
|
|
hkl, i = rugnux_intensities(mtz, t, cell_m, sg_m)
|
|
pos, found = lookup(fc_keys, hkl)
|
|
sel = found.copy()
|
|
sel[found] &= (fc_d[pos[found]] >= 3) & (fc_d[pos[found]] <= 6)
|
|
if sel.sum() < 200:
|
|
sel = found
|
|
c = cc(i[sel], fc2[pos[sel]])
|
|
if c is not None and (best is None or c > best[0]):
|
|
best = (c, hkl, i)
|
|
if best is None or best[0] < 0.1: # a model-only |Fc| correlates weakly at 4-6 A
|
|
return dict(res, dep_reason="the merge does not match the model in any setting")
|
|
_, hkl_r, i_r = best
|
|
pos, found = lookup(fc_keys, hkl_r)
|
|
i_r, f_r, d_r = i_r[found], fc2[pos[found]], fc_d[pos[found]]
|
|
k_r = fc_keys[pos[found]]
|
|
|
|
_, a, b = np.intersect1d(k_r, model_check.hkl_key(hkl_d), return_indices=True)
|
|
ir, idp, fcc, d = i_r[a], i_d[b], f_r[a], d_r[a]
|
|
sel = d < D_COMMON
|
|
if sel.sum() < SHELLS * 50:
|
|
sel = np.ones(len(d), bool)
|
|
ir, idp, fcc, d = ir[sel], idp[sel], fcc[sel], d[sel]
|
|
res["dep_n_common"] = int(len(d))
|
|
if len(d) < SHELLS * 20:
|
|
return dict(res, dep_reason=f"only {len(d)} reflections in common")
|
|
res["dep_cc_delta_all"] = round(cc(ir, fcc) - cc(idp, fcc), 4)
|
|
sh = shells(d, SHELLS)
|
|
res["dep_cc_delta_outer"] = round(float(np.mean(
|
|
[cc(ir[sh == k], fcc[sh == k]) - cc(idp[sh == k], fcc[sh == k]) for k in (SHELLS - 1, SHELLS - 2)])), 4)
|
|
|
|
past = d_r < d_dep - 1e-3 # ours only, past the deposited data's limit
|
|
if past.sum() >= 60:
|
|
n = max(1, min(3, int(past.sum() // 300)))
|
|
sh = shells(d_r[past], n)
|
|
out = sh == n - 1
|
|
res["dep_beyond_cc"] = round(cc(i_r[past][out], f_r[past][out]), 4)
|
|
res["dep_beyond_d"] = round(float(d_r[past][out].min()), 3)
|
|
res["dep_status"] = "ok"
|
|
return res
|
|
except Exception as e: # one bad entry must not stop a battery
|
|
if os.environ.get("MODEL_CHECK_RAISE"):
|
|
raise
|
|
return dict(res, dep_reason=f"{type(e).__name__}: {e}"[:300])
|
|
|
|
|
|
def check_set(wd, set_id, cache_dir=model_check.DEFAULT_CACHE):
|
|
"""check() on a battery work directory (p.mtz, and the wavelength from p_report.txt)."""
|
|
try:
|
|
wl = float(score.read_report(os.path.join(wd, "p_report.txt")).get("WAVELENGTH"))
|
|
except (TypeError, ValueError):
|
|
wl = None
|
|
return check(os.path.join(wd, "p.mtz"), set_id.split("_")[0], cache_dir, wl)
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
|
|
ap.add_argument("mtz", help="Rugnux merged MTZ (needs IMEAN, SIGIMEAN)")
|
|
ap.add_argument("pdb_id")
|
|
ap.add_argument("--cache", default=model_check.DEFAULT_CACHE)
|
|
ap.add_argument("--wavelength", type=float, help="picks the deposited block (multi-wavelength entries)")
|
|
a = ap.parse_args()
|
|
print(json.dumps(check(a.mtz, a.pdb_id, a.cache, a.wavelength), indent=1))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|