Files
Jungfraujoch/tools/battery/depdata_check.py
T
leonarski_f 84228bf8be
Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
v1.0.0-rc.173 (#83)
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports.
* jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls.
* Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results.
* Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable.
* Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate.
* Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do.
* Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence.
* Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags.
* Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check.
* Rugnux: Clear error messages when a data set needs more GPU or host memory than is available.

Reviewed-on: #83
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-09-29 15:57:32 +02:00

221 lines
10 KiB
Python

#!/usr/bin/env python3
"""depdata_check.py -- Rugnux's merged intensities against the depositor's, shell by shell.
A second, REFMAC-free quality check for the open arm, run AFTER Rugnux has written its MTZ. The
deposited model is the common yardstick: per resolution shell, the rank correlation of Rugnux's
merged intensities with |Fc|^2 and that of the depositor's own data are computed on the SAME
reflections, and the difference is reported. |Fc| comes from the deposited model as it is (gemmi,
no bulk solvent, no refinement), so the model was refined against the depositor's data and the
comparison carries a home advantage for the deposition; a difference near zero means our merge is
as good as the one the model was built from.
* Rugnux's merge (IMEAN) is carried into the model's setting exactly as model_check.py does it
(every change of basis between the two lattices, the one whose data correlate best with |Fc|^2
at 3-6 A kept) and reduced to the deposited group's asymmetric unit;
* the depositor's data are the first merged reflection block of the entry's -sf.cif (intensities
where given, else amplitudes squared; with several wavelengths the one nearest ours), in the
model's cell and group;
* the reflections both carry, at d < 4 A (all of them where that leaves fewer than 400), are cut
into SHELLS shells of equal count in 1/d^2.
Columns (report-only, never scored):
dep_cc_delta_all CC(ours) - CC(deposited) over the common reflections
dep_cc_delta_outer the same, mean of the two outermost common shells
dep_beyond_cc CC(ours, |Fc|^2) in the outermost of up to three equal-count shells PAST the
deposited data's limit (null when we do not reach past it); the model was
never refined against these, so a clearly positive value is signal
dep_beyond_d that shell's high-resolution limit
depdata_check.py rugnux.mtz 1abc [--cache DIR]
"""
import argparse
import json
import os
import gemmi
import numpy as np
import model_check
import score
SHELLS = 8
D_COMMON = 4.0 # low resolution left out: unmodelled bulk solvent dominates there
KEYS = ("dep_status", "dep_reason", "dep_kind", "dep_d_min", "dep_n_common", "dep_cc_delta_all",
"dep_cc_delta_outer", "dep_beyond_cc", "dep_beyond_d")
def cc(a, b):
"""Rank correlation: a handful of gross outliers (seen in merges of both kinds) would decide a
shell's Pearson CC on their own."""
if len(a) < 20:
return None
return float(np.corrcoef(np.argsort(np.argsort(a)), np.argsort(np.argsort(b)))[0, 1])
def shells(d, n):
"""Shell index (0 = lowest resolution) of each reflection, n shells of equal count in 1/d^2."""
s2 = 1.0 / d ** 2
edges = np.quantile(s2, np.linspace(0, 1, n + 1))
return np.clip(np.searchsorted(edges, s2, side="right") - 1, 0, n - 1)
def rugnux_intensities(mtz, t, cell_m, sg_m):
"""IMEAN of Rugnux's merge reindexed by t and reduced to the model group's asymmetric unit."""
i = np.array(mtz.column_with_label("IMEAN"), copy=False).astype(float)
s = np.array(mtz.column_with_label("SIGIMEAN"), copy=False).astype(float)
good = np.isfinite(i) & np.isfinite(s) & (s > 0)
hkl, src = model_check.expand_p1(mtz)
hm = hkl @ t
integral = np.all(np.abs(hm - np.round(hm)) < 1e-3, axis=1) & good[src]
hm, src = np.round(hm[integral]).astype(int), src[integral]
hkl, i, _, _ = model_check.to_group(hm, i[src], s[src], np.zeros(len(src), int), cell_m, sg_m, src)
return hkl, i
def deposited_intensities(path, wavelength, cell_m, sg_m):
"""(hkl, I, kind) of the entry's merged data in the model's asymmetric unit, or None, why."""
blocks = []
for rb in gemmi.as_refln_blocks(gemmi.cif.read(path)):
labels = rb.column_labels()
if rb.is_unmerged():
continue
for kind, cols in (("I", ("intensity_meas", "intensity_sigma")),
("F", ("F_meas_au", "F_meas_sigma_au")),
("I", ("pdbx_I_plus", "pdbx_I_minus")),
("F", ("pdbx_F_plus", "pdbx_F_minus"))):
if all(c in labels for c in cols):
blocks.append((rb, kind, cols))
break
if not blocks:
return None, "no measured data in the deposited structure factors"
rb, kind, cols = blocks[0]
if len(blocks) > 1 and wavelength:
near = [abs(b[0].wavelength - wavelength) if b[0].wavelength else 9.0 for b in blocks]
if min(near) < 0.02:
rb, kind, cols = blocks[int(np.argmin(near))]
if not np.allclose(rb.cell.parameters, cell_m.parameters, rtol=0.02, atol=0.5):
return None, "deposited structure factors in another cell"
a = np.array(rb.make_float_array(cols[0]))
if cols[0].startswith("pdbx_"): # anomalous pair: the mean of the two hands
b = np.array(rb.make_float_array(cols[1]))
a = np.where(np.isfinite(a) & np.isfinite(b), (a + b) / 2, np.where(np.isfinite(a), a, b))
x = a * a if kind == "F" else a
ok = np.isfinite(x)
if "status" in rb.column_labels(): # '-', 'x', '<': not measured
st = [v.strip("'\"") for v in rb.block.find_values("_refln.status")]
if len(st) == len(x):
ok &= np.array([v not in ("-", "x", "<") for v in st])
hkl = np.array(rb.make_miller_array())[ok]
x = x[ok]
d_min = float(rb.cell.calculate_d_array(hkl).min())
hkl, x, _, _ = model_check.to_group(hkl, x, np.ones(len(x)), np.zeros(len(x), int), cell_m, sg_m,
np.arange(len(x)))
return (hkl, x, kind, d_min), None
def lookup(keys, hkl):
"""Index into `keys` (sorted) of each reflection of hkl, and which of them were found."""
k = model_check.hkl_key(hkl)
pos = np.clip(np.searchsorted(keys, k), 0, len(keys) - 1)
return pos, keys[pos] == k
def check(mtz_path, pdb_id, cache_dir=model_check.DEFAULT_CACHE, wavelength=None):
res = {k: None for k in KEYS}
res["dep_status"] = "failed"
pdb_id = pdb_id.lower()
if not model_check.PDB_ID.match(pdb_id):
return dict(res, dep_status="n/a", dep_reason="not a PDB entry")
try:
meta = model_check.deposition(pdb_id, cache_dir)
if meta is None:
return dict(res, dep_reason="entry could not be downloaded")
sf = model_check.fetch(model_check.RCSB + pdb_id + "-sf.cif.gz",
os.path.join(cache_dir, pdb_id + "-sf.cif.gz"))
if sf is None:
return dict(res, dep_status="n/a", dep_reason="no deposited structure factors")
st = gemmi.read_structure(meta["xyz"])
sg_m, cell_m = st.find_spacegroup(), st.cell
if sg_m is None:
return dict(res, dep_reason=f"model space group '{st.spacegroup_hm}' not recognised")
dep, why = deposited_intensities(sf, wavelength, cell_m, sg_m)
if dep is None:
return dict(res, dep_status="n/a", dep_reason=why)
hkl_d, i_d, res["dep_kind"], d_dep = dep
res["dep_d_min"] = round(d_dep, 3)
mtz = gemmi.read_mtz_file(mtz_path)
fc_hkl, fc = model_check.fcalc(st, mtz.resolution_high() - 0.005)
order = np.argsort(model_check.hkl_key(fc_hkl))
fc_keys, fc_hkl, fc2 = model_check.hkl_key(fc_hkl)[order], fc_hkl[order], fc[order] ** 2
fc_d = cell_m.calculate_d_array(fc_hkl)
best = None # the setting: best CC(I, |Fc|^2) at 3-6 A (everything if that is too few)
for t in model_check.basis_changes(mtz.cell, mtz.spacegroup, cell_m, sg_m):
hkl, i = rugnux_intensities(mtz, t, cell_m, sg_m)
pos, found = lookup(fc_keys, hkl)
sel = found.copy()
sel[found] &= (fc_d[pos[found]] >= 3) & (fc_d[pos[found]] <= 6)
if sel.sum() < 200:
sel = found
c = cc(i[sel], fc2[pos[sel]])
if c is not None and (best is None or c > best[0]):
best = (c, hkl, i)
if best is None or best[0] < 0.1: # a model-only |Fc| correlates weakly at 4-6 A
return dict(res, dep_reason="the merge does not match the model in any setting")
_, hkl_r, i_r = best
pos, found = lookup(fc_keys, hkl_r)
i_r, f_r, d_r = i_r[found], fc2[pos[found]], fc_d[pos[found]]
k_r = fc_keys[pos[found]]
_, a, b = np.intersect1d(k_r, model_check.hkl_key(hkl_d), return_indices=True)
ir, idp, fcc, d = i_r[a], i_d[b], f_r[a], d_r[a]
sel = d < D_COMMON
if sel.sum() < SHELLS * 50:
sel = np.ones(len(d), bool)
ir, idp, fcc, d = ir[sel], idp[sel], fcc[sel], d[sel]
res["dep_n_common"] = int(len(d))
if len(d) < SHELLS * 20:
return dict(res, dep_reason=f"only {len(d)} reflections in common")
res["dep_cc_delta_all"] = round(cc(ir, fcc) - cc(idp, fcc), 4)
sh = shells(d, SHELLS)
res["dep_cc_delta_outer"] = round(float(np.mean(
[cc(ir[sh == k], fcc[sh == k]) - cc(idp[sh == k], fcc[sh == k]) for k in (SHELLS - 1, SHELLS - 2)])), 4)
past = d_r < d_dep - 1e-3 # ours only, past the deposited data's limit
if past.sum() >= 60:
n = max(1, min(3, int(past.sum() // 300)))
sh = shells(d_r[past], n)
out = sh == n - 1
res["dep_beyond_cc"] = round(cc(i_r[past][out], f_r[past][out]), 4)
res["dep_beyond_d"] = round(float(d_r[past][out].min()), 3)
res["dep_status"] = "ok"
return res
except Exception as e: # one bad entry must not stop a battery
if os.environ.get("MODEL_CHECK_RAISE"):
raise
return dict(res, dep_reason=f"{type(e).__name__}: {e}"[:300])
def check_set(wd, set_id, cache_dir=model_check.DEFAULT_CACHE):
"""check() on a battery work directory (p.mtz, and the wavelength from p_report.txt)."""
try:
wl = float(score.read_report(os.path.join(wd, "p_report.txt")).get("WAVELENGTH"))
except (TypeError, ValueError):
wl = None
return check(os.path.join(wd, "p.mtz"), set_id.split("_")[0], cache_dir, wl)
def main():
ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
ap.add_argument("mtz", help="Rugnux merged MTZ (needs IMEAN, SIGIMEAN)")
ap.add_argument("pdb_id")
ap.add_argument("--cache", default=model_check.DEFAULT_CACHE)
ap.add_argument("--wavelength", type=float, help="picks the deposited block (multi-wavelength entries)")
a = ap.parse_args()
print(json.dumps(check(a.mtz, a.pdb_id, a.cache, a.wavelength), indent=1))
if __name__ == "__main__":
main()