Files
Jungfraujoch/tools/battery/score.py
T
leonarski_f a395f358ef
Build Packages / Create release (push) Successful in 17s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m22s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m37s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 9m33s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 10m39s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 11m4s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 13m19s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 17m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 18m49s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 19m10s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m26s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m31s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 18m54s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m45s
Build Packages / Generate python client (push) Successful in 37s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 20m20s
Build Packages / Build documentation (push) Successful in 1m32s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m37s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m6s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 19m49s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 20m29s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 17m2s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 14m27s
Build Packages / Unit tests (push) Successful in 1h18m12s
1.0.0-rc.174 (#84)
* Rugnux: Performance improvements on GPU and CPU (more of the pre-scan and of scaling on the GPU, faster CPU spot finding and crystal refinement), with unchanged results.
* Rugnux: More robust processing - patches of persistently hot pixels are masked, an inconsistent merge triggers a retry at the measured beam centre, and builds targeting different CPU levels give the same results.
* Rugnux: Improved scaling and merging - reflections with an overloaded pixel are dropped, as in XDS, sparse rotation sweeps are scaled more reliably, and French-Wilson amplitudes use an anisotropic Wilson prior.
* Rugnux: Improved space-group determination - glide planes in groups without a centre of symmetry, screw axes from short or weak axial rows kept when a higher group is adopted, and more reliable decisions on twinned and pseudo-symmetric crystals.
* Rugnux: Improved small-molecule processing - spots that grow wider than the integration disk and split spots are integrated over their measured footprint, sparse lattices are integrated on every frame, and the `.hkl` file holds unmerged scaled reflections (SHELX HKLF 4).
* Rugnux: Reads Rigaku d*TREK SMV images (Saturn CCD), including detector 2theta and encoded pixel overflows; home-source (rotating-anode) datasets were added to the validation battery.
* jfjoch_viewer: Fixed processing failing at the end with "Wrong JPEG library version" on Linux; the merge window shows the space group with proper subscripts and a checklist of crystal pathologies.

Reviewed-on: #84
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-10-06 14:03:18 +02:00

376 lines
21 KiB
Python

"""Scoring one rugnux run against its reference.
One dataset, one verdict, one cause, decided in this order: did it run -> is the lattice right ->
is the symmetry right. Data quality is not scored: it is reported for a human to judge. The order matters:
a halved axis also makes the screw along it unobservable, and counting that row twice would hide
the defect that cost the reflections.
A row may accept more than one reference (`ref_alternatives`, see `alternatives`): where two
descriptions of the same crystal are both defensible and nothing available to us decides between
them, either answer passes, and the report says which rows those are.
"""
import itertools
import os
import re
import gemmi
import sgequiv
REPORT_LINE = re.compile(r"^([A-Z_0-9]+)=\s*(.*)$")
def read_report(path):
"""KEY= value lines of a rugnux _report.txt, plus the lowest-resolution shell of its two shell
tables: LOWRES_D / LOWRES_R_MEAS from the table over its own range, REFRES_LOWRES_D /
REFRES_LOWRES_R_MEAS from the one over the reference range (it follows REFRES_RANGE=)."""
out = {}
if not os.path.exists(path):
return out
header = None
for line in open(path, errors="replace"):
line = line.rstrip("\n")
m = REPORT_LINE.match(line)
if m:
out[m.group(1)] = m.group(2).strip()
continue
f = line.lstrip("#").split()
if f[:2] == ["d_min", "N_obs"]:
header = f
elif header and f and re.match(r"^\d+\.\d+$", f[0]):
# Three tables in report order: the run's own, the --report-resolution one, and the
# per-hand (FRIEDEL'S_LAW=FALSE) one, whose keys are written just above it.
prefix = ("ANOM_" if "ANOM_UNIQUE_REFLECTIONS" in out
else "REFRES_" if "REFRES_RANGE" in out else "")
out[prefix + "LOWRES_D"] = f[0]
r_meas = f[header.index("R_meas")].rstrip("%") # "3.7%", or "-" for none
out[prefix + "LOWRES_R_MEAS"] = "" if r_meas == "-" else str(round(float(r_meas) / 100, 4))
header = None # only the first row of each table
return out
# REFRES_* keys of the report read as numbers, stored lower-case under the same name
REFRES_KEYS = ("refres_shells_past_limit", "refres_unique_reflections", "refres_completeness",
"refres_multiplicity", "refres_i_over_sigma", "refres_r_meas", "refres_cc_half",
"refres_isa", "refres_r_meas_weighted")
def fnum(s):
try:
return float(s.split()[0])
except (AttributeError, ValueError, IndexError):
return None
def space_group(name, sgno):
"""The space group in the SETTING the cell was given in: by name when there is one (I 1 2 1 and
C 1 2 1 are both number 5, with different centring vectors), else the standard setting."""
sg = gemmi.find_spacegroup_by_name(name) if name else None
return sg or gemmi.find_spacegroup_by_number(int(sgno or 1))
def primitive_reduced(cell, sg):
"""Niggli-reduced PRIMITIVE cell of a cell given in space group sg (a gemmi.SpaceGroup), and
its volume.
Setting-invariant: two settings of one lattice (C2 vs I2, a/c swapped, beta vs 180-beta)
reduce to the same cell, so they compare equal."""
gv = gemmi.GruberVector(gemmi.UnitCell(*cell), sg)
gv.niggli_reduce()
red = gv.cell_parameters()
return red, gemmi.UnitCell(*red).volume
def lattice_match(cell, sg, ref_cell, ref_sg):
"""(primitive volume ratio, largest relative deviation of the reduced edges)."""
red, vol = primitive_reduced(cell, sg)
ref_red, ref_vol = primitive_reduced(ref_cell, ref_sg)
edge_dev = max(abs(a - b) / b for a, b in zip(sorted(red[:3]), sorted(ref_red[:3])))
return vol / ref_vol, edge_dev
def cell_dev_pct(cell, ref):
"""Largest relative deviation of a, b, c in percent, minimised over the six axis orders, so
P212121 with a and c exchanged is not scored as a 6% error."""
return min(max(abs(p[i] - ref[i]) / ref[i] * 100 for i in range(3))
for p in itertools.permutations(cell[:3]))
# XDS prints CC1/2 in % with one decimal; its value is taken at the bottom of that rounding, so a
# printed 100.0 does not demand a perfect CC1/2 of rugnux
XDS_CC_HALF_ROUNDING = 0.0005
def cc_half_noise_ratio(cc, cc_xds):
"""rugnux's half-set noise over XDS's, both read off CC1/2 over XDS's range; None without both.
CC1/2 = S / (S + E), with S the variance of the signal and E that of the half-set error, so
E / S = 1 / CC1/2 - 1 at any CC1/2. E goes as 1 / (observations): a ratio of 2 is a merge as
noisy as XDS's would be with half of its observations. Reported, not scored.
A CC1/2 at or below 0 is no signal at all, counted as 0.001."""
if cc is None or cc_xds is None:
return None
return round((1 / max(cc, 0.001) - 1) / (1 / (cc_xds - XDS_CC_HALF_ROUNDING) - 1), 3)
_AXIS_ROW = {"a": (1, 0, 0), "b": (0, 1, 0), "c": (0, 0, 1)}
def screw_axes_differing(sg, ref_sg):
"""The axes on which two space groups disagree about a screw: the principal row one of them
extinguishes and the other does not.
Only ever asked of two groups of the SAME point group on the same lattice, so their centrings
are the same and a row a centring already kills is killed in both - what is left is the screw."""
out = set()
for ax, hkl in _AXIS_ROW.items():
a, b = (s.operations().is_systematically_absent(hkl) for s in (sg, ref_sg))
if a != b:
out.add(ax)
return out
def point_group(sgno):
return gemmi.find_spacegroup_by_number(int(sgno)).point_group_hm()
def sg_name(sgno):
return gemmi.find_spacegroup_by_number(int(sgno)).hm if sgno else None
def alternatives(entry):
"""The other answers a row accepts, from its `ref_alternatives`.
A knife-edge row - one where the deposition and our reduction disagree, the disagreement is
real, and no test available to us settles it - says so here instead of being quietly overridden
to our answer or dropped from scoring. Each alternative replaces the reference fields it names
(a space group, a cell, or both) and the row passes if our answer matches ANY of the
references, the deposited one included. `ref` keeps the deposition verbatim either way.
Every alternative must carry `why`: a reason a reader can check. Without one this would be a
way to launder a failure into a pass, so it is a schema error, not a silent pass."""
out = []
for i, alt in enumerate(entry.get("ref_alternatives") or []):
where = f"{entry.get('id', '?')}: ref_alternatives[{i}]"
if not (alt.get("why") or "").strip():
raise ValueError(f"{where} has no 'why'; an accepted alternative must state why both "
"answers are acceptable")
if not set(alt) & {"sg", "sgno", "cell"}:
raise ValueError(f"{where} replaces nothing; give the sg, the cell, or both")
out.append(alt)
return out
def alt_label(alt):
"""How an accepted alternative is named in one line: its space group, its cell, or both."""
bits = []
if alt.get("sg") or alt.get("sgno"):
bits.append(alt.get("sg") or sg_name(alt["sgno"]))
if alt.get("cell"):
bits.append("cell " + " ".join(f"{x:g}" for x in alt["cell"][:3]))
return ", ".join(bits)
def judge(entry, rep, run_note):
"""Score one set. entry is its manifest row (with 'ref'), rep its parsed report, run_note
what the runner saw ('', 'timeout', 'exit N: message', 'no input')."""
ref = dict(entry.get("ref") or {})
if entry.get("ref_override"): # a reference measured by hand, replacing the file's
ref.update(entry["ref_override"])
alts = alternatives(entry) # raises on a row that accepts an answer for no reason
r = {
"sgno": None, "sg": None, "pg": None,
"sgno_ref": ref.get("sgno"), "sg_ref": ref.get("sg") or sg_name(ref.get("sgno")),
"pg_ref": point_group(ref["sgno"]) if ref.get("sgno") else None,
"cell": None, "cell_ref": ref.get("cell"), "cell_dev_pct": None, "volume_ratio": None,
"d_min": None, "d_min_ref": ref.get("dmin"), "d_min_ref_rule": ref.get("dmin_rule"),
"d_min_xds": ref.get("dmin_xds"), "res_gain_pct": None,
"r_meas": fnum(rep.get("R_MEAS")), "cc_half": fnum(rep.get("CC_HALF")),
# R_MEAS is counted as XDS counts it; the weighted one weights each observation as the
# merge does (absent from reports before the weighting)
"r_meas_weighted": fnum(rep.get("R_MEAS_WEIGHTED")),
"isa": fnum(rep.get("ISA")), "completeness": fnum(rep.get("COMPLETENESS")),
"mosaicity_deg": fnum(rep.get("MOSAICITY_DEG")), "extra_lattice_intensity_pct": fnum(rep.get("EXTRA_LATTICE_INTENSITY_PCT")),
"multiplicity": fnum(rep.get("MULTIPLICITY")), "i_over_sigma": fnum(rep.get("I_OVER_SIGMA")),
"indexing_rate": fnum(rep.get("INDEXING_RATE")), "images": fnum(rep.get("IMAGES_PROCESSED")),
"rugnux_wall_s": fnum(rep.get("WALL_TIME")), "rugnux_verdict": rep.get("VERDICT_TEXT"),
"r_meas_ref": ref.get("r_meas"), "cc_half_ref": ref.get("cc_half"), "isa_ref": ref.get("isa"),
"completeness_ref": ref.get("completeness"), "multiplicity_ref": ref.get("multiplicity"),
"sg_relation": None,
# rugnux's own fit of the deposited model (--model; the open arm, and the in-house sets that
# name a model): a rigid-body placement scored on rugnux's own free set, so a trend number,
# not the depositor's R-free
"rfree": fnum(rep.get("R_FREE")), "rwork": fnum(rep.get("R_WORK")),
# the anomalous difference map's mean height (sigma) at the model's anomalous scatterers
# (P and heavier: S, metals, Cl, I, ...) - how much anomalous signal the merge carries
"anom_sigma": fnum(rep.get("ANOMALOUS_SCATTERER_MEAN_SIGMA")),
"model_fit": rep.get("MODEL_FIT"), "cc_model": fnum(rep.get("CC_MODEL_OVERALL")),
# R over ALL the reflections, with and without one free scale per resolution shell, and the
# size of that rescale. rfree above describes one run and is not comparable with another's:
# the model's scale can only bend as k*exp(-B s^2), so a change in the amplitudes' radial
# profile alone moves it. rmodel_shell_scaled has that taken out and IS comparable; it is
# the R to read in a delta table. radial_misfit says how much there was to take out.
"rmodel": fnum(rep.get("R_MODEL")), "rmodel_shell_scaled": fnum(rep.get("R_MODEL_SHELL_SCALED")),
"radial_misfit": fnum(rep.get("MODEL_RADIAL_MISFIT")),
"sg_label": None, "accepted_alt": None, "accepted_alt_why": None,
}
# the same merge binned over the reference's range (--report-resolution, XDS arms)
r.update({k: fnum(rep.get(k.upper())) for k in REFRES_KEYS})
r["refres_range"] = rep.get("REFRES_RANGE")
# An XDS reference measured with FRIEDEL'S_LAW=FALSE counts its uniques, completeness,
# multiplicity, R_meas and CC1/2 per Bijvoet hand, so what is like for like is the per-hand
# table of the same rugnux merge - the Friedel-merged one has about twice the multiplicity.
# Both tables come out of one bare run: -A is presentation only and changes nothing that is
# compared here. ISa is not among them - there is one error model, fitted on the hands, and
# REFRES_ISA is it.
r["per_hand_reference"] = False
if ref.get("anomalous"):
for k in ("unique_reflections", "completeness", "multiplicity", "i_over_sigma",
"r_meas", "cc_half"):
v = fnum(rep.get("ANOM_REFRES_" + k.upper()))
if v is not None:
r["refres_" + k] = v
r["per_hand_reference"] = True
r["isa_ratio"] = round(r["refres_isa"] / r["isa_ref"], 4) if r["refres_isa"] and r["isa_ref"] else None
r["r_meas_ratio"] = (round(r["refres_r_meas"] / r["r_meas_ref"], 4)
if r["refres_r_meas"] and r["r_meas_ref"] else None)
# R_meas of the lowest-resolution shell: rugnux's own table, the reference-range table and XDS's
# (reported, not scored)
r.update(lowres_d=fnum(rep.get("LOWRES_D")), lowres_r_meas=fnum(rep.get("LOWRES_R_MEAS")),
refres_lowres_d=fnum(rep.get("REFRES_LOWRES_D")),
refres_lowres_r_meas=fnum(rep.get("REFRES_LOWRES_R_MEAS")),
lowres_d_ref=ref.get("dmin_low"), lowres_r_meas_ref=ref.get("r_meas_low"))
r["lowres_r_meas_ratio"] = (round(r["refres_lowres_r_meas"] / r["lowres_r_meas_ref"], 4)
if r["refres_lowres_r_meas"] and r["lowres_r_meas_ref"] else None)
r["cc_half_noise_ratio"] = cc_half_noise_ratio(r["refres_cc_half"], r["cc_half_ref"])
if rep.get("SPACE_GROUP_NUMBER"):
r["sgno"] = int(fnum(rep["SPACE_GROUP_NUMBER"]))
r["sg"] = rep.get("SPACE_GROUP_NAME") or sg_name(r["sgno"])
# With --model the reported group can be the model's enantiomorph, a label the data did not
# decide. The data's own answer is then the search's Sohncke group (same class up to hand).
if rep.get("SPACE_GROUP_ENANTIOMORPH") == "ASSUMED_FROM_MODEL" and rep.get("SOHNCKE_SPACE_GROUP"):
own = gemmi.find_spacegroup_by_name(rep["SOHNCKE_SPACE_GROUP"])
if own and own.number != r["sgno"] and sgequiv.indistinguishable(r["sgno"], own.number):
r["sg_label"] = r["sg"]
r["sgno"], r["sg"] = own.number, own.hm
r["pg"] = point_group(r["sgno"])
if rep.get("UNIT_CELL_CONSTANTS"):
r["cell"] = [float(x) for x in rep["UNIT_CELL_CONSTANTS"].split()[:6]]
rng = (rep.get("INCLUDE_RESOLUTION_RANGE") or "").split()
if len(rng) == 2:
r["d_min"] = float(rng[1])
if r["d_min"] and r["d_min_ref"]:
r["res_gain_pct"] = round((r["d_min_ref"] - r["d_min"]) / r["d_min_ref"] * 100, 1) + 0.0 # no -0.0
if r["cell"] and r["cell_ref"]:
r["cell_dev_pct"] = round(cell_dev_pct(r["cell"], r["cell_ref"]), 3)
def verdict(v, cause, reason):
r.update(verdict=v, cause=cause, reason=reason)
return r
have_lattice = r["sgno"] is not None and r["cell"] is not None
if entry.get("expect") == "no_lattice":
# a no-crystal control: the right answer is to refuse
if have_lattice:
return verdict("fail", "false_lattice",
f"reported a lattice ({r['sg']}) on a set with no crystal")
return verdict("pass", None, "no lattice reported, as expected")
if not have_lattice:
if run_note == "no input":
return verdict("not_run", "no_input", "input file not found")
if run_note == "timeout":
return verdict("fail", "timeout", "timed out")
if re.search(r"Error reading input|Cannot open", run_note):
return verdict("fail", "reader", run_note)
if re.search(r"index|lattice", run_note, re.I):
return verdict("fail", "indexing", run_note)
return verdict("fail", "crash", run_note or "no lattice in the report")
if not ref.get("sgno") or not ref.get("cell"):
return verdict("unscored", "no_reference", "no reference to score against")
if entry.get("unscored"): # a reference known to be wrong: run, but do not score
return verdict("unscored", "reference_problem", entry["unscored"])
def against(cand):
"""Score the answer against one reference: (the numbers it yields, None) if that reference
accepts the answer, (the numbers, (verdict, cause, reason)) if it does not."""
n = {"volume_ratio": None, "sg_relation": None}
vr, edge = lattice_match(r["cell"], space_group(rep.get("SPACE_GROUP_NAME"), r["sgno"]),
cand["cell"], space_group(cand.get("sg"), cand["sgno"]))
n["volume_ratio"] = round(vr, 4)
if not (0.95 < vr < 1.05 and edge < 0.02):
if 0.45 <= vr <= 0.55:
return n, ("fail", "lattice_halved", f"primitive volume ratio {vr:.2f}")
if 1.9 <= vr <= 2.1:
return n, ("fail", "lattice_doubled", f"primitive volume ratio {vr:.2f}")
return n, ("fail", "lattice_other",
f"primitive volume ratio {vr:.2f}, reduced edges off by {edge:.1%}")
# XDS never tests a screw axis, so against an XDS reference only the point group can be
# judged - unless the group was measured by hand (sg_measured).
by_sg = entry["arm"] == "open" or cand.get("sg_measured")
pg_cand = point_group(cand["sgno"])
if by_sg:
rel = sgequiv.describe(cand["sgno"], r["sgno"])
n["sg_relation"] = rel
ok = rel in ("match", "hand only (needs anomalous)", "UNDECIDABLE from intensities")
else:
ok = r["pg"] == pg_cand
if ok:
return n, None
sg_cand = cand.get("sg") or sg_name(cand["sgno"])
order = len(gemmi.find_spacegroup_by_number(r["sgno"]).operations().sym_ops)
order_ref = len(gemmi.find_spacegroup_by_number(cand["sgno"]).operations().sym_ops)
if r["pg"] == pg_cand:
# A screw the data never tested is not a wrong answer, it is an unanswerable question:
# the axial row it lives on was not recorded at all (the spindle's blind cone), so the
# run offers every member of the set as an alternative and writes the one claiming no
# screw, because a reflection file must carry one group. Scoring that as a failure
# measures the sweep, not the program - the same reason the hand of an enantiomorphic
# pair is not scored. Three conditions, all necessary: rugnux SAYS the axis is open,
# the reference is among the groups it offered, and the ONLY axis the two differ on is
# one it named. A screw called wrongly where the row WAS measured meets none of them
# and stays a failure.
# WHICH axis carries the disagreement is a property of the setting, and the two sides
# are named in different ones: the reference in the deposition's axis labels, ours - and
# SPACE_GROUP_SCREW_UNDETERMINED with it - in our own. Where the cells are a permutation
# of each other the two never meet: a deposited P 21 21 2 differs from our P 21 21 21 on
# c, while the same group written in our axes differs on b, the axis actually left open.
# So the axes are compared against the offered group, which is already in our setting;
# the reference being among the offered is a condition anyway, and a group offered under
# more than one setting passes if any of them fits.
undetermined = set((rep.get("SPACE_GROUP_SCREW_UNDETERMINED") or "NONE").split())
offered = [gemmi.find_spacegroup_by_name(x.strip())
for x in (rep.get("SPACE_GROUP_ALTERNATIVES") or "").split("|")]
ours = space_group(rep.get("SPACE_GROUP_NAME"), r["sgno"])
for o in (x for x in offered if x and x.number == cand["sgno"]):
differ = screw_axes_differing(ours, o)
if undetermined != {"NONE"} and differ and differ <= undetermined:
here = f" (here {o.hm})" if o.hm != sg_cand else ""
return n, ("unscored", "screw_undetermined",
f"{r['sg']} vs reference {sg_cand}{here}: the screw along "
f"{', '.join(sorted(differ))} is undeterminable from these data "
f"(offered {rep.get('SPACE_GROUP_ALTERNATIVES')})")
return n, ("fail", "sym_screw", f"{r['sg']} vs reference {sg_cand}")
cause = ("sym_under" if order < order_ref else
"sym_over" if order > order_ref else "sym_other")
return n, ("fail", cause, f"{r['sg']} vs reference {sg_cand}")
# The deposition (with any ref_override) decides, and its numbers are the ones reported: an
# accepted alternative changes the verdict, never what the row says about the deposition.
numbers, problem = against(ref)
r.update(numbers)
for alt in (alts if problem and problem[0] == "fail" else []):
if against(dict(ref, **{k: v for k, v in alt.items() if k != "why"}))[1] is None:
r["accepted_alt"], r["accepted_alt_why"] = alt_label(alt), alt["why"]
return verdict("pass", "accepted_alternative",
f"{r['sg']} vs reference {r['sg_ref']}: accepted alternative "
f"{r['accepted_alt']} - a knife-edge this battery does not decide")
if problem:
return verdict(*problem)
# Data quality (CC1/2, R_meas, resolution, ISa) is reported, never scored: that is for a human.
note = ""
if r["sgno"] != ref["sgno"]:
note = f"{r['sg']} vs reference {r['sg_ref']} ({r.get('sg_relation') or 'screws not judged against XDS'})"
return verdict("pass", None, note)