Files
Jungfraujoch/image_analysis/structure_refinement/ModelScaling.cpp
T
leonarski_fandClaude Opus 5.5 f956eb25e8 rugnux: model validation does the same work in less time
--model validation (battery-only for users) was 18% of the battery's time. Every
number it produces is unchanged to the bit (p.mtz, maps, placed model and every
model-validation line of the report md5/diff-identical on 11 open sets); only
when and where the work runs changes:

- The bulk-solvent grid fit (FitModelScale, most of the CPU time) fits each
  solvent pair on a copy of gemmi::Scaling's target that takes
  |Fcalc + k_sol exp(-b_sol s^2) Fmask| once per pair instead of at every
  solver evaluation; same expressions, same types (new test checks a grid
  point against gemmi's own Scaling fit with ==).
- Fcalc density and the solvent mask are made on two threads; the model's
  structure factors beside the GPU engine reservation.
- The indexing probe fits the relabellings concurrently.
- The null's replicates run beside the real model's placement (they start
  from a snapshot of the model as read); one GPU engine per replicate plus
  one for the real fit instead of a cap of 4 (engines are interchangeable
  and deterministic).
- The 2mFo-DFc, mFo-DFc and anomalous maps are made and written
  concurrently; the placed model is written beside the reflection files.
- A rigid-body zone whose solvent-mask grid needs gemmi's shrink is sent to
  the CPU when the engines are reserved (ModelMaskGPU::ShrinkIsNoOp), instead
  of failing on the GPU and validating everything again on the CPU - the
  same CPU result, without the wasted first attempt.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SVmAWnzCmRKAXVUCdc4iNi
2026-10-09 00:01:14 +02:00

212 lines
9.2 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "ModelScaling.h"
#include <algorithm>
#include <cmath>
#include <vector>
#include "../../common/ParallelFor.h"
namespace {
// gemmi::Scaling<float> as a grid point fits it: the same parameters, model values, derivatives
// and R. The solvent pair is fixed there, so |Fcalc + k_sol exp(-b_sol s^2) Fmask| of a reflection
// is the same at every evaluation of the solver; it is taken once per grid point here rather than
// at every evaluation, where it was most of the cost of the fit. Every expression is the one
// scaling.hpp evaluates, in the same types, so every number is the same to the bit.
struct FixedSolventFit {
struct Point {
gemmi::Miller hkl;
float fobs;
float fcalc_abs; // std::abs(Scaling::get_fcalc(p))
double get_y() const { return fobs; }
double get_weight() const { return 1.0; }
};
const std::vector<gemmi::Vec6> &constraint_matrix;
double k_overall = 1.0;
gemmi::SMat33<double> b_star{0, 0, 0, 0, 0, 0};
std::vector<Point> points;
explicit FixedSolventFit(const gemmi::Scaling<float> &s) : constraint_matrix(s.constraint_matrix) {}
// The reflections at the solvent pair of `s`, and its scale as the start of the fit.
void Start(const gemmi::Scaling<float> &s) {
k_overall = s.k_overall;
b_star = s.b_star;
points.resize(s.points.size());
for (size_t i = 0; i < points.size(); ++i)
points[i] = {s.points[i].hkl, s.points[i].fobs, std::abs(s.get_fcalc(s.points[i]))};
}
// Scaling::get_parameters(), set_parameters(), get_overall_scale_factor(), compute_value() and
// compute_value_and_derivatives(), with k_sol and b_sol fixed.
std::vector<double> get_parameters() const {
std::vector<double> ret;
ret.push_back(k_overall);
for (const gemmi::Vec6 &v : constraint_matrix)
ret.push_back(gemmi::vec6_dot(v, b_star));
return ret;
}
void set_parameters(const double *p) {
k_overall = p[0];
int n = 0;
b_star = {0, 0, 0, 0, 0, 0};
for (const gemmi::Vec6 &row : constraint_matrix) {
double d = p[++n];
b_star.u11 += row[0] * d;
b_star.u22 += row[1] * d;
b_star.u33 += row[2] * d;
b_star.u12 += row[3] * d;
b_star.u13 += row[4] * d;
b_star.u23 += row[5] * d;
}
}
void set_parameters(const std::vector<double> &p) { set_parameters(p.data()); }
double get_overall_scale_factor(const gemmi::Miller &hkl) const {
return k_overall * std::exp(-0.25 * b_star.r_u_r(hkl));
}
double compute_value(const Point &p) const {
return p.fcalc_abs * (float) get_overall_scale_factor(p.hkl);
}
double compute_value_and_derivatives(const Point &p, std::vector<double> &dy_da) const {
gemmi::Vec3 h(p.hkl);
double kaniso = std::exp(-0.25 * b_star.r_u_r(h));
double fcalc_abs = p.fcalc_abs;
int n = 1;
double fe = fcalc_abs * kaniso;
double y = k_overall * fe;
dy_da[0] = fe;
gemmi::SMat33<double> du = {
-0.25 * y * (h.x * h.x),
-0.25 * y * (h.y * h.y),
-0.25 * y * (h.z * h.z),
-0.5 * y * (h.x * h.y),
-0.5 * y * (h.x * h.z),
-0.5 * y * (h.y * h.z),
};
for (size_t j = 0; j < constraint_matrix.size(); ++j)
dy_da[n + j] = gemmi::vec6_dot(constraint_matrix[j], du);
return y;
}
};
// R-factor of the current parameters over the fitted reflections. This is what the grid is
// selected on, and it is the quantity the scale exists to make small.
double RFactor(const FixedSolventFit &fit) {
double num = 0, den = 0;
for (const auto &p : fit.points) {
num += std::fabs(p.fobs - fit.compute_value(p));
den += p.fobs;
}
return den > 0 ? num / den : 1.0;
}
} // namespace
// Following the phenix bulk-solvent and scaling procedure: k_sol and b_sol by a grid search, with
// the overall scale and the anisotropic B refitted at every grid point - Afonine, Grosse-Kunstleve
// & Adams, Acta Cryst. D61, 850-855, 2005, which searches b_sol over 10-80 A^2 in steps of 5.
// The fit is unweighted, as in both phenix and Refmac (Murshudov, Skubak, Lebedev, Pannu, Steiner,
// Nicholls, Winn, Long & Vagin, Acta Cryst. D67, 355-367, 2011, eq. 11). The physical range and the
// starting values are those of Fokine & Urzhumtsev, Acta Cryst. D58, 1387-1392, 2002.
//
// The point of the grid is that k_sol and b_sol cannot leave the physical box: gemmi's own
// fit_parameters() is an unbounded Levenberg-Marquardt, and on this corpus it reached b_sol of
// 1707 A^2 - a solvent term switched off in all but the lowest-resolution shell. Here the solvent
// pair is held fixed at each grid point and only the overall scale and the symmetry-constrained
// anisotropic B are refined, which is the well-conditioned half of the problem and is left to
// gemmi's solver rather than reimplemented.
//
// Every grid point is its own fit: fit_isotropic_b_approximately() sets k_overall and b_star from the
// data and the point's solvent pair alone, so a point does not depend on the one fitted before it, and
// the points run in parallel, each chunk on its own copy of `scaling`. The winner is then read off in
// grid order with the serial rule (lowest finite R, the first on a tie), so the answer is the serial
// loop's bit for bit. The one exception is fit_isotropic_b_approximately() finding five or fewer
// reflections to fit on - it then returns without setting anything and a point WOULD start from where
// the previous one ended - so there the grid is walked in order, on `scaling` itself, as it always was.
ModelScaleReport FitModelScale(gemmi::Scaling<float> &scaling, ModelScaleBox box, size_t nthreads) {
ModelScaleReport report;
report.n_points = static_cast<int>(scaling.points.size());
if (scaling.points.size() < 20)
return report;
const bool had_solvent = scaling.use_solvent;
scaling.fix_k_sol = true; // the grid owns the solvent pair; the solver never sees it
scaling.fix_b_sol = true;
// The reflections fit_isotropic_b_approximately() fits on (its own filter).
int n_isotropic = 0;
for (const auto &p : scaling.points)
if (!(p.fobs < 1 || p.fobs < p.sigma))
++n_isotropic;
const bool independent = n_isotropic > 5;
double best_r = -1, best_k_sol = 0.35, best_b_sol = 46.0, best_k_overall = 1.0;
gemmi::SMat33<double> best_b_star{0, 0, 0, 0, 0, 0};
struct PointFit { double k_sol, b_sol, r = NAN, k_overall = 1.0; gemmi::SMat33<double> b_star{0, 0, 0, 0, 0, 0}; };
auto fit_point = [](gemmi::Scaling<float> &s, FixedSolventFit &fit, PointFit &pf) {
s.k_sol = pf.k_sol;
s.b_sol = pf.b_sol;
s.fit_isotropic_b_approximately(); // a fresh starting point for this solvent pair
fit.Start(s);
gemmi::LevMar levmar;
levmar.fit(fit); // k_overall + anisotropic B only
s.k_overall = fit.k_overall;
s.b_star = fit.b_star;
pf.r = RFactor(fit);
pf.k_overall = fit.k_overall;
pf.b_star = fit.b_star;
};
auto try_points = [&](std::vector<PointFit> &pts) {
if (independent)
ParallelChunks(static_cast<int>(pts.size()), nthreads, [&](int lo, int hi) {
gemmi::Scaling<float> local = scaling;
FixedSolventFit fit(local);
for (int i = lo; i < hi; ++i)
fit_point(local, fit, pts[i]);
});
else {
FixedSolventFit fit(scaling);
for (auto &pf : pts)
fit_point(scaling, fit, pf);
}
for (const auto &pf : pts) {
++report.n_grid;
// A diverged fit gives r = NaN; latched as best_r it wins every later r < best_r.
if (std::isfinite(pf.r) && (best_r < 0 || pf.r < best_r)) {
best_r = pf.r;
best_k_sol = pf.k_sol;
best_b_sol = pf.b_sol;
best_k_overall = pf.k_overall;
best_b_star = pf.b_star;
}
}
};
// Coarse pass over the whole box, then one refinement pass around the winner.
std::vector<PointFit> coarse;
for (double ks = box.k_lo; ks <= box.k_hi + 1e-9; ks += 0.05)
for (double bs = box.b_lo; bs <= box.b_hi + 1e-9; bs += 10.0)
coarse.push_back(PointFit{ks, bs});
try_points(coarse);
const double k0 = best_k_sol, b0 = best_b_sol;
const double k_hi2 = std::min(box.k_hi, k0 + 0.05);
const double b_hi2 = std::min(box.b_hi, b0 + 10.0);
std::vector<PointFit> fine;
for (double ks = std::max(box.k_lo, k0 - 0.05); ks <= k_hi2 + 1e-9; ks += 0.025)
for (double bs = std::max(box.b_lo, b0 - 10.0); bs <= b_hi2 + 1e-9; bs += 5.0)
fine.push_back(PointFit{ks, bs});
try_points(fine);
scaling.k_sol = best_k_sol;
scaling.b_sol = best_b_sol;
scaling.k_overall = best_k_overall;
scaling.b_star = best_b_star;
scaling.use_solvent = had_solvent;
report.r_work_fit = best_r;
return report;
}