The merged reflections were written in one of mtz/cif/txt selected by --scaling-output. Write both an MTZ and an mmCIF unconditionally instead - each has its uses downstream (MTZ for the CCP4/phenix tools, mmCIF for deposition) - and remove the format selector, the plain-text .hkl writer, and the now-unused IntensityFormat enum / ScalingSettings::FileFormat plumbing. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
349 lines
15 KiB
C++
349 lines
15 KiB
C++
// SPDX-FileCopyrightText: 2025 Paul Scherrer Institute
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "WriteReflections.h"
|
|
#include "scale_merge/Merge.h"
|
|
#include "scale_merge/HKLKey.h"
|
|
#include "scale_merge/TwinningAnalysis.h"
|
|
|
|
#include <cmath>
|
|
#include <map>
|
|
#include <tuple>
|
|
#include <fstream>
|
|
#include <iomanip>
|
|
#include <sstream>
|
|
#include <stdexcept>
|
|
#include <ctime>
|
|
#include <chrono>
|
|
|
|
#include <gemmi/mtz.hpp>
|
|
|
|
#include "../common/GitInfo.h"
|
|
|
|
namespace {
|
|
|
|
/// Current date in ISO-8601 (YYYY-MM-DD) for the _audit block.
|
|
std::string CurrentDateISO() {
|
|
auto now = std::chrono::system_clock::now();
|
|
auto t = std::chrono::system_clock::to_time_t(now);
|
|
std::tm tm{};
|
|
#ifdef _WIN32
|
|
gmtime_s(&tm, &t);
|
|
#else
|
|
gmtime_r(&t, &tm);
|
|
#endif
|
|
char buf[32];
|
|
std::strftime(buf, sizeof(buf), "%Y-%m-%d", &tm);
|
|
return buf;
|
|
}
|
|
|
|
/// Format a double with given decimal places; returns "?" for non-finite.
|
|
std::string Fmt(double val, int decimals = 4) {
|
|
if (!std::isfinite(val))
|
|
return "?";
|
|
std::ostringstream ss;
|
|
ss << std::fixed << std::setprecision(decimals) << val;
|
|
return ss.str();
|
|
}
|
|
|
|
/// Quote a CIF string value; returns "?" for empty.
|
|
std::string CifStr(const std::string& s) {
|
|
if (s.empty())
|
|
return "?";
|
|
// If it contains spaces or special chars, single-quote it
|
|
if (s.find(' ') != std::string::npos ||
|
|
s.find('\'') != std::string::npos ||
|
|
s.find('#') != std::string::npos)
|
|
return "'" + s + "'";
|
|
return s;
|
|
}
|
|
|
|
} // namespace
|
|
|
|
|
|
void WriteMmcifReflections(const std::vector<MergedReflection> &reflections,
|
|
const UnitCell &unitCell,
|
|
const DiffractionExperiment &experiment,
|
|
const MergeStatistics &statistics,
|
|
const std::string &isa,
|
|
const TwinningAnalysisResult &twinning,
|
|
const std::string &filename) {
|
|
|
|
std::ofstream out(filename);
|
|
if (!out)
|
|
throw std::runtime_error("WriteMmcifReflections: cannot open " + filename);
|
|
|
|
out << std::fixed;
|
|
|
|
// ---------- data block ----------
|
|
out << "data_sample" << "\n";
|
|
out << "#\n";
|
|
|
|
// ---------- _audit ----------
|
|
out << "_audit.revision_id 1\n";
|
|
out << "_audit.creation_date " << CurrentDateISO() << "\n";
|
|
out << "_audit.update_record 'Initial release'\n";
|
|
out << "#\n";
|
|
|
|
// ---------- _software ----------
|
|
out << "_software.name 'Jungfraujoch'\n";
|
|
|
|
out << "_software.version " << CifStr(jfjoch_version()) << "\n";
|
|
out << "_software.classification reduction\n";
|
|
out << "#\n";
|
|
|
|
// ---------- _cell ----------
|
|
out << "_cell.length_a " << Fmt(unitCell.a, 3) << "\n";
|
|
out << "_cell.length_b " << Fmt(unitCell.b, 3) << "\n";
|
|
out << "_cell.length_c " << Fmt(unitCell.c, 3) << "\n";
|
|
out << "_cell.angle_alpha " << Fmt(unitCell.alpha, 2) << "\n";
|
|
out << "_cell.angle_beta " << Fmt(unitCell.beta, 2) << "\n";
|
|
out << "_cell.angle_gamma " << Fmt(unitCell.gamma, 2) << "\n";
|
|
|
|
auto *sg = gemmi::find_spacegroup_by_number(experiment.GetSpaceGroupNumber().value_or(1));
|
|
if (sg == nullptr)
|
|
throw std::runtime_error("WriteMmcifReflections: invalid space group number");
|
|
|
|
// ---------- _symmetry ----------
|
|
out << "_symmetry.space_group_name_H-M " << CifStr(sg->hm) << "\n";
|
|
out << "_symmetry.Int_Tables_number " << sg->number << "\n";
|
|
out << "#\n";
|
|
|
|
// ---------- _diffrn_source / _diffrn_detector ----------
|
|
if (!experiment.GetSourceName().empty())
|
|
out << "_diffrn_source.pdbx_synchrotron_site " << CifStr(experiment.GetSourceName()) << "\n";
|
|
|
|
if (!experiment.GetInstrumentName().empty())
|
|
out << "_diffrn_source.pdbx_synchrotron_beamline " << CifStr(experiment.GetInstrumentName()) << "\n";
|
|
|
|
out << "_diffrn_radiation_wavelength.wavelength " << Fmt(experiment.GetWavelength_A(), 5) << "\n";
|
|
out << "_diffrn_detector.detector " << CifStr(experiment.GetDetectorDescription()) << "\n";
|
|
out << "#\n";
|
|
|
|
// ---------- merging statistics (_reflns overall + _reflns_shell loop) ----------
|
|
// cc_half and r_meas are stored as fractions (0-1), which is the mmCIF convention. ISa (the
|
|
// Diederichs asymptotic I/sigma, 1/b of the a*sigma^2 + (b*I)^2 error model) and the twinning
|
|
// indicators below have no standard mmCIF item. They are written under the "jfjoch" reserved
|
|
// prefix (_reflns.jfjoch_*), the IUCr-sanctioned local-data-name extension for private items -
|
|
// NOT the "pdbx_" prefix, which is owned by the wwPDB PDBx/mmCIF dictionary and must not label
|
|
// items that dictionary does not define. (The other pdbx_ items here are genuine PDBx items.)
|
|
const auto mult = [](const MergeStatisticsShell &s) {
|
|
return s.unique_reflections > 0 ? static_cast<double>(s.total_observations) / s.unique_reflections : 0.0; };
|
|
const auto compl_pct = [](const MergeStatisticsShell &s) {
|
|
return s.possible_unique_reflections > 0
|
|
? 100.0 * static_cast<double>(s.unique_reflections) / s.possible_unique_reflections : 0.0; };
|
|
if (!statistics.shells.empty()) {
|
|
const auto &ov = statistics.overall;
|
|
out << "_reflns.d_resolution_high " << Fmt(ov.d_min, 2) << "\n";
|
|
out << "_reflns.d_resolution_low " << Fmt(ov.d_max, 2) << "\n";
|
|
out << "_reflns.number_obs " << ov.unique_reflections << "\n";
|
|
out << "_reflns.pdbx_number_measured_all " << ov.total_observations << "\n";
|
|
out << "_reflns.pdbx_redundancy " << Fmt(mult(ov), 2) << "\n";
|
|
out << "_reflns.percent_possible_obs " << Fmt(compl_pct(ov), 1) << "\n";
|
|
out << "_reflns.pdbx_netI_over_sigmaI " << Fmt(ov.mean_i_over_sigma, 2) << "\n";
|
|
out << "_reflns.pdbx_Rrim_I_all " << Fmt(ov.r_meas, 4) << "\n";
|
|
out << "_reflns.pdbx_CC_half " << Fmt(ov.cc_half, 4) << "\n";
|
|
out << "_reflns.jfjoch_diffrn_ISa " << CifStr(isa) << " # asymptotic I/sigma (Diederichs)\n";
|
|
// Twinning indicators (no standard mmCIF item; same jfjoch local prefix as ISa above).
|
|
if (twinning.l_test_pairs > 0) {
|
|
out << "_reflns.jfjoch_L_test_mean_abs_L " << Fmt(twinning.mean_abs_l, 3)
|
|
<< " # Padilla-Yeates <|L|> (untwinned 0.500, perfect twin 0.375)\n";
|
|
out << "_reflns.jfjoch_L_test_mean_L_squared " << Fmt(twinning.mean_l_squared, 3)
|
|
<< " # <L^2> (untwinned 0.333, perfect twin 0.200)\n";
|
|
}
|
|
if (twinning.moment_reflections > 0)
|
|
out << "_reflns.jfjoch_second_moment_I " << Fmt(twinning.second_moment, 3)
|
|
<< " # <I^2>/<I>^2 (untwinned 2.00, perfect twin 1.50)\n";
|
|
out << "#\n";
|
|
|
|
out << "loop_\n";
|
|
out << "_reflns_shell.d_res_high\n";
|
|
out << "_reflns_shell.d_res_low\n";
|
|
out << "_reflns_shell.number_measured_obs\n";
|
|
out << "_reflns_shell.number_unique_obs\n";
|
|
out << "_reflns_shell.pdbx_redundancy\n";
|
|
out << "_reflns_shell.percent_possible_obs\n";
|
|
out << "_reflns_shell.meanI_over_sigI_obs\n";
|
|
out << "_reflns_shell.pdbx_Rrim_I_all\n";
|
|
out << "_reflns_shell.pdbx_CC_half\n";
|
|
for (const auto &s : statistics.shells) {
|
|
if (s.unique_reflections == 0)
|
|
continue;
|
|
out << Fmt(s.d_min, 2) << " " << Fmt(s.d_max, 2) << " "
|
|
<< s.total_observations << " " << s.unique_reflections << " "
|
|
<< Fmt(mult(s), 2) << " " << Fmt(compl_pct(s), 1) << " "
|
|
<< Fmt(s.mean_i_over_sigma, 2) << " " << Fmt(s.r_meas, 4) << " " << Fmt(s.cc_half, 4) << "\n";
|
|
}
|
|
out << "#\n";
|
|
}
|
|
|
|
// ---------- _refln loop ----------
|
|
out << "loop_\n";
|
|
out << "_refln.index_h\n";
|
|
out << "_refln.index_k\n";
|
|
out << "_refln.index_l\n";
|
|
out << "_refln.intensity_meas\n";
|
|
out << "_refln.intensity_sigma\n";
|
|
out << "_refln.F_meas_au\n";
|
|
out << "_refln.F_meas_sigma_au\n";
|
|
out << "_refln.status_free\n";
|
|
out << "_refln.status\n";
|
|
|
|
for (const auto& r : reflections) {
|
|
out << std::setw(5) << r.h << " "
|
|
<< std::setw(5) << r.k << " "
|
|
<< std::setw(5) << r.l << " "
|
|
<< std::setw(14) << Fmt(r.I, 4) << " "
|
|
<< std::setw(14) << Fmt(r.sigma, 4) << " "
|
|
<< std::setw(14) << Fmt(r.F, 4) << " "
|
|
<< std::setw(14) << Fmt(r.sigmaF, 4) << " "
|
|
<< (r.rfree_flag ? 1 : 0) << " "
|
|
<< "o" // 'o' = observed
|
|
<< "\n";
|
|
}
|
|
|
|
out << "#\n";
|
|
out << "# End of reflections\n";
|
|
out.close();
|
|
}
|
|
|
|
void WriteMtzReflections(const std::vector<MergedReflection> &reflections,
|
|
const UnitCell &unitCell,
|
|
const DiffractionExperiment &experiment,
|
|
const std::string &filename) {
|
|
gemmi::Mtz mtz;
|
|
|
|
// Optional but recommended metadata
|
|
mtz.spacegroup = gemmi::find_spacegroup_by_number(
|
|
experiment.GetSpaceGroupNumber().value_or(1));
|
|
mtz.set_cell_for_all(unitCell);
|
|
|
|
// Add dataset
|
|
gemmi::Mtz::Dataset& ds = mtz.add_dataset("native");
|
|
ds.crystal_name = experiment.GetSampleName();
|
|
ds.wavelength = experiment.GetWavelength_A();
|
|
|
|
const int dataset_id = ds.id;
|
|
|
|
// In anomalous mode the merge keeps the two Friedel mates as separate rows (I+ under the ASU
|
|
// representative hkl, I- under -hkl). Emitting those verbatim gives a file with two rows per
|
|
// reflection that downstream tools have to re-collapse. Instead pair the mates into one row per
|
|
// reflection with the standard CCP4 anomalous layout (IMEAN + I(+)/I(-), and the same split for
|
|
// the French-Wilson amplitude), which aimless / ctruncate / mtz2sca / ANODE read directly.
|
|
if (experiment.GetScalingSettings().GetMergeFriedel()) {
|
|
mtz.add_column("H", 'H', dataset_id, -1, false);
|
|
mtz.add_column("K", 'H', dataset_id, -1, false);
|
|
mtz.add_column("L", 'H', dataset_id, -1, false);
|
|
mtz.add_column("IMEAN", 'J', dataset_id, -1, false);
|
|
mtz.add_column("SIGIMEAN", 'Q', dataset_id, -1, false);
|
|
mtz.add_column("F", 'F', dataset_id, -1, false); // French-Wilson amplitude
|
|
mtz.add_column("SIGF", 'Q', dataset_id, -1, false);
|
|
mtz.add_column("FreeR_flag", 'I', dataset_id, -1, false);
|
|
|
|
mtz.nreflections = static_cast<int>(reflections.size());
|
|
mtz.data.reserve(reflections.size() * 8);
|
|
for (const auto& r : reflections) {
|
|
mtz.data.push_back(static_cast<float>(r.h));
|
|
mtz.data.push_back(static_cast<float>(r.k));
|
|
mtz.data.push_back(static_cast<float>(r.l));
|
|
mtz.data.push_back(r.I);
|
|
mtz.data.push_back(r.sigma);
|
|
mtz.data.push_back(r.F);
|
|
mtz.data.push_back(r.sigmaF);
|
|
mtz.data.push_back(r.rfree_flag ? 1.0f : 0.0f);
|
|
}
|
|
mtz.write_to_file(filename);
|
|
return;
|
|
}
|
|
|
|
// Anomalous: group the two mates by their (shared) Friedel-merged ASU representative. A single
|
|
// generator gives both the group key (its hkl, identical for +hkl and -hkl) and which mate this
|
|
// row is (.plus).
|
|
const HKLKeyGenerator key_gen(false, experiment.GetSpaceGroupNumber().value_or(1));
|
|
|
|
struct AnomRow {
|
|
int h = 0, k = 0, l = 0;
|
|
float Ip = NAN, sIp = NAN, Im = NAN, sIm = NAN;
|
|
float Fp = NAN, sFp = NAN, Fm = NAN, sFm = NAN;
|
|
int rfree = 0;
|
|
};
|
|
std::map<std::tuple<int, int, int>, AnomRow> rows;
|
|
for (const auto& r : reflections) {
|
|
const HKLKey key = key_gen(r);
|
|
AnomRow& row = rows[{key.h, key.k, key.l}];
|
|
row.h = key.h; row.k = key.k; row.l = key.l;
|
|
row.rfree = r.rfree_flag ? 1 : 0;
|
|
if (key.plus) { row.Ip = r.I; row.sIp = r.sigma; row.Fp = r.F; row.sFp = r.sigmaF; }
|
|
else { row.Im = r.I; row.sIm = r.sigma; row.Fm = r.F; row.sFm = r.sigmaF; }
|
|
}
|
|
|
|
// Friedel-mean of the two mates by inverse variance (the single mate, if only one was measured).
|
|
const auto combine = [](float a, float sa, float b, float sb, float& val, float& sig) {
|
|
const bool ok_a = std::isfinite(a) && sa > 0.0f;
|
|
const bool ok_b = std::isfinite(b) && sb > 0.0f;
|
|
if (ok_a && ok_b) {
|
|
const double wa = 1.0 / (static_cast<double>(sa) * sa);
|
|
const double wb = 1.0 / (static_cast<double>(sb) * sb);
|
|
val = static_cast<float>((wa * a + wb * b) / (wa + wb));
|
|
sig = static_cast<float>(1.0 / std::sqrt(wa + wb));
|
|
} else if (ok_a) { val = a; sig = sa; }
|
|
else if (ok_b) { val = b; sig = sb; }
|
|
else { val = NAN; sig = NAN; }
|
|
};
|
|
|
|
mtz.add_column("H", 'H', dataset_id, -1, false);
|
|
mtz.add_column("K", 'H', dataset_id, -1, false);
|
|
mtz.add_column("L", 'H', dataset_id, -1, false);
|
|
mtz.add_column("IMEAN", 'J', dataset_id, -1, false);
|
|
mtz.add_column("SIGIMEAN", 'Q', dataset_id, -1, false);
|
|
mtz.add_column("I(+)", 'K', dataset_id, -1, false);
|
|
mtz.add_column("SIGI(+)", 'M', dataset_id, -1, false);
|
|
mtz.add_column("I(-)", 'K', dataset_id, -1, false);
|
|
mtz.add_column("SIGI(-)", 'M', dataset_id, -1, false);
|
|
mtz.add_column("F", 'F', dataset_id, -1, false); // French-Wilson amplitude (mean)
|
|
mtz.add_column("SIGF", 'Q', dataset_id, -1, false);
|
|
mtz.add_column("F(+)", 'G', dataset_id, -1, false);
|
|
mtz.add_column("SIGF(+)", 'L', dataset_id, -1, false);
|
|
mtz.add_column("F(-)", 'G', dataset_id, -1, false);
|
|
mtz.add_column("SIGF(-)", 'L', dataset_id, -1, false);
|
|
mtz.add_column("FreeR_flag", 'I', dataset_id, -1, false);
|
|
|
|
mtz.nreflections = static_cast<int>(rows.size());
|
|
mtz.data.reserve(rows.size() * 16);
|
|
for (const auto& [hkl, row] : rows) {
|
|
float i_mean, sig_i_mean, f_mean, sig_f_mean;
|
|
combine(row.Ip, row.sIp, row.Im, row.sIm, i_mean, sig_i_mean);
|
|
combine(row.Fp, row.sFp, row.Fm, row.sFm, f_mean, sig_f_mean);
|
|
mtz.data.push_back(static_cast<float>(row.h));
|
|
mtz.data.push_back(static_cast<float>(row.k));
|
|
mtz.data.push_back(static_cast<float>(row.l));
|
|
mtz.data.push_back(i_mean);
|
|
mtz.data.push_back(sig_i_mean);
|
|
mtz.data.push_back(row.Ip);
|
|
mtz.data.push_back(row.sIp);
|
|
mtz.data.push_back(row.Im);
|
|
mtz.data.push_back(row.sIm);
|
|
mtz.data.push_back(f_mean);
|
|
mtz.data.push_back(sig_f_mean);
|
|
mtz.data.push_back(row.Fp);
|
|
mtz.data.push_back(row.sFp);
|
|
mtz.data.push_back(row.Fm);
|
|
mtz.data.push_back(row.sFm);
|
|
mtz.data.push_back(static_cast<float>(row.rfree));
|
|
}
|
|
mtz.write_to_file(filename);
|
|
}
|
|
|
|
void WriteReflections(const std::vector<MergedReflection> &reflections,
|
|
const UnitCell &unitCell,
|
|
const DiffractionExperiment &experiment,
|
|
const MergeStatistics &statistics,
|
|
const std::string &isa,
|
|
const TwinningAnalysisResult &twinning,
|
|
const std::string &filename) {
|
|
// Always write both an MTZ and an mmCIF - each has its uses downstream (MTZ for the CCP4 /
|
|
// phenix reflection tools, mmCIF for deposition and as the self-describing native format).
|
|
WriteMtzReflections(reflections, unitCell, experiment, filename + ".mtz");
|
|
WriteMmcifReflections(reflections, unitCell, experiment, statistics, isa, twinning, filename + ".cif");
|
|
}
|