Files
Jungfraujoch/image_analysis/indexing/PostIndexingRefinement.cpp
T
leonarski_fandClaude Opus 5 7e47afe47f
Build Packages / build:viewer-tgz:cpu (push) Successful in 20m26s
Build Packages / build:viewer-tgz:cuda (push) Successful in 21m30s
Build Packages / build:rpm (ubuntu2404_nocuda) (push) Successful in 22m36s
Build Packages / build:rpm (rocky9_nocuda) (push) Successful in 24m4s
Build Packages / build:rpm (rocky8_nocuda) (push) Successful in 28m10s
Build Packages / build:rpm (ubuntu2204_nocuda) (push) Successful in 28m12s
Build Packages / build:rpm (rocky8_sls9) (push) Successful in 28m23s
Build Packages / XDS test (durin plugin) (push) Successful in 11m21s
Build Packages / build:rpm (rocky9_sls9) (push) Successful in 20m56s
Build Packages / build:rpm (rocky9) (push) Successful in 21m10s
Build Packages / Generate python client (push) Successful in 40s
Build Packages / Build documentation (push) Successful in 1m34s
Build Packages / Create release (push) Skipped
Build Packages / build:rpm (rocky8) (push) Successful in 25m28s
Build Packages / DIALS test (push) Successful in 21m15s
Build Packages / build:rpm (ubuntu2404) (push) Successful in 21m26s
Build Packages / build:rpm (ubuntu2204) (push) Successful in 25m51s
Build Packages / XDS test (JFJoch plugin) (push) Successful in 10m53s
Build Packages / XDS test (neggia plugin) (push) Successful in 9m41s
Build Packages / Unit tests (push) Successful in 2h21m29s
Build Packages / build:windows:nocuda (push) Successful in 1m15s
Build Packages / build:windows:cuda (push) Successful in 28m0s
rugnux: parallelise candidate-cell refinement, and stop repeating work in the tail
Three independent changes to the CPU-bound parts of an offline rotation run, none
of which alters a result.

Candidate-cell refinement now splits across threads. RefineCandidateCells already
took a (block, nblocks) partition, but the only call site passed nblocks=1, so the
whole first pass of a two-pass rotation run sat on one thread per scheme - two
threads, unchanged at every -N, for a third of the run. A block touches only its
own scores(j) and cells rows and holds its own scratch, so the split is exact.
The budget is a new IndexingSettings::RefineThreads, left at 1 by default and set
only where few indexer threads exist: raising it unconditionally would
oversubscribe the paths that already run one indexer per image across all workers.

The mmCIF writer built a std::ostringstream per formatted number, twelve per
reflection. snprintf gives the same digits for 0.535 -> 0.220 s per file.

The space-group search built the same orbit mapping twice per candidate point
group - once for the merge chi^2 and once for the systematic-error b, an
apply_to_hkl and Canonicalize per observation per operator each time. Build it
once and hand it to both.

18 Mpx rotation set 24.6 -> 18.7 s, 2.5 Mpx 13.0 -> 10.7 s, and the 37-crystal
battery 13m55s -> 10m47s with no failures, the same 34/37 space groups, and
statistics unchanged on 30 of 37 (the rest drift within the run-to-run spread the
binary already had, which a control build with the split disabled reproduces).

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-03 07:30:33 +02:00

323 lines
14 KiB
C++

// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "../../common/JFJochMath.h"
#include "PostIndexingRefinement.h"
#include <iostream>
#include <thread>
namespace {
struct config_ifssr final {
float threshold_contraction = .8; // contract error threshold by this value in every iteration
float max_distance = .00075; // max distance to reciprocal spots for inliers
unsigned min_spots = 8; // minimum number of spots to fit against
unsigned max_iter = 32; // max number of iterations
};
static std::pair<float, float> score_parts(float score) noexcept {
float nsp = -std::floor(score);
float s = score + nsp;
return std::make_pair(nsp - 1, s);
}
struct RefinedCandidate {
Eigen::Matrix3f cell;
float score;
float volume;
int64_t indexed_spot_count;
std::vector<uint8_t> indexed_mask;
};
static inline Eigen::MatrixX3<float> CalculateResiduals(
const Eigen::Ref<const Eigen::MatrixX3<float>> &spots,
const Eigen::Matrix3f &cell) {
Eigen::MatrixX3<float> miller = (spots * cell).array().round().matrix();
Eigen::MatrixX3<float> resid = miller * cell.inverse();
resid -= spots;
return resid;
}
static inline std::vector<uint8_t> ComputeIndexedMask(
const Eigen::Ref<const Eigen::MatrixX3<float>> &spots,
const Eigen::Matrix3f &cell,
float indexing_tolerance,
int64_t &indexed_spot_count) {
const float indexing_tolerance_sq = indexing_tolerance * indexing_tolerance;
// Compute fractional Miller indices. rint (round half to even) rather than round (round half
// away from zero): without SSE4.1 Eigen has no vector round, so each element is a libm call,
// while rint is a few inline instructions. Only the SQUARED residual is taken below and the two
// rules can differ only at an exact .5, where either leaves |frac| = 0.5 - so the mask and the
// count are the same. The refinement loop above keeps round: there the rounded value IS the
// Miller index that goes into the residual and the QR solve, so its tie rule does matter.
Eigen::MatrixX3<float> miller_frac = spots * cell;
Eigen::MatrixX3<float> miller_int = miller_frac.array().rint().matrix();
Eigen::MatrixX3<float> frac_resid = miller_frac - miller_int;
std::vector<uint8_t> mask(spots.rows(), 0);
indexed_spot_count = 0;
for (int i = 0; i < spots.rows(); ++i) {
if (frac_resid.row(i).squaredNorm() < indexing_tolerance_sq) {
mask[i] = 1;
indexed_spot_count++;
}
}
return mask;
}
template<typename MatX3, typename VecX>
static void RefineCandidateCells(const Eigen::Ref<const Eigen::MatrixX3<float>> &spots,
Eigen::DenseBase<MatX3> &cells,
Eigen::DenseBase<VecX> &scores,
const config_ifssr &cifssr,
unsigned block = 0, unsigned nblocks = 1) {
using namespace Eigen;
using Mx3 = MatrixX3<float>;
using M3 = Matrix3<float>;
const unsigned nspots = spots.rows();
const unsigned ncells = scores.rows();
VectorX<bool> below{nspots};
MatrixX3<bool> sel{nspots, 3u};
Mx3 resid{nspots, 3u};
Mx3 miller{nspots, 3u};
M3 cell;
const unsigned blocksize = (ncells + nblocks - 1u) / nblocks;
const unsigned startcell = block * blocksize;
const unsigned endcell = std::min(startcell + blocksize, ncells);
for (unsigned j = startcell; j < endcell; j++) {
if (nspots < cifssr.min_spots) {
scores(j) = float{1.};
continue;
}
cell = cells.block(3u * j, 0u, 3u, 3u).transpose(); // cell: col vectors
const float scale = cell.colwise().norm().minCoeff();
float threshold = score_parts(scores[j]).second / scale;
for (unsigned niter = 1; niter < cifssr.max_iter && threshold > cifssr.max_distance; niter++) {
miller = (spots * cell).array().round().matrix();
resid = miller * cell.inverse();
resid -= spots;
below = (resid.rowwise().norm().array() < threshold);
if (below.count() < cifssr.min_spots)
break;
threshold *= cifssr.threshold_contraction;
sel.colwise() = below;
HouseholderQR<Mx3> qr{sel.select(spots, .0f)};
cell = qr.solve(sel.select(miller, .0f));
}
resid = CalculateResiduals(spots, cell);
ArrayX<float> dist = resid.rowwise().norm();
auto nth = std::begin(dist) + (cifssr.min_spots - 1);
std::nth_element(std::begin(dist), nth, std::end(dist));
scores(j) = *nth;
cells.block(3u * j, 0u, 3u, 3u) = cell.transpose();
}
}
}
std::vector<CrystalLattice> Refine(const std::vector<Coord> &in_spots,
size_t nspots,
Eigen::MatrixX3<float> &oCell,
Eigen::VectorX<float> &scores,
RefineParameters &p) {
std::vector<CrystalLattice> ret;
Eigen::MatrixX3<float> spots(in_spots.size(), 3u);
for (int i = 0; i < in_spots.size(); i++) {
spots(i, 0u) = in_spots[i].x;
spots(i, 1u) = in_spots[i].y;
spots(i, 2u) = in_spots[i].z;
}
config_ifssr cifssr{
.min_spots = static_cast<uint32_t>(p.viable_cell_min_spots)
};
// Candidate cells refine independently - a block touches only its own scores(j) and cells rows, and
// holds its own scratch - so splitting them across threads gives the same numbers as one thread.
// Only worth it where few indexer threads run (the rotation first pass uses two, one per scheme,
// and leaves the rest of the machine idle); refine_threads stays 1 everywhere else.
const unsigned ncells = static_cast<unsigned>(scores.rows());
const unsigned nblocks = std::max(1u, std::min(p.refine_threads, ncells));
if (nblocks == 1) {
RefineCandidateCells(spots.topRows(nspots), oCell, scores, cifssr);
} else {
std::vector<std::thread> workers;
workers.reserve(nblocks - 1);
for (unsigned b = 1; b < nblocks; b++)
workers.emplace_back([&, b] {
RefineCandidateCells(spots.topRows(nspots), oCell, scores, cifssr, b, nblocks);
});
RefineCandidateCells(spots.topRows(nspots), oCell, scores, cifssr, 0, nblocks);
for (auto &w : workers)
w.join();
}
std::vector<RefinedCandidate> candidates;
// Angle bounds as cosines, once, for the per-candidate test below.
const float cos_min_angle = std::cos(p.min_angle_deg * PI / 180.0f);
const float cos_max_angle = std::cos(p.max_angle_deg * PI / 180.0f);
for (int i = 0; i < scores.size(); i++) {
Eigen::Matrix3f cell_rows = oCell.block(3u * i, 0u, 3u, 3u);
Eigen::Matrix3f cell_cols = cell_rows.transpose();
Eigen::Vector3f row_norms = cell_rows.rowwise().norm();
if (p.reference_unit_cell) {
std::array<float, 3> obs = {row_norms(0), row_norms(1), row_norms(2)};
std::array<float, 3> ref = {
static_cast<float>(p.reference_unit_cell->a),
static_cast<float>(p.reference_unit_cell->b),
static_cast<float>(p.reference_unit_cell->c)
};
std::sort(obs.begin(), obs.end());
std::sort(ref.begin(), ref.end());
bool lengths_ok = true;
for (int k = 0; k < 3; ++k) {
const float denom = std::max(ref[k], REFINE_MIN_REFERENCE_LENGTH_EPSILON);
const float rel_dev = std::abs(obs[k] - ref[k]) / denom;
if (rel_dev > p.dist_tolerance_vs_reference) {
lengths_ok = false;
break;
}
}
if (!lengths_ok)
continue;
// Also require the angles to match the reference. Fold each to its acute complement
// (min(x,180-x)) so the obtuse/acute setting choice is irrelevant, then compare the
// sorted triples. Guards against a right-edges/wrong-angle cell (a pseudo-symmetric
// near-metric, e.g. a monoclinic beta refined to the wrong value) passing on lengths.
auto fold = [](float deg) { return std::min(deg, 180.0f - deg); };
auto row_angle = [&](int i, int j) {
return std::acos(std::clamp(cell_rows.row(i).normalized().dot(cell_rows.row(j).normalized()),
-1.0f, 1.0f)) * 180.0f / PI;
};
std::array<float, 3> obs_ang = {fold(row_angle(1, 2)), fold(row_angle(0, 2)), fold(row_angle(0, 1))};
std::array<float, 3> ref_ang = {
fold(static_cast<float>(p.reference_unit_cell->alpha)),
fold(static_cast<float>(p.reference_unit_cell->beta)),
fold(static_cast<float>(p.reference_unit_cell->gamma))
};
std::sort(obs_ang.begin(), obs_ang.end());
std::sort(ref_ang.begin(), ref_ang.end());
bool angles_ok = true;
for (int k = 0; k < 3; ++k) {
if (std::abs(obs_ang[k] - ref_ang[k]) > REFINE_ANGLE_TOLERANCE_VS_REFERENCE_DEG) {
angles_ok = false;
break;
}
}
if (!angles_ok)
continue;
} else {
if (row_norms.minCoeff() < p.min_length_A || row_norms.maxCoeff() > p.max_length_A)
continue;
}
// Filter for wrong angles. Compared as COSINES, not angles: acos is strictly decreasing on
// [-1, 1], so "angle outside [min_angle, max_angle]" is exactly "cosine outside
// [cos(max_angle), cos(min_angle)]" with the ends swapped - and the three acos calls the
// comparison needed disappear. They were not cheap: this runs per candidate cell per image,
// and on a serial-stills run acos was 41% of the whole process.
const float cos_alpha = cell_rows.row(1).normalized().dot(cell_rows.row(2).normalized());
const float cos_beta = cell_rows.row(0).normalized().dot(cell_rows.row(2).normalized());
const float cos_gamma = cell_rows.row(0).normalized().dot(cell_rows.row(1).normalized());
if (cos_alpha > cos_min_angle || cos_alpha < cos_max_angle ||
cos_beta > cos_min_angle || cos_beta < cos_max_angle ||
cos_gamma > cos_min_angle || cos_gamma < cos_max_angle)
continue;
int64_t indexed_spot_count = 0;
auto indexed_mask = ComputeIndexedMask(spots.topRows(nspots), cell_cols, p.indexing_tolerance, indexed_spot_count);
if (indexed_spot_count < p.viable_cell_min_spots)
continue;
candidates.emplace_back(RefinedCandidate{
.cell = cell_rows,
.score = scores(i),
.volume = std::abs(cell_rows.determinant()),
.indexed_spot_count = indexed_spot_count,
.indexed_mask = std::move(indexed_mask)
});
}
std::sort(candidates.begin(), candidates.end(),
[](const RefinedCandidate &a, const RefinedCandidate &b) {
const auto max_spots = std::max(a.indexed_spot_count, b.indexed_spot_count);
const auto min_spots = std::min(a.indexed_spot_count, b.indexed_spot_count);
const bool spot_counts_close = (max_spots > 0)
&& (static_cast<float>(min_spots) / static_cast<float>(max_spots)
>= REFINE_CANDIDATE_SPOT_COUNT_RATIO_THRESHOLD);
if (!spot_counts_close)
return a.indexed_spot_count > b.indexed_spot_count;
const float max_volume = std::max(a.volume, b.volume);
const float min_volume = std::max(std::min(a.volume, b.volume), REFINE_MIN_VOLUME_EPSILON);
const bool volume_differs = (max_volume / min_volume) > REFINE_CANDIDATE_VOLUME_RATIO_THRESHOLD;
if (volume_differs)
return a.volume < b.volume;
if (a.score != b.score)
return a.score < b.score;
return a.indexed_spot_count > b.indexed_spot_count;
});
std::vector<RefinedCandidate> accepted;
for (const auto &candidate: candidates) {
int64_t overlap = 0;
// Check all already selected lattices and see how many spots are already indexed for the candidate
// If the overlap is more than 40% of indexed spots - we assume the lattice doesn't bring anything new
for (const auto &selected: accepted) {
for (size_t i = 0; i < candidate.indexed_mask.size(); ++i) {
if (candidate.indexed_mask[i] && selected.indexed_mask[i])
overlap++;
}
}
if (overlap < static_cast<int64_t>(REFINE_CANDIDATE_OVERLAP_RATIO_THRESHOLD
* static_cast<float>(candidate.indexed_spot_count))) {
accepted.emplace_back(candidate);
}
}
ret.reserve(accepted.size());
for (auto &candidate: accepted) {
auto cell = candidate.cell;
if (cell.determinant() < .0f)
cell = -cell;
ret.emplace_back(
Coord(cell(0, 0), cell(0, 1), cell(0, 2)),
Coord(cell(1, 0), cell(1, 1), cell(1, 2)),
Coord(cell(2, 0), cell(2, 1), cell(2, 2))
);
}
return ret;
}