The crystal refinement (XtalOptimizer, both the seven-block and the reduced beam+orientation form, and XtalOptimizerRotationOnly) no longer builds a ceres::Problem. XtalRefine holds the problem as data and solves it with LMSolver, which follows Ceres' trust-region LM step for step - Jacobi scaling, damping and radius updates, stopping rules, box projection, the projected Armijo line search with cubic interpolation on bounded problems, the SphereManifold for the spindle - but takes J^T J and J^T r directly instead of a Jacobian. The residual is the same XtalResidual code, now Ceres-free and evaluated on a forward-mode Dual (Dual.h); everything that depends on parameters alone (detector-angle trig, per-frame back-rotation, reciprocal basis, orientation rotation) is worked out once per evaluation, and the observed and predicted halves carry 6 and 9 derivative lanes rather than 16. The sums are cut into blocks that depend on the residual count alone, so the answer does not depend on the thread count. Because the line-search trial point is the candidate point, a bounded iteration costs one evaluation instead of Ceres' three. Validation (rc174 + this, -march=x86-64-v3): - p.mtz md5 identical to the Ceres build on myob/cytc/thau x10sa, GPU and CPU builds, and on the lyso8 stills reference. - Solve corpus (every 16-parameter solve and every 10th per-image solve of the three sets, 8.3k problems, inputs and Ceres results dumped from a run that reproduced the md5s): usable/failed agree on all, iteration counts identical on all, parameters agree to <2e-11 (in px / rad / 0.01 A units), costs to 1e-13. - Same process, same threads: 7-9x faster per solve than Ceres. - In-run (GPU, loaded box): xtal 16-parameter solves myob 22.2 -> 6.4 core-s, cytc 88 -> 26 core-s; per-image solves 5.4 -> 1.2 core-s (myob); cytc first pass indexing windows 1.9 -> 0.85 s, myob 1.1 -> 0.45 s; solver share of the whole cytc run 17% -> 4% of CPU samples. New tests compare the solver with Ceres on synthetic rotation problems (full/weighted/reduced) and check thread-count independence. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
366 lines
14 KiB
C++
366 lines
14 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
// The minimiser below follows Ceres Solver's trust-region Levenberg-Marquardt step for step - its
|
|
// options, its Jacobi scaling, its damping and radius updates, its stopping rules, its box projection,
|
|
// its projected Armijo line search on bounded problems and its SphereManifold - so that a problem moved
|
|
// off Ceres takes the same path to the same answer. Adapted from
|
|
// https://github.com/ceres-solver/ceres-solver (internal/ceres/trust_region_minimizer.cc,
|
|
// levenberg_marquardt_strategy.cc, trust_region_step_evaluator.cc, line_search.cc, polynomial.cc,
|
|
// include/ceres/internal/sphere_manifold_functions.h, householder_vector.h)
|
|
// Copyright 2023 Google Inc. All rights reserved.
|
|
// BSD-3-Clause, see licenses/ceres-solver.txt
|
|
//
|
|
// What it does NOT take from Ceres is the Jacobian: the caller hands over J^T J and J^T r directly,
|
|
// accumulated however it likes, and the minimiser never sees a row of J. Everything Ceres computes from
|
|
// the Jacobian - the column norms, the normal equations, the model cost change - is a function of those
|
|
// two alone. Only the options the crystal refinements use are reproduced: monotonic steps, no inner
|
|
// iterations, a dense Cholesky of the normal equations.
|
|
|
|
#pragma once
|
|
|
|
#include <chrono>
|
|
#include <cmath>
|
|
#include <limits>
|
|
#include <vector>
|
|
|
|
#include <Eigen/Dense>
|
|
|
|
struct LMBlock {
|
|
int offset = 0; // into the ambient parameter vector
|
|
int size = 0; // ambient size, at most 3
|
|
bool constant = false;
|
|
bool sphere = false; // Ceres' SphereManifold: the norm is kept, the tangent has size - 1 coordinates
|
|
double lower[3] = {-std::numeric_limits<double>::max(), -std::numeric_limits<double>::max(),
|
|
-std::numeric_limits<double>::max()};
|
|
double upper[3] = {std::numeric_limits<double>::max(), std::numeric_limits<double>::max(),
|
|
std::numeric_limits<double>::max()};
|
|
|
|
int TangentSize() const { return constant ? 0 : (sphere ? size - 1 : size); }
|
|
};
|
|
|
|
struct LMOptions {
|
|
int max_iterations = 50;
|
|
double max_time_s = 1e9;
|
|
};
|
|
|
|
enum class LMTermination { Convergence, NoConvergence, Failure };
|
|
|
|
struct LMSummary {
|
|
LMTermination termination = LMTermination::Failure;
|
|
// Iterations as Ceres counts them in Summary::iterations: iteration 0 included.
|
|
int iterations = 0;
|
|
int evaluations = 0;
|
|
int line_search_steps = 0;
|
|
double initial_cost = 0.0;
|
|
double final_cost = 0.0;
|
|
|
|
bool IsSolutionUsable() const { return termination != LMTermination::Failure; }
|
|
};
|
|
|
|
// Ceres' SphereManifold<3> for a three-vector: the Householder reflection that takes x to the pole, the
|
|
// Plus that walks a tangent step along the sphere, and the 3x2 Jacobian of that Plus at zero step.
|
|
void SpherePlus3(const double x[3], const double delta[2], double out[3]);
|
|
void SpherePlusJacobian3(const double x[3], double jacobian[3][2]);
|
|
|
|
// The polynomial step-size choice of Ceres' Armijo line search, cubic interpolation.
|
|
struct LMLineSample {
|
|
double x = 0.0;
|
|
double value = 0.0;
|
|
double gradient = 0.0;
|
|
bool value_is_valid = false;
|
|
bool gradient_is_valid = false;
|
|
};
|
|
double LMInterpolatedStepSize(const LMLineSample &lowerbound, const LMLineSample &previous,
|
|
const LMLineSample ¤t, double min_step, double max_step);
|
|
|
|
// Evaluate is called as eval(x, cost, g, H): x the ambient parameters, cost 1/2 sum of squared
|
|
// residuals, and - where g and H are not null - the gradient J^T r and J^T J in the TANGENT coordinates
|
|
// of the non-constant blocks, in block order. It returns false where anything came out non-finite.
|
|
// x is updated in place on success; it is left untouched on failure.
|
|
template<class Evaluate>
|
|
LMSummary SolveLM(std::vector<double> &x_io, const std::vector<LMBlock> &blocks, const LMOptions &options,
|
|
Evaluate &&eval) {
|
|
using Vec = Eigen::VectorXd;
|
|
using Mat = Eigen::MatrixXd;
|
|
constexpr double kMax = std::numeric_limits<double>::max();
|
|
|
|
// Ceres' defaults, which is what the callers always ran with.
|
|
constexpr double initial_radius = 1e4;
|
|
constexpr double max_radius = 1e16;
|
|
constexpr double min_radius = 1e-32;
|
|
constexpr double min_relative_decrease = 1e-3;
|
|
constexpr double min_lm_diagonal = 1e-6;
|
|
constexpr double max_lm_diagonal = 1e32;
|
|
constexpr int max_consecutive_invalid_steps = 5;
|
|
constexpr double function_tolerance = 1e-6;
|
|
constexpr double gradient_tolerance = 1e-10;
|
|
constexpr double parameter_tolerance = 1e-8;
|
|
constexpr double sufficient_decrease = 1e-4;
|
|
constexpr double max_step_contraction = 1e-3;
|
|
constexpr double min_step_contraction = 0.6;
|
|
constexpr double min_line_search_step = 1e-9;
|
|
constexpr int max_line_search_iterations = 20;
|
|
|
|
const auto start = std::chrono::steady_clock::now();
|
|
LMSummary summary;
|
|
|
|
int n = 0;
|
|
bool constrained = false;
|
|
for (const auto &b: blocks) {
|
|
for (int j = 0; j < b.size; j++)
|
|
if (!std::isfinite(x_io[b.offset + j]))
|
|
return summary;
|
|
n += b.TangentSize();
|
|
for (int j = 0; j < b.size; j++) {
|
|
if (b.constant) {
|
|
if (x_io[b.offset + j] < b.lower[j] || x_io[b.offset + j] > b.upper[j])
|
|
return summary;
|
|
} else {
|
|
if (b.lower[j] >= b.upper[j])
|
|
return summary;
|
|
if (b.lower[j] > -kMax || b.upper[j] < kMax)
|
|
constrained = true;
|
|
}
|
|
}
|
|
}
|
|
|
|
// x (+) delta, block by block, projected onto the bounds - Ceres' ParameterBlock::Plus.
|
|
const auto plus = [&](const Vec &x, const Vec &delta, Vec &out) {
|
|
out = x;
|
|
int t = 0;
|
|
for (const auto &b: blocks) {
|
|
if (b.constant)
|
|
continue;
|
|
if (b.sphere) {
|
|
SpherePlus3(x.data() + b.offset, delta.data() + t, out.data() + b.offset);
|
|
} else {
|
|
for (int j = 0; j < b.size; j++)
|
|
out[b.offset + j] = x[b.offset + j] + delta[t + j];
|
|
}
|
|
for (int j = 0; j < b.size; j++) {
|
|
out[b.offset + j] = std::max(out[b.offset + j], b.lower[j]);
|
|
out[b.offset + j] = std::min(out[b.offset + j], b.upper[j]);
|
|
}
|
|
t += b.TangentSize();
|
|
}
|
|
};
|
|
// Norms over the parameters Ceres keeps in its state: the non-constant blocks only.
|
|
const auto free_norm = [&](const Vec &v) {
|
|
double s = 0.0;
|
|
for (const auto &b: blocks)
|
|
if (!b.constant)
|
|
for (int j = 0; j < b.size; j++)
|
|
s += v[b.offset + j] * v[b.offset + j];
|
|
return std::sqrt(s);
|
|
};
|
|
const auto free_max_norm = [&](const Vec &v) {
|
|
double m = 0.0;
|
|
for (const auto &b: blocks)
|
|
if (!b.constant)
|
|
for (int j = 0; j < b.size; j++)
|
|
m = std::max(m, std::fabs(v[b.offset + j]));
|
|
return m;
|
|
};
|
|
|
|
Vec x = Eigen::Map<const Vec>(x_io.data(), static_cast<Eigen::Index>(x_io.size()));
|
|
if (constrained) {
|
|
Vec projected;
|
|
plus(x, Vec::Zero(n), projected);
|
|
x = projected;
|
|
}
|
|
|
|
// One evaluation point with everything Ceres computes there.
|
|
struct Point {
|
|
Vec x;
|
|
double cost = kMax;
|
|
Vec g;
|
|
Mat H;
|
|
bool valid = false;
|
|
};
|
|
const auto evaluate = [&](const Vec &at, Point &p) {
|
|
p.x = at;
|
|
p.g.setZero(n);
|
|
p.H.setZero(n, n);
|
|
summary.evaluations++;
|
|
p.valid = eval(at.data(), p.cost, &p.g, &p.H) && std::isfinite(p.cost);
|
|
if (!p.valid)
|
|
p.cost = kMax;
|
|
};
|
|
|
|
Point cur;
|
|
evaluate(x, cur);
|
|
if (!cur.valid)
|
|
return summary;
|
|
summary.initial_cost = cur.cost;
|
|
|
|
// Jacobi scaling, fixed from the Jacobian at the starting point.
|
|
Vec scale(n);
|
|
for (int i = 0; i < n; i++)
|
|
scale[i] = 1.0 / (1.0 + std::sqrt(cur.H(i, i)));
|
|
|
|
Vec gs, neg_g, projected;
|
|
Mat Hs;
|
|
double gradient_max_norm = 0.0;
|
|
const auto take_point = [&]() {
|
|
gs = scale.cwiseProduct(cur.g);
|
|
Hs = scale.asDiagonal() * cur.H * scale.asDiagonal();
|
|
neg_g = -cur.g;
|
|
plus(cur.x, neg_g, projected);
|
|
gradient_max_norm = free_max_norm(cur.x - projected);
|
|
};
|
|
take_point();
|
|
|
|
double radius = initial_radius;
|
|
double decrease_factor = 2.0;
|
|
bool reuse_diagonal = false;
|
|
Vec diagonal(n);
|
|
int consecutive_invalid = 0;
|
|
bool any_successful_step = false;
|
|
bool step_successful = true; // iteration 0
|
|
int iteration = 0;
|
|
|
|
Point trial; // the last point the line search evaluated, reused as the candidate when it is one
|
|
|
|
const auto step_rejected = [&]() {
|
|
radius = radius / decrease_factor;
|
|
decrease_factor *= 2.0;
|
|
reuse_diagonal = true;
|
|
};
|
|
|
|
const auto finish = [&](LMTermination t) {
|
|
summary.termination = t;
|
|
summary.final_cost = cur.cost;
|
|
if (t != LMTermination::Failure)
|
|
for (int i = 0; i < x.size(); i++)
|
|
x_io[i] = cur.x[i];
|
|
return summary;
|
|
};
|
|
|
|
for (;;) {
|
|
// FinalizeIterationAndCheckIfMinimizerCanContinue
|
|
summary.iterations++;
|
|
if (std::chrono::duration<double>(std::chrono::steady_clock::now() - start).count()
|
|
>= options.max_time_s)
|
|
return finish(LMTermination::NoConvergence);
|
|
if (iteration >= options.max_iterations)
|
|
return finish(LMTermination::NoConvergence);
|
|
if (step_successful && gradient_max_norm <= gradient_tolerance)
|
|
return finish(LMTermination::Convergence);
|
|
if (radius <= min_radius)
|
|
return finish(LMTermination::Convergence);
|
|
|
|
iteration++;
|
|
step_successful = false;
|
|
|
|
// ComputeTrustRegionStep: the damped normal equations of the scaled Jacobian.
|
|
if (!reuse_diagonal)
|
|
for (int i = 0; i < n; i++)
|
|
diagonal[i] = std::min(std::max(Hs(i, i), min_lm_diagonal), max_lm_diagonal);
|
|
Mat lhs = Hs;
|
|
for (int i = 0; i < n; i++) {
|
|
const double d = std::sqrt(diagonal[i] / radius);
|
|
lhs(i, i) += d * d;
|
|
}
|
|
reuse_diagonal = true;
|
|
Eigen::LLT<Mat, Eigen::Upper> llt(lhs);
|
|
bool step_valid = false;
|
|
Vec step;
|
|
double model_cost_change = 0.0;
|
|
if (llt.info() == Eigen::Success) {
|
|
step = -llt.solve(gs);
|
|
if (step.allFinite()) {
|
|
model_cost_change = -(step.dot(gs) + 0.5 * step.dot(Hs * step));
|
|
step_valid = model_cost_change > 0.0;
|
|
}
|
|
}
|
|
if (!step_valid) {
|
|
if (++consecutive_invalid >= max_consecutive_invalid_steps)
|
|
return finish(LMTermination::Failure);
|
|
step_rejected();
|
|
continue;
|
|
}
|
|
consecutive_invalid = 0;
|
|
Vec delta = step.cwiseProduct(scale);
|
|
|
|
bool have_trial = false;
|
|
if (constrained) {
|
|
// Projected Armijo line search along delta, cubic interpolation.
|
|
const double initial_gradient = cur.g.dot(delta);
|
|
const double direction_max_norm = delta.lpNorm<Eigen::Infinity>();
|
|
LMLineSample initial{0.0, cur.cost, initial_gradient, true, true};
|
|
LMLineSample previous, current;
|
|
const auto line_eval = [&](double alpha, LMLineSample &s) {
|
|
s = LMLineSample{};
|
|
s.x = alpha;
|
|
Vec moved;
|
|
plus(cur.x, Vec(alpha * delta), moved);
|
|
evaluate(moved, trial);
|
|
have_trial = true;
|
|
if (!trial.valid)
|
|
return;
|
|
s.value = trial.cost;
|
|
s.value_is_valid = true;
|
|
s.gradient = delta.dot(trial.g);
|
|
s.gradient_is_valid = std::isfinite(s.gradient);
|
|
};
|
|
line_eval(1.0, current);
|
|
bool success = true;
|
|
int ls_iterations = 0;
|
|
while (!current.value_is_valid
|
|
|| current.value > initial.value + sufficient_decrease * initial_gradient * current.x) {
|
|
++ls_iterations;
|
|
if (ls_iterations >= max_line_search_iterations) {
|
|
success = false;
|
|
break;
|
|
}
|
|
const double alpha = LMInterpolatedStepSize(initial, previous, current,
|
|
max_step_contraction * current.x,
|
|
min_step_contraction * current.x);
|
|
if (alpha * direction_max_norm < min_line_search_step) {
|
|
success = false;
|
|
break;
|
|
}
|
|
previous = current;
|
|
line_eval(alpha, current);
|
|
}
|
|
summary.line_search_steps += ls_iterations;
|
|
if (success)
|
|
delta *= current.x;
|
|
}
|
|
|
|
// ComputeCandidatePointAndEvaluateCost
|
|
Vec candidate_x;
|
|
plus(cur.x, delta, candidate_x);
|
|
Point cand;
|
|
if (have_trial && trial.x == candidate_x)
|
|
cand = std::move(trial);
|
|
else
|
|
evaluate(candidate_x, cand);
|
|
|
|
if (any_successful_step) {
|
|
const double step_norm = free_norm(cur.x - cand.x);
|
|
if (step_norm <= parameter_tolerance * (free_norm(cur.x) + parameter_tolerance))
|
|
return finish(LMTermination::Convergence);
|
|
}
|
|
if (std::fabs(cur.cost - cand.cost) <= function_tolerance * cur.cost)
|
|
return finish(LMTermination::Convergence);
|
|
|
|
const double relative_decrease = (cand.cost >= kMax)
|
|
? std::numeric_limits<double>::lowest()
|
|
: (cur.cost - cand.cost) / model_cost_change;
|
|
if (relative_decrease > min_relative_decrease) {
|
|
any_successful_step = true;
|
|
step_successful = true;
|
|
cur = std::move(cand);
|
|
take_point();
|
|
radius = radius / std::max(1.0 / 3.0, 1.0 - std::pow(2.0 * relative_decrease - 1.0, 3));
|
|
radius = std::min(max_radius, radius);
|
|
decrease_factor = 2.0;
|
|
reuse_diagonal = false;
|
|
} else {
|
|
step_rejected();
|
|
}
|
|
}
|
|
}
|