The crystal refinement (XtalOptimizer, both the seven-block and the reduced beam+orientation form, and XtalOptimizerRotationOnly) no longer builds a ceres::Problem. XtalRefine holds the problem as data and solves it with LMSolver, which follows Ceres' trust-region LM step for step - Jacobi scaling, damping and radius updates, stopping rules, box projection, the projected Armijo line search with cubic interpolation on bounded problems, the SphereManifold for the spindle - but takes J^T J and J^T r directly instead of a Jacobian. The residual is the same XtalResidual code, now Ceres-free and evaluated on a forward-mode Dual (Dual.h); everything that depends on parameters alone (detector-angle trig, per-frame back-rotation, reciprocal basis, orientation rotation) is worked out once per evaluation, and the observed and predicted halves carry 6 and 9 derivative lanes rather than 16. The sums are cut into blocks that depend on the residual count alone, so the answer does not depend on the thread count. Because the line-search trial point is the candidate point, a bounded iteration costs one evaluation instead of Ceres' three. Validation (rc174 + this, -march=x86-64-v3): - p.mtz md5 identical to the Ceres build on myob/cytc/thau x10sa, GPU and CPU builds, and on the lyso8 stills reference. - Solve corpus (every 16-parameter solve and every 10th per-image solve of the three sets, 8.3k problems, inputs and Ceres results dumped from a run that reproduced the md5s): usable/failed agree on all, iteration counts identical on all, parameters agree to <2e-11 (in px / rad / 0.01 A units), costs to 1e-13. - Same process, same threads: 7-9x faster per solve than Ceres. - In-run (GPU, loaded box): xtal 16-parameter solves myob 22.2 -> 6.4 core-s, cytc 88 -> 26 core-s; per-image solves 5.4 -> 1.2 core-s (myob); cytc first pass indexing windows 1.9 -> 0.85 s, myob 1.1 -> 0.45 s; solver share of the whole cytc run 17% -> 4% of CPU samples. New tests compare the solver with Ceres on synthetic rotation problems (full/weighted/reduced) and check thread-count independence. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
143 lines
5.6 KiB
C++
143 lines
5.6 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
// A forward-mode dual number with N derivative lanes: a value and its gradient with respect to N
|
|
// parameters. The residuals of the crystal refinement are written as templates over their scalar type,
|
|
// so the same code runs on a plain double and on this. The value part of every operation is the plain
|
|
// double arithmetic of the same expression - written the way ceres::Jet writes it, division through the
|
|
// reciprocal - so a residual evaluated on a Dual has the same value as on a Jet.
|
|
|
|
#include <cmath>
|
|
#include <limits>
|
|
|
|
#include <Eigen/Core>
|
|
|
|
template<int N>
|
|
struct Dual {
|
|
double a = 0.0;
|
|
double v[N] = {};
|
|
|
|
Dual() = default;
|
|
Dual(double value) : a(value) {} // NOLINT: implicit, a constant is a dual with zero derivatives
|
|
|
|
static Dual Variable(double value, int lane) {
|
|
Dual d(value);
|
|
d.v[lane] = 1.0;
|
|
return d;
|
|
}
|
|
|
|
Dual &operator+=(const Dual &o) { a += o.a; for (int i = 0; i < N; i++) v[i] += o.v[i]; return *this; }
|
|
Dual &operator-=(const Dual &o) { a -= o.a; for (int i = 0; i < N; i++) v[i] -= o.v[i]; return *this; }
|
|
Dual &operator*=(const Dual &o) { *this = *this * o; return *this; }
|
|
Dual &operator/=(const Dual &o) { *this = *this / o; return *this; }
|
|
|
|
friend Dual operator+(const Dual &x) { return x; }
|
|
friend Dual operator-(const Dual &x) {
|
|
Dual r(-x.a);
|
|
for (int i = 0; i < N; i++) r.v[i] = -x.v[i];
|
|
return r;
|
|
}
|
|
|
|
friend Dual operator+(const Dual &x, const Dual &y) {
|
|
Dual r(x.a + y.a);
|
|
for (int i = 0; i < N; i++) r.v[i] = x.v[i] + y.v[i];
|
|
return r;
|
|
}
|
|
friend Dual operator+(const Dual &x, double s) { Dual r = x; r.a += s; return r; }
|
|
friend Dual operator+(double s, const Dual &x) { Dual r = x; r.a += s; return r; }
|
|
|
|
friend Dual operator-(const Dual &x, const Dual &y) {
|
|
Dual r(x.a - y.a);
|
|
for (int i = 0; i < N; i++) r.v[i] = x.v[i] - y.v[i];
|
|
return r;
|
|
}
|
|
friend Dual operator-(const Dual &x, double s) { Dual r = x; r.a -= s; return r; }
|
|
friend Dual operator-(double s, const Dual &x) {
|
|
Dual r(s - x.a);
|
|
for (int i = 0; i < N; i++) r.v[i] = -x.v[i];
|
|
return r;
|
|
}
|
|
|
|
friend Dual operator*(const Dual &x, const Dual &y) {
|
|
Dual r(x.a * y.a);
|
|
for (int i = 0; i < N; i++) r.v[i] = x.a * y.v[i] + x.v[i] * y.a;
|
|
return r;
|
|
}
|
|
friend Dual operator*(const Dual &x, double s) {
|
|
Dual r(x.a * s);
|
|
for (int i = 0; i < N; i++) r.v[i] = x.v[i] * s;
|
|
return r;
|
|
}
|
|
friend Dual operator*(double s, const Dual &x) { return x * s; }
|
|
|
|
friend Dual operator/(const Dual &x, const Dual &y) {
|
|
const double y_inv = 1.0 / y.a;
|
|
const double q = x.a * y_inv;
|
|
Dual r(q);
|
|
for (int i = 0; i < N; i++) r.v[i] = (x.v[i] - q * y.v[i]) * y_inv;
|
|
return r;
|
|
}
|
|
friend Dual operator/(const Dual &x, double s) {
|
|
const double s_inv = 1.0 / s;
|
|
return x * s_inv;
|
|
}
|
|
friend Dual operator/(double s, const Dual &y) {
|
|
const double y_inv = 1.0 / y.a;
|
|
const double d = -s * y_inv * y_inv;
|
|
Dual r(s * y_inv);
|
|
for (int i = 0; i < N; i++) r.v[i] = d * y.v[i];
|
|
return r;
|
|
}
|
|
|
|
friend bool operator<(const Dual &x, const Dual &y) { return x.a < y.a; }
|
|
friend bool operator>(const Dual &x, const Dual &y) { return x.a > y.a; }
|
|
friend bool operator<=(const Dual &x, const Dual &y) { return x.a <= y.a; }
|
|
friend bool operator>=(const Dual &x, const Dual &y) { return x.a >= y.a; }
|
|
friend bool operator==(const Dual &x, const Dual &y) { return x.a == y.a; }
|
|
friend bool operator!=(const Dual &x, const Dual &y) { return x.a != y.a; }
|
|
|
|
// The chain rule for a function of one argument: value f, derivative df.
|
|
Dual Chain(double f, double df) const {
|
|
Dual r(f);
|
|
for (int i = 0; i < N; i++) r.v[i] = df * v[i];
|
|
return r;
|
|
}
|
|
|
|
friend Dual sqrt(const Dual &x) {
|
|
const double s = std::sqrt(x.a);
|
|
return x.Chain(s, 0.5 / s);
|
|
}
|
|
friend Dual cos(const Dual &x) { return x.Chain(std::cos(x.a), -std::sin(x.a)); }
|
|
friend Dual sin(const Dual &x) { return x.Chain(std::sin(x.a), std::cos(x.a)); }
|
|
friend Dual hypot(const Dual &x, const Dual &y, const Dual &z) {
|
|
// As ceres::hypot(Jet, Jet, Jet): the value is std::hypot, the derivative x/h dx + y/h dy + z/h dz.
|
|
const double h = std::hypot(x.a, y.a, z.a);
|
|
Dual r(h);
|
|
for (int i = 0; i < N; i++) r.v[i] = x.a / h * x.v[i] + y.a / h * y.v[i] + z.a / h * z.v[i];
|
|
return r;
|
|
}
|
|
friend int fpclassify(const Dual &x) { return std::fpclassify(x.a); }
|
|
};
|
|
|
|
// What Eigen needs to hold a Dual in a fixed-size matrix (the reciprocal basis is built in one).
|
|
namespace Eigen {
|
|
template<int N>
|
|
struct NumTraits<Dual<N>> : GenericNumTraits<double> {
|
|
typedef Dual<N> Real;
|
|
typedef Dual<N> NonInteger;
|
|
typedef Dual<N> Nested;
|
|
typedef Dual<N> Literal;
|
|
enum {
|
|
IsComplex = 0, IsInteger = 0, IsSigned = 1, RequireInitialization = 1,
|
|
ReadCost = 1, AddCost = 1, MulCost = 1
|
|
};
|
|
static inline Real epsilon() { return Real(std::numeric_limits<double>::epsilon()); }
|
|
static inline Real dummy_precision() { return Real(1e-12); }
|
|
static inline Real highest() { return Real(std::numeric_limits<double>::max()); }
|
|
static inline Real lowest() { return Real(-std::numeric_limits<double>::max()); }
|
|
static inline int digits10() { return NumTraits<double>::digits10(); }
|
|
};
|
|
}
|