Files
Jungfraujoch/image_analysis/geom_refinement/Dual.h
T
leonarski_fandClaude Opus 5.5 a4bb6f74f1 XtalOptimizer: own Levenberg-Marquardt solver in place of Ceres
The crystal refinement (XtalOptimizer, both the seven-block and the reduced
beam+orientation form, and XtalOptimizerRotationOnly) no longer builds a
ceres::Problem. XtalRefine holds the problem as data and solves it with
LMSolver, which follows Ceres' trust-region LM step for step - Jacobi scaling,
damping and radius updates, stopping rules, box projection, the projected
Armijo line search with cubic interpolation on bounded problems, the
SphereManifold for the spindle - but takes J^T J and J^T r directly instead of
a Jacobian. The residual is the same XtalResidual code, now Ceres-free and
evaluated on a forward-mode Dual (Dual.h); everything that depends on
parameters alone (detector-angle trig, per-frame back-rotation, reciprocal
basis, orientation rotation) is worked out once per evaluation, and the
observed and predicted halves carry 6 and 9 derivative lanes rather than 16.
The sums are cut into blocks that depend on the residual count alone, so the
answer does not depend on the thread count. Because the line-search trial point
is the candidate point, a bounded iteration costs one evaluation instead of
Ceres' three.

Validation (rc174 + this, -march=x86-64-v3):
- p.mtz md5 identical to the Ceres build on myob/cytc/thau x10sa, GPU and CPU
  builds, and on the lyso8 stills reference.
- Solve corpus (every 16-parameter solve and every 10th per-image solve of the
  three sets, 8.3k problems, inputs and Ceres results dumped from a run that
  reproduced the md5s): usable/failed agree on all, iteration counts identical
  on all, parameters agree to <2e-11 (in px / rad / 0.01 A units), costs to
  1e-13.
- Same process, same threads: 7-9x faster per solve than Ceres.
- In-run (GPU, loaded box): xtal 16-parameter solves myob 22.2 -> 6.4 core-s,
  cytc 88 -> 26 core-s; per-image solves 5.4 -> 1.2 core-s (myob); cytc first
  pass indexing windows 1.9 -> 0.85 s, myob 1.1 -> 0.45 s; solver share of the
  whole cytc run 17% -> 4% of CPU samples.

New tests compare the solver with Ceres on synthetic rotation problems
(full/weighted/reduced) and check thread-count independence.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
2026-10-03 09:46:26 +02:00

143 lines
5.6 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
// A forward-mode dual number with N derivative lanes: a value and its gradient with respect to N
// parameters. The residuals of the crystal refinement are written as templates over their scalar type,
// so the same code runs on a plain double and on this. The value part of every operation is the plain
// double arithmetic of the same expression - written the way ceres::Jet writes it, division through the
// reciprocal - so a residual evaluated on a Dual has the same value as on a Jet.
#include <cmath>
#include <limits>
#include <Eigen/Core>
template<int N>
struct Dual {
double a = 0.0;
double v[N] = {};
Dual() = default;
Dual(double value) : a(value) {} // NOLINT: implicit, a constant is a dual with zero derivatives
static Dual Variable(double value, int lane) {
Dual d(value);
d.v[lane] = 1.0;
return d;
}
Dual &operator+=(const Dual &o) { a += o.a; for (int i = 0; i < N; i++) v[i] += o.v[i]; return *this; }
Dual &operator-=(const Dual &o) { a -= o.a; for (int i = 0; i < N; i++) v[i] -= o.v[i]; return *this; }
Dual &operator*=(const Dual &o) { *this = *this * o; return *this; }
Dual &operator/=(const Dual &o) { *this = *this / o; return *this; }
friend Dual operator+(const Dual &x) { return x; }
friend Dual operator-(const Dual &x) {
Dual r(-x.a);
for (int i = 0; i < N; i++) r.v[i] = -x.v[i];
return r;
}
friend Dual operator+(const Dual &x, const Dual &y) {
Dual r(x.a + y.a);
for (int i = 0; i < N; i++) r.v[i] = x.v[i] + y.v[i];
return r;
}
friend Dual operator+(const Dual &x, double s) { Dual r = x; r.a += s; return r; }
friend Dual operator+(double s, const Dual &x) { Dual r = x; r.a += s; return r; }
friend Dual operator-(const Dual &x, const Dual &y) {
Dual r(x.a - y.a);
for (int i = 0; i < N; i++) r.v[i] = x.v[i] - y.v[i];
return r;
}
friend Dual operator-(const Dual &x, double s) { Dual r = x; r.a -= s; return r; }
friend Dual operator-(double s, const Dual &x) {
Dual r(s - x.a);
for (int i = 0; i < N; i++) r.v[i] = -x.v[i];
return r;
}
friend Dual operator*(const Dual &x, const Dual &y) {
Dual r(x.a * y.a);
for (int i = 0; i < N; i++) r.v[i] = x.a * y.v[i] + x.v[i] * y.a;
return r;
}
friend Dual operator*(const Dual &x, double s) {
Dual r(x.a * s);
for (int i = 0; i < N; i++) r.v[i] = x.v[i] * s;
return r;
}
friend Dual operator*(double s, const Dual &x) { return x * s; }
friend Dual operator/(const Dual &x, const Dual &y) {
const double y_inv = 1.0 / y.a;
const double q = x.a * y_inv;
Dual r(q);
for (int i = 0; i < N; i++) r.v[i] = (x.v[i] - q * y.v[i]) * y_inv;
return r;
}
friend Dual operator/(const Dual &x, double s) {
const double s_inv = 1.0 / s;
return x * s_inv;
}
friend Dual operator/(double s, const Dual &y) {
const double y_inv = 1.0 / y.a;
const double d = -s * y_inv * y_inv;
Dual r(s * y_inv);
for (int i = 0; i < N; i++) r.v[i] = d * y.v[i];
return r;
}
friend bool operator<(const Dual &x, const Dual &y) { return x.a < y.a; }
friend bool operator>(const Dual &x, const Dual &y) { return x.a > y.a; }
friend bool operator<=(const Dual &x, const Dual &y) { return x.a <= y.a; }
friend bool operator>=(const Dual &x, const Dual &y) { return x.a >= y.a; }
friend bool operator==(const Dual &x, const Dual &y) { return x.a == y.a; }
friend bool operator!=(const Dual &x, const Dual &y) { return x.a != y.a; }
// The chain rule for a function of one argument: value f, derivative df.
Dual Chain(double f, double df) const {
Dual r(f);
for (int i = 0; i < N; i++) r.v[i] = df * v[i];
return r;
}
friend Dual sqrt(const Dual &x) {
const double s = std::sqrt(x.a);
return x.Chain(s, 0.5 / s);
}
friend Dual cos(const Dual &x) { return x.Chain(std::cos(x.a), -std::sin(x.a)); }
friend Dual sin(const Dual &x) { return x.Chain(std::sin(x.a), std::cos(x.a)); }
friend Dual hypot(const Dual &x, const Dual &y, const Dual &z) {
// As ceres::hypot(Jet, Jet, Jet): the value is std::hypot, the derivative x/h dx + y/h dy + z/h dz.
const double h = std::hypot(x.a, y.a, z.a);
Dual r(h);
for (int i = 0; i < N; i++) r.v[i] = x.a / h * x.v[i] + y.a / h * y.v[i] + z.a / h * z.v[i];
return r;
}
friend int fpclassify(const Dual &x) { return std::fpclassify(x.a); }
};
// What Eigen needs to hold a Dual in a fixed-size matrix (the reciprocal basis is built in one).
namespace Eigen {
template<int N>
struct NumTraits<Dual<N>> : GenericNumTraits<double> {
typedef Dual<N> Real;
typedef Dual<N> NonInteger;
typedef Dual<N> Nested;
typedef Dual<N> Literal;
enum {
IsComplex = 0, IsInteger = 0, IsSigned = 1, RequireInitialization = 1,
ReadCost = 1, AddCost = 1, MulCost = 1
};
static inline Real epsilon() { return Real(std::numeric_limits<double>::epsilon()); }
static inline Real dummy_precision() { return Real(1e-12); }
static inline Real highest() { return Real(std::numeric_limits<double>::max()); }
static inline Real lowest() { return Real(-std::numeric_limits<double>::max()); }
static inline int digits10() { return NumTraits<double>::digits10(); }
};
}