None of this is worth its speedup - the routine is 0.5-0.8% of the pre-scan's cycles on two independent profiles, and the whole change is 0.9 ms per data set. It is worth having because the code was doing work that has no reason to exist. The isolation grid was a vector<vector<uint32_t>> over ISOLATION_PX cells: on a 4148x4362 detector that is 23244 std::vector objects constructed, heap-allocated and destroyed per image for a structure that is read once. It is now a counting sort - one offset array, one index array. This was the largest single item and it is not a pixel loop. The encircled-flux curve added every pixel into every bin at or beyond its own radius, about 7.5 adds per pixel, which sums the same aperture R_MAX/2 times over. Each pixel now lands in the one bin its radius falls in and the curve is the running total. The bins are double where the running totals were float, so the result is more accurate, not merely faster. No square roots remain in the pixel loops. The isolation and beam gates compare squared distances, and the radial bin is a table lookup over the 197 squared distances the aperture can produce. That is the same bin, not an approximation: floor(sqrt(floor(y))) == floor(sqrt(y)) for every real y >= 0, because k*k is an integer, so the table recovers ceil(sqrt(rc2)) exactly once the perfect-square case is separated out. The disks are also walked as disks rather than as their bounding boxes - constexpr per-row half-widths - which takes the background pass from 1681 to 1257 pixel visits per spot. Verified over the rotation battery: r80 identical to nine significant figures on 37 of 37 measurable crystals, the 38th unmeasurable in both arms, and the chosen r1 identical on 38 of 38. The pass feeds nothing but r1 into the run, so merged output is unchanged. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CHMmeM1d489zvNFT7ZMN2P
337 lines
16 KiB
C++
337 lines
16 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "SpotWidth.h"
|
|
|
|
#include <algorithm>
|
|
#include <cmath>
|
|
#include <cstddef>
|
|
#include <cstdint>
|
|
#include <limits>
|
|
#include <utility>
|
|
|
|
using namespace spot_width;
|
|
|
|
namespace {
|
|
|
|
// The engine reads pixels in the INT32_MIN(masked)/INT32_MAX(saturated) convention.
|
|
inline bool valid(int32_t v) { return v != INT32_MIN && v != INT32_MAX; }
|
|
|
|
// Nothing inside this radius of the beam centre: the beam stop and its halo are not spots.
|
|
constexpr float MIN_BEAM_DISTANCE_PX = 60.0f;
|
|
// A neighbour this close puts its own flux inside the aperture, which would read as extra width.
|
|
constexpr float ISOLATION_PX = 28.0f;
|
|
// Spots taken per resolution band per image, strongest first.
|
|
constexpr int PER_BAND_PER_IMAGE = 40;
|
|
// The r <= 4 px sum must be this many sigma above the background before the tail is believed.
|
|
constexpr double SNR_MIN = 15.0;
|
|
constexpr int R_CENTROID = 4;
|
|
constexpr double MAX_CENTROID_OFFSET_PX = 2.0;
|
|
// Spots needed before a band, and the crystal, are characterised at all.
|
|
constexpr size_t MIN_SPOTS_PER_BAND = 15;
|
|
constexpr size_t MIN_SPOTS_TOTAL = 20;
|
|
|
|
// Resolution bands, A. The quota is per band, so a crystal is characterised over its whole range
|
|
// and not wherever its strongest spots happen to sit.
|
|
constexpr int N_BAND = 5;
|
|
constexpr std::array<std::pair<float, float>, N_BAND> BANDS = {{
|
|
{2.0f, 3.0f}, {3.0f, 4.5f}, {4.5f, 7.0f}, {7.0f, 12.0f}, {12.0f, 30.0f}}};
|
|
|
|
// Every radius here is compared against an integer pixel offset, so all of it is exact integer
|
|
// arithmetic and no square root is needed anywhere in the pixel loops.
|
|
constexpr int isqrt_floor(int n) {
|
|
int r = 0;
|
|
while ((r + 1) * (r + 1) <= n) r++;
|
|
return r;
|
|
}
|
|
|
|
// Half-width of the disk of radius R on row dy: the largest |dx| with dx^2 + dy^2 <= R^2. Walking
|
|
// the rows by their own extent visits the disk itself rather than its bounding box.
|
|
template <int R>
|
|
constexpr std::array<int, R + 1> disk_row_half() {
|
|
std::array<int, R + 1> a{};
|
|
for (int dy = 0; dy <= R; dy++) a[dy] = isqrt_floor(R * R - dy * dy);
|
|
return a;
|
|
}
|
|
constexpr auto HALF_BKG = disk_row_half<R_BKG_OUT>();
|
|
constexpr auto HALF_CORE = disk_row_half<R_CENTROID>();
|
|
|
|
// The largest |dx| on row dy that is still INSIDE the background ring's inner edge, so |dx| beyond
|
|
// it is in the ring; -1 where the whole row is.
|
|
constexpr std::array<int, R_BKG_OUT + 1> ring_row_inner() {
|
|
std::array<int, R_BKG_OUT + 1> a{};
|
|
for (int dy = 0; dy <= R_BKG_OUT; dy++) {
|
|
const int rem = R_BKG_IN * R_BKG_IN - dy * dy - 1;
|
|
a[dy] = rem < 0 ? -1 : isqrt_floor(rem);
|
|
}
|
|
return a;
|
|
}
|
|
constexpr auto INNER_BKG = ring_row_inner();
|
|
|
|
// floor(sqrt(n)) for every squared distance the encircled-flux aperture can produce, so a pixel's
|
|
// radial bin - the smallest integer radius that contains it - is a table lookup and a compare.
|
|
constexpr std::array<int, R_MAX * R_MAX + 1> isqrt_lookup() {
|
|
std::array<int, R_MAX * R_MAX + 1> a{};
|
|
for (int n = 0; n <= R_MAX * R_MAX; n++) a[n] = isqrt_floor(n);
|
|
return a;
|
|
}
|
|
constexpr auto ISQRT = isqrt_lookup();
|
|
|
|
int band_of(float d_A) {
|
|
for (int b = 0; b < N_BAND; b++)
|
|
if (d_A >= BANDS[b].first && d_A < BANDS[b].second) return b;
|
|
return -1;
|
|
}
|
|
|
|
// The radius at which the curve reaches `frac`, linearly interpolated. prof[i] is the flux inside
|
|
// radius i+1.
|
|
float interpolate_radius(double frac, const std::array<float, R_MAX> &prof) {
|
|
if (prof[0] >= frac)
|
|
return prof[0] > 0.0f ? static_cast<float>(frac / prof[0]) : 1.0f;
|
|
for (int i = 1; i < R_MAX; i++)
|
|
if (prof[i] >= frac)
|
|
return static_cast<float>(i + (frac - prof[i - 1]) / (prof[i] - prof[i - 1]));
|
|
return static_cast<float>(R_MAX);
|
|
}
|
|
|
|
template <typename T>
|
|
double median_of(std::vector<T> &v) {
|
|
if (v.empty()) return 0.0;
|
|
const size_t mid = v.size() / 2;
|
|
std::nth_element(v.begin(), v.begin() + mid, v.end());
|
|
const double hi = v[mid];
|
|
if (v.size() % 2 == 1) return hi;
|
|
return 0.5 * (hi + *std::max_element(v.begin(), v.begin() + mid));
|
|
}
|
|
|
|
} // namespace
|
|
|
|
void MeasureSpotFluxCurves(const ImagePreprocessorBuffer &image, int width, int height,
|
|
const DiffractionGeometry &geometry,
|
|
const std::vector<DiffractionSpot> &spots,
|
|
std::vector<FluxCurve> &out) {
|
|
if (spots.empty()) return;
|
|
|
|
const int32_t *pixels = image.data();
|
|
if (pixels == nullptr) return;
|
|
const float beam_x = geometry.GetBeamX_pxl(), beam_y = geometry.GetBeamY_pxl();
|
|
|
|
// Where every spot of this image sits, so isolation can be tested against all of them and not
|
|
// only against the ones that survive the gates below.
|
|
std::vector<Coord> centre(spots.size());
|
|
for (size_t i = 0; i < spots.size(); i++)
|
|
centre[i] = spots[i].RawCoord();
|
|
|
|
// Isolation on a grid of ISOLATION_PX cells: a neighbour within that distance is in this cell or
|
|
// one of the eight around it. The grid is held as a counting sort - one index array and one
|
|
// offset array - rather than a vector per cell, which on a crowded detector is tens of thousands
|
|
// of allocations per image for a structure that is read once.
|
|
const int gw = static_cast<int>(width / ISOLATION_PX) + 1;
|
|
const int gh = static_cast<int>(height / ISOLATION_PX) + 1;
|
|
const size_t ncell = static_cast<size_t>(gw) * gh;
|
|
const auto cell_of = [&](const Coord &c) {
|
|
const int gx = std::clamp(static_cast<int>(c.x / ISOLATION_PX), 0, gw - 1);
|
|
const int gy = std::clamp(static_cast<int>(c.y / ISOLATION_PX), 0, gh - 1);
|
|
return static_cast<size_t>(gy) * gw + gx;
|
|
};
|
|
std::vector<uint32_t> cell_begin(ncell + 1, 0), cell_item(spots.size()), spot_cell(spots.size());
|
|
for (size_t i = 0; i < spots.size(); i++) {
|
|
spot_cell[i] = static_cast<uint32_t>(cell_of(centre[i]));
|
|
cell_begin[spot_cell[i] + 1]++;
|
|
}
|
|
for (size_t c = 0; c < ncell; c++) cell_begin[c + 1] += cell_begin[c];
|
|
{
|
|
std::vector<uint32_t> cursor(cell_begin.begin(), cell_begin.end() - 1);
|
|
for (size_t i = 0; i < spots.size(); i++)
|
|
cell_item[cursor[spot_cell[i]]++] = static_cast<uint32_t>(i);
|
|
}
|
|
constexpr double ISOLATION_PX2 = static_cast<double>(ISOLATION_PX) * ISOLATION_PX;
|
|
const auto isolated = [&](size_t i) {
|
|
const int gx = static_cast<int>(spot_cell[i] % gw), gy = static_cast<int>(spot_cell[i] / gw);
|
|
for (int y = std::max(0, gy - 1); y <= std::min(gh - 1, gy + 1); y++)
|
|
for (int x = std::max(0, gx - 1); x <= std::min(gw - 1, gx + 1); x++) {
|
|
const size_t c = static_cast<size_t>(y) * gw + x;
|
|
for (uint32_t k = cell_begin[c]; k < cell_begin[c + 1]; k++) {
|
|
const uint32_t j = cell_item[k];
|
|
if (j == i) continue;
|
|
const double ddx = centre[j].x - centre[i].x, ddy = centre[j].y - centre[i].y;
|
|
if (ddx * ddx + ddy * ddy < ISOLATION_PX2) return false;
|
|
}
|
|
}
|
|
return true;
|
|
};
|
|
|
|
// Candidates that pass the geometric gates, by band, strongest first.
|
|
struct Candidate { size_t index; int64_t count; float d_A; };
|
|
std::array<std::vector<Candidate>, N_BAND> candidates;
|
|
constexpr double MIN_BEAM_DISTANCE_PX2 = static_cast<double>(MIN_BEAM_DISTANCE_PX)
|
|
* MIN_BEAM_DISTANCE_PX;
|
|
for (size_t i = 0; i < spots.size(); i++) {
|
|
const Coord &c = centre[i];
|
|
const int cx = static_cast<int>(std::lround(c.x)), cy = static_cast<int>(std::lround(c.y));
|
|
if (cx < R_BKG_OUT || cy < R_BKG_OUT || cx >= width - R_BKG_OUT || cy >= height - R_BKG_OUT)
|
|
continue;
|
|
const double bx = c.x - beam_x, by = c.y - beam_y;
|
|
if (bx * bx + by * by < MIN_BEAM_DISTANCE_PX2) continue;
|
|
const float d_A = geometry.PxlToRes(c.x, c.y);
|
|
const int band = band_of(d_A);
|
|
if (band < 0) continue;
|
|
if (!isolated(i)) continue;
|
|
candidates[band].push_back({i, spots[i].Count(), d_A});
|
|
}
|
|
|
|
// The ring is gathered as the counts it is - the median of an int list is the same number, and
|
|
// half the bytes move through the partial sort.
|
|
std::vector<int32_t> ring;
|
|
ring.reserve(4 * (R_BKG_OUT + 1) * (R_BKG_OUT - R_BKG_IN + 1));
|
|
for (int band = 0; band < N_BAND; band++) {
|
|
auto &cand = candidates[band];
|
|
const size_t take = std::min<size_t>(cand.size(), PER_BAND_PER_IMAGE);
|
|
std::partial_sort(cand.begin(), cand.begin() + take, cand.end(),
|
|
[](const Candidate &a, const Candidate &b) { return a.count > b.count; });
|
|
for (size_t k = 0; k < take; k++) {
|
|
const Coord &c = centre[cand[k].index];
|
|
const int cx = static_cast<int>(std::lround(c.x)), cy = static_cast<int>(std::lround(c.y));
|
|
const int32_t *centre_px = pixels + static_cast<size_t>(cy) * width + cx;
|
|
|
|
// The background under the spot, and a check that the whole aperture is readable: a hole
|
|
// in it removes flux from one radius and not another, which is exactly the shape this
|
|
// measures.
|
|
ring.clear();
|
|
bool readable = true;
|
|
for (int dy = -R_BKG_OUT; dy <= R_BKG_OUT && readable; dy++) {
|
|
const int half = HALF_BKG[std::abs(dy)], inner = INNER_BKG[std::abs(dy)];
|
|
const int32_t *row = centre_px + static_cast<ptrdiff_t>(dy) * width;
|
|
for (int dx = -half; dx <= half; dx++) {
|
|
const int32_t px = row[dx];
|
|
if (!valid(px)) { readable = false; break; }
|
|
if (dx > inner || dx < -inner) ring.push_back(px);
|
|
}
|
|
}
|
|
if (!readable || ring.size() < 20) continue;
|
|
const size_t n_ring = ring.size();
|
|
const double bkg = median_of(ring);
|
|
|
|
// Flux and centroid over the r <= 4 px core, then the signal-to-noise gate. A weak spot's
|
|
// tail is background, and an encircled-flux curve built on it measures the background.
|
|
double core = 0.0, mx = 0.0, my = 0.0;
|
|
int n_core = 0;
|
|
for (int dy = -R_CENTROID; dy <= R_CENTROID; dy++) {
|
|
const int half = HALF_CORE[std::abs(dy)];
|
|
const int32_t *row = centre_px + static_cast<ptrdiff_t>(dy) * width;
|
|
for (int dx = -half; dx <= half; dx++) {
|
|
const double v = row[dx] - bkg;
|
|
core += v;
|
|
mx += v * dx;
|
|
my += v * dy;
|
|
++n_core;
|
|
}
|
|
}
|
|
if (core <= 0.0) continue;
|
|
const double noise = std::sqrt(core + n_core * std::max(bkg, 0.05)
|
|
* (1.0 + static_cast<double>(n_core) / n_ring));
|
|
if (core / noise < SNR_MIN) continue;
|
|
mx /= core;
|
|
my /= core;
|
|
if (std::abs(mx) > MAX_CENTROID_OFFSET_PX || std::abs(my) > MAX_CENTROID_OFFSET_PX)
|
|
continue;
|
|
|
|
// The encircled flux about that centroid, out to the fixed aperture. Each pixel is added
|
|
// to the one bin its own radius falls in and the curve is the running total over the
|
|
// bins: the encircled flux at t is everything inside t, so adding every pixel into every
|
|
// bin beyond it instead would sum the same aperture R_MAX/2 times over.
|
|
std::array<double, R_MAX + 1> bin{};
|
|
constexpr double R2_MAX = static_cast<double>(R_MAX) * R_MAX;
|
|
for (int dy = -R_MAX; dy <= R_MAX; dy++) {
|
|
const double ddy = dy - my, ddy2 = ddy * ddy;
|
|
if (ddy2 > R2_MAX) continue;
|
|
const double span = std::sqrt(R2_MAX - ddy2);
|
|
const int lo = std::max(-R_MAX, static_cast<int>(std::floor(mx - span)));
|
|
const int hi = std::min(R_MAX, static_cast<int>(std::ceil(mx + span)));
|
|
const int32_t *row = centre_px + static_cast<ptrdiff_t>(dy) * width;
|
|
for (int dx = lo; dx <= hi; dx++) {
|
|
const double ddx = dx - mx, rc2 = ddx * ddx + ddy2;
|
|
if (rc2 > R2_MAX) continue;
|
|
const int s = ISQRT[static_cast<int>(rc2)];
|
|
const int t = std::max(1, static_cast<double>(s) * s == rc2 ? s : s + 1);
|
|
bin[t] += row[dx] - bkg;
|
|
}
|
|
}
|
|
FluxCurve curve;
|
|
curve.d_A = cand[k].d_A;
|
|
double encircled = 0.0;
|
|
for (int t = 1; t <= R_MAX; t++) {
|
|
encircled += bin[t];
|
|
curve.c[t - 1] = static_cast<float>(encircled);
|
|
}
|
|
if (!(curve.c[R_NORM - 1] > 0.0f) || !(curve.c[R_MAX - 1] > 0.0f)) continue;
|
|
const float norm = curve.c[R_NORM - 1];
|
|
for (float &v : curve.c) v /= norm;
|
|
out.push_back(curve);
|
|
}
|
|
}
|
|
}
|
|
|
|
std::optional<float> spot_width::R80AtReference(const std::vector<FluxCurve> &curves) {
|
|
if (curves.size() < MIN_SPOTS_TOTAL) return std::nullopt;
|
|
|
|
// One point per band: the median curve of the band, the radius it holds 80 % of its flux at, and
|
|
// the median resolution it was measured at.
|
|
struct Point { double inv_d; double r80; double weight; };
|
|
std::vector<Point> points;
|
|
std::vector<double> values, band_d;
|
|
std::vector<uint32_t> members;
|
|
for (int b = 0; b < N_BAND; b++) {
|
|
members.clear();
|
|
band_d.clear();
|
|
for (uint32_t i = 0; i < curves.size(); i++)
|
|
if (curves[i].d_A >= BANDS[b].first && curves[i].d_A < BANDS[b].second) {
|
|
members.push_back(i);
|
|
band_d.push_back(curves[i].d_A);
|
|
}
|
|
if (band_d.size() < MIN_SPOTS_PER_BAND) continue;
|
|
std::array<float, R_MAX> profile{};
|
|
for (int t = 0; t < R_MAX; t++) {
|
|
values.clear();
|
|
for (uint32_t i : members) values.push_back(curves[i].c[t]);
|
|
profile[t] = static_cast<float>(median_of(values));
|
|
}
|
|
const double d_med = median_of(band_d);
|
|
if (d_med <= 0.0) continue;
|
|
points.push_back({1.0 / d_med, interpolate_radius(0.8, profile),
|
|
static_cast<double>(band_d.size())});
|
|
}
|
|
if (points.empty()) return std::nullopt;
|
|
if (points.size() == 1) return static_cast<float>(points[0].r80);
|
|
|
|
// The mosaic contribution to the detector footprint grows as 1/d, so r80 is linear in 1/d.
|
|
double sw = 0.0, sx = 0.0, sxx = 0.0, sy = 0.0, sxy = 0.0;
|
|
for (const auto &p : points) {
|
|
sw += p.weight;
|
|
sx += p.weight * p.inv_d;
|
|
sxx += p.weight * p.inv_d * p.inv_d;
|
|
sy += p.weight * p.r80;
|
|
sxy += p.weight * p.inv_d * p.r80;
|
|
}
|
|
const double det = sw * sxx - sx * sx;
|
|
double value = sy / sw;
|
|
if (std::abs(det) > 1e-12) {
|
|
const double c1 = (sw * sxy - sx * sy) / det;
|
|
value = (sy - c1 * sx) / sw + c1 / D_REF_A;
|
|
}
|
|
// Never extrapolate outside what the bands actually measured.
|
|
double lo = std::numeric_limits<double>::max(), hi = 0.0;
|
|
for (const auto &p : points) { lo = std::min(lo, p.r80); hi = std::max(hi, p.r80); }
|
|
return static_cast<float>(std::clamp(value, 0.8 * lo, 1.25 * hi));
|
|
}
|
|
|
|
float spot_width::R1ForWidth(float r80) {
|
|
return std::clamp(std::round(2.0f * r80), 4.0f, 6.0f);
|
|
}
|
|
|
|
bool spot_width::WidthSettled(float r80, float r80_before) {
|
|
return std::abs(r80 - r80_before) < SETTLED_STEP_PX
|
|
&& std::abs(r80 - 2.25f) > SWITCH_CLEARANCE_PX
|
|
&& std::abs(r80 - 2.75f) > SWITCH_CLEARANCE_PX;
|
|
}
|