Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports. * jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls. * Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results. * Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable. * Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate. * Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do. * Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence. * Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags. * Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check. * Rugnux: Clear error messages when a data set needs more GPU or host memory than is available. Reviewed-on: #83 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
144 lines
5.8 KiB
C++
144 lines
5.8 KiB
C++
// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "FFTIndexerCPU.h"
|
|
|
|
#include <cmath>
|
|
#include <algorithm>
|
|
#include <stdexcept>
|
|
#include <cassert>
|
|
#include <mutex>
|
|
#include <vector>
|
|
|
|
#include "../../common/FFTWPlannerLock.h"
|
|
#include "../../common/ParallelFor.h"
|
|
|
|
static inline double dot_abs(const Coord& a, const Coord& b) {
|
|
return std::fabs(a.x * b.x + a.y * b.y + a.z * b.z);
|
|
}
|
|
|
|
FFTIndexerCPU::FFTIndexerCPU(const IndexingSettings& settings)
|
|
: FFTIndexer(settings) {
|
|
|
|
// Allocate host buffers
|
|
h_input_fft.resize(input_size, 0.0);
|
|
h_output_fft.resize(output_size);
|
|
|
|
// Validate allocations vs. expected sizes
|
|
if (h_input_fft.size() != input_size)
|
|
throw std::runtime_error("FFTWIndexer: input buffer size mismatch");
|
|
if (h_output_fft.size() != output_size)
|
|
throw std::runtime_error("FFTWIndexer: output buffer size mismatch");
|
|
|
|
const int H = static_cast<int>(histogram_size);
|
|
const int out_len = (H / 2) + 1;
|
|
|
|
int n[1] = { H };
|
|
|
|
{
|
|
std::unique_lock ul(FFTWPlannerMutex());
|
|
plan = fftwf_plan_many_dft_r2c(
|
|
1, n, nDirections,
|
|
h_input_fft.data(), nullptr, 1, H,
|
|
reinterpret_cast<fftwf_complex*>(h_output_fft.data()), nullptr, 1, out_len,
|
|
FFTW_ESTIMATE);
|
|
}
|
|
|
|
if (!plan)
|
|
throw std::runtime_error("fftw_plan_many_dft_r2c failed");
|
|
}
|
|
|
|
|
|
FFTIndexerCPU::~FFTIndexerCPU() {
|
|
std::unique_lock ul(FFTWPlannerMutex());
|
|
fftwf_destroy_plan(plan);
|
|
}
|
|
|
|
void FFTIndexerCPU::ExecuteFFT(const std::vector<Coord> &coord, size_t nspots) {
|
|
// Build histograms: one per direction
|
|
const int H = static_cast<int>(histogram_size);
|
|
const int D = nDirections;
|
|
const int out_len = (H / 2) + 1;
|
|
|
|
std::fill(h_input_fft.begin(), h_input_fft.end(), 0.0);
|
|
|
|
// Every direction has a histogram and a spectrum of its own, so the directions are split over the
|
|
// refinement threads; each one is still filled and scanned in the same order.
|
|
const size_t nthreads = std::max(1u, refine_threads);
|
|
ParallelChunks(D, nthreads, [&](int d0, int d1) {
|
|
for (int d = d0; d < d1; ++d) {
|
|
float* hist = h_input_fft.data() + static_cast<size_t>(d) * H;
|
|
|
|
for (size_t i = 0; i < nspots; i++) {
|
|
const auto& r = coord[i];
|
|
double dot = dot_abs(direction_vectors[d], r);
|
|
long long bin = static_cast<long long>(dot / histogram_spacing);
|
|
if (bin >= 0 && bin < H) {
|
|
hist[bin] += 1.0;
|
|
}
|
|
}
|
|
}
|
|
});
|
|
|
|
// Plan and execute batched R2C FFT with FFTW
|
|
fftwf_execute(plan);
|
|
|
|
// Post-process: pick the peak past min_length_A by PROMINENCE above a local background.
|
|
// The projected histogram has a broad low-frequency ENVELOPE (spots cluster near the
|
|
// origin) whose magnitude can exceed the true lattice peaks; a plain argmax|spec| then
|
|
// returns a short envelope vector (~10A) on weak/pink-beam frames and the real axes are
|
|
// lost. Subtracting a running-mean background of half-width BG_HALF bins removes that
|
|
// smooth envelope (it cancels to ~0) while sharp lattice peaks - fundamentals AND
|
|
// harmonics - keep their height. The prominence is also reported as the magnitude so
|
|
// FilterFFTResults ranks directions by real-peak strength, not by envelope.
|
|
const double len_coeff = 2.0 * static_cast<double>(max_length_A) / static_cast<double>(H);
|
|
// Background half-window ~15 A (in length, so it is independent of the histogram sizing);
|
|
// wide enough to span the envelope yet narrow enough not to smooth real peaks. Validated
|
|
// optimum on serial-still jet data (de-novo indexing improved markedly vs FFBIDX); a second dataset unchanged.
|
|
constexpr double BG_HALF_WIDTH_A = 15.0;
|
|
const int bg_half = std::max(1, static_cast<int>(std::lround(BG_HALF_WIDTH_A / len_coeff)));
|
|
|
|
ParallelChunks(D, nthreads, [&](int d0, int d1) {
|
|
std::vector<double> mag(out_len), pref(out_len + 1);
|
|
|
|
for (int d = d0; d < d1; ++d) {
|
|
const auto* spec = h_output_fft.data() + static_cast<size_t>(d) * out_len;
|
|
|
|
pref[0] = 0.0;
|
|
for (int j = 0; j < out_len; ++j) {
|
|
mag[j] = std::hypot(spec[j][0], spec[j][1]);
|
|
pref[j + 1] = pref[j] + mag[j];
|
|
}
|
|
|
|
double best_prom = 0.0;
|
|
double best_len = -1.0;
|
|
|
|
for (int j = 0; j < out_len; ++j) {
|
|
double len = len_coeff * static_cast<double>(j);
|
|
if (len <= static_cast<double>(min_length_A)) continue;
|
|
|
|
// Slide the background window inward at the ends rather than truncating it. A peak within
|
|
// bg_half bins of either end - which is where the LONGEST cells sit, the last usable bin
|
|
// being max_length_A itself - otherwise gets its background from a one-sided window, and
|
|
// the prominence it is judged on is biased by however much the spectrum slopes there.
|
|
int lo = j - bg_half, hi = j + bg_half + 1;
|
|
if (lo < 0) { hi = std::min(out_len, hi - lo); lo = 0; }
|
|
if (hi > out_len) { lo = std::max(0, lo - (hi - out_len)); hi = out_len; }
|
|
const double background = (pref[hi] - pref[lo]) / static_cast<double>(hi - lo);
|
|
const double prominence = mag[j] - background;
|
|
|
|
if (prominence > best_prom) {
|
|
best_prom = prominence;
|
|
best_len = len;
|
|
}
|
|
}
|
|
|
|
result_fft[d] = FFTResult{
|
|
.magnitude = static_cast<float>(best_prom),
|
|
.direction = d,
|
|
.length = static_cast<float>(best_len)
|
|
};
|
|
}
|
|
});
|
|
}
|