Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports. * jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls. * Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results. * Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable. * Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate. * Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do. * Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence. * Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags. * Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check. * Rugnux: Clear error messages when a data set needs more GPU or host memory than is available. Reviewed-on: #83 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
212 lines
9.5 KiB
C++
212 lines
9.5 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "BeamCenterFFTCPU.h"
|
|
|
|
#include <algorithm>
|
|
#include <complex>
|
|
#include <cstring>
|
|
#include <future>
|
|
#include <memory>
|
|
#include <mutex>
|
|
#include <stdexcept>
|
|
|
|
#include <fftw3.h>
|
|
|
|
#include "../../common/FFTWPlannerLock.h"
|
|
|
|
namespace {
|
|
|
|
// One real-to-complex 2D FFT workspace of fixed padded size, reused for every forward transform.
|
|
class Rfft2 {
|
|
int64_t ny, nx, nxc;
|
|
std::vector<float> real;
|
|
fftwf_plan fwd = nullptr;
|
|
fftwf_plan bwd = nullptr;
|
|
|
|
public:
|
|
std::vector<std::complex<float>> spectrum;
|
|
|
|
Rfft2(int64_t ny_, int64_t nx_) : ny(ny_), nx(nx_), nxc(nx_ / 2 + 1) {
|
|
real.resize(static_cast<size_t>(ny * nx));
|
|
spectrum.resize(static_cast<size_t>(ny * nxc));
|
|
std::unique_lock lock(FFTWPlannerMutex());
|
|
fwd = fftwf_plan_dft_r2c_2d(static_cast<int>(ny), static_cast<int>(nx), real.data(),
|
|
reinterpret_cast<fftwf_complex *>(spectrum.data()),
|
|
FFTW_ESTIMATE);
|
|
bwd = fftwf_plan_dft_c2r_2d(static_cast<int>(ny), static_cast<int>(nx),
|
|
reinterpret_cast<fftwf_complex *>(spectrum.data()),
|
|
real.data(), FFTW_ESTIMATE);
|
|
if (!fwd || !bwd)
|
|
throw std::runtime_error("BeamCenterFFT: fftwf plan failed");
|
|
}
|
|
Rfft2(const Rfft2 &) = delete;
|
|
Rfft2 &operator=(const Rfft2 &) = delete;
|
|
~Rfft2() {
|
|
std::unique_lock lock(FFTWPlannerMutex());
|
|
if (fwd)
|
|
fftwf_destroy_plan(fwd);
|
|
if (bwd)
|
|
fftwf_destroy_plan(bwd);
|
|
}
|
|
|
|
// Zero-pad `src` (height x width) into the workspace and return its spectrum.
|
|
std::vector<std::complex<float>> Forward(const std::vector<float> &src, int64_t height,
|
|
int64_t width) {
|
|
std::fill(real.begin(), real.end(), 0.0f);
|
|
for (int64_t y = 0; y < height; y++)
|
|
std::memcpy(&real[static_cast<size_t>(y * nx)], &src[static_cast<size_t>(y * width)],
|
|
sizeof(float) * static_cast<size_t>(width));
|
|
fftwf_execute(fwd);
|
|
return spectrum; // copy: the workspace is reused
|
|
}
|
|
|
|
// Inverse-transform the elementwise product u*v, normalised, cropped to out_h x out_w.
|
|
std::vector<float> InverseProduct(const std::vector<std::complex<float>> &u,
|
|
const std::vector<std::complex<float>> &v, int64_t out_h,
|
|
int64_t out_w) {
|
|
const float norm = 1.0f / (static_cast<float>(ny) * static_cast<float>(nx));
|
|
for (size_t i = 0; i < spectrum.size(); i++)
|
|
spectrum[i] = u[i] * v[i];
|
|
fftwf_execute(bwd);
|
|
std::vector<float> out(static_cast<size_t>(out_h * out_w));
|
|
for (int64_t y = 0; y < out_h; y++)
|
|
for (int64_t x = 0; x < out_w; x++)
|
|
out[static_cast<size_t>(y * out_w + x)] =
|
|
real[static_cast<size_t>(y * nx + x)] * norm;
|
|
return out;
|
|
}
|
|
};
|
|
|
|
std::vector<float> Transpose(const std::vector<float> &src, int64_t h, int64_t w) {
|
|
std::vector<float> out(src.size());
|
|
for (int64_t y = 0; y < h; y++)
|
|
for (int64_t x = 0; x < w; x++)
|
|
out[static_cast<size_t>(x * h + y)] = src[static_cast<size_t>(y * w + x)];
|
|
return out;
|
|
}
|
|
|
|
} // namespace
|
|
|
|
BeamCenterConvSurfaces2D BeamCenterFFTCPU::PointSurfaces(const std::vector<float> &a,
|
|
const std::vector<float> &m, int64_t h,
|
|
int64_t w) {
|
|
// The three forward transforms are independent of each other, and so are the four inverse ones,
|
|
// so each group runs at the same time, one workspace per transform. A workspace is built exactly
|
|
// as a single one was - its own buffers and its own plan for them - so every transform is the one
|
|
// it was, bit for bit; the four inverse ones reuse the three forward workspaces and one more.
|
|
const int64_t ny = BeamCenterFFTPadSize(2 * h), nx = BeamCenterFFTPadSize(2 * w);
|
|
std::vector<std::unique_ptr<Rfft2>> fft;
|
|
for (int i = 0; i < 4; i++)
|
|
fft.push_back(std::make_unique<Rfft2>(ny, nx));
|
|
|
|
std::vector<float> a2(a.size());
|
|
for (size_t i = 0; i < a2.size(); i++)
|
|
a2[i] = a[i] * a[i];
|
|
auto fA = std::async(std::launch::async, [&] { return fft[0]->Forward(a, h, w); });
|
|
auto fM = std::async(std::launch::async, [&] { return fft[1]->Forward(m, h, w); });
|
|
auto fA2 = std::async(std::launch::async, [&] { return fft[2]->Forward(a2, h, w); });
|
|
const auto A = fA.get();
|
|
const auto M = fM.get();
|
|
const auto A2 = fA2.get();
|
|
|
|
BeamCenterConvSurfaces2D out;
|
|
auto fC = std::async(std::launch::async, [&] { return fft[0]->InverseProduct(A, A, 2 * h, 2 * w); });
|
|
auto fS = std::async(std::launch::async, [&] { return fft[1]->InverseProduct(A, M, 2 * h, 2 * w); });
|
|
auto fQ = std::async(std::launch::async, [&] { return fft[2]->InverseProduct(A2, M, 2 * h, 2 * w); });
|
|
auto fD = std::async(std::launch::async, [&] { return fft[3]->InverseProduct(M, M, 2 * h, 2 * w); });
|
|
out.C = fC.get();
|
|
out.S = fS.get();
|
|
out.Q = fQ.get();
|
|
out.D = fD.get();
|
|
return out;
|
|
}
|
|
|
|
// Each sequence is correlated with its own mirror and the four accumulators are summed over the
|
|
// other coordinate in double complex (the same cancellation argument as the 2D surface), so only
|
|
// four inverse transforms are needed however many sequences there are.
|
|
BeamCenterConvSurfaces1D BeamCenterFFTCPU::LineSurfaces(const std::vector<float> &a_in,
|
|
const std::vector<float> &m_in, int64_t h,
|
|
int64_t w, BeamCenterMirror mirror) {
|
|
// Mirroring the column coordinate is the same computation on the transposed image.
|
|
const std::vector<float> at = mirror == BeamCenterMirror::Rows ? std::vector<float>()
|
|
: Transpose(a_in, h, w);
|
|
const std::vector<float> mt = mirror == BeamCenterMirror::Rows ? std::vector<float>()
|
|
: Transpose(m_in, h, w);
|
|
const std::vector<float> &a = mirror == BeamCenterMirror::Rows ? a_in : at;
|
|
const std::vector<float> &m = mirror == BeamCenterMirror::Rows ? m_in : mt;
|
|
const int64_t nseq = mirror == BeamCenterMirror::Rows ? h : w;
|
|
const int64_t nbatch = mirror == BeamCenterMirror::Rows ? w : h;
|
|
|
|
const int64_t nfft = BeamCenterFFTPadSize(2 * nseq);
|
|
const int64_t nc = nfft / 2 + 1;
|
|
|
|
std::vector<float> in(static_cast<size_t>(nfft));
|
|
std::vector<std::complex<float>> A(static_cast<size_t>(nc)), M(static_cast<size_t>(nc)),
|
|
A2(static_cast<size_t>(nc));
|
|
std::vector<std::complex<double>> accC(static_cast<size_t>(nc)), accS(static_cast<size_t>(nc)),
|
|
accQ(static_cast<size_t>(nc)), accD(static_cast<size_t>(nc));
|
|
|
|
fftwf_plan fwd, bwd;
|
|
{
|
|
std::unique_lock lock(FFTWPlannerMutex());
|
|
fwd = fftwf_plan_dft_r2c_1d(static_cast<int>(nfft), in.data(),
|
|
reinterpret_cast<fftwf_complex *>(A.data()), FFTW_ESTIMATE);
|
|
bwd = fftwf_plan_dft_c2r_1d(static_cast<int>(nfft),
|
|
reinterpret_cast<fftwf_complex *>(A.data()), in.data(),
|
|
FFTW_ESTIMATE);
|
|
}
|
|
if (!fwd || !bwd)
|
|
throw std::runtime_error("BeamCenterFFT: fftwf 1D plan failed");
|
|
|
|
auto forward_into = [&](auto value_of, std::vector<std::complex<float>> &dst) {
|
|
for (int64_t y = 0; y < nseq; y++)
|
|
in[static_cast<size_t>(y)] = value_of(y);
|
|
std::fill(in.begin() + nseq, in.end(), 0.0f);
|
|
fftwf_execute_dft_r2c(fwd, in.data(), reinterpret_cast<fftwf_complex *>(A.data()));
|
|
dst = A;
|
|
};
|
|
|
|
for (int64_t x = 0; x < nbatch; x++) {
|
|
forward_into([&](int64_t y) { return a[static_cast<size_t>(y * nbatch + x)]; }, A2);
|
|
std::swap(A, A2); // A = spectrum of the sequence of a
|
|
std::vector<std::complex<float>> Acol = A;
|
|
forward_into([&](int64_t y) { return m[static_cast<size_t>(y * nbatch + x)]; }, M);
|
|
forward_into(
|
|
[&](int64_t y) {
|
|
const float v = a[static_cast<size_t>(y * nbatch + x)];
|
|
return v * v;
|
|
},
|
|
A2);
|
|
for (int64_t i = 0; i < nc; i++) {
|
|
const std::complex<double> ca(Acol[i]), cm(M[i]), ca2(A2[i]);
|
|
accC[i] += ca * ca;
|
|
accS[i] += ca * cm;
|
|
accQ[i] += ca2 * cm;
|
|
accD[i] += cm * cm;
|
|
}
|
|
}
|
|
|
|
const double norm = 1.0 / static_cast<double>(nfft);
|
|
auto inverse = [&](const std::vector<std::complex<double>> &acc) {
|
|
for (int64_t i = 0; i < nc; i++)
|
|
A[i] = std::complex<float>(acc[i]);
|
|
fftwf_execute_dft_c2r(bwd, reinterpret_cast<fftwf_complex *>(A.data()), in.data());
|
|
std::vector<double> out(static_cast<size_t>(2 * nseq));
|
|
for (int64_t i = 0; i < 2 * nseq; i++)
|
|
out[static_cast<size_t>(i)] = in[static_cast<size_t>(i)] * norm;
|
|
return out;
|
|
};
|
|
BeamCenterConvSurfaces1D out;
|
|
out.C = inverse(accC);
|
|
out.S = inverse(accS);
|
|
out.Q = inverse(accQ);
|
|
out.D = inverse(accD);
|
|
{
|
|
std::unique_lock lock(FFTWPlannerMutex());
|
|
fftwf_destroy_plan(fwd);
|
|
fftwf_destroy_plan(bwd);
|
|
}
|
|
return out;
|
|
}
|