Files
Jungfraujoch/image_analysis/geom_refinement/BeamCenterFFTGPU.cu
T
leonarski_f 84228bf8be
Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
v1.0.0-rc.173 (#83)
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports.
* jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls.
* Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results.
* Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable.
* Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate.
* Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do.
* Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence.
* Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags.
* Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check.
* Rugnux: Clear error messages when a data set needs more GPU or host memory than is available.

Reviewed-on: #83
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-09-29 15:57:32 +02:00

526 lines
25 KiB
Plaintext

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "BeamCenterFFTGPU.h"
#include <algorithm>
#include <cmath>
#include <limits>
#include <cuda_runtime.h>
#include <cufft.h>
#include "../indexing/CUDAMemHelpers.h"
namespace {
constexpr int BLOCK = 256;
// Threads per frequency in the 1D accumulation. A power of two, for the shared-memory reduction.
constexpr int ACCUMULATE_THREADS = 128;
// Blocks in the greedy search's reduction. Each returns one candidate, and the last step of the
// reduction is done on the host over that many - a thousand is far below what a download costs and
// far above what the card needs to stay busy.
constexpr int SHORTLIST_BLOCKS = 1024;
// Every buffer here is read by work on the legacy NULL stream - the kernels, the cuFFT plans and the
// copies below are all queued there - so none of them may come from the pool. A pooled buffer is
// freed with cudaFreeAsync on the thread's allocation stream, which is non-blocking and so not
// ordered after the NULL stream: the free completed at once while a Suppress or an inverse transform
// was still writing, and the pool handed those bytes to the analysis engines being built on other
// workers at the same moment, or unmapped them. cudaFree synchronises the device first.
constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous;
int Blocks(int64_t n) { return static_cast<int>((n + BLOCK - 1) / BLOCK); }
void Check(cudaError_t err, const char *what) {
if (err != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, what);
}
void Check(cufftResult res, const char *what) {
if (res != CUFFT_SUCCESS)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, what);
}
std::vector<int> Dims(int64_t a) { return {static_cast<int>(a)}; }
std::vector<int> Dims(int64_t a, int64_t b) { return {static_cast<int>(a), static_cast<int>(b)}; }
__global__ void SquareInPlace(float *data, int64_t n) {
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
if (i < n)
data[i] *= data[i];
}
__global__ void MultiplySpectra(const cufftComplex *u, const cufftComplex *v, cufftComplex *out,
int64_t n) {
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
if (i < n)
out[i] = cuCmulf(u[i], v[i]);
}
// The inverse transform leaves an unnormalised ny x nx result; only its 2h x 2w corner is the
// convolution surface. Scale that corner in place, for the one surface that stays where it is...
__global__ void ScaleCorner(float *data, int64_t nx, int64_t out_h, int64_t out_w, float norm) {
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
if (i >= out_h * out_w)
return;
const int64_t y = i / out_w;
data[y * nx + (i - y * out_w)] *= norm;
}
// ...and scale it into a surface of its own for the three that have to outlive the next inverse.
__global__ void CopyScaledCorner(const float *src, int64_t nx, int64_t out_h, int64_t out_w,
float norm, float *dst) {
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
if (i >= out_h * out_w)
return;
const int64_t y = i / out_w;
dst[i] = src[y * nx + (i - y * out_w)] * norm;
}
// Lay each sequence out contiguously, zero-padded to nfft, so the 1D transforms are a plain batch:
// out[b * nfft + s] = img[b * stride_batch + s * stride_seq], squared where asked.
__global__ void GatherSequences(const float *img, int64_t nseq, int64_t nbatch, int64_t stride_seq,
int64_t stride_batch, int64_t nfft, int square, float *out) {
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
if (i >= nbatch * nfft)
return;
const int64_t b = i / nfft;
const int64_t s = i - b * nfft;
float v = 0.0f;
if (s < nseq) {
v = img[b * stride_batch + s * stride_seq];
if (square)
v *= v;
}
out[i] = v;
}
// The four accumulators of the 1D score, summed over the batch in double - the same widening the
// host path does, and for the same reason: the numerator and denominator below are cancellations
// of large near-equal terms. One block per frequency.
__global__ void AccumulateSpectra(const cufftComplex *A, const cufftComplex *M,
const cufftComplex *A2, int64_t nbatch, int64_t nc, double2 *accC,
double2 *accS, double2 *accQ, double2 *accD) {
__shared__ double2 red[4][ACCUMULATE_THREADS];
const int64_t i = blockIdx.x;
double2 c = {0.0, 0.0}, s = {0.0, 0.0}, q = {0.0, 0.0}, d = {0.0, 0.0};
for (int64_t b = threadIdx.x; b < nbatch; b += blockDim.x) {
const cufftComplex fa = A[b * nc + i], fm = M[b * nc + i], fa2 = A2[b * nc + i];
const double ar = fa.x, ai = fa.y, mr = fm.x, mi = fm.y, qr = fa2.x, qi = fa2.y;
c.x += ar * ar - ai * ai;
c.y += ar * ai + ai * ar;
s.x += ar * mr - ai * mi;
s.y += ar * mi + ai * mr;
q.x += qr * mr - qi * mi;
q.y += qr * mi + qi * mr;
d.x += mr * mr - mi * mi;
d.y += mr * mi + mi * mr;
}
red[0][threadIdx.x] = c;
red[1][threadIdx.x] = s;
red[2][threadIdx.x] = q;
red[3][threadIdx.x] = d;
for (unsigned stride = blockDim.x / 2; stride > 0; stride /= 2) {
__syncthreads();
if (threadIdx.x < stride)
for (int k = 0; k < 4; k++) {
red[k][threadIdx.x].x += red[k][threadIdx.x + stride].x;
red[k][threadIdx.x].y += red[k][threadIdx.x + stride].y;
}
}
if (threadIdx.x == 0) {
accC[i] = red[0][0];
accS[i] = red[1][0];
accQ[i] = red[2][0];
accD[i] = red[3][0];
}
}
// The four accumulators narrowed back to single precision and packed as one batch of four, ready
// for the inverse transform - exactly what the host path does before its four inverses.
__global__ void PackAccumulators(const double2 *accC, const double2 *accS, const double2 *accQ,
const double2 *accD, int64_t nc, cufftComplex *out) {
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
if (i >= nc)
return;
const double2 *acc[4] = {accC, accS, accQ, accD};
for (int k = 0; k < 4; k++)
out[k * nc + i] = make_cuFloatComplex(static_cast<float>(acc[k][i].x),
static_cast<float>(acc[k][i].y));
}
// The masked Pearson, elementwise in double, exactly as BeamCenterPointScore does it on the host -
// the numerator and the denominator are cancellations of large near-equal terms. Q is read out of
// the transform buffer where the last inverse left it, so it needs no surface of its own. The score
// replaces C, which nothing wants afterwards.
__global__ void PearsonKernel(float *C, const float *S, const float *D, const float *Q_padded,
int64_t nx, int64_t out_h, int64_t out_w, double lim,
double variance_limit) {
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
if (i >= out_h * out_w)
return;
const double d = D[i];
if (d <= lim) {
C[i] = -INFINITY;
return;
}
const int64_t y = i / out_w;
const double q = Q_padded[y * nx + (i - y * out_w)];
const double s = S[i];
const double num = d * C[i] - s * s;
const double den = d * q - s * s;
C[i] = den > d * d * variance_limit ? static_cast<float>(num / den) : -INFINITY;
}
// The host's rule for the greedy search, so the device finds the same peak: a strictly greater
// value wins, among equal values the lower index does, and an index of -1 means "nothing found"
// (which is what a surface of -infinity leaves, and what ends the shortlist).
__device__ bool TakeRight(float lv, int64_t li, float rv, int64_t ri) {
if (ri < 0)
return false;
if (li < 0)
return true;
return rv > lv || (rv == lv && ri < li);
}
__global__ void BlockBest(const float *r, int64_t n, float *block_value, int64_t *block_index) {
__shared__ float sv[BLOCK];
__shared__ int64_t si[BLOCK];
float bv = -INFINITY;
int64_t bi = -1;
for (int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x; i < n;
i += static_cast<int64_t>(gridDim.x) * blockDim.x)
if (r[i] > bv) { // increasing i, so the lowest index among equal values is the one kept
bv = r[i];
bi = i;
}
sv[threadIdx.x] = bv;
si[threadIdx.x] = bi;
for (unsigned stride = blockDim.x / 2; stride > 0; stride /= 2) {
__syncthreads();
if (threadIdx.x < stride
&& TakeRight(sv[threadIdx.x], si[threadIdx.x], sv[threadIdx.x + stride],
si[threadIdx.x + stride])) {
sv[threadIdx.x] = sv[threadIdx.x + stride];
si[threadIdx.x] = si[threadIdx.x + stride];
}
}
if (threadIdx.x == 0) {
block_value[blockIdx.x] = sv[0];
block_index[blockIdx.x] = si[0];
}
}
// The non-maximum suppression square around a peak that has been taken.
__global__ void Suppress(float *r, int64_t w, int64_t x0, int64_t x1, int64_t y0, int64_t y1) {
const int64_t cols = x1 - x0 + 1;
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
if (i >= (y1 - y0 + 1) * cols)
return;
const int64_t y = i / cols;
r[(y0 + y) * w + x0 + (i - y * cols)] = -INFINITY;
}
void CheckLastKernel(const char *what) {
Check(cudaGetLastError(), what);
}
// The 2D transform set, left on the device. Only THREE 2h x 2w surfaces are ever allocated: the
// fourth is the last inverse's output, and nothing overwrites the transform buffer afterwards, so
// Q is read where it lies.
struct PointSurfacesDevice {
CudaDevicePtr<float> C, S, D; // 2h x 2w, normalised
CudaDevicePtr<float> real; // ny x nx; its 2h x 2w corner is Q, normalised
int64_t nx = 0;
};
PointSurfacesDevice TransformPoint(const std::vector<float> &a, const std::vector<float> &m,
int64_t h, int64_t w) {
const int64_t ny = BeamCenterFFTPadSize(2 * h);
const int64_t nx = BeamCenterFFTPadSize(2 * w);
const int64_t nxc = nx / 2 + 1;
const int64_t nreal = ny * nx;
const int64_t ncomplex = ny * nxc;
const int64_t ncrop = 4 * h * w;
PointSurfacesDevice out;
out.nx = nx;
out.real = CudaDevicePtr<float>(nreal, ALLOC);
CudaDevicePtr<cufftComplex> d_a(ncomplex, ALLOC), d_m(ncomplex, ALLOC), d_prod(ncomplex, ALLOC);
CudaFFTPlan fwd(2, Dims(ny, nx), Dims(ny, nx), 1, static_cast<int>(nreal), Dims(ny, nxc), 1,
static_cast<int>(ncomplex), CUFFT_R2C, 1);
CudaFFTPlan inv(2, Dims(ny, nx), Dims(ny, nxc), 1, static_cast<int>(ncomplex), Dims(ny, nx), 1,
static_cast<int>(nreal), CUFFT_C2R, 1);
auto upload = [&](const std::vector<float> &src) {
Check(cudaMemset(out.real.get(), 0, nreal * sizeof(float)),
"BeamCenterFFT: memset failed");
Check(cudaMemcpy2D(out.real.get(), nx * sizeof(float), src.data(), w * sizeof(float),
w * sizeof(float), h, cudaMemcpyHostToDevice),
"BeamCenterFFT: image upload failed");
};
upload(a);
Check(cufftExecR2C(fwd, out.real, d_a), "BeamCenterFFT: forward transform failed");
upload(m);
Check(cufftExecR2C(fwd, out.real, d_m), "BeamCenterFFT: forward transform failed");
const float norm = 1.0f / (static_cast<float>(ny) * static_cast<float>(nx));
// The inverse leaves an unnormalised ny x nx result whose 2h x 2w corner is the surface.
auto inverse_product = [&](const cufftComplex *u, const cufftComplex *v) {
MultiplySpectra<<<Blocks(ncomplex), BLOCK>>>(u, v, d_prod.get(), ncomplex);
CheckLastKernel("BeamCenterFFT: spectrum product failed");
Check(cufftExecC2R(inv, d_prod, out.real), "BeamCenterFFT: inverse transform failed");
};
auto take_corner = [&](CudaDevicePtr<float> &dst) {
dst = CudaDevicePtr<float>(ncrop, ALLOC);
CopyScaledCorner<<<Blocks(ncrop), BLOCK>>>(out.real.get(), nx, 2 * h, 2 * w, norm,
dst.get());
CheckLastKernel("BeamCenterFFT: surface scaling failed");
};
inverse_product(d_a, d_a);
take_corner(out.C);
inverse_product(d_a, d_m);
take_corner(out.S);
inverse_product(d_m, d_m);
take_corner(out.D);
// The spectrum of a is finished with, so a^2 goes in its place rather than in a fourth buffer.
upload(a);
SquareInPlace<<<Blocks(nreal), BLOCK>>>(out.real.get(), nreal);
CheckLastKernel("BeamCenterFFT: squaring failed");
Check(cufftExecR2C(fwd, out.real, d_a), "BeamCenterFFT: forward transform failed");
inverse_product(d_a, d_m);
ScaleCorner<<<Blocks(ncrop), BLOCK>>>(out.real.get(), nx, 2 * h, 2 * w, norm);
CheckLastKernel("BeamCenterFFT: surface scaling failed");
return out;
}
// The masked Pearson over the four surfaces, leaving the score in `surf.C`.
void ScorePoint(PointSurfacesDevice &surf, int64_t h, int64_t w, float min_pair_fraction,
double global_variance) {
const int64_t ncrop = 4 * h * w;
CudaDevicePtr<float> block_value(SHORTLIST_BLOCKS, ALLOC);
CudaDevicePtr<int64_t> block_index(SHORTLIST_BLOCKS, ALLOC);
BlockBest<<<SHORTLIST_BLOCKS, BLOCK>>>(surf.D.get(), ncrop, block_value.get(),
block_index.get());
CheckLastKernel("BeamCenterFFT: pair-count maximum failed");
std::vector<float> values(SHORTLIST_BLOCKS);
Check(cudaMemcpy(values.data(), block_value.get(), values.size() * sizeof(float),
cudaMemcpyDeviceToHost),
"BeamCenterFFT: pair-count maximum download failed");
float d_max = 0.0f; // as on the host, where the running maximum starts at zero
for (float v : values)
d_max = std::max(d_max, v);
PearsonKernel<<<Blocks(ncrop), BLOCK>>>(surf.C.get(), surf.S.get(), surf.D.get(),
surf.real.get(), surf.nx, 2 * h, 2 * w,
min_pair_fraction * static_cast<double>(d_max),
BEAM_CENTER_VARIANCE_FLOOR * global_variance);
CheckLastKernel("BeamCenterFFT: masked Pearson failed");
}
} // namespace
BeamCenterConvSurfaces2D BeamCenterFFTGPU::PointSurfaces(const std::vector<float> &a,
const std::vector<float> &m, int64_t h,
int64_t w) {
// The four surfaces brought back as the interface describes them. Nothing in the pipeline asks
// for this any more - PointShortlist below keeps them on the device - but the shared host code
// is written on them and the parity test compares them, so the engine still supplies them.
PointSurfacesDevice surf = TransformPoint(a, m, h, w);
const size_t ncrop = static_cast<size_t>(4 * h * w);
BeamCenterConvSurfaces2D out;
auto download = [&](const CudaDevicePtr<float> &src, std::vector<float> &dst) {
dst.resize(ncrop);
Check(cudaMemcpy(dst.data(), src.get(), ncrop * sizeof(float), cudaMemcpyDeviceToHost),
"BeamCenterFFT: surface download failed");
};
download(surf.C, out.C);
download(surf.S, out.S);
download(surf.D, out.D);
out.Q.resize(ncrop);
Check(cudaMemcpy2D(out.Q.data(), 2 * w * sizeof(float), surf.real.get(),
surf.nx * sizeof(float), 2 * w * sizeof(float), 2 * h,
cudaMemcpyDeviceToHost),
"BeamCenterFFT: surface download failed");
return out;
}
std::vector<float> BeamCenterFFTGPU::PointScoreSurface(const std::vector<float> &a,
const std::vector<float> &m, int64_t h,
int64_t w, float min_pair_fraction,
double global_variance) {
PointSurfacesDevice surf = TransformPoint(a, m, h, w);
ScorePoint(surf, h, w, min_pair_fraction, global_variance);
std::vector<float> out(static_cast<size_t>(4 * h * w));
Check(cudaMemcpy(out.data(), surf.C.get(), out.size() * sizeof(float), cudaMemcpyDeviceToHost),
"BeamCenterFFT: score download failed");
return out;
}
std::vector<BeamCenterFFTCandidate>
BeamCenterFFTGPU::PointShortlist(const std::vector<float> &a, const std::vector<float> &m,
int64_t h, int64_t w, const BeamCenterFFTSettings &settings,
double global_variance) {
PointSurfacesDevice surf = TransformPoint(a, m, h, w);
ScorePoint(surf, h, w, settings.min_pair_fraction, global_variance);
// The greedy search, peak by peak, where the surface is. Each round reduces it to one candidate
// per block and the host picks between those by the same rule, so the answer is the host's and
// does not depend on how the reduction was split.
const int64_t surface_h = 2 * h, surface_w = 2 * w;
const int64_t n = surface_h * surface_w;
const int64_t d = std::lround(2.0f * settings.nms_radius_pxl);
CudaDevicePtr<float> block_value(SHORTLIST_BLOCKS, ALLOC);
CudaDevicePtr<int64_t> block_index(SHORTLIST_BLOCKS, ALLOC);
std::vector<float> values(SHORTLIST_BLOCKS);
std::vector<int64_t> indices(SHORTLIST_BLOCKS);
std::vector<BeamCenterFFTCandidate> out;
for (int k = 0; k < settings.candidates_point; k++) {
BlockBest<<<SHORTLIST_BLOCKS, BLOCK>>>(surf.C.get(), n, block_value.get(),
block_index.get());
CheckLastKernel("BeamCenterFFT: peak search failed");
Check(cudaMemcpy(values.data(), block_value.get(), values.size() * sizeof(float),
cudaMemcpyDeviceToHost),
"BeamCenterFFT: peak download failed");
Check(cudaMemcpy(indices.data(), block_index.get(), indices.size() * sizeof(int64_t),
cudaMemcpyDeviceToHost),
"BeamCenterFFT: peak download failed");
int64_t best = -1;
float best_v = -std::numeric_limits<float>::infinity();
for (int b = 0; b < SHORTLIST_BLOCKS; b++)
if (indices[b] >= 0
&& (values[b] > best_v || (values[b] == best_v && indices[b] < best)))
best_v = values[b], best = indices[b];
if (best < 0 || !std::isfinite(best_v))
break;
const int64_t iy = best / surface_w, ix = best % surface_w;
out.push_back({static_cast<float>(ix) / 2.0f, static_cast<float>(iy) / 2.0f, best_v});
const int64_t y0 = std::max<int64_t>(0, iy - d), y1 = std::min(surface_h - 1, iy + d);
const int64_t x0 = std::max<int64_t>(0, ix - d), x1 = std::min(surface_w - 1, ix + d);
Suppress<<<Blocks((y1 - y0 + 1) * (x1 - x0 + 1)), BLOCK>>>(surf.C.get(), surface_w, x0, x1,
y0, y1);
CheckLastKernel("BeamCenterFFT: peak suppression failed");
}
return out;
}
BeamCenterConvSurfaces1D BeamCenterFFTGPU::LineSurfaces(const std::vector<float> &a,
const std::vector<float> &m, int64_t h,
int64_t w, BeamCenterMirror mirror) {
const bool rows = mirror == BeamCenterMirror::Rows;
const int64_t nseq = rows ? h : w;
const int64_t nbatch = rows ? w : h;
const int64_t stride_seq = rows ? w : 1; // step between two samples of one sequence
const int64_t stride_batch = rows ? 1 : w; // step between two sequences
const int64_t nfft = BeamCenterFFTPadSize(2 * nseq);
const int64_t nc = nfft / 2 + 1;
BeamCenterConvSurfaces1D out;
CudaDevicePtr<double2> accC(nc, ALLOC), accS(nc, ALLOC), accQ(nc, ALLOC), accD(nc, ALLOC);
{
CudaDevicePtr<float> d_img(h * w, ALLOC), d_in(nfft * nbatch, ALLOC);
CudaDevicePtr<cufftComplex> d_a(nbatch * nc, ALLOC), d_m(nbatch * nc, ALLOC), d_a2(nbatch * nc, ALLOC);
CudaFFTPlan fwd(1, Dims(nfft), Dims(nfft), 1, static_cast<int>(nfft), Dims(nc), 1,
static_cast<int>(nc), CUFFT_R2C, static_cast<int>(nbatch));
auto transform = [&](int square, cufftComplex *dst) {
GatherSequences<<<Blocks(nfft * nbatch), BLOCK>>>(d_img.get(), nseq, nbatch, stride_seq,
stride_batch, nfft, square,
d_in.get());
CheckLastKernel("BeamCenterFFT: sequence gather failed");
Check(cufftExecR2C(fwd, d_in, dst), "BeamCenterFFT: forward transform failed");
};
auto upload = [&](const std::vector<float> &src) {
Check(cudaMemcpy(d_img.get(), src.data(), h * w * sizeof(float),
cudaMemcpyHostToDevice),
"BeamCenterFFT: image upload failed");
};
upload(a);
transform(0, d_a);
transform(1, d_a2);
upload(m);
transform(0, d_m);
AccumulateSpectra<<<static_cast<int>(nc), ACCUMULATE_THREADS>>>(d_a, d_m, d_a2, nbatch, nc,
accC, accS, accQ, accD);
CheckLastKernel("BeamCenterFFT: spectrum accumulation failed");
}
CudaDevicePtr<cufftComplex> d_pack(4 * nc, ALLOC);
CudaDevicePtr<float> d_out(4 * nfft, ALLOC);
PackAccumulators<<<Blocks(nc), BLOCK>>>(accC, accS, accQ, accD, nc, d_pack.get());
CheckLastKernel("BeamCenterFFT: accumulator packing failed");
CudaFFTPlan inv(1, Dims(nfft), Dims(nc), 1, static_cast<int>(nc), Dims(nfft), 1,
static_cast<int>(nfft), CUFFT_C2R, 4);
Check(cufftExecC2R(inv, d_pack, d_out), "BeamCenterFFT: inverse transform failed");
std::vector<float> host(static_cast<size_t>(4 * nfft));
Check(cudaMemcpy(host.data(), d_out.get(), host.size() * sizeof(float),
cudaMemcpyDeviceToHost),
"BeamCenterFFT: line surface download failed");
const double norm = 1.0 / static_cast<double>(nfft);
auto take = [&](int k) {
std::vector<double> v(static_cast<size_t>(2 * nseq));
for (int64_t i = 0; i < 2 * nseq; i++)
v[static_cast<size_t>(i)] = host[static_cast<size_t>(k * nfft + i)] * norm;
return v;
};
out.C = take(0);
out.S = take(1);
out.Q = take(2);
out.D = take(3);
return out;
}
size_t BeamCenterFFTGPU::DeviceMemoryNeeded(int64_t width, int64_t height) {
const int64_t ny = BeamCenterFFTPadSize(2 * height);
const int64_t nx = BeamCenterFFTPadSize(2 * width);
const int64_t nxc = nx / 2 + 1;
size_t work_fwd = 0, work_inv = 0;
cufftEstimate2d(static_cast<int>(ny), static_cast<int>(nx), CUFFT_R2C, &work_fwd);
cufftEstimate2d(static_cast<int>(ny), static_cast<int>(nx), CUFFT_C2R, &work_inv);
// One transform buffer, three spectra, and three of the four convolution surfaces - the
// fourth stays in the transform buffer, where the last inverse leaves it.
size_t needed = ny * nx * sizeof(float) + 3 * ny * nxc * sizeof(cufftComplex)
+ 3 * 4 * static_cast<size_t>(width) * height * sizeof(float) + work_fwd
+ work_inv;
// The two 1D phases, the larger of them: they run after the point surfaces, with its buffers
// already freed.
for (int pass = 0; pass < 2; pass++) {
const int64_t nseq = pass == 0 ? height : width;
const int64_t nbatch = pass == 0 ? width : height;
const int64_t nfft = BeamCenterFFTPadSize(2 * nseq);
const int64_t nc = nfft / 2 + 1;
std::vector<int> n = Dims(nfft), embed_real = Dims(nfft), embed_complex = Dims(nc);
size_t work = 0;
cufftEstimateMany(1, n.data(), embed_real.data(), 1, static_cast<int>(nfft),
embed_complex.data(), 1, static_cast<int>(nc), CUFFT_R2C,
static_cast<int>(nbatch), &work);
const size_t line = (height * width + nfft * nbatch) * sizeof(float)
+ 3 * nbatch * nc * sizeof(cufftComplex) + work;
needed = std::max(needed, line);
}
return needed;
}
bool BeamCenterFFTGPU::FitsInDeviceMemory(int64_t width, int64_t height) {
size_t free_bytes = 0, total_bytes = 0;
if (cudaMemGetInfo(&free_bytes, &total_bytes) != cudaSuccess)
return false;
// Headroom for the allocator's own rounding, and for whatever else lands on a shared card
// between this question and the allocations it is asked about.
constexpr size_t HEADROOM = 512ull << 20;
return DeviceMemoryNeeded(width, height) + HEADROOM <= free_bytes;
}