Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports. * jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls. * Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results. * Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable. * Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate. * Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do. * Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence. * Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags. * Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check. * Rugnux: Clear error messages when a data set needs more GPU or host memory than is available. Reviewed-on: #83 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
526 lines
25 KiB
Plaintext
526 lines
25 KiB
Plaintext
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "BeamCenterFFTGPU.h"
|
|
|
|
#include <algorithm>
|
|
#include <cmath>
|
|
#include <limits>
|
|
|
|
#include <cuda_runtime.h>
|
|
#include <cufft.h>
|
|
|
|
#include "../indexing/CUDAMemHelpers.h"
|
|
|
|
namespace {
|
|
|
|
constexpr int BLOCK = 256;
|
|
// Threads per frequency in the 1D accumulation. A power of two, for the shared-memory reduction.
|
|
constexpr int ACCUMULATE_THREADS = 128;
|
|
// Blocks in the greedy search's reduction. Each returns one candidate, and the last step of the
|
|
// reduction is done on the host over that many - a thousand is far below what a download costs and
|
|
// far above what the card needs to stay busy.
|
|
constexpr int SHORTLIST_BLOCKS = 1024;
|
|
|
|
// Every buffer here is read by work on the legacy NULL stream - the kernels, the cuFFT plans and the
|
|
// copies below are all queued there - so none of them may come from the pool. A pooled buffer is
|
|
// freed with cudaFreeAsync on the thread's allocation stream, which is non-blocking and so not
|
|
// ordered after the NULL stream: the free completed at once while a Suppress or an inverse transform
|
|
// was still writing, and the pool handed those bytes to the analysis engines being built on other
|
|
// workers at the same moment, or unmapped them. cudaFree synchronises the device first.
|
|
constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous;
|
|
|
|
int Blocks(int64_t n) { return static_cast<int>((n + BLOCK - 1) / BLOCK); }
|
|
|
|
void Check(cudaError_t err, const char *what) {
|
|
if (err != cudaSuccess)
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, what);
|
|
}
|
|
|
|
void Check(cufftResult res, const char *what) {
|
|
if (res != CUFFT_SUCCESS)
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, what);
|
|
}
|
|
|
|
std::vector<int> Dims(int64_t a) { return {static_cast<int>(a)}; }
|
|
std::vector<int> Dims(int64_t a, int64_t b) { return {static_cast<int>(a), static_cast<int>(b)}; }
|
|
|
|
__global__ void SquareInPlace(float *data, int64_t n) {
|
|
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
|
|
if (i < n)
|
|
data[i] *= data[i];
|
|
}
|
|
|
|
__global__ void MultiplySpectra(const cufftComplex *u, const cufftComplex *v, cufftComplex *out,
|
|
int64_t n) {
|
|
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
|
|
if (i < n)
|
|
out[i] = cuCmulf(u[i], v[i]);
|
|
}
|
|
|
|
// The inverse transform leaves an unnormalised ny x nx result; only its 2h x 2w corner is the
|
|
// convolution surface. Scale that corner in place, for the one surface that stays where it is...
|
|
__global__ void ScaleCorner(float *data, int64_t nx, int64_t out_h, int64_t out_w, float norm) {
|
|
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= out_h * out_w)
|
|
return;
|
|
const int64_t y = i / out_w;
|
|
data[y * nx + (i - y * out_w)] *= norm;
|
|
}
|
|
|
|
// ...and scale it into a surface of its own for the three that have to outlive the next inverse.
|
|
__global__ void CopyScaledCorner(const float *src, int64_t nx, int64_t out_h, int64_t out_w,
|
|
float norm, float *dst) {
|
|
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= out_h * out_w)
|
|
return;
|
|
const int64_t y = i / out_w;
|
|
dst[i] = src[y * nx + (i - y * out_w)] * norm;
|
|
}
|
|
|
|
// Lay each sequence out contiguously, zero-padded to nfft, so the 1D transforms are a plain batch:
|
|
// out[b * nfft + s] = img[b * stride_batch + s * stride_seq], squared where asked.
|
|
__global__ void GatherSequences(const float *img, int64_t nseq, int64_t nbatch, int64_t stride_seq,
|
|
int64_t stride_batch, int64_t nfft, int square, float *out) {
|
|
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= nbatch * nfft)
|
|
return;
|
|
const int64_t b = i / nfft;
|
|
const int64_t s = i - b * nfft;
|
|
float v = 0.0f;
|
|
if (s < nseq) {
|
|
v = img[b * stride_batch + s * stride_seq];
|
|
if (square)
|
|
v *= v;
|
|
}
|
|
out[i] = v;
|
|
}
|
|
|
|
// The four accumulators of the 1D score, summed over the batch in double - the same widening the
|
|
// host path does, and for the same reason: the numerator and denominator below are cancellations
|
|
// of large near-equal terms. One block per frequency.
|
|
__global__ void AccumulateSpectra(const cufftComplex *A, const cufftComplex *M,
|
|
const cufftComplex *A2, int64_t nbatch, int64_t nc, double2 *accC,
|
|
double2 *accS, double2 *accQ, double2 *accD) {
|
|
__shared__ double2 red[4][ACCUMULATE_THREADS];
|
|
const int64_t i = blockIdx.x;
|
|
double2 c = {0.0, 0.0}, s = {0.0, 0.0}, q = {0.0, 0.0}, d = {0.0, 0.0};
|
|
for (int64_t b = threadIdx.x; b < nbatch; b += blockDim.x) {
|
|
const cufftComplex fa = A[b * nc + i], fm = M[b * nc + i], fa2 = A2[b * nc + i];
|
|
const double ar = fa.x, ai = fa.y, mr = fm.x, mi = fm.y, qr = fa2.x, qi = fa2.y;
|
|
c.x += ar * ar - ai * ai;
|
|
c.y += ar * ai + ai * ar;
|
|
s.x += ar * mr - ai * mi;
|
|
s.y += ar * mi + ai * mr;
|
|
q.x += qr * mr - qi * mi;
|
|
q.y += qr * mi + qi * mr;
|
|
d.x += mr * mr - mi * mi;
|
|
d.y += mr * mi + mi * mr;
|
|
}
|
|
red[0][threadIdx.x] = c;
|
|
red[1][threadIdx.x] = s;
|
|
red[2][threadIdx.x] = q;
|
|
red[3][threadIdx.x] = d;
|
|
for (unsigned stride = blockDim.x / 2; stride > 0; stride /= 2) {
|
|
__syncthreads();
|
|
if (threadIdx.x < stride)
|
|
for (int k = 0; k < 4; k++) {
|
|
red[k][threadIdx.x].x += red[k][threadIdx.x + stride].x;
|
|
red[k][threadIdx.x].y += red[k][threadIdx.x + stride].y;
|
|
}
|
|
}
|
|
if (threadIdx.x == 0) {
|
|
accC[i] = red[0][0];
|
|
accS[i] = red[1][0];
|
|
accQ[i] = red[2][0];
|
|
accD[i] = red[3][0];
|
|
}
|
|
}
|
|
|
|
// The four accumulators narrowed back to single precision and packed as one batch of four, ready
|
|
// for the inverse transform - exactly what the host path does before its four inverses.
|
|
__global__ void PackAccumulators(const double2 *accC, const double2 *accS, const double2 *accQ,
|
|
const double2 *accD, int64_t nc, cufftComplex *out) {
|
|
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= nc)
|
|
return;
|
|
const double2 *acc[4] = {accC, accS, accQ, accD};
|
|
for (int k = 0; k < 4; k++)
|
|
out[k * nc + i] = make_cuFloatComplex(static_cast<float>(acc[k][i].x),
|
|
static_cast<float>(acc[k][i].y));
|
|
}
|
|
|
|
// The masked Pearson, elementwise in double, exactly as BeamCenterPointScore does it on the host -
|
|
// the numerator and the denominator are cancellations of large near-equal terms. Q is read out of
|
|
// the transform buffer where the last inverse left it, so it needs no surface of its own. The score
|
|
// replaces C, which nothing wants afterwards.
|
|
__global__ void PearsonKernel(float *C, const float *S, const float *D, const float *Q_padded,
|
|
int64_t nx, int64_t out_h, int64_t out_w, double lim,
|
|
double variance_limit) {
|
|
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= out_h * out_w)
|
|
return;
|
|
const double d = D[i];
|
|
if (d <= lim) {
|
|
C[i] = -INFINITY;
|
|
return;
|
|
}
|
|
const int64_t y = i / out_w;
|
|
const double q = Q_padded[y * nx + (i - y * out_w)];
|
|
const double s = S[i];
|
|
const double num = d * C[i] - s * s;
|
|
const double den = d * q - s * s;
|
|
C[i] = den > d * d * variance_limit ? static_cast<float>(num / den) : -INFINITY;
|
|
}
|
|
|
|
// The host's rule for the greedy search, so the device finds the same peak: a strictly greater
|
|
// value wins, among equal values the lower index does, and an index of -1 means "nothing found"
|
|
// (which is what a surface of -infinity leaves, and what ends the shortlist).
|
|
__device__ bool TakeRight(float lv, int64_t li, float rv, int64_t ri) {
|
|
if (ri < 0)
|
|
return false;
|
|
if (li < 0)
|
|
return true;
|
|
return rv > lv || (rv == lv && ri < li);
|
|
}
|
|
|
|
__global__ void BlockBest(const float *r, int64_t n, float *block_value, int64_t *block_index) {
|
|
__shared__ float sv[BLOCK];
|
|
__shared__ int64_t si[BLOCK];
|
|
float bv = -INFINITY;
|
|
int64_t bi = -1;
|
|
for (int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x; i < n;
|
|
i += static_cast<int64_t>(gridDim.x) * blockDim.x)
|
|
if (r[i] > bv) { // increasing i, so the lowest index among equal values is the one kept
|
|
bv = r[i];
|
|
bi = i;
|
|
}
|
|
sv[threadIdx.x] = bv;
|
|
si[threadIdx.x] = bi;
|
|
for (unsigned stride = blockDim.x / 2; stride > 0; stride /= 2) {
|
|
__syncthreads();
|
|
if (threadIdx.x < stride
|
|
&& TakeRight(sv[threadIdx.x], si[threadIdx.x], sv[threadIdx.x + stride],
|
|
si[threadIdx.x + stride])) {
|
|
sv[threadIdx.x] = sv[threadIdx.x + stride];
|
|
si[threadIdx.x] = si[threadIdx.x + stride];
|
|
}
|
|
}
|
|
if (threadIdx.x == 0) {
|
|
block_value[blockIdx.x] = sv[0];
|
|
block_index[blockIdx.x] = si[0];
|
|
}
|
|
}
|
|
|
|
// The non-maximum suppression square around a peak that has been taken.
|
|
__global__ void Suppress(float *r, int64_t w, int64_t x0, int64_t x1, int64_t y0, int64_t y1) {
|
|
const int64_t cols = x1 - x0 + 1;
|
|
const int64_t i = blockIdx.x * static_cast<int64_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= (y1 - y0 + 1) * cols)
|
|
return;
|
|
const int64_t y = i / cols;
|
|
r[(y0 + y) * w + x0 + (i - y * cols)] = -INFINITY;
|
|
}
|
|
|
|
void CheckLastKernel(const char *what) {
|
|
Check(cudaGetLastError(), what);
|
|
}
|
|
|
|
// The 2D transform set, left on the device. Only THREE 2h x 2w surfaces are ever allocated: the
|
|
// fourth is the last inverse's output, and nothing overwrites the transform buffer afterwards, so
|
|
// Q is read where it lies.
|
|
struct PointSurfacesDevice {
|
|
CudaDevicePtr<float> C, S, D; // 2h x 2w, normalised
|
|
CudaDevicePtr<float> real; // ny x nx; its 2h x 2w corner is Q, normalised
|
|
int64_t nx = 0;
|
|
};
|
|
|
|
PointSurfacesDevice TransformPoint(const std::vector<float> &a, const std::vector<float> &m,
|
|
int64_t h, int64_t w) {
|
|
const int64_t ny = BeamCenterFFTPadSize(2 * h);
|
|
const int64_t nx = BeamCenterFFTPadSize(2 * w);
|
|
const int64_t nxc = nx / 2 + 1;
|
|
const int64_t nreal = ny * nx;
|
|
const int64_t ncomplex = ny * nxc;
|
|
const int64_t ncrop = 4 * h * w;
|
|
|
|
PointSurfacesDevice out;
|
|
out.nx = nx;
|
|
out.real = CudaDevicePtr<float>(nreal, ALLOC);
|
|
CudaDevicePtr<cufftComplex> d_a(ncomplex, ALLOC), d_m(ncomplex, ALLOC), d_prod(ncomplex, ALLOC);
|
|
|
|
CudaFFTPlan fwd(2, Dims(ny, nx), Dims(ny, nx), 1, static_cast<int>(nreal), Dims(ny, nxc), 1,
|
|
static_cast<int>(ncomplex), CUFFT_R2C, 1);
|
|
CudaFFTPlan inv(2, Dims(ny, nx), Dims(ny, nxc), 1, static_cast<int>(ncomplex), Dims(ny, nx), 1,
|
|
static_cast<int>(nreal), CUFFT_C2R, 1);
|
|
|
|
auto upload = [&](const std::vector<float> &src) {
|
|
Check(cudaMemset(out.real.get(), 0, nreal * sizeof(float)),
|
|
"BeamCenterFFT: memset failed");
|
|
Check(cudaMemcpy2D(out.real.get(), nx * sizeof(float), src.data(), w * sizeof(float),
|
|
w * sizeof(float), h, cudaMemcpyHostToDevice),
|
|
"BeamCenterFFT: image upload failed");
|
|
};
|
|
|
|
upload(a);
|
|
Check(cufftExecR2C(fwd, out.real, d_a), "BeamCenterFFT: forward transform failed");
|
|
upload(m);
|
|
Check(cufftExecR2C(fwd, out.real, d_m), "BeamCenterFFT: forward transform failed");
|
|
|
|
const float norm = 1.0f / (static_cast<float>(ny) * static_cast<float>(nx));
|
|
// The inverse leaves an unnormalised ny x nx result whose 2h x 2w corner is the surface.
|
|
auto inverse_product = [&](const cufftComplex *u, const cufftComplex *v) {
|
|
MultiplySpectra<<<Blocks(ncomplex), BLOCK>>>(u, v, d_prod.get(), ncomplex);
|
|
CheckLastKernel("BeamCenterFFT: spectrum product failed");
|
|
Check(cufftExecC2R(inv, d_prod, out.real), "BeamCenterFFT: inverse transform failed");
|
|
};
|
|
auto take_corner = [&](CudaDevicePtr<float> &dst) {
|
|
dst = CudaDevicePtr<float>(ncrop, ALLOC);
|
|
CopyScaledCorner<<<Blocks(ncrop), BLOCK>>>(out.real.get(), nx, 2 * h, 2 * w, norm,
|
|
dst.get());
|
|
CheckLastKernel("BeamCenterFFT: surface scaling failed");
|
|
};
|
|
|
|
inverse_product(d_a, d_a);
|
|
take_corner(out.C);
|
|
inverse_product(d_a, d_m);
|
|
take_corner(out.S);
|
|
inverse_product(d_m, d_m);
|
|
take_corner(out.D);
|
|
// The spectrum of a is finished with, so a^2 goes in its place rather than in a fourth buffer.
|
|
upload(a);
|
|
SquareInPlace<<<Blocks(nreal), BLOCK>>>(out.real.get(), nreal);
|
|
CheckLastKernel("BeamCenterFFT: squaring failed");
|
|
Check(cufftExecR2C(fwd, out.real, d_a), "BeamCenterFFT: forward transform failed");
|
|
inverse_product(d_a, d_m);
|
|
ScaleCorner<<<Blocks(ncrop), BLOCK>>>(out.real.get(), nx, 2 * h, 2 * w, norm);
|
|
CheckLastKernel("BeamCenterFFT: surface scaling failed");
|
|
return out;
|
|
}
|
|
|
|
// The masked Pearson over the four surfaces, leaving the score in `surf.C`.
|
|
void ScorePoint(PointSurfacesDevice &surf, int64_t h, int64_t w, float min_pair_fraction,
|
|
double global_variance) {
|
|
const int64_t ncrop = 4 * h * w;
|
|
CudaDevicePtr<float> block_value(SHORTLIST_BLOCKS, ALLOC);
|
|
CudaDevicePtr<int64_t> block_index(SHORTLIST_BLOCKS, ALLOC);
|
|
BlockBest<<<SHORTLIST_BLOCKS, BLOCK>>>(surf.D.get(), ncrop, block_value.get(),
|
|
block_index.get());
|
|
CheckLastKernel("BeamCenterFFT: pair-count maximum failed");
|
|
std::vector<float> values(SHORTLIST_BLOCKS);
|
|
Check(cudaMemcpy(values.data(), block_value.get(), values.size() * sizeof(float),
|
|
cudaMemcpyDeviceToHost),
|
|
"BeamCenterFFT: pair-count maximum download failed");
|
|
float d_max = 0.0f; // as on the host, where the running maximum starts at zero
|
|
for (float v : values)
|
|
d_max = std::max(d_max, v);
|
|
|
|
PearsonKernel<<<Blocks(ncrop), BLOCK>>>(surf.C.get(), surf.S.get(), surf.D.get(),
|
|
surf.real.get(), surf.nx, 2 * h, 2 * w,
|
|
min_pair_fraction * static_cast<double>(d_max),
|
|
BEAM_CENTER_VARIANCE_FLOOR * global_variance);
|
|
CheckLastKernel("BeamCenterFFT: masked Pearson failed");
|
|
}
|
|
|
|
} // namespace
|
|
|
|
BeamCenterConvSurfaces2D BeamCenterFFTGPU::PointSurfaces(const std::vector<float> &a,
|
|
const std::vector<float> &m, int64_t h,
|
|
int64_t w) {
|
|
// The four surfaces brought back as the interface describes them. Nothing in the pipeline asks
|
|
// for this any more - PointShortlist below keeps them on the device - but the shared host code
|
|
// is written on them and the parity test compares them, so the engine still supplies them.
|
|
PointSurfacesDevice surf = TransformPoint(a, m, h, w);
|
|
const size_t ncrop = static_cast<size_t>(4 * h * w);
|
|
|
|
BeamCenterConvSurfaces2D out;
|
|
auto download = [&](const CudaDevicePtr<float> &src, std::vector<float> &dst) {
|
|
dst.resize(ncrop);
|
|
Check(cudaMemcpy(dst.data(), src.get(), ncrop * sizeof(float), cudaMemcpyDeviceToHost),
|
|
"BeamCenterFFT: surface download failed");
|
|
};
|
|
download(surf.C, out.C);
|
|
download(surf.S, out.S);
|
|
download(surf.D, out.D);
|
|
out.Q.resize(ncrop);
|
|
Check(cudaMemcpy2D(out.Q.data(), 2 * w * sizeof(float), surf.real.get(),
|
|
surf.nx * sizeof(float), 2 * w * sizeof(float), 2 * h,
|
|
cudaMemcpyDeviceToHost),
|
|
"BeamCenterFFT: surface download failed");
|
|
return out;
|
|
}
|
|
|
|
std::vector<float> BeamCenterFFTGPU::PointScoreSurface(const std::vector<float> &a,
|
|
const std::vector<float> &m, int64_t h,
|
|
int64_t w, float min_pair_fraction,
|
|
double global_variance) {
|
|
PointSurfacesDevice surf = TransformPoint(a, m, h, w);
|
|
ScorePoint(surf, h, w, min_pair_fraction, global_variance);
|
|
std::vector<float> out(static_cast<size_t>(4 * h * w));
|
|
Check(cudaMemcpy(out.data(), surf.C.get(), out.size() * sizeof(float), cudaMemcpyDeviceToHost),
|
|
"BeamCenterFFT: score download failed");
|
|
return out;
|
|
}
|
|
|
|
std::vector<BeamCenterFFTCandidate>
|
|
BeamCenterFFTGPU::PointShortlist(const std::vector<float> &a, const std::vector<float> &m,
|
|
int64_t h, int64_t w, const BeamCenterFFTSettings &settings,
|
|
double global_variance) {
|
|
PointSurfacesDevice surf = TransformPoint(a, m, h, w);
|
|
ScorePoint(surf, h, w, settings.min_pair_fraction, global_variance);
|
|
|
|
// The greedy search, peak by peak, where the surface is. Each round reduces it to one candidate
|
|
// per block and the host picks between those by the same rule, so the answer is the host's and
|
|
// does not depend on how the reduction was split.
|
|
const int64_t surface_h = 2 * h, surface_w = 2 * w;
|
|
const int64_t n = surface_h * surface_w;
|
|
const int64_t d = std::lround(2.0f * settings.nms_radius_pxl);
|
|
CudaDevicePtr<float> block_value(SHORTLIST_BLOCKS, ALLOC);
|
|
CudaDevicePtr<int64_t> block_index(SHORTLIST_BLOCKS, ALLOC);
|
|
std::vector<float> values(SHORTLIST_BLOCKS);
|
|
std::vector<int64_t> indices(SHORTLIST_BLOCKS);
|
|
|
|
std::vector<BeamCenterFFTCandidate> out;
|
|
for (int k = 0; k < settings.candidates_point; k++) {
|
|
BlockBest<<<SHORTLIST_BLOCKS, BLOCK>>>(surf.C.get(), n, block_value.get(),
|
|
block_index.get());
|
|
CheckLastKernel("BeamCenterFFT: peak search failed");
|
|
Check(cudaMemcpy(values.data(), block_value.get(), values.size() * sizeof(float),
|
|
cudaMemcpyDeviceToHost),
|
|
"BeamCenterFFT: peak download failed");
|
|
Check(cudaMemcpy(indices.data(), block_index.get(), indices.size() * sizeof(int64_t),
|
|
cudaMemcpyDeviceToHost),
|
|
"BeamCenterFFT: peak download failed");
|
|
|
|
int64_t best = -1;
|
|
float best_v = -std::numeric_limits<float>::infinity();
|
|
for (int b = 0; b < SHORTLIST_BLOCKS; b++)
|
|
if (indices[b] >= 0
|
|
&& (values[b] > best_v || (values[b] == best_v && indices[b] < best)))
|
|
best_v = values[b], best = indices[b];
|
|
if (best < 0 || !std::isfinite(best_v))
|
|
break;
|
|
|
|
const int64_t iy = best / surface_w, ix = best % surface_w;
|
|
out.push_back({static_cast<float>(ix) / 2.0f, static_cast<float>(iy) / 2.0f, best_v});
|
|
const int64_t y0 = std::max<int64_t>(0, iy - d), y1 = std::min(surface_h - 1, iy + d);
|
|
const int64_t x0 = std::max<int64_t>(0, ix - d), x1 = std::min(surface_w - 1, ix + d);
|
|
Suppress<<<Blocks((y1 - y0 + 1) * (x1 - x0 + 1)), BLOCK>>>(surf.C.get(), surface_w, x0, x1,
|
|
y0, y1);
|
|
CheckLastKernel("BeamCenterFFT: peak suppression failed");
|
|
}
|
|
return out;
|
|
}
|
|
|
|
BeamCenterConvSurfaces1D BeamCenterFFTGPU::LineSurfaces(const std::vector<float> &a,
|
|
const std::vector<float> &m, int64_t h,
|
|
int64_t w, BeamCenterMirror mirror) {
|
|
const bool rows = mirror == BeamCenterMirror::Rows;
|
|
const int64_t nseq = rows ? h : w;
|
|
const int64_t nbatch = rows ? w : h;
|
|
const int64_t stride_seq = rows ? w : 1; // step between two samples of one sequence
|
|
const int64_t stride_batch = rows ? 1 : w; // step between two sequences
|
|
const int64_t nfft = BeamCenterFFTPadSize(2 * nseq);
|
|
const int64_t nc = nfft / 2 + 1;
|
|
|
|
BeamCenterConvSurfaces1D out;
|
|
CudaDevicePtr<double2> accC(nc, ALLOC), accS(nc, ALLOC), accQ(nc, ALLOC), accD(nc, ALLOC);
|
|
{
|
|
CudaDevicePtr<float> d_img(h * w, ALLOC), d_in(nfft * nbatch, ALLOC);
|
|
CudaDevicePtr<cufftComplex> d_a(nbatch * nc, ALLOC), d_m(nbatch * nc, ALLOC), d_a2(nbatch * nc, ALLOC);
|
|
CudaFFTPlan fwd(1, Dims(nfft), Dims(nfft), 1, static_cast<int>(nfft), Dims(nc), 1,
|
|
static_cast<int>(nc), CUFFT_R2C, static_cast<int>(nbatch));
|
|
|
|
auto transform = [&](int square, cufftComplex *dst) {
|
|
GatherSequences<<<Blocks(nfft * nbatch), BLOCK>>>(d_img.get(), nseq, nbatch, stride_seq,
|
|
stride_batch, nfft, square,
|
|
d_in.get());
|
|
CheckLastKernel("BeamCenterFFT: sequence gather failed");
|
|
Check(cufftExecR2C(fwd, d_in, dst), "BeamCenterFFT: forward transform failed");
|
|
};
|
|
auto upload = [&](const std::vector<float> &src) {
|
|
Check(cudaMemcpy(d_img.get(), src.data(), h * w * sizeof(float),
|
|
cudaMemcpyHostToDevice),
|
|
"BeamCenterFFT: image upload failed");
|
|
};
|
|
|
|
upload(a);
|
|
transform(0, d_a);
|
|
transform(1, d_a2);
|
|
upload(m);
|
|
transform(0, d_m);
|
|
|
|
AccumulateSpectra<<<static_cast<int>(nc), ACCUMULATE_THREADS>>>(d_a, d_m, d_a2, nbatch, nc,
|
|
accC, accS, accQ, accD);
|
|
CheckLastKernel("BeamCenterFFT: spectrum accumulation failed");
|
|
}
|
|
|
|
CudaDevicePtr<cufftComplex> d_pack(4 * nc, ALLOC);
|
|
CudaDevicePtr<float> d_out(4 * nfft, ALLOC);
|
|
PackAccumulators<<<Blocks(nc), BLOCK>>>(accC, accS, accQ, accD, nc, d_pack.get());
|
|
CheckLastKernel("BeamCenterFFT: accumulator packing failed");
|
|
CudaFFTPlan inv(1, Dims(nfft), Dims(nc), 1, static_cast<int>(nc), Dims(nfft), 1,
|
|
static_cast<int>(nfft), CUFFT_C2R, 4);
|
|
Check(cufftExecC2R(inv, d_pack, d_out), "BeamCenterFFT: inverse transform failed");
|
|
|
|
std::vector<float> host(static_cast<size_t>(4 * nfft));
|
|
Check(cudaMemcpy(host.data(), d_out.get(), host.size() * sizeof(float),
|
|
cudaMemcpyDeviceToHost),
|
|
"BeamCenterFFT: line surface download failed");
|
|
|
|
const double norm = 1.0 / static_cast<double>(nfft);
|
|
auto take = [&](int k) {
|
|
std::vector<double> v(static_cast<size_t>(2 * nseq));
|
|
for (int64_t i = 0; i < 2 * nseq; i++)
|
|
v[static_cast<size_t>(i)] = host[static_cast<size_t>(k * nfft + i)] * norm;
|
|
return v;
|
|
};
|
|
out.C = take(0);
|
|
out.S = take(1);
|
|
out.Q = take(2);
|
|
out.D = take(3);
|
|
return out;
|
|
}
|
|
|
|
size_t BeamCenterFFTGPU::DeviceMemoryNeeded(int64_t width, int64_t height) {
|
|
const int64_t ny = BeamCenterFFTPadSize(2 * height);
|
|
const int64_t nx = BeamCenterFFTPadSize(2 * width);
|
|
const int64_t nxc = nx / 2 + 1;
|
|
size_t work_fwd = 0, work_inv = 0;
|
|
cufftEstimate2d(static_cast<int>(ny), static_cast<int>(nx), CUFFT_R2C, &work_fwd);
|
|
cufftEstimate2d(static_cast<int>(ny), static_cast<int>(nx), CUFFT_C2R, &work_inv);
|
|
// One transform buffer, three spectra, and three of the four convolution surfaces - the
|
|
// fourth stays in the transform buffer, where the last inverse leaves it.
|
|
size_t needed = ny * nx * sizeof(float) + 3 * ny * nxc * sizeof(cufftComplex)
|
|
+ 3 * 4 * static_cast<size_t>(width) * height * sizeof(float) + work_fwd
|
|
+ work_inv;
|
|
|
|
// The two 1D phases, the larger of them: they run after the point surfaces, with its buffers
|
|
// already freed.
|
|
for (int pass = 0; pass < 2; pass++) {
|
|
const int64_t nseq = pass == 0 ? height : width;
|
|
const int64_t nbatch = pass == 0 ? width : height;
|
|
const int64_t nfft = BeamCenterFFTPadSize(2 * nseq);
|
|
const int64_t nc = nfft / 2 + 1;
|
|
std::vector<int> n = Dims(nfft), embed_real = Dims(nfft), embed_complex = Dims(nc);
|
|
size_t work = 0;
|
|
cufftEstimateMany(1, n.data(), embed_real.data(), 1, static_cast<int>(nfft),
|
|
embed_complex.data(), 1, static_cast<int>(nc), CUFFT_R2C,
|
|
static_cast<int>(nbatch), &work);
|
|
const size_t line = (height * width + nfft * nbatch) * sizeof(float)
|
|
+ 3 * nbatch * nc * sizeof(cufftComplex) + work;
|
|
needed = std::max(needed, line);
|
|
}
|
|
return needed;
|
|
}
|
|
|
|
bool BeamCenterFFTGPU::FitsInDeviceMemory(int64_t width, int64_t height) {
|
|
size_t free_bytes = 0, total_bytes = 0;
|
|
if (cudaMemGetInfo(&free_bytes, &total_bytes) != cudaSuccess)
|
|
return false;
|
|
// Headroom for the allocator's own rounding, and for whatever else lands on a shared card
|
|
// between this question and the allocations it is asked about.
|
|
constexpr size_t HEADROOM = 512ull << 20;
|
|
return DeviceMemoryNeeded(width, height) + HEADROOM <= free_bytes;
|
|
}
|