Files
Jungfraujoch/image_analysis/bragg_prediction/BraggPredictionGPU.cu
T
leonarski_f cb5a2f032a
Build Packages / Create release (push) Successful in 23s
Build Packages / build:rugnux-tgz (x86_64) (push) Successful in 10m6s
Build Packages / build:rugnux:aarch64 (cross) (push) Successful in 9m6s
Build Packages / build:viewer-tgz:cpu (push) Successful in 11m15s
Build Packages / build:viewer-tgz:cuda (push) Successful in 12m21s
Build Packages / build:windows:nocuda (push) Successful in 17m9s
Build Packages / build:windows:cuda (push) Successful in 19m49s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m42s
Build Packages / build:rpm (ubuntu2204_nocuda) (push) Successful in 16m0s
Build Packages / build:rpm (ubuntu2404_nocuda) (push) Successful in 14m54s
Build Packages / build:rpm (rocky8_nocuda) (push) Successful in 17m7s
Build Packages / build:rugnux:windows (push) Successful in 10m47s
Build Packages / build:rpm (rocky9_nocuda) (push) Successful in 17m4s
Build Packages / build:rpm (rocky8_sls9) (push) Successful in 17m8s
Build Packages / Generate python client (push) Successful in 45s
Build Packages / Build documentation (push) Successful in 1m45s
Build Packages / build:rpm (rocky9_sls9) (push) Successful in 19m23s
Build Packages / build:rpm (ubuntu2204) (push) Successful in 19m20s
Build Packages / build:rpm (ubuntu2404) (push) Successful in 18m43s
Build Packages / build:rpm (rocky8) (push) Successful in 19m31s
Build Packages / build:rpm (rocky9) (push) Successful in 20m16s
Build Packages / Unit tests (push) Successful in 1h41m19s
v1.0.0-rc.170 (#80)
* Fixed a `jfjoch_broker` crash during indexing: sorting no longer misbehaves on non-finite values, and GPU FFT indexer kernel launches are now error-checked.
* rugnux needs about a third less peak memory to scale, merge and post-refine rotation data, with identical results.
* `rugnux --model`: the placed coordinate file carries the space group its own coordinates obey, and says so when that is not the group the reflection files beside it carry.
* `jfjoch_viewer`: fixes in the dataset plots, inspector and layout; spot markers lose their black outline by default (a checkbox under "Image features" restores it) and the highest-pixel markers are white boxes around the pixel.

Reviewed-on: #80
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-09-16 18:17:46 +02:00

280 lines
12 KiB
Plaintext
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <algorithm>
#include "../../common/JFJochMath.h"
#include "BraggPredictionGPU.h"
#include "../SensorAbsorption.h"
#ifdef JFJOCH_USE_CUDA
#include "../indexing/CUDAMemHelpers.h"
#include <cuda_runtime.h>
static inline void cuda_err(cudaError_t val) {
if (val != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
}
namespace {
// Number of bandwidth sigmas included in the (radially thickened) Ewald-shell
// acceptance window. Mirrors the CPU BraggPrediction path.
constexpr float kBandwidthCutoffSigmas = 3.0f;
__device__ inline bool is_odd(int v) { return (v & 1) != 0; }
__device__ inline float angle_from_ewald_sphere_deg(const Coord &S0, float recip_x, float recip_y, float recip_z, float recip_sq) {
const float epsilon = 1e-5f;
const float rad_to_deg = 180.0f / static_cast<float>(PI);
const float s0_sq = S0.x * S0.x + S0.y * S0.y + S0.z * S0.z;
const float s0_p0 = S0.x * recip_x + S0.y * recip_y + S0.z * recip_z;
const float val = s0_sq * recip_sq - s0_p0 * s0_p0;
if (fabsf(val) < epsilon || s0_sq < epsilon) return NAN;
const float a_num = (s0_sq - 0.25f * recip_sq) * recip_sq;
if (a_num < 0.0f) return NAN;
const float A = sqrtf(a_num / val);
const float B = (A * s0_p0 + 0.5f * recip_sq) / s0_sq;
const float p_star_x = A * recip_x - B * S0.x;
const float p_star_y = A * recip_y - B * S0.y;
const float p_star_z = A * recip_z - B * S0.z;
const float p_star_sq = p_star_x * p_star_x + p_star_y * p_star_y + p_star_z * p_star_z;
const float denom = sqrtf(p_star_sq * recip_sq);
if (denom < epsilon) return NAN;
float c = (p_star_x * recip_x + p_star_y * recip_y + p_star_z * recip_z) / denom;
c = fmaxf(-1.0f, fminf(1.0f, c));
return acosf(c) * rad_to_deg;
}
__device__ inline bool compute_reflection(const KernelConsts &C, int h, int k, int l, Reflection &out) {
if (h == 0 && k == 0 && l == 0)
return false;
// Systematic absences (centering only)
// P, I, A, B, C, F supported
switch (C.centering) {
case 'I':
if (is_odd(h + k + l))
return false;
break;
case 'A':
if (is_odd(k + l))
return false;
break;
case 'B':
if (is_odd(h + l))
return false;
break;
case 'C':
if (is_odd(h + k))
return false;
break;
case 'F':
if ((is_odd(h + k)) || (is_odd(h + l)) || (is_odd(k + l)))
return false;
break;
case 'R': {
// Rhombohedral in hexagonal setting (hR, a_h=b_h, gamma=120°):
// Condition: -h + k + l = 3n
int mod = (-h + k + l) % 3;
if (mod < 0) mod += 3;
if (mod != 0) return false;
break;
}
default:
break;
}
float Ah_x = C.Astar.x * h;
float Ah_y = C.Astar.y * h;
float Ah_z = C.Astar.z * h;
float AhBk_x = Ah_x + C.Bstar.x * k;
float AhBk_y = Ah_y + C.Bstar.y * k;
float AhBk_z = Ah_z + C.Bstar.z * k;
float recip_x = AhBk_x + C.Cstar.x * l;
float recip_y = AhBk_y + C.Cstar.y * l;
float recip_z = AhBk_z + C.Cstar.z * l;
float recip_sq = recip_x * recip_x + recip_y * recip_y + recip_z * recip_z;
if (recip_sq > C.one_over_dmax_sq) return false;
float Sx = recip_x + C.S0.x;
float Sy = recip_y + C.S0.y;
float Sz = recip_z + C.S0.z;
float S_len = sqrtf(Sx * Sx + Sy * Sy + Sz * Sz);
float dist_ewald = fabsf(S_len - C.one_over_wavelength);
// Energy bandwidth thickens the Ewald shell radially: σ_bw = |recip_z|·(Δλ/λ)
// (= bλ/2d²). Broaden the acceptance window in quadrature (see CPU path).
float radial_cutoff = C.ewald_cutoff;
if (C.bandwidth_sigma > 0.0f) {
const float bw_tol = kBandwidthCutoffSigmas * C.bandwidth_sigma * fabsf(recip_z);
radial_cutoff = sqrtf(radial_cutoff * radial_cutoff + bw_tol * bw_tol);
}
if (dist_ewald > radial_cutoff) return false;
float Srx = C.rot[0] * Sx + C.rot[1] * Sy + C.rot[2] * Sz;
float Sry = C.rot[3] * Sx + C.rot[4] * Sy + C.rot[5] * Sz;
float Srz = C.rot[6] * Sx + C.rot[7] * Sy + C.rot[8] * Sz;
if (Srz <= 0.0f) return false;
float coeff = C.coeff_const / Srz;
float x = C.beam_x + Srx * coeff;
float y = C.beam_y + Sry * coeff;
if (x < 0.0f || x >= C.det_width_pxl || y < 0.0f || y >= C.det_height_pxl) return false;
// Sensor quantum efficiency and the air flight path at this reflection's angle of incidence
// on the detector. Sr is the diffracted direction in the detector's own frame, so Srz over
// its length is the cosine to the detector NORMAL - which carries detector tilt for free.
// Mirrors sensor_absorption::SensorQE::Factor and ::FlightPathAttenuation::Factor on the CPU
// side; the air term runs the other way, because an oblique reflection crossed more air.
float qe_corr = 1.0f;
float flight_corr = 1.0f;
{
float cos_alpha = Srz / S_len;
if (cos_alpha > 1e-3f) {
if (C.qe_a0 > 0.0f) {
float qe = 1.0f - expf(-C.qe_a0 / cos_alpha);
if (qe > 0.0f) qe_corr = C.qe_qe0 / qe;
}
if (C.flight_dL > 0.0f)
flight_corr = expf(C.flight_dL * (1.0f / cos_alpha - 1.0f));
}
}
out.h = h;
out.k = k;
out.l = l;
out.delta_phi_deg = angle_from_ewald_sphere_deg(C.S0, recip_x, recip_y, recip_z, recip_sq);
out.predicted_x = x;
out.predicted_y = y;
out.observed_x = NAN;
out.observed_y = NAN;
out.d = 1.0f / sqrtf(recip_sq);
out.dist_ewald = dist_ewald;
out.prescaling_corr = 1.0f;
out.qe_corr = qe_corr;
out.flight_corr = flight_corr;
out.partiality = 1.0f;
out.zeta = 1.0f;
out.image_scale_corr = qe_corr * flight_corr;
return true;
}
__global__ void bragg_kernel_3d(const KernelConsts *__restrict__ kc,
int max_h, int max_k, int max_l,
int max_reflections,
Reflection *__restrict__ out,
int *__restrict__ counter) {
int hi = blockIdx.x * blockDim.x + threadIdx.x;
int ki = blockIdx.y * blockDim.y + threadIdx.y;
int li = blockIdx.z * blockDim.z + threadIdx.z;
if (hi > 2 * max_h || ki > 2 * max_k || li > 2 * max_l) return;
int h = hi - max_h;
int k = ki - max_k;
int l = li - max_l;
Reflection r{};
if (!compute_reflection(*kc, h, k, l, r)) return;
// See the rotation kernel: clamping the counter hides an overflow from the host.
const int pos = atomicAdd(counter, 1);
if (pos < max_reflections) out[pos] = r;
}
inline KernelConsts BuildKernelConsts(const DiffractionExperiment &experiment,
const CrystalLattice &lattice,
float high_res_A,
float ewald_dist_cutoff,
char centering,
float bandwidth_sigma) {
KernelConsts kc{};
auto geom = experiment.GetDiffractionGeometry();
kc.det_width_pxl = static_cast<float>(experiment.GetXPixelsNum());
kc.det_height_pxl = static_cast<float>(experiment.GetYPixelsNum());
kc.beam_x = geom.GetBeamX_pxl();
kc.beam_y = geom.GetBeamY_pxl();
kc.coeff_const = geom.GetDetectorDistance_mm() / geom.GetPixelSize_mm();
float one_over_dmax = 1.0f / high_res_A;
kc.one_over_dmax_sq = one_over_dmax * one_over_dmax;
kc.one_over_wavelength = 1.0f / geom.GetWavelength_A();
kc.ewald_cutoff = ewald_dist_cutoff;
kc.bandwidth_sigma = bandwidth_sigma;
const auto &det = experiment.GetDetectorSetup();
const auto sensor_qe = sensor_absorption::SensorQE::Build(
det.GetSensorMaterial(), det.GetSensorThickness_um(), geom.GetWavelength_A());
kc.qe_a0 = sensor_qe.active ? sensor_qe.a0 : 0.0f;
kc.qe_qe0 = sensor_qe.qe0;
kc.flight_dL = sensor_absorption::FlightPathAttenuation::Build(
experiment.GetBraggIntegrationSettings().GetFlightPath(),
geom.GetDetectorDistance_mm(), geom.GetWavelength_A()).d_over_L;
kc.Astar = lattice.Astar();
kc.Bstar = lattice.Bstar();
kc.Cstar = lattice.Cstar();
kc.S0 = geom.GetScatteringVector();
kc.centering = centering;
auto rotT = geom.GetDetectorMatrix().transpose().arr();
for (int i = 0; i < 9; ++i) kc.rot[i] = rotT[i];
return kc;
}
} // namespace
void BraggPredictionGPU::GrowCapacity(int count) {
reg_out = CudaRegisteredVector<Reflection>();
BraggPrediction::GrowCapacity(count);
reg_out = CudaRegisteredVector<Reflection>(reflections);
d_out = CudaDevicePtr<Reflection>(count);
}
BraggPredictionGPU::BraggPredictionGPU(int max_reflections)
: BraggPrediction(max_reflections),
reg_out(reflections), d_out(max_reflections),
dK(1), d_count(1), h_count(1) {
}
int BraggPredictionGPU::Calc(const DiffractionExperiment &experiment,
const CrystalLattice &lattice,
const BraggPredictionSettings &settings) {
// Build constants on host
KernelConsts hK = BuildKernelConsts(experiment, lattice, settings.high_res_A, settings.ewald_dist_cutoff,
settings.centering, settings.bandwidth_sigma);
cudaMemcpyAsync(dK, &hK, sizeof(KernelConsts), cudaMemcpyHostToDevice, stream);
cudaMemsetAsync(d_count, 0, sizeof(int), stream);
// Configure and launch on the stream
// Inclusive on both ends, matching the kernel's own bounds and the CPU loops (-max_i .. +max_i).
dim3 block(8, 8, 8);
dim3 grid((2 * settings.max_h + 1 + block.x - 1) / block.x,
(2 * settings.max_k + 1 + block.y - 1) / block.y,
(2 * settings.max_l + 1 + block.z - 1) / block.z);
bragg_kernel_3d<<<grid, block, 0, stream>>>(dK, settings.max_h, settings.max_k, settings.max_l, max_reflections, d_out, d_count);
cuda_err(cudaGetLastError());
// Async D2H count and synchronize
cudaMemcpyAsync(h_count, d_count, sizeof(int), cudaMemcpyDeviceToHost, stream);
cudaStreamSynchronize(stream);
int count = *h_count.get();
if (count > max_reflections) {
GrowCapacity(count); // see the rotation predictor
cudaMemsetAsync(d_count, 0, sizeof(int), stream);
bragg_kernel_3d<<<grid, block, 0, stream>>>(dK, settings.max_h, settings.max_k, settings.max_l, max_reflections, d_out, d_count);
cuda_err(cudaGetLastError());
cudaMemcpyAsync(h_count, d_count, sizeof(int), cudaMemcpyDeviceToHost, stream);
cudaStreamSynchronize(stream);
count = std::min(*h_count.get(), max_reflections);
}
if (count == 0)
return {};
cudaMemcpyAsync(reflections.data(), d_out, sizeof(Reflection) * count, cudaMemcpyDeviceToHost, stream);
cudaStreamSynchronize(stream);
OrderOutput(count);
return TruncateToOutput(count);
}
#endif