Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports. * jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls. * Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results. * Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable. * Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate. * Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do. * Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence. * Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags. * Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check. * Rugnux: Clear error messages when a data set needs more GPU or host memory than is available. Reviewed-on: #83 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
245 lines
12 KiB
Plaintext
245 lines
12 KiB
Plaintext
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "HotPixelsGPU.h"
|
|
|
|
#include "../common/JFJochException.h"
|
|
|
|
namespace {
|
|
|
|
void cuda_err(cudaError_t val) {
|
|
if (val != cudaSuccess)
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val));
|
|
}
|
|
|
|
constexpr int THREADS = 256;
|
|
|
|
// Masked or error value, and saturated: neither has a count to rank.
|
|
__device__ bool is_valid(int32_t v) {
|
|
return v != INT32_MIN && v != INT32_MAX;
|
|
}
|
|
|
|
// The rank-th smallest of the keys a block's elements j in [begin, end) give, where key_of(j, key)
|
|
// says whether element j has one. Radix selection, eight bits at a time from the top: a histogram of
|
|
// the next digit among the keys that share the digits already fixed says which digit the rank-th
|
|
// falls in. Exact for any keys, so no range has to be assumed. Called by all THREADS of the block.
|
|
template <class F>
|
|
__device__ uint32_t block_select(uint32_t begin, uint32_t end, uint32_t rank, F key_of) {
|
|
__shared__ uint32_t hist[256];
|
|
__shared__ uint32_t s_prefix, s_rank;
|
|
uint32_t prefix = 0;
|
|
for (int shift = 24; shift >= 0; shift -= 8) {
|
|
hist[threadIdx.x] = 0;
|
|
__syncthreads();
|
|
const uint32_t fixed = shift == 24 ? 0u : ~0u << (shift + 8);
|
|
for (uint32_t j = begin + threadIdx.x; j < end; j += THREADS) {
|
|
uint32_t k;
|
|
if (key_of(j, k) && (k & fixed) == prefix)
|
|
atomicAdd(&hist[(k >> shift) & 255u], 1u);
|
|
}
|
|
__syncthreads();
|
|
if (threadIdx.x == 0) {
|
|
uint32_t d = 0;
|
|
while (rank >= hist[d]) {
|
|
rank -= hist[d];
|
|
d++;
|
|
}
|
|
s_prefix = prefix | (d << shift);
|
|
s_rank = rank;
|
|
}
|
|
__syncthreads();
|
|
prefix = s_prefix;
|
|
rank = s_rank;
|
|
}
|
|
return prefix;
|
|
}
|
|
|
|
// int32 to a uint32 that sorts the same way, and back.
|
|
__device__ uint32_t order_key(int32_t v) { return static_cast<uint32_t>(v) ^ 0x80000000u; }
|
|
__device__ int32_t from_order_key(uint32_t k) { return static_cast<int32_t>(k ^ 0x80000000u); }
|
|
|
|
// One block per ring-sector: its valid pixels and their lower median.
|
|
__global__ void sector_kernel(const int32_t *__restrict__ image, const uint32_t *__restrict__ pixels,
|
|
const uint32_t *__restrict__ key_begin, uint32_t *__restrict__ count,
|
|
int32_t *__restrict__ median) {
|
|
const uint32_t begin = key_begin[blockIdx.x], end = key_begin[blockIdx.x + 1];
|
|
__shared__ uint32_t s_n;
|
|
if (threadIdx.x == 0) s_n = 0;
|
|
__syncthreads();
|
|
uint32_t n = 0;
|
|
for (uint32_t j = begin + threadIdx.x; j < end; j += THREADS)
|
|
n += is_valid(image[pixels[j]]);
|
|
atomicAdd(&s_n, n);
|
|
__syncthreads();
|
|
n = s_n;
|
|
if (threadIdx.x == 0) {
|
|
count[blockIdx.x] = n;
|
|
median[blockIdx.x] = 0;
|
|
}
|
|
if (n == 0)
|
|
return;
|
|
const uint32_t m = block_select(begin, end, n / 2, [&](uint32_t j, uint32_t &k) {
|
|
const int32_t v = image[pixels[j]];
|
|
k = order_key(v);
|
|
return is_valid(v);
|
|
});
|
|
if (threadIdx.x == 0)
|
|
median[blockIdx.x] = from_order_key(m);
|
|
}
|
|
|
|
// One block per ring: the lower median of its valid pixels, and the lower median of their absolute
|
|
// deviations from it - written as the host writes it, |v - median| in int.
|
|
__global__ void ring_kernel(const int32_t *__restrict__ image, const uint32_t *__restrict__ pixels,
|
|
const uint32_t *__restrict__ key_begin, const uint32_t *__restrict__ count,
|
|
int sectors, int32_t *__restrict__ ring_median, int32_t *__restrict__ ring_mad) {
|
|
const int first = blockIdx.x * sectors;
|
|
const uint32_t begin = key_begin[first], end = key_begin[first + sectors];
|
|
uint32_t n = 0;
|
|
for (int s = 0; s < sectors; s++)
|
|
n += count[first + s];
|
|
if (threadIdx.x == 0) {
|
|
ring_median[blockIdx.x] = 0;
|
|
ring_mad[blockIdx.x] = 0;
|
|
}
|
|
if (n == 0)
|
|
return;
|
|
const int32_t m = from_order_key(block_select(begin, end, n / 2, [&](uint32_t j, uint32_t &k) {
|
|
const int32_t v = image[pixels[j]];
|
|
k = order_key(v);
|
|
return is_valid(v);
|
|
}));
|
|
const uint32_t mad = block_select(begin, end, n / 2, [&](uint32_t j, uint32_t &k) {
|
|
const int32_t v = image[pixels[j]];
|
|
k = static_cast<uint32_t>(abs(v - m));
|
|
return is_valid(v);
|
|
});
|
|
if (threadIdx.x == 0) {
|
|
ring_median[blockIdx.x] = m;
|
|
ring_mad[blockIdx.x] = static_cast<int32_t>(mad);
|
|
}
|
|
}
|
|
|
|
// HotPixelFinder::AddImage's per-pixel loop, line for line.
|
|
__global__ void accumulate_kernel(const int32_t *__restrict__ image, const int32_t *__restrict__ key,
|
|
size_t npixels, int sectors, const int32_t *__restrict__ level,
|
|
const float *__restrict__ threshold, const char *__restrict__ ring_ok,
|
|
uint16_t *__restrict__ n_lit, uint16_t *__restrict__ n_error,
|
|
int64_t *__restrict__ sum_value, uint16_t *__restrict__ n_error_ring_ok,
|
|
int64_t *__restrict__ error_level_sum) {
|
|
for (size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x; i < npixels;
|
|
i += static_cast<size_t>(blockDim.x) * gridDim.x) {
|
|
const int32_t k = key[i], v = image[i];
|
|
if (k < 0) continue;
|
|
if (v == INT32_MIN) {
|
|
n_error[i]++;
|
|
if (ring_ok[k / sectors]) {
|
|
n_error_ring_ok[i]++;
|
|
error_level_sum[i] += level[k];
|
|
}
|
|
continue;
|
|
}
|
|
if (!ring_ok[k / sectors]) continue;
|
|
sum_value[i] += v;
|
|
if (v == INT32_MAX || static_cast<float>(v) > threshold[k])
|
|
n_lit[i]++;
|
|
}
|
|
}
|
|
|
|
// The shared tables and sums are filled on a stream of their own, added to on the workers' streams
|
|
// and downloaded on the NULL stream, so they are allocated synchronously rather than from the pool:
|
|
// a pooled buffer is freed on the thread's allocation stream, which none of those is ordered before.
|
|
// Each of those steps is waited for on the host, so this costs nothing but the device-wide sync of
|
|
// eight allocations per run, and it leaves compute-sanitizer --track-stream-ordered-races nothing
|
|
// to report here, where it flagged every one of them.
|
|
constexpr CudaAlloc ALLOC = CudaAlloc::Synchronous;
|
|
|
|
} // namespace
|
|
|
|
HotPixelFinderGPU::HotPixelFinderGPU(const int32_t *host_key, size_t npixels,
|
|
const std::vector<uint32_t> &host_key_begin, int nrings, int sectors)
|
|
: npixels(npixels), nkeys(static_cast<size_t>(nrings) * sectors), nrings(nrings),
|
|
sectors(sectors),
|
|
key(npixels, ALLOC), pixels_by_key(host_key_begin.back(), ALLOC), key_begin(host_key_begin.size(), ALLOC),
|
|
n_lit(npixels, ALLOC), n_error(npixels, ALLOC), n_error_ring_ok(npixels, ALLOC),
|
|
sum_value(npixels, ALLOC), error_level_sum(npixels, ALLOC) {
|
|
std::vector<uint32_t> pixels(host_key_begin.back());
|
|
std::vector<uint32_t> filled(host_key_begin.begin(), host_key_begin.end() - 1);
|
|
for (size_t i = 0; i < npixels; i++)
|
|
if (host_key[i] >= 0)
|
|
pixels[filled[host_key[i]]++] = static_cast<uint32_t>(i);
|
|
|
|
CudaStream stream;
|
|
cuda_err(cudaMemcpyAsync(key, host_key, npixels * sizeof(int32_t), cudaMemcpyHostToDevice, stream));
|
|
cuda_err(cudaMemcpyAsync(pixels_by_key, pixels.data(), pixels.size() * sizeof(uint32_t),
|
|
cudaMemcpyHostToDevice, stream));
|
|
cuda_err(cudaMemcpyAsync(key_begin, host_key_begin.data(), host_key_begin.size() * sizeof(uint32_t),
|
|
cudaMemcpyHostToDevice, stream));
|
|
cuda_err(cudaMemsetAsync(n_lit, 0, npixels * sizeof(uint16_t), stream));
|
|
cuda_err(cudaMemsetAsync(n_error, 0, npixels * sizeof(uint16_t), stream));
|
|
cuda_err(cudaMemsetAsync(n_error_ring_ok, 0, npixels * sizeof(uint16_t), stream));
|
|
cuda_err(cudaMemsetAsync(sum_value, 0, npixels * sizeof(int64_t), stream));
|
|
cuda_err(cudaMemsetAsync(error_level_sum, 0, npixels * sizeof(int64_t), stream));
|
|
cuda_err(cudaStreamSynchronize(stream));
|
|
}
|
|
|
|
void HotPixelFinderGPU::Statistics(const int32_t *device_image, Frame &frame, std::vector<uint32_t> &count,
|
|
std::vector<int32_t> §or_median, std::vector<int32_t> &ring_median,
|
|
std::vector<int32_t> &ring_mad) {
|
|
count.resize(nkeys);
|
|
sector_median.resize(nkeys);
|
|
ring_median.resize(nrings);
|
|
ring_mad.resize(nrings);
|
|
if (nrings == 0)
|
|
return;
|
|
if (!frame.count.get()) {
|
|
frame.count = CudaDevicePtr<uint32_t>(nkeys);
|
|
frame.sector_median = CudaDevicePtr<int32_t>(nkeys);
|
|
frame.ring_median = CudaDevicePtr<int32_t>(nrings);
|
|
frame.ring_mad = CudaDevicePtr<int32_t>(nrings);
|
|
frame.level = CudaDevicePtr<int32_t>(nkeys);
|
|
frame.threshold = CudaDevicePtr<float>(nkeys);
|
|
frame.ring_ok = CudaDevicePtr<char>(nrings);
|
|
}
|
|
const cudaStream_t stream = *frame.stream;
|
|
sector_kernel<<<static_cast<unsigned>(nkeys), THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin, frame.count,
|
|
frame.sector_median);
|
|
cuda_err(cudaGetLastError());
|
|
ring_kernel<<<nrings, THREADS, 0, stream>>>(device_image, pixels_by_key, key_begin, frame.count, sectors,
|
|
frame.ring_median, frame.ring_mad);
|
|
cuda_err(cudaGetLastError());
|
|
|
|
cuda_err(cudaMemcpyAsync(count.data(), frame.count, nkeys * sizeof(uint32_t), cudaMemcpyDeviceToHost, stream));
|
|
cuda_err(cudaMemcpyAsync(sector_median.data(), frame.sector_median, nkeys * sizeof(int32_t),
|
|
cudaMemcpyDeviceToHost, stream));
|
|
cuda_err(cudaMemcpyAsync(ring_median.data(), frame.ring_median, nrings * sizeof(int32_t),
|
|
cudaMemcpyDeviceToHost, stream));
|
|
cuda_err(cudaMemcpyAsync(ring_mad.data(), frame.ring_mad, nrings * sizeof(int32_t),
|
|
cudaMemcpyDeviceToHost, stream));
|
|
cuda_err(cudaStreamSynchronize(stream));
|
|
}
|
|
|
|
void HotPixelFinderGPU::Accumulate(const int32_t *device_image, Frame &frame, const std::vector<int32_t> &level,
|
|
const std::vector<float> &threshold, const std::vector<char> &ring_ok) {
|
|
const cudaStream_t stream = *frame.stream;
|
|
cuda_err(cudaMemcpyAsync(frame.level, level.data(), nkeys * sizeof(int32_t), cudaMemcpyHostToDevice, stream));
|
|
cuda_err(cudaMemcpyAsync(frame.threshold, threshold.data(), nkeys * sizeof(float), cudaMemcpyHostToDevice,
|
|
stream));
|
|
cuda_err(cudaMemcpyAsync(frame.ring_ok, ring_ok.data(), nrings * sizeof(char), cudaMemcpyHostToDevice, stream));
|
|
|
|
std::lock_guard lock(accumulate_mutex);
|
|
accumulate_kernel<<<static_cast<unsigned>((npixels + THREADS - 1) / THREADS), THREADS, 0, stream>>>(
|
|
device_image, key, npixels, sectors, frame.level, frame.threshold, frame.ring_ok,
|
|
n_lit, n_error, sum_value, n_error_ring_ok, error_level_sum);
|
|
cuda_err(cudaGetLastError());
|
|
cuda_err(cudaStreamSynchronize(stream));
|
|
}
|
|
|
|
void HotPixelFinderGPU::Download(uint16_t *host_n_lit, uint16_t *host_n_error, int64_t *host_sum_value,
|
|
uint16_t *host_n_error_ring_ok, int64_t *host_error_level_sum) {
|
|
cuda_err(cudaMemcpy(host_n_lit, n_lit, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
|
|
cuda_err(cudaMemcpy(host_n_error, n_error, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
|
|
cuda_err(cudaMemcpy(host_sum_value, sum_value, npixels * sizeof(int64_t), cudaMemcpyDeviceToHost));
|
|
cuda_err(cudaMemcpy(host_n_error_ring_ok, n_error_ring_ok, npixels * sizeof(uint16_t), cudaMemcpyDeviceToHost));
|
|
cuda_err(cudaMemcpy(host_error_level_sum, error_level_sum, npixels * sizeof(int64_t), cudaMemcpyDeviceToHost));
|
|
}
|