Files
Jungfraujoch/image_analysis/beam_stop/ShadowMaskGPU.cu
T
leonarski_f a395f358ef
Build Packages / Create release (push) Successful in 17s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m22s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m37s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 9m33s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 10m39s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 11m4s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 13m19s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 17m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 18m49s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 19m10s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m26s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m31s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 18m54s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m45s
Build Packages / Generate python client (push) Successful in 37s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 20m20s
Build Packages / Build documentation (push) Successful in 1m32s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m37s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m6s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 19m49s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 20m29s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 17m2s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 14m27s
Build Packages / Unit tests (push) Successful in 1h18m12s
1.0.0-rc.174 (#84)
* Rugnux: Performance improvements on GPU and CPU (more of the pre-scan and of scaling on the GPU, faster CPU spot finding and crystal refinement), with unchanged results.
* Rugnux: More robust processing - patches of persistently hot pixels are masked, an inconsistent merge triggers a retry at the measured beam centre, and builds targeting different CPU levels give the same results.
* Rugnux: Improved scaling and merging - reflections with an overloaded pixel are dropped, as in XDS, sparse rotation sweeps are scaled more reliably, and French-Wilson amplitudes use an anisotropic Wilson prior.
* Rugnux: Improved space-group determination - glide planes in groups without a centre of symmetry, screw axes from short or weak axial rows kept when a higher group is adopted, and more reliable decisions on twinned and pseudo-symmetric crystals.
* Rugnux: Improved small-molecule processing - spots that grow wider than the integration disk and split spots are integrated over their measured footprint, sparse lattices are integrated on every frame, and the `.hkl` file holds unmerged scaled reflections (SHELX HKLF 4).
* Rugnux: Reads Rigaku d*TREK SMV images (Saturn CCD), including detector 2theta and encoded pixel overflows; home-source (rotating-anode) datasets were added to the validation battery.
* jfjoch_viewer: Fixed processing failing at the end with "Wrong JPEG library version" on Linux; the merge window shows the space group with proper subscripts and a checklist of crystal pathologies.

Reviewed-on: #84
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-10-06 14:03:18 +02:00

637 lines
30 KiB
Plaintext

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "ShadowMaskGPU.h"
#include <cub/device/device_radix_sort.cuh>
#include "ShadowFinderInternal.h"
#include "../../common/JFJochMath.h"
#include "../indexing/CUDAMemHelpers.h"
#include "../../common/JFJochException.h"
using namespace shadow_finder;
namespace {
constexpr int THREADS = 256;
void check(cudaError_t err, const char *what) {
if (err != cudaSuccess)
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
std::string("Beam stop mask: ") + what + ": " + cudaGetErrorString(err));
}
unsigned grid(size_t n) {
return static_cast<unsigned>((n + THREADS - 1) / THREADS);
}
// A float as an unsigned integer that sorts the same way, and back.
__device__ uint32_t float_key(float f) {
const uint32_t u = __float_as_uint(f);
return (u & 0x80000000u) ? ~u : (u | 0x80000000u);
}
__device__ float key_float(uint32_t k) {
return __uint_as_float((k & 0x80000000u) ? (k & 0x7fffffffu) : ~k);
}
// The first index in a sorted key array whose key is not below `value`.
__device__ size_t lower_bound(const uint64_t *keys, size_t n, uint64_t value) {
size_t lo = 0, hi = n;
while (lo < hi) {
const size_t mid = (lo + hi) / 2;
if (keys[mid] < value) lo = mid + 1;
else hi = mid;
}
return lo;
}
__device__ double poisson_deficit_sigma(double observed, double expected) {
if (expected <= 0.0 || observed >= expected)
return 0.0;
const double ll = 2.0 * (expected - observed + (observed > 0.0 ? observed * log(observed / expected) : 0.0));
return ll > 0.0 ? sqrt(ll) : 0.0;
}
// DiffractionGeometry::CalcAzIntPolarizationCorr about the centre the rings are drawn about.
__device__ float polarization_factor(const ShadowMaskSetup &s, float x, float y) {
const float u = (x - s.beam_x) * s.pixel_size_mm;
const float v = (y - s.beam_y) * s.pixel_size_mm;
const float *m = s.det_matrix;
const float lx = m[0] * u + m[1] * v + m[2] * s.distance_mm;
const float ly = m[3] * u + m[4] * v + m[5] * s.distance_mm;
const float lz = m[6] * u + m[7] * v + m[8] * s.distance_mm;
const float two_theta = atan2f(sqrtf(lx * lx + ly * ly), lz);
float phi = atan2f(ly, lx);
if (phi < 0)
phi += 2.0f * PI;
const float cos_2theta = cosf(two_theta);
const float cos_2theta_2 = cos_2theta * cos_2theta;
const float cos_2phi = cosf(2.0f * phi);
return 0.5f * (1.0f + cos_2theta_2 - s.polarization * cos_2phi * (1.0f - cos_2theta_2));
}
__global__ void setup_kernel(ShadowMaskSetup s, const uint32_t *__restrict__ pixel_mask,
const int64_t *__restrict__ sum_value, const uint32_t *__restrict__ valid_count,
float *__restrict__ pol, char *__restrict__ valid, int *__restrict__ radius,
double *__restrict__ num, int32_t *__restrict__ den, int *__restrict__ max_radius) {
const size_t n = static_cast<size_t>(s.width) * s.height;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n)
return;
const int x = static_cast<int>(i % s.width), y = static_cast<int>(i / s.width);
const float dx = x - s.beam_x, dy = y - s.beam_y;
const float p = s.has_polarization ? polarization_factor(s, static_cast<float>(x), static_cast<float>(y)) : 1.0f;
pol[i] = p;
float mean = 0.0f;
char v = 0;
if (valid_count[i] > 0 && pixel_mask[i] == 0 && p > 0.0f) {
mean = static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i] / p);
v = 1;
}
valid[i] = v;
num[i] = v ? mean : 0.0;
den[i] = v ? 1 : 0;
const int r = static_cast<int>(lroundf(sqrtf(dx * dx + dy * dy)));
radius[i] = r;
atomicMax(max_radius, r);
}
// Sum over the k x k box about each pixel, zero outside the frame: one running sum per row, then one
// per column, each with exactly the terms and order of the host's (box_sum in ShadowFinder.cpp).
template <typename T>
__global__ void box_rows(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) {
const int y = blockIdx.x * blockDim.x + threadIdx.x;
if (y >= H) return;
const T *src = in + static_cast<size_t>(y) * W;
T *dst = out + static_cast<size_t>(y) * W;
T s = 0;
for (int x = 0; x <= min(half, W - 1); x++)
s += src[x];
for (int x = 0; x < W; x++) {
dst[x] = s;
if (x + half + 1 < W) s += src[x + half + 1];
if (x - half >= 0) s -= src[x - half];
}
}
template <typename T>
__global__ void box_columns(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) {
const int x = blockIdx.x * blockDim.x + threadIdx.x;
if (x >= W) return;
T s = 0;
for (int y = 0; y <= min(half, H - 1); y++)
s += in[static_cast<size_t>(y) * W + x];
for (int y = 0; y < H; y++) {
out[static_cast<size_t>(y) * W + x] = s;
if (y + half + 1 < H) s += in[static_cast<size_t>(y + half + 1) * W + x];
if (y - half >= 0) s -= in[static_cast<size_t>(y - half) * W + x];
}
}
// Dilation of a 0/1 plane by the (2r+1) square clipped to the frame, as a count over a sliding window
// along rows and then along columns (dilate in ShadowFinder.cpp).
__global__ void dilate_rows(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) {
const int y = blockIdx.x * blockDim.x + threadIdx.x;
if (y >= H) return;
const char *src = in + static_cast<size_t>(y) * W;
char *dst = out + static_cast<size_t>(y) * W;
int count = 0;
for (int x = 0; x <= min(r, W - 1); x++)
count += src[x];
for (int x = 0; x < W; x++) {
dst[x] = count > 0;
if (x + r + 1 < W) count += src[x + r + 1];
if (x - r >= 0) count -= src[x - r];
}
}
__global__ void dilate_columns(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) {
const int x = blockIdx.x * blockDim.x + threadIdx.x;
if (x >= W) return;
int count = 0;
for (int y = 0; y <= min(r, H - 1); y++)
count += in[static_cast<size_t>(y) * W + x];
for (int y = 0; y < H; y++) {
out[static_cast<size_t>(y) * W + x] = count > 0;
if (y + r + 1 < H) count += in[static_cast<size_t>(y + r + 1) * W + x];
if (y - r >= 0) count -= in[static_cast<size_t>(y - r) * W + x];
}
}
__global__ void pooled_kernel(size_t n, const double *__restrict__ pooled_sum, const int32_t *__restrict__ pooled_count,
const char *__restrict__ valid, const int *__restrict__ radius,
float *__restrict__ pooled, uint64_t *__restrict__ ring_key) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
const float p = pooled_count[i] > 0 ? static_cast<float>(pooled_sum[i] / pooled_count[i]) : 0.0f;
pooled[i] = p;
ring_key[i] = valid[i] ? (static_cast<uint64_t>(radius[i]) << 32) | float_key(p) : UINT64_MAX;
}
// Where each of the keys' leading 32-bit groups (ring or sector) starts in a sorted key array.
__global__ void group_offsets(const uint64_t *__restrict__ keys, size_t n, int groups, int *__restrict__ offset) {
const int g = blockIdx.x * blockDim.x + threadIdx.x;
if (g > groups) return;
offset[g] = static_cast<int>(lower_bound(keys, n, static_cast<uint64_t>(g) << 32));
}
// The ring's baseline, three iterations of an order statistic over its sorted values (GetMask).
__global__ void baseline_kernel(int rings, const uint64_t *__restrict__ keys, const int *__restrict__ offset,
float *__restrict__ baseline) {
const int r = blockIdx.x * blockDim.x + threadIdx.x;
if (r >= rings) return;
const int lo = offset[r], n = offset[r + 1] - offset[r];
int excluded = 0;
float b = 0.0f;
for (int iter = 0; iter < 3; iter++) {
const int avail = n - excluded;
b = (avail <= 0) ? 0.0f : key_float(static_cast<uint32_t>(keys[lo + excluded + avail / 2]));
const float d = fmaxf(b, 1e-6f);
int excl = 0;
while (excl < n && key_float(static_cast<uint32_t>(keys[lo + excl])) / d < SHADOW_RATIO)
excl++;
excluded = excl;
}
baseline[r] = b;
}
__global__ void low_kernel(size_t n, uint32_t frames, const char *__restrict__ valid, const float *__restrict__ pooled,
const int32_t *__restrict__ pooled_count, const float *__restrict__ pol,
const int *__restrict__ radius, const float *__restrict__ baseline,
float *__restrict__ ratio, float *__restrict__ deficit, char *__restrict__ low) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
const float base = baseline[radius[i]];
const float rt = valid[i] ? pooled[i] / fmaxf(base, 1e-6f) : 1.0f;
ratio[i] = rt;
float df = 0.0f;
char l = 0;
if (valid[i]) {
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
df = static_cast<float>(poisson_deficit_sigma(pooled[i] * counted, base * counted));
l = rt < SHADOW_RATIO && df > MIN_DEFICIT_SIGMA;
}
deficit[i] = df;
low[i] = l;
}
// 8-connected components by union-find: every component ends up named by its smallest pixel index,
// whatever order the unions ran in.
__device__ int find_root(const int *parent, int x) {
while (parent[x] != x)
x = parent[x];
return x;
}
__global__ void cc_init(size_t n, const char *__restrict__ member, int *__restrict__ parent) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
parent[i] = member[i] ? static_cast<int>(i) : -1;
}
__device__ void cc_unite(int *parent, int a, int b) {
while (true) {
a = find_root(parent, a);
b = find_root(parent, b);
if (a == b) return;
if (a < b) { const int t = a; a = b; b = t; }
if (atomicCAS(&parent[a], a, b) == a) return;
}
}
__global__ void cc_union(int W, int H, const char *__restrict__ member, int *parent) {
const size_t n = static_cast<size_t>(W) * H;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n || !member[i]) return;
const int x = static_cast<int>(i % W), y = static_cast<int>(i / W);
// The four neighbours before this pixel; the other four see it from their side.
if (x > 0 && member[i - 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(i - 1));
if (y > 0) {
const size_t up = i - W;
if (member[up]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up));
if (x > 0 && member[up - 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up - 1));
if (x + 1 < W && member[up + 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up + 1));
}
}
__global__ void cc_flatten(size_t n, int *parent) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n || parent[i] < 0) return;
parent[i] = find_root(parent, static_cast<int>(i));
}
// Per component (by root): how many of its pixels are `a`, and how many are both `a` and `b`.
__global__ void cc_count(size_t n, const int *__restrict__ root, const char *__restrict__ a, const char *__restrict__ b,
int *__restrict__ count_a, int *__restrict__ count_ab) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n || root[i] < 0 || !a[i]) return;
atomicAdd(&count_a[root[i]], 1);
if (count_ab && b[i])
atomicAdd(&count_ab[root[i]], 1);
}
__global__ void region_kernel(size_t n, const int *__restrict__ root, const int *__restrict__ n_low,
const char *__restrict__ low, const char *__restrict__ valid, const int *__restrict__ radius,
int blocked_out_to, char *__restrict__ region) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
char r = root[i] >= 0 && n_low[root[i]] >= MIN_SHADOW_PIXELS ? low[i] : 0;
if (valid[i] && radius[i] <= blocked_out_to)
r = 1;
region[i] = r;
}
__global__ void lit_kernel(size_t n, const uint32_t *__restrict__ valid_count, const int64_t *__restrict__ max_value,
char *__restrict__ lit) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
lit[i] = (valid_count[i] > 0) && (max_value[i] >= MIN_REFLECTION);
}
__global__ void reflection_kernel(int W, int H, const char *__restrict__ lit, char *__restrict__ reflection) {
const size_t n = static_cast<size_t>(W) * H;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
const int x = static_cast<int>(i % W), y = static_cast<int>(i / W);
char r = 0;
if (lit[i]) {
int neighbours = 0;
for (int dy = -1; dy <= 1; dy++)
for (int dx = -1; dx <= 1; dx++) {
const int yy = y + dy, xx = x + dx;
if ((dx || dy) && yy >= 0 && yy < H && xx >= 0 && xx < W && lit[static_cast<size_t>(yy) * W + xx])
neighbours++;
}
r = neighbours >= 2;
}
reflection[i] = r;
}
__global__ void penumbra_kernel(size_t n, const char *__restrict__ penumbra, const char *__restrict__ valid,
const float *__restrict__ ratio, const float *__restrict__ deficit, char *__restrict__ region) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
if (penumbra[i] && valid[i] && ratio[i] < PENUMBRA_RATIO && deficit[i] > MIN_DEFICIT_SIGMA)
region[i] = 1;
}
__global__ void invert_kernel(size_t n, const char *__restrict__ in, char *__restrict__ out) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
out[i] = !in[i];
}
// Background components that touch the frame's edge are outside; the rest are holes.
__global__ void outside_kernel(int W, int H, const int *__restrict__ root, int *__restrict__ outside) {
const int k = blockIdx.x * blockDim.x + threadIdx.x;
const int perimeter = 2 * W + 2 * H;
if (k >= perimeter) return;
int x, y;
if (k < W) { x = k; y = 0; }
else if (k < 2 * W) { x = k - W; y = H - 1; }
else if (k < 2 * W + H) { x = 0; y = k - 2 * W; }
else { x = W - 1; y = k - 2 * W - H; }
const int r = root[static_cast<size_t>(y) * W + x];
if (r >= 0) outside[r] = 1;
}
__global__ void final_kernel(size_t n, const int *__restrict__ background_root, const int *__restrict__ outside,
const char *__restrict__ reflection_grown, char *__restrict__ region,
uint32_t *__restrict__ mask) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
char r = region[i];
if (background_root[i] >= 0 && !outside[background_root[i]])
r = 1; // a hole
if (reflection_grown[i])
r = 0;
region[i] = r;
mask[i] = r ? MASK_SHADOW : 0;
}
__global__ void sector_key_kernel(ShadowMaskSetup s, const char *__restrict__ valid, const char *__restrict__ region,
const int *__restrict__ radius, const float *__restrict__ ratio,
uint64_t *__restrict__ key) {
const size_t n = static_cast<size_t>(s.width) * s.height;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
if (!valid[i] || region[i]) {
key[i] = UINT64_MAX;
return;
}
const float dx = static_cast<float>(i % s.width) - s.beam_x, dy = static_cast<float>(i / s.width) - s.beam_y;
const double phi = atan2f(dy, dx) + PI;
const int sector = min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * PI) * HARMONIC_SECTORS));
const uint64_t k = static_cast<uint64_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector;
key[i] = (k << 32) | float_key(ratio[i]);
}
__global__ void sector_median_kernel(int sectors, const uint64_t *__restrict__ keys, const int *__restrict__ offset,
double *__restrict__ median) {
const int k = blockIdx.x * blockDim.x + threadIdx.x;
if (k >= sectors) return;
const int n = offset[k + 1] - offset[k];
median[k] = n >= MIN_SECTOR_PIXELS ? key_float(static_cast<uint32_t>(keys[offset[k] + n / 2])) : -1.0;
}
__global__ void dim_kernel(ShadowMaskSetup s, uint32_t frames, const char *__restrict__ valid,
const char *__restrict__ region, const float *__restrict__ ratio,
const float *__restrict__ deficit, const int *__restrict__ radius,
const float *__restrict__ harm_c, const float *__restrict__ harm_s,
const int32_t *__restrict__ pooled_count, const float *__restrict__ pol,
const float *__restrict__ pooled, const float *__restrict__ baseline,
char *__restrict__ dim) {
const size_t n = static_cast<size_t>(s.width) * s.height;
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
char d = 0;
if (valid[i] && !region[i] && ratio[i] < PENUMBRA_RATIO) {
// cos 2phi and sin 2phi from the offset to the beam.
const float dx = static_cast<float>(i % s.width) - s.beam_x, dy = static_cast<float>(i / s.width) - s.beam_y;
const float r2 = fmaxf(dx * dx + dy * dy, 1e-6f);
const int band = radius[i] / HARMONIC_BAND_PX;
const float model = fminf(1.0f, 1.0f + harm_c[band] * (dx * dx - dy * dy) / r2
+ harm_s[band] * 2.0f * dx * dy / r2);
if (model >= 1.0f) {
d = deficit[i] > MIN_DEFICIT_SIGMA;
} else {
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
d = ratio[i] < PENUMBRA_RATIO * model
&& poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA;
}
}
dim[i] = d;
}
// Join a region across the module gaps it crosses, one line per thread (bridge_gaps in
// ShadowFinder.cpp). Both directions read `region` and only ever set pixels of `out` to 1.
__global__ void bridge_rows(int W, int H, const char *__restrict__ region, const char *__restrict__ valid,
char *out) {
const int y = blockIdx.x * blockDim.x + threadIdx.x;
if (y >= H) return;
const size_t row = static_cast<size_t>(y) * W;
int k = 0;
while (k < W) {
if (valid[row + k]) { k++; continue; }
const int start = k;
while (k < W && !valid[row + k]) k++;
if (start > 0 && k < W && region[row + start - 1] && region[row + k])
for (int j = start; j < k; j++) out[row + j] = 1;
}
}
__global__ void bridge_columns(int W, int H, const char *__restrict__ region, const char *__restrict__ valid,
char *out) {
const int x = blockIdx.x * blockDim.x + threadIdx.x;
if (x >= W) return;
const auto at = [&](int y) { return static_cast<size_t>(y) * W + x; };
int k = 0;
while (k < H) {
if (valid[at(k)]) { k++; continue; }
const int start = k;
while (k < H && !valid[at(k)]) k++;
if (start > 0 && k < H && region[at(start - 1)] && region[at(k)])
for (int j = start; j < k; j++) out[at(j)] = 1;
}
}
__global__ void transmitting_kernel(size_t n, const int *__restrict__ root, const char *__restrict__ dim,
const int *__restrict__ n_dim, const int *__restrict__ n_low,
uint32_t *__restrict__ mask) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
const int c = root[i];
if (c >= 0 && dim[i] && n_dim[c] >= MIN_SHADOW_PIXELS && n_low[c] >= MIN_CORE_PIXELS)
mask[i] = MASK_TRANSMITTING;
}
__global__ void mean_kernel(size_t n, const uint32_t *__restrict__ pixel_mask, const int64_t *__restrict__ sum_value,
const uint32_t *__restrict__ valid_count, float *__restrict__ mean) {
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
if (i >= n) return;
mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0
? static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]) : NAN;
}
// The device side of one GetMask: planes, sort scratch and the stream to run on.
class MaskEngine {
public:
const ShadowMaskSetup s;
const int W, H;
const size_t n;
cudaStream_t stream;
MaskEngine(const ShadowMaskSetup &setup, cudaStream_t st)
: s(setup), W(setup.width), H(setup.height), n(static_cast<size_t>(setup.width) * setup.height), stream(st) {}
void Check(const char *what) const {
check(cudaGetLastError(), what);
}
void Dilate(const char *in, char *out, char *scratch, int r) const {
dilate_rows<<<grid(H), THREADS, 0, stream>>>(in, scratch, W, H, r);
dilate_columns<<<grid(W), THREADS, 0, stream>>>(scratch, out, W, H, r);
Check("dilate");
}
// Components of `member`, each pixel's root in `root` (-1 outside every component).
void Label(const char *member, int *root) const {
cc_init<<<grid(n), THREADS, 0, stream>>>(n, member, root);
cc_union<<<grid(n), THREADS, 0, stream>>>(W, H, member, root);
cc_flatten<<<grid(n), THREADS, 0, stream>>>(n, root);
Check("components");
}
void SortKeys(uint64_t *keys, uint64_t *sorted) const {
size_t bytes = 0;
check(cub::DeviceRadixSort::SortKeys(nullptr, bytes, keys, sorted, n, 0, 64, stream), "sort size");
CudaDevicePtr<uint8_t> scratch(bytes);
check(cub::DeviceRadixSort::SortKeys(scratch.get(), bytes, keys, sorted, n, 0, 64, stream), "sort");
// The scratch is freed on the allocation stream, which knows nothing of this one.
check(cudaStreamSynchronize(stream), "sort");
}
template <typename T>
std::vector<T> Download(const T *device, size_t count) const {
std::vector<T> host(count);
check(cudaMemcpyAsync(host.data(), device, count * sizeof(T), cudaMemcpyDeviceToHost, stream), "download");
check(cudaStreamSynchronize(stream), "download");
return host;
}
template <typename T>
void Upload(T *device, const std::vector<T> &host) const {
check(cudaMemcpyAsync(device, host.data(), host.size() * sizeof(T), cudaMemcpyHostToDevice, stream), "upload");
}
};
} // namespace
std::vector<uint32_t> ShadowMaskOnDevice(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask,
const int64_t *max_value, const int64_t *sum_value,
const uint32_t *valid_count, uint32_t frames, cudaStream_t stream) {
const MaskEngine e(setup, stream);
const size_t n = e.n;
const int W = e.W, H = e.H;
CudaDevicePtr<uint32_t> d_pixel_mask(n), mask(n);
e.Upload(d_pixel_mask.get(), pixel_mask);
// Mean projection over the polarization factor, usable pixels and radius from the beam centre.
CudaDevicePtr<float> pol(n), pooled(n), ratio(n), deficit(n);
CudaDevicePtr<char> valid(n), low(n), region(n), a(n), b(n), c(n);
CudaDevicePtr<int> radius(n), root(n), count_a(n), count_b(n), max_radius(1);
CudaDevicePtr<double> num(n), dsum(n);
CudaDevicePtr<int32_t> den(n), dcount(n);
check(cudaMemsetAsync(max_radius.get(), 0, sizeof(int), stream), "memset");
setup_kernel<<<grid(n), THREADS, 0, stream>>>(setup, d_pixel_mask, sum_value, valid_count, pol, valid, radius,
num, den, max_radius);
e.Check("setup");
const int max_r = e.Download(max_radius.get(), 1)[0];
const int rings = max_r + 1;
// The background pooled over a small box.
box_rows<double><<<grid(H), THREADS, 0, stream>>>(num, dsum, W, H, POOL_PX / 2);
box_columns<double><<<grid(W), THREADS, 0, stream>>>(dsum, num, W, H, POOL_PX / 2);
box_rows<int32_t><<<grid(H), THREADS, 0, stream>>>(den, dcount, W, H, POOL_PX / 2);
box_columns<int32_t><<<grid(W), THREADS, 0, stream>>>(dcount, den, W, H, POOL_PX / 2);
e.Check("pooling");
const double *pooled_sum = num;
const int32_t *pooled_count = den;
// The rings, each sorted once; the baseline is an order statistic of them.
CudaDevicePtr<uint64_t> keys(n), sorted(n);
pooled_kernel<<<grid(n), THREADS, 0, stream>>>(n, pooled_sum, pooled_count, valid, radius, pooled, keys);
e.Check("pooled");
e.SortKeys(keys, sorted);
CudaDevicePtr<int> ring_offset(rings + 1);
group_offsets<<<grid(rings + 1), THREADS, 0, stream>>>(sorted, n, rings, ring_offset);
CudaDevicePtr<float> baseline(rings);
baseline_kernel<<<grid(rings), THREADS, 0, stream>>>(rings, sorted, ring_offset, baseline);
e.Check("baseline");
const auto host_baseline = e.Download(baseline.get(), rings);
const auto offsets = e.Download(ring_offset.get(), rings + 1);
std::vector<int> ring_pixels(rings);
for (int r = 0; r < rings; r++)
ring_pixels[r] = offsets[r + 1] - offsets[r];
const int blocked_out_to = BlockedOutTo(host_baseline, ring_pixels);
// Low pixels, and the regions of them large enough to be hardware.
low_kernel<<<grid(n), THREADS, 0, stream>>>(n, frames, valid, pooled, pooled_count, pol, radius, baseline,
ratio, deficit, low);
e.Check("low");
e.Dilate(low, a, c, BRIDGE_PX);
e.Label(a, root);
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
cc_count<<<grid(n), THREADS, 0, stream>>>(n, root, low, low, count_a, nullptr);
region_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, count_a, low, valid, radius, blocked_out_to, region);
e.Check("region");
// Recorded reflections; `b` holds them until they are given back at the end.
lit_kernel<<<grid(n), THREADS, 0, stream>>>(n, valid_count, max_value, a);
reflection_kernel<<<grid(n), THREADS, 0, stream>>>(W, H, a, b);
e.Check("reflections");
// Penumbra, round and fill.
e.Dilate(region, a, c, PENUMBRA_MAX_PX);
penumbra_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, valid, ratio, deficit, region);
e.Dilate(region, a, c, 2);
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, region);
e.Dilate(region, a, c, 2);
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, region); // region = erode(dilate(region))
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, region, a); // the background
e.Label(a, root);
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
outside_kernel<<<grid(2 * W + 2 * H), THREADS, 0, stream>>>(W, H, root, count_a);
e.Dilate(b, c, a, 1); // the reflections, grown by one
final_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, count_a, c, region, mask);
e.Check("fill");
// The arm search: sector medians of the ratio, the harmonic of each band, the dim pixels.
const int n_bands = max_r / HARMONIC_BAND_PX + 1;
const int n_sectors = n_bands * HARMONIC_SECTORS;
sector_key_kernel<<<grid(n), THREADS, 0, stream>>>(setup, valid, region, radius, ratio, keys);
e.Check("sectors");
e.SortKeys(keys, sorted);
CudaDevicePtr<int> sector_offset(n_sectors + 1);
group_offsets<<<grid(n_sectors + 1), THREADS, 0, stream>>>(sorted, n, n_sectors, sector_offset);
CudaDevicePtr<double> sector_median(n_sectors);
sector_median_kernel<<<grid(n_sectors), THREADS, 0, stream>>>(n_sectors, sorted, sector_offset, sector_median);
e.Check("sector medians");
std::vector<float> harm_c, harm_s;
HarmonicFit(e.Download(sector_median.get(), n_sectors), n_bands, harm_c, harm_s);
CudaDevicePtr<float> d_harm_c(n_bands), d_harm_s(n_bands);
e.Upload(d_harm_c.get(), harm_c);
e.Upload(d_harm_s.get(), harm_s);
dim_kernel<<<grid(n), THREADS, 0, stream>>>(setup, frames, valid, region, ratio, deficit, radius, d_harm_c, d_harm_s,
pooled_count, pol, pooled, baseline, a);
e.Check("dim");
e.Dilate(a, b, c, BRIDGE_PX);
check(cudaMemcpyAsync(c.get(), b.get(), n, cudaMemcpyDeviceToDevice, stream), "copy");
bridge_rows<<<grid(H), THREADS, 0, stream>>>(W, H, b, valid, c);
bridge_columns<<<grid(W), THREADS, 0, stream>>>(W, H, b, valid, c);
e.Label(c, root);
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
check(cudaMemsetAsync(count_b.get(), 0, n * sizeof(int), stream), "memset");
cc_count<<<grid(n), THREADS, 0, stream>>>(n, root, a, low, count_a, count_b);
transmitting_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, a, count_a, count_b, mask);
e.Check("transmitting");
return e.Download(mask.get(), n);
}
std::vector<float> MeanProjectionOnDevice(const std::vector<uint32_t> &pixel_mask, const int64_t *sum_value,
const uint32_t *valid_count, size_t npixels, cudaStream_t stream) {
CudaDevicePtr<uint32_t> d_pixel_mask(npixels);
CudaDevicePtr<float> mean(npixels);
check(cudaMemcpyAsync(d_pixel_mask.get(), pixel_mask.data(), npixels * sizeof(uint32_t), cudaMemcpyHostToDevice,
stream), "upload");
mean_kernel<<<grid(npixels), THREADS, 0, stream>>>(npixels, d_pixel_mask, sum_value, valid_count, mean);
check(cudaGetLastError(), "mean");
std::vector<float> host(npixels);
check(cudaMemcpyAsync(host.data(), mean.get(), npixels * sizeof(float), cudaMemcpyDeviceToHost, stream), "download");
check(cudaStreamSynchronize(stream), "mean");
return host;
}