The two pre-scan steps that were still CPU-bound in a GPU build now run where the projection already is. - FindBeamCenterFromBackground: the per-iteration binning pass and the two clip rounds run on the device (BeamCenterBackgroundGPU); the fit itself stays on the host. Each cell is summed in the host's order (pixel order within the host's row blocks, blocks in order), and the per-pixel cell/derivative formula is shared (BackgroundBand.h). The angles come from BackgroundAtan2 (IEEE ops only) instead of atan2f, and both translation units are compiled without FMA contraction, so host and device give the same bits: 0 of 6.5 M pixels in a different cell, identical walks on the three in-house rotation sets. With glibc/CUDA atan2f and default contraction ~30 pixels per 16 Mpx sweep changed cell and the fitted centre moved by up to 0.05 px. - ShadowFinder::GetMask: the whole mask (pooling, ring medians, components, morphology, hole fill, arm search) runs on the device from ShadowAccumulatorGPU's projection (ShadowMaskGPU), so the 360 MB projection no longer comes back; the mean projection is divided on the device too (same bits). The two small fits over rings and sectors (BlockedOutTo, HarmonicFit) are shared with the host path in ShadowFinderInternal.h. Integers, comparisons, sorts and components are exact; the polarization trig, the Poisson log and the arm-search azimuth are not, so a pixel at a threshold can differ. The one-time change against the previous CPU arithmetic (BackgroundAtan2, no contraction), measured on the myoglobin, cytochrome C and thaumatin rotation sets: ring centre moves 0.002-0.045 px (fit sigma 0.75-1.2 px), beam-centre capture 0.01-0.04 px; beam-stop mask differs on 31 / 144 / 53 pixels of 259k / 144k / 198k (25 of the myoglobin ones are GPU-vs-CPU arithmetic in the mask, the rest follow the centre); hot-pixel mask identical. Spot width, integration radii, bandwidth, beam-centre arbitration, indexing, space group, cell, resolution and the merged statistics table are identical; only the error model moves in its 4th digit. CPU build: the same centres and decisions. Timing (GPU, box at load 30-38): ring walk 0.54 -> 0.23-0.27 s, mask 1.24-1.44 -> 0.18-0.22 s, beam-centre capture walk 1.1-1.3 -> 0.31-0.35 s. Tests: ShadowFinder_DeviceMaskMatchesHost, BeamCenterFromBackground_DeviceMatchesHost (bit-exact), plus [ShadowFinder], [BeamCenter], [HotPixelFinder]. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
637 lines
30 KiB
Plaintext
637 lines
30 KiB
Plaintext
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include "ShadowMaskGPU.h"
|
|
|
|
#include <cub/device/device_radix_sort.cuh>
|
|
|
|
#include "ShadowFinderInternal.h"
|
|
#include "../../common/JFJochMath.h"
|
|
#include "../indexing/CUDAMemHelpers.h"
|
|
#include "../../common/JFJochException.h"
|
|
|
|
using namespace shadow_finder;
|
|
|
|
namespace {
|
|
|
|
constexpr int THREADS = 256;
|
|
|
|
void check(cudaError_t err, const char *what) {
|
|
if (err != cudaSuccess)
|
|
throw JFJochException(JFJochExceptionCategory::GPUCUDAError,
|
|
std::string("Beam stop mask: ") + what + ": " + cudaGetErrorString(err));
|
|
}
|
|
|
|
unsigned grid(size_t n) {
|
|
return static_cast<unsigned>((n + THREADS - 1) / THREADS);
|
|
}
|
|
|
|
// A float as an unsigned integer that sorts the same way, and back.
|
|
__device__ uint32_t float_key(float f) {
|
|
const uint32_t u = __float_as_uint(f);
|
|
return (u & 0x80000000u) ? ~u : (u | 0x80000000u);
|
|
}
|
|
__device__ float key_float(uint32_t k) {
|
|
return __uint_as_float((k & 0x80000000u) ? (k & 0x7fffffffu) : ~k);
|
|
}
|
|
|
|
// The first index in a sorted key array whose key is not below `value`.
|
|
__device__ size_t lower_bound(const uint64_t *keys, size_t n, uint64_t value) {
|
|
size_t lo = 0, hi = n;
|
|
while (lo < hi) {
|
|
const size_t mid = (lo + hi) / 2;
|
|
if (keys[mid] < value) lo = mid + 1;
|
|
else hi = mid;
|
|
}
|
|
return lo;
|
|
}
|
|
|
|
__device__ double poisson_deficit_sigma(double observed, double expected) {
|
|
if (expected <= 0.0 || observed >= expected)
|
|
return 0.0;
|
|
const double ll = 2.0 * (expected - observed + (observed > 0.0 ? observed * log(observed / expected) : 0.0));
|
|
return ll > 0.0 ? sqrt(ll) : 0.0;
|
|
}
|
|
|
|
// DiffractionGeometry::CalcAzIntPolarizationCorr about the centre the rings are drawn about.
|
|
__device__ float polarization_factor(const ShadowMaskSetup &s, float x, float y) {
|
|
const float u = (x - s.beam_x) * s.pixel_size_mm;
|
|
const float v = (y - s.beam_y) * s.pixel_size_mm;
|
|
const float *m = s.det_matrix;
|
|
const float lx = m[0] * u + m[1] * v + m[2] * s.distance_mm;
|
|
const float ly = m[3] * u + m[4] * v + m[5] * s.distance_mm;
|
|
const float lz = m[6] * u + m[7] * v + m[8] * s.distance_mm;
|
|
const float two_theta = atan2f(sqrtf(lx * lx + ly * ly), lz);
|
|
float phi = atan2f(ly, lx);
|
|
if (phi < 0)
|
|
phi += 2.0f * PI;
|
|
const float cos_2theta = cosf(two_theta);
|
|
const float cos_2theta_2 = cos_2theta * cos_2theta;
|
|
const float cos_2phi = cosf(2.0f * phi);
|
|
return 0.5f * (1.0f + cos_2theta_2 - s.polarization * cos_2phi * (1.0f - cos_2theta_2));
|
|
}
|
|
|
|
__global__ void setup_kernel(ShadowMaskSetup s, const uint32_t *__restrict__ pixel_mask,
|
|
const int64_t *__restrict__ sum_value, const uint32_t *__restrict__ valid_count,
|
|
float *__restrict__ pol, char *__restrict__ valid, int *__restrict__ radius,
|
|
double *__restrict__ num, int32_t *__restrict__ den, int *__restrict__ max_radius) {
|
|
const size_t n = static_cast<size_t>(s.width) * s.height;
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n)
|
|
return;
|
|
const int x = static_cast<int>(i % s.width), y = static_cast<int>(i / s.width);
|
|
const float dx = x - s.beam_x, dy = y - s.beam_y;
|
|
const float p = s.has_polarization ? polarization_factor(s, static_cast<float>(x), static_cast<float>(y)) : 1.0f;
|
|
pol[i] = p;
|
|
float mean = 0.0f;
|
|
char v = 0;
|
|
if (valid_count[i] > 0 && pixel_mask[i] == 0 && p > 0.0f) {
|
|
mean = static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i] / p);
|
|
v = 1;
|
|
}
|
|
valid[i] = v;
|
|
num[i] = v ? mean : 0.0;
|
|
den[i] = v ? 1 : 0;
|
|
const int r = static_cast<int>(lroundf(sqrtf(dx * dx + dy * dy)));
|
|
radius[i] = r;
|
|
atomicMax(max_radius, r);
|
|
}
|
|
|
|
// Sum over the k x k box about each pixel, zero outside the frame: one running sum per row, then one
|
|
// per column, each with exactly the terms and order of the host's (box_sum in ShadowFinder.cpp).
|
|
template <typename T>
|
|
__global__ void box_rows(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) {
|
|
const int y = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (y >= H) return;
|
|
const T *src = in + static_cast<size_t>(y) * W;
|
|
T *dst = out + static_cast<size_t>(y) * W;
|
|
T s = 0;
|
|
for (int x = 0; x <= min(half, W - 1); x++)
|
|
s += src[x];
|
|
for (int x = 0; x < W; x++) {
|
|
dst[x] = s;
|
|
if (x + half + 1 < W) s += src[x + half + 1];
|
|
if (x - half >= 0) s -= src[x - half];
|
|
}
|
|
}
|
|
|
|
template <typename T>
|
|
__global__ void box_columns(const T *__restrict__ in, T *__restrict__ out, int W, int H, int half) {
|
|
const int x = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (x >= W) return;
|
|
T s = 0;
|
|
for (int y = 0; y <= min(half, H - 1); y++)
|
|
s += in[static_cast<size_t>(y) * W + x];
|
|
for (int y = 0; y < H; y++) {
|
|
out[static_cast<size_t>(y) * W + x] = s;
|
|
if (y + half + 1 < H) s += in[static_cast<size_t>(y + half + 1) * W + x];
|
|
if (y - half >= 0) s -= in[static_cast<size_t>(y - half) * W + x];
|
|
}
|
|
}
|
|
|
|
// Dilation of a 0/1 plane by the (2r+1) square clipped to the frame, as a count over a sliding window
|
|
// along rows and then along columns (dilate in ShadowFinder.cpp).
|
|
__global__ void dilate_rows(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) {
|
|
const int y = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (y >= H) return;
|
|
const char *src = in + static_cast<size_t>(y) * W;
|
|
char *dst = out + static_cast<size_t>(y) * W;
|
|
int count = 0;
|
|
for (int x = 0; x <= min(r, W - 1); x++)
|
|
count += src[x];
|
|
for (int x = 0; x < W; x++) {
|
|
dst[x] = count > 0;
|
|
if (x + r + 1 < W) count += src[x + r + 1];
|
|
if (x - r >= 0) count -= src[x - r];
|
|
}
|
|
}
|
|
|
|
__global__ void dilate_columns(const char *__restrict__ in, char *__restrict__ out, int W, int H, int r) {
|
|
const int x = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (x >= W) return;
|
|
int count = 0;
|
|
for (int y = 0; y <= min(r, H - 1); y++)
|
|
count += in[static_cast<size_t>(y) * W + x];
|
|
for (int y = 0; y < H; y++) {
|
|
out[static_cast<size_t>(y) * W + x] = count > 0;
|
|
if (y + r + 1 < H) count += in[static_cast<size_t>(y + r + 1) * W + x];
|
|
if (y - r >= 0) count -= in[static_cast<size_t>(y - r) * W + x];
|
|
}
|
|
}
|
|
|
|
__global__ void pooled_kernel(size_t n, const double *__restrict__ pooled_sum, const int32_t *__restrict__ pooled_count,
|
|
const char *__restrict__ valid, const int *__restrict__ radius,
|
|
float *__restrict__ pooled, uint64_t *__restrict__ ring_key) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
const float p = pooled_count[i] > 0 ? static_cast<float>(pooled_sum[i] / pooled_count[i]) : 0.0f;
|
|
pooled[i] = p;
|
|
ring_key[i] = valid[i] ? (static_cast<uint64_t>(radius[i]) << 32) | float_key(p) : UINT64_MAX;
|
|
}
|
|
|
|
// Where each of the keys' leading 32-bit groups (ring or sector) starts in a sorted key array.
|
|
__global__ void group_offsets(const uint64_t *__restrict__ keys, size_t n, int groups, int *__restrict__ offset) {
|
|
const int g = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (g > groups) return;
|
|
offset[g] = static_cast<int>(lower_bound(keys, n, static_cast<uint64_t>(g) << 32));
|
|
}
|
|
|
|
// The ring's baseline, three iterations of an order statistic over its sorted values (GetMask).
|
|
__global__ void baseline_kernel(int rings, const uint64_t *__restrict__ keys, const int *__restrict__ offset,
|
|
float *__restrict__ baseline) {
|
|
const int r = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (r >= rings) return;
|
|
const int lo = offset[r], n = offset[r + 1] - offset[r];
|
|
int excluded = 0;
|
|
float b = 0.0f;
|
|
for (int iter = 0; iter < 3; iter++) {
|
|
const int avail = n - excluded;
|
|
b = (avail <= 0) ? 0.0f : key_float(static_cast<uint32_t>(keys[lo + excluded + avail / 2]));
|
|
const float d = fmaxf(b, 1e-6f);
|
|
int excl = 0;
|
|
while (excl < n && key_float(static_cast<uint32_t>(keys[lo + excl])) / d < SHADOW_RATIO)
|
|
excl++;
|
|
excluded = excl;
|
|
}
|
|
baseline[r] = b;
|
|
}
|
|
|
|
__global__ void low_kernel(size_t n, uint32_t frames, const char *__restrict__ valid, const float *__restrict__ pooled,
|
|
const int32_t *__restrict__ pooled_count, const float *__restrict__ pol,
|
|
const int *__restrict__ radius, const float *__restrict__ baseline,
|
|
float *__restrict__ ratio, float *__restrict__ deficit, char *__restrict__ low) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
const float base = baseline[radius[i]];
|
|
const float rt = valid[i] ? pooled[i] / fmaxf(base, 1e-6f) : 1.0f;
|
|
ratio[i] = rt;
|
|
float df = 0.0f;
|
|
char l = 0;
|
|
if (valid[i]) {
|
|
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
|
|
df = static_cast<float>(poisson_deficit_sigma(pooled[i] * counted, base * counted));
|
|
l = rt < SHADOW_RATIO && df > MIN_DEFICIT_SIGMA;
|
|
}
|
|
deficit[i] = df;
|
|
low[i] = l;
|
|
}
|
|
|
|
// 8-connected components by union-find: every component ends up named by its smallest pixel index,
|
|
// whatever order the unions ran in.
|
|
__device__ int find_root(const int *parent, int x) {
|
|
while (parent[x] != x)
|
|
x = parent[x];
|
|
return x;
|
|
}
|
|
|
|
__global__ void cc_init(size_t n, const char *__restrict__ member, int *__restrict__ parent) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
parent[i] = member[i] ? static_cast<int>(i) : -1;
|
|
}
|
|
|
|
__device__ void cc_unite(int *parent, int a, int b) {
|
|
while (true) {
|
|
a = find_root(parent, a);
|
|
b = find_root(parent, b);
|
|
if (a == b) return;
|
|
if (a < b) { const int t = a; a = b; b = t; }
|
|
if (atomicCAS(&parent[a], a, b) == a) return;
|
|
}
|
|
}
|
|
|
|
__global__ void cc_union(int W, int H, const char *__restrict__ member, int *parent) {
|
|
const size_t n = static_cast<size_t>(W) * H;
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n || !member[i]) return;
|
|
const int x = static_cast<int>(i % W), y = static_cast<int>(i / W);
|
|
// The four neighbours before this pixel; the other four see it from their side.
|
|
if (x > 0 && member[i - 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(i - 1));
|
|
if (y > 0) {
|
|
const size_t up = i - W;
|
|
if (member[up]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up));
|
|
if (x > 0 && member[up - 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up - 1));
|
|
if (x + 1 < W && member[up + 1]) cc_unite(parent, static_cast<int>(i), static_cast<int>(up + 1));
|
|
}
|
|
}
|
|
|
|
__global__ void cc_flatten(size_t n, int *parent) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n || parent[i] < 0) return;
|
|
parent[i] = find_root(parent, static_cast<int>(i));
|
|
}
|
|
|
|
// Per component (by root): how many of its pixels are `a`, and how many are both `a` and `b`.
|
|
__global__ void cc_count(size_t n, const int *__restrict__ root, const char *__restrict__ a, const char *__restrict__ b,
|
|
int *__restrict__ count_a, int *__restrict__ count_ab) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n || root[i] < 0 || !a[i]) return;
|
|
atomicAdd(&count_a[root[i]], 1);
|
|
if (count_ab && b[i])
|
|
atomicAdd(&count_ab[root[i]], 1);
|
|
}
|
|
|
|
__global__ void region_kernel(size_t n, const int *__restrict__ root, const int *__restrict__ n_low,
|
|
const char *__restrict__ low, const char *__restrict__ valid, const int *__restrict__ radius,
|
|
int blocked_out_to, char *__restrict__ region) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
char r = root[i] >= 0 && n_low[root[i]] >= MIN_SHADOW_PIXELS ? low[i] : 0;
|
|
if (valid[i] && radius[i] <= blocked_out_to)
|
|
r = 1;
|
|
region[i] = r;
|
|
}
|
|
|
|
__global__ void lit_kernel(size_t n, const uint32_t *__restrict__ valid_count, const int64_t *__restrict__ max_value,
|
|
char *__restrict__ lit) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
lit[i] = (valid_count[i] > 0) && (max_value[i] >= MIN_REFLECTION);
|
|
}
|
|
|
|
__global__ void reflection_kernel(int W, int H, const char *__restrict__ lit, char *__restrict__ reflection) {
|
|
const size_t n = static_cast<size_t>(W) * H;
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
const int x = static_cast<int>(i % W), y = static_cast<int>(i / W);
|
|
char r = 0;
|
|
if (lit[i]) {
|
|
int neighbours = 0;
|
|
for (int dy = -1; dy <= 1; dy++)
|
|
for (int dx = -1; dx <= 1; dx++) {
|
|
const int yy = y + dy, xx = x + dx;
|
|
if ((dx || dy) && yy >= 0 && yy < H && xx >= 0 && xx < W && lit[static_cast<size_t>(yy) * W + xx])
|
|
neighbours++;
|
|
}
|
|
r = neighbours >= 2;
|
|
}
|
|
reflection[i] = r;
|
|
}
|
|
|
|
__global__ void penumbra_kernel(size_t n, const char *__restrict__ penumbra, const char *__restrict__ valid,
|
|
const float *__restrict__ ratio, const float *__restrict__ deficit, char *__restrict__ region) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
if (penumbra[i] && valid[i] && ratio[i] < PENUMBRA_RATIO && deficit[i] > MIN_DEFICIT_SIGMA)
|
|
region[i] = 1;
|
|
}
|
|
|
|
__global__ void invert_kernel(size_t n, const char *__restrict__ in, char *__restrict__ out) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
out[i] = !in[i];
|
|
}
|
|
|
|
// Background components that touch the frame's edge are outside; the rest are holes.
|
|
__global__ void outside_kernel(int W, int H, const int *__restrict__ root, int *__restrict__ outside) {
|
|
const int k = blockIdx.x * blockDim.x + threadIdx.x;
|
|
const int perimeter = 2 * W + 2 * H;
|
|
if (k >= perimeter) return;
|
|
int x, y;
|
|
if (k < W) { x = k; y = 0; }
|
|
else if (k < 2 * W) { x = k - W; y = H - 1; }
|
|
else if (k < 2 * W + H) { x = 0; y = k - 2 * W; }
|
|
else { x = W - 1; y = k - 2 * W - H; }
|
|
const int r = root[static_cast<size_t>(y) * W + x];
|
|
if (r >= 0) outside[r] = 1;
|
|
}
|
|
|
|
__global__ void final_kernel(size_t n, const int *__restrict__ background_root, const int *__restrict__ outside,
|
|
const char *__restrict__ reflection_grown, char *__restrict__ region,
|
|
uint32_t *__restrict__ mask) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
char r = region[i];
|
|
if (background_root[i] >= 0 && !outside[background_root[i]])
|
|
r = 1; // a hole
|
|
if (reflection_grown[i])
|
|
r = 0;
|
|
region[i] = r;
|
|
mask[i] = r ? MASK_SHADOW : 0;
|
|
}
|
|
|
|
__global__ void sector_key_kernel(ShadowMaskSetup s, const char *__restrict__ valid, const char *__restrict__ region,
|
|
const int *__restrict__ radius, const float *__restrict__ ratio,
|
|
uint64_t *__restrict__ key) {
|
|
const size_t n = static_cast<size_t>(s.width) * s.height;
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
if (!valid[i] || region[i]) {
|
|
key[i] = UINT64_MAX;
|
|
return;
|
|
}
|
|
const float dx = static_cast<float>(i % s.width) - s.beam_x, dy = static_cast<float>(i / s.width) - s.beam_y;
|
|
const double phi = atan2f(dy, dx) + PI;
|
|
const int sector = min(HARMONIC_SECTORS - 1, static_cast<int>(phi / (2.0 * PI) * HARMONIC_SECTORS));
|
|
const uint64_t k = static_cast<uint64_t>(radius[i] / HARMONIC_BAND_PX) * HARMONIC_SECTORS + sector;
|
|
key[i] = (k << 32) | float_key(ratio[i]);
|
|
}
|
|
|
|
__global__ void sector_median_kernel(int sectors, const uint64_t *__restrict__ keys, const int *__restrict__ offset,
|
|
double *__restrict__ median) {
|
|
const int k = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (k >= sectors) return;
|
|
const int n = offset[k + 1] - offset[k];
|
|
median[k] = n >= MIN_SECTOR_PIXELS ? key_float(static_cast<uint32_t>(keys[offset[k] + n / 2])) : -1.0;
|
|
}
|
|
|
|
__global__ void dim_kernel(ShadowMaskSetup s, uint32_t frames, const char *__restrict__ valid,
|
|
const char *__restrict__ region, const float *__restrict__ ratio,
|
|
const float *__restrict__ deficit, const int *__restrict__ radius,
|
|
const float *__restrict__ harm_c, const float *__restrict__ harm_s,
|
|
const int32_t *__restrict__ pooled_count, const float *__restrict__ pol,
|
|
const float *__restrict__ pooled, const float *__restrict__ baseline,
|
|
char *__restrict__ dim) {
|
|
const size_t n = static_cast<size_t>(s.width) * s.height;
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
char d = 0;
|
|
if (valid[i] && !region[i] && ratio[i] < PENUMBRA_RATIO) {
|
|
// cos 2phi and sin 2phi from the offset to the beam.
|
|
const float dx = static_cast<float>(i % s.width) - s.beam_x, dy = static_cast<float>(i / s.width) - s.beam_y;
|
|
const float r2 = fmaxf(dx * dx + dy * dy, 1e-6f);
|
|
const int band = radius[i] / HARMONIC_BAND_PX;
|
|
const float model = fminf(1.0f, 1.0f + harm_c[band] * (dx * dx - dy * dy) / r2
|
|
+ harm_s[band] * 2.0f * dx * dy / r2);
|
|
if (model >= 1.0f) {
|
|
d = deficit[i] > MIN_DEFICIT_SIGMA;
|
|
} else {
|
|
const double counted = static_cast<double>(frames) * pooled_count[i] * pol[i];
|
|
d = ratio[i] < PENUMBRA_RATIO * model
|
|
&& poisson_deficit_sigma(pooled[i] * counted, baseline[radius[i]] * model * counted) > MIN_DEFICIT_SIGMA;
|
|
}
|
|
}
|
|
dim[i] = d;
|
|
}
|
|
|
|
// Join a region across the module gaps it crosses, one line per thread (bridge_gaps in
|
|
// ShadowFinder.cpp). Both directions read `region` and only ever set pixels of `out` to 1.
|
|
__global__ void bridge_rows(int W, int H, const char *__restrict__ region, const char *__restrict__ valid,
|
|
char *out) {
|
|
const int y = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (y >= H) return;
|
|
const size_t row = static_cast<size_t>(y) * W;
|
|
int k = 0;
|
|
while (k < W) {
|
|
if (valid[row + k]) { k++; continue; }
|
|
const int start = k;
|
|
while (k < W && !valid[row + k]) k++;
|
|
if (start > 0 && k < W && region[row + start - 1] && region[row + k])
|
|
for (int j = start; j < k; j++) out[row + j] = 1;
|
|
}
|
|
}
|
|
|
|
__global__ void bridge_columns(int W, int H, const char *__restrict__ region, const char *__restrict__ valid,
|
|
char *out) {
|
|
const int x = blockIdx.x * blockDim.x + threadIdx.x;
|
|
if (x >= W) return;
|
|
const auto at = [&](int y) { return static_cast<size_t>(y) * W + x; };
|
|
int k = 0;
|
|
while (k < H) {
|
|
if (valid[at(k)]) { k++; continue; }
|
|
const int start = k;
|
|
while (k < H && !valid[at(k)]) k++;
|
|
if (start > 0 && k < H && region[at(start - 1)] && region[at(k)])
|
|
for (int j = start; j < k; j++) out[at(j)] = 1;
|
|
}
|
|
}
|
|
|
|
__global__ void transmitting_kernel(size_t n, const int *__restrict__ root, const char *__restrict__ dim,
|
|
const int *__restrict__ n_dim, const int *__restrict__ n_low,
|
|
uint32_t *__restrict__ mask) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
const int c = root[i];
|
|
if (c >= 0 && dim[i] && n_dim[c] >= MIN_SHADOW_PIXELS && n_low[c] >= MIN_CORE_PIXELS)
|
|
mask[i] = MASK_TRANSMITTING;
|
|
}
|
|
|
|
__global__ void mean_kernel(size_t n, const uint32_t *__restrict__ pixel_mask, const int64_t *__restrict__ sum_value,
|
|
const uint32_t *__restrict__ valid_count, float *__restrict__ mean) {
|
|
const size_t i = blockIdx.x * static_cast<size_t>(blockDim.x) + threadIdx.x;
|
|
if (i >= n) return;
|
|
mean[i] = valid_count[i] > 0 && pixel_mask[i] == 0
|
|
? static_cast<float>(static_cast<double>(sum_value[i]) / valid_count[i]) : NAN;
|
|
}
|
|
|
|
// The device side of one GetMask: planes, sort scratch and the stream to run on.
|
|
class MaskEngine {
|
|
public:
|
|
const ShadowMaskSetup s;
|
|
const int W, H;
|
|
const size_t n;
|
|
cudaStream_t stream;
|
|
|
|
MaskEngine(const ShadowMaskSetup &setup, cudaStream_t st)
|
|
: s(setup), W(setup.width), H(setup.height), n(static_cast<size_t>(setup.width) * setup.height), stream(st) {}
|
|
|
|
void Check(const char *what) const {
|
|
check(cudaGetLastError(), what);
|
|
}
|
|
|
|
void Dilate(const char *in, char *out, char *scratch, int r) const {
|
|
dilate_rows<<<grid(H), THREADS, 0, stream>>>(in, scratch, W, H, r);
|
|
dilate_columns<<<grid(W), THREADS, 0, stream>>>(scratch, out, W, H, r);
|
|
Check("dilate");
|
|
}
|
|
|
|
// Components of `member`, each pixel's root in `root` (-1 outside every component).
|
|
void Label(const char *member, int *root) const {
|
|
cc_init<<<grid(n), THREADS, 0, stream>>>(n, member, root);
|
|
cc_union<<<grid(n), THREADS, 0, stream>>>(W, H, member, root);
|
|
cc_flatten<<<grid(n), THREADS, 0, stream>>>(n, root);
|
|
Check("components");
|
|
}
|
|
|
|
void SortKeys(uint64_t *keys, uint64_t *sorted) const {
|
|
size_t bytes = 0;
|
|
check(cub::DeviceRadixSort::SortKeys(nullptr, bytes, keys, sorted, n, 0, 64, stream), "sort size");
|
|
CudaDevicePtr<uint8_t> scratch(bytes);
|
|
check(cub::DeviceRadixSort::SortKeys(scratch.get(), bytes, keys, sorted, n, 0, 64, stream), "sort");
|
|
// The scratch is freed on the allocation stream, which knows nothing of this one.
|
|
check(cudaStreamSynchronize(stream), "sort");
|
|
}
|
|
|
|
template <typename T>
|
|
std::vector<T> Download(const T *device, size_t count) const {
|
|
std::vector<T> host(count);
|
|
check(cudaMemcpyAsync(host.data(), device, count * sizeof(T), cudaMemcpyDeviceToHost, stream), "download");
|
|
check(cudaStreamSynchronize(stream), "download");
|
|
return host;
|
|
}
|
|
|
|
template <typename T>
|
|
void Upload(T *device, const std::vector<T> &host) const {
|
|
check(cudaMemcpyAsync(device, host.data(), host.size() * sizeof(T), cudaMemcpyHostToDevice, stream), "upload");
|
|
}
|
|
};
|
|
|
|
} // namespace
|
|
|
|
std::vector<uint32_t> ShadowMaskOnDevice(const ShadowMaskSetup &setup, const std::vector<uint32_t> &pixel_mask,
|
|
const int64_t *max_value, const int64_t *sum_value,
|
|
const uint32_t *valid_count, uint32_t frames, cudaStream_t stream) {
|
|
const MaskEngine e(setup, stream);
|
|
const size_t n = e.n;
|
|
const int W = e.W, H = e.H;
|
|
|
|
CudaDevicePtr<uint32_t> d_pixel_mask(n), mask(n);
|
|
e.Upload(d_pixel_mask.get(), pixel_mask);
|
|
|
|
// Mean projection over the polarization factor, usable pixels and radius from the beam centre.
|
|
CudaDevicePtr<float> pol(n), pooled(n), ratio(n), deficit(n);
|
|
CudaDevicePtr<char> valid(n), low(n), region(n), a(n), b(n), c(n);
|
|
CudaDevicePtr<int> radius(n), root(n), count_a(n), count_b(n), max_radius(1);
|
|
CudaDevicePtr<double> num(n), dsum(n);
|
|
CudaDevicePtr<int32_t> den(n), dcount(n);
|
|
check(cudaMemsetAsync(max_radius.get(), 0, sizeof(int), stream), "memset");
|
|
setup_kernel<<<grid(n), THREADS, 0, stream>>>(setup, d_pixel_mask, sum_value, valid_count, pol, valid, radius,
|
|
num, den, max_radius);
|
|
e.Check("setup");
|
|
const int max_r = e.Download(max_radius.get(), 1)[0];
|
|
const int rings = max_r + 1;
|
|
|
|
// The background pooled over a small box.
|
|
box_rows<double><<<grid(H), THREADS, 0, stream>>>(num, dsum, W, H, POOL_PX / 2);
|
|
box_columns<double><<<grid(W), THREADS, 0, stream>>>(dsum, num, W, H, POOL_PX / 2);
|
|
box_rows<int32_t><<<grid(H), THREADS, 0, stream>>>(den, dcount, W, H, POOL_PX / 2);
|
|
box_columns<int32_t><<<grid(W), THREADS, 0, stream>>>(dcount, den, W, H, POOL_PX / 2);
|
|
e.Check("pooling");
|
|
const double *pooled_sum = num;
|
|
const int32_t *pooled_count = den;
|
|
|
|
// The rings, each sorted once; the baseline is an order statistic of them.
|
|
CudaDevicePtr<uint64_t> keys(n), sorted(n);
|
|
pooled_kernel<<<grid(n), THREADS, 0, stream>>>(n, pooled_sum, pooled_count, valid, radius, pooled, keys);
|
|
e.Check("pooled");
|
|
e.SortKeys(keys, sorted);
|
|
CudaDevicePtr<int> ring_offset(rings + 1);
|
|
group_offsets<<<grid(rings + 1), THREADS, 0, stream>>>(sorted, n, rings, ring_offset);
|
|
CudaDevicePtr<float> baseline(rings);
|
|
baseline_kernel<<<grid(rings), THREADS, 0, stream>>>(rings, sorted, ring_offset, baseline);
|
|
e.Check("baseline");
|
|
const auto host_baseline = e.Download(baseline.get(), rings);
|
|
const auto offsets = e.Download(ring_offset.get(), rings + 1);
|
|
std::vector<int> ring_pixels(rings);
|
|
for (int r = 0; r < rings; r++)
|
|
ring_pixels[r] = offsets[r + 1] - offsets[r];
|
|
const int blocked_out_to = BlockedOutTo(host_baseline, ring_pixels);
|
|
|
|
// Low pixels, and the regions of them large enough to be hardware.
|
|
low_kernel<<<grid(n), THREADS, 0, stream>>>(n, frames, valid, pooled, pooled_count, pol, radius, baseline,
|
|
ratio, deficit, low);
|
|
e.Check("low");
|
|
e.Dilate(low, a, c, BRIDGE_PX);
|
|
e.Label(a, root);
|
|
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
|
|
cc_count<<<grid(n), THREADS, 0, stream>>>(n, root, low, low, count_a, nullptr);
|
|
region_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, count_a, low, valid, radius, blocked_out_to, region);
|
|
e.Check("region");
|
|
|
|
// Recorded reflections; `b` holds them until they are given back at the end.
|
|
lit_kernel<<<grid(n), THREADS, 0, stream>>>(n, valid_count, max_value, a);
|
|
reflection_kernel<<<grid(n), THREADS, 0, stream>>>(W, H, a, b);
|
|
e.Check("reflections");
|
|
|
|
// Penumbra, round and fill.
|
|
e.Dilate(region, a, c, PENUMBRA_MAX_PX);
|
|
penumbra_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, valid, ratio, deficit, region);
|
|
e.Dilate(region, a, c, 2);
|
|
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, region);
|
|
e.Dilate(region, a, c, 2);
|
|
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, a, region); // region = erode(dilate(region))
|
|
invert_kernel<<<grid(n), THREADS, 0, stream>>>(n, region, a); // the background
|
|
e.Label(a, root);
|
|
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
|
|
outside_kernel<<<grid(2 * W + 2 * H), THREADS, 0, stream>>>(W, H, root, count_a);
|
|
e.Dilate(b, c, a, 1); // the reflections, grown by one
|
|
final_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, count_a, c, region, mask);
|
|
e.Check("fill");
|
|
|
|
// The arm search: sector medians of the ratio, the harmonic of each band, the dim pixels.
|
|
const int n_bands = max_r / HARMONIC_BAND_PX + 1;
|
|
const int n_sectors = n_bands * HARMONIC_SECTORS;
|
|
sector_key_kernel<<<grid(n), THREADS, 0, stream>>>(setup, valid, region, radius, ratio, keys);
|
|
e.Check("sectors");
|
|
e.SortKeys(keys, sorted);
|
|
CudaDevicePtr<int> sector_offset(n_sectors + 1);
|
|
group_offsets<<<grid(n_sectors + 1), THREADS, 0, stream>>>(sorted, n, n_sectors, sector_offset);
|
|
CudaDevicePtr<double> sector_median(n_sectors);
|
|
sector_median_kernel<<<grid(n_sectors), THREADS, 0, stream>>>(n_sectors, sorted, sector_offset, sector_median);
|
|
e.Check("sector medians");
|
|
std::vector<float> harm_c, harm_s;
|
|
HarmonicFit(e.Download(sector_median.get(), n_sectors), n_bands, harm_c, harm_s);
|
|
CudaDevicePtr<float> d_harm_c(n_bands), d_harm_s(n_bands);
|
|
e.Upload(d_harm_c.get(), harm_c);
|
|
e.Upload(d_harm_s.get(), harm_s);
|
|
dim_kernel<<<grid(n), THREADS, 0, stream>>>(setup, frames, valid, region, ratio, deficit, radius, d_harm_c, d_harm_s,
|
|
pooled_count, pol, pooled, baseline, a);
|
|
e.Check("dim");
|
|
e.Dilate(a, b, c, BRIDGE_PX);
|
|
check(cudaMemcpyAsync(c.get(), b.get(), n, cudaMemcpyDeviceToDevice, stream), "copy");
|
|
bridge_rows<<<grid(H), THREADS, 0, stream>>>(W, H, b, valid, c);
|
|
bridge_columns<<<grid(W), THREADS, 0, stream>>>(W, H, b, valid, c);
|
|
e.Label(c, root);
|
|
check(cudaMemsetAsync(count_a.get(), 0, n * sizeof(int), stream), "memset");
|
|
check(cudaMemsetAsync(count_b.get(), 0, n * sizeof(int), stream), "memset");
|
|
cc_count<<<grid(n), THREADS, 0, stream>>>(n, root, a, low, count_a, count_b);
|
|
transmitting_kernel<<<grid(n), THREADS, 0, stream>>>(n, root, a, count_a, count_b, mask);
|
|
e.Check("transmitting");
|
|
|
|
return e.Download(mask.get(), n);
|
|
}
|
|
|
|
std::vector<float> MeanProjectionOnDevice(const std::vector<uint32_t> &pixel_mask, const int64_t *sum_value,
|
|
const uint32_t *valid_count, size_t npixels, cudaStream_t stream) {
|
|
CudaDevicePtr<uint32_t> d_pixel_mask(npixels);
|
|
CudaDevicePtr<float> mean(npixels);
|
|
check(cudaMemcpyAsync(d_pixel_mask.get(), pixel_mask.data(), npixels * sizeof(uint32_t), cudaMemcpyHostToDevice,
|
|
stream), "upload");
|
|
mean_kernel<<<grid(npixels), THREADS, 0, stream>>>(npixels, d_pixel_mask, sum_value, valid_count, mean);
|
|
check(cudaGetLastError(), "mean");
|
|
std::vector<float> host(npixels);
|
|
check(cudaMemcpyAsync(host.data(), mean.get(), npixels * sizeof(float), cudaMemcpyDeviceToHost, stream), "download");
|
|
check(cudaStreamSynchronize(stream), "mean");
|
|
return host;
|
|
}
|