// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #include #include "../common/CUDAWrapper.h" #ifdef JFJOCH_USE_CUDA #include #include #include #include #include "../common/ParallelFor.h" #include "../image_analysis/scale_merge/RotationScaleMergeGPU.h" namespace { using Term = RotationScaleMergeGPU::SurfaceTerm; // The roundings RotationScaleMerge::ApplyCellSurface makes on the host (x86-64-v3 build), spelled out: // a volatile result is rounded on its own and never fused into the next operation, std::fma is fused. double Mul(double a, double b) { volatile double r = a * b; return r; } double Add(double a, double b) { volatile double r = a + b; return r; } constexpr int SURFACE_BLOCK = 32768; // ApplyCellSurface's reduction block struct Surface { int n_groups = 0, ncell = 0; std::vector term; std::vector parity; std::vector gperm, gstart; std::vector sel[3]; // even, odd, all - in term order }; Surface MakeSurface(int n_terms, int n_groups, int ncell, uint32_t seed) { std::mt19937 rng(seed); std::uniform_int_distribution group(0, n_groups - 1), cell(0, ncell - 1), bit(0, 1); std::uniform_real_distribution u(0.0f, 1.0f); Surface s; s.n_groups = n_groups; s.ncell = ncell; for (int i = 0; i < n_terms; ++i) { // Negative intensities, and now and then a sigma of zero: both reach the host sums as they are. const float I = 1000.0f * u(rng) - 100.0f; const float sigma = (i % 997 == 0) ? 0.0f : 1.0f + 30.0f * u(rng); // Every 50th group gets no terms at all, so its reference is empty. int g = group(rng); if (g % 50 == 0) g = (g + 1) % n_groups; s.term.push_back({I, sigma, 0.5f + u(rng), 1.0f + 3.0f * u(rng), cell(rng), g}); s.parity.push_back(static_cast(bit(rng))); s.sel[s.parity.back()].push_back(i); s.sel[2].push_back(i); } s.gstart.assign(n_groups + 1, 0); for (const Term &t : s.term) ++s.gstart[t.group + 1]; for (int g = 0; g < n_groups; ++g) s.gstart[g + 1] += s.gstart[g]; s.gperm.resize(n_terms); std::vector fill(s.gstart.begin(), s.gstart.end() - 1); for (int i = 0; i < n_terms; ++i) s.gperm[fill[s.term[i].group]++] = i; return s; } void HostReference(const Surface &s, int parity, const std::vector &A, std::vector &sw, std::vector &swI) { sw.assign(s.n_groups, 0.0); swI.assign(s.n_groups, 0.0); for (int g = 0; g < s.n_groups; ++g) { double s_w = 0.0, s_wI = 0.0; for (int k = s.gstart[g]; k < s.gstart[g + 1]; ++k) { const int i = s.gperm[k]; if (parity >= 0 && s.parity[i] != parity) continue; const Term &t = s.term[i]; const double a = A[t.cell]; const double Is = Mul(Mul(t.I, t.corr), a), sc = Mul(Mul(t.sigma, t.corr), a); const double w = 1.0 / Mul(sc, sc); s_w = Add(s_w, w); s_wI = parity >= 0 ? std::fma(Is, w, s_wI) : Add(s_wI, Mul(Is, w)); } sw[g] = s_w; swI[g] = s_wI; } } void HostFitSums(const Surface &s, const std::vector &sel, const std::vector &A, const std::vector &sw, const std::vector &swI, std::vector &cross, std::vector &ref2) { cross.assign(s.ncell, 0.0); ref2.assign(s.ncell, 0.0); const int n = static_cast(sel.size()), nb = ReductionBlocks(n, SURFACE_BLOCK); for (int b = 0; b < nb; ++b) { std::vector xcross(s.ncell, 0.0), xref2(s.ncell, 0.0); const int lo = static_cast(int64_t(n) * b / nb), hi = static_cast(int64_t(n) * (b + 1) / nb); for (int k = lo; k < hi; ++k) { const Term &t = s.term[sel[k]]; if (sw[t.group] <= 0.0) continue; const double Iref = swI[t.group] / sw[t.group], a = A[t.cell]; const double Is = Mul(Mul(t.I, t.corr), a), sc = Mul(Mul(t.sigma, t.corr), a); if (!std::isfinite(Iref) || Iref <= 0.0 || !(sc > 0.0)) continue; const double w = 1.0 / Mul(sc, sc); xcross[t.cell] = std::fma(Mul(w, Is), Iref, xcross[t.cell]); xref2[t.cell] = std::fma(Mul(w, Iref), Iref, xref2[t.cell]); } for (int c = 0; c < s.ncell; ++c) { cross[c] = Add(cross[c], xcross[c]); ref2[c] = Add(ref2[c], xref2[c]); } } } // The device side of one subset: ApplyCellSurface's per-block counting sort by cell. void UploadSubset(RotationScaleMergeGPU &gpu, const Surface &s, int id, const std::vector &sel) { const int n = static_cast(sel.size()), nb = ReductionBlocks(n, SURFACE_BLOCK); std::vector perm(n), seg_start(static_cast(nb) * s.ncell + 1, n); for (int b = 0; b < nb; ++b) { const int lo = static_cast(int64_t(n) * b / nb), hi = static_cast(int64_t(n) * (b + 1) / nb); std::vector pos(s.ncell + 1, 0); for (int k = lo; k < hi; ++k) ++pos[s.term[sel[k]].cell + 1]; for (int c = 0; c < s.ncell; ++c) pos[c + 1] += pos[c]; for (int c = 0; c < s.ncell; ++c) seg_start[size_t(b) * s.ncell + c] = lo + pos[c]; for (int k = lo; k < hi; ++k) perm[lo + pos[s.term[sel[k]].cell]++] = sel[k]; } gpu.SurfaceSetSubset(id, nb, perm.data(), seg_start.data()); } bool SameBits(const std::vector &a, const std::vector &b) { return a.size() == b.size() && std::memcmp(a.data(), b.data(), a.size() * sizeof(double)) == 0; } } // namespace TEST_CASE("CorrectionSurfaceGPU_SumsMatchHostBitForBit", "[RotationScale][gpu]") { if (get_gpu_count() == 0) SKIP("No GPU"); // Enough terms for several reduction blocks in every subset. const Surface s = MakeSurface(250000, 4000, 144, 7); RotationScaleMergeGPU gpu; REQUIRE(gpu.Available()); gpu.SurfaceSetTerms(static_cast(s.term.size()), s.term.data(), s.parity.data(), s.n_groups, s.gperm.data(), s.gstart.data(), s.ncell); for (int id = 0; id < 3; ++id) UploadSubset(gpu, s, id, s.sel[id]); std::mt19937 rng(11); std::uniform_real_distribution u(0.7, 1.4); std::vector A(s.ncell); for (double &a : A) a = u(rng); for (int parity : {0, 1, -1}) { const int id = parity < 0 ? 2 : parity; std::vector sw, swI, cross, ref2; HostReference(s, parity, A, sw, swI); HostFitSums(s, s.sel[id], A, sw, swI, cross, ref2); REQUIRE(ReductionBlocks(static_cast(s.sel[id].size()), SURFACE_BLOCK) > 1); gpu.SurfaceReference(parity, A.data()); std::vector dsw(s.n_groups), dswI(s.n_groups), dcross(s.ncell), dref2(s.ncell); gpu.SurfaceGetReference(dsw.data(), dswI.data()); gpu.SurfaceFitSums(id, dcross.data(), dref2.data()); CHECK(SameBits(sw, dsw)); CHECK(SameBits(swI, dswI)); CHECK(SameBits(cross, dcross)); CHECK(SameBits(ref2, dref2)); } } #endif