Files
Jungfraujoch/image_analysis/scale_merge/HKLKey.cpp
T
leonarski_fandClaude Opus 5.5 ea8132ceca RotationScaleMerge: with the GPU resident, build the per-space-group grouping on the device
Exact (tier E): the device gets the same group ids and the same group-ordered permutation
(ascending observation index within a group) the host's counting sort produced.

ComputeAsuGroups still finds the groups on the host (one ASU reduction per raw-hkl run, the
(key, run) sort, the dense ids, the representatives - now unpacked from the packed key instead of
a second ASU reduction). With the GPU resident it then hands the device only the runs in group
order (two arrays as long as the runs, not the observations): BuildGroups stamps every
observation's group from its run (the device evaluates the same finiteness test on the same
uploaded floats), counts each group's observations, and fills each group's segment and puts it in
index order. The host no longer builds or uploads the two observation-length arrays; the device
keeps its per-observation group array across Runs instead of reallocating it beside the old one.
SetGroups (host-built upload) is gone; the CPU path keeps the host histogram CSR.

Measured (prototype, GPU RTX 5080, Run-span sections): ComputeAsuGroups summed over a run's
merges 8tyy 9.0 -> 4.2 s, 8a1a 2.2 -> 0.85 s, cytc 0.51 -> 0.19 s. md5 of p.mtz identical to
production on myob/cytc/thau/8a1a/8tyy and cytc -N 4.

Clean branch rebuilt from scratch (GPU and CPU) and re-verified: p.mtz md5 identical to production on myob/cytc/thau/8a1a/8tyy/8qaw/9gdj (GPU), cytc -N 4, myob/cytc and myob -N 4 (CPU).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SVmAWnzCmRKAXVUCdc4iNi
2026-10-07 06:30:20 +02:00

119 lines
3.9 KiB
C++

// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include <cmath>
#include "HKLKey.h"
#include "gemmi/symmetry.hpp"
uint64_t HKLKey::pack() const {
constexpr int bits = 21;
constexpr int bias = 1 << (bits - 1); // 1,048,576
constexpr int max_value = bias - 1;
constexpr int min_value = -bias;
constexpr std::uint64_t mask = (1ULL << bits) - 1ULL;
if (h < min_value || h > max_value ||
k < min_value || k > max_value ||
l < min_value || l > max_value) {
throw std::out_of_range("HKL index outside packable range");
}
const std::uint64_t hh = static_cast<std::uint64_t>(h + bias) & mask;
const std::uint64_t kk = static_cast<std::uint64_t>(k + bias) & mask;
const std::uint64_t ll = static_cast<std::uint64_t>(l + bias) & mask;
return (hh << 1) | (kk << (bits + 1)) | (ll << (2 * bits + 1)) | (plus ? 1ULL : 0ULL);
}
HKLKey HKLKey::unpack(uint64_t packed) {
constexpr int bits = 21;
constexpr int bias = 1 << (bits - 1);
constexpr std::uint64_t mask = (1ULL << bits) - 1ULL;
return HKLKey{static_cast<int>((packed >> 1) & mask) - bias,
static_cast<int>((packed >> (bits + 1)) & mask) - bias,
static_cast<int>((packed >> (2 * bits + 1)) & mask) - bias,
(packed & 1ULL) != 0};
}
HKLKeyGenerator::HKLKeyGenerator(bool merge_friedel, const gemmi::SpaceGroup &sg)
: merge_friedel(merge_friedel),
sg(sg),
ops(sg.operations()),
asu(&sg) {
}
HKLKey HKLKeyGenerator::operator()(const MergedReflection &r) const {
return operator()(r.h, r.k, r.l);
}
HKLKey HKLKeyGenerator::operator()(const Reflection &r) const {
return operator()(r.h, r.k, r.l);
}
HKLKey HKLKeyGenerator::operator()(int32_t h, int32_t k, int32_t l) const {
HKLKey key{h, k, l, true};
if (sg.number == 1) {
const HKLKey neg{-h, -k, -l, true};
if (std::tie(key.h, key.k, key.l) < std::tie(neg.h, neg.k, neg.l)) {
key.h = -key.h;
key.k = -key.k;
key.l = -key.l;
key.plus = merge_friedel;
}
} else {
const gemmi::Op::Miller in{h, k, l};
const auto [hkl, sign_plus] = asu.to_asu_sign(in, ops);
key.h = hkl[0];
key.k = hkl[1];
key.l = hkl[2];
key.plus = merge_friedel ? true : sign_plus;
}
return key;
}
bool HKLKeyGenerator::IsSystematicallyAbsent(int32_t h, int32_t k, int32_t l) const {
return ops.is_systematically_absent(gemmi::Op::Miller{h, k, l});
}
bool HKLKeyGenerator::IsSystematicallyAbsent(const Reflection &r) const {
return IsSystematicallyAbsent(r.h, r.k, r.l);
}
bool AcceptReflection(const Reflection &r, std::optional<double> d_min_limit, std::optional<double> d_max_limit) {
if (!std::isfinite(r.I) || r.overloaded) // a saturated spot is not a measurement
return false;
if (!std::isfinite(r.d) || r.d <= 0.0f)
return false;
if (d_min_limit && r.d < d_min_limit)
return false;
if (d_max_limit && r.d > d_max_limit)
return false;
const float corr = r.prescaling_corr * r.qe_corr * r.flight_corr;
if (!std::isfinite(corr) || corr == 0.0f)
return false;
if (!std::isfinite(r.sigma) || r.sigma <= 0.0)
return false;
return true;
}
bool AcceptReflection(const Reflection &r, double d_min_limit, double d_max_limit) {
if (!std::isfinite(r.I) || r.overloaded) // a saturated spot is not a measurement
return false;
if (!std::isfinite(r.d) || r.d <= 0.0f)
return false;
if (d_min_limit > 0.0 && r.d < d_min_limit)
return false;
if (d_max_limit > 0.0 && r.d > d_max_limit)
return false;
const float corr = r.prescaling_corr * r.qe_corr * r.flight_corr;
if (!std::isfinite(corr) || corr == 0.0f)
return false;
if (!std::isfinite(r.sigma) || r.sigma <= 0.0)
return false;
return true;
}