Exact (tier E): the device gets the same group ids and the same group-ordered permutation (ascending observation index within a group) the host's counting sort produced. ComputeAsuGroups still finds the groups on the host (one ASU reduction per raw-hkl run, the (key, run) sort, the dense ids, the representatives - now unpacked from the packed key instead of a second ASU reduction). With the GPU resident it then hands the device only the runs in group order (two arrays as long as the runs, not the observations): BuildGroups stamps every observation's group from its run (the device evaluates the same finiteness test on the same uploaded floats), counts each group's observations, and fills each group's segment and puts it in index order. The host no longer builds or uploads the two observation-length arrays; the device keeps its per-observation group array across Runs instead of reallocating it beside the old one. SetGroups (host-built upload) is gone; the CPU path keeps the host histogram CSR. Measured (prototype, GPU RTX 5080, Run-span sections): ComputeAsuGroups summed over a run's merges 8tyy 9.0 -> 4.2 s, 8a1a 2.2 -> 0.85 s, cytc 0.51 -> 0.19 s. md5 of p.mtz identical to production on myob/cytc/thau/8a1a/8tyy and cytc -N 4. Clean branch rebuilt from scratch (GPU and CPU) and re-verified: p.mtz md5 identical to production on myob/cytc/thau/8a1a/8tyy/8qaw/9gdj (GPU), cytc -N 4, myob/cytc and myob -N 4 (CPU). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SVmAWnzCmRKAXVUCdc4iNi
119 lines
3.9 KiB
C++
119 lines
3.9 KiB
C++
// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#include <cmath>
|
|
|
|
#include "HKLKey.h"
|
|
#include "gemmi/symmetry.hpp"
|
|
|
|
uint64_t HKLKey::pack() const {
|
|
constexpr int bits = 21;
|
|
constexpr int bias = 1 << (bits - 1); // 1,048,576
|
|
constexpr int max_value = bias - 1;
|
|
constexpr int min_value = -bias;
|
|
constexpr std::uint64_t mask = (1ULL << bits) - 1ULL;
|
|
|
|
if (h < min_value || h > max_value ||
|
|
k < min_value || k > max_value ||
|
|
l < min_value || l > max_value) {
|
|
throw std::out_of_range("HKL index outside packable range");
|
|
}
|
|
|
|
const std::uint64_t hh = static_cast<std::uint64_t>(h + bias) & mask;
|
|
const std::uint64_t kk = static_cast<std::uint64_t>(k + bias) & mask;
|
|
const std::uint64_t ll = static_cast<std::uint64_t>(l + bias) & mask;
|
|
|
|
return (hh << 1) | (kk << (bits + 1)) | (ll << (2 * bits + 1)) | (plus ? 1ULL : 0ULL);
|
|
}
|
|
|
|
HKLKey HKLKey::unpack(uint64_t packed) {
|
|
constexpr int bits = 21;
|
|
constexpr int bias = 1 << (bits - 1);
|
|
constexpr std::uint64_t mask = (1ULL << bits) - 1ULL;
|
|
return HKLKey{static_cast<int>((packed >> 1) & mask) - bias,
|
|
static_cast<int>((packed >> (bits + 1)) & mask) - bias,
|
|
static_cast<int>((packed >> (2 * bits + 1)) & mask) - bias,
|
|
(packed & 1ULL) != 0};
|
|
}
|
|
|
|
HKLKeyGenerator::HKLKeyGenerator(bool merge_friedel, const gemmi::SpaceGroup &sg)
|
|
: merge_friedel(merge_friedel),
|
|
sg(sg),
|
|
ops(sg.operations()),
|
|
asu(&sg) {
|
|
}
|
|
|
|
HKLKey HKLKeyGenerator::operator()(const MergedReflection &r) const {
|
|
return operator()(r.h, r.k, r.l);
|
|
}
|
|
|
|
HKLKey HKLKeyGenerator::operator()(const Reflection &r) const {
|
|
return operator()(r.h, r.k, r.l);
|
|
}
|
|
|
|
HKLKey HKLKeyGenerator::operator()(int32_t h, int32_t k, int32_t l) const {
|
|
HKLKey key{h, k, l, true};
|
|
|
|
if (sg.number == 1) {
|
|
const HKLKey neg{-h, -k, -l, true};
|
|
if (std::tie(key.h, key.k, key.l) < std::tie(neg.h, neg.k, neg.l)) {
|
|
key.h = -key.h;
|
|
key.k = -key.k;
|
|
key.l = -key.l;
|
|
key.plus = merge_friedel;
|
|
}
|
|
} else {
|
|
const gemmi::Op::Miller in{h, k, l};
|
|
const auto [hkl, sign_plus] = asu.to_asu_sign(in, ops);
|
|
|
|
key.h = hkl[0];
|
|
key.k = hkl[1];
|
|
key.l = hkl[2];
|
|
key.plus = merge_friedel ? true : sign_plus;
|
|
}
|
|
return key;
|
|
}
|
|
|
|
bool HKLKeyGenerator::IsSystematicallyAbsent(int32_t h, int32_t k, int32_t l) const {
|
|
return ops.is_systematically_absent(gemmi::Op::Miller{h, k, l});
|
|
}
|
|
|
|
bool HKLKeyGenerator::IsSystematicallyAbsent(const Reflection &r) const {
|
|
return IsSystematicallyAbsent(r.h, r.k, r.l);
|
|
}
|
|
|
|
bool AcceptReflection(const Reflection &r, std::optional<double> d_min_limit, std::optional<double> d_max_limit) {
|
|
if (!std::isfinite(r.I) || r.overloaded) // a saturated spot is not a measurement
|
|
return false;
|
|
if (!std::isfinite(r.d) || r.d <= 0.0f)
|
|
return false;
|
|
if (d_min_limit && r.d < d_min_limit)
|
|
return false;
|
|
if (d_max_limit && r.d > d_max_limit)
|
|
return false;
|
|
const float corr = r.prescaling_corr * r.qe_corr * r.flight_corr;
|
|
if (!std::isfinite(corr) || corr == 0.0f)
|
|
return false;
|
|
if (!std::isfinite(r.sigma) || r.sigma <= 0.0)
|
|
return false;
|
|
return true;
|
|
}
|
|
|
|
bool AcceptReflection(const Reflection &r, double d_min_limit, double d_max_limit) {
|
|
if (!std::isfinite(r.I) || r.overloaded) // a saturated spot is not a measurement
|
|
return false;
|
|
if (!std::isfinite(r.d) || r.d <= 0.0f)
|
|
return false;
|
|
if (d_min_limit > 0.0 && r.d < d_min_limit)
|
|
return false;
|
|
if (d_max_limit > 0.0 && r.d > d_max_limit)
|
|
return false;
|
|
const float corr = r.prescaling_corr * r.qe_corr * r.flight_corr;
|
|
if (!std::isfinite(corr) || corr == 0.0f)
|
|
return false;
|
|
if (!std::isfinite(r.sigma) || r.sigma <= 0.0)
|
|
return false;
|
|
return true;
|
|
}
|
|
|