rugnux: fixes from the rc173 pre-push scan

- RigidBodyGPU gather: AxisBrickBound undercounted by one where a box wraps
  past a partial last brick (n = 17: points 15, 16, 0 fall in bricks 1, 2, 0);
  the bound is now +2, and axis_bricks reports an overflow that Density()
  turns into a failure (-> CPU fallback) instead of dropping a brick.
  Exhaustive host test over n <= 64.
- RigidBodyTargetGPU::Residuals tests for an empty zone before launching
  Fcalc/Fmask (a 0-block grid).
- The pool engine is held by an RAII lease, returned if the constructor throws.
- The engine's stream is a CudaStream member (priority argument added), so it
  is not leaked when an allocation in the constructor throws.
- FitSolvent's "five or fewer strong reflections" case throws its own type,
  and only that is caught for the host scale.
- Hot pixels: the per-pixel counters are 16-bit, so the pre-scan sample is
  clamped to 65535 frames; no rings -> no 0-block sector/ring launches.
- PostRefine commit gate: fewer than two held-out positions refuse with
  their own reason; an excitation family of fewer than two values no longer
  refuses (its SE was NaN). No set of the r6-pooled battery reaches either
  case (smallest joint fit: ~2400 events, all SEs finite).
- Tests: ModelScaleGPU uploads on the fits' stream and accepts a near-tie's
  neighbouring grid point at the CPU winner's R; [gpu] rigid-body cases SKIP
  when the card is busy rather than fail.
- THIRD_PARTY_NOTICES: bitshuffle h-perf copyright holder is Kal Conley.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
This commit is contained in:
2026-09-29 00:24:45 +02:00
co-authored by Claude Opus 5.5
parent edca7a96ab
commit 941a822f76
14 changed files with 146 additions and 53 deletions
+43 -9
View File
@@ -10,6 +10,7 @@
#include <fstream>
#include <map>
#include <random>
#include <set>
#include <sstream>
#include <gemmi/mmread_gz.hpp>
@@ -23,6 +24,7 @@
#include "../rugnux/RigidBodyRefine.h"
#ifdef JFJOCH_USE_CUDA
#include "../rugnux/RigidBodyGPU.h"
#include "../rugnux/RigidBodyGPUEngine.h"
#include "../common/CUDAWrapper.h"
#endif
#include "../rugnux/SigmaA.h"
@@ -1025,6 +1027,26 @@ TEST_CASE("ModelValidation_RigidBodyTranslationDerivativeIsExact", "[ModelValida
}
}
#ifdef JFJOCH_USE_CUDA
namespace {
// A pool of `engines`, or a SKIP where there is none because the card is busy - over half its memory
// taken by something else, where the pool falls back to the CPU by design. Anywhere else no pool fails.
std::unique_ptr<RigidBodyGPUPool> TestPool(const gemmi::Model &model, const gemmi::UnitCell &cell,
const gemmi::SpaceGroup &sg, double d_min, size_t observations,
size_t engines, Logger &logger) {
auto pool = RigidBodyGPUPool::Create(model, cell, sg, d_min, observations, engines, logger);
if (!pool) {
size_t free = 0, total = 0;
RigidBodyGPUEngine::MemoryInfo(free, total);
if (free < total / 2)
SKIP("No GPU engine: the card is busy (" << free / 1000000 << " of " << total / 1000000 << " MB free)");
}
REQUIRE(pool);
return pool;
}
}
#endif
namespace {
// The rigid body's Jacobian at q against a central difference of its own residuals - which
// re-fit the scale at every evaluation - with the bulk-solvent mask held, as the Jacobian holds
@@ -1082,8 +1104,7 @@ namespace {
#ifdef JFJOCH_USE_CUDA
std::unique_ptr<RigidBodyGPUPool> engines;
if (gpu) {
engines = RigidBodyGPUPool::Create(model, st.cell, *sg, zone, fobs.v.size(), 1, logger);
REQUIRE(engines);
engines = TestPool(model, st.cell, *sg, zone, fobs.v.size(), 1, logger);
pool = engines.get();
}
#else
@@ -1235,6 +1256,23 @@ TEST_CASE("ModelValidation_RigidBodyReportsTheRotationItApplied", "[ModelValidat
#ifdef JFJOCH_USE_CUDA
// The gather's bound on the bricks a box falls in along one axis, against every box on small grids -
// wrapped boxes on grids that are not a multiple of the brick included (n = 17: 15, 16, 0 fall in bricks
// 1, 2 and 0). Host arithmetic, no device needed.
TEST_CASE("RigidBodyGPU_AxisBrickBoundCoversWrappedBoxes", "[ModelValidation][gpu]") {
CHECK(RigidBodyGPUEngine::AxisBrickBound(1, 17) == 3);
const int brick = 8;
for (int n = 1; n <= 64; n++)
for (int d = 0; 2 * d + 1 <= n; d++)
for (int c = 0; c < n; c++) {
std::set<int> bricks;
for (int p = c - d; p <= c + d; p++)
bricks.insert(((p % n) + n) % n / brick);
INFO("n " << n << ", d " << d << ", c " << c);
CHECK(bricks.size() <= RigidBodyGPUEngine::AxisBrickBound(d, n));
}
}
// The GPU target is the CPU target's function, computed on the device: at the same placement the two
// give the same residuals and the same Jacobian to rounding (float distances on the device, cuFFT for
// FFTW), on every group of the composition tests, both zones, anisotropic atoms included.
@@ -1249,8 +1287,7 @@ TEST_CASE("RigidBodyGPU_MatchesCPU", "[ModelValidation][gpu]") {
for (double zone : {6.0, 3.5}) {
const auto fobs = OwnAmplitudes(cryst, st, zone, logger);
gemmi::Model cpu_model = st.models[0], gpu_model = st.models[0];
auto pool = RigidBodyGPUPool::Create(gpu_model, st.cell, *sg, zone, fobs.v.size(), 1, logger);
REQUIRE(pool);
auto pool = TestPool(gpu_model, st.cell, *sg, zone, fobs.v.size(), 1, logger);
RigidBodyTarget cpu(cpu_model, st.cell, *sg, 4);
RigidBodyTargetGPU gpu(*pool, gpu_model, st.cell, *sg, 4);
cpu.SetZone(fobs, zone);
@@ -1337,9 +1374,7 @@ TEST_CASE("RigidBodyGPU_FitAgreesWithCPU", "[ModelValidation][gpu]") {
Logger logger("RigidBodyGPU_FitAgreesWithCPU");
for (const char *cryst : {kCryst, kPolarCryst, kRigidBodyCrysts[3]}) {
gemmi::Structure cpu_st = AnisoCluster(cryst), gpu_st = AnisoCluster(cryst);
auto pool = RigidBodyGPUPool::Create(gpu_st.models[0], gpu_st.cell, *gpu_st.find_spacegroup(), 3.0, 100000, 1,
logger);
REQUIRE(pool);
auto pool = TestPool(gpu_st.models[0], gpu_st.cell, *gpu_st.find_spacegroup(), 3.0, 100000, 1, logger);
const RigidBodyRefineResult cpu = DisplacedFit(cryst, cpu_st, nullptr, logger);
const RigidBodyRefineResult gpu = DisplacedFit(cryst, gpu_st, pool.get(), logger);
CHECK(gpu.converged == cpu.converged);
@@ -1366,8 +1401,7 @@ TEST_CASE("RigidBodyGPU_Deterministic", "[ModelValidation][gpu]") {
std::vector<std::vector<gemmi::Position>> placed;
for (size_t engines : {1, 1, 4}) {
gemmi::Structure st = AnisoCluster(cryst);
auto pool = RigidBodyGPUPool::Create(st.models[0], st.cell, *st.find_spacegroup(), 3.0, 100000, engines, logger);
REQUIRE(pool);
auto pool = TestPool(st.models[0], st.cell, *st.find_spacegroup(), 3.0, 100000, engines, logger);
const RigidBodyRefineResult r = DisplacedFit(cryst, st, pool.get(), logger);
CHECK(r.evaluations > 0);
placed.push_back(ModelPositions(st.models[0]));