rugnux: fixes from the rc173 pre-push scan
- RigidBodyGPU gather: AxisBrickBound undercounted by one where a box wraps past a partial last brick (n = 17: points 15, 16, 0 fall in bricks 1, 2, 0); the bound is now +2, and axis_bricks reports an overflow that Density() turns into a failure (-> CPU fallback) instead of dropping a brick. Exhaustive host test over n <= 64. - RigidBodyTargetGPU::Residuals tests for an empty zone before launching Fcalc/Fmask (a 0-block grid). - The pool engine is held by an RAII lease, returned if the constructor throws. - The engine's stream is a CudaStream member (priority argument added), so it is not leaked when an allocation in the constructor throws. - FitSolvent's "five or fewer strong reflections" case throws its own type, and only that is caught for the host scale. - Hot pixels: the per-pixel counters are 16-bit, so the pre-scan sample is clamped to 65535 frames; no rings -> no 0-block sector/ring launches. - PostRefine commit gate: fewer than two held-out positions refuse with their own reason; an excitation family of fewer than two values no longer refuses (its SE was NaN). No set of the r6-pooled battery reaches either case (smallest joint fit: ~2400 events, all SEs finite). - Tests: ModelScaleGPU uploads on the fits' stream and accepts a near-tie's neighbouring grid point at the CPU winner's R; [gpu] rigid-body cases SKIP when the card is busy rather than fail. - THIRD_PARTY_NOTICES: bitshuffle h-perf copyright holder is Kal Conley. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
This commit is contained in:
@@ -10,6 +10,7 @@
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
#include <random>
|
||||
#include <set>
|
||||
#include <sstream>
|
||||
|
||||
#include <gemmi/mmread_gz.hpp>
|
||||
@@ -23,6 +24,7 @@
|
||||
#include "../rugnux/RigidBodyRefine.h"
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
#include "../rugnux/RigidBodyGPU.h"
|
||||
#include "../rugnux/RigidBodyGPUEngine.h"
|
||||
#include "../common/CUDAWrapper.h"
|
||||
#endif
|
||||
#include "../rugnux/SigmaA.h"
|
||||
@@ -1025,6 +1027,26 @@ TEST_CASE("ModelValidation_RigidBodyTranslationDerivativeIsExact", "[ModelValida
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
namespace {
|
||||
// A pool of `engines`, or a SKIP where there is none because the card is busy - over half its memory
|
||||
// taken by something else, where the pool falls back to the CPU by design. Anywhere else no pool fails.
|
||||
std::unique_ptr<RigidBodyGPUPool> TestPool(const gemmi::Model &model, const gemmi::UnitCell &cell,
|
||||
const gemmi::SpaceGroup &sg, double d_min, size_t observations,
|
||||
size_t engines, Logger &logger) {
|
||||
auto pool = RigidBodyGPUPool::Create(model, cell, sg, d_min, observations, engines, logger);
|
||||
if (!pool) {
|
||||
size_t free = 0, total = 0;
|
||||
RigidBodyGPUEngine::MemoryInfo(free, total);
|
||||
if (free < total / 2)
|
||||
SKIP("No GPU engine: the card is busy (" << free / 1000000 << " of " << total / 1000000 << " MB free)");
|
||||
}
|
||||
REQUIRE(pool);
|
||||
return pool;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
namespace {
|
||||
// The rigid body's Jacobian at q against a central difference of its own residuals - which
|
||||
// re-fit the scale at every evaluation - with the bulk-solvent mask held, as the Jacobian holds
|
||||
@@ -1082,8 +1104,7 @@ namespace {
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
std::unique_ptr<RigidBodyGPUPool> engines;
|
||||
if (gpu) {
|
||||
engines = RigidBodyGPUPool::Create(model, st.cell, *sg, zone, fobs.v.size(), 1, logger);
|
||||
REQUIRE(engines);
|
||||
engines = TestPool(model, st.cell, *sg, zone, fobs.v.size(), 1, logger);
|
||||
pool = engines.get();
|
||||
}
|
||||
#else
|
||||
@@ -1235,6 +1256,23 @@ TEST_CASE("ModelValidation_RigidBodyReportsTheRotationItApplied", "[ModelValidat
|
||||
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
|
||||
// The gather's bound on the bricks a box falls in along one axis, against every box on small grids -
|
||||
// wrapped boxes on grids that are not a multiple of the brick included (n = 17: 15, 16, 0 fall in bricks
|
||||
// 1, 2 and 0). Host arithmetic, no device needed.
|
||||
TEST_CASE("RigidBodyGPU_AxisBrickBoundCoversWrappedBoxes", "[ModelValidation][gpu]") {
|
||||
CHECK(RigidBodyGPUEngine::AxisBrickBound(1, 17) == 3);
|
||||
const int brick = 8;
|
||||
for (int n = 1; n <= 64; n++)
|
||||
for (int d = 0; 2 * d + 1 <= n; d++)
|
||||
for (int c = 0; c < n; c++) {
|
||||
std::set<int> bricks;
|
||||
for (int p = c - d; p <= c + d; p++)
|
||||
bricks.insert(((p % n) + n) % n / brick);
|
||||
INFO("n " << n << ", d " << d << ", c " << c);
|
||||
CHECK(bricks.size() <= RigidBodyGPUEngine::AxisBrickBound(d, n));
|
||||
}
|
||||
}
|
||||
|
||||
// The GPU target is the CPU target's function, computed on the device: at the same placement the two
|
||||
// give the same residuals and the same Jacobian to rounding (float distances on the device, cuFFT for
|
||||
// FFTW), on every group of the composition tests, both zones, anisotropic atoms included.
|
||||
@@ -1249,8 +1287,7 @@ TEST_CASE("RigidBodyGPU_MatchesCPU", "[ModelValidation][gpu]") {
|
||||
for (double zone : {6.0, 3.5}) {
|
||||
const auto fobs = OwnAmplitudes(cryst, st, zone, logger);
|
||||
gemmi::Model cpu_model = st.models[0], gpu_model = st.models[0];
|
||||
auto pool = RigidBodyGPUPool::Create(gpu_model, st.cell, *sg, zone, fobs.v.size(), 1, logger);
|
||||
REQUIRE(pool);
|
||||
auto pool = TestPool(gpu_model, st.cell, *sg, zone, fobs.v.size(), 1, logger);
|
||||
RigidBodyTarget cpu(cpu_model, st.cell, *sg, 4);
|
||||
RigidBodyTargetGPU gpu(*pool, gpu_model, st.cell, *sg, 4);
|
||||
cpu.SetZone(fobs, zone);
|
||||
@@ -1337,9 +1374,7 @@ TEST_CASE("RigidBodyGPU_FitAgreesWithCPU", "[ModelValidation][gpu]") {
|
||||
Logger logger("RigidBodyGPU_FitAgreesWithCPU");
|
||||
for (const char *cryst : {kCryst, kPolarCryst, kRigidBodyCrysts[3]}) {
|
||||
gemmi::Structure cpu_st = AnisoCluster(cryst), gpu_st = AnisoCluster(cryst);
|
||||
auto pool = RigidBodyGPUPool::Create(gpu_st.models[0], gpu_st.cell, *gpu_st.find_spacegroup(), 3.0, 100000, 1,
|
||||
logger);
|
||||
REQUIRE(pool);
|
||||
auto pool = TestPool(gpu_st.models[0], gpu_st.cell, *gpu_st.find_spacegroup(), 3.0, 100000, 1, logger);
|
||||
const RigidBodyRefineResult cpu = DisplacedFit(cryst, cpu_st, nullptr, logger);
|
||||
const RigidBodyRefineResult gpu = DisplacedFit(cryst, gpu_st, pool.get(), logger);
|
||||
CHECK(gpu.converged == cpu.converged);
|
||||
@@ -1366,8 +1401,7 @@ TEST_CASE("RigidBodyGPU_Deterministic", "[ModelValidation][gpu]") {
|
||||
std::vector<std::vector<gemmi::Position>> placed;
|
||||
for (size_t engines : {1, 1, 4}) {
|
||||
gemmi::Structure st = AnisoCluster(cryst);
|
||||
auto pool = RigidBodyGPUPool::Create(st.models[0], st.cell, *st.find_spacegroup(), 3.0, 100000, engines, logger);
|
||||
REQUIRE(pool);
|
||||
auto pool = TestPool(st.models[0], st.cell, *st.find_spacegroup(), 3.0, 100000, engines, logger);
|
||||
const RigidBodyRefineResult r = DisplacedFit(cryst, st, pool.get(), logger);
|
||||
CHECK(r.evaluations > 0);
|
||||
placed.push_back(ModelPositions(st.models[0]));
|
||||
|
||||
Reference in New Issue
Block a user