rugnux: model validation's structure factors and maps on the GPU
ModelStructureFactorsGPU computes what compute_model_factors() and
map_from_coefficients() compute on the CPU - F_calc from the model's
density (IT92, Refmac-compatible blur, unblurred as prepare_asu_data()
does) and F_mask from the Refmac bulk-solvent mask, both on the
reflections prepare_asu_data(d_min) lists, in its order; and a map from
ASU coefficients on the grid get_size_for_hkl(coef, 0, 3.0) sizes - on a
device. Made once per cell, group, resolution and model, then evaluated
as often as the coordinates change, so refinement or MR can call it in a
loop. The device is an explicit parameter; every call leaves the calling
thread's current device as it found it.
Pieces:
- ModelDensityGPU: the rigid body's deterministic brick gather, moved
out of RigidBodyGPU.cu into a component of its own (ModelMaskGPU's
pattern); the rigid body uses it unchanged. MAX_BRICKS_PER_AXIS 8 ->
16, so fine grids with high-B atoms (lysozyme at 1.2 A, a 0.9 A P1
cell) are no longer refused; existing zones are gridded identically.
- One copy of the content is gridded and the symmetry composed in
reciprocal space (SymmetryComposition), operators applied on the fly;
the mask is ModelMaskGPU (every image of every atom, islands, shrink).
- Maps: gemmi's get_f_phi_on_grid() in ZYX order on the host (the
coefficients written are the same), in-place cuFFT c2r, transposed back
to XYZ on the device. One map at a time, in the engine's buffers.
Decided once, up front, per card, from its TOTAL memory: the engine's
bytes (16 N + cuFFT work + reflections, N the larger of the structure-
factor and map grids) must be at most half the card - the rigid body's
engines take at most a quarter beside it. Otherwise, or where the gather
cannot grid the cell, the CPU path runs, logged with needed vs total.
Anything to a resolution other than d_min (the null's 3.5 A fits) stays
on the CPU, so all replicates and the real model's side of the null are
computed the same way. A CUDA failure takes the existing path: the
validation restarts on the CPU.
Measured, model validation total per run (CPU path -> GPU), 16 GB card:
F432 215 A cubic, 1.30 A, 500^3 grid: 47.7 -> 15.6 s (two validations;
14.3 -> 2.4 and 33.4 -> 13.2, the rest of the second is writing the
three 0.5 GB maps); F_calc + F_mask 5.7 s -> 0.05 s
C2 1.11 A: 23.9 -> 13.6 s; P3_2 1.55 A: 18.2 -> 10.2 s;
P2_1 1.25 A: 13.4 -> 6.5 s; F4_132 328 A: 13.0 -> 5.5 s;
P6_5: 8.8 -> 4.2 s; P4_3 0.97 A: 4.6 -> 2.5 s; P1 0.92 A: 4.2 -> 2.2 s;
small P1: 3.2 -> 1.3 s; P6_1: 8.8 -> 6.2 s; lysozyme: 1.8 -> 1.4 s.
p.mtz md5-identical on all 13 sets. Against the CPU path: FC within
1e-4 of mean |F|, phases of the strong half within 0.003 deg, maps within
1e-4 (2mFo-DFc) and 7e-4 (mFo-DFc) of their rms; every logged R, CC,
FOM, k_sol and anomalous site list identical at the printed precision,
except where a rigid-body commit sat on an exact R-free tie (0.2155 ->
0.2155) and fell the other way (R-work 0.2127 vs 0.2129). The GPU result
is bit-identical run to run and with -N 8 (maps, map MTZ, placed model).
Peak device memory of the engine: 2.5 GB at 500^3 (process total peaked
at 14.4 GB with what the merge still holds).
Tests: ModelStructureFactorsGPU_MatchesCPU (five groups, 3.5 and 1.5 A:
same reflections, F_calc <= 1e-5 of mean |F|, F_mask 2e-7 rms, repeat
bit-identical), ModelStructureFactorsGPU_MapMatchesCPU (<= 5e-6 of rms).
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SVmAWnzCmRKAXVUCdc4iNi
This commit is contained in:
@@ -23,6 +23,7 @@
|
||||
#include "../image_analysis/structure_refinement/ModelValidation.h"
|
||||
#include "../image_analysis/structure_refinement/RigidBodyRefine.h"
|
||||
#ifdef JFJOCH_USE_CUDA
|
||||
#include "../image_analysis/structure_refinement/ModelStructureFactorsGPU.h"
|
||||
#include "../image_analysis/structure_refinement/RigidBodyGPU.h"
|
||||
#include "../image_analysis/structure_refinement/RigidBodyGPUEngine.h"
|
||||
#include "../common/CUDAWrapper.h"
|
||||
@@ -1414,4 +1415,98 @@ TEST_CASE("RigidBodyGPU_Deterministic", "[ModelValidation][gpu]") {
|
||||
}
|
||||
}
|
||||
|
||||
// The model's structure factors on the GPU against the CPU path model validation takes - gemmi's density and
|
||||
// mask on the symmetrized grid, transformed by FFTW and read off by prepare_asu_data() - in the five groups,
|
||||
// at a resolution where the mask's shrink step does nothing and at one where it does. The same reflections
|
||||
// in the same order; F_calc to rounding; F_mask to the few grid points at an atom's mask radius that float
|
||||
// and double distances put on different sides. The same atoms twice give the same bits.
|
||||
TEST_CASE("ModelStructureFactorsGPU_MatchesCPU", "[ModelValidation][gpu]") {
|
||||
if (get_gpu_count() == 0)
|
||||
SKIP("No GPU");
|
||||
for (const char *cryst : kRigidBodyCrysts) {
|
||||
const gemmi::Structure st = AnisoCluster(cryst);
|
||||
const gemmi::Model &model = st.models[0];
|
||||
const gemmi::SpaceGroup *sg = st.find_spacegroup();
|
||||
for (double d_min : {3.5, 1.5}) {
|
||||
auto dc = ZoneDensity(st, model, d_min);
|
||||
dc.put_model_density_on_grid(model);
|
||||
const auto ref_fc = MapToFPhi(dc.grid).prepare_asu_data(d_min, dc.blur, false, false, false);
|
||||
gemmi::Grid<float> mask;
|
||||
mask.unit_cell = st.cell;
|
||||
mask.spacegroup = sg;
|
||||
mask.set_size_from_spacing(dc.requested_grid_spacing(), gemmi::GridSizeRounding::Up);
|
||||
gemmi::SolventMasker(gemmi::AtomicRadiiSet::Refmac).put_mask_on_grid(mask, model);
|
||||
const auto ref_fm = MapToFPhi(mask).prepare_asu_data(d_min, 0);
|
||||
|
||||
ModelStructureFactorsGPU gpu(0, model, st.cell, *sg, d_min);
|
||||
REQUIRE(gpu.Supported());
|
||||
gpu.Reserve();
|
||||
gemmi::AsuData<std::complex<float>> fc, fm, fc2, fm2;
|
||||
gpu.Compute(model, fc, fm);
|
||||
gpu.Compute(model, fc2, fm2);
|
||||
REQUIRE(fc.v.size() == ref_fc.v.size());
|
||||
REQUIRE(fm.v.size() == ref_fm.v.size());
|
||||
double fc_mean = 0, fc_worst = 0, fm_norm = 0, fm_diff = 0;
|
||||
bool repeat_same = true;
|
||||
for (size_t i = 0; i < fc.v.size(); i++) {
|
||||
REQUIRE(fc.v[i].hkl == ref_fc.v[i].hkl);
|
||||
REQUIRE(fm.v[i].hkl == ref_fm.v[i].hkl);
|
||||
fc_mean += std::abs(ref_fc.v[i].value) / static_cast<double>(fc.v.size());
|
||||
fc_worst = std::max(fc_worst, static_cast<double>(std::abs(fc.v[i].value - ref_fc.v[i].value)));
|
||||
fm_norm += std::norm(ref_fm.v[i].value);
|
||||
fm_diff += std::norm(fm.v[i].value - ref_fm.v[i].value);
|
||||
repeat_same = repeat_same && fc2.v[i].value == fc.v[i].value && fm2.v[i].value == fm.v[i].value;
|
||||
}
|
||||
INFO(cryst << " at " << d_min << " A: F_calc worst " << fc_worst / fc_mean << " of the mean |F|, F_mask "
|
||||
<< std::sqrt(fm_diff / fm_norm) << " relative rms");
|
||||
CHECK(fc_worst <= 1e-4 * fc_mean);
|
||||
CHECK(std::sqrt(fm_diff) <= 1e-3 * std::sqrt(fm_norm));
|
||||
CHECK(repeat_same);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A map from coefficients on the GPU against MapFromFPhi() of gemmi's get_f_phi_on_grid(): the same grid,
|
||||
// the same values to rounding, the same bits twice.
|
||||
TEST_CASE("ModelStructureFactorsGPU_MapMatchesCPU", "[ModelValidation][gpu]") {
|
||||
if (get_gpu_count() == 0)
|
||||
SKIP("No GPU");
|
||||
for (const char *cryst : kRigidBodyCrysts) {
|
||||
const gemmi::Structure st = AnisoCluster(cryst);
|
||||
const gemmi::Model &model = st.models[0];
|
||||
const gemmi::SpaceGroup *sg = st.find_spacegroup();
|
||||
const double d_min = 2.0;
|
||||
ModelStructureFactorsGPU gpu(0, model, st.cell, *sg, d_min);
|
||||
REQUIRE(gpu.Supported());
|
||||
gpu.Reserve();
|
||||
gemmi::AsuData<std::complex<float>> coef, fm;
|
||||
gpu.Compute(model, coef, fm);
|
||||
// Coefficients on every second reflection only, as a map's are on the observed ones.
|
||||
gemmi::AsuData<std::complex<float>> some = coef;
|
||||
some.v.clear();
|
||||
for (size_t i = 0; i < coef.v.size(); i += 2)
|
||||
some.v.push_back(coef.v[i]);
|
||||
|
||||
gemmi::AsuData<std::complex<float>> cpu_coef = some;
|
||||
cpu_coef.ensure_sorted();
|
||||
const auto size = gemmi::get_size_for_hkl(cpu_coef, {{0, 0, 0}}, 3.0);
|
||||
const gemmi::Grid<float> cpu = MapFromFPhi(gemmi::get_f_phi_on_grid<float>(cpu_coef, size, true));
|
||||
gemmi::AsuData<std::complex<float>> gpu_coef = some;
|
||||
const gemmi::Grid<float> map = gpu.Map(gpu_coef);
|
||||
const gemmi::Grid<float> map2 = gpu.Map(gpu_coef);
|
||||
REQUIRE(map.nu == cpu.nu);
|
||||
REQUIRE(map.nv == cpu.nv);
|
||||
REQUIRE(map.nw == cpu.nw);
|
||||
double rms = 0, worst = 0;
|
||||
for (size_t i = 0; i < cpu.data.size(); i++) {
|
||||
rms += gemmi::sq(cpu.data[i]) / static_cast<double>(cpu.data.size());
|
||||
worst = std::max(worst, static_cast<double>(std::fabs(map.data[i] - cpu.data[i])));
|
||||
}
|
||||
INFO(cryst << ": map differs by at most " << worst / std::sqrt(rms) << " of its rms");
|
||||
CHECK(worst <= 1e-4 * std::sqrt(rms));
|
||||
CHECK(std::memcmp(map.data.data(), map2.data.data(), map.data.size() * sizeof(float)) == 0);
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
|
||||
Reference in New Issue
Block a user