// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #pragma once // The density of a model's atoms on a grid, on the GPU (CUDA builds only): gemmi's // DensityCalculator::put_model_density_on_grid() without the symmetrize - ONE copy of the content. The // rigid body (RigidBodyGPU.cu) and the model's structure factors (ModelStructureFactorsGPU.cu) both grid // with it, and compose the crystal's symmetry in reciprocal space (SymmetryComposition, RigidBodyRefine.h). // // A gather, not a scatter: the grid is cut into bricks of BRICK^3 points, each brick is one block, and // every point adds, in model order, each atom whose sphere it is inside. Nothing is summed with a // floating-point atomic, so the grid is the same, bit for bit, on every run. #include #include #include #include "../indexing/CUDAMemHelpers.h" // One atom's density on a grid, as PutModelDensityOnGrid() (ModelGrid.cpp) sets it up: gemmi's // precalculated five-Gaussian sum and the radius it cuts the sum at. struct ModelDensityAtom { float a[5]; float b[5][6]; // isotropic: b[k][0] multiplies r^2; anisotropic: the matrix, u11 u22 u33 u12 u13 u23 float occ; float radius; int aniso; }; // The grid the density is put on, u fastest: index = u + nu * (v + nv * w). struct ModelDensityGrid { int nu = 0, nv = 0, nw = 0; double orth[9] = {}, frac[9] = {}; // the cell's, row-major }; class ModelDensityGPU { public: // The bytes an instance of this capacity reserves on the device. static size_t DeviceBytes(size_t max_atoms, size_t max_pairs, size_t max_bricks); static size_t Bricks(int nu, int nv, int nw); // The most gather bricks the 2 d + 1 points of an atom's box can fall in along an axis of n points. static size_t AxisBrickBound(int d, int n); // gemmi's box around each atom (MakeAtomBox, ModelGrid.cpp): the points within ceil(radius / spacing) // of the nearest one along each axis, three per atom. static std::vector Boxes(const ModelDensityGrid &grid, const std::vector &atoms); // (brick, atom) pairs of the gather over at most. static size_t PairBound(const ModelDensityGrid &grid, const std::vector &atoms); // Whether the gather reproduces gemmi's box walk on this grid: every atom's box narrower than the cell, // so that no point is reached by two images of one atom. static bool Supports(const ModelDensityGrid &grid, const std::vector &atoms); ModelDensityGPU(cudaStream_t stream, size_t max_atoms, size_t max_pairs, size_t max_bricks); // The grid and the atoms' densities. Throws if they are over the capacity or if !Supports(). void SetAtoms(const ModelDensityGrid &grid, const std::vector &atoms); // The density into d_grid (nu * nv * nw floats, every point written). d_pos: each atom's position in // grid units - fractional times the grid size, wrapped into [0, n) - in the order SetAtoms() got them; // w is not read. Waits on the stream once, for the number of (brick, atom) pairs. void Compute(const float4 *d_pos, float *d_grid); // What the gather kernel needs of the grid. struct Geometry { int nu, nv, nw; float orth_n[9]; // orth * diag(1/nu, 1/nv, 1/nw), row-major: grid offset -> Cartesian int narrow; // the cell is too small for one image of an atom to serve a whole brick }; private: cudaStream_t stream; size_t max_atoms, max_pairs, max_bricks; Geometry geom{}; int n_atoms = 0; CudaDevicePtr atoms; CudaDevicePtr box; CudaDevicePtr overflow; // an atom's box over MAX_BRICKS_PER_AXIS CudaDevicePtr count, offset, key, value, key_sorted, value_sorted, brick_start, brick_end; CudaDevicePtr cub_temp; size_t cub_bytes = 0; };