// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #pragma once #include #include #include #include #include "../indexing/CUDAMemHelpers.h" // The rigid body's scale fit on the GPU: what RigidBodyTarget::Residuals does with a gemmi // Scaling (use_solvent, k_sol and b_sol fixed) - fit_isotropic_b_approximately() followed by // gemmi's Levenberg-Marquardt, and FitModelScale's k_sol/b_sol grid. The sums over the reflections run // on the device, accumulated in double in a fixed order; the Levenberg-Marquardt control (the damping, // the 7x7 solve, the stop rules) is gemmi's own, ported line for line and run on the host. struct ModelScaleParams { double k_overall = 1.0; double b_star[6] = {0, 0, 0, 0, 0, 0}; // gemmi SMat33 order u11 u22 u33 u12 u13 u23 }; struct ModelSolventFit { double k_sol = 0.35, b_sol = 46.0; // gemmi's Scaling defaults when the fit does not run double r = 1.0; // ModelScaleReport::r_work_fit int n_grid = 0; ModelScaleParams scale; // the winner's k_overall and b_star }; // One fit's parameters for one launch, and which sums to take over the points. struct ModelScaleFitState { double k_overall; double b_star[6]; double k_sol, b_sol; int mode; int column; // the fit's row of |Fcalc + solvent| on the device }; class ModelScaleGPU { public: static size_t DeviceBytes(size_t max_points); ModelScaleGPU(cudaStream_t stream, size_t max_points); // Per zone: the Scaling points in gemmi prepare_points() order, adp_symmetry_constraints(sg) rows // and the cell's fractionalization matrix (UnitCell::frac.mat, row-major). n <= max_points. void SetPoints(const std::vector> &hkl, const std::vector &stol2, const std::vector &fobs, const std::vector &sigma, const std::vector> &constraints, const double frac[9]); // fit_isotropic_b_approximately() + fit_parameters() at a fixed solvent pair, from k_overall = 1 and // b_star = 0. d_fcmol, d_fmask: one float2 per point on the device. ModelScaleParams Fit(const float2 *d_fcmol, const float2 *d_fmask, double k_sol, double b_sol); // FitModelScale (ModelScaling.cpp) with the default box: the same grid, the same winner. Throws // ModelScaleGPUTooFewReflections where FitModelScale would chain the grid points (five or fewer reflections for the // isotropic fit). ModelSolventFit FitSolvent(const float2 *d_fcmol, const float2 *d_fmask); private: struct LevMarRun; // Queues the copy of `fits` to the device. host_fits_ is written only after the stream has been // synchronized, which every Reduce() ends with. void UploadFits(const std::vector &fits); // The sums of the first `nfits` uploaded fits over all points, one row of SLOTS doubles per fit, on // the host. const double *Reduce(int nfits); const double *Sums(const std::vector &fits); void FitBatch(const float2 *d_fcmol, const float2 *d_fmask, std::vector &fits); cudaStream_t stream_; size_t max_points_; int n_ = 0; int n_strong_ = 0; // points fit_isotropic_b_approximately() fits on int n_params_ = 1; // k_overall + one per constraint row double constraints_[6][6] = {}; double frac_[9] = {}; CudaDevicePtr hkl_; // 3 per point CudaDevicePtr stol2_; CudaDevicePtr fobs_; CudaDevicePtr strong_; CudaDevicePtr f_abs_; // per fit of a batch, per point CudaDevicePtr fits_; CudaDevicePtr partial_; // per fit, per block, per slot CudaDevicePtr sums_; // per fit, per slot CudaHostPtr host_fits_; CudaHostPtr host_sums_; };