Files
Jungfraujoch/image_analysis/structure_refinement/ModelScaleGPU.h
T
leonarski_fandClaude Opus 5.5 9ad92b6bfe Move the atomic-model code to image_analysis/structure_refinement/ and WriteModel to writer/
A pure move. ModelValidation, RigidBodyRefine, RigidBodyGPU, ModelFFT, ModelGrid,
ModelScaling, ModelMaskGPU, ModelScaleGPU and SigmaA - everything that works on an
atomic model - become the JFJochStructureRefinement library, linked by
JFJochImageAnalysis. WriteModel (the placed-model mmCIF/PDB writer) goes to writer/
as its own small JFJochModelWriter target, so JFJochWriter, which a writer-only build
compiles, does not gain a gemmi dependency. Only include paths and CMake lists change.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SVmAWnzCmRKAXVUCdc4iNi
2026-10-07 14:05:37 +02:00

95 lines
4.1 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
#include <array>
#include <cstddef>
#include <vector>
#include <cuda_runtime.h>
#include "../indexing/CUDAMemHelpers.h"
// The rigid body's scale fit on the GPU: what RigidBodyTarget::Residuals does with a gemmi
// Scaling<float> (use_solvent, k_sol and b_sol fixed) - fit_isotropic_b_approximately() followed by
// gemmi's Levenberg-Marquardt, and FitModelScale's k_sol/b_sol grid. The sums over the reflections run
// on the device, accumulated in double in a fixed order; the Levenberg-Marquardt control (the damping,
// the 7x7 solve, the stop rules) is gemmi's own, ported line for line and run on the host.
struct ModelScaleParams {
double k_overall = 1.0;
double b_star[6] = {0, 0, 0, 0, 0, 0}; // gemmi SMat33 order u11 u22 u33 u12 u13 u23
};
struct ModelSolventFit {
double k_sol = 0.35, b_sol = 46.0; // gemmi's Scaling defaults when the fit does not run
double r = 1.0; // ModelScaleReport::r_work_fit
int n_grid = 0;
ModelScaleParams scale; // the winner's k_overall and b_star
};
// One fit's parameters for one launch, and which sums to take over the points.
struct ModelScaleFitState {
double k_overall;
double b_star[6];
double k_sol, b_sol;
int mode;
int column; // the fit's row of |Fcalc + solvent| on the device
};
class ModelScaleGPU {
public:
static size_t DeviceBytes(size_t max_points);
ModelScaleGPU(cudaStream_t stream, size_t max_points);
// Per zone: the Scaling points in gemmi prepare_points() order, adp_symmetry_constraints(sg) rows
// and the cell's fractionalization matrix (UnitCell::frac.mat, row-major). n <= max_points.
void SetPoints(const std::vector<std::array<int, 3>> &hkl,
const std::vector<double> &stol2,
const std::vector<float> &fobs,
const std::vector<float> &sigma,
const std::vector<std::array<double, 6>> &constraints,
const double frac[9]);
// fit_isotropic_b_approximately() + fit_parameters() at a fixed solvent pair, from k_overall = 1 and
// b_star = 0. d_fcmol, d_fmask: one float2 per point on the device.
ModelScaleParams Fit(const float2 *d_fcmol, const float2 *d_fmask, double k_sol, double b_sol);
// FitModelScale (ModelScaling.cpp) with the default box: the same grid, the same winner. Throws
// ModelScaleGPUTooFewReflections where FitModelScale would chain the grid points (five or fewer reflections for the
// isotropic fit).
ModelSolventFit FitSolvent(const float2 *d_fcmol, const float2 *d_fmask);
private:
struct LevMarRun;
// Queues the copy of `fits` to the device. host_fits_ is written only after the stream has been
// synchronized, which every Reduce() ends with.
void UploadFits(const std::vector<ModelScaleFitState> &fits);
// The sums of the first `nfits` uploaded fits over all points, one row of SLOTS doubles per fit, on
// the host.
const double *Reduce(int nfits);
const double *Sums(const std::vector<ModelScaleFitState> &fits);
void FitBatch(const float2 *d_fcmol, const float2 *d_fmask, std::vector<ModelScaleFitState> &fits);
cudaStream_t stream_;
size_t max_points_;
int n_ = 0;
int n_strong_ = 0; // points fit_isotropic_b_approximately() fits on
int n_params_ = 1; // k_overall + one per constraint row
double constraints_[6][6] = {};
double frac_[9] = {};
CudaDevicePtr<int> hkl_; // 3 per point
CudaDevicePtr<double> stol2_;
CudaDevicePtr<float> fobs_;
CudaDevicePtr<unsigned char> strong_;
CudaDevicePtr<float> f_abs_; // per fit of a batch, per point
CudaDevicePtr<ModelScaleFitState> fits_;
CudaDevicePtr<double> partial_; // per fit, per block, per slot
CudaDevicePtr<double> sums_; // per fit, per slot
CudaHostPtr<ModelScaleFitState> host_fits_;
CudaHostPtr<double> host_sums_;
};