Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports. * jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls. * Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results. * Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable. * Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate. * Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do. * Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence. * Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags. * Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check. * Rugnux: Clear error messages when a data set needs more GPU or host memory than is available. Reviewed-on: #83 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
95 lines
4.1 KiB
C++
95 lines
4.1 KiB
C++
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <array>
|
|
#include <cstddef>
|
|
#include <vector>
|
|
|
|
#include <cuda_runtime.h>
|
|
|
|
#include "../image_analysis/indexing/CUDAMemHelpers.h"
|
|
|
|
// The rigid body's scale fit on the GPU: what RigidBodyTarget::Residuals does with a gemmi
|
|
// Scaling<float> (use_solvent, k_sol and b_sol fixed) - fit_isotropic_b_approximately() followed by
|
|
// gemmi's Levenberg-Marquardt, and FitModelScale's k_sol/b_sol grid. The sums over the reflections run
|
|
// on the device, accumulated in double in a fixed order; the Levenberg-Marquardt control (the damping,
|
|
// the 7x7 solve, the stop rules) is gemmi's own, ported line for line and run on the host.
|
|
|
|
struct ModelScaleParams {
|
|
double k_overall = 1.0;
|
|
double b_star[6] = {0, 0, 0, 0, 0, 0}; // gemmi SMat33 order u11 u22 u33 u12 u13 u23
|
|
};
|
|
|
|
struct ModelSolventFit {
|
|
double k_sol = 0.35, b_sol = 46.0; // gemmi's Scaling defaults when the fit does not run
|
|
double r = 1.0; // ModelScaleReport::r_work_fit
|
|
int n_grid = 0;
|
|
ModelScaleParams scale; // the winner's k_overall and b_star
|
|
};
|
|
|
|
// One fit's parameters for one launch, and which sums to take over the points.
|
|
struct ModelScaleFitState {
|
|
double k_overall;
|
|
double b_star[6];
|
|
double k_sol, b_sol;
|
|
int mode;
|
|
int column; // the fit's row of |Fcalc + solvent| on the device
|
|
};
|
|
|
|
class ModelScaleGPU {
|
|
public:
|
|
static size_t DeviceBytes(size_t max_points);
|
|
ModelScaleGPU(cudaStream_t stream, size_t max_points);
|
|
|
|
// Per zone: the Scaling points in gemmi prepare_points() order, adp_symmetry_constraints(sg) rows
|
|
// and the cell's fractionalization matrix (UnitCell::frac.mat, row-major). n <= max_points.
|
|
void SetPoints(const std::vector<std::array<int, 3>> &hkl,
|
|
const std::vector<double> &stol2,
|
|
const std::vector<float> &fobs,
|
|
const std::vector<float> &sigma,
|
|
const std::vector<std::array<double, 6>> &constraints,
|
|
const double frac[9]);
|
|
|
|
// fit_isotropic_b_approximately() + fit_parameters() at a fixed solvent pair, from k_overall = 1 and
|
|
// b_star = 0. d_fcmol, d_fmask: one float2 per point on the device.
|
|
ModelScaleParams Fit(const float2 *d_fcmol, const float2 *d_fmask, double k_sol, double b_sol);
|
|
|
|
// FitModelScale (ModelScaling.cpp) with the default box: the same grid, the same winner. Throws
|
|
// ModelScaleGPUTooFewReflections where FitModelScale would chain the grid points (five or fewer reflections for the
|
|
// isotropic fit).
|
|
ModelSolventFit FitSolvent(const float2 *d_fcmol, const float2 *d_fmask);
|
|
|
|
private:
|
|
struct LevMarRun;
|
|
|
|
// Queues the copy of `fits` to the device. host_fits_ is written only after the stream has been
|
|
// synchronized, which every Reduce() ends with.
|
|
void UploadFits(const std::vector<ModelScaleFitState> &fits);
|
|
// The sums of the first `nfits` uploaded fits over all points, one row of SLOTS doubles per fit, on
|
|
// the host.
|
|
const double *Reduce(int nfits);
|
|
const double *Sums(const std::vector<ModelScaleFitState> &fits);
|
|
void FitBatch(const float2 *d_fcmol, const float2 *d_fmask, std::vector<ModelScaleFitState> &fits);
|
|
|
|
cudaStream_t stream_;
|
|
size_t max_points_;
|
|
int n_ = 0;
|
|
int n_strong_ = 0; // points fit_isotropic_b_approximately() fits on
|
|
int n_params_ = 1; // k_overall + one per constraint row
|
|
double constraints_[6][6] = {};
|
|
double frac_[9] = {};
|
|
|
|
CudaDevicePtr<int> hkl_; // 3 per point
|
|
CudaDevicePtr<double> stol2_;
|
|
CudaDevicePtr<float> fobs_;
|
|
CudaDevicePtr<unsigned char> strong_;
|
|
CudaDevicePtr<float> f_abs_; // per fit of a batch, per point
|
|
CudaDevicePtr<ModelScaleFitState> fits_;
|
|
CudaDevicePtr<double> partial_; // per fit, per block, per slot
|
|
CudaDevicePtr<double> sums_; // per fit, per slot
|
|
CudaHostPtr<ModelScaleFitState> host_fits_;
|
|
CudaHostPtr<double> host_sums_;
|
|
};
|