From ed2d557bb7cc6b0e0f355e83db1eb7e03a7dd2cd Mon Sep 17 00:00:00 2001 From: Filip Leonarski Date: Fri, 9 Oct 2026 12:47:04 +0200 Subject: [PATCH] ModelValidation: a CUDA failure fails the validation instead of restarting it on the CPU A CUDA error on the validation's GPU path (rigid body, structure factors, maps) used to start the whole validation again on the CPU - a migration part way through, which makes how long a run takes, and in principle what its rigid bodies converge to, depend on a device fault. It now ends the validation with ok = false and a logged reason ("the GPU failed during the validation (...)"), which the run reports as MODEL_NOT_VALIDATED. A validation that did not finish decides nothing, so the reflection files are those of a run without a model. Where there is no GPU at all, or the up-front rule sends the work to the CPU, nothing changes. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01SVmAWnzCmRKAXVUCdc4iNi --- .../structure_refinement/ModelValidation.cpp | 16 ++++++++++------ .../structure_refinement/RigidBodyGPU.h | 2 +- 2 files changed, 11 insertions(+), 7 deletions(-) diff --git a/image_analysis/structure_refinement/ModelValidation.cpp b/image_analysis/structure_refinement/ModelValidation.cpp index 304484ffb..95a4550a3 100644 --- a/image_analysis/structure_refinement/ModelValidation.cpp +++ b/image_analysis/structure_refinement/ModelValidation.cpp @@ -1570,20 +1570,24 @@ ModelValidationResult ValidateAgainstModel(const std::vector & double wavelength_A, const std::vector &report_shell_d_min) { #ifdef JFJOCH_USE_CUDA - // A CUDA failure on the GPU path does not end the run: the validation is started again from the - // model as read, on the CPU throughout. It is re-runnable, and a dead model check should not take a - // finished merge with it. + // A CUDA failure ends the validation, not the run, and nothing is moved to the CPU part way through: + // the validation reports why it did not finish, and a validation that did not finish decides nothing, + // so the reflection files are those of a run without a model. try { return Validate(merged, cell, model_path, output_prefix, logger, data_space_group, probe_indexing_ambiguity, nthreads, wavelength_A, report_shell_d_min, true); } catch (const RigidBodyGPUFailure &e) { cuda_clear_error(); - logger.Warning("Model validation: the rigid body failed on the GPU ({}); validating again on the CPU", - e.what()); + ModelValidationResult failed; + failed.model_path = model_path; + failed.failure_reason = fmt::format("the GPU failed during the validation ({})", e.what()); + logger.Error("Model validation: {}", failed.failure_reason); + return failed; } -#endif +#else return Validate(merged, cell, model_path, output_prefix, logger, data_space_group, probe_indexing_ambiguity, nthreads, wavelength_A, report_shell_d_min, false); +#endif } namespace { diff --git a/image_analysis/structure_refinement/RigidBodyGPU.h b/image_analysis/structure_refinement/RigidBodyGPU.h index bf312db45..c9c38d855 100644 --- a/image_analysis/structure_refinement/RigidBodyGPU.h +++ b/image_analysis/structure_refinement/RigidBodyGPU.h @@ -25,7 +25,7 @@ class RigidBodyGPUEngine; struct RigidBodyGPUZone; // A CUDA failure inside the rigid body, or in model validation's structure factors and maps -// (ModelStructureFactorsGPU). Model validation catches it and starts again on the CPU. +// (ModelStructureFactorsGPU). Model validation catches it and reports the validation as failed. class RigidBodyGPUFailure : public std::runtime_error { public: explicit RigidBodyGPUFailure(const std::string &what) : std::runtime_error(what) {}