ModelValidation: use every visible GPU - the null's engines and the second validation
The rigid-body pool put all of its engines on the calling thread's current card, so a validation used one GPU whatever the machine had. Now, by fixed rules decided up front and never by momentary free memory: - RigidBodyGPUPool::Create puts engine i on card (d + i) % count, d being the calling thread's card; the pool restores that card afterwards (an engine's constructor sets its own) and an engine is released on its own card. - The threads that run the null's replicates are pinned with pin_gpu (which also binds them to the card's NUMA node where that is enabled), replicate thread t to card (d + 1 + t) % count, and Acquire() hands a thread an idle engine on its own card where there is one, any other otherwise. With one card this is exactly the previous back-of-the-list choice. - The validation started on the forecast runs on card 1 % count, so with two cards or more it is on the other card from the first; the up-front memory rule becomes: twice the planned engine bytes within a quarter of the cards' total memory taken together. The engines' kernels are deterministic (no floating-point atomics) and an engine's result does not depend on which engine it is, so on cards of one model the numbers are those of one card; what changes is only where the work runs. On this one-card workstation the multi-card path cannot be exercised: md5 of p.mtz and the validation outputs are identical to the base, and the card-count arithmetic was checked by reading only. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SVmAWnzCmRKAXVUCdc4iNi
This commit is contained in:
@@ -810,10 +810,11 @@ ModelValidationResult Validate(const std::vector<MergedReflection> &merged,
|
||||
|| !(result.change_of_basis_op == gemmi::Op::identity());
|
||||
if (schedule.on_forecast) {
|
||||
// Two validations at once take twice the engines, and are allowed where twice what this one
|
||||
// asked for fits the share of the card one validation may take (RigidBodyGPUPool::Create).
|
||||
// asked for fits the share of the cards one validation may take (RigidBodyGPUPool::Create).
|
||||
bool may_run_beside = true;
|
||||
if (rigid_body_pool != nullptr)
|
||||
may_run_beside = 2 * rigid_body_pool->PlannedBytes() <= rigid_body_pool->CardBytes() / 4;
|
||||
may_run_beside = 2 * rigid_body_pool->PlannedBytes()
|
||||
<= static_cast<size_t>(get_gpu_count()) * rigid_body_pool->CardBytes() / 4;
|
||||
schedule.on_forecast({result.change_of_basis_op, indexing.op, result.model_enantiomorph_candidate,
|
||||
may_run_beside});
|
||||
}
|
||||
@@ -953,7 +954,11 @@ ModelValidationResult Validate(const std::vector<MergedReflection> &merged,
|
||||
std::vector<std::future<void>> replicate_threads;
|
||||
if (decision_pending)
|
||||
for (size_t t = 0; t < std::min<size_t>(std::max<size_t>(nthreads, 1), NULL_REPLICATES); t++)
|
||||
replicate_threads.push_back(std::async(std::launch::async, [&] {
|
||||
replicate_threads.push_back(std::async(std::launch::async, [&, t] {
|
||||
// Each on a card in turn from the one after the real fit's, which its engines are
|
||||
// spread the same way over (RigidBodyGPUPool::Create).
|
||||
if (rigid_body_pool != nullptr)
|
||||
pin_gpu((rigid_body_pool->Device() + 1 + static_cast<int>(t)) % get_gpu_count());
|
||||
for (int i = next_replicate++; i < NULL_REPLICATES; i = next_replicate++)
|
||||
run_replicate(i);
|
||||
}));
|
||||
|
||||
@@ -209,11 +209,16 @@ std::unique_ptr<RigidBodyGPUPool> RigidBodyGPUPool::Create(const gemmi::Model &m
|
||||
const size_t budget = std::min(total / 4, free > HEADROOM ? free - HEADROOM : 0);
|
||||
const size_t fit = bytes > 0 ? budget / bytes : 0;
|
||||
const size_t want = std::min<size_t>(fit, std::max<size_t>(max_engines, 1));
|
||||
// Engine i on the i-th card from the calling thread's, so a validation uses every card there is;
|
||||
// the engines give the same bits on any card of one model, so this changes only where they run.
|
||||
const int device = RigidBodyGPUEngine::CurrentDevice();
|
||||
const int cards = get_gpu_count();
|
||||
pool->device_ = device;
|
||||
pool->planned_bytes_ = bytes * std::max<size_t>(max_engines, 1);
|
||||
pool->card_bytes_ = total;
|
||||
for (size_t i = 0; i < want; i++)
|
||||
pool->engines_.push_back(std::make_unique<RigidBodyGPUEngine>(cap, device));
|
||||
pool->engines_.push_back(std::make_unique<RigidBodyGPUEngine>(cap, (device + static_cast<int>(i)) % cards));
|
||||
set_gpu(device); // each engine was made on its own card
|
||||
if (pool->engines_.empty()) {
|
||||
logger.Info("Model validation: the rigid body runs on the CPU - one GPU engine needs {:.0f} MB and the "
|
||||
"budget is {:.0f} MB", bytes / 1e6, budget / 1e6);
|
||||
@@ -264,10 +269,19 @@ const RigidBodyGPUZone &RigidBodyGPUPool::Zone(const gemmi::Model &model, const
|
||||
}
|
||||
|
||||
RigidBodyGPUEngine &RigidBodyGPUPool::Acquire() {
|
||||
// One on the calling thread's own card where one is idle - the threads that drive a validation are
|
||||
// pinned to the cards in turn - and any other otherwise.
|
||||
const int device = RigidBodyGPUEngine::CurrentDevice();
|
||||
std::unique_lock lock(m_);
|
||||
cv_.wait(lock, [this] { return !idle_.empty(); });
|
||||
RigidBodyGPUEngine *e = idle_.back();
|
||||
idle_.pop_back();
|
||||
size_t pick = idle_.size() - 1;
|
||||
for (size_t i = idle_.size(); i-- > 0;)
|
||||
if (idle_[i]->Device() == device) {
|
||||
pick = i;
|
||||
break;
|
||||
}
|
||||
RigidBodyGPUEngine *e = idle_[pick];
|
||||
idle_.erase(idle_.begin() + static_cast<std::ptrdiff_t>(pick));
|
||||
return *e;
|
||||
}
|
||||
|
||||
|
||||
@@ -781,7 +781,18 @@ RigidBodyGPUEngine::RigidBodyGPUEngine(const RigidBodyGPUCapacity &capacity, int
|
||||
impl_ = std::make_unique<RigidBodyGPUEngineImpl>(capacity, device);
|
||||
}
|
||||
|
||||
RigidBodyGPUEngine::~RigidBodyGPUEngine() = default;
|
||||
// Released on its own card, which need not be the one the destroying thread is on.
|
||||
RigidBodyGPUEngine::~RigidBodyGPUEngine() {
|
||||
int current = 0;
|
||||
cudaGetDevice(¤t);
|
||||
cudaSetDevice(impl_->device);
|
||||
impl_.reset();
|
||||
cudaSetDevice(current);
|
||||
}
|
||||
|
||||
int RigidBodyGPUEngine::Device() const {
|
||||
return impl_->device;
|
||||
}
|
||||
|
||||
void RigidBodyGPUEngine::SetBody(const std::vector<std::array<double, 3>> &relative) {
|
||||
RigidBodyGPUEngineImpl &e = *impl_;
|
||||
|
||||
@@ -49,6 +49,8 @@ public:
|
||||
~RigidBodyGPUPool();
|
||||
|
||||
size_t Engines() const { return engines_.size(); }
|
||||
// The card the pool was made from, which its first engine is on and the others count on from.
|
||||
int Device() const { return device_; }
|
||||
// What max_engines engines take, however many fitted, and the memory of the card: the sizes a
|
||||
// caller decides by whether a second validation may run beside this one.
|
||||
size_t PlannedBytes() const { return planned_bytes_; }
|
||||
@@ -68,6 +70,7 @@ public:
|
||||
|
||||
private:
|
||||
RigidBodyGPUPool() = default;
|
||||
int device_ = 0;
|
||||
size_t planned_bytes_ = 0;
|
||||
size_t card_bytes_ = 0;
|
||||
std::vector<std::unique_ptr<RigidBodyGPUEngine>> engines_;
|
||||
|
||||
@@ -108,6 +108,7 @@ public:
|
||||
|
||||
RigidBodyGPUEngine(const RigidBodyGPUCapacity &capacity, int device);
|
||||
~RigidBodyGPUEngine();
|
||||
int Device() const;
|
||||
|
||||
// Per fit: each atom's position relative to the model centroid, which the placements rotate.
|
||||
void SetBody(const std::vector<std::array<double, 3>> &relative);
|
||||
|
||||
@@ -3249,7 +3249,8 @@ bool Rugnux::ScaleMergeAndSymmetry(PipelineLocals &p) {
|
||||
report_shell_d_min.push_back(sh.d_min);
|
||||
// The validation in the model's setting further down depends on this one only through the
|
||||
// setting and the indexing it settles, and both are forecast well before this one has finished
|
||||
// (ModelFrameForecast). So it is started on that forecast, beside this one, and kept only where the decision is the one forecast; its log is held,
|
||||
// (ModelFrameForecast). So it is started on that forecast, beside this one - on the next card
|
||||
// where there is one - and kept only where the decision is the one forecast; its log is held,
|
||||
// and its files wait, until then. On any other decision it is dropped, unwritten. The relabelling
|
||||
// it runs on is the one AdoptModelFrame and relabel_output below make, made here on a copy.
|
||||
// A model asserting the other enantiomorph is not forecast: whether the label is taken is
|
||||
@@ -3274,6 +3275,8 @@ bool Rugnux::ScaleMergeAndSymmetry(PipelineLocals &p) {
|
||||
rfree_fraction = experiment_.GetScalingSettings().GetRfreeFraction(),
|
||||
wavelength = experiment_.GetWavelength_A(),
|
||||
gate = ahead.write.get_future().share()]() mutable {
|
||||
if (const int32_t cards = get_gpu_count(); cards > 0)
|
||||
pin_gpu(1 % cards);
|
||||
if (!(ahead.indexing_op == gemmi::Op::identity()))
|
||||
merged = ReindexMergedIntoAsu(merged, ahead.indexing_op, sg, friedel);
|
||||
gemmi::Op to_model = ahead.change_of_basis_op;
|
||||
|
||||
Reference in New Issue
Block a user