A rotation run asks the rotation indexer the same question many times: the canonical pass's first pass repeats the rotation-scale probe at the stored angles exactly (identical validation evidence on every set checked), and on a crystal that does not index, pass 2 and every probe repeat pass 1's rescue ladder rung for rung. Each such RunIndexing is an FFT search plus a serial Ceres fixed-point chain, ~1-3 s on the GPU build. RunIndexing is deterministic in its inputs, so its outcome - every member it sets - is now kept process-wide under a key of all of them: the accumulated spots (every field) and their angles, both geometries and the axis, the experiment's indexing settings, cell and space group, and the settings of the pool it indexes with (IndexerThreadPool::Settings). A RotationIndexer asking with the same key takes the outcome. RUGNUX_VERIFY_FIRST_PASS_MEMO recomputes and throws on a difference. md5-identical output on four sets; GPU wall 28.9 -> 26.1 s, 43.9 -> 41.0 s, 30.3 -> 26.9 s and 143.6 -> 120.5 s (the set that does not index); the verify mode found no difference on the last. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
97 lines
4.0 KiB
C++
97 lines
4.0 KiB
C++
// SPDX-FileCopyrightText: 2025 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
#pragma once
|
|
|
|
|
|
#include <thread>
|
|
#include <mutex>
|
|
#include <condition_variable>
|
|
#include <queue>
|
|
#include <functional>
|
|
#include <future>
|
|
#include <vector>
|
|
#include <optional>
|
|
#include <memory>
|
|
#include <latch>
|
|
|
|
#include "../common/JFJochMessages.h"
|
|
#include "../common/DiffractionSpot.h"
|
|
#include "../common/DiffractionExperiment.h"
|
|
#include "Indexer.h"
|
|
|
|
// When a worker builds its indexers.
|
|
//
|
|
// Preconstruct (the default, and what the online service needs): every indexer the requested
|
|
// algorithm could resolve to is built while the pool is being constructed, so it is resident and
|
|
// planned before the first frame arrives. Building one means a cuFFT plan plus ~0.6 GB of device
|
|
// allocation, and the first cuFFT call in a process also pays the library's one-time init (0.3 s
|
|
// measured); jfjoch_broker cannot have any of that land on the live data path, so holding an indexer
|
|
// the resolved algorithm may never dispatch to is an accepted cost there.
|
|
//
|
|
// OnFirstUse (offline batch - rugnux, jfjoch_viewer): nothing is latency-critical, so the indexer is
|
|
// built on the first frame that needs it. The algorithm is only resolved from the frame's
|
|
// DiffractionExperiment (rotation -> FFT, stills with a known cell -> FFBIDX), so preconstructing
|
|
// leaves each worker holding a fully allocated FFTIndexerGPU - 0.6 GB of cuFFT plan and histograms -
|
|
// that can never be dispatched to.
|
|
enum class IndexerConstruction { Preconstruct, OnFirstUse };
|
|
|
|
// Force cuFFT's one-time library initialisation - 0.3 s measured, paid by whichever call happens to
|
|
// be first - so it can be got out of the way on a background thread at startup rather than landing in
|
|
// the middle of the first pass with nothing overlapping it. No-op without a GPU.
|
|
void WarmUpCuFFT();
|
|
|
|
class IndexerThread {
|
|
struct TaskInput {
|
|
const DiffractionExperiment &experiment;
|
|
const std::vector<Coord> &recip;
|
|
const bool severity_only;
|
|
};
|
|
|
|
// Held by value: with IndexerConstruction::OnFirstUse the worker builds its indexer long after
|
|
// the constructor returned, and pools are routinely built from a temporary - for instance
|
|
// IndexerThreadPool(experiment.GetIndexingSettings()), which returns by value.
|
|
const IndexingSettings settings_;
|
|
const IndexerConstruction construction_;
|
|
|
|
bool stop = false;
|
|
enum class TaskState {STARTING, IDLE, READY, RUNNING, COMPLETED, ERROR} state = TaskState::STARTING;
|
|
std::mutex m;
|
|
std::condition_variable c_running;
|
|
std::condition_variable c_start;
|
|
std::condition_variable c_done;
|
|
std::unique_ptr<IndexerResult> result = nullptr;
|
|
std::unique_ptr<TaskInput> task_input = nullptr;
|
|
std::thread worker_thread;
|
|
|
|
void Worker(int threadid);
|
|
public:
|
|
IndexerThread(const IndexingSettings& settings, int threadid, IndexerConstruction construction);
|
|
~IndexerThread();
|
|
std::unique_ptr<IndexerResult> Run(const DiffractionExperiment &experiment, const std::vector<Coord> &recip,
|
|
bool severity_only = false);
|
|
void Finalize();
|
|
};
|
|
|
|
class IndexerThreadPool {
|
|
std::mutex m;
|
|
std::condition_variable c;
|
|
std::vector<uint8_t> worker_busy;
|
|
size_t worker_free_count;
|
|
std::vector<std::unique_ptr<IndexerThread>> tasks;
|
|
const int64_t viable_cell_min_spots;
|
|
const bool blocking;
|
|
const IndexingSettings settings_;
|
|
int GetFreeWorker();
|
|
public:
|
|
IndexerThreadPool(const IndexingSettings& settings,
|
|
IndexerConstruction construction = IndexerConstruction::Preconstruct);
|
|
// The settings the pool's indexers were built with.
|
|
[[nodiscard]] const IndexingSettings &Settings() const { return settings_; }
|
|
// severity_only skips indexing and produces just the spindle severity - see Indexer::Run.
|
|
IndexerResult Run(const DiffractionExperiment& experiment, const std::vector<Coord>& recip,
|
|
bool severity_only = false);
|
|
};
|
|
|
|
|
|
|