Pre-scan: parallel beam-stop mask, leaner background beam-centre fit
Exact: p.mtz and the pre-scan products (shadow mask, mean projection, defective-pixel mask, ring and capture centres, compared as hashes and hex floats) are bit-identical to rc174 on three in-house rotation sets, GPU and CPU builds. - ShadowFinder::GetMask: the serial parts run in parallel - connected components by row band joined with union-find (both the shadow and the transmitting-arm searches, and the hole fill), ring binning and the harmonic sector gather by blocks, gap bridging by line; ring pixel counts read off the ring offsets. Mean projection filled in parallel. - ShadowFinder host accumulation: one band-locked projection instead of a 20 B/px shard per pre-scan worker (2.7 GB zeroed and folded on a 16M detector); SetShardCount and the shard argument are gone. - FindBeamCenterFromBackground: the usable-pixel test is made once, the in-band pixels are kept in pixel order so the clipping rounds no longer sweep the whole detector, the 67 MB cell map is gone and the per-iteration block fold runs in parallel - same sums, same order. - HotPixelFinder::GetMask: the chance-rate counts in parallel (integers). Measured on a loaded box (load ~25 from other jobs), pre-scan window: GPU 5.9-6.5 s -> 3.2-3.4 s, CPU 8.4-9.0 s -> 6.1-7.4 s. The GPU-build pre-scan now ends with its background spot measurement (CPU spot finder on ~120 frames, ~13 core-s on 8 workers). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01K5K8jvPPbmCrbqnWkddTuB
This commit is contained in:
+24
-14
@@ -8,6 +8,7 @@
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <optional>
|
||||
#include <thread>
|
||||
#include <vector>
|
||||
|
||||
#include "../common/DetectorSetup.h"
|
||||
@@ -168,16 +169,15 @@ TEST_CASE("ShadowFinder_MaskDoesNotDependOnTheThreadCount", "[ShadowFinder]") {
|
||||
CHECK(finder.GetMask(8) == one);
|
||||
}
|
||||
|
||||
// Workers accumulate into shards of their own and the shards are summed when the projection is read,
|
||||
// so which worker saw which frame must not reach the answer - including the maximum, which only one
|
||||
// shard holds when the reflection is on a single frame.
|
||||
TEST_CASE("ShadowFinder_ShardingDoesNotChangeTheProjection", "[ShadowFinder]") {
|
||||
// Workers add their frames concurrently, each starting at a different band of the projection, so
|
||||
// which worker added which frame, and in what order, must not reach the answer - including the
|
||||
// maximum, which one frame alone holds when the reflection is on a single frame.
|
||||
TEST_CASE("ShadowFinder_ConcurrentWorkersDoNotChangeTheProjection", "[ShadowFinder]") {
|
||||
const DiffractionExperiment x = TestExperiment();
|
||||
const PixelMask pixel_mask(x);
|
||||
|
||||
ShadowFinder serial(x, pixel_mask);
|
||||
ShadowFinder sharded(x, pixel_mask);
|
||||
sharded.SetShardCount(4);
|
||||
ShadowFinder concurrent(x, pixel_mask);
|
||||
|
||||
std::vector<std::vector<int32_t>> frames;
|
||||
std::vector<uint8_t> buffer;
|
||||
@@ -185,22 +185,32 @@ TEST_CASE("ShadowFinder_ShardingDoesNotChangeTheProjection", "[ShadowFinder]") {
|
||||
frames.push_back(Scene(/*cross=*/false, /*reflection=*/f == 0));
|
||||
DataMessage msg{};
|
||||
msg.image = CompressedImage(frames.back(), W, H);
|
||||
serial.AddImage(msg, buffer, 0);
|
||||
sharded.AddImage(msg, buffer, static_cast<size_t>(f) % 4);
|
||||
serial.AddImage(msg, buffer);
|
||||
}
|
||||
std::vector<std::thread> workers;
|
||||
for (int t = 0; t < 4; t++)
|
||||
workers.emplace_back([&, t] {
|
||||
std::vector<uint8_t> worker_buffer;
|
||||
for (int f = NFRAMES - 1 - t; f >= 0; f -= 4) {
|
||||
DataMessage msg{};
|
||||
msg.image = CompressedImage(frames[f], W, H);
|
||||
concurrent.AddImage(msg, worker_buffer);
|
||||
}
|
||||
});
|
||||
for (auto &w : workers) w.join();
|
||||
|
||||
CHECK(serial.GetFrameCount() == sharded.GetFrameCount());
|
||||
CHECK(serial.GetFrameCount() == concurrent.GetFrameCount());
|
||||
|
||||
const auto a = serial.GetMeanProjection();
|
||||
const auto b = sharded.GetMeanProjection();
|
||||
const auto b = concurrent.GetMeanProjection();
|
||||
REQUIRE(a.size() == b.size());
|
||||
// NAN marks a pixel nothing counted, and NAN != NAN, so compare the bits rather than the values.
|
||||
CHECK(memcmp(a.data(), b.data(), a.size() * sizeof(float)) == 0);
|
||||
|
||||
// The reflection is on one frame, so its maximum lives in a single shard. If the fold lost it,
|
||||
// the mask would swallow the reflection instead of giving it back.
|
||||
CHECK(serial.GetMask(1) == sharded.GetMask(1));
|
||||
CHECK(sharded.GetMask(1)[I(C - 14, C - 2)] == 0);
|
||||
// The reflection is on one frame, so only that frame's maximum sees it. If it were lost, the mask
|
||||
// would swallow the reflection instead of giving it back.
|
||||
CHECK(serial.GetMask(1) == concurrent.GetMask(1));
|
||||
CHECK(concurrent.GetMask(1)[I(C - 14, C - 2)] == 0);
|
||||
}
|
||||
|
||||
// Four opaque arms and a centred disk: the scene is invariant under a quarter turn, so the mask must
|
||||
|
||||
Reference in New Issue
Block a user