// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #pragma once #include #include #include #include #include #include "CUDAMemHelpers.h" #include "../../common/TableChecksum.h" // Read-only lookup tables that depend only on the detector geometry (pixel -> azimuthal bin, the // per-pixel correction factors, the pixel mask). One analysis engine is built per worker thread, so // each of those used to upload its own copy: on a 18 Mpx detector that is ~220 MB per thread, and // 32 threads spent ~7 GB of device memory on identical data. // // Upload once per GPU instead and hand every engine on that GPU a shared pointer to the same table. // The cache is keyed by (device, key) because a worker thread is pinned round-robin to a device // (pin_gpu()), so on a multi-GPU node each device keeps its own copy - a kernel may only read memory // resident on the device it runs on. `key` identifies the table's source data; use the address of the // host vector that produced it, which lives in the experiment / integration mapping and therefore // outlives every engine. // // A bare address is not enough on its own to say "same table", though: a host buffer can be mutated // in place, or freed and a new one allocated at the same address, and either would hand the caller a // device copy of something else - silently, since the data is only ever read. So the byte length and // a checksum of the bytes actually uploaded are part of the key too. // // The checksum is part of the KEY, so it is computed before the lookup and a cache hit pays for it as // well: the tables are tens of megabytes and one engine is built per worker per pass, which put the // hashing alone at several percent of all CPU samples. The overload below therefore takes a checksum // the caller already has. That does not weaken anything, because it is the buffer's OWNER that // computes it - AzimuthalIntegrationMapping and PixelMask hash their tables whenever they write them // and hand out the result - so the checksum still describes the bytes as they are now. What must not // be done instead is to memoise the checksum on the address here: that is precisely the reused-address // hazard the checksum exists to catch. // // Entries are held weakly, so the tables are released once the last engine using them is gone. namespace jfjoch_cuda_shared_tables { // (device, source address, byte length, checksum of the bytes) using TableKey = std::tuple; struct Registry { std::mutex m; std::map> tables; }; inline Registry ®istry() { static Registry r; return r; } // Not called cuda_err: the .cu files that include this header define their own such helper in an // anonymous namespace, and a second one at global scope would make every call ambiguous. inline void check(cudaError_t val) { if (val != cudaSuccess) throw JFJochException(JFJochExceptionCategory::GPUCUDAError, cudaGetErrorString(val)); } } // Return the device-resident copy of `host` (`count` elements) for the calling thread's GPU, // uploading it on `stream` the first time it is asked for. `host_checksum` is TableChecksum() of // those `count * sizeof(T)` bytes, computed by whoever owns the buffer. template std::shared_ptr> SharedDeviceTable(const void *key, size_t count, const T *host, uint64_t host_checksum, cudaStream_t stream) { int device = 0; jfjoch_cuda_shared_tables::check(cudaGetDevice(&device)); const size_t bytes = count * sizeof(T); const jfjoch_cuda_shared_tables::TableKey table_key{device, key, bytes, host_checksum}; auto ® = jfjoch_cuda_shared_tables::registry(); // The upload happens while the lock is held: another worker must not obtain the pointer before // its content is on the device. std::lock_guard lock(reg.m); if (auto it = reg.tables.find(table_key); it != reg.tables.end()) { if (auto cached = it->second.lock()) return std::static_pointer_cast>(cached); } // Drop entries whose table is gone before adding one. Without this a long session that reloads // masks or remaps geometry accumulates a dead entry per distinct content, for ever. for (auto it = reg.tables.begin(); it != reg.tables.end();) it = it->second.expired() ? reg.tables.erase(it) : std::next(it); // Free on the device that allocated it - the last engine to drop the table may well be a worker // pinned to a different GPU. And synchronously, for the same reason: the thread that drops the // last reference is not the one whose kernels read the table, so a stream-ordered free would be // ordered against the wrong work. The table is allocated once per GPU, so it has nothing to gain // from the pool anyway. std::shared_ptr> table(new CudaDevicePtr(count, CudaAlloc::Synchronous), [device](CudaDevicePtr *p) { int current = 0; cudaGetDevice(¤t); cudaSetDevice(device); delete p; cudaSetDevice(current); }); jfjoch_cuda_shared_tables::check( cudaMemcpyAsync(table->get(), host, bytes, cudaMemcpyHostToDevice, stream)); jfjoch_cuda_shared_tables::check(cudaStreamSynchronize(stream)); reg.tables[table_key] = std::shared_ptr(table); return table; } // Same, for a caller with no checksum of its own to hand over. template std::shared_ptr> SharedDeviceTable(const void *key, size_t count, const T *host, cudaStream_t stream) { return SharedDeviceTable(key, count, host, TableChecksum(host, count * sizeof(T)), stream); }