Files
leonarski_f 84228bf8be
Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
v1.0.0-rc.173 (#83)
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports.
* jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls.
* Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results.
* Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable.
* Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate.
* Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do.
* Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence.
* Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags.
* Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check.
* Rugnux: Clear error messages when a data set needs more GPU or host memory than is available.

Reviewed-on: #83
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-09-29 15:57:32 +02:00

185 lines
8.0 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
// Checks that the results which depend on the architecture, the compiler or the standard library are
// the same everywhere: the bitshuffle/LZ4 path (hperf SSE2/AVX2 on x86, NEON on aarch64, SSE2 under
// MSVC) and an end-to-end run of the analysis. Golden values were taken on x86-64 Linux; an arm64
// or MSVC build that disagrees has a portability defect, not a new reference.
#include <catch2/catch_all.hpp>
#include <cstring>
#include <thread>
#include "TestData.h"
#include "../compression/JFJochCompressor.h"
#include "../compression/JFJochDecompress.h"
#include "../common/JFJochMath.h"
#include "../writer/HDF5Objects.h"
#include "../image_analysis/indexing/IndexerFactory.h"
#include "../reader/JFJochHDF5Reader.h"
#include "../rugnux/Rugnux.h"
namespace {
// FNV-1a, spelled out: std::hash is not the same function on every standard library.
uint64_t Fnv1a(const void *data, size_t size) {
auto p = static_cast<const uint8_t *>(data);
uint64_t h = 1469598103934665603ULL;
for (size_t i = 0; i < size; i++) {
h ^= p[i];
h *= 1099511628211ULL;
}
return h;
}
uint64_t SplitMix64(uint64_t &state) {
uint64_t z = (state += 0x9e3779b97f4a7c15ULL);
z = (z ^ (z >> 30)) * 0xbf58476d1ce4e5b9ULL;
z = (z ^ (z >> 27)) * 0x94d049bb133111ebULL;
return z ^ (z >> 31);
}
}
// compression_benchmark.h5 holds 25 int16 frames, bitshuffle/LZ4 (HDF5 filter 32008). One frame is
// read three ways - the raw chunk decoded by the hperf path and by the classic bitshuffle, and
// through the HDF5 filter - and all three must give the frame whose hash was recorded on x86.
TEST_CASE("Portable_BSLZ4_BenchmarkFrameGoldenHash", "[portable][compression]") {
RegisterHDF5Filter();
HDF5ReadOnlyFile file("../../tests/test_data/compression_benchmark.h5");
HDF5DataSet dataset(file, "/entry/data/data");
HDF5DataSpace file_space(dataset);
const auto dims = file_space.GetDimensions();
REQUIRE(dims.size() == 3);
const size_t npixel = dims[1] * dims[2];
const std::pair<hsize_t, uint64_t> golden[] = {
{0, 0xaffea29c511b6ec2ULL},
{24, 0xe46913009c95a1f1ULL}
};
for (const auto &[frame, hash]: golden) {
std::vector<uint8_t> chunk;
dataset.ReadDirectChunk(chunk, {frame, 0, 0});
std::vector<int16_t> hperf, classic, filtered(npixel);
JFJochDecompress(hperf, CompressionAlgorithm::BSHUF_LZ4, chunk, npixel, true);
JFJochDecompress(classic, CompressionAlgorithm::BSHUF_LZ4, chunk, npixel, false);
dataset.ReadVector(filtered, {frame, 0, 0}, {1, dims[1], dims[2]});
CHECK(Fnv1a(hperf.data(), hperf.size() * sizeof(int16_t)) == hash);
CHECK(hperf == classic);
CHECK(hperf == filtered);
}
}
// The encoder must write the same bytes everywhere: a file is decoded on a different machine from
// the one that wrote it, and the header of every block is big-endian regardless of the host.
TEST_CASE("Portable_BSLZ4_EncodeGoldenHash", "[portable][compression]") {
uint64_t state = 42;
std::vector<uint32_t> image(1024 * 1024 + 13); // a tail that is not a whole block
for (auto &v: image)
v = (SplitMix64(state) % 16 == 0) ? SplitMix64(state) % 200 : 0;
JFJochBitShuffleCompressor compressor(CompressionAlgorithm::BSHUF_LZ4);
auto compressed = compressor.Compress(image);
CHECK(compressed.size() == 562349);
CHECK(Fnv1a(compressed.data(), compressed.size()) == 0x874e45eb17a51ed2ULL);
std::vector<uint32_t> decoded;
JFJochDecompress(decoded, CompressionAlgorithm::BSHUF_LZ4, compressed, image.size());
CHECK(decoded == image);
}
// One bitshuffle block through the shipped selector (hperf on x86, classic NEON on aarch64) against
// the classic reference transform, over the element sizes and block tails the file format allows.
TEST_CASE("Portable_BitShuffleBlock_MatchesReference", "[portable][compression]") {
uint64_t state = 7;
for (size_t elem: {1, 2, 4, 8}) {
for (size_t size: {8, 16, 24, 136, 1032, 2040, 2048, 8192}) {
for (int pattern = 0; pattern < 3; pattern++) {
std::vector<char> orig(size * elem), ref(size * elem), enc(size * elem),
scratch(size * elem), out(size * elem);
for (size_t i = 0; i < orig.size(); i++) {
if (pattern == 0) orig[i] = static_cast<char>(SplitMix64(state)); // noise
else if (pattern == 1) orig[i] = (i % elem == 0) ? static_cast<char>(i % 7) : 0; // counts
else orig[i] = static_cast<char>((i % elem == elem - 1) ? 0x80 : 0); // sign bit
}
INFO("elem " << elem << " size " << size << " pattern " << pattern);
REQUIRE(bshuf_trans_bit_elem(orig.data(), ref.data(), size, elem) >= 0);
REQUIRE(JFJochBitShuffleBlock(enc.data(), orig.data(), scratch.data(), size, elem) >= 0);
CHECK(enc == ref);
REQUIRE(JFJochBitUnshuffleBlock(out.data(), ref.data(), scratch.data(), size, elem) >= 0);
CHECK(out == orig);
}
}
}
}
// The CPU indexer (FFTW, NEON codelets on arm64) on a synthetic orthorhombic lattice. Named
// explicitly rather than through Auto, so a CUDA build on a runner without a GPU tests it too.
TEST_CASE("Portable_FFTWIndexer_SyntheticLattice", "[portable][Indexing]") {
const UnitCell uc(39, 45, 78, 90, 90, 90);
const CrystalLattice cl(uc);
DiffractionExperiment experiment;
IndexingSettings settings;
settings.Algorithm(IndexingAlgorithmEnum::FFTW).FFT_MaxUnitCell_A(250.0).FFT_HighResolution_A(2 * PI / 3.0);
experiment.ImportIndexingSettings(settings).SetUnitCell(uc);
auto indexer = CreateIndexer(experiment);
REQUIRE(indexer);
std::vector<Coord> spots;
for (int h = -2; h < 10; h++)
for (int k = -5; k < 10; k++)
for (int l = -3; l < 10; l++)
spots.push_back(h * cl.Astar() + k * cl.Bstar() + l * cl.Cstar());
const auto result = indexer->Run(spots);
REQUIRE(result.lattice.size() == 1);
const auto out = result.lattice[0].GetUnitCell();
std::array<double, 3> lengths = {out.a, out.b, out.c};
std::sort(lengths.begin(), lengths.end());
CHECK(lengths[0] == Catch::Approx(39.0));
CHECK(lengths[1] == Catch::Approx(45.0));
CHECK(lengths[2] == Catch::Approx(78.0));
CHECK(out.alpha == Catch::Approx(90.0));
CHECK(out.beta == Catch::Approx(90.0));
CHECK(out.gamma == Catch::Approx(90.0));
}
// End to end on the git-LFS rotation dataset (tests/data, see README.md there): HDF5 read through
// external links, CPU spot finding, the default indexer (FFTW on a CPU build), refinement and
// integration, on a short wedge. SKIPs when LFS was not pulled.
TEST_CASE("Portable_Rugnux_RotationWedge", "[portable][lfs]") {
const auto master = jfjoch_test::LargeDataFile("rotation_master.h5");
if (!master)
SKIP("rotation_master.h5 not available (git-lfs data not pulled)");
RegisterHDF5Filter();
JFJochHDF5Reader reader;
REQUIRE_NOTHROW(reader.ReadFile(*master));
auto dataset = reader.GetDataset();
REQUIRE(dataset);
DiffractionExperiment experiment(dataset->experiment);
IndexingSettings indexing;
indexing.Algorithm(IndexingAlgorithmEnum::Auto);
experiment.ImportIndexingSettings(indexing);
ProcessConfig config;
config.mode = ProcessMode::FullAnalysis;
config.nthreads = std::max(1u, std::thread::hardware_concurrency());
config.end_image = 4;
config.spot_finding = DiffractionExperiment::DefaultDataProcessingSettings();
config.spot_finding.indexing = true;
Rugnux process(reader, experiment, *dataset->pixel_mask, config);
ProcessResult result;
REQUIRE_NOTHROW(result = process.Run());
CHECK(result.images_processed == 4);
REQUIRE(result.indexing_rate.has_value());
CHECK(result.indexing_rate.value() > 0.5f);
reader.Close();
}