// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only // Checks that the results which depend on the architecture, the compiler or the standard library are // the same everywhere: the bitshuffle/LZ4 path (hperf SSE2/AVX2 on x86, NEON on aarch64, SSE2 under // MSVC) and an end-to-end run of the analysis. Golden values were taken on x86-64 Linux; an arm64 // or MSVC build that disagrees has a portability defect, not a new reference. #include #include #include #include "TestData.h" #include "../compression/JFJochCompressor.h" #include "../compression/JFJochDecompress.h" #include "../common/JFJochMath.h" #include "../writer/HDF5Objects.h" #include "../image_analysis/indexing/IndexerFactory.h" #include "../reader/JFJochHDF5Reader.h" #include "../rugnux/Rugnux.h" namespace { // FNV-1a, spelled out: std::hash is not the same function on every standard library. uint64_t Fnv1a(const void *data, size_t size) { auto p = static_cast(data); uint64_t h = 1469598103934665603ULL; for (size_t i = 0; i < size; i++) { h ^= p[i]; h *= 1099511628211ULL; } return h; } uint64_t SplitMix64(uint64_t &state) { uint64_t z = (state += 0x9e3779b97f4a7c15ULL); z = (z ^ (z >> 30)) * 0xbf58476d1ce4e5b9ULL; z = (z ^ (z >> 27)) * 0x94d049bb133111ebULL; return z ^ (z >> 31); } } // compression_benchmark.h5 holds 25 int16 frames, bitshuffle/LZ4 (HDF5 filter 32008). One frame is // read three ways - the raw chunk decoded by the hperf path and by the classic bitshuffle, and // through the HDF5 filter - and all three must give the frame whose hash was recorded on x86. TEST_CASE("Portable_BSLZ4_BenchmarkFrameGoldenHash", "[portable][compression]") { RegisterHDF5Filter(); HDF5ReadOnlyFile file("../../tests/test_data/compression_benchmark.h5"); HDF5DataSet dataset(file, "/entry/data/data"); HDF5DataSpace file_space(dataset); const auto dims = file_space.GetDimensions(); REQUIRE(dims.size() == 3); const size_t npixel = dims[1] * dims[2]; const std::pair golden[] = { {0, 0xaffea29c511b6ec2ULL}, {24, 0xe46913009c95a1f1ULL} }; for (const auto &[frame, hash]: golden) { std::vector chunk; dataset.ReadDirectChunk(chunk, {frame, 0, 0}); std::vector hperf, classic, filtered(npixel); JFJochDecompress(hperf, CompressionAlgorithm::BSHUF_LZ4, chunk, npixel, true); JFJochDecompress(classic, CompressionAlgorithm::BSHUF_LZ4, chunk, npixel, false); dataset.ReadVector(filtered, {frame, 0, 0}, {1, dims[1], dims[2]}); CHECK(Fnv1a(hperf.data(), hperf.size() * sizeof(int16_t)) == hash); CHECK(hperf == classic); CHECK(hperf == filtered); } } // The encoder must write the same bytes everywhere: a file is decoded on a different machine from // the one that wrote it, and the header of every block is big-endian regardless of the host. TEST_CASE("Portable_BSLZ4_EncodeGoldenHash", "[portable][compression]") { uint64_t state = 42; std::vector image(1024 * 1024 + 13); // a tail that is not a whole block for (auto &v: image) v = (SplitMix64(state) % 16 == 0) ? SplitMix64(state) % 200 : 0; JFJochBitShuffleCompressor compressor(CompressionAlgorithm::BSHUF_LZ4); auto compressed = compressor.Compress(image); CHECK(compressed.size() == 562349); CHECK(Fnv1a(compressed.data(), compressed.size()) == 0x874e45eb17a51ed2ULL); std::vector decoded; JFJochDecompress(decoded, CompressionAlgorithm::BSHUF_LZ4, compressed, image.size()); CHECK(decoded == image); } // One bitshuffle block through the shipped selector (hperf on x86, classic NEON on aarch64) against // the classic reference transform, over the element sizes and block tails the file format allows. TEST_CASE("Portable_BitShuffleBlock_MatchesReference", "[portable][compression]") { uint64_t state = 7; for (size_t elem: {1, 2, 4, 8}) { for (size_t size: {8, 16, 24, 136, 1032, 2040, 2048, 8192}) { for (int pattern = 0; pattern < 3; pattern++) { std::vector orig(size * elem), ref(size * elem), enc(size * elem), scratch(size * elem), out(size * elem); for (size_t i = 0; i < orig.size(); i++) { if (pattern == 0) orig[i] = static_cast(SplitMix64(state)); // noise else if (pattern == 1) orig[i] = (i % elem == 0) ? static_cast(i % 7) : 0; // counts else orig[i] = static_cast((i % elem == elem - 1) ? 0x80 : 0); // sign bit } INFO("elem " << elem << " size " << size << " pattern " << pattern); REQUIRE(bshuf_trans_bit_elem(orig.data(), ref.data(), size, elem) >= 0); REQUIRE(JFJochBitShuffleBlock(enc.data(), orig.data(), scratch.data(), size, elem) >= 0); CHECK(enc == ref); REQUIRE(JFJochBitUnshuffleBlock(out.data(), ref.data(), scratch.data(), size, elem) >= 0); CHECK(out == orig); } } } } // The CPU indexer (FFTW, NEON codelets on arm64) on a synthetic orthorhombic lattice. Named // explicitly rather than through Auto, so a CUDA build on a runner without a GPU tests it too. TEST_CASE("Portable_FFTWIndexer_SyntheticLattice", "[portable][Indexing]") { const UnitCell uc(39, 45, 78, 90, 90, 90); const CrystalLattice cl(uc); DiffractionExperiment experiment; IndexingSettings settings; settings.Algorithm(IndexingAlgorithmEnum::FFTW).FFT_MaxUnitCell_A(250.0).FFT_HighResolution_A(2 * PI / 3.0); experiment.ImportIndexingSettings(settings).SetUnitCell(uc); auto indexer = CreateIndexer(experiment); REQUIRE(indexer); std::vector spots; for (int h = -2; h < 10; h++) for (int k = -5; k < 10; k++) for (int l = -3; l < 10; l++) spots.push_back(h * cl.Astar() + k * cl.Bstar() + l * cl.Cstar()); const auto result = indexer->Run(spots); REQUIRE(result.lattice.size() == 1); const auto out = result.lattice[0].GetUnitCell(); std::array lengths = {out.a, out.b, out.c}; std::sort(lengths.begin(), lengths.end()); CHECK(lengths[0] == Catch::Approx(39.0)); CHECK(lengths[1] == Catch::Approx(45.0)); CHECK(lengths[2] == Catch::Approx(78.0)); CHECK(out.alpha == Catch::Approx(90.0)); CHECK(out.beta == Catch::Approx(90.0)); CHECK(out.gamma == Catch::Approx(90.0)); } // End to end on the git-LFS rotation dataset (tests/data, see README.md there): HDF5 read through // external links, CPU spot finding, the default indexer (FFTW on a CPU build), refinement and // integration, on a short wedge. SKIPs when LFS was not pulled. TEST_CASE("Portable_Rugnux_RotationWedge", "[portable][lfs]") { const auto master = jfjoch_test::LargeDataFile("rotation_master.h5"); if (!master) SKIP("rotation_master.h5 not available (git-lfs data not pulled)"); RegisterHDF5Filter(); JFJochHDF5Reader reader; REQUIRE_NOTHROW(reader.ReadFile(*master)); auto dataset = reader.GetDataset(); REQUIRE(dataset); DiffractionExperiment experiment(dataset->experiment); IndexingSettings indexing; indexing.Algorithm(IndexingAlgorithmEnum::Auto); experiment.ImportIndexingSettings(indexing); ProcessConfig config; config.mode = ProcessMode::FullAnalysis; config.nthreads = std::max(1u, std::thread::hardware_concurrency()); config.end_image = 4; config.spot_finding = DiffractionExperiment::DefaultDataProcessingSettings(); config.spot_finding.indexing = true; Rugnux process(reader, experiment, *dataset->pixel_mask, config); ProcessResult result; REQUIRE_NOTHROW(result = process.Run()); CHECK(result.images_processed == 4); REQUIRE(result.indexing_rate.has_value()); CHECK(result.indexing_rate.value() > 0.5f); reader.Close(); }