Files
Jungfraujoch/tests/PortableTest.cpp
T
leonarski_fandClaude Opus 5.5 a8c559bf6b tests: jfjoch_portable_test, quick [portable] checks for the macOS and Windows jobs
A Catch2 executable that builds in the portable configurations (JFJOCH_VIEWER_ONLY /
JFJOCH_RUGNUX_ONLY) and links only what those build - JFJochRugnux, JFJochReader,
JFJochImageAnalysis, JFJochWriter, JFJochCommon - so it can run on the macOS arm64 and
Windows x64 jobs, where the receiver/broker/FPGA/HLS sources are not built and there is no GPU.
EXCLUDE_FROM_ALL, so a product build does not pay for it; catch2 is now made available in the
portable configure as well (it is EXCLUDE_FROM_ALL too).

The cases tagged [portable] cover what depends on the architecture, the compiler or the
standard library: bitshuffle/LZ4/zstd, HDF5 read-back (legacy/VDS/integrated, the direct-chunk
path), miniCBF/marCCD/SMV header parsing, CBOR, CPU spot finding, azimuthal mapping, Bragg
prediction/integration, gemmi MTZ/mmCIF. New in tests/PortableTest.cpp:
- a golden FNV-1a hash of two frames of compression_benchmark.h5, decoded from the raw chunk by
  the hperf and the classic bitshuffle and through the HDF5 filter (x86 hashes
  affea29c511b6ec2 / e46913009c95a1f1);
- a golden hash of a bitshuffle/LZ4 encode (the writer must produce the same bytes everywhere);
- the shipped bitshuffle block selector against the classic reference over elem 1/2/4/8 and
  block tails;
- the FFTW indexer, named explicitly, on a synthetic orthorhombic lattice (the existing FFT
  indexer lattice tests run only under CUDA);
- a 4-frame end-to-end rugnux run on the git-LFS rotation dataset (HDF5 via external links, CPU
  spot finding, indexing, integration); SKIPs when LFS was not pulled.
All 45 take ~4 s on Linux (~2 s without the LFS case), CPU-only build.

M_PI replaced by PI (common/JFJochMath.h) in the two tagged files that used it, as MSVC does not
define M_PI. The CBF gzip test shells out to gzip and is left untagged.

CI: build and run jfjoch_portable_test "[portable]" in build-windows (both variants),
build-rugnux-windows, build-macos-viewer and build-rugnux-macos, after the build and before
packaging.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
2026-09-27 23:44:53 +02:00

185 lines
8.0 KiB
C++

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
// Checks that the results which depend on the architecture, the compiler or the standard library are
// the same everywhere: the bitshuffle/LZ4 path (hperf SSE2/AVX2 on x86, NEON on aarch64, SSE2 under
// MSVC) and an end-to-end run of the analysis. Golden values were taken on x86-64 Linux; an arm64
// or MSVC build that disagrees has a portability defect, not a new reference.
#include <catch2/catch_all.hpp>
#include <cstring>
#include <thread>
#include "TestData.h"
#include "../compression/JFJochCompressor.h"
#include "../compression/JFJochDecompress.h"
#include "../common/JFJochMath.h"
#include "../writer/HDF5Objects.h"
#include "../image_analysis/indexing/IndexerFactory.h"
#include "../reader/JFJochHDF5Reader.h"
#include "../rugnux/Rugnux.h"
namespace {
// FNV-1a, spelled out: std::hash is not the same function on every standard library.
uint64_t Fnv1a(const void *data, size_t size) {
auto p = static_cast<const uint8_t *>(data);
uint64_t h = 1469598103934665603ULL;
for (size_t i = 0; i < size; i++) {
h ^= p[i];
h *= 1099511628211ULL;
}
return h;
}
uint64_t SplitMix64(uint64_t &state) {
uint64_t z = (state += 0x9e3779b97f4a7c15ULL);
z = (z ^ (z >> 30)) * 0xbf58476d1ce4e5b9ULL;
z = (z ^ (z >> 27)) * 0x94d049bb133111ebULL;
return z ^ (z >> 31);
}
}
// compression_benchmark.h5 holds 25 int16 frames, bitshuffle/LZ4 (HDF5 filter 32008). One frame is
// read three ways - the raw chunk decoded by the hperf path and by the classic bitshuffle, and
// through the HDF5 filter - and all three must give the frame whose hash was recorded on x86.
TEST_CASE("Portable_BSLZ4_BenchmarkFrameGoldenHash", "[portable][compression]") {
RegisterHDF5Filter();
HDF5ReadOnlyFile file("../../tests/test_data/compression_benchmark.h5");
HDF5DataSet dataset(file, "/entry/data/data");
HDF5DataSpace file_space(dataset);
const auto dims = file_space.GetDimensions();
REQUIRE(dims.size() == 3);
const size_t npixel = dims[1] * dims[2];
const std::pair<hsize_t, uint64_t> golden[] = {
{0, 0xaffea29c511b6ec2ULL},
{24, 0xe46913009c95a1f1ULL}
};
for (const auto &[frame, hash]: golden) {
std::vector<uint8_t> chunk;
dataset.ReadDirectChunk(chunk, {frame, 0, 0});
std::vector<int16_t> hperf, classic, filtered(npixel);
JFJochDecompress(hperf, CompressionAlgorithm::BSHUF_LZ4, chunk, npixel, true);
JFJochDecompress(classic, CompressionAlgorithm::BSHUF_LZ4, chunk, npixel, false);
dataset.ReadVector(filtered, {frame, 0, 0}, {1, dims[1], dims[2]});
CHECK(Fnv1a(hperf.data(), hperf.size() * sizeof(int16_t)) == hash);
CHECK(hperf == classic);
CHECK(hperf == filtered);
}
}
// The encoder must write the same bytes everywhere: a file is decoded on a different machine from
// the one that wrote it, and the header of every block is big-endian regardless of the host.
TEST_CASE("Portable_BSLZ4_EncodeGoldenHash", "[portable][compression]") {
uint64_t state = 42;
std::vector<uint32_t> image(1024 * 1024 + 13); // a tail that is not a whole block
for (auto &v: image)
v = (SplitMix64(state) % 16 == 0) ? SplitMix64(state) % 200 : 0;
JFJochBitShuffleCompressor compressor(CompressionAlgorithm::BSHUF_LZ4);
auto compressed = compressor.Compress(image);
CHECK(compressed.size() == 562349);
CHECK(Fnv1a(compressed.data(), compressed.size()) == 0x874e45eb17a51ed2ULL);
std::vector<uint32_t> decoded;
JFJochDecompress(decoded, CompressionAlgorithm::BSHUF_LZ4, compressed, image.size());
CHECK(decoded == image);
}
// One bitshuffle block through the shipped selector (hperf on x86, classic NEON on aarch64) against
// the classic reference transform, over the element sizes and block tails the file format allows.
TEST_CASE("Portable_BitShuffleBlock_MatchesReference", "[portable][compression]") {
uint64_t state = 7;
for (size_t elem: {1, 2, 4, 8}) {
for (size_t size: {8, 16, 24, 136, 1032, 2040, 2048, 8192}) {
for (int pattern = 0; pattern < 3; pattern++) {
std::vector<char> orig(size * elem), ref(size * elem), enc(size * elem),
scratch(size * elem), out(size * elem);
for (size_t i = 0; i < orig.size(); i++) {
if (pattern == 0) orig[i] = static_cast<char>(SplitMix64(state)); // noise
else if (pattern == 1) orig[i] = (i % elem == 0) ? static_cast<char>(i % 7) : 0; // counts
else orig[i] = static_cast<char>((i % elem == elem - 1) ? 0x80 : 0); // sign bit
}
INFO("elem " << elem << " size " << size << " pattern " << pattern);
REQUIRE(bshuf_trans_bit_elem(orig.data(), ref.data(), size, elem) >= 0);
REQUIRE(JFJochBitShuffleBlock(enc.data(), orig.data(), scratch.data(), size, elem) >= 0);
CHECK(enc == ref);
REQUIRE(JFJochBitUnshuffleBlock(out.data(), ref.data(), scratch.data(), size, elem) >= 0);
CHECK(out == orig);
}
}
}
}
// The CPU indexer (FFTW, NEON codelets on arm64) on a synthetic orthorhombic lattice. Named
// explicitly rather than through Auto, so a CUDA build on a runner without a GPU tests it too.
TEST_CASE("Portable_FFTWIndexer_SyntheticLattice", "[portable][Indexing]") {
const UnitCell uc(39, 45, 78, 90, 90, 90);
const CrystalLattice cl(uc);
DiffractionExperiment experiment;
IndexingSettings settings;
settings.Algorithm(IndexingAlgorithmEnum::FFTW).FFT_MaxUnitCell_A(250.0).FFT_HighResolution_A(2 * PI / 3.0);
experiment.ImportIndexingSettings(settings).SetUnitCell(uc);
auto indexer = CreateIndexer(experiment);
REQUIRE(indexer);
std::vector<Coord> spots;
for (int h = -2; h < 10; h++)
for (int k = -5; k < 10; k++)
for (int l = -3; l < 10; l++)
spots.push_back(h * cl.Astar() + k * cl.Bstar() + l * cl.Cstar());
const auto result = indexer->Run(spots);
REQUIRE(result.lattice.size() == 1);
const auto out = result.lattice[0].GetUnitCell();
std::array<double, 3> lengths = {out.a, out.b, out.c};
std::sort(lengths.begin(), lengths.end());
CHECK(lengths[0] == Catch::Approx(39.0));
CHECK(lengths[1] == Catch::Approx(45.0));
CHECK(lengths[2] == Catch::Approx(78.0));
CHECK(out.alpha == Catch::Approx(90.0));
CHECK(out.beta == Catch::Approx(90.0));
CHECK(out.gamma == Catch::Approx(90.0));
}
// End to end on the git-LFS rotation dataset (tests/data, see README.md there): HDF5 read through
// external links, CPU spot finding, the default indexer (FFTW on a CPU build), refinement and
// integration, on a short wedge. SKIPs when LFS was not pulled.
TEST_CASE("Portable_Rugnux_RotationWedge", "[portable][lfs]") {
const auto master = jfjoch_test::LargeDataFile("rotation_master.h5");
if (!master)
SKIP("rotation_master.h5 not available (git-lfs data not pulled)");
RegisterHDF5Filter();
JFJochHDF5Reader reader;
REQUIRE_NOTHROW(reader.ReadFile(*master));
auto dataset = reader.GetDataset();
REQUIRE(dataset);
DiffractionExperiment experiment(dataset->experiment);
IndexingSettings indexing;
indexing.Algorithm(IndexingAlgorithmEnum::Auto);
experiment.ImportIndexingSettings(indexing);
ProcessConfig config;
config.mode = ProcessMode::FullAnalysis;
config.nthreads = std::max(1u, std::thread::hardware_concurrency());
config.end_image = 4;
config.spot_finding = DiffractionExperiment::DefaultDataProcessingSettings();
config.spot_finding.indexing = true;
Rugnux process(reader, experiment, *dataset->pixel_mask, config);
ProcessResult result;
REQUIRE_NOTHROW(result = process.Run());
CHECK(result.images_processed == 4);
REQUIRE(result.indexing_rate.has_value());
CHECK(result.indexing_rate.value() > 0.5f);
reader.Close();
}