Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports. * jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls. * Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results. * Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable. * Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate. * Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do. * Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence. * Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags. * Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check. * Rugnux: Clear error messages when a data set needs more GPU or host memory than is available. Reviewed-on: #83 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
289 lines
14 KiB
C++
289 lines
14 KiB
C++
// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <vector>
|
|
#include <cstring>
|
|
|
|
#include <bitshuffle/bitshuffle.h>
|
|
#include <bitshuffle/bitshuffle_internals.h>
|
|
#include <lz4/lz4.h>
|
|
#include <zstd.h>
|
|
|
|
#include "BitShuffleBlock.h"
|
|
#include "../compression/CompressionAlgorithmEnum.h"
|
|
|
|
#include "../common/JFJochException.h"
|
|
#include "../common/CompressedImage.h"
|
|
|
|
extern "C" {
|
|
uint64_t bshuf_read_uint64_BE(const void* buf);
|
|
};
|
|
|
|
// Decode the bitshuffle blocks one after another and hand each to block_done(first_element, data,
|
|
// nelements), in element order. With output set, each block is decoded in place into output; with
|
|
// output = nullptr, into a block-sized buffer that is reused for the next block, so a caller that
|
|
// consumes the block straight away reads it from cache and the whole image is never written out.
|
|
template <class F>
|
|
size_t JFJochDecompressHperfBlocks(uint8_t *output,
|
|
CompressionAlgorithm algorithm,
|
|
const uint8_t *source,
|
|
size_t source_size,
|
|
size_t nelements,
|
|
size_t elem_size,
|
|
size_t block_size,
|
|
F &&block_done) {
|
|
if ((algorithm != CompressionAlgorithm::BSHUF_LZ4) &&
|
|
(algorithm != CompressionAlgorithm::BSHUF_ZSTD) &&
|
|
(algorithm != CompressionAlgorithm::BSHUF_ZSTD_RLE) &&
|
|
(algorithm != CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF))
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Algorithm not supported by hperf decompressor");
|
|
|
|
if ((block_size == 0) || ((block_size % BSHUF_BLOCKED_MULT) != 0))
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Invalid block size");
|
|
|
|
std::vector<char> decompressed_block(block_size * elem_size);
|
|
std::vector<char> scratch(block_size * elem_size);
|
|
std::vector<uint8_t> unshuffled(output ? 0 : block_size * elem_size);
|
|
|
|
const uint8_t *src_ptr = source;
|
|
const uint8_t *const source_end = source + source_size;
|
|
size_t first_element = 0;
|
|
|
|
const size_t num_full_blocks = nelements / block_size;
|
|
const size_t reminder_size = nelements - num_full_blocks * block_size;
|
|
const size_t last_block_size = reminder_size - reminder_size % BSHUF_BLOCKED_MULT;
|
|
|
|
auto decode_block = [&](size_t current_nelements) {
|
|
// Both the block length and the block itself come from the stream, which for the CBOR path
|
|
// and for the XDS plugin is data we did not produce.
|
|
if (source_end - src_ptr < 4)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Truncated compressed block header");
|
|
|
|
const auto compressed_size = static_cast<size_t>(bshuf_read_uint32_BE(src_ptr));
|
|
src_ptr += 4;
|
|
|
|
if (static_cast<size_t>(source_end - src_ptr) < compressed_size)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Compressed block extends past the input buffer");
|
|
|
|
const size_t expected_size = current_nelements * elem_size;
|
|
size_t decompressed_size = 0;
|
|
|
|
switch (algorithm) {
|
|
case CompressionAlgorithm::BSHUF_LZ4: {
|
|
const int ret = LZ4_decompress_safe(reinterpret_cast<const char *>(src_ptr),
|
|
decompressed_block.data(),
|
|
static_cast<int>(compressed_size),
|
|
static_cast<int>(expected_size));
|
|
if (ret < 0 || static_cast<size_t>(ret) != expected_size)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "LZ4 decompression error");
|
|
decompressed_size = static_cast<size_t>(ret);
|
|
break;
|
|
}
|
|
case CompressionAlgorithm::BSHUF_ZSTD:
|
|
case CompressionAlgorithm::BSHUF_ZSTD_RLE:
|
|
case CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF: {
|
|
const size_t ret = ZSTD_decompress(decompressed_block.data(),
|
|
expected_size,
|
|
src_ptr,
|
|
compressed_size);
|
|
if (ZSTD_isError(ret) || ret != expected_size)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "ZSTD decompression error");
|
|
decompressed_size = ret;
|
|
break;
|
|
}
|
|
default:
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Algorithm not supported");
|
|
}
|
|
|
|
uint8_t *dst_ptr = output ? output + first_element * elem_size : unshuffled.data();
|
|
if (JFJochBitUnshuffleBlock(reinterpret_cast<char *>(dst_ptr),
|
|
decompressed_block.data(),
|
|
scratch.data(),
|
|
current_nelements,
|
|
elem_size) < 0)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "bitshuffle block decode error");
|
|
block_done(first_element, dst_ptr, current_nelements);
|
|
|
|
src_ptr += compressed_size;
|
|
first_element += decompressed_size / elem_size;
|
|
};
|
|
|
|
for (size_t i = 0; i < num_full_blocks; ++i)
|
|
decode_block(block_size);
|
|
|
|
if (last_block_size > 0)
|
|
decode_block(last_block_size);
|
|
|
|
const size_t leftover_bytes = (reminder_size % BSHUF_BLOCKED_MULT) * elem_size;
|
|
if (leftover_bytes > 0) {
|
|
if (static_cast<size_t>(source_end - src_ptr) < leftover_bytes)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Truncated trailing bytes");
|
|
uint8_t *dst_ptr = output ? output + first_element * elem_size : unshuffled.data();
|
|
memcpy(dst_ptr, src_ptr, leftover_bytes);
|
|
block_done(first_element, dst_ptr, leftover_bytes / elem_size);
|
|
src_ptr += leftover_bytes;
|
|
}
|
|
|
|
return static_cast<size_t>(src_ptr - source);
|
|
}
|
|
|
|
inline size_t JFJochDecompressHperfPtr(uint8_t *output,
|
|
CompressionAlgorithm algorithm,
|
|
const uint8_t *source,
|
|
size_t source_size,
|
|
size_t nelements,
|
|
size_t elem_size,
|
|
size_t block_size) {
|
|
return JFJochDecompressHperfBlocks(output, algorithm, source, source_size, nelements, elem_size, block_size,
|
|
[](size_t, const uint8_t *, size_t) {});
|
|
}
|
|
|
|
// Plain LZ4, HDF5 filter 32004, as DECTRIS Eiger firmware 1.x wrote it. The framing is the same as
|
|
// bitshuffle's - a 64-bit big-endian total size, a 32-bit big-endian block size IN BYTES, then each
|
|
// block prefixed by its 32-bit big-endian compressed size - and the only difference is that no bit
|
|
// shuffle was applied, so each block decompresses straight into place. It cannot go through the
|
|
// bitshuffle path: that one requires the block to be a multiple of BSHUF_BLOCKED_MULT elements, and
|
|
// these files put the WHOLE image in one block (measured: 40666360 bytes, i.e. 10166590 uint32).
|
|
inline void JFJochDecompressLZ4Ptr(uint8_t *output,
|
|
const uint8_t *source,
|
|
size_t source_size,
|
|
size_t nelements,
|
|
size_t elem_size) {
|
|
const size_t expected_total = nelements * elem_size;
|
|
|
|
if (source_size < 12)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Buffer too short for the LZ4 header");
|
|
if (bshuf_read_uint64_BE(const_cast<uint8_t *>(source)) != expected_total)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Mismatch in size");
|
|
|
|
size_t block_bytes = bshuf_read_uint32_BE(source + 8);
|
|
if (block_bytes == 0)
|
|
block_bytes = expected_total; // some writers leave it zero and mean "one block"
|
|
|
|
const uint8_t *src_ptr = source + 12;
|
|
const uint8_t *const source_end = source + source_size;
|
|
size_t written = 0;
|
|
|
|
while (written < expected_total) {
|
|
if (source_end - src_ptr < 4)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Truncated LZ4 block header");
|
|
|
|
const auto compressed_size = static_cast<size_t>(bshuf_read_uint32_BE(src_ptr));
|
|
src_ptr += 4;
|
|
if (compressed_size == 0 || static_cast<size_t>(source_end - src_ptr) < compressed_size)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "LZ4 block extends past the input buffer");
|
|
|
|
const size_t this_block = std::min(block_bytes, expected_total - written);
|
|
const int ret = LZ4_decompress_safe(reinterpret_cast<const char *>(src_ptr),
|
|
reinterpret_cast<char *>(output + written),
|
|
static_cast<int>(compressed_size),
|
|
static_cast<int>(this_block));
|
|
if (ret < 0 || static_cast<size_t>(ret) != this_block)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "LZ4 decompression error");
|
|
|
|
src_ptr += compressed_size;
|
|
written += this_block;
|
|
}
|
|
}
|
|
|
|
// Check the 12-byte bitshuffle header and return the block size (in elements) it gives.
|
|
inline size_t JFJochBitshuffleHeaderBlockSize(const uint8_t *source, size_t source_size,
|
|
size_t nelements, size_t elem_size) {
|
|
// The header must be there before it can be read, and before source_size - 12 is handed to the
|
|
// decompressors.
|
|
if (source_size < 12)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Buffer too short for the bitshuffle header");
|
|
if (bshuf_read_uint64_BE(const_cast<uint8_t *>(source)) != nelements * elem_size)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Mismatch in size");
|
|
return bshuf_read_uint32_BE(source + 8) / elem_size;
|
|
}
|
|
|
|
inline bool JFJochIsBitshuffle(CompressionAlgorithm algorithm) {
|
|
return (algorithm == CompressionAlgorithm::BSHUF_LZ4)
|
|
|| (algorithm == CompressionAlgorithm::BSHUF_ZSTD)
|
|
|| (algorithm == CompressionAlgorithm::BSHUF_ZSTD_RLE)
|
|
|| (algorithm == CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF);
|
|
}
|
|
|
|
// A bitshuffle-compressed image, handed block by block to block_done(first_element, data, nelements)
|
|
// in element order (see JFJochDecompressHperfBlocks), without the image ever being written out whole.
|
|
template <class F>
|
|
void JFJochDecompressBlocks(CompressionAlgorithm algorithm,
|
|
const uint8_t *source,
|
|
size_t source_size,
|
|
size_t nelements,
|
|
size_t elem_size,
|
|
F &&block_done) {
|
|
const size_t block_size = JFJochBitshuffleHeaderBlockSize(source, source_size, nelements, elem_size);
|
|
if (JFJochDecompressHperfBlocks(nullptr, algorithm, source + 12, source_size - 12,
|
|
nelements, elem_size, block_size, block_done) != source_size - 12)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
|
|
}
|
|
|
|
inline void JFJochDecompressPtr(uint8_t *output,
|
|
CompressionAlgorithm algorithm,
|
|
const uint8_t *source,
|
|
size_t source_size,
|
|
size_t nelements,
|
|
size_t elem_size,
|
|
bool use_hperf = true) {
|
|
if (algorithm == CompressionAlgorithm::LZ4_NO_SHUFFLE) {
|
|
JFJochDecompressLZ4Ptr(output, source, source_size, nelements, elem_size);
|
|
return;
|
|
}
|
|
|
|
size_t block_size = 0;
|
|
if (algorithm != CompressionAlgorithm::NO_COMPRESSION)
|
|
block_size = JFJochBitshuffleHeaderBlockSize(source, source_size, nelements, elem_size);
|
|
|
|
switch (algorithm) {
|
|
case CompressionAlgorithm::NO_COMPRESSION:
|
|
if (source_size != nelements * elem_size)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Mismatch in size");
|
|
memcpy(output, source, source_size);
|
|
break;
|
|
case CompressionAlgorithm::BSHUF_LZ4:
|
|
if (use_hperf) {
|
|
if (JFJochDecompressHperfPtr(output, algorithm, source + 12, source_size - 12,
|
|
nelements, elem_size, block_size) != source_size - 12)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
|
|
} else {
|
|
if (bshuf_decompress_lz4(source + 12, output, nelements,
|
|
elem_size, block_size) != source_size - 12)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
|
|
}
|
|
break;
|
|
case CompressionAlgorithm::BSHUF_ZSTD_RLE:
|
|
case CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF:
|
|
case CompressionAlgorithm::BSHUF_ZSTD:
|
|
if (use_hperf) {
|
|
if (JFJochDecompressHperfPtr(output, algorithm, source + 12, source_size - 12,
|
|
nelements, elem_size, block_size) != source_size - 12)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
|
|
} else {
|
|
if (bshuf_decompress_zstd(source + 12, output, nelements,
|
|
elem_size, block_size) != source_size - 12)
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
|
|
}
|
|
break;
|
|
default:
|
|
throw JFJochException(JFJochExceptionCategory::Compression, "Not implemented algorithm");
|
|
}
|
|
}
|
|
|
|
template <class Td, class Ts>
|
|
void JFJochDecompress(std::vector<Td> &output, CompressionAlgorithm algorithm, const Ts *source_v, size_t source_size,
|
|
size_t nelements, bool use_hperf = true) {
|
|
output.resize(nelements);
|
|
JFJochDecompressPtr((uint8_t *) output.data(), algorithm, (uint8_t *) source_v, source_size,
|
|
nelements, sizeof(Td), use_hperf);
|
|
}
|
|
|
|
template <class Td, class Ts>
|
|
void JFJochDecompress(std::vector<Td> &output, CompressionAlgorithm algorithm, const std::vector<Ts> source_v,
|
|
size_t nelements, bool use_hperf = true) {
|
|
JFJochDecompress(output, algorithm, source_v.data(), source_v.size() * sizeof(Ts), nelements, use_hperf);
|
|
}
|