Files
Jungfraujoch/compression/JFJochDecompress.h
T
leonarski_f 84228bf8be
Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
v1.0.0-rc.173 (#83)
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports.
* jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls.
* Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results.
* Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable.
* Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate.
* Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do.
* Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence.
* Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags.
* Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check.
* Rugnux: Clear error messages when a data set needs more GPU or host memory than is available.

Reviewed-on: #83
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-09-29 15:57:32 +02:00

289 lines
14 KiB
C++

// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
#include <vector>
#include <cstring>
#include <bitshuffle/bitshuffle.h>
#include <bitshuffle/bitshuffle_internals.h>
#include <lz4/lz4.h>
#include <zstd.h>
#include "BitShuffleBlock.h"
#include "../compression/CompressionAlgorithmEnum.h"
#include "../common/JFJochException.h"
#include "../common/CompressedImage.h"
extern "C" {
uint64_t bshuf_read_uint64_BE(const void* buf);
};
// Decode the bitshuffle blocks one after another and hand each to block_done(first_element, data,
// nelements), in element order. With output set, each block is decoded in place into output; with
// output = nullptr, into a block-sized buffer that is reused for the next block, so a caller that
// consumes the block straight away reads it from cache and the whole image is never written out.
template <class F>
size_t JFJochDecompressHperfBlocks(uint8_t *output,
CompressionAlgorithm algorithm,
const uint8_t *source,
size_t source_size,
size_t nelements,
size_t elem_size,
size_t block_size,
F &&block_done) {
if ((algorithm != CompressionAlgorithm::BSHUF_LZ4) &&
(algorithm != CompressionAlgorithm::BSHUF_ZSTD) &&
(algorithm != CompressionAlgorithm::BSHUF_ZSTD_RLE) &&
(algorithm != CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF))
throw JFJochException(JFJochExceptionCategory::Compression, "Algorithm not supported by hperf decompressor");
if ((block_size == 0) || ((block_size % BSHUF_BLOCKED_MULT) != 0))
throw JFJochException(JFJochExceptionCategory::Compression, "Invalid block size");
std::vector<char> decompressed_block(block_size * elem_size);
std::vector<char> scratch(block_size * elem_size);
std::vector<uint8_t> unshuffled(output ? 0 : block_size * elem_size);
const uint8_t *src_ptr = source;
const uint8_t *const source_end = source + source_size;
size_t first_element = 0;
const size_t num_full_blocks = nelements / block_size;
const size_t reminder_size = nelements - num_full_blocks * block_size;
const size_t last_block_size = reminder_size - reminder_size % BSHUF_BLOCKED_MULT;
auto decode_block = [&](size_t current_nelements) {
// Both the block length and the block itself come from the stream, which for the CBOR path
// and for the XDS plugin is data we did not produce.
if (source_end - src_ptr < 4)
throw JFJochException(JFJochExceptionCategory::Compression, "Truncated compressed block header");
const auto compressed_size = static_cast<size_t>(bshuf_read_uint32_BE(src_ptr));
src_ptr += 4;
if (static_cast<size_t>(source_end - src_ptr) < compressed_size)
throw JFJochException(JFJochExceptionCategory::Compression, "Compressed block extends past the input buffer");
const size_t expected_size = current_nelements * elem_size;
size_t decompressed_size = 0;
switch (algorithm) {
case CompressionAlgorithm::BSHUF_LZ4: {
const int ret = LZ4_decompress_safe(reinterpret_cast<const char *>(src_ptr),
decompressed_block.data(),
static_cast<int>(compressed_size),
static_cast<int>(expected_size));
if (ret < 0 || static_cast<size_t>(ret) != expected_size)
throw JFJochException(JFJochExceptionCategory::Compression, "LZ4 decompression error");
decompressed_size = static_cast<size_t>(ret);
break;
}
case CompressionAlgorithm::BSHUF_ZSTD:
case CompressionAlgorithm::BSHUF_ZSTD_RLE:
case CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF: {
const size_t ret = ZSTD_decompress(decompressed_block.data(),
expected_size,
src_ptr,
compressed_size);
if (ZSTD_isError(ret) || ret != expected_size)
throw JFJochException(JFJochExceptionCategory::Compression, "ZSTD decompression error");
decompressed_size = ret;
break;
}
default:
throw JFJochException(JFJochExceptionCategory::Compression, "Algorithm not supported");
}
uint8_t *dst_ptr = output ? output + first_element * elem_size : unshuffled.data();
if (JFJochBitUnshuffleBlock(reinterpret_cast<char *>(dst_ptr),
decompressed_block.data(),
scratch.data(),
current_nelements,
elem_size) < 0)
throw JFJochException(JFJochExceptionCategory::Compression, "bitshuffle block decode error");
block_done(first_element, dst_ptr, current_nelements);
src_ptr += compressed_size;
first_element += decompressed_size / elem_size;
};
for (size_t i = 0; i < num_full_blocks; ++i)
decode_block(block_size);
if (last_block_size > 0)
decode_block(last_block_size);
const size_t leftover_bytes = (reminder_size % BSHUF_BLOCKED_MULT) * elem_size;
if (leftover_bytes > 0) {
if (static_cast<size_t>(source_end - src_ptr) < leftover_bytes)
throw JFJochException(JFJochExceptionCategory::Compression, "Truncated trailing bytes");
uint8_t *dst_ptr = output ? output + first_element * elem_size : unshuffled.data();
memcpy(dst_ptr, src_ptr, leftover_bytes);
block_done(first_element, dst_ptr, leftover_bytes / elem_size);
src_ptr += leftover_bytes;
}
return static_cast<size_t>(src_ptr - source);
}
inline size_t JFJochDecompressHperfPtr(uint8_t *output,
CompressionAlgorithm algorithm,
const uint8_t *source,
size_t source_size,
size_t nelements,
size_t elem_size,
size_t block_size) {
return JFJochDecompressHperfBlocks(output, algorithm, source, source_size, nelements, elem_size, block_size,
[](size_t, const uint8_t *, size_t) {});
}
// Plain LZ4, HDF5 filter 32004, as DECTRIS Eiger firmware 1.x wrote it. The framing is the same as
// bitshuffle's - a 64-bit big-endian total size, a 32-bit big-endian block size IN BYTES, then each
// block prefixed by its 32-bit big-endian compressed size - and the only difference is that no bit
// shuffle was applied, so each block decompresses straight into place. It cannot go through the
// bitshuffle path: that one requires the block to be a multiple of BSHUF_BLOCKED_MULT elements, and
// these files put the WHOLE image in one block (measured: 40666360 bytes, i.e. 10166590 uint32).
inline void JFJochDecompressLZ4Ptr(uint8_t *output,
const uint8_t *source,
size_t source_size,
size_t nelements,
size_t elem_size) {
const size_t expected_total = nelements * elem_size;
if (source_size < 12)
throw JFJochException(JFJochExceptionCategory::Compression, "Buffer too short for the LZ4 header");
if (bshuf_read_uint64_BE(const_cast<uint8_t *>(source)) != expected_total)
throw JFJochException(JFJochExceptionCategory::Compression, "Mismatch in size");
size_t block_bytes = bshuf_read_uint32_BE(source + 8);
if (block_bytes == 0)
block_bytes = expected_total; // some writers leave it zero and mean "one block"
const uint8_t *src_ptr = source + 12;
const uint8_t *const source_end = source + source_size;
size_t written = 0;
while (written < expected_total) {
if (source_end - src_ptr < 4)
throw JFJochException(JFJochExceptionCategory::Compression, "Truncated LZ4 block header");
const auto compressed_size = static_cast<size_t>(bshuf_read_uint32_BE(src_ptr));
src_ptr += 4;
if (compressed_size == 0 || static_cast<size_t>(source_end - src_ptr) < compressed_size)
throw JFJochException(JFJochExceptionCategory::Compression, "LZ4 block extends past the input buffer");
const size_t this_block = std::min(block_bytes, expected_total - written);
const int ret = LZ4_decompress_safe(reinterpret_cast<const char *>(src_ptr),
reinterpret_cast<char *>(output + written),
static_cast<int>(compressed_size),
static_cast<int>(this_block));
if (ret < 0 || static_cast<size_t>(ret) != this_block)
throw JFJochException(JFJochExceptionCategory::Compression, "LZ4 decompression error");
src_ptr += compressed_size;
written += this_block;
}
}
// Check the 12-byte bitshuffle header and return the block size (in elements) it gives.
inline size_t JFJochBitshuffleHeaderBlockSize(const uint8_t *source, size_t source_size,
size_t nelements, size_t elem_size) {
// The header must be there before it can be read, and before source_size - 12 is handed to the
// decompressors.
if (source_size < 12)
throw JFJochException(JFJochExceptionCategory::Compression, "Buffer too short for the bitshuffle header");
if (bshuf_read_uint64_BE(const_cast<uint8_t *>(source)) != nelements * elem_size)
throw JFJochException(JFJochExceptionCategory::Compression, "Mismatch in size");
return bshuf_read_uint32_BE(source + 8) / elem_size;
}
inline bool JFJochIsBitshuffle(CompressionAlgorithm algorithm) {
return (algorithm == CompressionAlgorithm::BSHUF_LZ4)
|| (algorithm == CompressionAlgorithm::BSHUF_ZSTD)
|| (algorithm == CompressionAlgorithm::BSHUF_ZSTD_RLE)
|| (algorithm == CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF);
}
// A bitshuffle-compressed image, handed block by block to block_done(first_element, data, nelements)
// in element order (see JFJochDecompressHperfBlocks), without the image ever being written out whole.
template <class F>
void JFJochDecompressBlocks(CompressionAlgorithm algorithm,
const uint8_t *source,
size_t source_size,
size_t nelements,
size_t elem_size,
F &&block_done) {
const size_t block_size = JFJochBitshuffleHeaderBlockSize(source, source_size, nelements, elem_size);
if (JFJochDecompressHperfBlocks(nullptr, algorithm, source + 12, source_size - 12,
nelements, elem_size, block_size, block_done) != source_size - 12)
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
}
inline void JFJochDecompressPtr(uint8_t *output,
CompressionAlgorithm algorithm,
const uint8_t *source,
size_t source_size,
size_t nelements,
size_t elem_size,
bool use_hperf = true) {
if (algorithm == CompressionAlgorithm::LZ4_NO_SHUFFLE) {
JFJochDecompressLZ4Ptr(output, source, source_size, nelements, elem_size);
return;
}
size_t block_size = 0;
if (algorithm != CompressionAlgorithm::NO_COMPRESSION)
block_size = JFJochBitshuffleHeaderBlockSize(source, source_size, nelements, elem_size);
switch (algorithm) {
case CompressionAlgorithm::NO_COMPRESSION:
if (source_size != nelements * elem_size)
throw JFJochException(JFJochExceptionCategory::Compression, "Mismatch in size");
memcpy(output, source, source_size);
break;
case CompressionAlgorithm::BSHUF_LZ4:
if (use_hperf) {
if (JFJochDecompressHperfPtr(output, algorithm, source + 12, source_size - 12,
nelements, elem_size, block_size) != source_size - 12)
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
} else {
if (bshuf_decompress_lz4(source + 12, output, nelements,
elem_size, block_size) != source_size - 12)
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
}
break;
case CompressionAlgorithm::BSHUF_ZSTD_RLE:
case CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF:
case CompressionAlgorithm::BSHUF_ZSTD:
if (use_hperf) {
if (JFJochDecompressHperfPtr(output, algorithm, source + 12, source_size - 12,
nelements, elem_size, block_size) != source_size - 12)
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
} else {
if (bshuf_decompress_zstd(source + 12, output, nelements,
elem_size, block_size) != source_size - 12)
throw JFJochException(JFJochExceptionCategory::Compression, "Decompression error");
}
break;
default:
throw JFJochException(JFJochExceptionCategory::Compression, "Not implemented algorithm");
}
}
template <class Td, class Ts>
void JFJochDecompress(std::vector<Td> &output, CompressionAlgorithm algorithm, const Ts *source_v, size_t source_size,
size_t nelements, bool use_hperf = true) {
output.resize(nelements);
JFJochDecompressPtr((uint8_t *) output.data(), algorithm, (uint8_t *) source_v, source_size,
nelements, sizeof(Td), use_hperf);
}
template <class Td, class Ts>
void JFJochDecompress(std::vector<Td> &output, CompressionAlgorithm algorithm, const std::vector<Ts> source_v,
size_t nelements, bool use_hperf = true) {
JFJochDecompress(output, algorithm, source_v.data(), source_v.size() * sizeof(Ts), nelements, use_hperf);
}