Files
Jungfraujoch/compression/JFJochCompressor.cpp
T
leonarski_fandClaude Opus 5 fbc5078399 compression: use the NEON bitshuffle on aarch64
bitshuffle_hperf is an x86-only implementation - its entire vector body sits behind
__i386__/__x86_64__, so on aarch64 every entry point compiles down to the scalar
fallback. Measured against its own SIMD path that costs 8.3x on encode and 3.8x on
decode, and it is the transform behind every compressed image the writer produces and
every one the reader, preview and XDS plugin take apart again.

The classic bitshuffle vendored beside it does have an aarch64 NEON path, and is already
compiled into the same target, so this costs nothing new. BitShuffleBlock.h picks
bshuf_trans_bit_elem/bshuf_untrans_bit_elem there and keeps bitshuf_encode_block /
bitshuf_decode_block everywhere else, where hperf is about twice classic SSE2 and remains
the better choice. The expected aarch64 gain is ~2.5x encode / ~1.7x decode: classic NEON
is 128-bit and carries an extra pass, so it recovers part of the gap rather than all of
it. The condition mirrors USEARMNEON in bitshuffle_core.c exactly, because with NEON off
the classic scalar path is slower than hperf's and must not be selected.

Swapping implementations is only safe while the two agree bit for bit - otherwise an ARM
build would write files an x86 build could not read. They do: verified byte-identical
output and mutual cross-decoding for elem_size 1/2/4/8 over block sizes from 8 to 65536
elements. Both are always compiled in, so the new test holds them to it on every
architecture, not just the one that would notice.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-03 21:27:27 +02:00

147 lines
6.7 KiB
C++

// SPDX-FileCopyrightText: 2024 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#include "JFJochCompressor.h"
#include <stdexcept>
#include <cstring>
#include <bitshuffle/bitshuffle_internals.h>
#include <zstd.h>
#include <lz4/lz4.h>
#include "BitShuffleBlock.h"
#include "../common/JFJochException.h"
extern "C" {
void bshuf_write_uint64_BE(void* buf, uint64_t num);
}
// Necessary condition for BlockSize() to be a valid bitshuffle block: with elem_size a power of
// two, a byte target that is a multiple of BSHUF_BLOCKED_MULT keeps the element count a multiple
// of BSHUF_BLOCKED_MULT too.
static_assert(JFJochBitShuffleCompressor::DefaultBlockSizeBytes(CompressionAlgorithm::BSHUF_LZ4) % BSHUF_BLOCKED_MULT == 0
&& JFJochBitShuffleCompressor::DefaultBlockSizeBytes(CompressionAlgorithm::BSHUF_ZSTD) % BSHUF_BLOCKED_MULT == 0,
"block byte target must be a multiple of the bitshuffle block multiple");
// Worst-case size of one compressed block, including its 4-byte length prefix. Mirrors the
// per-block term of MaxCompressedSize(), so a dest sized to MaxCompressedSize() never fails.
static size_t MaxCompressedBlockSize(CompressionAlgorithm algorithm, size_t src_size) {
switch (algorithm) {
case CompressionAlgorithm::BSHUF_LZ4:
return LZ4_compressBound(src_size) + 4;
case CompressionAlgorithm::BSHUF_ZSTD:
case CompressionAlgorithm::BSHUF_ZSTD_RLE:
case CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF:
return ZSTD_compressBound(src_size) + 4;
default:
return src_size + 4;
}
}
JFJochBitShuffleCompressor::JFJochBitShuffleCompressor(CompressionAlgorithm in_algorithm) {
algorithm = in_algorithm;
}
size_t JFJochBitShuffleCompressor::CompressBlock(char *dest, const char *source, size_t nelements, size_t elem_size) {
// Assert nelements < block_size
const char *src_ptr;
int64_t bshuf_ret = JFJochBitShuffleBlock(tmp_space.data(), source, scratch.data(), nelements, elem_size);
if (bshuf_ret < 0)
throw JFJochException(JFJochExceptionCategory::Compression, "bshuf_trans_bit_elem error");
src_ptr = tmp_space.data();
size_t compressed_size;
size_t src_size = nelements * elem_size;
switch (algorithm) {
case CompressionAlgorithm::BSHUF_LZ4:
compressed_size = LZ4_compress_default(src_ptr, dest + 4, src_size, LZ4_compressBound(src_size));
break;
case CompressionAlgorithm::BSHUF_ZSTD:
compressed_size = ZSTD_compress(dest + 4, ZSTD_compressBound(src_size), src_ptr, src_size, 0);
if (ZSTD_isError(compressed_size))
throw(JFJochException(JFJochExceptionCategory::Compression, ZSTD_getErrorName(compressed_size)));
break;
case CompressionAlgorithm::BSHUF_ZSTD_RLE:
try {
compressed_size = zstd_compressor.Compress(((uint8_t *) dest) + 4, (uint64_t *) src_ptr,
src_size, src_size);
} catch (const std::runtime_error &e) {
throw JFJochException(JFJochExceptionCategory::ZSTDCompressionError, e.what());
}
break;
case CompressionAlgorithm::BSHUF_ZSTD_RLE_HUFF:
compressed_size = huff_compressor.Compress(((uint8_t *) dest) + 4, (const uint64_t *) src_ptr, src_size);
break;
default:
throw JFJochException(JFJochExceptionCategory::Compression, "Algorithm not supported");
}
bshuf_write_uint32_BE(dest, compressed_size);
return compressed_size + 4;
}
std::vector<uint8_t> JFJochBitShuffleCompressor::Compress(const void *source, size_t nelements, size_t elem_size) {
std::vector<uint8_t> tmp(MaxCompressedSize(algorithm, nelements, elem_size));
size_t tmp_size = Compress(tmp.data(), tmp.size(), source, nelements, elem_size);
tmp.resize(tmp_size);
return tmp;
}
size_t JFJochBitShuffleCompressor::Compress(void *dest, size_t dest_size, const void *source, size_t nelements, size_t elem_size) {
auto c_dest = (char *) dest;
auto c_source = (char *) source;
if (algorithm == CompressionAlgorithm::NO_COMPRESSION) {
// Trivial case if no compression - copy content
if (nelements * elem_size > dest_size)
throw CompressionBufferTooSmallException("compressed output exceeds destination buffer");
memcpy(dest, source, nelements * elem_size);
return nelements * elem_size;
}
if (dest_size < 12)
throw CompressionBufferTooSmallException("compressed output exceeds destination buffer");
const size_t block_size = BlockSize(algorithm, elem_size);
bshuf_write_uint64_BE(c_dest, nelements * elem_size);
bshuf_write_uint32_BE(c_dest + 8, block_size * elem_size);
if (tmp_space.size() < block_size * elem_size)
tmp_space.resize(block_size * elem_size);
if (scratch.size() < block_size * elem_size)
scratch.resize(block_size * elem_size);
size_t num_full_blocks = nelements / block_size;
size_t reminder_size = nelements - num_full_blocks * block_size;
size_t compressed_size = 12;
// Blocks are small relative to the image, so before each one we just check that the
// remaining space still covers that block's worst case, and throw if not.
for (int i = 0; i < num_full_blocks; i++) {
if (compressed_size + MaxCompressedBlockSize(algorithm, block_size * elem_size) > dest_size)
throw CompressionBufferTooSmallException("compressed output exceeds destination buffer");
compressed_size += CompressBlock(c_dest + compressed_size,
c_source + i * block_size * elem_size, block_size, elem_size);
}
size_t last_block_size = reminder_size - reminder_size % BSHUF_BLOCKED_MULT;
if (last_block_size > 0) {
if (compressed_size + MaxCompressedBlockSize(algorithm, last_block_size * elem_size) > dest_size)
throw CompressionBufferTooSmallException("compressed output exceeds destination buffer");
compressed_size += CompressBlock(c_dest + compressed_size,
c_source + num_full_blocks * block_size * elem_size, last_block_size, elem_size);
}
size_t leftover_bytes = (reminder_size % BSHUF_BLOCKED_MULT) * elem_size;
if (leftover_bytes > 0) {
if (compressed_size + leftover_bytes > dest_size)
throw CompressionBufferTooSmallException("compressed output exceeds destination buffer");
memcpy(c_dest + compressed_size, c_source + (num_full_blocks * block_size + last_block_size) * elem_size, leftover_bytes);
compressed_size += leftover_bytes;
}
return compressed_size;
}