// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute // SPDX-License-Identifier: GPL-3.0-only #pragma once #include #include // One bitshuffle block, transformed by whichever of the two vendored implementations is SIMD on // this architecture. The two write byte-identical output and each decodes the other's, so the file // format does not depend on the build host. // // bitshuffle_hperf is x86-only: outside SSE2 its whole vector body is compiled out and what remains // is a scalar fallback. The classic bitshuffle has an aarch64 NEON path (bitshuffle_core.c, // USEARMNEON), so aarch64 encodes with that instead - measured against the hperf scalar fallback it // is ~2.5x on encode and ~1.7x on decode. Decode, which is what rugnux spends its time on, goes on // aarch64 through a NEON port of hperf's decoder (bitshuffle_hperf/bitshuffle_neon.c) for the // element sizes it covers (1, 2, 4 bytes), and through the classic NEON stages for the rest. // Everywhere else hperf wins outright (~2x over classic SSE2), so it stays the default. // // The condition mirrors USEARMNEON in bitshuffle_core.c exactly. With NEON off the classic path // falls back to a scalar of its own that is slower than hperf's, so it must not be selected then. // It is a preprocessor test rather than a CMake one on purpose: Apple Silicon defines the same two // macros as aarch64 Linux, and a macOS universal build compiles this header once per architecture, // which a single configure-time answer could not follow. #if (defined(__ARM_NEON__) || (__ARM_NEON)) && defined(__aarch64__) #include // The two stages of bshuf_untrans_bit_elem_NEON, which bitshuffle_core.c defines but no header // declares. Calling them directly lets the decode use the caller's scratch instead of the block-sized // buffer the classic entry point mallocs and frees for every block. extern "C" { int64_t bshuf_trans_byte_bitrow_NEON(const void *in, void *out, size_t size, size_t elem_size); int64_t bshuf_shuffle_bit_eightelem_NEON(const void *in, void *out, size_t size, size_t elem_size); } // The classic encode entry point allocates its own block-sized scratch, so the caller's goes unused. inline int64_t JFJochBitShuffleBlock(char *out, const char *in, char *, size_t size, size_t elem_size) { return bshuf_trans_bit_elem(in, out, size, elem_size); } inline int64_t JFJochBitUnshuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) { if (elem_size == 1 || elem_size == 2 || elem_size == 4) return bitshuf_decode_block_neon(out, in, scratch, size, elem_size); const int64_t count = bshuf_trans_byte_bitrow_NEON(in, scratch, size, elem_size); if (count < 0) return count; return bshuf_shuffle_bit_eightelem_NEON(scratch, out, size, elem_size); } #else inline int64_t JFJochBitShuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) { return bitshuf_encode_block(out, in, scratch, size, elem_size); } inline int64_t JFJochBitUnshuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) { return bitshuf_decode_block(out, in, scratch, size, elem_size); } #endif