Files
Jungfraujoch/compression/BitShuffleBlock.h
T
leonarski_fandClaude Opus 5.5 f8e5d0c3fa BitShuffleBlock: the aarch64 decode uses the caller's scratch instead of a malloc per block
On aarch64 JFJochBitUnshuffleBlock went through bshuf_untrans_bit_elem_NEON, which mallocs and
frees a block-sized temporary for every block, while the caller (JFJochDecompressHperfPtr) already
owns an unused scratch buffer of exactly that size. Call the two NEON stages it wraps
(bshuf_trans_byte_bitrow_NEON, bshuf_shuffle_bit_eightelem_NEON) directly with that scratch.
Same code, same output; x86 is untouched.

Checked byte-identical under qemu-aarch64 against the classic entry point, hperf and the x86
decoders: element sizes 1/2/3/4/8, every block length 8..1040 elements plus 1536..16384, seven
data patterns, and real EIGER2/PILATUS4 bslz4 chunks.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
2026-09-27 21:08:41 +02:00

57 lines
2.8 KiB
C

// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
// SPDX-License-Identifier: GPL-3.0-only
#pragma once
#include <bitshuffle/bitshuffle_internals.h>
#include <bitshuffle_hperf/bitshuffle.h>
// One bitshuffle block, transformed by whichever of the two vendored implementations is SIMD on
// this architecture. The two write byte-identical output and each decodes the other's, so the file
// format does not depend on the build host.
//
// bitshuffle_hperf is x86-only: outside SSE2 its whole vector body is compiled out and what remains
// is a scalar fallback. The classic bitshuffle has an aarch64 NEON path (bitshuffle_core.c,
// USEARMNEON), so aarch64 uses that instead - measured against the hperf scalar fallback it is
// ~2.5x on encode and ~1.7x on decode. Everywhere else hperf wins outright (~2x over classic SSE2),
// so it stays the default.
//
// The condition mirrors USEARMNEON in bitshuffle_core.c exactly. With NEON off the classic path
// falls back to a scalar of its own that is slower than hperf's, so it must not be selected then.
// It is a preprocessor test rather than a CMake one on purpose: Apple Silicon defines the same two
// macros as aarch64 Linux, and a macOS universal build compiles this header once per architecture,
// which a single configure-time answer could not follow.
#if (defined(__ARM_NEON__) || (__ARM_NEON)) && defined(__aarch64__)
// The two stages of bshuf_untrans_bit_elem_NEON, which bitshuffle_core.c defines but no header
// declares. Calling them directly lets the decode use the caller's scratch instead of the block-sized
// buffer the classic entry point mallocs and frees for every block.
extern "C" {
int64_t bshuf_trans_byte_bitrow_NEON(const void *in, void *out, size_t size, size_t elem_size);
int64_t bshuf_shuffle_bit_eightelem_NEON(const void *in, void *out, size_t size, size_t elem_size);
}
// The classic encode entry point allocates its own block-sized scratch, so the caller's goes unused.
inline int64_t JFJochBitShuffleBlock(char *out, const char *in, char *, size_t size, size_t elem_size) {
return bshuf_trans_bit_elem(in, out, size, elem_size);
}
inline int64_t JFJochBitUnshuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
const int64_t count = bshuf_trans_byte_bitrow_NEON(in, scratch, size, elem_size);
if (count < 0)
return count;
return bshuf_shuffle_bit_eightelem_NEON(scratch, out, size, elem_size);
}
#else
inline int64_t JFJochBitShuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
return bitshuf_encode_block(out, in, scratch, size, elem_size);
}
inline int64_t JFJochBitUnshuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
return bitshuf_decode_block(out, in, scratch, size, elem_size);
}
#endif