On aarch64 JFJochBitUnshuffleBlock went through bshuf_untrans_bit_elem_NEON, which mallocs and frees a block-sized temporary for every block, while the caller (JFJochDecompressHperfPtr) already owns an unused scratch buffer of exactly that size. Call the two NEON stages it wraps (bshuf_trans_byte_bitrow_NEON, bshuf_shuffle_bit_eightelem_NEON) directly with that scratch. Same code, same output; x86 is untouched. Checked byte-identical under qemu-aarch64 against the classic entry point, hperf and the x86 decoders: element sizes 1/2/3/4/8, every block length 8..1040 elements plus 1536..16384, seven data patterns, and real EIGER2/PILATUS4 bslz4 chunks. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
57 lines
2.8 KiB
C
57 lines
2.8 KiB
C
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <bitshuffle/bitshuffle_internals.h>
|
|
#include <bitshuffle_hperf/bitshuffle.h>
|
|
|
|
// One bitshuffle block, transformed by whichever of the two vendored implementations is SIMD on
|
|
// this architecture. The two write byte-identical output and each decodes the other's, so the file
|
|
// format does not depend on the build host.
|
|
//
|
|
// bitshuffle_hperf is x86-only: outside SSE2 its whole vector body is compiled out and what remains
|
|
// is a scalar fallback. The classic bitshuffle has an aarch64 NEON path (bitshuffle_core.c,
|
|
// USEARMNEON), so aarch64 uses that instead - measured against the hperf scalar fallback it is
|
|
// ~2.5x on encode and ~1.7x on decode. Everywhere else hperf wins outright (~2x over classic SSE2),
|
|
// so it stays the default.
|
|
//
|
|
// The condition mirrors USEARMNEON in bitshuffle_core.c exactly. With NEON off the classic path
|
|
// falls back to a scalar of its own that is slower than hperf's, so it must not be selected then.
|
|
// It is a preprocessor test rather than a CMake one on purpose: Apple Silicon defines the same two
|
|
// macros as aarch64 Linux, and a macOS universal build compiles this header once per architecture,
|
|
// which a single configure-time answer could not follow.
|
|
#if (defined(__ARM_NEON__) || (__ARM_NEON)) && defined(__aarch64__)
|
|
|
|
// The two stages of bshuf_untrans_bit_elem_NEON, which bitshuffle_core.c defines but no header
|
|
// declares. Calling them directly lets the decode use the caller's scratch instead of the block-sized
|
|
// buffer the classic entry point mallocs and frees for every block.
|
|
extern "C" {
|
|
int64_t bshuf_trans_byte_bitrow_NEON(const void *in, void *out, size_t size, size_t elem_size);
|
|
int64_t bshuf_shuffle_bit_eightelem_NEON(const void *in, void *out, size_t size, size_t elem_size);
|
|
}
|
|
|
|
// The classic encode entry point allocates its own block-sized scratch, so the caller's goes unused.
|
|
inline int64_t JFJochBitShuffleBlock(char *out, const char *in, char *, size_t size, size_t elem_size) {
|
|
return bshuf_trans_bit_elem(in, out, size, elem_size);
|
|
}
|
|
|
|
inline int64_t JFJochBitUnshuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
|
|
const int64_t count = bshuf_trans_byte_bitrow_NEON(in, scratch, size, elem_size);
|
|
if (count < 0)
|
|
return count;
|
|
return bshuf_shuffle_bit_eightelem_NEON(scratch, out, size, elem_size);
|
|
}
|
|
|
|
#else
|
|
|
|
inline int64_t JFJochBitShuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
|
|
return bitshuf_encode_block(out, in, scratch, size, elem_size);
|
|
}
|
|
|
|
inline int64_t JFJochBitUnshuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
|
|
return bitshuf_decode_block(out, in, scratch, size, elem_size);
|
|
}
|
|
|
|
#endif
|