bitshuffle_hperf's vector code is x86-only, so aarch64 decoded with the classic NEON path, whose bit stage emulates x86 movemask (8 and + 8 cmeq + 2 x 9-instruction pairwise-add reductions per 16 bytes, then eight scattered 16-bit stores): ~29 instructions per 8x8 bit block plus a byte transpose pass with scattered 8-byte stores, and a scalar path for 1-byte elements. bitshuffle_neon.c follows hperf's structure (per byte plane bit-untranspose, then byte interleave) but does the 8x8 bit transpose NEON-natively: the eight rows stay in eight registers, bits are exchanged between them with shift + bit-select in three levels, and one zip level plus two st4 write the 128 contiguous output bytes. GCC 13 -O3 gives ~75 instructions per 16 blocks (~4.7 per block) in the bit stage; the byte interleave is one st2/st4 per 16 elements. Other element sizes keep the classic stages with the caller's scratch. x86 is unchanged (the file compiles empty). Byte-identical under qemu-aarch64 (GCC 13, -O3 and -O0) to the classic decoder, hperf and the x86 hperf/AVX2 path: element sizes 1/2/3/4/8, every block of 8..1040 elements plus 1536..16384, seven data patterns, and real EIGER2/PILATUS4 bslz4 chunks (whose decoded image hash also matches an hdf5plugin read). Same licence and component as bitshuffle_hperf, so licenses/ and THIRD_PARTY_NOTICES.md are unchanged. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
64 lines
3.3 KiB
C
64 lines
3.3 KiB
C
// SPDX-FileCopyrightText: 2026 Filip Leonarski, Paul Scherrer Institute <filip.leonarski@psi.ch>
|
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
|
|
#pragma once
|
|
|
|
#include <bitshuffle/bitshuffle_internals.h>
|
|
#include <bitshuffle_hperf/bitshuffle.h>
|
|
|
|
// One bitshuffle block, transformed by whichever of the two vendored implementations is SIMD on
|
|
// this architecture. The two write byte-identical output and each decodes the other's, so the file
|
|
// format does not depend on the build host.
|
|
//
|
|
// bitshuffle_hperf is x86-only: outside SSE2 its whole vector body is compiled out and what remains
|
|
// is a scalar fallback. The classic bitshuffle has an aarch64 NEON path (bitshuffle_core.c,
|
|
// USEARMNEON), so aarch64 encodes with that instead - measured against the hperf scalar fallback it
|
|
// is ~2.5x on encode and ~1.7x on decode. Decode, which is what rugnux spends its time on, goes on
|
|
// aarch64 through a NEON port of hperf's decoder (bitshuffle_hperf/bitshuffle_neon.c) for the
|
|
// element sizes it covers (1, 2, 4 bytes), and through the classic NEON stages for the rest.
|
|
// Everywhere else hperf wins outright (~2x over classic SSE2), so it stays the default.
|
|
//
|
|
// The condition mirrors USEARMNEON in bitshuffle_core.c exactly. With NEON off the classic path
|
|
// falls back to a scalar of its own that is slower than hperf's, so it must not be selected then.
|
|
// It is a preprocessor test rather than a CMake one on purpose: Apple Silicon defines the same two
|
|
// macros as aarch64 Linux, and a macOS universal build compiles this header once per architecture,
|
|
// which a single configure-time answer could not follow.
|
|
#if (defined(__ARM_NEON__) || (__ARM_NEON)) && defined(__aarch64__)
|
|
|
|
#include <bitshuffle_hperf/bitshuffle_neon.h>
|
|
|
|
// The two stages of bshuf_untrans_bit_elem_NEON, which bitshuffle_core.c defines but no header
|
|
// declares. Calling them directly lets the decode use the caller's scratch instead of the block-sized
|
|
// buffer the classic entry point mallocs and frees for every block.
|
|
extern "C" {
|
|
int64_t bshuf_trans_byte_bitrow_NEON(const void *in, void *out, size_t size, size_t elem_size);
|
|
int64_t bshuf_shuffle_bit_eightelem_NEON(const void *in, void *out, size_t size, size_t elem_size);
|
|
}
|
|
|
|
// The classic encode entry point allocates its own block-sized scratch, so the caller's goes unused.
|
|
inline int64_t JFJochBitShuffleBlock(char *out, const char *in, char *, size_t size, size_t elem_size) {
|
|
return bshuf_trans_bit_elem(in, out, size, elem_size);
|
|
}
|
|
|
|
inline int64_t JFJochBitUnshuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
|
|
if (elem_size == 1 || elem_size == 2 || elem_size == 4)
|
|
return bitshuf_decode_block_neon(out, in, scratch, size, elem_size);
|
|
|
|
const int64_t count = bshuf_trans_byte_bitrow_NEON(in, scratch, size, elem_size);
|
|
if (count < 0)
|
|
return count;
|
|
return bshuf_shuffle_bit_eightelem_NEON(scratch, out, size, elem_size);
|
|
}
|
|
|
|
#else
|
|
|
|
inline int64_t JFJochBitShuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
|
|
return bitshuf_encode_block(out, in, scratch, size, elem_size);
|
|
}
|
|
|
|
inline int64_t JFJochBitUnshuffleBlock(char *out, const char *in, char *scratch, size_t size, size_t elem_size) {
|
|
return bitshuf_decode_block(out, in, scratch, size, elem_size);
|
|
}
|
|
|
|
#endif
|