diff --git a/compression/BitShuffleBlock.h b/compression/BitShuffleBlock.h index b740a4f6a..a4a73f76a 100644 --- a/compression/BitShuffleBlock.h +++ b/compression/BitShuffleBlock.h @@ -18,6 +18,9 @@ // // The condition mirrors USEARMNEON in bitshuffle_core.c exactly. With NEON off the classic path // falls back to a scalar of its own that is slower than hperf's, so it must not be selected then. +// It is a preprocessor test rather than a CMake one on purpose: Apple Silicon defines the same two +// macros as aarch64 Linux, and a macOS universal build compiles this header once per architecture, +// which a single configure-time answer could not follow. #if (defined(__ARM_NEON__) || (__ARM_NEON)) && defined(__aarch64__) // The classic entry points allocate their own block-sized scratch, so the caller's goes unused. diff --git a/tests/FrameTransformationTest.cpp b/tests/FrameTransformationTest.cpp index 22f997521..3abebda59 100644 --- a/tests/FrameTransformationTest.cpp +++ b/tests/FrameTransformationTest.cpp @@ -22,8 +22,10 @@ inline uint64_t read_be64(const void *ptr) { return __builtin_bswap64(ptr64[0]); } -TEST_CASE("Bshuf_SSE", "[bitshuffle]") { - REQUIRE (bshuf_using_SSE2() == 1); +// The classic bitshuffle runs behind the HDF5 filter everywhere, and on aarch64 it is also the one +// BitShuffleBlock.h picks, so a build that left it on its scalar path must not go unnoticed. +TEST_CASE("Bshuf_SIMD", "[bitshuffle]") { + REQUIRE ((bshuf_using_SSE2() == 1 || bshuf_using_NEON() == 1)); } TEST_CASE("FrameTransformation_Raw_NoCompression" ,"") {