// Bitshuffle block-decode benchmark: classic NEON (malloc per block), classic NEON with the caller's // scratch (S1), the NEON port of hperf, hperf itself (scalar fallback on arm64), and the selector // rugnux uses (JFJochBitUnshuffleBlock). Optionally the full bitshuffle/LZ4 chunk decode. // // Build: ./build.sh /path/to/nextgendcu [clang] [clang++] (checkout of branch arm64-bitshuffle), or by hand: // REPO=/path/to/nextgendcu // S="$REPO/compression/bitshuffle/bitshuffle_core.c $REPO/compression/bitshuffle/iochain.c \ // $REPO/compression/bitshuffle_hperf/bitshuffle.c $REPO/compression/bitshuffle_hperf/bitshuffle_neon.c \ // $REPO/compression/lz4/lz4.c" // for f in $S; do clang -O3 -std=gnu11 -I$REPO/compression -c $f -o $(basename $f .c).o; done // clang++ -O3 -std=c++20 -I$REPO/compression bs_bench.cpp *.o -o bs_bench // ./bs_bench [-m MB] [chunk.bslz4 ...] # MB of synthetic blocks (default 64); 4-byte chunks, // # e.g. eiger2_0.bslz4 pilatus4_0.bslz4 // (on Linux use gcc/g++ the same way; the same objects also build bs_exact_test.cpp, the // byte-identity check, which is worth running natively once: ./bs_exact_test *.bslz4) #include #include #include #include #include #include #include #include #include #include #include extern "C" uint64_t bshuf_read_uint64_BE(const void *buf); #if defined(__aarch64__) #include #endif using Decoder = std::function; static double best_seconds(const std::function &f, int reps) { double best = 1e30; for (int r = 0; r < reps; r++) { auto t0 = std::chrono::steady_clock::now(); f(); best = std::min(best, std::chrono::duration(std::chrono::steady_clock::now() - t0).count()); } return best; } static std::vector> decoders() { std::vector> d; d.push_back({"classic bshuf_untrans_bit_elem", [](char *o, const char *i, char *, size_t n, size_t e) { bshuf_untrans_bit_elem(i, o, n, e); }}); d.push_back({"hperf bitshuf_decode_block", [](char *o, const char *i, char *s, size_t n, size_t e) { bitshuf_decode_block(o, i, s, n, e); }}); #if defined(__aarch64__) d.push_back({"S1 classic NEON, caller scratch", [](char *o, const char *i, char *s, size_t n, size_t e) { bshuf_trans_byte_bitrow_NEON(i, s, n, e); bshuf_shuffle_bit_eightelem_NEON(s, o, n, e); }}); d.push_back({"NEON port of hperf", [](char *o, const char *i, char *s, size_t n, size_t e) { bitshuf_decode_block_neon(o, i, s, n, e); }}); #endif d.push_back({"JFJochBitUnshuffleBlock (shipped)", [](char *o, const char *i, char *s, size_t n, size_t e) { JFJochBitUnshuffleBlock(o, i, s, n, e); }}); return d; } // Block transform alone: 8192-byte blocks (the EIGER/rugnux default) over `total` bytes. static void bench_blocks(size_t total) { uint64_t x = 12345; for (size_t elem : {1, 2, 4}) { const size_t block = 8192 / elem; std::vector raw(total), enc(total), out(total), scratch(block * elem); for (size_t i = 0; i < total / elem; i++) { // sparse photon counts x = x * 6364136223846793005ULL + 1442695040888963407ULL; const uint32_t v = (x >> 60) < 3 ? (uint32_t) ((x >> 40) % 17) : 0; memcpy(&raw[i * elem], &v, elem); } for (size_t off = 0; off < total; off += block * elem) bshuf_trans_bit_elem(&raw[off], &enc[off], block, elem); printf("block decode, elem_size %zu, block %zu elements, %zu MB:\n", elem, block, total >> 20); for (auto &[name, dec] : decoders()) { memset(out.data(), 0, total); const double t = best_seconds([&] { for (size_t off = 0; off < total; off += block * elem) dec(&out[off], &enc[off], scratch.data(), block, elem); }, 5); printf(" %-36s %8.0f MB/s %s\n", name.c_str(), total / t / 1e6, memcmp(out.data(), raw.data(), total) == 0 ? "" : " MISMATCH"); } } } // Full bitshuffle/LZ4 chunk decode (LZ4 + block transform), as JFJochDecompressHperfPtr does it. static void bench_chunk(const std::string &path, size_t elem) { std::ifstream f(path, std::ios::binary); std::vector src((std::istreambuf_iterator(f)), std::istreambuf_iterator()); if (src.size() < 12) { printf("cannot read %s\n", path.c_str()); return; } const uint8_t *p = (const uint8_t *) src.data(); const size_t total = bshuf_read_uint64_BE(p), block = bshuf_read_uint32_BE(p + 8) / elem, nelem = total / elem; std::vector image(total), dec(block * elem), scratch(block * elem), ref; printf("chunk %s (%zu MB decoded):\n", path.c_str(), total >> 20); for (auto &[name, fn] : decoders()) { const double t = best_seconds([&] { size_t pos = 12, done = 0; while (nelem - done >= 8) { size_t cur = std::min(block, nelem - done); cur -= cur % 8; const size_t csize = bshuf_read_uint32_BE(p + pos); pos += 4; LZ4_decompress_safe(src.data() + pos, dec.data(), (int) csize, (int) (cur * elem)); pos += csize; fn(&image[done * elem], dec.data(), scratch.data(), cur, elem); done += cur; } memcpy(&image[done * elem], src.data() + pos, (nelem - done) * elem); }, 5); if (ref.empty()) ref = image; printf(" %-36s %8.0f MB/s %s\n", name.c_str(), total / t / 1e6, image == ref ? "" : " MISMATCH"); } } int main(int argc, char **argv) { int first = 1; size_t mb = 64; if (argc > 2 && strcmp(argv[1], "-m") == 0) { mb = strtoul(argv[2], nullptr, 10); first = 3; } bench_blocks(mb << 20); for (int i = first; i < argc; i++) bench_chunk(argv[i], 4); }