Files
Jungfraujoch/compression/bitshuffle_hperf/bitshuffle_neon.c
T
leonarski_f 84228bf8be
Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
v1.0.0-rc.173 (#83)
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports.
* jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls.
* Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results.
* Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable.
* Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate.
* Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do.
* Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence.
* Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags.
* Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check.
* Rugnux: Clear error messages when a data set needs more GPU or host memory than is available.

Reviewed-on: #83
Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
2026-09-29 15:57:32 +02:00

175 lines
7.2 KiB
C

// SPDX-License-Identifier: MIT OR Apache-2.0
// Copyright (c) 2023 Kal Conley
// Copyright (c) 2026 Filip Leonarski, Paul Scherrer Institute
//
// Arm NEON port of the decode direction of bitshuffle_hperf (bitshuffle.c in this directory,
// https://github.com/kalcutter/bitshuffle), whose vector code is x86-only. The structure is hperf's:
// each byte plane of the block is bit-untransposed on its own (into `out` for 1-byte elements, into
// `scratch` otherwise), then the planes are byte-interleaved into `out`. The scalar tail and
// transpose8() are hperf's unchanged.
//
// What differs is the 8x8 bit transpose in the vector body. hperf's SSE2/AVX2 code gathers the eight
// row bytes of a column into one 64-bit lane and runs transpose8() on it (shift/xor/and on 64-bit
// lanes). Here the eight rows stay in eight registers, one column per byte lane, and the bits are
// exchanged between registers with shift + bit-select (three levels: 4, 2, 1 bits); the rows then
// become output bytes through one zip level and a four-way interleaving store (vst4q).
// The output is byte-identical to bitshuf_decode_block() and to the classic bitshuffle decoder.
#include "bitshuffle_neon.h"
#if (defined(__ARM_NEON__) || (__ARM_NEON)) && defined(__aarch64__)
#include <arm_neon.h>
#include <stdint.h>
#include <string.h>
// Computes the transpose of an 8x8 bit matrix.
// Ref: "Hacker's Delight" 7-3 by Henry Warren.
static uint64_t transpose8(uint64_t x) {
uint64_t t;
t = (x ^ (x >> 7)) & 0x00aa00aa00aa00aa;
x = (x ^ t ^ (t << 7));
t = (x ^ (x >> 14)) & 0x0000cccc0000cccc;
x = (x ^ t ^ (t << 14));
t = (x ^ (x >> 28)) & 0x00000000f0f0f0f0;
x = (x ^ t ^ (t << 28));
return x;
}
static void bitshuf_untrans_bit_tail(char* restrict out,
const char* restrict in,
size_t size,
size_t index) {
size /= 8;
for (size_t i = index; i < size; i++) {
const uint64_t a = (uint64_t)(uint8_t)in[0 * size + i] |
(uint64_t)(uint8_t)in[1 * size + i] << 8 * 1 |
(uint64_t)(uint8_t)in[2 * size + i] << 8 * 2 |
(uint64_t)(uint8_t)in[3 * size + i] << 8 * 3 |
(uint64_t)(uint8_t)in[4 * size + i] << 8 * 4 |
(uint64_t)(uint8_t)in[5 * size + i] << 8 * 5 |
(uint64_t)(uint8_t)in[6 * size + i] << 8 * 6 |
(uint64_t)(uint8_t)in[7 * size + i] << 8 * 7;
const uint64_t x = transpose8(a);
memcpy(&out[i * 8], &x, sizeof(x));
}
}
// In every byte lane, splits the bits into groups of 2 * shift, each a low and a high field of `shift`
// bits (low_mask selects the low fields), and swaps a's high fields with b's low fields: a keeps its
// low fields and takes b's low fields as its high ones, b keeps its high fields and takes a's high
// fields as its low ones.
static inline void swap_bits(uint8x16_t* a, uint8x16_t* b, int8_t shift, uint8_t low_mask) {
const uint8x16_t low = vdupq_n_u8(low_mask);
const uint8x16_t a_new = vbslq_u8(low, *a, vshlq_u8(*b, vdupq_n_s8(shift)));
const uint8x16_t b_new = vbslq_u8(low, vshlq_u8(*a, vdupq_n_s8(-shift)), *b);
*a = a_new;
*b = b_new;
}
// Bit untranspose of one byte plane: `size` elements, stored as 8 rows of size / 8 bytes. Output byte
// 8 * i + b holds, in bit k, bit b of byte i of row k.
static void bitshuf_untrans_bit_neon(char* restrict out, const char* restrict in, size_t size) {
const size_t row = size / 8;
const uint8_t* src = (const uint8_t*)in;
uint8_t* dst = (uint8_t*)out;
size_t i = 0;
for (; i + 16 <= row; i += 16) {
// r[k] lane c = byte i + c of row k.
uint8x16_t r0 = vld1q_u8(src + 0 * row + i);
uint8x16_t r1 = vld1q_u8(src + 1 * row + i);
uint8x16_t r2 = vld1q_u8(src + 2 * row + i);
uint8x16_t r3 = vld1q_u8(src + 3 * row + i);
uint8x16_t r4 = vld1q_u8(src + 4 * row + i);
uint8x16_t r5 = vld1q_u8(src + 5 * row + i);
uint8x16_t r6 = vld1q_u8(src + 6 * row + i);
uint8x16_t r7 = vld1q_u8(src + 7 * row + i);
// Transpose the 8x8 bit matrix of each lane (register index x bit index), a quarter at a time.
swap_bits(&r0, &r4, 4, 0x0f);
swap_bits(&r1, &r5, 4, 0x0f);
swap_bits(&r2, &r6, 4, 0x0f);
swap_bits(&r3, &r7, 4, 0x0f);
swap_bits(&r0, &r2, 2, 0x33);
swap_bits(&r1, &r3, 2, 0x33);
swap_bits(&r4, &r6, 2, 0x33);
swap_bits(&r5, &r7, 2, 0x33);
swap_bits(&r0, &r1, 1, 0x55);
swap_bits(&r2, &r3, 1, 0x55);
swap_bits(&r4, &r5, 1, 0x55);
swap_bits(&r6, &r7, 1, 0x55);
// Now r[b] lane c is output byte 8 * (i + c) + b. Interleave the eight registers: zipping
// r[b] with r[b + 4] and storing four such vectors interleaved gives r0, r1, ..., r7 per lane.
const uint8x16x4_t lo = {{vzip1q_u8(r0, r4), vzip1q_u8(r1, r5), vzip1q_u8(r2, r6), vzip1q_u8(r3, r7)}};
const uint8x16x4_t hi = {{vzip2q_u8(r0, r4), vzip2q_u8(r1, r5), vzip2q_u8(r2, r6), vzip2q_u8(r3, r7)}};
vst4q_u8(dst + 8 * i, lo);
vst4q_u8(dst + 8 * i + 64, hi);
}
if (i < row)
bitshuf_untrans_bit_tail(out, in, size, i);
}
static void bitshuf_untrans_byte_2_neon(char* restrict out, const char* restrict in, size_t size) {
const uint8_t* src = (const uint8_t*)in;
uint8_t* dst = (uint8_t*)out;
size_t i = 0;
for (; i + 16 <= size; i += 16) {
const uint8x16x2_t v = {{vld1q_u8(src + 0 * size + i), vld1q_u8(src + 1 * size + i)}};
vst2q_u8(dst + 2 * i, v);
}
if (i + 8 <= size) {
const uint8x8x2_t v = {{vld1_u8(src + 0 * size + i), vld1_u8(src + 1 * size + i)}};
vst2_u8(dst + 2 * i, v);
}
}
static void bitshuf_untrans_byte_4_neon(char* restrict out, const char* restrict in, size_t size) {
const uint8_t* src = (const uint8_t*)in;
uint8_t* dst = (uint8_t*)out;
size_t i = 0;
for (; i + 16 <= size; i += 16) {
const uint8x16x4_t v = {{vld1q_u8(src + 0 * size + i), vld1q_u8(src + 1 * size + i),
vld1q_u8(src + 2 * size + i), vld1q_u8(src + 3 * size + i)}};
vst4q_u8(dst + 4 * i, v);
}
if (i + 8 <= size) {
const uint8x8x4_t v = {{vld1_u8(src + 0 * size + i), vld1_u8(src + 1 * size + i),
vld1_u8(src + 2 * size + i), vld1_u8(src + 3 * size + i)}};
vst4_u8(dst + 4 * i, v);
}
}
int bitshuf_decode_block_neon(char* restrict out,
const char* restrict in,
char* restrict scratch,
size_t size,
size_t elem_size) {
if (size & 7)
return -1;
if (elem_size == 1) {
bitshuf_untrans_bit_neon(out, in, size);
return 0;
}
if (!scratch || (elem_size != 2 && elem_size != 4))
return -1;
for (size_t i = 0; i < elem_size; i++)
bitshuf_untrans_bit_neon(&scratch[i * size], &in[i * size], size);
if (elem_size == 2)
bitshuf_untrans_byte_2_neon(out, scratch, size);
else
bitshuf_untrans_byte_4_neon(out, scratch, size);
return 0;
}
#endif