Build Packages / Create release (push) Successful in 24s
Build Packages / build:viewer:macos-arm64:nocuda (push) Successful in 3m29s
Build Packages / build:rugnux:macos-arm64:nocuda (push) Successful in 2m43s
Build Packages / build:rugnux:linux-aarch64:cuda (push) Successful in 8m27s
Build Packages / build:rugnux:linux-x86_64:cuda (push) Successful in 9m53s
Build Packages / build:viewer:linux-x86_64:nocuda (push) Successful in 9m58s
Build Packages / build:viewer:linux-x86_64:cuda (push) Successful in 11m22s
Build Packages / build:jfjoch:rocky8:nocuda (push) Successful in 13m39s
Build Packages / build:viewer:windows-x86_64:nocuda (push) Successful in 18m37s
Build Packages / build:jfjoch:rocky9:nocuda (push) Successful in 16m32s
Build Packages / build:viewer:windows-x86_64:cuda (push) Successful in 24m11s
Build Packages / HDF5 consumer tests (DIALS, XDS) (push) Successful in 25m30s
Build Packages / build:jfjoch:ubuntu2404:nocuda (push) Successful in 19m3s
Build Packages / build:jfjoch:ubuntu2204:nocuda (push) Successful in 20m23s
Build Packages / build:jfjoch:rocky8:cuda-sls9 (push) Successful in 19m41s
Build Packages / Generate python client (push) Successful in 50s
Build Packages / Build documentation (push) Successful in 1m16s
Build Packages / build:jfjoch:rocky9:cuda-sls9 (push) Successful in 21m0s
Build Packages / build:jfjoch:rocky8:cuda (push) Successful in 18m38s
Build Packages / build:rugnux:windows-x86_64:cuda (push) Successful in 14m33s
Build Packages / build:jfjoch:rocky9:cuda (push) Successful in 17m55s
Build Packages / build:jfjoch:ubuntu2204:cuda (push) Successful in 20m50s
Build Packages / build:jfjoch:ubuntu2404:cuda (push) Successful in 18m38s
Build Packages / Unit tests (push) Successful in 1h46m14s
* jfjoch_broker: Optional per-dataset authentication - statistics, images and plots can require a bearer token, which jfjoch_viewer supports. * jfjoch_viewer: Dark mode and a theme-matched colour scheme, a magnifier panel, and simpler contrast and background controls. * Rugnux: Multiple performance improvements on GPU and CPU (CPU-only processing up to 40% faster, faster image decoding on ARM), with unchanged results. * Rugnux: `--model` rigid-body refinement runs on the GPU, and the model-validation check is faster and more reliable. * Rugnux: Improved scaling and merging - error model, outlier rejection, absorption correction and French-Wilson amplitudes now agree more closely with XDS and ctruncate. * Rugnux: Improved integration - radial background on powder and ice rings, crowded rotation data keep their reflections, and CPU-only builds integrate large unit cells as GPU builds do. * Rugnux: More robust detector geometry - measured beam centre, X-ray bandwidth and goniometer rate, and geometry refinement accepted only on significant evidence. * Rugnux: Merged files are written in the standard setting, or in the setting of a reference MTZ, structure-factor mmCIF or model, with its free-R flags. * Rugnux: Richer report - ice and powder rings, further lattices, superstructure candidates and mosaicity, with warnings worded as prompts to check. * Rugnux: Clear error messages when a data set needs more GPU or host memory than is available. Reviewed-on: #83 Co-authored-by: Filip Leonarski <filip.leonarski@psi.ch>
175 lines
7.2 KiB
C
175 lines
7.2 KiB
C
// SPDX-License-Identifier: MIT OR Apache-2.0
|
|
// Copyright (c) 2023 Kal Conley
|
|
// Copyright (c) 2026 Filip Leonarski, Paul Scherrer Institute
|
|
//
|
|
// Arm NEON port of the decode direction of bitshuffle_hperf (bitshuffle.c in this directory,
|
|
// https://github.com/kalcutter/bitshuffle), whose vector code is x86-only. The structure is hperf's:
|
|
// each byte plane of the block is bit-untransposed on its own (into `out` for 1-byte elements, into
|
|
// `scratch` otherwise), then the planes are byte-interleaved into `out`. The scalar tail and
|
|
// transpose8() are hperf's unchanged.
|
|
//
|
|
// What differs is the 8x8 bit transpose in the vector body. hperf's SSE2/AVX2 code gathers the eight
|
|
// row bytes of a column into one 64-bit lane and runs transpose8() on it (shift/xor/and on 64-bit
|
|
// lanes). Here the eight rows stay in eight registers, one column per byte lane, and the bits are
|
|
// exchanged between registers with shift + bit-select (three levels: 4, 2, 1 bits); the rows then
|
|
// become output bytes through one zip level and a four-way interleaving store (vst4q).
|
|
// The output is byte-identical to bitshuf_decode_block() and to the classic bitshuffle decoder.
|
|
#include "bitshuffle_neon.h"
|
|
|
|
#if (defined(__ARM_NEON__) || (__ARM_NEON)) && defined(__aarch64__)
|
|
|
|
#include <arm_neon.h>
|
|
#include <stdint.h>
|
|
#include <string.h>
|
|
|
|
// Computes the transpose of an 8x8 bit matrix.
|
|
// Ref: "Hacker's Delight" 7-3 by Henry Warren.
|
|
static uint64_t transpose8(uint64_t x) {
|
|
uint64_t t;
|
|
t = (x ^ (x >> 7)) & 0x00aa00aa00aa00aa;
|
|
x = (x ^ t ^ (t << 7));
|
|
t = (x ^ (x >> 14)) & 0x0000cccc0000cccc;
|
|
x = (x ^ t ^ (t << 14));
|
|
t = (x ^ (x >> 28)) & 0x00000000f0f0f0f0;
|
|
x = (x ^ t ^ (t << 28));
|
|
return x;
|
|
}
|
|
|
|
static void bitshuf_untrans_bit_tail(char* restrict out,
|
|
const char* restrict in,
|
|
size_t size,
|
|
size_t index) {
|
|
size /= 8;
|
|
|
|
for (size_t i = index; i < size; i++) {
|
|
const uint64_t a = (uint64_t)(uint8_t)in[0 * size + i] |
|
|
(uint64_t)(uint8_t)in[1 * size + i] << 8 * 1 |
|
|
(uint64_t)(uint8_t)in[2 * size + i] << 8 * 2 |
|
|
(uint64_t)(uint8_t)in[3 * size + i] << 8 * 3 |
|
|
(uint64_t)(uint8_t)in[4 * size + i] << 8 * 4 |
|
|
(uint64_t)(uint8_t)in[5 * size + i] << 8 * 5 |
|
|
(uint64_t)(uint8_t)in[6 * size + i] << 8 * 6 |
|
|
(uint64_t)(uint8_t)in[7 * size + i] << 8 * 7;
|
|
const uint64_t x = transpose8(a);
|
|
memcpy(&out[i * 8], &x, sizeof(x));
|
|
}
|
|
}
|
|
|
|
// In every byte lane, splits the bits into groups of 2 * shift, each a low and a high field of `shift`
|
|
// bits (low_mask selects the low fields), and swaps a's high fields with b's low fields: a keeps its
|
|
// low fields and takes b's low fields as its high ones, b keeps its high fields and takes a's high
|
|
// fields as its low ones.
|
|
static inline void swap_bits(uint8x16_t* a, uint8x16_t* b, int8_t shift, uint8_t low_mask) {
|
|
const uint8x16_t low = vdupq_n_u8(low_mask);
|
|
const uint8x16_t a_new = vbslq_u8(low, *a, vshlq_u8(*b, vdupq_n_s8(shift)));
|
|
const uint8x16_t b_new = vbslq_u8(low, vshlq_u8(*a, vdupq_n_s8(-shift)), *b);
|
|
*a = a_new;
|
|
*b = b_new;
|
|
}
|
|
|
|
// Bit untranspose of one byte plane: `size` elements, stored as 8 rows of size / 8 bytes. Output byte
|
|
// 8 * i + b holds, in bit k, bit b of byte i of row k.
|
|
static void bitshuf_untrans_bit_neon(char* restrict out, const char* restrict in, size_t size) {
|
|
const size_t row = size / 8;
|
|
const uint8_t* src = (const uint8_t*)in;
|
|
uint8_t* dst = (uint8_t*)out;
|
|
|
|
size_t i = 0;
|
|
for (; i + 16 <= row; i += 16) {
|
|
// r[k] lane c = byte i + c of row k.
|
|
uint8x16_t r0 = vld1q_u8(src + 0 * row + i);
|
|
uint8x16_t r1 = vld1q_u8(src + 1 * row + i);
|
|
uint8x16_t r2 = vld1q_u8(src + 2 * row + i);
|
|
uint8x16_t r3 = vld1q_u8(src + 3 * row + i);
|
|
uint8x16_t r4 = vld1q_u8(src + 4 * row + i);
|
|
uint8x16_t r5 = vld1q_u8(src + 5 * row + i);
|
|
uint8x16_t r6 = vld1q_u8(src + 6 * row + i);
|
|
uint8x16_t r7 = vld1q_u8(src + 7 * row + i);
|
|
|
|
// Transpose the 8x8 bit matrix of each lane (register index x bit index), a quarter at a time.
|
|
swap_bits(&r0, &r4, 4, 0x0f);
|
|
swap_bits(&r1, &r5, 4, 0x0f);
|
|
swap_bits(&r2, &r6, 4, 0x0f);
|
|
swap_bits(&r3, &r7, 4, 0x0f);
|
|
|
|
swap_bits(&r0, &r2, 2, 0x33);
|
|
swap_bits(&r1, &r3, 2, 0x33);
|
|
swap_bits(&r4, &r6, 2, 0x33);
|
|
swap_bits(&r5, &r7, 2, 0x33);
|
|
|
|
swap_bits(&r0, &r1, 1, 0x55);
|
|
swap_bits(&r2, &r3, 1, 0x55);
|
|
swap_bits(&r4, &r5, 1, 0x55);
|
|
swap_bits(&r6, &r7, 1, 0x55);
|
|
|
|
// Now r[b] lane c is output byte 8 * (i + c) + b. Interleave the eight registers: zipping
|
|
// r[b] with r[b + 4] and storing four such vectors interleaved gives r0, r1, ..., r7 per lane.
|
|
const uint8x16x4_t lo = {{vzip1q_u8(r0, r4), vzip1q_u8(r1, r5), vzip1q_u8(r2, r6), vzip1q_u8(r3, r7)}};
|
|
const uint8x16x4_t hi = {{vzip2q_u8(r0, r4), vzip2q_u8(r1, r5), vzip2q_u8(r2, r6), vzip2q_u8(r3, r7)}};
|
|
vst4q_u8(dst + 8 * i, lo);
|
|
vst4q_u8(dst + 8 * i + 64, hi);
|
|
}
|
|
if (i < row)
|
|
bitshuf_untrans_bit_tail(out, in, size, i);
|
|
}
|
|
|
|
static void bitshuf_untrans_byte_2_neon(char* restrict out, const char* restrict in, size_t size) {
|
|
const uint8_t* src = (const uint8_t*)in;
|
|
uint8_t* dst = (uint8_t*)out;
|
|
|
|
size_t i = 0;
|
|
for (; i + 16 <= size; i += 16) {
|
|
const uint8x16x2_t v = {{vld1q_u8(src + 0 * size + i), vld1q_u8(src + 1 * size + i)}};
|
|
vst2q_u8(dst + 2 * i, v);
|
|
}
|
|
if (i + 8 <= size) {
|
|
const uint8x8x2_t v = {{vld1_u8(src + 0 * size + i), vld1_u8(src + 1 * size + i)}};
|
|
vst2_u8(dst + 2 * i, v);
|
|
}
|
|
}
|
|
|
|
static void bitshuf_untrans_byte_4_neon(char* restrict out, const char* restrict in, size_t size) {
|
|
const uint8_t* src = (const uint8_t*)in;
|
|
uint8_t* dst = (uint8_t*)out;
|
|
|
|
size_t i = 0;
|
|
for (; i + 16 <= size; i += 16) {
|
|
const uint8x16x4_t v = {{vld1q_u8(src + 0 * size + i), vld1q_u8(src + 1 * size + i),
|
|
vld1q_u8(src + 2 * size + i), vld1q_u8(src + 3 * size + i)}};
|
|
vst4q_u8(dst + 4 * i, v);
|
|
}
|
|
if (i + 8 <= size) {
|
|
const uint8x8x4_t v = {{vld1_u8(src + 0 * size + i), vld1_u8(src + 1 * size + i),
|
|
vld1_u8(src + 2 * size + i), vld1_u8(src + 3 * size + i)}};
|
|
vst4_u8(dst + 4 * i, v);
|
|
}
|
|
}
|
|
|
|
int bitshuf_decode_block_neon(char* restrict out,
|
|
const char* restrict in,
|
|
char* restrict scratch,
|
|
size_t size,
|
|
size_t elem_size) {
|
|
if (size & 7)
|
|
return -1;
|
|
|
|
if (elem_size == 1) {
|
|
bitshuf_untrans_bit_neon(out, in, size);
|
|
return 0;
|
|
}
|
|
|
|
if (!scratch || (elem_size != 2 && elem_size != 4))
|
|
return -1;
|
|
|
|
for (size_t i = 0; i < elem_size; i++)
|
|
bitshuf_untrans_bit_neon(&scratch[i * size], &in[i * size], size);
|
|
|
|
if (elem_size == 2)
|
|
bitshuf_untrans_byte_2_neon(out, scratch, size);
|
|
else
|
|
bitshuf_untrans_byte_4_neon(out, scratch, size);
|
|
return 0;
|
|
}
|
|
|
|
#endif
|