superintervals 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: superintervals
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Rapid interval intersections
5
5
  Author: Kez Cleal
6
6
  Author-email: Kez Cleal <clealk@cardiff.ac.uk>
@@ -91,11 +91,11 @@ exceptions. Coitrees-s was faster for one test (ONT reads, sorted DB53 reads).
91
91
 
92
92
  Datasets:
93
93
 
94
- 1. `rna / anno` RNA-seq reads and annotations from cgranges repository
95
- 2. `ONT reads` nanopore alignments from sample PAO33946 chr1, converted to bed format
96
- 3. `DB53 reads` paired-end reads from sample DB53, NCBI BioProject PRJNA417592, chr1, converted to bed format
97
- 4. `mito-b, mito-a` paired-end reads from sample DB53 chrM, converted to bed format (mito-b and mito-a are the same)
98
- 5. `genes` UCSC genes from hg19
94
+ - `rna / anno` RNA-seq reads and annotations from cgranges repository
95
+ - `ONT reads` nanopore alignments from sample PAO33946 chr1, converted to bed format
96
+ - `DB53 reads` paired-end reads from sample DB53, NCBI BioProject PRJNA417592, chr1, converted to bed format
97
+ - `mito-b, mito-a` paired-end reads from sample DB53 chrM, converted to bed format (mito-b and mito-a are the same)
98
+ - `genes` UCSC genes from hg19
99
99
 
100
100
  Test programs use internal timers and print data to stdout, measuring the index time, and time to find all intersections. Other steps such as file IO are ignored. Test programs also only assess chr1 bed records - other chromosomes are ignored. For 'chrM' records, the M was replaced with 1 using sed. Data were assessed in position sorted and random order. Datasets can be found on the Releases page, and the test/run_tools.sh script has instructions for how to repeat the benchmark.
101
101
 
@@ -8,7 +8,7 @@ build-backend = "setuptools.build_meta"
8
8
 
9
9
  [project]
10
10
  name = "superintervals"
11
- version = "0.2.2"
11
+ version = "0.2.3"
12
12
  description = "Rapid interval intersections"
13
13
  dependencies = ['Cython']
14
14
  authors = [{name = "Kez Cleal", email = "clealk@cardiff.ac.uk"}]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: superintervals
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Rapid interval intersections
5
5
  Author: Kez Cleal
6
6
  Author-email: Kez Cleal <clealk@cardiff.ac.uk>
@@ -6,10 +6,14 @@
6
6
  #include <climits>
7
7
  #include <iostream>
8
8
  #include <limits>
9
- #if defined(__AVX2__)
10
- #include <immintrin.h>
11
- #elif defined(__ARM_NEON__)
12
- #include <arm_neon.h>
9
+ #ifndef SI_NOSIMD
10
+ #if defined(__AVX2__)
11
+ #include <immintrin.h>
12
+ #elif defined(__ARM_NEON__) || defined(__aarch64__)
13
+ #include <arm_neon.h>
14
+ #else
15
+ #define SI_NOSIMD
16
+ #endif
13
17
  #endif
14
18
 
15
19
  /**
@@ -282,25 +286,37 @@ class SuperIntervals {
282
286
  size_t found = 0;
283
287
  size_t i = idx;
284
288
 
285
- #ifdef __AVX2__
289
+ #ifdef SI_NOSIMD
290
+ constexpr size_t block = 16;
291
+ #elif defined(__AVX2__)
286
292
  __m256i start_vec = _mm256_set1_epi32(start);
287
293
  constexpr size_t simd_width = 256 / (sizeof(S) * 8);
288
294
  constexpr size_t block = simd_width * 4;
289
- #elif defined __ARM_NEON
295
+ #elif defined(__ARM_NEON__) || defined(__aarch64__)
290
296
  int32x4_t start_vec = vdupq_n_s32(start);
291
297
  constexpr size_t simd_width = 128 / (sizeof(S) * 8);
292
298
  uint32x4_t ones = vdupq_n_u32(1);
293
299
  constexpr size_t block = simd_width * 4;
294
- #else
295
- constexpr size_t block = 16;
296
300
  #endif
297
301
 
298
302
  while (i > 0) {
299
303
  if (start <= ends[i]) {
300
304
  ++found;
301
305
  --i;
306
+ #ifdef SI_NOSIMD
307
+ while (i > block) { // Rely on compiler auto vectorize
308
+ size_t count = 0;
309
+ for (size_t j = i; j > i - block; --j) {
310
+ count += (start <= ends[j]) ? 1 : 0;
311
+ }
312
+ found += count;
313
+ i -= block;
314
+ if (count < block && start > ends[i + 1]) { // check for a branch
315
+ break;
316
+ }
317
+ }
302
318
 
303
- #ifdef __AVX2__
319
+ #elif defined(__AVX2__)
304
320
  while (i > block) {
305
321
  size_t count = 0;
306
322
  for (size_t j = i; j > i - block; j -= simd_width) {
@@ -315,7 +331,7 @@ class SuperIntervals {
315
331
  break;
316
332
  }
317
333
  }
318
- #elif defined(__ARM_NEON__)
334
+ #elif defined(__ARM_NEON__) || defined(__aarch64__)
319
335
  while (i > block) {
320
336
  size_t count = 0;
321
337
  uint32x4_t mask, bool_mask;
@@ -332,19 +348,6 @@ class SuperIntervals {
332
348
  break;
333
349
  }
334
350
  }
335
- #else // Rely on compiler auto vectorize
336
- while (i > block) {
337
- size_t count = 0;
338
- for (size_t j = i; j > i - block; --j) {
339
- count += (start <= ends[j]) ? 1 : 0;
340
- }
341
- found += count;
342
- i -= block;
343
- if (count < block && start > ends[i + 1]) { // check for a branch
344
- // if (count < block) { // check for a branch
345
- break;
346
- }
347
- }
348
351
  #endif
349
352
  } else {
350
353
  if (branch[i] >= i) {
@@ -419,17 +422,17 @@ class SuperIntervals {
419
422
  size_t found = 0;
420
423
  size_t i = idx;
421
424
 
422
- #ifdef __AVX2__
425
+ #ifdef SI_NOSIMD
426
+ constexpr size_t block = 16;
427
+ #elif defined(__AVX2__)
423
428
  __m256i start_vec = _mm256_set1_epi32(point);
424
429
  constexpr size_t simd_width = 256 / (sizeof(S) * 8);
425
430
  constexpr size_t block = simd_width * 4;
426
- #elif defined __ARM_NEON
431
+ #elif defined(__ARM_NEON__) || defined(__aarch64__)
427
432
  int32x4_t start_vec = vdupq_n_s32(point);
428
433
  constexpr size_t simd_width = 128 / (sizeof(S) * 8);
429
434
  uint32x4_t ones = vdupq_n_u32(1);
430
435
  constexpr size_t block = simd_width * 4;
431
- #else
432
- constexpr size_t block = 16;
433
436
  #endif
434
437
 
435
438
  while (i > 0) {
@@ -437,7 +440,19 @@ class SuperIntervals {
437
440
  ++found;
438
441
  --i;
439
442
 
440
- #ifdef __AVX2__
443
+ #ifdef SI_NOSIMD
444
+ while (i > block) {
445
+ size_t count = 0;
446
+ for (size_t j = i; j > i - block; --j) {
447
+ count += (point <= ends[j]) ? 1 : 0;
448
+ }
449
+ found += count;
450
+ i -= block;
451
+ if (count < block) { // go check for a branch
452
+ break;
453
+ }
454
+ }
455
+ #elif defined(__AVX2__)
441
456
  while (i > block) {
442
457
  size_t count = 0;
443
458
  for (size_t j = i; j > i - block; j -= simd_width) {
@@ -452,7 +467,7 @@ class SuperIntervals {
452
467
  break;
453
468
  }
454
469
  }
455
- #elif defined __ARM_NEON
470
+ #elif defined(__ARM_NEON__) || defined(__aarch64__)
456
471
  while (i > block) {
457
472
  size_t count = 0;
458
473
  uint32x4_t bool_mask;
@@ -468,18 +483,6 @@ class SuperIntervals {
468
483
  break;
469
484
  }
470
485
  }
471
- #else
472
- while (i > block) {
473
- size_t count = 0;
474
- for (size_t j = i; j > i - block; --j) {
475
- count += (point <= ends[j]) ? 1 : 0;
476
- }
477
- found += count;
478
- i -= block;
479
- if (count < block) { // go check for a branch
480
- break;
481
- }
482
- }
483
486
  #endif
484
487
  } else {
485
488
  if (branch[i] >= i) {
File without changes
File without changes
File without changes