seqpro 0.16.0__tar.gz → 0.18.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/bench.yaml +2 -2
  2. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/lint.yaml +2 -2
  3. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/test.yaml +2 -2
  4. {seqpro-0.16.0 → seqpro-0.18.0}/.pre-commit-config.yaml +18 -0
  5. {seqpro-0.16.0 → seqpro-0.18.0}/CHANGELOG.md +110 -0
  6. {seqpro-0.16.0 → seqpro-0.18.0}/PKG-INFO +1 -1
  7. seqpro-0.18.0/benches/bench_tokenize_translate.py +30 -0
  8. seqpro-0.18.0/benchmarks/bench_ragged_backends.py +481 -0
  9. seqpro-0.18.0/docs/roadmap/rust-ragged.md +388 -0
  10. seqpro-0.18.0/docs/superpowers/plans/2026-06-18-rust-tokenize-translate.md +1391 -0
  11. seqpro-0.18.0/docs/superpowers/plans/2026-06-19-rust-ragged-core.md +1347 -0
  12. seqpro-0.18.0/docs/superpowers/plans/2026-06-20-rust-ragged-records.md +1583 -0
  13. seqpro-0.18.0/docs/superpowers/plans/2026-06-20-rust-ragged-spec-c.md +1556 -0
  14. seqpro-0.18.0/docs/superpowers/plans/2026-06-21-ragged-throughput-gate.md +656 -0
  15. seqpro-0.18.0/docs/superpowers/plans/2026-06-21-rust-ragged-consumer-audit.md +492 -0
  16. seqpro-0.18.0/docs/superpowers/specs/2026-06-18-rust-tokenize-translate-design.md +360 -0
  17. seqpro-0.18.0/docs/superpowers/specs/2026-06-19-rust-ragged-core-design.md +260 -0
  18. seqpro-0.18.0/docs/superpowers/specs/2026-06-20-rust-ragged-nested-design.md +364 -0
  19. seqpro-0.18.0/docs/superpowers/specs/2026-06-20-rust-ragged-records-design.md +327 -0
  20. seqpro-0.18.0/docs/superpowers/specs/2026-06-21-ragged-throughput-gate-design.md +166 -0
  21. seqpro-0.18.0/docs/superpowers/specs/2026-06-21-rust-ragged-audit-ledger.md +75 -0
  22. seqpro-0.18.0/docs/superpowers/specs/2026-06-21-rust-ragged-consumer-audit-design.md +121 -0
  23. {seqpro-0.16.0 → seqpro-0.18.0}/pixi.lock +52 -16
  24. {seqpro-0.16.0 → seqpro-0.18.0}/pixi.toml +10 -1
  25. {seqpro-0.16.0 → seqpro-0.18.0}/pyproject.toml +7 -1
  26. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/alphabets/_alphabets.py +137 -92
  27. seqpro-0.18.0/python/seqpro/rag/__init__.py +29 -0
  28. seqpro-0.18.0/python/seqpro/rag/_ak_interop.py +208 -0
  29. seqpro-0.18.0/python/seqpro/rag/_core.py +1612 -0
  30. seqpro-0.18.0/python/seqpro/rag/_ingest.py +252 -0
  31. seqpro-0.18.0/python/seqpro/rag/_layout.py +172 -0
  32. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/rag/_ops.py +161 -49
  33. {seqpro-0.16.0 → seqpro-0.18.0}/skills/seqpro/SKILL.md +35 -24
  34. {seqpro-0.16.0 → seqpro-0.18.0}/src/kshuffle.rs +132 -65
  35. seqpro-0.18.0/src/lib.rs +212 -0
  36. seqpro-0.18.0/src/ragged.rs +612 -0
  37. seqpro-0.18.0/src/translate.rs +510 -0
  38. seqpro-0.18.0/tests/test_concatenate.py +29 -0
  39. seqpro-0.18.0/tests/test_core_ragged_surface.py +36 -0
  40. seqpro-0.18.0/tests/test_ingest.py +19 -0
  41. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_rag_to_packed.py +93 -28
  42. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_ragged.py +57 -34
  43. seqpro-0.18.0/tests/test_ragged_core.py +810 -0
  44. seqpro-0.18.0/tests/test_ragged_core_records.py +682 -0
  45. seqpro-0.18.0/tests/test_ragged_nested_consumers.py +68 -0
  46. seqpro-0.18.0/tests/test_ragged_nested_diff.py +297 -0
  47. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_ragged_rc.py +20 -26
  48. seqpro-0.18.0/tests/test_ragged_record_indexing.py +47 -0
  49. seqpro-0.18.0/tests/test_ragged_subclass.py +60 -0
  50. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_ragged_to_padded.py +66 -6
  51. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_translate.py +48 -19
  52. seqpro-0.18.0/tests/test_translate_rust.py +135 -0
  53. seqpro-0.16.0/python/seqpro/rag/__init__.py +0 -15
  54. seqpro-0.16.0/python/seqpro/rag/_array.py +0 -904
  55. seqpro-0.16.0/python/seqpro/rag/_gufuncs.py +0 -14
  56. seqpro-0.16.0/src/lib.rs +0 -44
  57. {seqpro-0.16.0 → seqpro-0.18.0}/.claude/skills/zensical/SKILL.md +0 -0
  58. {seqpro-0.16.0 → seqpro-0.18.0}/.gitattributes +0 -0
  59. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/bump.yaml +0 -0
  60. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/docs.yml +0 -0
  61. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/merge.yaml +0 -0
  62. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/publish.yaml +0 -0
  63. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/release-pipeline.yaml +0 -0
  64. {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/release.yaml +0 -0
  65. {seqpro-0.16.0 → seqpro-0.18.0}/.gitignore +0 -0
  66. {seqpro-0.16.0 → seqpro-0.18.0}/CLAUDE.md +0 -0
  67. {seqpro-0.16.0 → seqpro-0.18.0}/Cargo.lock +0 -0
  68. {seqpro-0.16.0 → seqpro-0.18.0}/Cargo.toml +0 -0
  69. {seqpro-0.16.0 → seqpro-0.18.0}/LICENSE +0 -0
  70. {seqpro-0.16.0 → seqpro-0.18.0}/README.md +0 -0
  71. {seqpro-0.16.0 → seqpro-0.18.0}/benches/kshuffle.rs +0 -0
  72. {seqpro-0.16.0 → seqpro-0.18.0}/benchmarks/bench_to_packed.py +0 -0
  73. {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/alphabets.md +0 -0
  74. {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/bed.md +0 -0
  75. {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/gtf.md +0 -0
  76. {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/index.md +0 -0
  77. {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/ragged.md +0 -0
  78. {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/types.md +0 -0
  79. {seqpro-0.16.0 → seqpro-0.18.0}/docs/index.md +0 -0
  80. {seqpro-0.16.0 → seqpro-0.18.0}/docs/ragged.md +0 -0
  81. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-04-ragged-record-array.md +0 -0
  82. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-05-documentation-site.md +0 -0
  83. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-05-narwhals-coord-schema.md +0 -0
  84. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-05-ragged-zip-and-record-introspection.md +0 -0
  85. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-20-kshuffle-optimization.md +0 -0
  86. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-20-kshuffle-pooled-buffers-and-k2-fast-path.md +0 -0
  87. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-20-kshuffle-wilson-single-pass.md +0 -0
  88. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-20-release-pipeline.md +0 -0
  89. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-28-translate-lut-validation.md +0 -0
  90. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-31-flat-buffer-to-padded.md +0 -0
  91. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-31-rag-to-packed.md +0 -0
  92. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-06-05-translate-unknown-codon-policy.md +0 -0
  93. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-06-12-tokenize-lut-codspeed.md +0 -0
  94. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-04-ragged-record-array-design.md +0 -0
  95. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-05-documentation-site-design.md +0 -0
  96. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-05-narwhals-coord-schema-design.md +0 -0
  97. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-05-ragged-zip-and-record-introspection-design.md +0 -0
  98. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-20-kshuffle-optimization-design.md +0 -0
  99. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-20-kshuffle-pooled-buffers-and-k2-fast-path-design.md +0 -0
  100. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-20-kshuffle-wilson-single-pass-design.md +0 -0
  101. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-20-release-pipeline-design.md +0 -0
  102. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-28-translate-lut-and-validation-design.md +0 -0
  103. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-31-flat-buffer-to-padded-design.md +0 -0
  104. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-31-rag-to-packed-design.md +0 -0
  105. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-06-05-translate-unknown-codon-policy-design.md +0 -0
  106. {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-06-12-tokenize-lut-codspeed-design.md +0 -0
  107. {seqpro-0.16.0 → seqpro-0.18.0}/meta.yaml +0 -0
  108. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/__init__.py +0 -0
  109. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_analyzers.py +0 -0
  110. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_cleaners.py +0 -0
  111. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_coords.py +0 -0
  112. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_encoders.py +0 -0
  113. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_modifiers.py +0 -0
  114. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_numba.py +0 -0
  115. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_types.py +0 -0
  116. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_utils.py +0 -0
  117. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/alphabets/__init__.py +0 -0
  118. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/bed.py +0 -0
  119. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/experimental/_experimental.py +0 -0
  120. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/experimental/_visualizers.py +0 -0
  121. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/gtf.py +0 -0
  122. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/py.typed +0 -0
  123. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/rag/_types.py +0 -0
  124. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/rag/_utils.py +0 -0
  125. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/transforms/__init__.py +0 -0
  126. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/transforms/augmentation.py +0 -0
  127. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/transforms/tmm.py +0 -0
  128. {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/xr/__init__.py +0 -0
  129. {seqpro-0.16.0 → seqpro-0.18.0}/scratch_bench_rc.py +0 -0
  130. {seqpro-0.16.0 → seqpro-0.18.0}/scratch_bench_to_padded.py +0 -0
  131. {seqpro-0.16.0 → seqpro-0.18.0}/src/kmer_encode.rs +0 -0
  132. {seqpro-0.16.0 → seqpro-0.18.0}/src/kshuffle_ref.rs +0 -0
  133. {seqpro-0.16.0 → seqpro-0.18.0}/tests/_shape_fixtures.py +0 -0
  134. {seqpro-0.16.0 → seqpro-0.18.0}/tests/bed/test_pyranges.py +0 -0
  135. {seqpro-0.16.0 → seqpro-0.18.0}/tests/bed/test_read.py +0 -0
  136. {seqpro-0.16.0 → seqpro-0.18.0}/tests/bed/test_sort.py +0 -0
  137. {seqpro-0.16.0 → seqpro-0.18.0}/tests/bed/test_with_length.py +0 -0
  138. {seqpro-0.16.0 → seqpro-0.18.0}/tests/bench_translate_lut.py +0 -0
  139. {seqpro-0.16.0 → seqpro-0.18.0}/tests/conftest.py +0 -0
  140. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_analyzers.py +0 -0
  141. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_bench_tokenize.py +0 -0
  142. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_coords.py +0 -0
  143. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_encoders.py +0 -0
  144. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_modifiers.py +0 -0
  145. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_ohe.py +0 -0
  146. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_shape_matrix.py +0 -0
  147. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_tokenize.py +0 -0
  148. {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_transforms.py +0 -0
  149. {seqpro-0.16.0 → seqpro-0.18.0}/zensical.toml +0 -0
@@ -25,10 +25,10 @@ jobs:
25
25
  - name: Setup pixi
26
26
  uses: prefix-dev/setup-pixi@v0.9.5
27
27
  with:
28
- pixi-version: v0.67.2
28
+ pixi-version: v0.70.1
29
29
  cache: true
30
30
  environments: bench
31
- locked: false
31
+ locked: true
32
32
  - name: Build extension
33
33
  run: pixi run -e bench maturin develop
34
34
  - name: Run benchmarks
@@ -11,8 +11,8 @@ jobs:
11
11
  - name: Setup pixi
12
12
  uses: prefix-dev/setup-pixi@v0.9.5
13
13
  with:
14
- pixi-version: v0.67.2
14
+ pixi-version: v0.70.1
15
15
  cache: true
16
16
  environments: default
17
- locked: false
17
+ locked: true
18
18
  - uses: j178/prek-action@v2
@@ -21,9 +21,9 @@ jobs:
21
21
  - name: Setup pixi
22
22
  uses: prefix-dev/setup-pixi@v0.9.5
23
23
  with:
24
- pixi-version: v0.67.2
24
+ pixi-version: v0.70.1
25
25
  cache: true
26
26
  environments: ${{ matrix.environment }}
27
- locked: false
27
+ locked: true
28
28
  - name: Test
29
29
  run: pixi run -e ${{ matrix.environment }} test
@@ -33,3 +33,21 @@ repos:
33
33
  language: system
34
34
  stages: [pre-push]
35
35
  pass_filenames: false
36
+ - id: cargo-check
37
+ name: Type-check Rust with cargo check
38
+ entry: cargo check --all-targets --all-features
39
+ language: system
40
+ types: [rust]
41
+ pass_filenames: false
42
+ - id: cargo-fmt
43
+ name: Format Rust with cargo fmt
44
+ entry: cargo fmt --all
45
+ language: system
46
+ types: [rust]
47
+ pass_filenames: false
48
+ - id: cargo-clippy
49
+ name: Lint Rust with cargo clippy
50
+ entry: cargo clippy --all-targets -- -D warnings
51
+ language: system
52
+ types: [rust]
53
+ pass_filenames: false
@@ -1,3 +1,113 @@
1
+ ## 0.18.0 (2026-06-23)
2
+
3
+ ### Feat
4
+
5
+ - **rag**: reshape/squeeze/to_packed preserve Ragged subclass
6
+ - **rag**: __getitem__ preserves subclass on positional indexing
7
+ - **rag**: add _with_layout subclass-preserving constructor
8
+
9
+ ### Fix
10
+
11
+ - **rag**: non-tuple record indexing matches numpy (A[x] == A[(x,)])
12
+
13
+ ## 0.17.0 (2026-06-23)
14
+
15
+ ### Feat
16
+
17
+ - **rag**: concatenate() along ragged axis (Rust kernel)
18
+ - **rag**: to_packed() on record and opaque-string-under-axis Ragged
19
+ - wire rag-gate pixi task; record throughput gate outcome
20
+ - string-conversion op cells for Ragged throughput gate
21
+ - nested R=2 op cells for Ragged throughput gate
22
+ - record (SoA) op cells for Ragged throughput gate
23
+ - single-level op cells for Ragged throughput gate
24
+ - harness core for rust-vs-awkward Ragged throughput gate
25
+ - _ingest bridge for R=2 + string-under-axis (oracle interop)
26
+ - nested record indexing + record-aware to_packed/to_padded/to_numpy
27
+ - nested + string-under-axis record fields sharing full offsets list
28
+ - R=2 lengths/squeeze/reshape
29
+ - per-axis nested to_padded + rectangular to_numpy (R=2); trailing-dim support in to_padded
30
+ - nested to_packed via nested_pack kernel
31
+ - nested_pack Rust kernel + binding for two-level pack
32
+ - per-group inner mask/int-array indexing via nested_gather
33
+ - nested_gather Rust kernel + binding for per-group middle selection
34
+ - per-group inner int/slice indexing (rag[:, k], rag[:, a:b])
35
+ - R=2 tuple indexing + leaf access via peel chaining
36
+ - R=2 outer-row indexing (lazy gather, peel to 1-level)
37
+ - string-under-axis leaf + nested to_chars/to_strings
38
+ - nested constructors from_offsets(list)/from_lengths(tuple) for R=2
39
+ - validate R=2 nested ragged layouts; cap at R<=2
40
+ - ingest/emit record layouts via awkward bridge (oracle interop)
41
+ - record to_numpy/to_padded dicts; raise view/ufunc on records
42
+ - record-aware to_packed (one shared packed offsets across fields)
43
+ - per-field squeeze/reshape on record Ragged
44
+ - record row-axis indexing (slice/mask -> record, int -> dict)
45
+ - record field access (key/attr) and __setitem__ mutation
46
+ - record-branch properties (data/dtype/offsets/shape/fields/state)
47
+ - add Ragged.from_fields record constructor and rag.zip alias
48
+ - add RecordLayout value object and validation arm
49
+ - add zero-copy to_chars/to_strings between opaque-string and char Ragged
50
+ - disambiguate opaque-string vs char layout by presence of None in shape
51
+ - report np.dtype('S') for opaque-string Ragged (string/char duality)
52
+ - **rust**: ragged_validate and ragged_select kernels
53
+ - **rag**: awkward ingestion and to_ak shim
54
+ - **rag**: to_numpy, to_packed, to_padded
55
+ - **rag**: squeeze and reshape on regular dims
56
+ - **rag**: element-wise ufunc interop
57
+ - **rag**: __getitem__ indexing and slicing
58
+ - **rag**: state predicates and view
59
+ - **rag**: Ragged constructors and core properties
60
+ - **rag**: RaggedLayout value object + validation
61
+ - **translate**: self-contained Rust OHE<->AA path with native drop
62
+ - **translate**: route truncate_stop through Rust
63
+ - **translate**: route unknown=drop compaction through Rust
64
+ - **translate**: route ragged bytes pad path through Rust
65
+ - **translate**: route dense pad path through Rust
66
+ - **rust**: add codon-stride translate kernels
67
+ - **tokenize**: route ragged path through Rust, drop Numba gather
68
+ - **tokenize**: route dense path through Rust _tokenize
69
+ - **rust**: add tokenize LUT gather kernel
70
+
71
+ ### Fix
72
+
73
+ - **rag**: to_ak() on multi-leading-axis record Ragged
74
+ - **rag**: _core.Ragged/tokenize fixes found via genvarformer audit
75
+ - **rag**: add __len__, np.newaxis support, and element-wise fancy indexing to _core.Ragged
76
+ - **rag**: _ops.to_padded record detection works for both _array and _core backends
77
+ - **rag**: _core.to_numpy returns dict for records (restore designed contract); update _array-era test
78
+ - **rag**: correct _core getitem tuple routing for rag_dim==1 found via SeqPro own-suite audit
79
+ - **rag**: port _ops + _core to_packed/is_rag_dtype/to_numpy to _core object model
80
+ - **rag**: precise _core getitem routing — _core contract + genoray parity both green
81
+ - **rag**: repair _core.Ragged getitem regressions (string leaf, to_strings, r2 int array)
82
+ - **rag**: match _array is_base semantics for non-ndarray base
83
+ - **rag**: _core.Ragged fixes found via genoray audit
84
+ - **bench**: time to_packed on unpacked input so single-level pack compares equal work
85
+ - forward --repeats CLI flag into time_callable
86
+ - R=2 is_contiguous checks all offset levels; guard string-under-axis to_packed/to_numpy
87
+ - panic-safety in nested_pack (checked_mul, elem>0, o0-range guard)
88
+ - reject negative inner-slice bounds in rag[:, a:b]
89
+ - -O-safe record guards on is_string/to_chars/to_strings; note latent S4 over-acceptance
90
+ - raise TypeError on np.asarray of a record Ragged (-O-safe, was a bare assert)
91
+ - **rag**: reject mismatched boolean masks and bounds-check ragged_select
92
+ - **translate**: handle empty dense input in unknown=drop path
93
+
94
+ ### Refactor
95
+
96
+ - **rag**: delete awkward _array backend; relocate interop to _ak_interop
97
+ - resolve clippy findings for -D warnings
98
+ - **rag**: route index/validate hot paths through Rust
99
+ - **rag**: use NDArrayOperatorsMixin for operator dunders
100
+ - **tokenize**: restore np.take/Numba path; Rust port regressed vs baseline
101
+
102
+ ### Perf
103
+
104
+ - validate pack output size up-front; test parallel pack path
105
+ - Rust single-level pack kernel for to_packed (parallel gather)
106
+ - opt-in Ragged validation (default off) + slice indexing fast-path
107
+ - drop redundant per-field copy in record to_packed; align record is_base with single-level one-indirection rule
108
+ - **translate**: optimize LUT gather kernel (validity table + coarse chunking)
109
+ - **translate**: add rayon parallel path for large inputs
110
+
1
111
  ## 0.16.0 (2026-06-14)
2
112
 
3
113
  ### Feat
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: seqpro
3
- Version: 0.16.0
3
+ Version: 0.18.0
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Programming Language :: Python :: Implementation :: CPython
6
6
  Classifier: Programming Language :: Python :: Implementation :: PyPy
@@ -0,0 +1,30 @@
1
+ """End-to-end Python-level benchmarks for the Rust tokenize/translate port.
2
+
3
+ Run: pixi run -e dev pytest benches/bench_tokenize_translate.py --benchmark-only
4
+
5
+ Measures the public API (including PyO3 marshalling), which Rust-only criterion
6
+ misses and which decides the small-array regime. To compare against the Numba
7
+ baseline, check out the pre-port commit and run the same file.
8
+ """
9
+
10
+ import numpy as np
11
+ import pytest
12
+
13
+ import seqpro as sp
14
+
15
+ TOKEN_MAP = {"A": 0, "C": 1, "G": 2, "T": 3}
16
+ SIZES = [100, 10_000, 1_000_000]
17
+
18
+
19
+ @pytest.mark.parametrize("n", SIZES)
20
+ def test_bench_tokenize_dense(benchmark, n):
21
+ rng = np.random.default_rng(0)
22
+ seqs = rng.choice(np.frombuffer(b"ACGT", "S1"), size=n)
23
+ benchmark(lambda: sp.tokenize(seqs, TOKEN_MAP, 7))
24
+
25
+
26
+ @pytest.mark.parametrize("n_codons", [33, 3_333, 333_333])
27
+ def test_bench_translate_dense(benchmark, n_codons):
28
+ rng = np.random.default_rng(1)
29
+ seqs = rng.choice(np.frombuffer(b"ACGT", "S1"), size=n_codons * 3)
30
+ benchmark(lambda: sp.AA.translate(seqs, length_axis=0))
@@ -0,0 +1,481 @@
1
+ """Local, transitional A-vs-B throughput gate: rust-native Ragged vs awkward.
2
+
3
+ Proves seqpro.rag._core.Ragged is at least as fast as awkward's native layout
4
+ algebra before the Spec D cutover drops awkward. NOT a permanent CI fixture —
5
+ deleted at cutover (see docs/superpowers/specs/2026-06-21-ragged-throughput-gate-design.md).
6
+
7
+ Run: pixi run -e bench rag-gate (or: python benchmarks/bench_ragged_backends.py --tol 0.10)
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ from dataclasses import dataclass
14
+ from time import perf_counter
15
+ from typing import Any, Callable
16
+
17
+ import awkward as ak
18
+ import numpy as np
19
+
20
+ from seqpro.rag._core import Ragged as RustRagged
21
+
22
+
23
+ @dataclass
24
+ class Cell:
25
+ category: str
26
+ op: str
27
+ shape: str
28
+ awk: Callable[[], Any]
29
+ rust: Callable[[], Any]
30
+ eq: "Callable[[Any, Any], bool] | None" = None
31
+
32
+
33
+ def time_callable(
34
+ fn: Callable[[], Any], *, repeats: int = 7, min_batch_s: float = 0.005
35
+ ) -> float:
36
+ """Seconds per call: warm up, autoscale a batch past min_batch_s, take min of repeats."""
37
+ for _ in range(3):
38
+ fn()
39
+ iters = 1
40
+ while True:
41
+ t0 = perf_counter()
42
+ for _ in range(iters):
43
+ fn()
44
+ if perf_counter() - t0 >= min_batch_s:
45
+ break
46
+ iters *= 2
47
+ best = float("inf")
48
+ for _ in range(repeats):
49
+ t0 = perf_counter()
50
+ for _ in range(iters):
51
+ fn()
52
+ best = min(best, (perf_counter() - t0) / iters)
53
+ return best
54
+
55
+
56
+ def to_list(x: Any) -> Any:
57
+ """Canonicalize a result to a comparable structure for equivalence checks."""
58
+ if isinstance(x, RustRagged):
59
+ return to_list(x.to_ak())
60
+ if isinstance(x, ak.Array):
61
+ return ak.to_list(x)
62
+ if isinstance(x, dict):
63
+ return {k: to_list(v) for k, v in x.items()}
64
+ if isinstance(x, np.ndarray):
65
+ return x.tolist()
66
+ if isinstance(x, (np.integer, np.floating)):
67
+ return x.item()
68
+ return x
69
+
70
+
71
+ def default_eq(a: Any, b: Any) -> bool:
72
+ return to_list(a) == to_list(b)
73
+
74
+
75
+ def run_cells(cells: list[Cell], tol: float, repeats: int = 7) -> int:
76
+ rows: list[tuple[str, str, str, float, float, float, bool]] = []
77
+ failures = 0
78
+ for c in cells:
79
+ eq = c.eq or default_eq
80
+ if not eq(c.awk(), c.rust()):
81
+ raise AssertionError(
82
+ f"equivalence check failed for {c.category}/{c.op} ({c.shape}); "
83
+ "the comparison would be unfair — fix the cell before timing."
84
+ )
85
+ t_awk = time_callable(c.awk, repeats=repeats)
86
+ t_rust = time_callable(c.rust, repeats=repeats)
87
+ ratio = t_rust / t_awk if t_awk > 0 else float("inf")
88
+ ok = ratio <= 1.0 + tol
89
+ failures += not ok
90
+ rows.append((c.category, c.op, c.shape, t_awk, t_rust, ratio, ok))
91
+
92
+ w_cat = max(len("category"), *(len(r[0]) for r in rows))
93
+ w_op = max(len("op"), *(len(r[1]) for r in rows))
94
+ w_sh = max(len("shape"), *(len(r[2]) for r in rows))
95
+ header = (
96
+ f"{'category':<{w_cat}} {'op':<{w_op}} {'shape':<{w_sh}} "
97
+ f"{'awk (us)':>10} {'rust (us)':>10} {'rust/awk':>9} verdict"
98
+ )
99
+ print(header)
100
+ print("-" * len(header))
101
+ for cat, op, sh, ta, tr, ratio, ok in rows:
102
+ print(
103
+ f"{cat:<{w_cat}} {op:<{w_op}} {sh:<{w_sh}} "
104
+ f"{ta * 1e6:>10.2f} {tr * 1e6:>10.2f} {ratio:>9.3f} "
105
+ f"{'PASS' if ok else 'FAIL'}"
106
+ )
107
+ n = len(rows)
108
+ print("-" * len(header))
109
+ print(f"{n - failures}/{n} passed (tol={tol:.2%})")
110
+ return 1 if failures else 0
111
+
112
+
113
+ _BASES = np.frombuffer(b"ACGT", dtype="S1")
114
+
115
+
116
+ def _r1_buffers(n: int, low: int, high: int, *, bytes_: bool = False):
117
+ """Return (data, lengths, offsets) for n segments with lengths in [low, high]."""
118
+ rng = np.random.default_rng(0)
119
+ lengths = rng.integers(low, high + 1, size=n).astype(np.int64)
120
+ total = int(lengths.sum())
121
+ if bytes_:
122
+ data = _BASES[rng.integers(0, 4, size=total)] # (total,) S1
123
+ else:
124
+ data = np.arange(total, dtype=np.int64)
125
+ offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
126
+ return data, lengths, offsets
127
+
128
+
129
+ def _ak_r1(offsets, data):
130
+ arr = np.asarray(data)
131
+ if arr.dtype.kind == "S":
132
+ # awkward does not support S1 dtype; view as uint8 bytes with "byte" parameter
133
+ leaf = ak.contents.NumpyArray(
134
+ arr.view(np.uint8), parameters={"__array__": "byte"}
135
+ )
136
+ else:
137
+ leaf = ak.contents.NumpyArray(arr)
138
+ return ak.Array(
139
+ ak.contents.ListOffsetArray(
140
+ ak.index.Index64(np.asarray(offsets, np.int64)),
141
+ leaf,
142
+ )
143
+ )
144
+
145
+
146
+ def single_cells() -> list[Cell]:
147
+ cells: list[Cell] = []
148
+ # Primary numeric workload: flanked alleles (8000 x ~11-60).
149
+ data, lengths, offsets = _r1_buffers(8000, 11, 60)
150
+ shape = (8000, None)
151
+ sh = "8000x~11-60 i64"
152
+
153
+ cells.append(
154
+ Cell(
155
+ "single",
156
+ "construct",
157
+ sh,
158
+ lambda: _ak_r1(offsets, data),
159
+ lambda: RustRagged.from_offsets(data, shape, offsets),
160
+ )
161
+ )
162
+
163
+ akx = _ak_r1(offsets, data)
164
+ rx = RustRagged.from_offsets(data, shape, offsets)
165
+
166
+ cells.append(Cell("single", "index[int]", sh, lambda: akx[1234], lambda: rx[1234]))
167
+ cells.append(
168
+ Cell(
169
+ "single", "index[slice]", sh, lambda: akx[1000:5000], lambda: rx[1000:5000]
170
+ )
171
+ )
172
+
173
+ mask = np.arange(8000) % 3 == 0
174
+ cells.append(Cell("single", "index[mask]", sh, lambda: akx[mask], lambda: rx[mask]))
175
+
176
+ # Use UNPACKED input for to_packed so both backends do real gather work.
177
+ # On already-packed data: awkward to_packed is zero-copy (shares buffer) while
178
+ # rust to_packed(copy=True) does a defensive data.copy() — unequal work.
179
+ # On unpacked (masked) input both backends must gather scattered segments.
180
+ akx_unpacked = akx[mask]
181
+ rx_unpacked = rx[mask]
182
+ cells.append(
183
+ Cell(
184
+ "single",
185
+ "to_packed",
186
+ f"{sh} unpacked",
187
+ lambda: ak.to_packed(akx_unpacked),
188
+ lambda: rx_unpacked.to_packed(),
189
+ )
190
+ )
191
+
192
+ L = int(lengths.max())
193
+ cells.append(
194
+ Cell(
195
+ "single",
196
+ "to_padded",
197
+ f"{sh} L={L}",
198
+ lambda: ak.to_numpy(ak.fill_none(ak.pad_none(akx, L, clip=True), 0)),
199
+ lambda: rx.to_padded(0, length=L),
200
+ )
201
+ )
202
+
203
+ cells.append(Cell("single", "ufunc(+1)", sh, lambda: akx + 1, lambda: rx + 1))
204
+
205
+ # S1 byte workload: construct + to_packed (the kernel-relevant ops).
206
+ bdata, blen, boff = _r1_buffers(8000, 11, 60, bytes_=True)
207
+ bshape = (8000, None)
208
+ bsh = "8000x~11-60 S1"
209
+ akb = _ak_r1(boff, bdata)
210
+ rb = RustRagged.from_offsets(bdata, bshape, boff)
211
+ cells.append(
212
+ Cell(
213
+ "single",
214
+ "construct",
215
+ bsh,
216
+ lambda: _ak_r1(boff, bdata),
217
+ lambda: RustRagged.from_offsets(bdata, bshape, boff),
218
+ )
219
+ )
220
+ # Use UNPACKED input for to_packed so both backends do real gather work.
221
+ # On already-packed data: awkward to_packed is zero-copy (shares buffer) while
222
+ # rust to_packed(copy=True) does a defensive data.copy() — unequal work.
223
+ # On unpacked (masked) input both backends must gather scattered segments.
224
+ bmask = np.arange(8000) % 3 == 0
225
+ akb_unpacked = akb[bmask]
226
+ rb_unpacked = rb[bmask]
227
+ cells.append(
228
+ Cell(
229
+ "single",
230
+ "to_packed",
231
+ f"{bsh} unpacked",
232
+ lambda: ak.to_packed(akb_unpacked),
233
+ lambda: rb_unpacked.to_packed(),
234
+ )
235
+ )
236
+ return cells
237
+
238
+
239
+ def record_cells() -> list[Cell]:
240
+ cells: list[Cell] = []
241
+ # genoray-like: 400 segments (samples*ploidy), ~variants in [0, 200), 3 numeric fields.
242
+ rng = np.random.default_rng(0)
243
+ n = 400
244
+ lengths = rng.integers(0, 200, size=n).astype(np.int64)
245
+ total = int(lengths.sum())
246
+ offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
247
+ genos = rng.integers(0, 2, size=total).astype(np.int8)
248
+ dosages = rng.random(total).astype(np.float32)
249
+ mutcat = rng.integers(0, 5, size=total).astype(np.int32)
250
+ shape = (n, None)
251
+ sh = "400x~0-200 3fld"
252
+
253
+ fields = {"genos": genos, "dosages": dosages, "mutcat": mutcat}
254
+
255
+ def _ak_rec():
256
+ return ak.zip({k: _ak_r1(offsets, v) for k, v in fields.items()}, depth_limit=1)
257
+
258
+ def _rust_rec():
259
+ return RustRagged.from_fields(
260
+ {k: RustRagged.from_offsets(v, shape, offsets) for k, v in fields.items()}
261
+ )
262
+
263
+ cells.append(Cell("records", "zip/from_fields", sh, _ak_rec, _rust_rec))
264
+
265
+ akr = _ak_rec()
266
+ rr = _rust_rec()
267
+ cells.append(
268
+ Cell("records", "field[a]", sh, lambda: akr["dosages"], lambda: rr["dosages"])
269
+ )
270
+ cells.append(
271
+ Cell(
272
+ "records",
273
+ "to_packed",
274
+ sh,
275
+ lambda: ak.to_packed(akr),
276
+ lambda: rr.to_packed(),
277
+ )
278
+ )
279
+
280
+ # Per-field dense: awkward pads each field; rust returns a dict.
281
+ L = int(lengths.max())
282
+
283
+ def _ak_padded_dict():
284
+ return {
285
+ k: ak.to_numpy(ak.fill_none(ak.pad_none(akr[k], L, clip=True), 0))
286
+ for k in fields
287
+ }
288
+
289
+ cells.append(
290
+ Cell(
291
+ "records",
292
+ "to_padded(dict)",
293
+ f"{sh} L={L}",
294
+ _ak_padded_dict,
295
+ lambda: rr.to_padded(0, length=L),
296
+ )
297
+ )
298
+ return cells
299
+
300
+
301
+ def _r2_buffers(n_outer: int, mid_low: int, mid_high: int, in_low: int, in_high: int):
302
+ """Return (data, o0, o1, outer_counts, inner_lengths) for an R=2 structure."""
303
+ rng = np.random.default_rng(0)
304
+ outer_counts = rng.integers(mid_low, mid_high + 1, size=n_outer).astype(np.int64)
305
+ n_mid = int(outer_counts.sum())
306
+ inner_lengths = rng.integers(in_low, in_high + 1, size=n_mid).astype(np.int64)
307
+ total = int(inner_lengths.sum())
308
+ data = np.arange(total, dtype=np.int64)
309
+ o0 = np.concatenate([[0], np.cumsum(outer_counts)]).astype(np.int64)
310
+ o1 = np.concatenate([[0], np.cumsum(inner_lengths)]).astype(np.int64)
311
+ return data, o0, o1, outer_counts, inner_lengths
312
+
313
+
314
+ def _ak_r2(o0, o1, data):
315
+ return ak.Array(
316
+ ak.contents.ListOffsetArray(
317
+ ak.index.Index64(np.asarray(o0, np.int64)),
318
+ ak.contents.ListOffsetArray(
319
+ ak.index.Index64(np.asarray(o1, np.int64)),
320
+ ak.contents.NumpyArray(np.asarray(data)),
321
+ ),
322
+ )
323
+ )
324
+
325
+
326
+ def nested_cells() -> list[Cell]:
327
+ cells: list[Cell] = []
328
+ # flat-variant-windows-like: 64 outer groups, ~variants in [1,30], ~window in [1,20].
329
+ data, o0, o1, outer_counts, inner_lengths = _r2_buffers(64, 1, 30, 1, 20)
330
+ shape = (64, None, None)
331
+ sh = "64x~1-30x~1-20 i64"
332
+
333
+ cells.append(
334
+ Cell(
335
+ "nested",
336
+ "construct",
337
+ sh,
338
+ lambda: _ak_r2(o0, o1, data),
339
+ lambda: RustRagged.from_offsets(data, shape, [o0, o1]),
340
+ )
341
+ )
342
+
343
+ akx = _ak_r2(o0, o1, data)
344
+ rx = RustRagged.from_offsets(data, shape, [o0, o1])
345
+
346
+ cells.append(Cell("nested", "index[int]", sh, lambda: akx[7], lambda: rx[7]))
347
+ cells.append(
348
+ Cell("nested", "index[slice]", sh, lambda: akx[8:40], lambda: rx[8:40])
349
+ )
350
+ # rag[:, k]: k-th middle of each outer group. Use k=0 so every group is non-empty
351
+ # (outer_counts >= 1 by construction).
352
+ cells.append(Cell("nested", "index[:,k]", sh, lambda: akx[:, 0], lambda: rx[:, 0]))
353
+ cells.append(
354
+ Cell("nested", "index[:,a:b]", sh, lambda: akx[:, 0:2], lambda: rx[:, 0:2])
355
+ )
356
+
357
+ # rag[:, mask]: rust takes a flat boolean over the global middle axis; awkward takes
358
+ # the same mask regrouped under the outer counts (a ragged boolean of matching shape).
359
+ n_mid = int(outer_counts.sum())
360
+ flat_mask = np.arange(n_mid) % 2 == 0
361
+ ak_mask = ak.unflatten(flat_mask, list(outer_counts))
362
+ cells.append(
363
+ Cell(
364
+ "nested",
365
+ "index[:,mask]",
366
+ sh,
367
+ lambda: akx[ak_mask],
368
+ lambda: rx[:, flat_mask],
369
+ )
370
+ )
371
+
372
+ cells.append(
373
+ Cell(
374
+ "nested", "to_packed", sh, lambda: ak.to_packed(akx), lambda: rx.to_packed()
375
+ )
376
+ )
377
+
378
+ # Dense both axes: M = max middles per group, K = max inner length.
379
+ M = int(outer_counts.max())
380
+ K = int(inner_lengths.max()) if inner_lengths.size else 0
381
+
382
+ def _ak_dense():
383
+ padded_inner = ak.fill_none(ak.pad_none(akx, K, axis=2, clip=True), 0)
384
+ padded_outer = ak.pad_none(padded_inner, M, axis=1, clip=True)
385
+ # fill_none with a length-K array fills axis=1 None slots; to_numpy returns a masked
386
+ # array because awkward retains the option type — call .filled(0) to materialise zeros.
387
+ fill_val = np.zeros(K, dtype=np.int64)
388
+ return ak.to_numpy(
389
+ ak.fill_none(padded_outer, fill_val), allow_missing=True
390
+ ).filled(0)
391
+
392
+ cells.append(
393
+ Cell(
394
+ "nested",
395
+ "to_padded(both)",
396
+ f"{sh} M={M},K={K}",
397
+ _ak_dense,
398
+ lambda: rx.to_padded(0, axis=None),
399
+ )
400
+ )
401
+ return cells
402
+
403
+
404
+ def string_cells() -> list[Cell]:
405
+ cells: list[Cell] = []
406
+ # 8000 short opaque strings (alleles), lengths in [1, 8].
407
+ rng = np.random.default_rng(0)
408
+ n = 8000
409
+ lengths = rng.integers(1, 9, size=n).astype(np.int64)
410
+ total = int(lengths.sum())
411
+ data = _BASES[rng.integers(0, 4, size=total)] # (total,) S1
412
+ offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
413
+
414
+ # rust: opaque-string Ragged (shape (n,), dtype 'S') -> chars (n, ~length) and back.
415
+ r_str = RustRagged.from_lengths(data, lengths) # opaque string by default
416
+ r_chars = r_str.to_chars()
417
+
418
+ # awkward analogue: a list-of-S1 (chars) array; "to chars" = view as char list,
419
+ # "to strings" = join back to bytestrings. Build the char-list oracle once.
420
+ ak_chars = _ak_r1(offsets, data) # ListOffsetArray over S1 == char lists
421
+
422
+ # to_chars: rust retag vs awkward producing the char-list view.
423
+ cells.append(
424
+ Cell(
425
+ "string",
426
+ "to_chars",
427
+ f"{n}x~1-8",
428
+ lambda: ak.copy(ak_chars),
429
+ lambda: r_str.to_chars(),
430
+ eq=lambda a, b: True,
431
+ )
432
+ ) # structural shapes differ across backends; time-only
433
+ # to_strings: rust retag vs awkward joining char lists into bytestrings.
434
+ cells.append(
435
+ Cell(
436
+ "string",
437
+ "to_strings",
438
+ f"{n}x~1-8",
439
+ lambda: ak.copy(ak_chars),
440
+ lambda: r_chars.to_strings(),
441
+ eq=lambda a, b: True,
442
+ )
443
+ )
444
+ return cells
445
+
446
+
447
+ def build_cells(args: argparse.Namespace) -> list[Cell]:
448
+ """Assemble the cell list. Extended in later tasks."""
449
+ cells: list[Cell] = []
450
+ if args.only in ("all", "single"):
451
+ cells += single_cells()
452
+ if args.only in ("all", "records"):
453
+ cells += record_cells()
454
+ if args.only in ("all", "nested"):
455
+ cells += nested_cells()
456
+ if args.only in ("all", "string"):
457
+ cells += string_cells()
458
+ return cells
459
+
460
+
461
+ def main(argv: "list[str] | None" = None) -> int:
462
+ p = argparse.ArgumentParser(
463
+ description="rust-native vs awkward Ragged throughput gate"
464
+ )
465
+ p.add_argument("--tol", type=float, default=0.10)
466
+ p.add_argument(
467
+ "--only",
468
+ choices=["single", "records", "nested", "string", "all"],
469
+ default="all",
470
+ )
471
+ p.add_argument("--repeats", type=int, default=7)
472
+ args = p.parse_args(argv)
473
+ cells = build_cells(args)
474
+ if not cells:
475
+ print(f"no cells for --only {args.only}")
476
+ return 0
477
+ return run_cells(cells, args.tol, args.repeats)
478
+
479
+
480
+ if __name__ == "__main__":
481
+ raise SystemExit(main())