seqpro 0.16.0__tar.gz → 0.17.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/bench.yaml +3 -3
  2. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/lint.yaml +2 -2
  3. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/test.yaml +2 -2
  4. {seqpro-0.16.0 → seqpro-0.17.0}/.pre-commit-config.yaml +18 -0
  5. {seqpro-0.16.0 → seqpro-0.17.0}/CHANGELOG.md +98 -0
  6. {seqpro-0.16.0 → seqpro-0.17.0}/PKG-INFO +1 -1
  7. seqpro-0.17.0/benches/bench_tokenize_translate.py +30 -0
  8. seqpro-0.17.0/benchmarks/bench_ragged_backends.py +481 -0
  9. seqpro-0.17.0/docs/roadmap/rust-ragged.md +388 -0
  10. seqpro-0.17.0/docs/superpowers/plans/2026-06-18-rust-tokenize-translate.md +1391 -0
  11. seqpro-0.17.0/docs/superpowers/plans/2026-06-19-rust-ragged-core.md +1347 -0
  12. seqpro-0.17.0/docs/superpowers/plans/2026-06-20-rust-ragged-records.md +1583 -0
  13. seqpro-0.17.0/docs/superpowers/plans/2026-06-20-rust-ragged-spec-c.md +1556 -0
  14. seqpro-0.17.0/docs/superpowers/plans/2026-06-21-ragged-throughput-gate.md +656 -0
  15. seqpro-0.17.0/docs/superpowers/plans/2026-06-21-rust-ragged-consumer-audit.md +492 -0
  16. seqpro-0.17.0/docs/superpowers/specs/2026-06-18-rust-tokenize-translate-design.md +360 -0
  17. seqpro-0.17.0/docs/superpowers/specs/2026-06-19-rust-ragged-core-design.md +260 -0
  18. seqpro-0.17.0/docs/superpowers/specs/2026-06-20-rust-ragged-nested-design.md +364 -0
  19. seqpro-0.17.0/docs/superpowers/specs/2026-06-20-rust-ragged-records-design.md +327 -0
  20. seqpro-0.17.0/docs/superpowers/specs/2026-06-21-ragged-throughput-gate-design.md +166 -0
  21. seqpro-0.17.0/docs/superpowers/specs/2026-06-21-rust-ragged-audit-ledger.md +75 -0
  22. seqpro-0.17.0/docs/superpowers/specs/2026-06-21-rust-ragged-consumer-audit-design.md +121 -0
  23. {seqpro-0.16.0 → seqpro-0.17.0}/pixi.lock +38 -16
  24. {seqpro-0.16.0 → seqpro-0.17.0}/pixi.toml +6 -1
  25. {seqpro-0.16.0 → seqpro-0.17.0}/pyproject.toml +7 -1
  26. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/alphabets/_alphabets.py +137 -92
  27. seqpro-0.17.0/python/seqpro/rag/__init__.py +29 -0
  28. seqpro-0.17.0/python/seqpro/rag/_ak_interop.py +208 -0
  29. seqpro-0.17.0/python/seqpro/rag/_core.py +1559 -0
  30. seqpro-0.17.0/python/seqpro/rag/_ingest.py +252 -0
  31. seqpro-0.17.0/python/seqpro/rag/_layout.py +172 -0
  32. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/rag/_ops.py +161 -49
  33. {seqpro-0.16.0 → seqpro-0.17.0}/skills/seqpro/SKILL.md +35 -24
  34. {seqpro-0.16.0 → seqpro-0.17.0}/src/kshuffle.rs +132 -65
  35. seqpro-0.17.0/src/lib.rs +212 -0
  36. seqpro-0.17.0/src/ragged.rs +612 -0
  37. seqpro-0.17.0/src/translate.rs +510 -0
  38. seqpro-0.17.0/tests/test_concatenate.py +29 -0
  39. seqpro-0.17.0/tests/test_core_ragged_surface.py +36 -0
  40. seqpro-0.17.0/tests/test_ingest.py +19 -0
  41. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_rag_to_packed.py +93 -28
  42. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_ragged.py +57 -34
  43. seqpro-0.17.0/tests/test_ragged_core.py +810 -0
  44. seqpro-0.17.0/tests/test_ragged_core_records.py +682 -0
  45. seqpro-0.17.0/tests/test_ragged_nested_consumers.py +68 -0
  46. seqpro-0.17.0/tests/test_ragged_nested_diff.py +297 -0
  47. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_ragged_rc.py +20 -26
  48. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_ragged_to_padded.py +66 -6
  49. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_translate.py +48 -19
  50. seqpro-0.17.0/tests/test_translate_rust.py +135 -0
  51. seqpro-0.16.0/python/seqpro/rag/__init__.py +0 -15
  52. seqpro-0.16.0/python/seqpro/rag/_array.py +0 -904
  53. seqpro-0.16.0/python/seqpro/rag/_gufuncs.py +0 -14
  54. seqpro-0.16.0/src/lib.rs +0 -44
  55. {seqpro-0.16.0 → seqpro-0.17.0}/.claude/skills/zensical/SKILL.md +0 -0
  56. {seqpro-0.16.0 → seqpro-0.17.0}/.gitattributes +0 -0
  57. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/bump.yaml +0 -0
  58. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/docs.yml +0 -0
  59. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/merge.yaml +0 -0
  60. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/publish.yaml +0 -0
  61. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/release-pipeline.yaml +0 -0
  62. {seqpro-0.16.0 → seqpro-0.17.0}/.github/workflows/release.yaml +0 -0
  63. {seqpro-0.16.0 → seqpro-0.17.0}/.gitignore +0 -0
  64. {seqpro-0.16.0 → seqpro-0.17.0}/CLAUDE.md +0 -0
  65. {seqpro-0.16.0 → seqpro-0.17.0}/Cargo.lock +0 -0
  66. {seqpro-0.16.0 → seqpro-0.17.0}/Cargo.toml +0 -0
  67. {seqpro-0.16.0 → seqpro-0.17.0}/LICENSE +0 -0
  68. {seqpro-0.16.0 → seqpro-0.17.0}/README.md +0 -0
  69. {seqpro-0.16.0 → seqpro-0.17.0}/benches/kshuffle.rs +0 -0
  70. {seqpro-0.16.0 → seqpro-0.17.0}/benchmarks/bench_to_packed.py +0 -0
  71. {seqpro-0.16.0 → seqpro-0.17.0}/docs/api/alphabets.md +0 -0
  72. {seqpro-0.16.0 → seqpro-0.17.0}/docs/api/bed.md +0 -0
  73. {seqpro-0.16.0 → seqpro-0.17.0}/docs/api/gtf.md +0 -0
  74. {seqpro-0.16.0 → seqpro-0.17.0}/docs/api/index.md +0 -0
  75. {seqpro-0.16.0 → seqpro-0.17.0}/docs/api/ragged.md +0 -0
  76. {seqpro-0.16.0 → seqpro-0.17.0}/docs/api/types.md +0 -0
  77. {seqpro-0.16.0 → seqpro-0.17.0}/docs/index.md +0 -0
  78. {seqpro-0.16.0 → seqpro-0.17.0}/docs/ragged.md +0 -0
  79. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-04-ragged-record-array.md +0 -0
  80. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-05-documentation-site.md +0 -0
  81. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-05-narwhals-coord-schema.md +0 -0
  82. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-05-ragged-zip-and-record-introspection.md +0 -0
  83. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-20-kshuffle-optimization.md +0 -0
  84. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-20-kshuffle-pooled-buffers-and-k2-fast-path.md +0 -0
  85. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-20-kshuffle-wilson-single-pass.md +0 -0
  86. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-20-release-pipeline.md +0 -0
  87. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-28-translate-lut-validation.md +0 -0
  88. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-31-flat-buffer-to-padded.md +0 -0
  89. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-05-31-rag-to-packed.md +0 -0
  90. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-06-05-translate-unknown-codon-policy.md +0 -0
  91. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/plans/2026-06-12-tokenize-lut-codspeed.md +0 -0
  92. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-04-ragged-record-array-design.md +0 -0
  93. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-05-documentation-site-design.md +0 -0
  94. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-05-narwhals-coord-schema-design.md +0 -0
  95. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-05-ragged-zip-and-record-introspection-design.md +0 -0
  96. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-20-kshuffle-optimization-design.md +0 -0
  97. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-20-kshuffle-pooled-buffers-and-k2-fast-path-design.md +0 -0
  98. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-20-kshuffle-wilson-single-pass-design.md +0 -0
  99. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-20-release-pipeline-design.md +0 -0
  100. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-28-translate-lut-and-validation-design.md +0 -0
  101. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-31-flat-buffer-to-padded-design.md +0 -0
  102. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-05-31-rag-to-packed-design.md +0 -0
  103. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-06-05-translate-unknown-codon-policy-design.md +0 -0
  104. {seqpro-0.16.0 → seqpro-0.17.0}/docs/superpowers/specs/2026-06-12-tokenize-lut-codspeed-design.md +0 -0
  105. {seqpro-0.16.0 → seqpro-0.17.0}/meta.yaml +0 -0
  106. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/__init__.py +0 -0
  107. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/_analyzers.py +0 -0
  108. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/_cleaners.py +0 -0
  109. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/_coords.py +0 -0
  110. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/_encoders.py +0 -0
  111. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/_modifiers.py +0 -0
  112. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/_numba.py +0 -0
  113. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/_types.py +0 -0
  114. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/_utils.py +0 -0
  115. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/alphabets/__init__.py +0 -0
  116. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/bed.py +0 -0
  117. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/experimental/_experimental.py +0 -0
  118. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/experimental/_visualizers.py +0 -0
  119. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/gtf.py +0 -0
  120. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/py.typed +0 -0
  121. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/rag/_types.py +0 -0
  122. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/rag/_utils.py +0 -0
  123. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/transforms/__init__.py +0 -0
  124. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/transforms/augmentation.py +0 -0
  125. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/transforms/tmm.py +0 -0
  126. {seqpro-0.16.0 → seqpro-0.17.0}/python/seqpro/xr/__init__.py +0 -0
  127. {seqpro-0.16.0 → seqpro-0.17.0}/scratch_bench_rc.py +0 -0
  128. {seqpro-0.16.0 → seqpro-0.17.0}/scratch_bench_to_padded.py +0 -0
  129. {seqpro-0.16.0 → seqpro-0.17.0}/src/kmer_encode.rs +0 -0
  130. {seqpro-0.16.0 → seqpro-0.17.0}/src/kshuffle_ref.rs +0 -0
  131. {seqpro-0.16.0 → seqpro-0.17.0}/tests/_shape_fixtures.py +0 -0
  132. {seqpro-0.16.0 → seqpro-0.17.0}/tests/bed/test_pyranges.py +0 -0
  133. {seqpro-0.16.0 → seqpro-0.17.0}/tests/bed/test_read.py +0 -0
  134. {seqpro-0.16.0 → seqpro-0.17.0}/tests/bed/test_sort.py +0 -0
  135. {seqpro-0.16.0 → seqpro-0.17.0}/tests/bed/test_with_length.py +0 -0
  136. {seqpro-0.16.0 → seqpro-0.17.0}/tests/bench_translate_lut.py +0 -0
  137. {seqpro-0.16.0 → seqpro-0.17.0}/tests/conftest.py +0 -0
  138. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_analyzers.py +0 -0
  139. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_bench_tokenize.py +0 -0
  140. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_coords.py +0 -0
  141. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_encoders.py +0 -0
  142. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_modifiers.py +0 -0
  143. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_ohe.py +0 -0
  144. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_shape_matrix.py +0 -0
  145. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_tokenize.py +0 -0
  146. {seqpro-0.16.0 → seqpro-0.17.0}/tests/test_transforms.py +0 -0
  147. {seqpro-0.16.0 → seqpro-0.17.0}/zensical.toml +0 -0
@@ -25,12 +25,12 @@ jobs:
25
25
  - name: Setup pixi
26
26
  uses: prefix-dev/setup-pixi@v0.9.5
27
27
  with:
28
- pixi-version: v0.67.2
28
+ pixi-version: v0.70.1
29
29
  cache: true
30
30
  environments: bench
31
- locked: false
31
+ locked: true
32
32
  - name: Build extension
33
- run: pixi run -e bench maturin develop
33
+ run: pixi run -e bench maturin develop --uv
34
34
  - name: Run benchmarks
35
35
  uses: CodSpeedHQ/action@v4
36
36
  with:
@@ -11,8 +11,8 @@ jobs:
11
11
  - name: Setup pixi
12
12
  uses: prefix-dev/setup-pixi@v0.9.5
13
13
  with:
14
- pixi-version: v0.67.2
14
+ pixi-version: v0.70.1
15
15
  cache: true
16
16
  environments: default
17
- locked: false
17
+ locked: true
18
18
  - uses: j178/prek-action@v2
@@ -21,9 +21,9 @@ jobs:
21
21
  - name: Setup pixi
22
22
  uses: prefix-dev/setup-pixi@v0.9.5
23
23
  with:
24
- pixi-version: v0.67.2
24
+ pixi-version: v0.70.1
25
25
  cache: true
26
26
  environments: ${{ matrix.environment }}
27
- locked: false
27
+ locked: true
28
28
  - name: Test
29
29
  run: pixi run -e ${{ matrix.environment }} test
@@ -33,3 +33,21 @@ repos:
33
33
  language: system
34
34
  stages: [pre-push]
35
35
  pass_filenames: false
36
+ - id: cargo-check
37
+ name: Type-check Rust with cargo check
38
+ entry: cargo check --all-targets --all-features
39
+ language: system
40
+ types: [rust]
41
+ pass_filenames: false
42
+ - id: cargo-fmt
43
+ name: Format Rust with cargo fmt
44
+ entry: cargo fmt --all
45
+ language: system
46
+ types: [rust]
47
+ pass_filenames: false
48
+ - id: cargo-clippy
49
+ name: Lint Rust with cargo clippy
50
+ entry: cargo clippy --all-targets -- -D warnings
51
+ language: system
52
+ types: [rust]
53
+ pass_filenames: false
@@ -1,3 +1,101 @@
1
+ ## 0.17.0 (2026-06-23)
2
+
3
+ ### Feat
4
+
5
+ - **rag**: concatenate() along ragged axis (Rust kernel)
6
+ - **rag**: to_packed() on record and opaque-string-under-axis Ragged
7
+ - wire rag-gate pixi task; record throughput gate outcome
8
+ - string-conversion op cells for Ragged throughput gate
9
+ - nested R=2 op cells for Ragged throughput gate
10
+ - record (SoA) op cells for Ragged throughput gate
11
+ - single-level op cells for Ragged throughput gate
12
+ - harness core for rust-vs-awkward Ragged throughput gate
13
+ - _ingest bridge for R=2 + string-under-axis (oracle interop)
14
+ - nested record indexing + record-aware to_packed/to_padded/to_numpy
15
+ - nested + string-under-axis record fields sharing full offsets list
16
+ - R=2 lengths/squeeze/reshape
17
+ - per-axis nested to_padded + rectangular to_numpy (R=2); trailing-dim support in to_padded
18
+ - nested to_packed via nested_pack kernel
19
+ - nested_pack Rust kernel + binding for two-level pack
20
+ - per-group inner mask/int-array indexing via nested_gather
21
+ - nested_gather Rust kernel + binding for per-group middle selection
22
+ - per-group inner int/slice indexing (rag[:, k], rag[:, a:b])
23
+ - R=2 tuple indexing + leaf access via peel chaining
24
+ - R=2 outer-row indexing (lazy gather, peel to 1-level)
25
+ - string-under-axis leaf + nested to_chars/to_strings
26
+ - nested constructors from_offsets(list)/from_lengths(tuple) for R=2
27
+ - validate R=2 nested ragged layouts; cap at R<=2
28
+ - ingest/emit record layouts via awkward bridge (oracle interop)
29
+ - record to_numpy/to_padded dicts; raise view/ufunc on records
30
+ - record-aware to_packed (one shared packed offsets across fields)
31
+ - per-field squeeze/reshape on record Ragged
32
+ - record row-axis indexing (slice/mask -> record, int -> dict)
33
+ - record field access (key/attr) and __setitem__ mutation
34
+ - record-branch properties (data/dtype/offsets/shape/fields/state)
35
+ - add Ragged.from_fields record constructor and rag.zip alias
36
+ - add RecordLayout value object and validation arm
37
+ - add zero-copy to_chars/to_strings between opaque-string and char Ragged
38
+ - disambiguate opaque-string vs char layout by presence of None in shape
39
+ - report np.dtype('S') for opaque-string Ragged (string/char duality)
40
+ - **rust**: ragged_validate and ragged_select kernels
41
+ - **rag**: awkward ingestion and to_ak shim
42
+ - **rag**: to_numpy, to_packed, to_padded
43
+ - **rag**: squeeze and reshape on regular dims
44
+ - **rag**: element-wise ufunc interop
45
+ - **rag**: __getitem__ indexing and slicing
46
+ - **rag**: state predicates and view
47
+ - **rag**: Ragged constructors and core properties
48
+ - **rag**: RaggedLayout value object + validation
49
+ - **translate**: self-contained Rust OHE<->AA path with native drop
50
+ - **translate**: route truncate_stop through Rust
51
+ - **translate**: route unknown=drop compaction through Rust
52
+ - **translate**: route ragged bytes pad path through Rust
53
+ - **translate**: route dense pad path through Rust
54
+ - **rust**: add codon-stride translate kernels
55
+ - **tokenize**: route ragged path through Rust, drop Numba gather
56
+ - **tokenize**: route dense path through Rust _tokenize
57
+ - **rust**: add tokenize LUT gather kernel
58
+
59
+ ### Fix
60
+
61
+ - **rag**: to_ak() on multi-leading-axis record Ragged
62
+ - **rag**: _core.Ragged/tokenize fixes found via genvarformer audit
63
+ - **rag**: add __len__, np.newaxis support, and element-wise fancy indexing to _core.Ragged
64
+ - **rag**: _ops.to_padded record detection works for both _array and _core backends
65
+ - **rag**: _core.to_numpy returns dict for records (restore designed contract); update _array-era test
66
+ - **rag**: correct _core getitem tuple routing for rag_dim==1 found via SeqPro own-suite audit
67
+ - **rag**: port _ops + _core to_packed/is_rag_dtype/to_numpy to _core object model
68
+ - **rag**: precise _core getitem routing — _core contract + genoray parity both green
69
+ - **rag**: repair _core.Ragged getitem regressions (string leaf, to_strings, r2 int array)
70
+ - **rag**: match _array is_base semantics for non-ndarray base
71
+ - **rag**: _core.Ragged fixes found via genoray audit
72
+ - **bench**: time to_packed on unpacked input so single-level pack compares equal work
73
+ - forward --repeats CLI flag into time_callable
74
+ - R=2 is_contiguous checks all offset levels; guard string-under-axis to_packed/to_numpy
75
+ - panic-safety in nested_pack (checked_mul, elem>0, o0-range guard)
76
+ - reject negative inner-slice bounds in rag[:, a:b]
77
+ - -O-safe record guards on is_string/to_chars/to_strings; note latent S4 over-acceptance
78
+ - raise TypeError on np.asarray of a record Ragged (-O-safe, was a bare assert)
79
+ - **rag**: reject mismatched boolean masks and bounds-check ragged_select
80
+ - **translate**: handle empty dense input in unknown=drop path
81
+
82
+ ### Refactor
83
+
84
+ - **rag**: delete awkward _array backend; relocate interop to _ak_interop
85
+ - resolve clippy findings for -D warnings
86
+ - **rag**: route index/validate hot paths through Rust
87
+ - **rag**: use NDArrayOperatorsMixin for operator dunders
88
+ - **tokenize**: restore np.take/Numba path; Rust port regressed vs baseline
89
+
90
+ ### Perf
91
+
92
+ - validate pack output size up-front; test parallel pack path
93
+ - Rust single-level pack kernel for to_packed (parallel gather)
94
+ - opt-in Ragged validation (default off) + slice indexing fast-path
95
+ - drop redundant per-field copy in record to_packed; align record is_base with single-level one-indirection rule
96
+ - **translate**: optimize LUT gather kernel (validity table + coarse chunking)
97
+ - **translate**: add rayon parallel path for large inputs
98
+
1
99
  ## 0.16.0 (2026-06-14)
2
100
 
3
101
  ### Feat
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: seqpro
3
- Version: 0.16.0
3
+ Version: 0.17.0
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Programming Language :: Python :: Implementation :: CPython
6
6
  Classifier: Programming Language :: Python :: Implementation :: PyPy
@@ -0,0 +1,30 @@
1
+ """End-to-end Python-level benchmarks for the Rust tokenize/translate port.
2
+
3
+ Run: pixi run -e dev pytest benches/bench_tokenize_translate.py --benchmark-only
4
+
5
+ Measures the public API (including PyO3 marshalling), which Rust-only criterion
6
+ misses and which decides the small-array regime. To compare against the Numba
7
+ baseline, check out the pre-port commit and run the same file.
8
+ """
9
+
10
+ import numpy as np
11
+ import pytest
12
+
13
+ import seqpro as sp
14
+
15
+ TOKEN_MAP = {"A": 0, "C": 1, "G": 2, "T": 3}
16
+ SIZES = [100, 10_000, 1_000_000]
17
+
18
+
19
+ @pytest.mark.parametrize("n", SIZES)
20
+ def test_bench_tokenize_dense(benchmark, n):
21
+ rng = np.random.default_rng(0)
22
+ seqs = rng.choice(np.frombuffer(b"ACGT", "S1"), size=n)
23
+ benchmark(lambda: sp.tokenize(seqs, TOKEN_MAP, 7))
24
+
25
+
26
+ @pytest.mark.parametrize("n_codons", [33, 3_333, 333_333])
27
+ def test_bench_translate_dense(benchmark, n_codons):
28
+ rng = np.random.default_rng(1)
29
+ seqs = rng.choice(np.frombuffer(b"ACGT", "S1"), size=n_codons * 3)
30
+ benchmark(lambda: sp.AA.translate(seqs, length_axis=0))
@@ -0,0 +1,481 @@
1
+ """Local, transitional A-vs-B throughput gate: rust-native Ragged vs awkward.
2
+
3
+ Proves seqpro.rag._core.Ragged is at least as fast as awkward's native layout
4
+ algebra before the Spec D cutover drops awkward. NOT a permanent CI fixture —
5
+ deleted at cutover (see docs/superpowers/specs/2026-06-21-ragged-throughput-gate-design.md).
6
+
7
+ Run: pixi run -e bench rag-gate (or: python benchmarks/bench_ragged_backends.py --tol 0.10)
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ from dataclasses import dataclass
14
+ from time import perf_counter
15
+ from typing import Any, Callable
16
+
17
+ import awkward as ak
18
+ import numpy as np
19
+
20
+ from seqpro.rag._core import Ragged as RustRagged
21
+
22
+
23
+ @dataclass
24
+ class Cell:
25
+ category: str
26
+ op: str
27
+ shape: str
28
+ awk: Callable[[], Any]
29
+ rust: Callable[[], Any]
30
+ eq: "Callable[[Any, Any], bool] | None" = None
31
+
32
+
33
+ def time_callable(
34
+ fn: Callable[[], Any], *, repeats: int = 7, min_batch_s: float = 0.005
35
+ ) -> float:
36
+ """Seconds per call: warm up, autoscale a batch past min_batch_s, take min of repeats."""
37
+ for _ in range(3):
38
+ fn()
39
+ iters = 1
40
+ while True:
41
+ t0 = perf_counter()
42
+ for _ in range(iters):
43
+ fn()
44
+ if perf_counter() - t0 >= min_batch_s:
45
+ break
46
+ iters *= 2
47
+ best = float("inf")
48
+ for _ in range(repeats):
49
+ t0 = perf_counter()
50
+ for _ in range(iters):
51
+ fn()
52
+ best = min(best, (perf_counter() - t0) / iters)
53
+ return best
54
+
55
+
56
+ def to_list(x: Any) -> Any:
57
+ """Canonicalize a result to a comparable structure for equivalence checks."""
58
+ if isinstance(x, RustRagged):
59
+ return to_list(x.to_ak())
60
+ if isinstance(x, ak.Array):
61
+ return ak.to_list(x)
62
+ if isinstance(x, dict):
63
+ return {k: to_list(v) for k, v in x.items()}
64
+ if isinstance(x, np.ndarray):
65
+ return x.tolist()
66
+ if isinstance(x, (np.integer, np.floating)):
67
+ return x.item()
68
+ return x
69
+
70
+
71
+ def default_eq(a: Any, b: Any) -> bool:
72
+ return to_list(a) == to_list(b)
73
+
74
+
75
+ def run_cells(cells: list[Cell], tol: float, repeats: int = 7) -> int:
76
+ rows: list[tuple[str, str, str, float, float, float, bool]] = []
77
+ failures = 0
78
+ for c in cells:
79
+ eq = c.eq or default_eq
80
+ if not eq(c.awk(), c.rust()):
81
+ raise AssertionError(
82
+ f"equivalence check failed for {c.category}/{c.op} ({c.shape}); "
83
+ "the comparison would be unfair — fix the cell before timing."
84
+ )
85
+ t_awk = time_callable(c.awk, repeats=repeats)
86
+ t_rust = time_callable(c.rust, repeats=repeats)
87
+ ratio = t_rust / t_awk if t_awk > 0 else float("inf")
88
+ ok = ratio <= 1.0 + tol
89
+ failures += not ok
90
+ rows.append((c.category, c.op, c.shape, t_awk, t_rust, ratio, ok))
91
+
92
+ w_cat = max(len("category"), *(len(r[0]) for r in rows))
93
+ w_op = max(len("op"), *(len(r[1]) for r in rows))
94
+ w_sh = max(len("shape"), *(len(r[2]) for r in rows))
95
+ header = (
96
+ f"{'category':<{w_cat}} {'op':<{w_op}} {'shape':<{w_sh}} "
97
+ f"{'awk (us)':>10} {'rust (us)':>10} {'rust/awk':>9} verdict"
98
+ )
99
+ print(header)
100
+ print("-" * len(header))
101
+ for cat, op, sh, ta, tr, ratio, ok in rows:
102
+ print(
103
+ f"{cat:<{w_cat}} {op:<{w_op}} {sh:<{w_sh}} "
104
+ f"{ta * 1e6:>10.2f} {tr * 1e6:>10.2f} {ratio:>9.3f} "
105
+ f"{'PASS' if ok else 'FAIL'}"
106
+ )
107
+ n = len(rows)
108
+ print("-" * len(header))
109
+ print(f"{n - failures}/{n} passed (tol={tol:.2%})")
110
+ return 1 if failures else 0
111
+
112
+
113
+ _BASES = np.frombuffer(b"ACGT", dtype="S1")
114
+
115
+
116
+ def _r1_buffers(n: int, low: int, high: int, *, bytes_: bool = False):
117
+ """Return (data, lengths, offsets) for n segments with lengths in [low, high]."""
118
+ rng = np.random.default_rng(0)
119
+ lengths = rng.integers(low, high + 1, size=n).astype(np.int64)
120
+ total = int(lengths.sum())
121
+ if bytes_:
122
+ data = _BASES[rng.integers(0, 4, size=total)] # (total,) S1
123
+ else:
124
+ data = np.arange(total, dtype=np.int64)
125
+ offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
126
+ return data, lengths, offsets
127
+
128
+
129
+ def _ak_r1(offsets, data):
130
+ arr = np.asarray(data)
131
+ if arr.dtype.kind == "S":
132
+ # awkward does not support S1 dtype; view as uint8 bytes with "byte" parameter
133
+ leaf = ak.contents.NumpyArray(
134
+ arr.view(np.uint8), parameters={"__array__": "byte"}
135
+ )
136
+ else:
137
+ leaf = ak.contents.NumpyArray(arr)
138
+ return ak.Array(
139
+ ak.contents.ListOffsetArray(
140
+ ak.index.Index64(np.asarray(offsets, np.int64)),
141
+ leaf,
142
+ )
143
+ )
144
+
145
+
146
+ def single_cells() -> list[Cell]:
147
+ cells: list[Cell] = []
148
+ # Primary numeric workload: flanked alleles (8000 x ~11-60).
149
+ data, lengths, offsets = _r1_buffers(8000, 11, 60)
150
+ shape = (8000, None)
151
+ sh = "8000x~11-60 i64"
152
+
153
+ cells.append(
154
+ Cell(
155
+ "single",
156
+ "construct",
157
+ sh,
158
+ lambda: _ak_r1(offsets, data),
159
+ lambda: RustRagged.from_offsets(data, shape, offsets),
160
+ )
161
+ )
162
+
163
+ akx = _ak_r1(offsets, data)
164
+ rx = RustRagged.from_offsets(data, shape, offsets)
165
+
166
+ cells.append(Cell("single", "index[int]", sh, lambda: akx[1234], lambda: rx[1234]))
167
+ cells.append(
168
+ Cell(
169
+ "single", "index[slice]", sh, lambda: akx[1000:5000], lambda: rx[1000:5000]
170
+ )
171
+ )
172
+
173
+ mask = np.arange(8000) % 3 == 0
174
+ cells.append(Cell("single", "index[mask]", sh, lambda: akx[mask], lambda: rx[mask]))
175
+
176
+ # Use UNPACKED input for to_packed so both backends do real gather work.
177
+ # On already-packed data: awkward to_packed is zero-copy (shares buffer) while
178
+ # rust to_packed(copy=True) does a defensive data.copy() — unequal work.
179
+ # On unpacked (masked) input both backends must gather scattered segments.
180
+ akx_unpacked = akx[mask]
181
+ rx_unpacked = rx[mask]
182
+ cells.append(
183
+ Cell(
184
+ "single",
185
+ "to_packed",
186
+ f"{sh} unpacked",
187
+ lambda: ak.to_packed(akx_unpacked),
188
+ lambda: rx_unpacked.to_packed(),
189
+ )
190
+ )
191
+
192
+ L = int(lengths.max())
193
+ cells.append(
194
+ Cell(
195
+ "single",
196
+ "to_padded",
197
+ f"{sh} L={L}",
198
+ lambda: ak.to_numpy(ak.fill_none(ak.pad_none(akx, L, clip=True), 0)),
199
+ lambda: rx.to_padded(0, length=L),
200
+ )
201
+ )
202
+
203
+ cells.append(Cell("single", "ufunc(+1)", sh, lambda: akx + 1, lambda: rx + 1))
204
+
205
+ # S1 byte workload: construct + to_packed (the kernel-relevant ops).
206
+ bdata, blen, boff = _r1_buffers(8000, 11, 60, bytes_=True)
207
+ bshape = (8000, None)
208
+ bsh = "8000x~11-60 S1"
209
+ akb = _ak_r1(boff, bdata)
210
+ rb = RustRagged.from_offsets(bdata, bshape, boff)
211
+ cells.append(
212
+ Cell(
213
+ "single",
214
+ "construct",
215
+ bsh,
216
+ lambda: _ak_r1(boff, bdata),
217
+ lambda: RustRagged.from_offsets(bdata, bshape, boff),
218
+ )
219
+ )
220
+ # Use UNPACKED input for to_packed so both backends do real gather work.
221
+ # On already-packed data: awkward to_packed is zero-copy (shares buffer) while
222
+ # rust to_packed(copy=True) does a defensive data.copy() — unequal work.
223
+ # On unpacked (masked) input both backends must gather scattered segments.
224
+ bmask = np.arange(8000) % 3 == 0
225
+ akb_unpacked = akb[bmask]
226
+ rb_unpacked = rb[bmask]
227
+ cells.append(
228
+ Cell(
229
+ "single",
230
+ "to_packed",
231
+ f"{bsh} unpacked",
232
+ lambda: ak.to_packed(akb_unpacked),
233
+ lambda: rb_unpacked.to_packed(),
234
+ )
235
+ )
236
+ return cells
237
+
238
+
239
+ def record_cells() -> list[Cell]:
240
+ cells: list[Cell] = []
241
+ # genoray-like: 400 segments (samples*ploidy), ~variants in [0, 200), 3 numeric fields.
242
+ rng = np.random.default_rng(0)
243
+ n = 400
244
+ lengths = rng.integers(0, 200, size=n).astype(np.int64)
245
+ total = int(lengths.sum())
246
+ offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
247
+ genos = rng.integers(0, 2, size=total).astype(np.int8)
248
+ dosages = rng.random(total).astype(np.float32)
249
+ mutcat = rng.integers(0, 5, size=total).astype(np.int32)
250
+ shape = (n, None)
251
+ sh = "400x~0-200 3fld"
252
+
253
+ fields = {"genos": genos, "dosages": dosages, "mutcat": mutcat}
254
+
255
+ def _ak_rec():
256
+ return ak.zip({k: _ak_r1(offsets, v) for k, v in fields.items()}, depth_limit=1)
257
+
258
+ def _rust_rec():
259
+ return RustRagged.from_fields(
260
+ {k: RustRagged.from_offsets(v, shape, offsets) for k, v in fields.items()}
261
+ )
262
+
263
+ cells.append(Cell("records", "zip/from_fields", sh, _ak_rec, _rust_rec))
264
+
265
+ akr = _ak_rec()
266
+ rr = _rust_rec()
267
+ cells.append(
268
+ Cell("records", "field[a]", sh, lambda: akr["dosages"], lambda: rr["dosages"])
269
+ )
270
+ cells.append(
271
+ Cell(
272
+ "records",
273
+ "to_packed",
274
+ sh,
275
+ lambda: ak.to_packed(akr),
276
+ lambda: rr.to_packed(),
277
+ )
278
+ )
279
+
280
+ # Per-field dense: awkward pads each field; rust returns a dict.
281
+ L = int(lengths.max())
282
+
283
+ def _ak_padded_dict():
284
+ return {
285
+ k: ak.to_numpy(ak.fill_none(ak.pad_none(akr[k], L, clip=True), 0))
286
+ for k in fields
287
+ }
288
+
289
+ cells.append(
290
+ Cell(
291
+ "records",
292
+ "to_padded(dict)",
293
+ f"{sh} L={L}",
294
+ _ak_padded_dict,
295
+ lambda: rr.to_padded(0, length=L),
296
+ )
297
+ )
298
+ return cells
299
+
300
+
301
+ def _r2_buffers(n_outer: int, mid_low: int, mid_high: int, in_low: int, in_high: int):
302
+ """Return (data, o0, o1, outer_counts, inner_lengths) for an R=2 structure."""
303
+ rng = np.random.default_rng(0)
304
+ outer_counts = rng.integers(mid_low, mid_high + 1, size=n_outer).astype(np.int64)
305
+ n_mid = int(outer_counts.sum())
306
+ inner_lengths = rng.integers(in_low, in_high + 1, size=n_mid).astype(np.int64)
307
+ total = int(inner_lengths.sum())
308
+ data = np.arange(total, dtype=np.int64)
309
+ o0 = np.concatenate([[0], np.cumsum(outer_counts)]).astype(np.int64)
310
+ o1 = np.concatenate([[0], np.cumsum(inner_lengths)]).astype(np.int64)
311
+ return data, o0, o1, outer_counts, inner_lengths
312
+
313
+
314
+ def _ak_r2(o0, o1, data):
315
+ return ak.Array(
316
+ ak.contents.ListOffsetArray(
317
+ ak.index.Index64(np.asarray(o0, np.int64)),
318
+ ak.contents.ListOffsetArray(
319
+ ak.index.Index64(np.asarray(o1, np.int64)),
320
+ ak.contents.NumpyArray(np.asarray(data)),
321
+ ),
322
+ )
323
+ )
324
+
325
+
326
+ def nested_cells() -> list[Cell]:
327
+ cells: list[Cell] = []
328
+ # flat-variant-windows-like: 64 outer groups, ~variants in [1,30], ~window in [1,20].
329
+ data, o0, o1, outer_counts, inner_lengths = _r2_buffers(64, 1, 30, 1, 20)
330
+ shape = (64, None, None)
331
+ sh = "64x~1-30x~1-20 i64"
332
+
333
+ cells.append(
334
+ Cell(
335
+ "nested",
336
+ "construct",
337
+ sh,
338
+ lambda: _ak_r2(o0, o1, data),
339
+ lambda: RustRagged.from_offsets(data, shape, [o0, o1]),
340
+ )
341
+ )
342
+
343
+ akx = _ak_r2(o0, o1, data)
344
+ rx = RustRagged.from_offsets(data, shape, [o0, o1])
345
+
346
+ cells.append(Cell("nested", "index[int]", sh, lambda: akx[7], lambda: rx[7]))
347
+ cells.append(
348
+ Cell("nested", "index[slice]", sh, lambda: akx[8:40], lambda: rx[8:40])
349
+ )
350
+ # rag[:, k]: k-th middle of each outer group. Use k=0 so every group is non-empty
351
+ # (outer_counts >= 1 by construction).
352
+ cells.append(Cell("nested", "index[:,k]", sh, lambda: akx[:, 0], lambda: rx[:, 0]))
353
+ cells.append(
354
+ Cell("nested", "index[:,a:b]", sh, lambda: akx[:, 0:2], lambda: rx[:, 0:2])
355
+ )
356
+
357
+ # rag[:, mask]: rust takes a flat boolean over the global middle axis; awkward takes
358
+ # the same mask regrouped under the outer counts (a ragged boolean of matching shape).
359
+ n_mid = int(outer_counts.sum())
360
+ flat_mask = np.arange(n_mid) % 2 == 0
361
+ ak_mask = ak.unflatten(flat_mask, list(outer_counts))
362
+ cells.append(
363
+ Cell(
364
+ "nested",
365
+ "index[:,mask]",
366
+ sh,
367
+ lambda: akx[ak_mask],
368
+ lambda: rx[:, flat_mask],
369
+ )
370
+ )
371
+
372
+ cells.append(
373
+ Cell(
374
+ "nested", "to_packed", sh, lambda: ak.to_packed(akx), lambda: rx.to_packed()
375
+ )
376
+ )
377
+
378
+ # Dense both axes: M = max middles per group, K = max inner length.
379
+ M = int(outer_counts.max())
380
+ K = int(inner_lengths.max()) if inner_lengths.size else 0
381
+
382
+ def _ak_dense():
383
+ padded_inner = ak.fill_none(ak.pad_none(akx, K, axis=2, clip=True), 0)
384
+ padded_outer = ak.pad_none(padded_inner, M, axis=1, clip=True)
385
+ # fill_none with a length-K array fills axis=1 None slots; to_numpy returns a masked
386
+ # array because awkward retains the option type — call .filled(0) to materialise zeros.
387
+ fill_val = np.zeros(K, dtype=np.int64)
388
+ return ak.to_numpy(
389
+ ak.fill_none(padded_outer, fill_val), allow_missing=True
390
+ ).filled(0)
391
+
392
+ cells.append(
393
+ Cell(
394
+ "nested",
395
+ "to_padded(both)",
396
+ f"{sh} M={M},K={K}",
397
+ _ak_dense,
398
+ lambda: rx.to_padded(0, axis=None),
399
+ )
400
+ )
401
+ return cells
402
+
403
+
404
+ def string_cells() -> list[Cell]:
405
+ cells: list[Cell] = []
406
+ # 8000 short opaque strings (alleles), lengths in [1, 8].
407
+ rng = np.random.default_rng(0)
408
+ n = 8000
409
+ lengths = rng.integers(1, 9, size=n).astype(np.int64)
410
+ total = int(lengths.sum())
411
+ data = _BASES[rng.integers(0, 4, size=total)] # (total,) S1
412
+ offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
413
+
414
+ # rust: opaque-string Ragged (shape (n,), dtype 'S') -> chars (n, ~length) and back.
415
+ r_str = RustRagged.from_lengths(data, lengths) # opaque string by default
416
+ r_chars = r_str.to_chars()
417
+
418
+ # awkward analogue: a list-of-S1 (chars) array; "to chars" = view as char list,
419
+ # "to strings" = join back to bytestrings. Build the char-list oracle once.
420
+ ak_chars = _ak_r1(offsets, data) # ListOffsetArray over S1 == char lists
421
+
422
+ # to_chars: rust retag vs awkward producing the char-list view.
423
+ cells.append(
424
+ Cell(
425
+ "string",
426
+ "to_chars",
427
+ f"{n}x~1-8",
428
+ lambda: ak.copy(ak_chars),
429
+ lambda: r_str.to_chars(),
430
+ eq=lambda a, b: True,
431
+ )
432
+ ) # structural shapes differ across backends; time-only
433
+ # to_strings: rust retag vs awkward joining char lists into bytestrings.
434
+ cells.append(
435
+ Cell(
436
+ "string",
437
+ "to_strings",
438
+ f"{n}x~1-8",
439
+ lambda: ak.copy(ak_chars),
440
+ lambda: r_chars.to_strings(),
441
+ eq=lambda a, b: True,
442
+ )
443
+ )
444
+ return cells
445
+
446
+
447
+ def build_cells(args: argparse.Namespace) -> list[Cell]:
448
+ """Assemble the cell list. Extended in later tasks."""
449
+ cells: list[Cell] = []
450
+ if args.only in ("all", "single"):
451
+ cells += single_cells()
452
+ if args.only in ("all", "records"):
453
+ cells += record_cells()
454
+ if args.only in ("all", "nested"):
455
+ cells += nested_cells()
456
+ if args.only in ("all", "string"):
457
+ cells += string_cells()
458
+ return cells
459
+
460
+
461
+ def main(argv: "list[str] | None" = None) -> int:
462
+ p = argparse.ArgumentParser(
463
+ description="rust-native vs awkward Ragged throughput gate"
464
+ )
465
+ p.add_argument("--tol", type=float, default=0.10)
466
+ p.add_argument(
467
+ "--only",
468
+ choices=["single", "records", "nested", "string", "all"],
469
+ default="all",
470
+ )
471
+ p.add_argument("--repeats", type=int, default=7)
472
+ args = p.parse_args(argv)
473
+ cells = build_cells(args)
474
+ if not cells:
475
+ print(f"no cells for --only {args.only}")
476
+ return 0
477
+ return run_cells(cells, args.tol, args.repeats)
478
+
479
+
480
+ if __name__ == "__main__":
481
+ raise SystemExit(main())