seqpro 0.16.0__tar.gz → 0.18.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/bench.yaml +2 -2
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/lint.yaml +2 -2
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/test.yaml +2 -2
- {seqpro-0.16.0 → seqpro-0.18.0}/.pre-commit-config.yaml +18 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/CHANGELOG.md +110 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/PKG-INFO +1 -1
- seqpro-0.18.0/benches/bench_tokenize_translate.py +30 -0
- seqpro-0.18.0/benchmarks/bench_ragged_backends.py +481 -0
- seqpro-0.18.0/docs/roadmap/rust-ragged.md +388 -0
- seqpro-0.18.0/docs/superpowers/plans/2026-06-18-rust-tokenize-translate.md +1391 -0
- seqpro-0.18.0/docs/superpowers/plans/2026-06-19-rust-ragged-core.md +1347 -0
- seqpro-0.18.0/docs/superpowers/plans/2026-06-20-rust-ragged-records.md +1583 -0
- seqpro-0.18.0/docs/superpowers/plans/2026-06-20-rust-ragged-spec-c.md +1556 -0
- seqpro-0.18.0/docs/superpowers/plans/2026-06-21-ragged-throughput-gate.md +656 -0
- seqpro-0.18.0/docs/superpowers/plans/2026-06-21-rust-ragged-consumer-audit.md +492 -0
- seqpro-0.18.0/docs/superpowers/specs/2026-06-18-rust-tokenize-translate-design.md +360 -0
- seqpro-0.18.0/docs/superpowers/specs/2026-06-19-rust-ragged-core-design.md +260 -0
- seqpro-0.18.0/docs/superpowers/specs/2026-06-20-rust-ragged-nested-design.md +364 -0
- seqpro-0.18.0/docs/superpowers/specs/2026-06-20-rust-ragged-records-design.md +327 -0
- seqpro-0.18.0/docs/superpowers/specs/2026-06-21-ragged-throughput-gate-design.md +166 -0
- seqpro-0.18.0/docs/superpowers/specs/2026-06-21-rust-ragged-audit-ledger.md +75 -0
- seqpro-0.18.0/docs/superpowers/specs/2026-06-21-rust-ragged-consumer-audit-design.md +121 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/pixi.lock +52 -16
- {seqpro-0.16.0 → seqpro-0.18.0}/pixi.toml +10 -1
- {seqpro-0.16.0 → seqpro-0.18.0}/pyproject.toml +7 -1
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/alphabets/_alphabets.py +137 -92
- seqpro-0.18.0/python/seqpro/rag/__init__.py +29 -0
- seqpro-0.18.0/python/seqpro/rag/_ak_interop.py +208 -0
- seqpro-0.18.0/python/seqpro/rag/_core.py +1612 -0
- seqpro-0.18.0/python/seqpro/rag/_ingest.py +252 -0
- seqpro-0.18.0/python/seqpro/rag/_layout.py +172 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/rag/_ops.py +161 -49
- {seqpro-0.16.0 → seqpro-0.18.0}/skills/seqpro/SKILL.md +35 -24
- {seqpro-0.16.0 → seqpro-0.18.0}/src/kshuffle.rs +132 -65
- seqpro-0.18.0/src/lib.rs +212 -0
- seqpro-0.18.0/src/ragged.rs +612 -0
- seqpro-0.18.0/src/translate.rs +510 -0
- seqpro-0.18.0/tests/test_concatenate.py +29 -0
- seqpro-0.18.0/tests/test_core_ragged_surface.py +36 -0
- seqpro-0.18.0/tests/test_ingest.py +19 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_rag_to_packed.py +93 -28
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_ragged.py +57 -34
- seqpro-0.18.0/tests/test_ragged_core.py +810 -0
- seqpro-0.18.0/tests/test_ragged_core_records.py +682 -0
- seqpro-0.18.0/tests/test_ragged_nested_consumers.py +68 -0
- seqpro-0.18.0/tests/test_ragged_nested_diff.py +297 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_ragged_rc.py +20 -26
- seqpro-0.18.0/tests/test_ragged_record_indexing.py +47 -0
- seqpro-0.18.0/tests/test_ragged_subclass.py +60 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_ragged_to_padded.py +66 -6
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_translate.py +48 -19
- seqpro-0.18.0/tests/test_translate_rust.py +135 -0
- seqpro-0.16.0/python/seqpro/rag/__init__.py +0 -15
- seqpro-0.16.0/python/seqpro/rag/_array.py +0 -904
- seqpro-0.16.0/python/seqpro/rag/_gufuncs.py +0 -14
- seqpro-0.16.0/src/lib.rs +0 -44
- {seqpro-0.16.0 → seqpro-0.18.0}/.claude/skills/zensical/SKILL.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/.gitattributes +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/bump.yaml +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/docs.yml +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/merge.yaml +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/publish.yaml +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/release-pipeline.yaml +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/.github/workflows/release.yaml +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/.gitignore +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/CLAUDE.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/Cargo.lock +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/Cargo.toml +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/LICENSE +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/README.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/benches/kshuffle.rs +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/benchmarks/bench_to_packed.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/alphabets.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/bed.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/gtf.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/index.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/ragged.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/api/types.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/index.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/ragged.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-04-ragged-record-array.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-05-documentation-site.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-05-narwhals-coord-schema.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-05-ragged-zip-and-record-introspection.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-20-kshuffle-optimization.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-20-kshuffle-pooled-buffers-and-k2-fast-path.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-20-kshuffle-wilson-single-pass.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-20-release-pipeline.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-28-translate-lut-validation.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-31-flat-buffer-to-padded.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-05-31-rag-to-packed.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-06-05-translate-unknown-codon-policy.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/plans/2026-06-12-tokenize-lut-codspeed.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-04-ragged-record-array-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-05-documentation-site-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-05-narwhals-coord-schema-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-05-ragged-zip-and-record-introspection-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-20-kshuffle-optimization-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-20-kshuffle-pooled-buffers-and-k2-fast-path-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-20-kshuffle-wilson-single-pass-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-20-release-pipeline-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-28-translate-lut-and-validation-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-31-flat-buffer-to-padded-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-05-31-rag-to-packed-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-06-05-translate-unknown-codon-policy-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/docs/superpowers/specs/2026-06-12-tokenize-lut-codspeed-design.md +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/meta.yaml +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/__init__.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_analyzers.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_cleaners.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_coords.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_encoders.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_modifiers.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_numba.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_types.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/_utils.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/alphabets/__init__.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/bed.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/experimental/_experimental.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/experimental/_visualizers.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/gtf.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/py.typed +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/rag/_types.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/rag/_utils.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/transforms/__init__.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/transforms/augmentation.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/transforms/tmm.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/python/seqpro/xr/__init__.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/scratch_bench_rc.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/scratch_bench_to_padded.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/src/kmer_encode.rs +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/src/kshuffle_ref.rs +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/_shape_fixtures.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/bed/test_pyranges.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/bed/test_read.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/bed/test_sort.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/bed/test_with_length.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/bench_translate_lut.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/conftest.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_analyzers.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_bench_tokenize.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_coords.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_encoders.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_modifiers.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_ohe.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_shape_matrix.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_tokenize.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/tests/test_transforms.py +0 -0
- {seqpro-0.16.0 → seqpro-0.18.0}/zensical.toml +0 -0
|
@@ -25,10 +25,10 @@ jobs:
|
|
|
25
25
|
- name: Setup pixi
|
|
26
26
|
uses: prefix-dev/setup-pixi@v0.9.5
|
|
27
27
|
with:
|
|
28
|
-
pixi-version: v0.
|
|
28
|
+
pixi-version: v0.70.1
|
|
29
29
|
cache: true
|
|
30
30
|
environments: bench
|
|
31
|
-
locked:
|
|
31
|
+
locked: true
|
|
32
32
|
- name: Build extension
|
|
33
33
|
run: pixi run -e bench maturin develop
|
|
34
34
|
- name: Run benchmarks
|
|
@@ -21,9 +21,9 @@ jobs:
|
|
|
21
21
|
- name: Setup pixi
|
|
22
22
|
uses: prefix-dev/setup-pixi@v0.9.5
|
|
23
23
|
with:
|
|
24
|
-
pixi-version: v0.
|
|
24
|
+
pixi-version: v0.70.1
|
|
25
25
|
cache: true
|
|
26
26
|
environments: ${{ matrix.environment }}
|
|
27
|
-
locked:
|
|
27
|
+
locked: true
|
|
28
28
|
- name: Test
|
|
29
29
|
run: pixi run -e ${{ matrix.environment }} test
|
|
@@ -33,3 +33,21 @@ repos:
|
|
|
33
33
|
language: system
|
|
34
34
|
stages: [pre-push]
|
|
35
35
|
pass_filenames: false
|
|
36
|
+
- id: cargo-check
|
|
37
|
+
name: Type-check Rust with cargo check
|
|
38
|
+
entry: cargo check --all-targets --all-features
|
|
39
|
+
language: system
|
|
40
|
+
types: [rust]
|
|
41
|
+
pass_filenames: false
|
|
42
|
+
- id: cargo-fmt
|
|
43
|
+
name: Format Rust with cargo fmt
|
|
44
|
+
entry: cargo fmt --all
|
|
45
|
+
language: system
|
|
46
|
+
types: [rust]
|
|
47
|
+
pass_filenames: false
|
|
48
|
+
- id: cargo-clippy
|
|
49
|
+
name: Lint Rust with cargo clippy
|
|
50
|
+
entry: cargo clippy --all-targets -- -D warnings
|
|
51
|
+
language: system
|
|
52
|
+
types: [rust]
|
|
53
|
+
pass_filenames: false
|
|
@@ -1,3 +1,113 @@
|
|
|
1
|
+
## 0.18.0 (2026-06-23)
|
|
2
|
+
|
|
3
|
+
### Feat
|
|
4
|
+
|
|
5
|
+
- **rag**: reshape/squeeze/to_packed preserve Ragged subclass
|
|
6
|
+
- **rag**: __getitem__ preserves subclass on positional indexing
|
|
7
|
+
- **rag**: add _with_layout subclass-preserving constructor
|
|
8
|
+
|
|
9
|
+
### Fix
|
|
10
|
+
|
|
11
|
+
- **rag**: non-tuple record indexing matches numpy (A[x] == A[(x,)])
|
|
12
|
+
|
|
13
|
+
## 0.17.0 (2026-06-23)
|
|
14
|
+
|
|
15
|
+
### Feat
|
|
16
|
+
|
|
17
|
+
- **rag**: concatenate() along ragged axis (Rust kernel)
|
|
18
|
+
- **rag**: to_packed() on record and opaque-string-under-axis Ragged
|
|
19
|
+
- wire rag-gate pixi task; record throughput gate outcome
|
|
20
|
+
- string-conversion op cells for Ragged throughput gate
|
|
21
|
+
- nested R=2 op cells for Ragged throughput gate
|
|
22
|
+
- record (SoA) op cells for Ragged throughput gate
|
|
23
|
+
- single-level op cells for Ragged throughput gate
|
|
24
|
+
- harness core for rust-vs-awkward Ragged throughput gate
|
|
25
|
+
- _ingest bridge for R=2 + string-under-axis (oracle interop)
|
|
26
|
+
- nested record indexing + record-aware to_packed/to_padded/to_numpy
|
|
27
|
+
- nested + string-under-axis record fields sharing full offsets list
|
|
28
|
+
- R=2 lengths/squeeze/reshape
|
|
29
|
+
- per-axis nested to_padded + rectangular to_numpy (R=2); trailing-dim support in to_padded
|
|
30
|
+
- nested to_packed via nested_pack kernel
|
|
31
|
+
- nested_pack Rust kernel + binding for two-level pack
|
|
32
|
+
- per-group inner mask/int-array indexing via nested_gather
|
|
33
|
+
- nested_gather Rust kernel + binding for per-group middle selection
|
|
34
|
+
- per-group inner int/slice indexing (rag[:, k], rag[:, a:b])
|
|
35
|
+
- R=2 tuple indexing + leaf access via peel chaining
|
|
36
|
+
- R=2 outer-row indexing (lazy gather, peel to 1-level)
|
|
37
|
+
- string-under-axis leaf + nested to_chars/to_strings
|
|
38
|
+
- nested constructors from_offsets(list)/from_lengths(tuple) for R=2
|
|
39
|
+
- validate R=2 nested ragged layouts; cap at R<=2
|
|
40
|
+
- ingest/emit record layouts via awkward bridge (oracle interop)
|
|
41
|
+
- record to_numpy/to_padded dicts; raise view/ufunc on records
|
|
42
|
+
- record-aware to_packed (one shared packed offsets across fields)
|
|
43
|
+
- per-field squeeze/reshape on record Ragged
|
|
44
|
+
- record row-axis indexing (slice/mask -> record, int -> dict)
|
|
45
|
+
- record field access (key/attr) and __setitem__ mutation
|
|
46
|
+
- record-branch properties (data/dtype/offsets/shape/fields/state)
|
|
47
|
+
- add Ragged.from_fields record constructor and rag.zip alias
|
|
48
|
+
- add RecordLayout value object and validation arm
|
|
49
|
+
- add zero-copy to_chars/to_strings between opaque-string and char Ragged
|
|
50
|
+
- disambiguate opaque-string vs char layout by presence of None in shape
|
|
51
|
+
- report np.dtype('S') for opaque-string Ragged (string/char duality)
|
|
52
|
+
- **rust**: ragged_validate and ragged_select kernels
|
|
53
|
+
- **rag**: awkward ingestion and to_ak shim
|
|
54
|
+
- **rag**: to_numpy, to_packed, to_padded
|
|
55
|
+
- **rag**: squeeze and reshape on regular dims
|
|
56
|
+
- **rag**: element-wise ufunc interop
|
|
57
|
+
- **rag**: __getitem__ indexing and slicing
|
|
58
|
+
- **rag**: state predicates and view
|
|
59
|
+
- **rag**: Ragged constructors and core properties
|
|
60
|
+
- **rag**: RaggedLayout value object + validation
|
|
61
|
+
- **translate**: self-contained Rust OHE<->AA path with native drop
|
|
62
|
+
- **translate**: route truncate_stop through Rust
|
|
63
|
+
- **translate**: route unknown=drop compaction through Rust
|
|
64
|
+
- **translate**: route ragged bytes pad path through Rust
|
|
65
|
+
- **translate**: route dense pad path through Rust
|
|
66
|
+
- **rust**: add codon-stride translate kernels
|
|
67
|
+
- **tokenize**: route ragged path through Rust, drop Numba gather
|
|
68
|
+
- **tokenize**: route dense path through Rust _tokenize
|
|
69
|
+
- **rust**: add tokenize LUT gather kernel
|
|
70
|
+
|
|
71
|
+
### Fix
|
|
72
|
+
|
|
73
|
+
- **rag**: to_ak() on multi-leading-axis record Ragged
|
|
74
|
+
- **rag**: _core.Ragged/tokenize fixes found via genvarformer audit
|
|
75
|
+
- **rag**: add __len__, np.newaxis support, and element-wise fancy indexing to _core.Ragged
|
|
76
|
+
- **rag**: _ops.to_padded record detection works for both _array and _core backends
|
|
77
|
+
- **rag**: _core.to_numpy returns dict for records (restore designed contract); update _array-era test
|
|
78
|
+
- **rag**: correct _core getitem tuple routing for rag_dim==1 found via SeqPro own-suite audit
|
|
79
|
+
- **rag**: port _ops + _core to_packed/is_rag_dtype/to_numpy to _core object model
|
|
80
|
+
- **rag**: precise _core getitem routing — _core contract + genoray parity both green
|
|
81
|
+
- **rag**: repair _core.Ragged getitem regressions (string leaf, to_strings, r2 int array)
|
|
82
|
+
- **rag**: match _array is_base semantics for non-ndarray base
|
|
83
|
+
- **rag**: _core.Ragged fixes found via genoray audit
|
|
84
|
+
- **bench**: time to_packed on unpacked input so single-level pack compares equal work
|
|
85
|
+
- forward --repeats CLI flag into time_callable
|
|
86
|
+
- R=2 is_contiguous checks all offset levels; guard string-under-axis to_packed/to_numpy
|
|
87
|
+
- panic-safety in nested_pack (checked_mul, elem>0, o0-range guard)
|
|
88
|
+
- reject negative inner-slice bounds in rag[:, a:b]
|
|
89
|
+
- -O-safe record guards on is_string/to_chars/to_strings; note latent S4 over-acceptance
|
|
90
|
+
- raise TypeError on np.asarray of a record Ragged (-O-safe, was a bare assert)
|
|
91
|
+
- **rag**: reject mismatched boolean masks and bounds-check ragged_select
|
|
92
|
+
- **translate**: handle empty dense input in unknown=drop path
|
|
93
|
+
|
|
94
|
+
### Refactor
|
|
95
|
+
|
|
96
|
+
- **rag**: delete awkward _array backend; relocate interop to _ak_interop
|
|
97
|
+
- resolve clippy findings for -D warnings
|
|
98
|
+
- **rag**: route index/validate hot paths through Rust
|
|
99
|
+
- **rag**: use NDArrayOperatorsMixin for operator dunders
|
|
100
|
+
- **tokenize**: restore np.take/Numba path; Rust port regressed vs baseline
|
|
101
|
+
|
|
102
|
+
### Perf
|
|
103
|
+
|
|
104
|
+
- validate pack output size up-front; test parallel pack path
|
|
105
|
+
- Rust single-level pack kernel for to_packed (parallel gather)
|
|
106
|
+
- opt-in Ragged validation (default off) + slice indexing fast-path
|
|
107
|
+
- drop redundant per-field copy in record to_packed; align record is_base with single-level one-indirection rule
|
|
108
|
+
- **translate**: optimize LUT gather kernel (validity table + coarse chunking)
|
|
109
|
+
- **translate**: add rayon parallel path for large inputs
|
|
110
|
+
|
|
1
111
|
## 0.16.0 (2026-06-14)
|
|
2
112
|
|
|
3
113
|
### Feat
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""End-to-end Python-level benchmarks for the Rust tokenize/translate port.
|
|
2
|
+
|
|
3
|
+
Run: pixi run -e dev pytest benches/bench_tokenize_translate.py --benchmark-only
|
|
4
|
+
|
|
5
|
+
Measures the public API (including PyO3 marshalling), which Rust-only criterion
|
|
6
|
+
misses and which decides the small-array regime. To compare against the Numba
|
|
7
|
+
baseline, check out the pre-port commit and run the same file.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
import pytest
|
|
12
|
+
|
|
13
|
+
import seqpro as sp
|
|
14
|
+
|
|
15
|
+
TOKEN_MAP = {"A": 0, "C": 1, "G": 2, "T": 3}
|
|
16
|
+
SIZES = [100, 10_000, 1_000_000]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@pytest.mark.parametrize("n", SIZES)
|
|
20
|
+
def test_bench_tokenize_dense(benchmark, n):
|
|
21
|
+
rng = np.random.default_rng(0)
|
|
22
|
+
seqs = rng.choice(np.frombuffer(b"ACGT", "S1"), size=n)
|
|
23
|
+
benchmark(lambda: sp.tokenize(seqs, TOKEN_MAP, 7))
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@pytest.mark.parametrize("n_codons", [33, 3_333, 333_333])
|
|
27
|
+
def test_bench_translate_dense(benchmark, n_codons):
|
|
28
|
+
rng = np.random.default_rng(1)
|
|
29
|
+
seqs = rng.choice(np.frombuffer(b"ACGT", "S1"), size=n_codons * 3)
|
|
30
|
+
benchmark(lambda: sp.AA.translate(seqs, length_axis=0))
|
|
@@ -0,0 +1,481 @@
|
|
|
1
|
+
"""Local, transitional A-vs-B throughput gate: rust-native Ragged vs awkward.
|
|
2
|
+
|
|
3
|
+
Proves seqpro.rag._core.Ragged is at least as fast as awkward's native layout
|
|
4
|
+
algebra before the Spec D cutover drops awkward. NOT a permanent CI fixture —
|
|
5
|
+
deleted at cutover (see docs/superpowers/specs/2026-06-21-ragged-throughput-gate-design.md).
|
|
6
|
+
|
|
7
|
+
Run: pixi run -e bench rag-gate (or: python benchmarks/bench_ragged_backends.py --tol 0.10)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from time import perf_counter
|
|
15
|
+
from typing import Any, Callable
|
|
16
|
+
|
|
17
|
+
import awkward as ak
|
|
18
|
+
import numpy as np
|
|
19
|
+
|
|
20
|
+
from seqpro.rag._core import Ragged as RustRagged
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class Cell:
|
|
25
|
+
category: str
|
|
26
|
+
op: str
|
|
27
|
+
shape: str
|
|
28
|
+
awk: Callable[[], Any]
|
|
29
|
+
rust: Callable[[], Any]
|
|
30
|
+
eq: "Callable[[Any, Any], bool] | None" = None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def time_callable(
|
|
34
|
+
fn: Callable[[], Any], *, repeats: int = 7, min_batch_s: float = 0.005
|
|
35
|
+
) -> float:
|
|
36
|
+
"""Seconds per call: warm up, autoscale a batch past min_batch_s, take min of repeats."""
|
|
37
|
+
for _ in range(3):
|
|
38
|
+
fn()
|
|
39
|
+
iters = 1
|
|
40
|
+
while True:
|
|
41
|
+
t0 = perf_counter()
|
|
42
|
+
for _ in range(iters):
|
|
43
|
+
fn()
|
|
44
|
+
if perf_counter() - t0 >= min_batch_s:
|
|
45
|
+
break
|
|
46
|
+
iters *= 2
|
|
47
|
+
best = float("inf")
|
|
48
|
+
for _ in range(repeats):
|
|
49
|
+
t0 = perf_counter()
|
|
50
|
+
for _ in range(iters):
|
|
51
|
+
fn()
|
|
52
|
+
best = min(best, (perf_counter() - t0) / iters)
|
|
53
|
+
return best
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def to_list(x: Any) -> Any:
|
|
57
|
+
"""Canonicalize a result to a comparable structure for equivalence checks."""
|
|
58
|
+
if isinstance(x, RustRagged):
|
|
59
|
+
return to_list(x.to_ak())
|
|
60
|
+
if isinstance(x, ak.Array):
|
|
61
|
+
return ak.to_list(x)
|
|
62
|
+
if isinstance(x, dict):
|
|
63
|
+
return {k: to_list(v) for k, v in x.items()}
|
|
64
|
+
if isinstance(x, np.ndarray):
|
|
65
|
+
return x.tolist()
|
|
66
|
+
if isinstance(x, (np.integer, np.floating)):
|
|
67
|
+
return x.item()
|
|
68
|
+
return x
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def default_eq(a: Any, b: Any) -> bool:
|
|
72
|
+
return to_list(a) == to_list(b)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def run_cells(cells: list[Cell], tol: float, repeats: int = 7) -> int:
|
|
76
|
+
rows: list[tuple[str, str, str, float, float, float, bool]] = []
|
|
77
|
+
failures = 0
|
|
78
|
+
for c in cells:
|
|
79
|
+
eq = c.eq or default_eq
|
|
80
|
+
if not eq(c.awk(), c.rust()):
|
|
81
|
+
raise AssertionError(
|
|
82
|
+
f"equivalence check failed for {c.category}/{c.op} ({c.shape}); "
|
|
83
|
+
"the comparison would be unfair — fix the cell before timing."
|
|
84
|
+
)
|
|
85
|
+
t_awk = time_callable(c.awk, repeats=repeats)
|
|
86
|
+
t_rust = time_callable(c.rust, repeats=repeats)
|
|
87
|
+
ratio = t_rust / t_awk if t_awk > 0 else float("inf")
|
|
88
|
+
ok = ratio <= 1.0 + tol
|
|
89
|
+
failures += not ok
|
|
90
|
+
rows.append((c.category, c.op, c.shape, t_awk, t_rust, ratio, ok))
|
|
91
|
+
|
|
92
|
+
w_cat = max(len("category"), *(len(r[0]) for r in rows))
|
|
93
|
+
w_op = max(len("op"), *(len(r[1]) for r in rows))
|
|
94
|
+
w_sh = max(len("shape"), *(len(r[2]) for r in rows))
|
|
95
|
+
header = (
|
|
96
|
+
f"{'category':<{w_cat}} {'op':<{w_op}} {'shape':<{w_sh}} "
|
|
97
|
+
f"{'awk (us)':>10} {'rust (us)':>10} {'rust/awk':>9} verdict"
|
|
98
|
+
)
|
|
99
|
+
print(header)
|
|
100
|
+
print("-" * len(header))
|
|
101
|
+
for cat, op, sh, ta, tr, ratio, ok in rows:
|
|
102
|
+
print(
|
|
103
|
+
f"{cat:<{w_cat}} {op:<{w_op}} {sh:<{w_sh}} "
|
|
104
|
+
f"{ta * 1e6:>10.2f} {tr * 1e6:>10.2f} {ratio:>9.3f} "
|
|
105
|
+
f"{'PASS' if ok else 'FAIL'}"
|
|
106
|
+
)
|
|
107
|
+
n = len(rows)
|
|
108
|
+
print("-" * len(header))
|
|
109
|
+
print(f"{n - failures}/{n} passed (tol={tol:.2%})")
|
|
110
|
+
return 1 if failures else 0
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
_BASES = np.frombuffer(b"ACGT", dtype="S1")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _r1_buffers(n: int, low: int, high: int, *, bytes_: bool = False):
|
|
117
|
+
"""Return (data, lengths, offsets) for n segments with lengths in [low, high]."""
|
|
118
|
+
rng = np.random.default_rng(0)
|
|
119
|
+
lengths = rng.integers(low, high + 1, size=n).astype(np.int64)
|
|
120
|
+
total = int(lengths.sum())
|
|
121
|
+
if bytes_:
|
|
122
|
+
data = _BASES[rng.integers(0, 4, size=total)] # (total,) S1
|
|
123
|
+
else:
|
|
124
|
+
data = np.arange(total, dtype=np.int64)
|
|
125
|
+
offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
|
|
126
|
+
return data, lengths, offsets
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _ak_r1(offsets, data):
|
|
130
|
+
arr = np.asarray(data)
|
|
131
|
+
if arr.dtype.kind == "S":
|
|
132
|
+
# awkward does not support S1 dtype; view as uint8 bytes with "byte" parameter
|
|
133
|
+
leaf = ak.contents.NumpyArray(
|
|
134
|
+
arr.view(np.uint8), parameters={"__array__": "byte"}
|
|
135
|
+
)
|
|
136
|
+
else:
|
|
137
|
+
leaf = ak.contents.NumpyArray(arr)
|
|
138
|
+
return ak.Array(
|
|
139
|
+
ak.contents.ListOffsetArray(
|
|
140
|
+
ak.index.Index64(np.asarray(offsets, np.int64)),
|
|
141
|
+
leaf,
|
|
142
|
+
)
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def single_cells() -> list[Cell]:
|
|
147
|
+
cells: list[Cell] = []
|
|
148
|
+
# Primary numeric workload: flanked alleles (8000 x ~11-60).
|
|
149
|
+
data, lengths, offsets = _r1_buffers(8000, 11, 60)
|
|
150
|
+
shape = (8000, None)
|
|
151
|
+
sh = "8000x~11-60 i64"
|
|
152
|
+
|
|
153
|
+
cells.append(
|
|
154
|
+
Cell(
|
|
155
|
+
"single",
|
|
156
|
+
"construct",
|
|
157
|
+
sh,
|
|
158
|
+
lambda: _ak_r1(offsets, data),
|
|
159
|
+
lambda: RustRagged.from_offsets(data, shape, offsets),
|
|
160
|
+
)
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
akx = _ak_r1(offsets, data)
|
|
164
|
+
rx = RustRagged.from_offsets(data, shape, offsets)
|
|
165
|
+
|
|
166
|
+
cells.append(Cell("single", "index[int]", sh, lambda: akx[1234], lambda: rx[1234]))
|
|
167
|
+
cells.append(
|
|
168
|
+
Cell(
|
|
169
|
+
"single", "index[slice]", sh, lambda: akx[1000:5000], lambda: rx[1000:5000]
|
|
170
|
+
)
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
mask = np.arange(8000) % 3 == 0
|
|
174
|
+
cells.append(Cell("single", "index[mask]", sh, lambda: akx[mask], lambda: rx[mask]))
|
|
175
|
+
|
|
176
|
+
# Use UNPACKED input for to_packed so both backends do real gather work.
|
|
177
|
+
# On already-packed data: awkward to_packed is zero-copy (shares buffer) while
|
|
178
|
+
# rust to_packed(copy=True) does a defensive data.copy() — unequal work.
|
|
179
|
+
# On unpacked (masked) input both backends must gather scattered segments.
|
|
180
|
+
akx_unpacked = akx[mask]
|
|
181
|
+
rx_unpacked = rx[mask]
|
|
182
|
+
cells.append(
|
|
183
|
+
Cell(
|
|
184
|
+
"single",
|
|
185
|
+
"to_packed",
|
|
186
|
+
f"{sh} unpacked",
|
|
187
|
+
lambda: ak.to_packed(akx_unpacked),
|
|
188
|
+
lambda: rx_unpacked.to_packed(),
|
|
189
|
+
)
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
L = int(lengths.max())
|
|
193
|
+
cells.append(
|
|
194
|
+
Cell(
|
|
195
|
+
"single",
|
|
196
|
+
"to_padded",
|
|
197
|
+
f"{sh} L={L}",
|
|
198
|
+
lambda: ak.to_numpy(ak.fill_none(ak.pad_none(akx, L, clip=True), 0)),
|
|
199
|
+
lambda: rx.to_padded(0, length=L),
|
|
200
|
+
)
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
cells.append(Cell("single", "ufunc(+1)", sh, lambda: akx + 1, lambda: rx + 1))
|
|
204
|
+
|
|
205
|
+
# S1 byte workload: construct + to_packed (the kernel-relevant ops).
|
|
206
|
+
bdata, blen, boff = _r1_buffers(8000, 11, 60, bytes_=True)
|
|
207
|
+
bshape = (8000, None)
|
|
208
|
+
bsh = "8000x~11-60 S1"
|
|
209
|
+
akb = _ak_r1(boff, bdata)
|
|
210
|
+
rb = RustRagged.from_offsets(bdata, bshape, boff)
|
|
211
|
+
cells.append(
|
|
212
|
+
Cell(
|
|
213
|
+
"single",
|
|
214
|
+
"construct",
|
|
215
|
+
bsh,
|
|
216
|
+
lambda: _ak_r1(boff, bdata),
|
|
217
|
+
lambda: RustRagged.from_offsets(bdata, bshape, boff),
|
|
218
|
+
)
|
|
219
|
+
)
|
|
220
|
+
# Use UNPACKED input for to_packed so both backends do real gather work.
|
|
221
|
+
# On already-packed data: awkward to_packed is zero-copy (shares buffer) while
|
|
222
|
+
# rust to_packed(copy=True) does a defensive data.copy() — unequal work.
|
|
223
|
+
# On unpacked (masked) input both backends must gather scattered segments.
|
|
224
|
+
bmask = np.arange(8000) % 3 == 0
|
|
225
|
+
akb_unpacked = akb[bmask]
|
|
226
|
+
rb_unpacked = rb[bmask]
|
|
227
|
+
cells.append(
|
|
228
|
+
Cell(
|
|
229
|
+
"single",
|
|
230
|
+
"to_packed",
|
|
231
|
+
f"{bsh} unpacked",
|
|
232
|
+
lambda: ak.to_packed(akb_unpacked),
|
|
233
|
+
lambda: rb_unpacked.to_packed(),
|
|
234
|
+
)
|
|
235
|
+
)
|
|
236
|
+
return cells
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def record_cells() -> list[Cell]:
|
|
240
|
+
cells: list[Cell] = []
|
|
241
|
+
# genoray-like: 400 segments (samples*ploidy), ~variants in [0, 200), 3 numeric fields.
|
|
242
|
+
rng = np.random.default_rng(0)
|
|
243
|
+
n = 400
|
|
244
|
+
lengths = rng.integers(0, 200, size=n).astype(np.int64)
|
|
245
|
+
total = int(lengths.sum())
|
|
246
|
+
offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
|
|
247
|
+
genos = rng.integers(0, 2, size=total).astype(np.int8)
|
|
248
|
+
dosages = rng.random(total).astype(np.float32)
|
|
249
|
+
mutcat = rng.integers(0, 5, size=total).astype(np.int32)
|
|
250
|
+
shape = (n, None)
|
|
251
|
+
sh = "400x~0-200 3fld"
|
|
252
|
+
|
|
253
|
+
fields = {"genos": genos, "dosages": dosages, "mutcat": mutcat}
|
|
254
|
+
|
|
255
|
+
def _ak_rec():
|
|
256
|
+
return ak.zip({k: _ak_r1(offsets, v) for k, v in fields.items()}, depth_limit=1)
|
|
257
|
+
|
|
258
|
+
def _rust_rec():
|
|
259
|
+
return RustRagged.from_fields(
|
|
260
|
+
{k: RustRagged.from_offsets(v, shape, offsets) for k, v in fields.items()}
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
cells.append(Cell("records", "zip/from_fields", sh, _ak_rec, _rust_rec))
|
|
264
|
+
|
|
265
|
+
akr = _ak_rec()
|
|
266
|
+
rr = _rust_rec()
|
|
267
|
+
cells.append(
|
|
268
|
+
Cell("records", "field[a]", sh, lambda: akr["dosages"], lambda: rr["dosages"])
|
|
269
|
+
)
|
|
270
|
+
cells.append(
|
|
271
|
+
Cell(
|
|
272
|
+
"records",
|
|
273
|
+
"to_packed",
|
|
274
|
+
sh,
|
|
275
|
+
lambda: ak.to_packed(akr),
|
|
276
|
+
lambda: rr.to_packed(),
|
|
277
|
+
)
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
# Per-field dense: awkward pads each field; rust returns a dict.
|
|
281
|
+
L = int(lengths.max())
|
|
282
|
+
|
|
283
|
+
def _ak_padded_dict():
|
|
284
|
+
return {
|
|
285
|
+
k: ak.to_numpy(ak.fill_none(ak.pad_none(akr[k], L, clip=True), 0))
|
|
286
|
+
for k in fields
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
cells.append(
|
|
290
|
+
Cell(
|
|
291
|
+
"records",
|
|
292
|
+
"to_padded(dict)",
|
|
293
|
+
f"{sh} L={L}",
|
|
294
|
+
_ak_padded_dict,
|
|
295
|
+
lambda: rr.to_padded(0, length=L),
|
|
296
|
+
)
|
|
297
|
+
)
|
|
298
|
+
return cells
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _r2_buffers(n_outer: int, mid_low: int, mid_high: int, in_low: int, in_high: int):
|
|
302
|
+
"""Return (data, o0, o1, outer_counts, inner_lengths) for an R=2 structure."""
|
|
303
|
+
rng = np.random.default_rng(0)
|
|
304
|
+
outer_counts = rng.integers(mid_low, mid_high + 1, size=n_outer).astype(np.int64)
|
|
305
|
+
n_mid = int(outer_counts.sum())
|
|
306
|
+
inner_lengths = rng.integers(in_low, in_high + 1, size=n_mid).astype(np.int64)
|
|
307
|
+
total = int(inner_lengths.sum())
|
|
308
|
+
data = np.arange(total, dtype=np.int64)
|
|
309
|
+
o0 = np.concatenate([[0], np.cumsum(outer_counts)]).astype(np.int64)
|
|
310
|
+
o1 = np.concatenate([[0], np.cumsum(inner_lengths)]).astype(np.int64)
|
|
311
|
+
return data, o0, o1, outer_counts, inner_lengths
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _ak_r2(o0, o1, data):
|
|
315
|
+
return ak.Array(
|
|
316
|
+
ak.contents.ListOffsetArray(
|
|
317
|
+
ak.index.Index64(np.asarray(o0, np.int64)),
|
|
318
|
+
ak.contents.ListOffsetArray(
|
|
319
|
+
ak.index.Index64(np.asarray(o1, np.int64)),
|
|
320
|
+
ak.contents.NumpyArray(np.asarray(data)),
|
|
321
|
+
),
|
|
322
|
+
)
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def nested_cells() -> list[Cell]:
|
|
327
|
+
cells: list[Cell] = []
|
|
328
|
+
# flat-variant-windows-like: 64 outer groups, ~variants in [1,30], ~window in [1,20].
|
|
329
|
+
data, o0, o1, outer_counts, inner_lengths = _r2_buffers(64, 1, 30, 1, 20)
|
|
330
|
+
shape = (64, None, None)
|
|
331
|
+
sh = "64x~1-30x~1-20 i64"
|
|
332
|
+
|
|
333
|
+
cells.append(
|
|
334
|
+
Cell(
|
|
335
|
+
"nested",
|
|
336
|
+
"construct",
|
|
337
|
+
sh,
|
|
338
|
+
lambda: _ak_r2(o0, o1, data),
|
|
339
|
+
lambda: RustRagged.from_offsets(data, shape, [o0, o1]),
|
|
340
|
+
)
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
akx = _ak_r2(o0, o1, data)
|
|
344
|
+
rx = RustRagged.from_offsets(data, shape, [o0, o1])
|
|
345
|
+
|
|
346
|
+
cells.append(Cell("nested", "index[int]", sh, lambda: akx[7], lambda: rx[7]))
|
|
347
|
+
cells.append(
|
|
348
|
+
Cell("nested", "index[slice]", sh, lambda: akx[8:40], lambda: rx[8:40])
|
|
349
|
+
)
|
|
350
|
+
# rag[:, k]: k-th middle of each outer group. Use k=0 so every group is non-empty
|
|
351
|
+
# (outer_counts >= 1 by construction).
|
|
352
|
+
cells.append(Cell("nested", "index[:,k]", sh, lambda: akx[:, 0], lambda: rx[:, 0]))
|
|
353
|
+
cells.append(
|
|
354
|
+
Cell("nested", "index[:,a:b]", sh, lambda: akx[:, 0:2], lambda: rx[:, 0:2])
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
# rag[:, mask]: rust takes a flat boolean over the global middle axis; awkward takes
|
|
358
|
+
# the same mask regrouped under the outer counts (a ragged boolean of matching shape).
|
|
359
|
+
n_mid = int(outer_counts.sum())
|
|
360
|
+
flat_mask = np.arange(n_mid) % 2 == 0
|
|
361
|
+
ak_mask = ak.unflatten(flat_mask, list(outer_counts))
|
|
362
|
+
cells.append(
|
|
363
|
+
Cell(
|
|
364
|
+
"nested",
|
|
365
|
+
"index[:,mask]",
|
|
366
|
+
sh,
|
|
367
|
+
lambda: akx[ak_mask],
|
|
368
|
+
lambda: rx[:, flat_mask],
|
|
369
|
+
)
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
cells.append(
|
|
373
|
+
Cell(
|
|
374
|
+
"nested", "to_packed", sh, lambda: ak.to_packed(akx), lambda: rx.to_packed()
|
|
375
|
+
)
|
|
376
|
+
)
|
|
377
|
+
|
|
378
|
+
# Dense both axes: M = max middles per group, K = max inner length.
|
|
379
|
+
M = int(outer_counts.max())
|
|
380
|
+
K = int(inner_lengths.max()) if inner_lengths.size else 0
|
|
381
|
+
|
|
382
|
+
def _ak_dense():
|
|
383
|
+
padded_inner = ak.fill_none(ak.pad_none(akx, K, axis=2, clip=True), 0)
|
|
384
|
+
padded_outer = ak.pad_none(padded_inner, M, axis=1, clip=True)
|
|
385
|
+
# fill_none with a length-K array fills axis=1 None slots; to_numpy returns a masked
|
|
386
|
+
# array because awkward retains the option type — call .filled(0) to materialise zeros.
|
|
387
|
+
fill_val = np.zeros(K, dtype=np.int64)
|
|
388
|
+
return ak.to_numpy(
|
|
389
|
+
ak.fill_none(padded_outer, fill_val), allow_missing=True
|
|
390
|
+
).filled(0)
|
|
391
|
+
|
|
392
|
+
cells.append(
|
|
393
|
+
Cell(
|
|
394
|
+
"nested",
|
|
395
|
+
"to_padded(both)",
|
|
396
|
+
f"{sh} M={M},K={K}",
|
|
397
|
+
_ak_dense,
|
|
398
|
+
lambda: rx.to_padded(0, axis=None),
|
|
399
|
+
)
|
|
400
|
+
)
|
|
401
|
+
return cells
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def string_cells() -> list[Cell]:
|
|
405
|
+
cells: list[Cell] = []
|
|
406
|
+
# 8000 short opaque strings (alleles), lengths in [1, 8].
|
|
407
|
+
rng = np.random.default_rng(0)
|
|
408
|
+
n = 8000
|
|
409
|
+
lengths = rng.integers(1, 9, size=n).astype(np.int64)
|
|
410
|
+
total = int(lengths.sum())
|
|
411
|
+
data = _BASES[rng.integers(0, 4, size=total)] # (total,) S1
|
|
412
|
+
offsets = np.concatenate([[0], np.cumsum(lengths)]).astype(np.int64)
|
|
413
|
+
|
|
414
|
+
# rust: opaque-string Ragged (shape (n,), dtype 'S') -> chars (n, ~length) and back.
|
|
415
|
+
r_str = RustRagged.from_lengths(data, lengths) # opaque string by default
|
|
416
|
+
r_chars = r_str.to_chars()
|
|
417
|
+
|
|
418
|
+
# awkward analogue: a list-of-S1 (chars) array; "to chars" = view as char list,
|
|
419
|
+
# "to strings" = join back to bytestrings. Build the char-list oracle once.
|
|
420
|
+
ak_chars = _ak_r1(offsets, data) # ListOffsetArray over S1 == char lists
|
|
421
|
+
|
|
422
|
+
# to_chars: rust retag vs awkward producing the char-list view.
|
|
423
|
+
cells.append(
|
|
424
|
+
Cell(
|
|
425
|
+
"string",
|
|
426
|
+
"to_chars",
|
|
427
|
+
f"{n}x~1-8",
|
|
428
|
+
lambda: ak.copy(ak_chars),
|
|
429
|
+
lambda: r_str.to_chars(),
|
|
430
|
+
eq=lambda a, b: True,
|
|
431
|
+
)
|
|
432
|
+
) # structural shapes differ across backends; time-only
|
|
433
|
+
# to_strings: rust retag vs awkward joining char lists into bytestrings.
|
|
434
|
+
cells.append(
|
|
435
|
+
Cell(
|
|
436
|
+
"string",
|
|
437
|
+
"to_strings",
|
|
438
|
+
f"{n}x~1-8",
|
|
439
|
+
lambda: ak.copy(ak_chars),
|
|
440
|
+
lambda: r_chars.to_strings(),
|
|
441
|
+
eq=lambda a, b: True,
|
|
442
|
+
)
|
|
443
|
+
)
|
|
444
|
+
return cells
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def build_cells(args: argparse.Namespace) -> list[Cell]:
|
|
448
|
+
"""Assemble the cell list. Extended in later tasks."""
|
|
449
|
+
cells: list[Cell] = []
|
|
450
|
+
if args.only in ("all", "single"):
|
|
451
|
+
cells += single_cells()
|
|
452
|
+
if args.only in ("all", "records"):
|
|
453
|
+
cells += record_cells()
|
|
454
|
+
if args.only in ("all", "nested"):
|
|
455
|
+
cells += nested_cells()
|
|
456
|
+
if args.only in ("all", "string"):
|
|
457
|
+
cells += string_cells()
|
|
458
|
+
return cells
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def main(argv: "list[str] | None" = None) -> int:
|
|
462
|
+
p = argparse.ArgumentParser(
|
|
463
|
+
description="rust-native vs awkward Ragged throughput gate"
|
|
464
|
+
)
|
|
465
|
+
p.add_argument("--tol", type=float, default=0.10)
|
|
466
|
+
p.add_argument(
|
|
467
|
+
"--only",
|
|
468
|
+
choices=["single", "records", "nested", "string", "all"],
|
|
469
|
+
default="all",
|
|
470
|
+
)
|
|
471
|
+
p.add_argument("--repeats", type=int, default=7)
|
|
472
|
+
args = p.parse_args(argv)
|
|
473
|
+
cells = build_cells(args)
|
|
474
|
+
if not cells:
|
|
475
|
+
print(f"no cells for --only {args.only}")
|
|
476
|
+
return 0
|
|
477
|
+
return run_cells(cells, args.tol, args.repeats)
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
if __name__ == "__main__":
|
|
481
|
+
raise SystemExit(main())
|