openomicsbench 2.0.0__tar.gz → 2.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- openomicsbench-2.2.0/CITATION.cff +16 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/MANIFEST.in +3 -0
- {openomicsbench-2.0.0/src/openomicsbench.egg-info → openomicsbench-2.2.0}/PKG-INFO +62 -1
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/README.md +61 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/pyproject.toml +1 -1
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/setup.py +6 -2
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/__init__.py +1 -1
- openomicsbench-2.2.0/src/omicsbench/bundle.py +193 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/cli.py +62 -0
- openomicsbench-2.2.0/src/omicsbench/matrix.py +87 -0
- openomicsbench-2.2.0/src/omicsbench/regression.py +126 -0
- openomicsbench-2.2.0/src/omicsbench/suite.py +268 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0/src/openomicsbench.egg-info}/PKG-INFO +62 -1
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/openomicsbench.egg-info/SOURCES.txt +10 -1
- openomicsbench-2.2.0/tests/test_bundle.py +72 -0
- openomicsbench-2.2.0/tests/test_matrix.py +63 -0
- openomicsbench-2.2.0/tests/test_regression.py +79 -0
- openomicsbench-2.2.0/tests/test_suite.py +89 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/LICENSE +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/LICENSE-METADATA +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/NOTICE +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/LICENSE-DATA.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/expected/baseline.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/expected/size-fidelity.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/nano/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/fixture-001/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-001/LICENSE-DATA.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-001/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-001/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-002/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-003/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-004/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-005/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-006/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-007/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-008/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-009/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-010/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-011/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-012/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/README.md +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/attribution.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/expected/deseq2.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/expected/reference-effects.tsv.gz +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/expected/source-counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/pocket/counts.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/provenance/transform.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/reference.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/rights.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/rnaseq/rnaseq-013/samples.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-001/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-001/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-001/nano/sequences.fasta +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-002/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-002/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-002/nano/sequences.fasta +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-003/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-003/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-003/nano/sequences.fasta +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-004/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-004/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-004/nano/reads_R1.fastq +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-004/nano/reads_R2.fastq +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-005/expected/validation.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-005/expected/variants.tsv +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-005/manifest.json +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-005/nano/reads_R1.fastq +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-005/nano/reads_R2.fastq +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/datasets/sequences/sequence-005/nano/reference.fasta +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/setup.cfg +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/__main__.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/assays.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/cache.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/compare.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/download.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/expression_atlas.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/hashing.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/models.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/proteins.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/registry.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/rnaseq.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/sequence_analysis.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/sequence_cli.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/sequence_tools.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/sequences.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/omicsbench/validate.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/openomicsbench.egg-info/dependency_links.txt +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/openomicsbench.egg-info/entry_points.txt +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/openomicsbench.egg-info/requires.txt +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/src/openomicsbench.egg-info/top_level.txt +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/tests/test_assays.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/tests/test_core.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/tests/test_proteins.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/tests/test_sequence_fixtures.py +0 -0
- {openomicsbench-2.0.0 → openomicsbench-2.2.0}/tests/test_sequences.py +0 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
message: "Please cite the archived release corresponding to the version you used."
|
|
3
|
+
title: "OpenOmicsBench"
|
|
4
|
+
type: software
|
|
5
|
+
version: "2.2.0"
|
|
6
|
+
license: Apache-2.0
|
|
7
|
+
doi: "10.5281/zenodo.22551734"
|
|
8
|
+
authors:
|
|
9
|
+
- family-names: "Patni"
|
|
10
|
+
given-names: "Vivaan"
|
|
11
|
+
orcid: "https://orcid.org/0009-0005-1859-5107"
|
|
12
|
+
repository-code: "https://github.com/vxxqv/openomicsbench"
|
|
13
|
+
identifiers:
|
|
14
|
+
- type: doi
|
|
15
|
+
value: "10.5281/zenodo.22551734"
|
|
16
|
+
description: "Zenodo archive containing all released versions"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: openomicsbench
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.2.0
|
|
4
4
|
Summary: Traceable omics benchmarks and deterministic FASTA and FASTQ tools
|
|
5
5
|
Author: Vivaan Patni
|
|
6
6
|
Maintainer: Vivaan Patni
|
|
@@ -35,6 +35,8 @@ Dynamic: license-file
|
|
|
35
35
|
|
|
36
36
|
OpenOmicsBench is a compact benchmark collection and local sequence toolkit for bioinformatics software testing, method checks and teaching. Version 2 keeps the 12 certified bulk RNA-seq objects from version 1 and adds strict FASTA and FASTQ handling, DNA, RNA and protein support, paired-read checks, preprocessing, sequence QC and five deterministic sequence benchmarks.
|
|
37
37
|
|
|
38
|
+
Version 2.2 adds multi-method benchmark matrices, report-to-report regression gates and portable benchmark bundles. Suite reports can be written as JSON, Markdown, JUnit XML, CSV or a self-contained HTML page.
|
|
39
|
+
|
|
38
40
|
The package runs offline after installation. Sequence files stay on the local computer. The built-in tools cover inspection and lightweight preprocessing; they do not claim to replace aligners, variant callers, taxonomic classifiers or assay-specific statistical workflows.
|
|
39
41
|
|
|
40
42
|
## Install
|
|
@@ -104,6 +106,65 @@ The FASTA validator follows the nucleotide symbol expectations described by [NCB
|
|
|
104
106
|
|
|
105
107
|
The full command reference is in [docs/sequence-tools.md](docs/sequence-tools.md).
|
|
106
108
|
|
|
109
|
+
## Benchmark suites
|
|
110
|
+
|
|
111
|
+
Validate the complete installed collection and save reports that can be attached to a CI run:
|
|
112
|
+
|
|
113
|
+
```sh
|
|
114
|
+
omicsbench suite validate --json reports/validation.json --markdown reports/validation.md --junit reports/validation.xml
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
The selection can be narrowed by assay or by repeating `--id`:
|
|
118
|
+
|
|
119
|
+
```sh
|
|
120
|
+
omicsbench suite validate --assay bulk_rna_seq
|
|
121
|
+
omicsbench suite validate --id rnaseq-002 --id sequence-005
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
To test a differential-expression pipeline across several biological objects, place each result in a directory under its benchmark ID. CSV and TSV are accepted, with optional gzip compression:
|
|
125
|
+
|
|
126
|
+
```text
|
|
127
|
+
results/
|
|
128
|
+
rnaseq-002.tsv.gz
|
|
129
|
+
rnaseq-003.csv
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
Then run:
|
|
133
|
+
|
|
134
|
+
```sh
|
|
135
|
+
omicsbench suite compare results --id rnaseq-002 --id rnaseq-003 --junit reports/comparison.xml
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Compare several tools or parameter sets in one matrix:
|
|
139
|
+
|
|
140
|
+
```sh
|
|
141
|
+
omicsbench suite matrix \
|
|
142
|
+
--method deseq2=results/deseq2 \
|
|
143
|
+
--method edger=results/edger \
|
|
144
|
+
--json reports/matrix.json \
|
|
145
|
+
--csv reports/matrix.csv \
|
|
146
|
+
--html reports/matrix.html
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
Use a previously accepted report as a regression baseline:
|
|
150
|
+
|
|
151
|
+
```sh
|
|
152
|
+
omicsbench suite regress accepted.json candidate.json --absolute-tolerance 0.01 --junit reports/regression.xml
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
Omit `--id` to require results for all 12 comparable RNA-seq objects. A completed suite exits with status 0 when every check passes, status 2 when a benchmark or regression gate fails, and status 1 for invalid input. Existing report files are protected unless `--force` is supplied. The [suite reference](docs/benchmark-suites.md) describes the file convention and report fields.
|
|
156
|
+
|
|
157
|
+
## Portable bundles
|
|
158
|
+
|
|
159
|
+
Create a deterministic ZIP containing selected benchmarks, their manifests, declared files and the project licence material:
|
|
160
|
+
|
|
161
|
+
```sh
|
|
162
|
+
omicsbench bundle create rna-and-dna.zip --id rnaseq-002 --id sequence-005
|
|
163
|
+
omicsbench bundle verify rna-and-dna.zip
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
Bundle verification checks the complete file inventory, byte counts, SHA-256 digests and embedded dataset manifests without extracting the archive. Recreating the same selection with the same package version produces the same archive bytes. See [docs/portable-bundles.md](docs/portable-bundles.md).
|
|
167
|
+
|
|
107
168
|
## Assay profiles
|
|
108
169
|
|
|
109
170
|
The assay profiles explain what the local tools can check and where a dedicated workflow becomes necessary:
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
OpenOmicsBench is a compact benchmark collection and local sequence toolkit for bioinformatics software testing, method checks and teaching. Version 2 keeps the 12 certified bulk RNA-seq objects from version 1 and adds strict FASTA and FASTQ handling, DNA, RNA and protein support, paired-read checks, preprocessing, sequence QC and five deterministic sequence benchmarks.
|
|
4
4
|
|
|
5
|
+
Version 2.2 adds multi-method benchmark matrices, report-to-report regression gates and portable benchmark bundles. Suite reports can be written as JSON, Markdown, JUnit XML, CSV or a self-contained HTML page.
|
|
6
|
+
|
|
5
7
|
The package runs offline after installation. Sequence files stay on the local computer. The built-in tools cover inspection and lightweight preprocessing; they do not claim to replace aligners, variant callers, taxonomic classifiers or assay-specific statistical workflows.
|
|
6
8
|
|
|
7
9
|
## Install
|
|
@@ -71,6 +73,65 @@ The FASTA validator follows the nucleotide symbol expectations described by [NCB
|
|
|
71
73
|
|
|
72
74
|
The full command reference is in [docs/sequence-tools.md](docs/sequence-tools.md).
|
|
73
75
|
|
|
76
|
+
## Benchmark suites
|
|
77
|
+
|
|
78
|
+
Validate the complete installed collection and save reports that can be attached to a CI run:
|
|
79
|
+
|
|
80
|
+
```sh
|
|
81
|
+
omicsbench suite validate --json reports/validation.json --markdown reports/validation.md --junit reports/validation.xml
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
The selection can be narrowed by assay or by repeating `--id`:
|
|
85
|
+
|
|
86
|
+
```sh
|
|
87
|
+
omicsbench suite validate --assay bulk_rna_seq
|
|
88
|
+
omicsbench suite validate --id rnaseq-002 --id sequence-005
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
To test a differential-expression pipeline across several biological objects, place each result in a directory under its benchmark ID. CSV and TSV are accepted, with optional gzip compression:
|
|
92
|
+
|
|
93
|
+
```text
|
|
94
|
+
results/
|
|
95
|
+
rnaseq-002.tsv.gz
|
|
96
|
+
rnaseq-003.csv
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Then run:
|
|
100
|
+
|
|
101
|
+
```sh
|
|
102
|
+
omicsbench suite compare results --id rnaseq-002 --id rnaseq-003 --junit reports/comparison.xml
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Compare several tools or parameter sets in one matrix:
|
|
106
|
+
|
|
107
|
+
```sh
|
|
108
|
+
omicsbench suite matrix \
|
|
109
|
+
--method deseq2=results/deseq2 \
|
|
110
|
+
--method edger=results/edger \
|
|
111
|
+
--json reports/matrix.json \
|
|
112
|
+
--csv reports/matrix.csv \
|
|
113
|
+
--html reports/matrix.html
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Use a previously accepted report as a regression baseline:
|
|
117
|
+
|
|
118
|
+
```sh
|
|
119
|
+
omicsbench suite regress accepted.json candidate.json --absolute-tolerance 0.01 --junit reports/regression.xml
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
Omit `--id` to require results for all 12 comparable RNA-seq objects. A completed suite exits with status 0 when every check passes, status 2 when a benchmark or regression gate fails, and status 1 for invalid input. Existing report files are protected unless `--force` is supplied. The [suite reference](docs/benchmark-suites.md) describes the file convention and report fields.
|
|
123
|
+
|
|
124
|
+
## Portable bundles
|
|
125
|
+
|
|
126
|
+
Create a deterministic ZIP containing selected benchmarks, their manifests, declared files and the project licence material:
|
|
127
|
+
|
|
128
|
+
```sh
|
|
129
|
+
omicsbench bundle create rna-and-dna.zip --id rnaseq-002 --id sequence-005
|
|
130
|
+
omicsbench bundle verify rna-and-dna.zip
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
Bundle verification checks the complete file inventory, byte counts, SHA-256 digests and embedded dataset manifests without extracting the archive. Recreating the same selection with the same package version produces the same archive bytes. See [docs/portable-bundles.md](docs/portable-bundles.md).
|
|
134
|
+
|
|
74
135
|
## Assay profiles
|
|
75
136
|
|
|
76
137
|
The assay profiles explain what the local tools can check and where a dedicated workflow becomes necessary:
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "openomicsbench"
|
|
7
|
-
version = "2.
|
|
7
|
+
version = "2.2.0"
|
|
8
8
|
description = "Traceable omics benchmarks and deterministic FASTA and FASTQ tools"
|
|
9
9
|
readme = {file = "README.md", content-type = "text/markdown"}
|
|
10
10
|
license = "Apache-2.0"
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
from pathlib import Path
|
|
2
|
-
from shutil import copytree
|
|
2
|
+
from shutil import copy2, copytree
|
|
3
3
|
|
|
4
4
|
from setuptools import setup
|
|
5
5
|
from setuptools.command.build_py import build_py as setuptools_build_py
|
|
@@ -9,8 +9,12 @@ class build_py(setuptools_build_py):
|
|
|
9
9
|
def run(self):
|
|
10
10
|
super().run()
|
|
11
11
|
source = Path(__file__).parent / "datasets"
|
|
12
|
-
|
|
12
|
+
collection = Path(self.build_lib) / "omicsbench" / "_collection"
|
|
13
|
+
destination = collection / "datasets"
|
|
13
14
|
copytree(source, destination, dirs_exist_ok=True)
|
|
15
|
+
collection.mkdir(parents=True, exist_ok=True)
|
|
16
|
+
for name in ("CITATION.cff", "LICENSE", "LICENSE-METADATA", "NOTICE"):
|
|
17
|
+
copy2(Path(__file__).parent / name, collection / name)
|
|
14
18
|
|
|
15
19
|
|
|
16
20
|
setup(cmdclass={"build_py": build_py})
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
"""Discovery, validation and local analysis of compact omics benchmarks."""
|
|
2
|
-
__version__ = "2.
|
|
2
|
+
__version__ = "2.2.0"
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""Create and verify portable, deterministic benchmark bundles."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import zipfile
|
|
8
|
+
from pathlib import Path, PurePosixPath
|
|
9
|
+
|
|
10
|
+
from . import __version__
|
|
11
|
+
from .models import Dataset
|
|
12
|
+
from .registry import registry
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
BUNDLE_FORMAT = "openomicsbench-bundle-v1"
|
|
16
|
+
SUPPORT_FILES = ("CITATION.cff", "LICENSE", "LICENSE-METADATA", "NOTICE")
|
|
17
|
+
FIXED_TIME = (1980, 1, 1, 0, 0, 0)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _sha256(data: bytes) -> str:
|
|
21
|
+
return hashlib.sha256(data).hexdigest()
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _safe_name(value: str) -> str:
|
|
25
|
+
if "\\" in value or "\x00" in value:
|
|
26
|
+
raise ValueError(f"unsafe bundle path: {value}")
|
|
27
|
+
path = PurePosixPath(value)
|
|
28
|
+
if path.is_absolute() or not path.parts or any(part in {"", ".", ".."} for part in path.parts):
|
|
29
|
+
raise ValueError(f"unsafe bundle path: {value}")
|
|
30
|
+
return path.as_posix()
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _selected(root: Path, dataset_ids: list[str] | None, assay: str | None):
|
|
34
|
+
records = registry(root)
|
|
35
|
+
requested = list(dict.fromkeys(dataset_ids or []))
|
|
36
|
+
unknown = [dataset_id for dataset_id in requested if dataset_id not in records]
|
|
37
|
+
if unknown:
|
|
38
|
+
raise ValueError(f"unknown dataset IDs: {', '.join(unknown)}")
|
|
39
|
+
selected = [(model, folder) for model, folder in records.values() if not assay or model.assay == assay]
|
|
40
|
+
if requested:
|
|
41
|
+
wanted = set(requested)
|
|
42
|
+
selected = [(model, folder) for model, folder in selected if model.id in wanted]
|
|
43
|
+
excluded = [dataset_id for dataset_id in requested if dataset_id not in {model.id for model, _ in selected}]
|
|
44
|
+
if excluded:
|
|
45
|
+
raise ValueError(f"dataset IDs do not match assay {assay}: {', '.join(excluded)}")
|
|
46
|
+
if not selected:
|
|
47
|
+
raise ValueError("no datasets matched the bundle selection")
|
|
48
|
+
return selected
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _zip_info(name: str) -> zipfile.ZipInfo:
|
|
52
|
+
info = zipfile.ZipInfo(_safe_name(name), FIXED_TIME)
|
|
53
|
+
info.compress_type = zipfile.ZIP_DEFLATED
|
|
54
|
+
info.create_system = 3
|
|
55
|
+
info.external_attr = 0o100644 << 16
|
|
56
|
+
return info
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def create_bundle(
|
|
60
|
+
root: Path,
|
|
61
|
+
destination: Path,
|
|
62
|
+
dataset_ids: list[str] | None = None,
|
|
63
|
+
assay: str | None = None,
|
|
64
|
+
force: bool = False,
|
|
65
|
+
) -> dict:
|
|
66
|
+
root = Path(root).resolve()
|
|
67
|
+
destination = Path(destination)
|
|
68
|
+
if destination.exists() and not force:
|
|
69
|
+
raise ValueError(f"{destination}: output already exists; use --force to replace it")
|
|
70
|
+
selected = _selected(root, dataset_ids, assay)
|
|
71
|
+
files: dict[str, bytes] = {}
|
|
72
|
+
for relative in SUPPORT_FILES:
|
|
73
|
+
path = root / relative
|
|
74
|
+
if not path.is_file():
|
|
75
|
+
raise ValueError(f"{path}: required bundle support file is missing")
|
|
76
|
+
files[relative] = path.read_bytes()
|
|
77
|
+
for model, folder in selected:
|
|
78
|
+
prefix = folder.resolve().relative_to(root).as_posix()
|
|
79
|
+
manifest_path = (folder / "manifest.json").resolve()
|
|
80
|
+
included = [manifest_path, *(folder / record.path for record in model.files)]
|
|
81
|
+
for path in included:
|
|
82
|
+
resolved = path.resolve()
|
|
83
|
+
try:
|
|
84
|
+
relative = resolved.relative_to(root).as_posix()
|
|
85
|
+
except ValueError:
|
|
86
|
+
raise ValueError(f"{path}: bundle input escapes the collection root") from None
|
|
87
|
+
if not resolved.is_file():
|
|
88
|
+
raise ValueError(f"{relative}: declared bundle file is missing")
|
|
89
|
+
data = resolved.read_bytes()
|
|
90
|
+
if resolved != manifest_path:
|
|
91
|
+
record = next(item for item in model.files if (folder / item.path).resolve() == resolved)
|
|
92
|
+
if len(data) != record.bytes or _sha256(data) != record.sha256:
|
|
93
|
+
raise ValueError(f"{relative}: declared size or SHA-256 does not match")
|
|
94
|
+
files[_safe_name(relative)] = data
|
|
95
|
+
if f"{prefix}/manifest.json" not in files:
|
|
96
|
+
raise ValueError(f"{model.id}: manifest was not added to the bundle")
|
|
97
|
+
|
|
98
|
+
file_records = [
|
|
99
|
+
{"path": name, "bytes": len(files[name]), "sha256": _sha256(files[name])}
|
|
100
|
+
for name in sorted(files)
|
|
101
|
+
]
|
|
102
|
+
manifest = {
|
|
103
|
+
"format": BUNDLE_FORMAT,
|
|
104
|
+
"software_version": __version__,
|
|
105
|
+
"selection": {"assay": assay, "ids": [model.id for model, _ in selected] if dataset_ids else []},
|
|
106
|
+
"datasets": [model.id for model, _ in selected],
|
|
107
|
+
"files": file_records,
|
|
108
|
+
}
|
|
109
|
+
manifest_data = (json.dumps(manifest, indent=2, sort_keys=True) + "\n").encode("utf-8")
|
|
110
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
111
|
+
temporary = destination.with_name(destination.name + ".tmp")
|
|
112
|
+
try:
|
|
113
|
+
with zipfile.ZipFile(temporary, "w", compression=zipfile.ZIP_DEFLATED, compresslevel=9) as archive:
|
|
114
|
+
archive.writestr(_zip_info("bundle.json"), manifest_data, compresslevel=9)
|
|
115
|
+
for name in sorted(files):
|
|
116
|
+
archive.writestr(_zip_info(name), files[name], compresslevel=9)
|
|
117
|
+
os.replace(temporary, destination)
|
|
118
|
+
finally:
|
|
119
|
+
temporary.unlink(missing_ok=True)
|
|
120
|
+
result = verify_bundle(destination)
|
|
121
|
+
result["path"] = str(destination)
|
|
122
|
+
return result
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def verify_bundle(path: Path) -> dict:
|
|
126
|
+
path = Path(path)
|
|
127
|
+
if not path.is_file():
|
|
128
|
+
raise ValueError(f"{path}: bundle does not exist")
|
|
129
|
+
try:
|
|
130
|
+
with zipfile.ZipFile(path) as archive:
|
|
131
|
+
infos = [info for info in archive.infolist() if not info.is_dir()]
|
|
132
|
+
names = [_safe_name(info.filename) for info in infos]
|
|
133
|
+
if len(names) != len(set(names)):
|
|
134
|
+
raise ValueError(f"{path}: bundle contains duplicate paths")
|
|
135
|
+
if names.count("bundle.json") != 1:
|
|
136
|
+
raise ValueError(f"{path}: bundle.json is missing or duplicated")
|
|
137
|
+
for info in infos:
|
|
138
|
+
if ((info.external_attr >> 16) & 0o170000) == 0o120000:
|
|
139
|
+
raise ValueError(f"{path}: symbolic links are not permitted")
|
|
140
|
+
try:
|
|
141
|
+
manifest = json.loads(archive.read("bundle.json").decode("utf-8"))
|
|
142
|
+
except (json.JSONDecodeError, UnicodeDecodeError) as error:
|
|
143
|
+
raise ValueError(f"{path}: invalid bundle.json: {error}") from None
|
|
144
|
+
if manifest.get("format") != BUNDLE_FORMAT:
|
|
145
|
+
raise ValueError(f"{path}: unsupported bundle format")
|
|
146
|
+
datasets = manifest.get("datasets")
|
|
147
|
+
records = manifest.get("files")
|
|
148
|
+
if not isinstance(datasets, list) or not datasets or len(datasets) != len(set(datasets)):
|
|
149
|
+
raise ValueError(f"{path}: dataset list is empty or duplicated")
|
|
150
|
+
if not isinstance(records, list):
|
|
151
|
+
raise ValueError(f"{path}: file inventory is missing")
|
|
152
|
+
declared = {}
|
|
153
|
+
for record in records:
|
|
154
|
+
if not isinstance(record, dict):
|
|
155
|
+
raise ValueError(f"{path}: invalid file inventory record")
|
|
156
|
+
name = _safe_name(record.get("path", ""))
|
|
157
|
+
if name == "bundle.json" or name in declared:
|
|
158
|
+
raise ValueError(f"{path}: duplicate or reserved inventory path {name}")
|
|
159
|
+
if not isinstance(record.get("bytes"), int) or record["bytes"] < 0:
|
|
160
|
+
raise ValueError(f"{path}: invalid byte count for {name}")
|
|
161
|
+
sha256 = record.get("sha256", "")
|
|
162
|
+
if not isinstance(sha256, str) or len(sha256) != 64 or any(char not in "0123456789abcdef" for char in sha256):
|
|
163
|
+
raise ValueError(f"{path}: invalid SHA-256 for {name}")
|
|
164
|
+
declared[name] = record
|
|
165
|
+
expected = set(declared) | {"bundle.json"}
|
|
166
|
+
if set(names) != expected:
|
|
167
|
+
missing = sorted(expected - set(names))
|
|
168
|
+
extra = sorted(set(names) - expected)
|
|
169
|
+
raise ValueError(f"{path}: archive inventory differs: missing={missing}, extra={extra}")
|
|
170
|
+
for name, record in declared.items():
|
|
171
|
+
data = archive.read(name)
|
|
172
|
+
if len(data) != record["bytes"] or _sha256(data) != record["sha256"]:
|
|
173
|
+
raise ValueError(f"{path}: size or SHA-256 differs for {name}")
|
|
174
|
+
manifest_ids = []
|
|
175
|
+
for name in sorted(declared):
|
|
176
|
+
if name.startswith("datasets/") and name.endswith("/manifest.json"):
|
|
177
|
+
model = Dataset.model_validate_json(archive.read(name))
|
|
178
|
+
manifest_ids.append(model.id)
|
|
179
|
+
if sorted(manifest_ids) != sorted(datasets):
|
|
180
|
+
raise ValueError(f"{path}: dataset manifests differ from the dataset list")
|
|
181
|
+
except zipfile.BadZipFile as error:
|
|
182
|
+
raise ValueError(f"{path}: invalid ZIP archive: {error}") from None
|
|
183
|
+
data = path.read_bytes()
|
|
184
|
+
return {
|
|
185
|
+
"status": "pass",
|
|
186
|
+
"format": BUNDLE_FORMAT,
|
|
187
|
+
"software_version": manifest.get("software_version"),
|
|
188
|
+
"datasets": datasets,
|
|
189
|
+
"dataset_count": len(datasets),
|
|
190
|
+
"file_count": len(declared),
|
|
191
|
+
"archive_bytes": len(data),
|
|
192
|
+
"sha256": _sha256(data),
|
|
193
|
+
}
|
|
@@ -11,6 +11,10 @@ from .cache import get, verify_cache
|
|
|
11
11
|
from .compare import compare
|
|
12
12
|
from .sequence_cli import add_sequence_parser, run_sequence_command
|
|
13
13
|
from .assays import build_plan, get_profile, list_profiles
|
|
14
|
+
from .suite import compare_suite, validate_suite, write_reports
|
|
15
|
+
from .matrix import compare_matrix, parse_method
|
|
16
|
+
from .regression import compare_reports
|
|
17
|
+
from .bundle import create_bundle, verify_bundle
|
|
14
18
|
from .validate import validate
|
|
15
19
|
|
|
16
20
|
def main():
|
|
@@ -43,10 +47,48 @@ def main():
|
|
|
43
47
|
assay_plan.add_argument("input", type=Path, nargs="+")
|
|
44
48
|
assay_plan.add_argument("--output-directory", type=Path, default=Path("omicsbench-output"))
|
|
45
49
|
assay_plan.add_argument("--adapter", action="append", default=[])
|
|
50
|
+
suite = sub.add_parser("suite", help="Run collection-wide checks and write CI reports.")
|
|
51
|
+
suite_sub = suite.add_subparsers(dest="suite_command", required=True)
|
|
52
|
+
suite_validate = suite_sub.add_parser("validate", help="Validate several bundled benchmarks in one run.")
|
|
53
|
+
suite_validate.add_argument("--assay", choices=["bulk_rna_seq", "sequence_dna", "sequence_rna", "sequence_protein", "short_read_dna", "whole_genome_dna_seq"])
|
|
54
|
+
suite_compare = suite_sub.add_parser("compare", help="Compare a directory of differential-expression results.")
|
|
55
|
+
suite_compare.add_argument("results", type=Path, help="Directory containing files named <dataset-id>.csv or .tsv, optionally gzip-compressed.")
|
|
56
|
+
suite_compare.add_argument("--detail-limit", type=int, default=20, help="Maximum missing and unexpected gene examples per benchmark.")
|
|
57
|
+
suite_matrix = suite_sub.add_parser("matrix", help="Compare several analysis methods across the same benchmarks.")
|
|
58
|
+
suite_matrix.add_argument("--method", action="append", required=True, help="Method and result directory as NAME=PATH. Repeat for every method.")
|
|
59
|
+
suite_matrix.add_argument("--detail-limit", type=int, default=20, help="Maximum missing and unexpected gene examples per benchmark.")
|
|
60
|
+
suite_regress = suite_sub.add_parser("regress", help="Fail when a candidate suite report regresses from a baseline report.")
|
|
61
|
+
suite_regress.add_argument("baseline", type=Path, help="Previously accepted JSON suite report.")
|
|
62
|
+
suite_regress.add_argument("candidate", type=Path, help="Candidate JSON suite report to check.")
|
|
63
|
+
suite_regress.add_argument("--absolute-tolerance", type=float, default=0.0, help="Largest permitted absolute decrease in a tracked metric.")
|
|
64
|
+
suite_regress.add_argument("--allow-missing", action="store_true", help="Do not fail when a baseline case is absent from the candidate.")
|
|
65
|
+
for command in (suite_validate, suite_compare, suite_matrix):
|
|
66
|
+
command.add_argument("--id", action="append", default=[], help="Benchmark ID to include. Repeat to select several; omit to run all eligible benchmarks.")
|
|
67
|
+
for command in (suite_validate, suite_compare, suite_matrix, suite_regress):
|
|
68
|
+
command.add_argument("--json", type=Path, help="Write the complete report as JSON.")
|
|
69
|
+
command.add_argument("--markdown", type=Path, help="Write a concise Markdown report.")
|
|
70
|
+
command.add_argument("--junit", type=Path, help="Write a JUnit XML report for CI systems.")
|
|
71
|
+
command.add_argument("--csv", type=Path, help="Write a flat CSV report for analysis or plotting.")
|
|
72
|
+
command.add_argument("--html", type=Path, help="Write a self-contained HTML report.")
|
|
73
|
+
command.add_argument("--force", action="store_true", help="Replace existing report files.")
|
|
74
|
+
bundle = sub.add_parser("bundle", help="Create or verify a portable benchmark bundle.")
|
|
75
|
+
bundle_sub = bundle.add_subparsers(dest="bundle_command", required=True)
|
|
76
|
+
bundle_create = bundle_sub.add_parser("create", help="Create a deterministic ZIP containing selected benchmarks.")
|
|
77
|
+
bundle_create.add_argument("output", type=Path)
|
|
78
|
+
bundle_create.add_argument("--id", action="append", default=[], help="Benchmark ID to include. Repeat to select several.")
|
|
79
|
+
bundle_create.add_argument("--assay", choices=["bulk_rna_seq", "sequence_dna", "sequence_rna", "sequence_protein", "short_read_dna", "whole_genome_dna_seq"])
|
|
80
|
+
bundle_create.add_argument("--force", action="store_true", help="Replace an existing bundle.")
|
|
81
|
+
bundle_verify = bundle_sub.add_parser("verify", help="Verify a bundle inventory, hashes and dataset manifests.")
|
|
82
|
+
bundle_verify.add_argument("archive", type=Path)
|
|
46
83
|
args = p.parse_args()
|
|
47
84
|
try:
|
|
48
85
|
if args.command == "seq":
|
|
49
86
|
out = run_sequence_command(args)
|
|
87
|
+
elif args.command == "bundle":
|
|
88
|
+
if args.bundle_command == "create":
|
|
89
|
+
out = create_bundle(args.root, args.output, args.id, args.assay, args.force)
|
|
90
|
+
else:
|
|
91
|
+
out = verify_bundle(args.archive)
|
|
50
92
|
elif args.command == "assay":
|
|
51
93
|
if args.assay_command == "list":
|
|
52
94
|
out = list_profiles()
|
|
@@ -54,6 +96,24 @@ def main():
|
|
|
54
96
|
out = get_profile(args.profile)
|
|
55
97
|
else:
|
|
56
98
|
out = build_plan(args.profile, args.input, args.output_directory, args.adapter)
|
|
99
|
+
elif args.command == "suite":
|
|
100
|
+
if args.suite_command == "validate":
|
|
101
|
+
out = validate_suite(args.root, args.assay, args.id)
|
|
102
|
+
elif args.suite_command == "compare":
|
|
103
|
+
out = compare_suite(args.root, args.results, args.id, args.detail_limit)
|
|
104
|
+
elif args.suite_command == "matrix":
|
|
105
|
+
out = compare_matrix(args.root, [parse_method(value) for value in args.method], args.id, args.detail_limit)
|
|
106
|
+
else:
|
|
107
|
+
out = compare_reports(args.baseline, args.candidate, args.absolute_tolerance, args.allow_missing)
|
|
108
|
+
out["reports"] = write_reports(
|
|
109
|
+
out,
|
|
110
|
+
json_path=args.json,
|
|
111
|
+
markdown_path=args.markdown,
|
|
112
|
+
junit_path=args.junit,
|
|
113
|
+
force=args.force,
|
|
114
|
+
csv_path=args.csv,
|
|
115
|
+
html_path=args.html,
|
|
116
|
+
)
|
|
57
117
|
elif args.command == "list":
|
|
58
118
|
out = [{"id":m.id,"title":m.title,"kind":m.kind,"status":m.status,"rights":m.rights.status,"tiers":sorted({f.tier for f in m.files})} for m,_ in registry(args.root).values() if not args.assay or m.assay==args.assay]
|
|
59
119
|
elif args.command == "doctor":
|
|
@@ -75,6 +135,8 @@ def main():
|
|
|
75
135
|
print(json.dumps(out,indent=2,allow_nan=False))
|
|
76
136
|
if args.command == "compare" and out["status"] == "fail":
|
|
77
137
|
sys.exit(2)
|
|
138
|
+
if args.command == "suite" and out["summary"]["status"] == "fail":
|
|
139
|
+
sys.exit(2)
|
|
78
140
|
except (ValueError,OSError,KeyError) as exc:
|
|
79
141
|
if args.debug:
|
|
80
142
|
raise
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Compare several analysis methods across the RNA-seq benchmark collection."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from .suite import _summary, compare_suite
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
METHOD_NAME = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.-]{0,63}$")
|
|
11
|
+
SCORED_METRICS = ("spearman_logfc", "top_k_jaccard", "sign_concordance")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def parse_method(value: str) -> tuple[str, Path]:
|
|
15
|
+
if "=" not in value:
|
|
16
|
+
raise ValueError(f"{value}: method must use NAME=RESULTS_DIRECTORY")
|
|
17
|
+
name, directory = value.split("=", 1)
|
|
18
|
+
if not METHOD_NAME.fullmatch(name):
|
|
19
|
+
raise ValueError(f"{name}: method name must use letters, numbers, dots, underscores or hyphens")
|
|
20
|
+
if not directory:
|
|
21
|
+
raise ValueError(f"{value}: results directory is empty")
|
|
22
|
+
return name, Path(directory)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _method_summary(name: str, directory: Path, results: list[dict]) -> dict:
|
|
26
|
+
summary = _summary(results)
|
|
27
|
+
completed = [item for item in results if "details" in item]
|
|
28
|
+
mean_metrics = {
|
|
29
|
+
metric: (sum(item["details"]["metrics"][metric] for item in completed) / len(completed) if completed else None)
|
|
30
|
+
for metric in SCORED_METRICS
|
|
31
|
+
}
|
|
32
|
+
mean_coverage = (
|
|
33
|
+
sum(item["details"]["genes"]["coverage"] for item in completed) / len(completed)
|
|
34
|
+
if completed else None
|
|
35
|
+
)
|
|
36
|
+
return {
|
|
37
|
+
"name": name,
|
|
38
|
+
"results_directory": str(directory),
|
|
39
|
+
**summary,
|
|
40
|
+
"pass_rate": summary["passed"] / summary["total"] if summary["total"] else 0.0,
|
|
41
|
+
"mean_metrics": mean_metrics,
|
|
42
|
+
"mean_coverage": mean_coverage,
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def compare_matrix(
|
|
47
|
+
root: Path,
|
|
48
|
+
methods: list[tuple[str, Path]],
|
|
49
|
+
dataset_ids: list[str] | None = None,
|
|
50
|
+
detail_limit: int = 20,
|
|
51
|
+
) -> dict:
|
|
52
|
+
if len(methods) < 2:
|
|
53
|
+
raise ValueError("a benchmark matrix requires at least two --method entries")
|
|
54
|
+
names = [name for name, _ in methods]
|
|
55
|
+
if len(names) != len(set(names)):
|
|
56
|
+
raise ValueError("method names must be unique")
|
|
57
|
+
cells = []
|
|
58
|
+
method_summaries = []
|
|
59
|
+
benchmark_ids = None
|
|
60
|
+
for name, directory in methods:
|
|
61
|
+
if not METHOD_NAME.fullmatch(name):
|
|
62
|
+
raise ValueError(f"{name}: invalid method name")
|
|
63
|
+
report = compare_suite(root, directory, dataset_ids, detail_limit)
|
|
64
|
+
ids = [item["id"] for item in report["results"]]
|
|
65
|
+
if benchmark_ids is None:
|
|
66
|
+
benchmark_ids = ids
|
|
67
|
+
elif ids != benchmark_ids:
|
|
68
|
+
raise ValueError(f"{name}: benchmark selection differs from the first method")
|
|
69
|
+
method_results = []
|
|
70
|
+
for item in report["results"]:
|
|
71
|
+
cell = {"method": name, **item}
|
|
72
|
+
cells.append(cell)
|
|
73
|
+
method_results.append(item)
|
|
74
|
+
method_summaries.append(_method_summary(name, directory, method_results))
|
|
75
|
+
|
|
76
|
+
def rank_key(item: dict):
|
|
77
|
+
spearman = item["mean_metrics"]["spearman_logfc"]
|
|
78
|
+
return (-item["passed"], item["failed"], -(spearman if spearman is not None else -1.0), item["name"])
|
|
79
|
+
|
|
80
|
+
leaderboard = sorted(method_summaries, key=rank_key)
|
|
81
|
+
return {
|
|
82
|
+
"operation": "matrix",
|
|
83
|
+
"selection": {"assay": "bulk_rna_seq", "ids": dataset_ids or []},
|
|
84
|
+
"summary": {**_summary(cells), "methods": len(methods), "benchmarks": len(benchmark_ids or [])},
|
|
85
|
+
"leaderboard": leaderboard,
|
|
86
|
+
"results": cells,
|
|
87
|
+
}
|