openomicsbench 2.2.0__tar.gz → 3.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (229) hide show
  1. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/CITATION.cff +1 -1
  2. {openomicsbench-2.2.0/src/openomicsbench.egg-info → openomicsbench-3.0.0}/PKG-INFO +49 -7
  3. openomicsbench-2.2.0/PKG-INFO → openomicsbench-3.0.0/README.md +46 -37
  4. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-005/expected/validation.json +7 -1
  5. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-005/manifest.json +4 -4
  6. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/pyproject.toml +3 -3
  7. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/__init__.py +1 -1
  8. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/cli.py +38 -3
  9. openomicsbench-3.0.0/src/omicsbench/evaluate.py +211 -0
  10. openomicsbench-3.0.0/src/omicsbench/evidence.py +240 -0
  11. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/regression.py +3 -3
  12. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/suite.py +151 -37
  13. openomicsbench-3.0.0/src/omicsbench/variants.py +210 -0
  14. openomicsbench-2.2.0/README.md → openomicsbench-3.0.0/src/openomicsbench.egg-info/PKG-INFO +79 -4
  15. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/openomicsbench.egg-info/SOURCES.txt +7 -1
  16. openomicsbench-3.0.0/tests/test_evaluate.py +108 -0
  17. openomicsbench-3.0.0/tests/test_evidence.py +80 -0
  18. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_regression.py +13 -0
  19. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_suite.py +5 -1
  20. openomicsbench-3.0.0/tests/test_variants.py +83 -0
  21. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/LICENSE +0 -0
  22. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/LICENSE-METADATA +0 -0
  23. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/MANIFEST.in +0 -0
  24. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/NOTICE +0 -0
  25. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/LICENSE-DATA.md +0 -0
  26. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/README.md +0 -0
  27. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/expected/baseline.json +0 -0
  28. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/expected/size-fidelity.json +0 -0
  29. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/expected/source-counts.tsv +0 -0
  30. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/expected/validation.json +0 -0
  31. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/manifest.json +0 -0
  32. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/nano/counts.tsv +0 -0
  33. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/pocket/counts.tsv +0 -0
  34. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/fixture-001/provenance/transform.json +0 -0
  35. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-001/LICENSE-DATA.md +0 -0
  36. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-001/README.md +0 -0
  37. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-001/manifest.json +0 -0
  38. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/README.md +0 -0
  39. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/attribution.json +0 -0
  40. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/expected/deseq2.json +0 -0
  41. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/expected/reference-effects.tsv.gz +0 -0
  42. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/expected/source-counts.tsv +0 -0
  43. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/expected/validation.json +0 -0
  44. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/manifest.json +0 -0
  45. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/pocket/counts.tsv +0 -0
  46. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/provenance/transform.json +0 -0
  47. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/reference.json +0 -0
  48. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/rights.json +0 -0
  49. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-002/samples.json +0 -0
  50. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/README.md +0 -0
  51. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/attribution.json +0 -0
  52. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/expected/deseq2.json +0 -0
  53. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/expected/reference-effects.tsv.gz +0 -0
  54. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/expected/source-counts.tsv +0 -0
  55. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/expected/validation.json +0 -0
  56. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/manifest.json +0 -0
  57. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/pocket/counts.tsv +0 -0
  58. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/provenance/transform.json +0 -0
  59. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/reference.json +0 -0
  60. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/rights.json +0 -0
  61. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-003/samples.json +0 -0
  62. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/README.md +0 -0
  63. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/attribution.json +0 -0
  64. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/expected/deseq2.json +0 -0
  65. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/expected/reference-effects.tsv.gz +0 -0
  66. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/expected/source-counts.tsv +0 -0
  67. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/expected/validation.json +0 -0
  68. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/manifest.json +0 -0
  69. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/pocket/counts.tsv +0 -0
  70. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/provenance/transform.json +0 -0
  71. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/reference.json +0 -0
  72. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/rights.json +0 -0
  73. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-004/samples.json +0 -0
  74. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/README.md +0 -0
  75. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/attribution.json +0 -0
  76. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/expected/deseq2.json +0 -0
  77. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/expected/reference-effects.tsv.gz +0 -0
  78. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/expected/source-counts.tsv +0 -0
  79. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/expected/validation.json +0 -0
  80. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/manifest.json +0 -0
  81. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/pocket/counts.tsv +0 -0
  82. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/provenance/transform.json +0 -0
  83. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/reference.json +0 -0
  84. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/rights.json +0 -0
  85. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-005/samples.json +0 -0
  86. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/README.md +0 -0
  87. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/attribution.json +0 -0
  88. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/expected/deseq2.json +0 -0
  89. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/expected/reference-effects.tsv.gz +0 -0
  90. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/expected/source-counts.tsv +0 -0
  91. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/expected/validation.json +0 -0
  92. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/manifest.json +0 -0
  93. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/pocket/counts.tsv +0 -0
  94. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/provenance/transform.json +0 -0
  95. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/reference.json +0 -0
  96. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/rights.json +0 -0
  97. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-006/samples.json +0 -0
  98. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/README.md +0 -0
  99. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/attribution.json +0 -0
  100. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/expected/deseq2.json +0 -0
  101. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/expected/reference-effects.tsv.gz +0 -0
  102. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/expected/source-counts.tsv +0 -0
  103. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/expected/validation.json +0 -0
  104. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/manifest.json +0 -0
  105. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/pocket/counts.tsv +0 -0
  106. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/provenance/transform.json +0 -0
  107. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/reference.json +0 -0
  108. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/rights.json +0 -0
  109. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-007/samples.json +0 -0
  110. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/README.md +0 -0
  111. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/attribution.json +0 -0
  112. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/expected/deseq2.json +0 -0
  113. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/expected/reference-effects.tsv.gz +0 -0
  114. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/expected/source-counts.tsv +0 -0
  115. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/expected/validation.json +0 -0
  116. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/manifest.json +0 -0
  117. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/pocket/counts.tsv +0 -0
  118. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/provenance/transform.json +0 -0
  119. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/reference.json +0 -0
  120. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/rights.json +0 -0
  121. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-008/samples.json +0 -0
  122. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/README.md +0 -0
  123. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/attribution.json +0 -0
  124. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/expected/deseq2.json +0 -0
  125. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/expected/reference-effects.tsv.gz +0 -0
  126. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/expected/source-counts.tsv +0 -0
  127. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/expected/validation.json +0 -0
  128. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/manifest.json +0 -0
  129. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/pocket/counts.tsv +0 -0
  130. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/provenance/transform.json +0 -0
  131. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/reference.json +0 -0
  132. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/rights.json +0 -0
  133. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-009/samples.json +0 -0
  134. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/README.md +0 -0
  135. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/attribution.json +0 -0
  136. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/expected/deseq2.json +0 -0
  137. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/expected/reference-effects.tsv.gz +0 -0
  138. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/expected/source-counts.tsv +0 -0
  139. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/expected/validation.json +0 -0
  140. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/manifest.json +0 -0
  141. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/pocket/counts.tsv +0 -0
  142. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/provenance/transform.json +0 -0
  143. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/reference.json +0 -0
  144. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/rights.json +0 -0
  145. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-010/samples.json +0 -0
  146. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/README.md +0 -0
  147. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/attribution.json +0 -0
  148. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/expected/deseq2.json +0 -0
  149. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/expected/reference-effects.tsv.gz +0 -0
  150. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/expected/source-counts.tsv +0 -0
  151. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/expected/validation.json +0 -0
  152. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/manifest.json +0 -0
  153. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/pocket/counts.tsv +0 -0
  154. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/provenance/transform.json +0 -0
  155. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/reference.json +0 -0
  156. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/rights.json +0 -0
  157. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-011/samples.json +0 -0
  158. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/README.md +0 -0
  159. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/attribution.json +0 -0
  160. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/expected/deseq2.json +0 -0
  161. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/expected/reference-effects.tsv.gz +0 -0
  162. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/expected/source-counts.tsv +0 -0
  163. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/expected/validation.json +0 -0
  164. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/manifest.json +0 -0
  165. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/pocket/counts.tsv +0 -0
  166. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/provenance/transform.json +0 -0
  167. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/reference.json +0 -0
  168. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/rights.json +0 -0
  169. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-012/samples.json +0 -0
  170. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/README.md +0 -0
  171. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/attribution.json +0 -0
  172. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/expected/deseq2.json +0 -0
  173. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/expected/reference-effects.tsv.gz +0 -0
  174. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/expected/source-counts.tsv +0 -0
  175. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/expected/validation.json +0 -0
  176. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/manifest.json +0 -0
  177. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/pocket/counts.tsv +0 -0
  178. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/provenance/transform.json +0 -0
  179. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/reference.json +0 -0
  180. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/rights.json +0 -0
  181. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/rnaseq/rnaseq-013/samples.json +0 -0
  182. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-001/expected/validation.json +0 -0
  183. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-001/manifest.json +0 -0
  184. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-001/nano/sequences.fasta +0 -0
  185. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-002/expected/validation.json +0 -0
  186. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-002/manifest.json +0 -0
  187. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-002/nano/sequences.fasta +0 -0
  188. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-003/expected/validation.json +0 -0
  189. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-003/manifest.json +0 -0
  190. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-003/nano/sequences.fasta +0 -0
  191. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-004/expected/validation.json +0 -0
  192. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-004/manifest.json +0 -0
  193. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-004/nano/reads_R1.fastq +0 -0
  194. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-004/nano/reads_R2.fastq +0 -0
  195. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-005/expected/variants.tsv +0 -0
  196. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-005/nano/reads_R1.fastq +0 -0
  197. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-005/nano/reads_R2.fastq +0 -0
  198. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/datasets/sequences/sequence-005/nano/reference.fasta +0 -0
  199. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/setup.cfg +0 -0
  200. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/setup.py +0 -0
  201. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/__main__.py +0 -0
  202. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/assays.py +0 -0
  203. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/bundle.py +0 -0
  204. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/cache.py +0 -0
  205. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/compare.py +0 -0
  206. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/download.py +0 -0
  207. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/expression_atlas.py +0 -0
  208. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/hashing.py +0 -0
  209. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/matrix.py +0 -0
  210. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/models.py +0 -0
  211. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/proteins.py +0 -0
  212. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/registry.py +0 -0
  213. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/rnaseq.py +0 -0
  214. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/sequence_analysis.py +0 -0
  215. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/sequence_cli.py +0 -0
  216. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/sequence_tools.py +0 -0
  217. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/sequences.py +0 -0
  218. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/omicsbench/validate.py +0 -0
  219. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/openomicsbench.egg-info/dependency_links.txt +0 -0
  220. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/openomicsbench.egg-info/entry_points.txt +0 -0
  221. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/openomicsbench.egg-info/requires.txt +0 -0
  222. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/src/openomicsbench.egg-info/top_level.txt +0 -0
  223. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_assays.py +0 -0
  224. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_bundle.py +0 -0
  225. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_core.py +0 -0
  226. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_matrix.py +0 -0
  227. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_proteins.py +0 -0
  228. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_sequence_fixtures.py +0 -0
  229. {openomicsbench-2.2.0 → openomicsbench-3.0.0}/tests/test_sequences.py +0 -0
@@ -2,7 +2,7 @@ cff-version: 1.2.0
2
2
  message: "Please cite the archived release corresponding to the version you used."
3
3
  title: "OpenOmicsBench"
4
4
  type: software
5
- version: "2.2.0"
5
+ version: "3.0.0"
6
6
  license: Apache-2.0
7
7
  doi: "10.5281/zenodo.22551734"
8
8
  authors:
@@ -1,7 +1,7 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: openomicsbench
3
- Version: 2.2.0
4
- Summary: Traceable omics benchmarks and deterministic FASTA and FASTQ tools
3
+ Version: 3.0.0
4
+ Summary: Traceable multi-assay benchmarks, VCF evaluation and deterministic sequence tools
5
5
  Author: Vivaan Patni
6
6
  Maintainer: Vivaan Patni
7
7
  License-Expression: Apache-2.0
@@ -11,7 +11,7 @@ Project-URL: Issues, https://github.com/vxxqv/openomicsbench/issues
11
11
  Project-URL: Changelog, https://github.com/vxxqv/openomicsbench/blob/main/CHANGELOG.md
12
12
  Project-URL: Zenodo, https://doi.org/10.5281/zenodo.22551734
13
13
  Project-URL: ORCID, https://orcid.org/0009-0005-1859-5107
14
- Keywords: bioinformatics,computational-biology,genomics,transcriptomics,proteomics,bulk-rna-seq,dna-seq,fasta,fastq,dna,rna,protein-sequence,amino-acid-analysis,sequence-analysis,quality-control,kmer,variant-truth,benchmarking,benchmark-data,test-data,data-validation,data-provenance,data-integrity,reproducibility,scientific-workflows,research-software,expression-atlas,ensembl,deseq2
14
+ Keywords: bioinformatics,computational-biology,genomics,transcriptomics,proteomics,bulk-rna-seq,dna-seq,fasta,fastq,vcf,dna,rna,protein-sequence,amino-acid-analysis,sequence-analysis,quality-control,kmer,variant-truth,variant-benchmarking,multi-assay,ro-crate,benchmarking,benchmark-data,test-data,data-validation,data-provenance,data-integrity,reproducibility,scientific-workflows,research-software,expression-atlas,ensembl,deseq2
15
15
  Classifier: Development Status :: 5 - Production/Stable
16
16
  Classifier: Environment :: Console
17
17
  Classifier: Intended Audience :: Science/Research
@@ -33,9 +33,9 @@ Dynamic: license-file
33
33
 
34
34
  # OpenOmicsBench
35
35
 
36
- OpenOmicsBench is a compact benchmark collection and local sequence toolkit for bioinformatics software testing, method checks and teaching. Version 2 keeps the 12 certified bulk RNA-seq objects from version 1 and adds strict FASTA and FASTQ handling, DNA, RNA and protein support, paired-read checks, preprocessing, sequence QC and five deterministic sequence benchmarks.
36
+ OpenOmicsBench is a compact benchmark collection and local sequence toolkit for bioinformatics software testing, method checks and teaching. It contains 12 certified bulk RNA-seq objects, five deterministic DNA, RNA and protein sequence fixtures, strict FASTA and FASTQ tools, and a CI-ready evaluation system.
37
37
 
38
- Version 2.2 adds multi-method benchmark matrices, report-to-report regression gates and portable benchmark bundles. Suite reports can be written as JSON, Markdown, JUnit XML, CSV or a self-contained HTML page.
38
+ Version 3 evaluates RNA differential-expression results and DNA variant calls in the same run. It adds exact-SNV VCF scoring, method and resource receipts, cross-method multi-assay matrices, richer offline dashboards, and deterministic evidence crates with RO-Crate metadata and checksums.
39
39
 
40
40
  The package runs offline after installation. Sequence files stay on the local computer. The built-in tools cover inspection and lightweight preprocessing; they do not claim to replace aligners, variant callers, taxonomic classifiers or assay-specific statistical workflows.
41
41
 
@@ -54,6 +54,7 @@ List and validate the bundled benchmarks:
54
54
  omicsbench list
55
55
  omicsbench validate rnaseq-002
56
56
  omicsbench validate sequence-004
57
+ omicsbench variant compare sequence-005 calls.vcf.gz
57
58
  ```
58
59
 
59
60
  The wheel contains the complete collection. A repository checkout and network connection are not required.
@@ -135,6 +136,26 @@ Then run:
135
136
  omicsbench suite compare results --id rnaseq-002 --id rnaseq-003 --junit reports/comparison.xml
136
137
  ```
137
138
 
139
+ For a mixed RNA and DNA run, add benchmark-named VCF output alongside the expression tables:
140
+
141
+ ```text
142
+ results/
143
+ method.json
144
+ rnaseq-002.tsv.gz
145
+ sequence-005.vcf.gz
146
+ ```
147
+
148
+ ```sh
149
+ omicsbench suite evaluate results \
150
+ --id rnaseq-002 \
151
+ --id sequence-005 \
152
+ --json reports/evaluation.json \
153
+ --html reports/evaluation.html \
154
+ --evidence reports/evaluation-evidence.zip
155
+ ```
156
+
157
+ `method.json` is optional. When supplied, it records the exact method version, command, container or source revision, parameters, runtime, peak memory and thread count. Missing selected outputs fail as benchmark cases.
158
+
138
159
  Compare several tools or parameter sets in one matrix:
139
160
 
140
161
  ```sh
@@ -146,13 +167,32 @@ omicsbench suite matrix \
146
167
  --html reports/matrix.html
147
168
  ```
148
169
 
170
+ Use `evaluate-matrix` when each method directory contains a mixture of RNA tables and VCF calls:
171
+
172
+ ```sh
173
+ omicsbench suite evaluate-matrix \
174
+ --method current=results/current \
175
+ --method candidate=results/candidate \
176
+ --id rnaseq-002 \
177
+ --id sequence-005 \
178
+ --html reports/multi-assay.html
179
+ ```
180
+
181
+ The leaderboard uses pass counts and keeps assay metrics separate. It does not average unlike measurements into a single score.
182
+
149
183
  Use a previously accepted report as a regression baseline:
150
184
 
151
185
  ```sh
152
186
  omicsbench suite regress accepted.json candidate.json --absolute-tolerance 0.01 --junit reports/regression.xml
153
187
  ```
154
188
 
155
- Omit `--id` to require results for all 12 comparable RNA-seq objects. A completed suite exits with status 0 when every check passes, status 2 when a benchmark or regression gate fails, and status 1 for invalid input. Existing report files are protected unless `--force` is supplied. The [suite reference](docs/benchmark-suites.md) describes the file convention and report fields.
189
+ Omit `--id` from `suite compare` to require results for all 12 comparable RNA-seq objects. Omit it from `suite evaluate` to require every benchmark with a supported comparator. A completed suite exits with status 0 when every check passes, status 2 when a benchmark or regression gate fails, and status 1 for invalid input. Existing report files are protected unless `--force` is supplied. The [suite reference](docs/benchmark-suites.md) describes the file convention and report fields.
190
+
191
+ Verify an evaluation evidence crate without extracting it:
192
+
193
+ ```sh
194
+ omicsbench evidence verify reports/evaluation-evidence.zip
195
+ ```
156
196
 
157
197
  ## Portable bundles
158
198
 
@@ -189,9 +229,11 @@ Profiles are available for whole-genome, exome, targeted-panel, bulk RNA-seq, si
189
229
 
190
230
  These five objects are project-authored synthetic fixtures. Each has an exact file inventory, checksums, format and molecule declarations, expected summary metrics and a deterministic rebuild workflow. `sequence-005` also verifies that every truth-set reference allele matches the bundled reference and that every alternate allele is supported by the paired reads. The objects test software behavior and do not represent a biological cohort or sequencing instrument.
191
231
 
232
+ The version 3 VCF comparator for `sequence-005` matches contig, one-based position, REF and ALT exactly. It reports precision, recall and F1 for A/C/G/T substitutions and verifies REF against the packaged synthetic reference. Genotypes, indels, complex representations and confident-region stratification are outside this fixture. Use a haplotype-aware benchmarking engine such as [hap.py](https://github.com/Illumina/hap.py) or vcfeval with an appropriate truth set and confident regions for real germline benchmarking; the [GA4GH benchmarking project](https://github.com/ga4gh/benchmarking-tools) documents that broader problem.
233
+
192
234
  ## Bulk RNA-seq benchmarks
193
235
 
194
- Version 2 retains the complete version 1 biological collection unchanged.
236
+ Version 3 retains the complete version 1 biological collection unchanged.
195
237
 
196
238
  | ID | Design | Samples | Pocket genes |
197
239
  |---|---|---:|---:|
@@ -1,41 +1,8 @@
1
- Metadata-Version: 2.4
2
- Name: openomicsbench
3
- Version: 2.2.0
4
- Summary: Traceable omics benchmarks and deterministic FASTA and FASTQ tools
5
- Author: Vivaan Patni
6
- Maintainer: Vivaan Patni
7
- License-Expression: Apache-2.0
8
- Project-URL: Homepage, https://github.com/vxxqv/openomicsbench
9
- Project-URL: Repository, https://github.com/vxxqv/openomicsbench
10
- Project-URL: Issues, https://github.com/vxxqv/openomicsbench/issues
11
- Project-URL: Changelog, https://github.com/vxxqv/openomicsbench/blob/main/CHANGELOG.md
12
- Project-URL: Zenodo, https://doi.org/10.5281/zenodo.22551734
13
- Project-URL: ORCID, https://orcid.org/0009-0005-1859-5107
14
- Keywords: bioinformatics,computational-biology,genomics,transcriptomics,proteomics,bulk-rna-seq,dna-seq,fasta,fastq,dna,rna,protein-sequence,amino-acid-analysis,sequence-analysis,quality-control,kmer,variant-truth,benchmarking,benchmark-data,test-data,data-validation,data-provenance,data-integrity,reproducibility,scientific-workflows,research-software,expression-atlas,ensembl,deseq2
15
- Classifier: Development Status :: 5 - Production/Stable
16
- Classifier: Environment :: Console
17
- Classifier: Intended Audience :: Science/Research
18
- Classifier: Operating System :: OS Independent
19
- Classifier: Programming Language :: Python :: 3
20
- Classifier: Programming Language :: Python :: 3.11
21
- Classifier: Programming Language :: Python :: 3.12
22
- Classifier: Programming Language :: Python :: 3.13
23
- Classifier: Programming Language :: Python :: 3.14
24
- Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
25
- Requires-Python: >=3.11
26
- Description-Content-Type: text/markdown
27
- License-File: LICENSE
28
- License-File: LICENSE-METADATA
29
- License-File: NOTICE
30
- Requires-Dist: pydantic<3,>=2.13
31
- Requires-Dist: numpy<3,>=2.3
32
- Dynamic: license-file
33
-
34
1
  # OpenOmicsBench
35
2
 
36
- OpenOmicsBench is a compact benchmark collection and local sequence toolkit for bioinformatics software testing, method checks and teaching. Version 2 keeps the 12 certified bulk RNA-seq objects from version 1 and adds strict FASTA and FASTQ handling, DNA, RNA and protein support, paired-read checks, preprocessing, sequence QC and five deterministic sequence benchmarks.
3
+ OpenOmicsBench is a compact benchmark collection and local sequence toolkit for bioinformatics software testing, method checks and teaching. It contains 12 certified bulk RNA-seq objects, five deterministic DNA, RNA and protein sequence fixtures, strict FASTA and FASTQ tools, and a CI-ready evaluation system.
37
4
 
38
- Version 2.2 adds multi-method benchmark matrices, report-to-report regression gates and portable benchmark bundles. Suite reports can be written as JSON, Markdown, JUnit XML, CSV or a self-contained HTML page.
5
+ Version 3 evaluates RNA differential-expression results and DNA variant calls in the same run. It adds exact-SNV VCF scoring, method and resource receipts, cross-method multi-assay matrices, richer offline dashboards, and deterministic evidence crates with RO-Crate metadata and checksums.
39
6
 
40
7
  The package runs offline after installation. Sequence files stay on the local computer. The built-in tools cover inspection and lightweight preprocessing; they do not claim to replace aligners, variant callers, taxonomic classifiers or assay-specific statistical workflows.
41
8
 
@@ -54,6 +21,7 @@ List and validate the bundled benchmarks:
54
21
  omicsbench list
55
22
  omicsbench validate rnaseq-002
56
23
  omicsbench validate sequence-004
24
+ omicsbench variant compare sequence-005 calls.vcf.gz
57
25
  ```
58
26
 
59
27
  The wheel contains the complete collection. A repository checkout and network connection are not required.
@@ -135,6 +103,26 @@ Then run:
135
103
  omicsbench suite compare results --id rnaseq-002 --id rnaseq-003 --junit reports/comparison.xml
136
104
  ```
137
105
 
106
+ For a mixed RNA and DNA run, add benchmark-named VCF output alongside the expression tables:
107
+
108
+ ```text
109
+ results/
110
+ method.json
111
+ rnaseq-002.tsv.gz
112
+ sequence-005.vcf.gz
113
+ ```
114
+
115
+ ```sh
116
+ omicsbench suite evaluate results \
117
+ --id rnaseq-002 \
118
+ --id sequence-005 \
119
+ --json reports/evaluation.json \
120
+ --html reports/evaluation.html \
121
+ --evidence reports/evaluation-evidence.zip
122
+ ```
123
+
124
+ `method.json` is optional. When supplied, it records the exact method version, command, container or source revision, parameters, runtime, peak memory and thread count. Missing selected outputs fail as benchmark cases.
125
+
138
126
  Compare several tools or parameter sets in one matrix:
139
127
 
140
128
  ```sh
@@ -146,13 +134,32 @@ omicsbench suite matrix \
146
134
  --html reports/matrix.html
147
135
  ```
148
136
 
137
+ Use `evaluate-matrix` when each method directory contains a mixture of RNA tables and VCF calls:
138
+
139
+ ```sh
140
+ omicsbench suite evaluate-matrix \
141
+ --method current=results/current \
142
+ --method candidate=results/candidate \
143
+ --id rnaseq-002 \
144
+ --id sequence-005 \
145
+ --html reports/multi-assay.html
146
+ ```
147
+
148
+ The leaderboard uses pass counts and keeps assay metrics separate. It does not average unlike measurements into a single score.
149
+
149
150
  Use a previously accepted report as a regression baseline:
150
151
 
151
152
  ```sh
152
153
  omicsbench suite regress accepted.json candidate.json --absolute-tolerance 0.01 --junit reports/regression.xml
153
154
  ```
154
155
 
155
- Omit `--id` to require results for all 12 comparable RNA-seq objects. A completed suite exits with status 0 when every check passes, status 2 when a benchmark or regression gate fails, and status 1 for invalid input. Existing report files are protected unless `--force` is supplied. The [suite reference](docs/benchmark-suites.md) describes the file convention and report fields.
156
+ Omit `--id` from `suite compare` to require results for all 12 comparable RNA-seq objects. Omit it from `suite evaluate` to require every benchmark with a supported comparator. A completed suite exits with status 0 when every check passes, status 2 when a benchmark or regression gate fails, and status 1 for invalid input. Existing report files are protected unless `--force` is supplied. The [suite reference](docs/benchmark-suites.md) describes the file convention and report fields.
157
+
158
+ Verify an evaluation evidence crate without extracting it:
159
+
160
+ ```sh
161
+ omicsbench evidence verify reports/evaluation-evidence.zip
162
+ ```
156
163
 
157
164
  ## Portable bundles
158
165
 
@@ -189,9 +196,11 @@ Profiles are available for whole-genome, exome, targeted-panel, bulk RNA-seq, si
189
196
 
190
197
  These five objects are project-authored synthetic fixtures. Each has an exact file inventory, checksums, format and molecule declarations, expected summary metrics and a deterministic rebuild workflow. `sequence-005` also verifies that every truth-set reference allele matches the bundled reference and that every alternate allele is supported by the paired reads. The objects test software behavior and do not represent a biological cohort or sequencing instrument.
191
198
 
199
+ The version 3 VCF comparator for `sequence-005` matches contig, one-based position, REF and ALT exactly. It reports precision, recall and F1 for A/C/G/T substitutions and verifies REF against the packaged synthetic reference. Genotypes, indels, complex representations and confident-region stratification are outside this fixture. Use a haplotype-aware benchmarking engine such as [hap.py](https://github.com/Illumina/hap.py) or vcfeval with an appropriate truth set and confident regions for real germline benchmarking; the [GA4GH benchmarking project](https://github.com/ga4gh/benchmarking-tools) documents that broader problem.
200
+
192
201
  ## Bulk RNA-seq benchmarks
193
202
 
194
- Version 2 retains the complete version 1 biological collection unchanged.
203
+ Version 3 retains the complete version 1 biological collection unchanged.
195
204
 
196
205
  | ID | Design | Samples | Pocket genes |
197
206
  |---|---|---:|---:|
@@ -74,5 +74,11 @@
74
74
  "reference": "nano/reference.fasta",
75
75
  "variant_truth": "expected/variants.tsv",
76
76
  "variant_count": 3,
77
- "context_bases": 10
77
+ "context_bases": 10,
78
+ "variant_comparison": {
79
+ "mode": "exact_allele",
80
+ "scope": "single_nucleotide_substitutions",
81
+ "minimum_precision": 1.0,
82
+ "minimum_recall": 1.0
83
+ }
78
84
  }
@@ -2,7 +2,7 @@
2
2
  "schema_version": "2.0",
3
3
  "id": "sequence-005",
4
4
  "title": "Synthetic paired DNA-seq variant benchmark",
5
- "release": "2.0.0",
5
+ "release": "3.0.0",
6
6
  "assay": "whole_genome_dna_seq",
7
7
  "kind": "synthetic_fixture",
8
8
  "status": "validated",
@@ -93,13 +93,13 @@
93
93
  "tier": "expected",
94
94
  "role": "metrics",
95
95
  "media_type": "application/json",
96
- "bytes": 1739,
97
- "sha256": "9b21fb89fe782b9e7ced92f0bebba8e84f4d3e50188e7af18ad2a429ba87b9bb"
96
+ "bytes": 1902,
97
+ "sha256": "da0813bf715df2a455a7c0eb86dc804dbb8cf582c9aa09119e8d976e3548cd88"
98
98
  }
99
99
  ],
100
100
  "validation": {
101
101
  "profile": "expected/validation.json",
102
- "baseline_version": "dnaseq-truth-v1",
102
+ "baseline_version": "dnaseq-truth-v2",
103
103
  "metrics": []
104
104
  },
105
105
  "sequence": {
@@ -4,13 +4,13 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "openomicsbench"
7
- version = "2.2.0"
8
- description = "Traceable omics benchmarks and deterministic FASTA and FASTQ tools"
7
+ version = "3.0.0"
8
+ description = "Traceable multi-assay benchmarks, VCF evaluation and deterministic sequence tools"
9
9
  readme = {file = "README.md", content-type = "text/markdown"}
10
10
  license = "Apache-2.0"
11
11
  authors = [{name = "Vivaan Patni"}]
12
12
  maintainers = [{name = "Vivaan Patni"}]
13
- keywords = ["bioinformatics", "computational-biology", "genomics", "transcriptomics", "proteomics", "bulk-rna-seq", "dna-seq", "fasta", "fastq", "dna", "rna", "protein-sequence", "amino-acid-analysis", "sequence-analysis", "quality-control", "kmer", "variant-truth", "benchmarking", "benchmark-data", "test-data", "data-validation", "data-provenance", "data-integrity", "reproducibility", "scientific-workflows", "research-software", "expression-atlas", "ensembl", "deseq2"]
13
+ keywords = ["bioinformatics", "computational-biology", "genomics", "transcriptomics", "proteomics", "bulk-rna-seq", "dna-seq", "fasta", "fastq", "vcf", "dna", "rna", "protein-sequence", "amino-acid-analysis", "sequence-analysis", "quality-control", "kmer", "variant-truth", "variant-benchmarking", "multi-assay", "ro-crate", "benchmarking", "benchmark-data", "test-data", "data-validation", "data-provenance", "data-integrity", "reproducibility", "scientific-workflows", "research-software", "expression-atlas", "ensembl", "deseq2"]
14
14
  requires-python = ">=3.11"
15
15
  dependencies = ["pydantic>=2.13,<3", "numpy>=2.3,<3"]
16
16
  classifiers = [
@@ -1,2 +1,2 @@
1
1
  """Discovery, validation and local analysis of compact omics benchmarks."""
2
- __version__ = "2.2.0"
2
+ __version__ = "3.0.0"
@@ -16,6 +16,9 @@ from .matrix import compare_matrix, parse_method
16
16
  from .regression import compare_reports
17
17
  from .bundle import create_bundle, verify_bundle
18
18
  from .validate import validate
19
+ from .variants import compare_variants
20
+ from .evaluate import evaluate_matrix, evaluate_suite
21
+ from .evidence import create_evidence_crate, verify_evidence_crate
19
22
 
20
23
  def main():
21
24
  p = argparse.ArgumentParser(description="Find, verify and process compact omics benchmarks.",epilog="Example: omicsbench info rnaseq-002")
@@ -37,6 +40,12 @@ def main():
37
40
  command.add_argument("results",type=Path,help="CSV or TSV file with gene_id and log2_fold_change columns.")
38
41
  command.add_argument("--detail-limit",type=int,default=20,help="Maximum missing and unexpected gene examples to return.")
39
42
  add_sequence_parser(sub)
43
+ variant = sub.add_parser("variant", help="Compare VCF calls with a bundled small-variant truth set.")
44
+ variant_sub = variant.add_subparsers(dest="variant_command", required=True)
45
+ variant_compare = variant_sub.add_parser("compare", help="Score exact SNV alleles in a VCF or VCF.GZ file.")
46
+ variant_compare.add_argument("id", help="Benchmark ID with a declared variant truth set.")
47
+ variant_compare.add_argument("results", type=Path, help="VCF or VCF.GZ call set.")
48
+ variant_compare.add_argument("--detail-limit", type=int, default=20, help="Maximum false-positive and false-negative examples to return.")
40
49
  assay = sub.add_parser("assay", help="Inspect sequencing assay profiles and build local command plans.")
41
50
  assay_sub = assay.add_subparsers(dest="assay_command", required=True)
42
51
  assay_sub.add_parser("list", help="List supported assay profiles.")
@@ -54,17 +63,23 @@ def main():
54
63
  suite_compare = suite_sub.add_parser("compare", help="Compare a directory of differential-expression results.")
55
64
  suite_compare.add_argument("results", type=Path, help="Directory containing files named <dataset-id>.csv or .tsv, optionally gzip-compressed.")
56
65
  suite_compare.add_argument("--detail-limit", type=int, default=20, help="Maximum missing and unexpected gene examples per benchmark.")
66
+ suite_evaluate = suite_sub.add_parser("evaluate", help="Evaluate one result directory across RNA and DNA benchmarks.")
67
+ suite_evaluate.add_argument("results", type=Path, help="Directory containing benchmark-named RNA tables and VCF files.")
68
+ suite_evaluate.add_argument("--detail-limit", type=int, default=20, help="Maximum mismatch examples per benchmark.")
57
69
  suite_matrix = suite_sub.add_parser("matrix", help="Compare several analysis methods across the same benchmarks.")
58
70
  suite_matrix.add_argument("--method", action="append", required=True, help="Method and result directory as NAME=PATH. Repeat for every method.")
59
71
  suite_matrix.add_argument("--detail-limit", type=int, default=20, help="Maximum missing and unexpected gene examples per benchmark.")
72
+ suite_evaluate_matrix = suite_sub.add_parser("evaluate-matrix", help="Compare several methods across RNA and DNA benchmarks.")
73
+ suite_evaluate_matrix.add_argument("--method", action="append", required=True, help="Method and result directory as NAME=PATH. Repeat for every method.")
74
+ suite_evaluate_matrix.add_argument("--detail-limit", type=int, default=20, help="Maximum mismatch examples per benchmark.")
60
75
  suite_regress = suite_sub.add_parser("regress", help="Fail when a candidate suite report regresses from a baseline report.")
61
76
  suite_regress.add_argument("baseline", type=Path, help="Previously accepted JSON suite report.")
62
77
  suite_regress.add_argument("candidate", type=Path, help="Candidate JSON suite report to check.")
63
78
  suite_regress.add_argument("--absolute-tolerance", type=float, default=0.0, help="Largest permitted absolute decrease in a tracked metric.")
64
79
  suite_regress.add_argument("--allow-missing", action="store_true", help="Do not fail when a baseline case is absent from the candidate.")
65
- for command in (suite_validate, suite_compare, suite_matrix):
80
+ for command in (suite_validate, suite_compare, suite_evaluate, suite_matrix, suite_evaluate_matrix):
66
81
  command.add_argument("--id", action="append", default=[], help="Benchmark ID to include. Repeat to select several; omit to run all eligible benchmarks.")
67
- for command in (suite_validate, suite_compare, suite_matrix, suite_regress):
82
+ for command in (suite_validate, suite_compare, suite_evaluate, suite_matrix, suite_evaluate_matrix, suite_regress):
68
83
  command.add_argument("--json", type=Path, help="Write the complete report as JSON.")
69
84
  command.add_argument("--markdown", type=Path, help="Write a concise Markdown report.")
70
85
  command.add_argument("--junit", type=Path, help="Write a JUnit XML report for CI systems.")
@@ -80,10 +95,21 @@ def main():
80
95
  bundle_create.add_argument("--force", action="store_true", help="Replace an existing bundle.")
81
96
  bundle_verify = bundle_sub.add_parser("verify", help="Verify a bundle inventory, hashes and dataset manifests.")
82
97
  bundle_verify.add_argument("archive", type=Path)
98
+ evidence = sub.add_parser("evidence", help="Verify a portable evaluation evidence crate.")
99
+ evidence_sub = evidence.add_subparsers(dest="evidence_command", required=True)
100
+ evidence_verify = evidence_sub.add_parser("verify", help="Verify checksums, inventory, report and RO-Crate metadata.")
101
+ evidence_verify.add_argument("archive", type=Path)
102
+ for command in (suite_evaluate, suite_evaluate_matrix):
103
+ command.add_argument("--evidence", type=Path, help="Write a deterministic ZIP with the reports, submitted results and RO-Crate metadata.")
83
104
  args = p.parse_args()
84
105
  try:
85
106
  if args.command == "seq":
86
107
  out = run_sequence_command(args)
108
+ elif args.command == "variant":
109
+ model, folder = lookup(args.root, args.id)
110
+ out = compare_variants(model, folder, args.results, args.detail_limit)
111
+ elif args.command == "evidence":
112
+ out = verify_evidence_crate(args.archive)
87
113
  elif args.command == "bundle":
88
114
  if args.bundle_command == "create":
89
115
  out = create_bundle(args.root, args.output, args.id, args.assay, args.force)
@@ -101,11 +127,15 @@ def main():
101
127
  out = validate_suite(args.root, args.assay, args.id)
102
128
  elif args.suite_command == "compare":
103
129
  out = compare_suite(args.root, args.results, args.id, args.detail_limit)
130
+ elif args.suite_command == "evaluate":
131
+ out = evaluate_suite(args.root, args.results, args.id, args.detail_limit)
104
132
  elif args.suite_command == "matrix":
105
133
  out = compare_matrix(args.root, [parse_method(value) for value in args.method], args.id, args.detail_limit)
134
+ elif args.suite_command == "evaluate-matrix":
135
+ out = evaluate_matrix(args.root, [parse_method(value) for value in args.method], args.id, args.detail_limit)
106
136
  else:
107
137
  out = compare_reports(args.baseline, args.candidate, args.absolute_tolerance, args.allow_missing)
108
- out["reports"] = write_reports(
138
+ reports = write_reports(
109
139
  out,
110
140
  json_path=args.json,
111
141
  markdown_path=args.markdown,
@@ -114,6 +144,9 @@ def main():
114
144
  csv_path=args.csv,
115
145
  html_path=args.html,
116
146
  )
147
+ if getattr(args, "evidence", None) is not None:
148
+ reports["evidence"] = create_evidence_crate(out, args.evidence, args.force)
149
+ out["reports"] = reports
117
150
  elif args.command == "list":
118
151
  out = [{"id":m.id,"title":m.title,"kind":m.kind,"status":m.status,"rights":m.rights.status,"tiers":sorted({f.tier for f in m.files})} for m,_ in registry(args.root).values() if not args.assay or m.assay==args.assay]
119
152
  elif args.command == "doctor":
@@ -135,6 +168,8 @@ def main():
135
168
  print(json.dumps(out,indent=2,allow_nan=False))
136
169
  if args.command == "compare" and out["status"] == "fail":
137
170
  sys.exit(2)
171
+ if args.command == "variant" and out["status"] == "fail":
172
+ sys.exit(2)
138
173
  if args.command == "suite" and out["summary"]["status"] == "fail":
139
174
  sys.exit(2)
140
175
  except (ValueError,OSError,KeyError) as exc:
@@ -0,0 +1,211 @@
1
+ """Unified evaluation of RNA effects and exact synthetic SNV calls."""
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ import math
6
+ import re
7
+ from pathlib import Path
8
+
9
+ from pydantic import Field, field_validator
10
+
11
+ from .compare import compare
12
+ from .matrix import METHOD_NAME
13
+ from .models import StrictModel
14
+ from .registry import registry
15
+ from .suite import RESULT_SUFFIXES, _summary
16
+ from .variants import compare_variants
17
+
18
+
19
+ VCF_SUFFIXES = (".vcf", ".vcf.gz")
20
+
21
+
22
+ class MethodReceipt(StrictModel):
23
+ schema_version: str = Field(pattern=r"^1\.0$")
24
+ name: str = Field(min_length=1, max_length=128)
25
+ version: str = Field(min_length=1, max_length=128)
26
+ description: str | None = Field(default=None, max_length=500)
27
+ command: str | None = Field(default=None, max_length=4000)
28
+ container: str | None = Field(default=None, max_length=500)
29
+ source_revision: str | None = Field(default=None, max_length=128)
30
+ runtime_seconds: float | None = Field(default=None, ge=0)
31
+ peak_memory_mb: float | None = Field(default=None, ge=0)
32
+ threads: int | None = Field(default=None, gt=0, strict=True)
33
+ parameters: dict[str, str | int | float | bool | None] = Field(default_factory=dict)
34
+
35
+ @field_validator("runtime_seconds", "peak_memory_mb")
36
+ @classmethod
37
+ def finite_resource_value(cls, value):
38
+ if value is not None and not math.isfinite(value):
39
+ raise ValueError("resource measurements must be finite")
40
+ return value
41
+
42
+
43
+ def read_method_receipt(directory: Path) -> dict | None:
44
+ path = Path(directory) / "method.json"
45
+ if not path.exists():
46
+ return None
47
+ if not path.is_file():
48
+ raise ValueError(f"{path}: method receipt is not a file")
49
+ try:
50
+ receipt = MethodReceipt.model_validate_json(path.read_text(encoding="utf-8"))
51
+ except (ValueError, UnicodeDecodeError) as error:
52
+ raise ValueError(f"{path}: invalid method receipt: {error}") from None
53
+ return {"path": str(path), **receipt.model_dump(mode="json")}
54
+
55
+
56
+ def _candidate_files(directory: Path, dataset_id: str, suffixes: tuple[str, ...]) -> list[Path]:
57
+ return [directory / f"{dataset_id}{suffix}" for suffix in suffixes if (directory / f"{dataset_id}{suffix}").is_file()]
58
+
59
+
60
+ def _result_file(directory: Path, dataset_id: str, suffixes: tuple[str, ...]) -> Path | None:
61
+ matches = _candidate_files(directory, dataset_id, suffixes)
62
+ if len(matches) > 1:
63
+ raise ValueError(f"{dataset_id}: multiple result files found: {', '.join(path.name for path in matches)}")
64
+ return matches[0] if matches else None
65
+
66
+
67
+ def _evaluator(model) -> tuple[str, tuple[str, ...]] | None:
68
+ if model.assay == "bulk_rna_seq" and model.kind == "real" and model.validation is not None:
69
+ return "differential_expression", RESULT_SUFFIXES
70
+ if model.assay == "whole_genome_dna_seq" and model.validation is not None:
71
+ return "exact_snv", VCF_SUFFIXES
72
+ return None
73
+
74
+
75
+ def _selection(root: Path, dataset_ids: list[str] | None) -> list[tuple]:
76
+ records = registry(Path(root))
77
+ requested = list(dict.fromkeys(dataset_ids or []))
78
+ unknown = [dataset_id for dataset_id in requested if dataset_id not in records]
79
+ if unknown:
80
+ raise ValueError(f"unknown dataset IDs: {', '.join(unknown)}")
81
+ if requested:
82
+ unsupported = [dataset_id for dataset_id in requested if _evaluator(records[dataset_id][0]) is None]
83
+ if unsupported:
84
+ raise ValueError(f"no result comparator for: {', '.join(unsupported)}")
85
+ return [records[dataset_id] for dataset_id in requested]
86
+ selected = [record for record in records.values() if _evaluator(record[0]) is not None]
87
+ if not selected:
88
+ raise ValueError("the collection has no comparable benchmarks")
89
+ return selected
90
+
91
+
92
+ def _assay_summary(results: list[dict]) -> dict:
93
+ assays = {}
94
+ for assay in sorted({item["assay"] for item in results}):
95
+ assays[assay] = _summary([item for item in results if item["assay"] == assay])
96
+ return assays
97
+
98
+
99
+ def evaluate_suite(
100
+ root: Path,
101
+ results_directory: Path,
102
+ dataset_ids: list[str] | None = None,
103
+ detail_limit: int = 20,
104
+ ) -> dict:
105
+ directory = Path(results_directory)
106
+ if not directory.is_dir():
107
+ raise ValueError(f"{directory}: results directory does not exist")
108
+ if detail_limit < 0:
109
+ raise ValueError("detail limit must be zero or greater")
110
+ receipt = read_method_receipt(directory)
111
+ results = []
112
+ for model, folder in _selection(Path(root), dataset_ids):
113
+ comparator, suffixes = _evaluator(model) # type: ignore[misc]
114
+ try:
115
+ path = _result_file(directory, model.id, suffixes)
116
+ if path is None:
117
+ expected = " or ".join(f"{model.id}{suffix}" for suffix in suffixes)
118
+ results.append({
119
+ "id": model.id,
120
+ "assay": model.assay,
121
+ "comparator": comparator,
122
+ "status": "fail",
123
+ "reason": f"missing result file; expected {expected}",
124
+ })
125
+ continue
126
+ details = compare(model, folder, path, detail_limit) if comparator == "differential_expression" else compare_variants(model, folder, path, detail_limit)
127
+ item = {
128
+ "id": model.id,
129
+ "assay": model.assay,
130
+ "comparator": comparator,
131
+ "status": details["status"],
132
+ "input": str(path),
133
+ "details": details,
134
+ }
135
+ if details["status"] == "fail":
136
+ item["reason"] = "; ".join(details["reasons"])
137
+ results.append(item)
138
+ except (ValueError, OSError, KeyError) as error:
139
+ results.append({
140
+ "id": model.id,
141
+ "assay": model.assay,
142
+ "comparator": comparator,
143
+ "status": "fail",
144
+ "reason": str(error),
145
+ })
146
+ summary = _summary(results)
147
+ summary["assays"] = _assay_summary(results)
148
+ return {
149
+ "operation": "evaluate",
150
+ "selection": {"ids": dataset_ids or [], "default": "all comparable benchmarks"},
151
+ "results_directory": str(directory),
152
+ "method": receipt,
153
+ "summary": summary,
154
+ "results": results,
155
+ }
156
+
157
+
158
+ def _leader(name: str, directory: Path, report: dict) -> dict:
159
+ summary = report["summary"]
160
+ receipt = report.get("method")
161
+ return {
162
+ "name": name,
163
+ "results_directory": str(directory),
164
+ "receipt": receipt,
165
+ "status": summary["status"],
166
+ "total": summary["total"],
167
+ "passed": summary["passed"],
168
+ "failed": summary["failed"],
169
+ "skipped": summary["skipped"],
170
+ "pass_rate": summary["passed"] / summary["total"] if summary["total"] else 0.0,
171
+ "assays": summary["assays"],
172
+ }
173
+
174
+
175
+ def evaluate_matrix(
176
+ root: Path,
177
+ methods: list[tuple[str, Path]],
178
+ dataset_ids: list[str] | None = None,
179
+ detail_limit: int = 20,
180
+ ) -> dict:
181
+ if len(methods) < 2:
182
+ raise ValueError("a multi-assay matrix requires at least two --method entries")
183
+ names = [name for name, _ in methods]
184
+ if len(names) != len(set(names)):
185
+ raise ValueError("method names must be unique")
186
+ for name in names:
187
+ if not METHOD_NAME.fullmatch(name):
188
+ raise ValueError(f"{name}: invalid method name")
189
+ cells = []
190
+ leaders = []
191
+ benchmark_ids = None
192
+ for name, directory in methods:
193
+ report = evaluate_suite(root, directory, dataset_ids, detail_limit)
194
+ ids = [item["id"] for item in report["results"]]
195
+ if benchmark_ids is None:
196
+ benchmark_ids = ids
197
+ elif ids != benchmark_ids:
198
+ raise ValueError(f"{name}: benchmark selection differs from the first method")
199
+ cells.extend({"method": name, **item} for item in report["results"])
200
+ leaders.append(_leader(name, directory, report))
201
+ leaderboard = sorted(leaders, key=lambda item: (-item["passed"], item["failed"], item["name"]))
202
+ summary = _summary(cells)
203
+ summary.update({"methods": len(methods), "benchmarks": len(benchmark_ids or []), "assays": _assay_summary(cells)})
204
+ return {
205
+ "operation": "evaluate_matrix",
206
+ "selection": {"ids": dataset_ids or [], "default": "all comparable benchmarks"},
207
+ "ranking": "passed cases descending, failed cases ascending, then method name; assay metrics are not pooled",
208
+ "summary": summary,
209
+ "leaderboard": leaderboard,
210
+ "results": cells,
211
+ }