biotapy 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. biotapy-0.0.1/.codecov.yaml +17 -0
  2. biotapy-0.0.1/.cruft.json +44 -0
  3. biotapy-0.0.1/.editorconfig +15 -0
  4. biotapy-0.0.1/.github/ISSUE_TEMPLATE/bug_report.yml +93 -0
  5. biotapy-0.0.1/.github/ISSUE_TEMPLATE/config.yml +5 -0
  6. biotapy-0.0.1/.github/ISSUE_TEMPLATE/feature_request.yml +11 -0
  7. biotapy-0.0.1/.github/dependabot.yml +12 -0
  8. biotapy-0.0.1/.github/pull_request_template.md +13 -0
  9. biotapy-0.0.1/.github/workflows/build.yaml +30 -0
  10. biotapy-0.0.1/.github/workflows/release.yaml +33 -0
  11. biotapy-0.0.1/.github/workflows/test.yaml +168 -0
  12. biotapy-0.0.1/.gitignore +23 -0
  13. biotapy-0.0.1/.knowledge/contracts/data-model-slots.md +82 -0
  14. biotapy-0.0.1/.knowledge/contracts/engine-parity.md +38 -0
  15. biotapy-0.0.1/.knowledge/contracts/function-shape.md +73 -0
  16. biotapy-0.0.1/.knowledge/contracts/index.md +8 -0
  17. biotapy-0.0.1/.knowledge/contracts/module-boundaries.md +65 -0
  18. biotapy-0.0.1/.knowledge/contracts/r-golden-parity.md +45 -0
  19. biotapy-0.0.1/.knowledge/contracts/tree-access.md +50 -0
  20. biotapy-0.0.1/.knowledge/decisions/docs-okf-and-sphinx.md +53 -0
  21. biotapy-0.0.1/.knowledge/decisions/index.md +11 -0
  22. biotapy-0.0.1/.knowledge/decisions/no-bundled-kegg.md +28 -0
  23. biotapy-0.0.1/.knowledge/decisions/optional-heavy-dependencies.md +56 -0
  24. biotapy-0.0.1/.knowledge/decisions/package-name-biotapy.md +30 -0
  25. biotapy-0.0.1/.knowledge/decisions/pure-by-default.md +43 -0
  26. biotapy-0.0.1/.knowledge/decisions/python-first-compiled-last.md +47 -0
  27. biotapy-0.0.1/.knowledge/decisions/r-bridge-before-ports.md +36 -0
  28. biotapy-0.0.1/.knowledge/decisions/samples-as-rows.md +35 -0
  29. biotapy-0.0.1/.knowledge/decisions/treedata-as-container.md +41 -0
  30. biotapy-0.0.1/.knowledge/index.md +23 -0
  31. biotapy-0.0.1/.knowledge/log.md +7 -0
  32. biotapy-0.0.1/.knowledge/playbooks/add-a-function.md +54 -0
  33. biotapy-0.0.1/.knowledge/playbooks/cut-a-release.md +54 -0
  34. biotapy-0.0.1/.knowledge/playbooks/index.md +7 -0
  35. biotapy-0.0.1/.knowledge/playbooks/maintain-knowledge.md +64 -0
  36. biotapy-0.0.1/.knowledge/roadmap/index.md +15 -0
  37. biotapy-0.0.1/.knowledge/roadmap/phase-0-foundation.md +535 -0
  38. biotapy-0.0.1/.knowledge/roadmap/phase-1-core.md +1053 -0
  39. biotapy-0.0.1/.knowledge/roadmap/phase-2-function.md +84 -0
  40. biotapy-0.0.1/.knowledge/roadmap/phase-3-stats.md +66 -0
  41. biotapy-0.0.1/.knowledge/roadmap/phase-4-ml-multiomics.md +59 -0
  42. biotapy-0.0.1/.knowledge/roadmap/phase-5-beyond.md +33 -0
  43. biotapy-0.0.1/.knowledge/roadmap/spec-review.md +52 -0
  44. biotapy-0.0.1/.pre-commit-config.yaml +59 -0
  45. biotapy-0.0.1/.readthedocs.yaml +16 -0
  46. biotapy-0.0.1/.vscode/extensions.json +18 -0
  47. biotapy-0.0.1/.vscode/launch.json +33 -0
  48. biotapy-0.0.1/.vscode/settings.json +18 -0
  49. biotapy-0.0.1/CHANGELOG.md +17 -0
  50. biotapy-0.0.1/CLAUDE.md +13 -0
  51. biotapy-0.0.1/LICENSE +29 -0
  52. biotapy-0.0.1/PKG-INFO +117 -0
  53. biotapy-0.0.1/README.md +68 -0
  54. biotapy-0.0.1/biome.jsonc +17 -0
  55. biotapy-0.0.1/docs/_static/.gitkeep +0 -0
  56. biotapy-0.0.1/docs/_static/css/custom.css +4 -0
  57. biotapy-0.0.1/docs/_templates/.gitkeep +0 -0
  58. biotapy-0.0.1/docs/api.md +3 -0
  59. biotapy-0.0.1/docs/changelog.md +3 -0
  60. biotapy-0.0.1/docs/conf.py +137 -0
  61. biotapy-0.0.1/docs/contributing.md +353 -0
  62. biotapy-0.0.1/docs/design.md +13 -0
  63. biotapy-0.0.1/docs/index.md +14 -0
  64. biotapy-0.0.1/docs/references.bib +10 -0
  65. biotapy-0.0.1/docs/references.md +5 -0
  66. biotapy-0.0.1/plan.md +259 -0
  67. biotapy-0.0.1/pyproject.toml +190 -0
  68. biotapy-0.0.1/rules.md +244 -0
  69. biotapy-0.0.1/scripts/knowledge_stale.sh +141 -0
  70. biotapy-0.0.1/src/biotapy/__init__.py +5 -0
  71. biotapy-0.0.1/src/biotapy/_core/__init__.py +6 -0
  72. biotapy-0.0.1/src/biotapy/_core/_optional.py +13 -0
  73. biotapy-0.0.1/src/biotapy/_core/_rng.py +13 -0
  74. biotapy-0.0.1/tests/core/test_optional.py +12 -0
  75. biotapy-0.0.1/tests/core/test_rng.py +22 -0
  76. biotapy-0.0.1/tests/test_ci.py +23 -0
  77. biotapy-0.0.1/tests/test_knowledge_bundle.py +44 -0
  78. biotapy-0.0.1/tests/test_knowledge_stale.py +22 -0
@@ -0,0 +1,17 @@
1
+ # Based on pydata/xarray
2
+ codecov:
3
+ require_ci_to_pass: no
4
+
5
+ coverage:
6
+ status:
7
+ project:
8
+ default:
9
+ # Require 1% coverage, i.e., always succeed
10
+ target: 1
11
+ patch: false
12
+ changes: false
13
+
14
+ comment:
15
+ layout: diff, flags, files
16
+ behavior: once
17
+ require_base: no
@@ -0,0 +1,44 @@
1
+ {
2
+ "template": "https://github.com/scverse/cookiecutter-scverse",
3
+ "commit": "6518dfa1abde7379ea7255daf0ce09c23f2b4c94",
4
+ "checkout": "v0.8.0",
5
+ "context": {
6
+ "cookiecutter": {
7
+ "project_name": "biotapy",
8
+ "package_name": "biotapy",
9
+ "project_description": "mia-style microbiome toolkit for Python on AnnData/TreeData",
10
+ "author_full_name": "Pedro Ribeiro",
11
+ "author_email": "pedrocasalribeiro@gmail.com",
12
+ "github_user": "pedrocr83",
13
+ "github_repo": "biotapy",
14
+ "license": "BSD 3-Clause License",
15
+ "ide_integration": true,
16
+ "issue_categorization": "labels",
17
+ "_copy_without_render": [
18
+ ".github/workflows/build.yaml",
19
+ ".github/workflows/test.yaml",
20
+ "docs/_templates/autosummary/**.rst"
21
+ ],
22
+ "_exclude_on_template_update": [
23
+ "CHANGELOG.md",
24
+ "LICENSE",
25
+ "README.md",
26
+ "docs/api.md",
27
+ "docs/index.md",
28
+ "docs/notebooks/example.ipynb",
29
+ "docs/references.bib",
30
+ "docs/references.md",
31
+ "src/**",
32
+ "tests/**"
33
+ ],
34
+ "_render_devdocs": false,
35
+ "_jinja2_env_vars": {
36
+ "lstrip_blocks": true,
37
+ "trim_blocks": true
38
+ },
39
+ "_template": "https://github.com/scverse/cookiecutter-scverse",
40
+ "_commit": "6518dfa1abde7379ea7255daf0ce09c23f2b4c94"
41
+ }
42
+ },
43
+ "directory": null
44
+ }
@@ -0,0 +1,15 @@
1
+ root = true
2
+
3
+ [*]
4
+ indent_style = space
5
+ indent_size = 4
6
+ end_of_line = lf
7
+ charset = utf-8
8
+ trim_trailing_whitespace = true
9
+ insert_final_newline = true
10
+
11
+ [{*.{yml,yaml,toml},.cruft.json}]
12
+ indent_size = 2
13
+
14
+ [Makefile]
15
+ indent_style = tab
@@ -0,0 +1,93 @@
1
+ name: Bug report
2
+ description: Report something that is broken or incorrect
3
+ labels: bug
4
+ body:
5
+ - type: markdown
6
+ attributes:
7
+ value: |
8
+ **Note**: Please read [this guide](https://matthewrocklin.com/blog/work/2018/02/28/minimal-bug-reports)
9
+ detailing how to provide the necessary information for us to reproduce your bug. In brief:
10
+ * Please provide exact steps how to reproduce the bug in a clean Python environment.
11
+ * In case it's not clear what's causing this bug, please provide the data or the data generation procedure.
12
+ * Replicate problems on public datasets or share data subsets when full sharing isn't possible.
13
+
14
+ - type: textarea
15
+ id: report
16
+ attributes:
17
+ label: Report
18
+ description: A clear and concise description of what the bug is.
19
+ validations:
20
+ required: true
21
+
22
+ - type: textarea
23
+ id: versions
24
+ attributes:
25
+ label: Versions
26
+ description: |
27
+ Which version of packages.
28
+
29
+ Please install `session-info2`, run the following command in a notebook,
30
+ click the “Copy as Markdown” button, then paste the results into the text box below.
31
+
32
+ ```python
33
+ In[1]: import session_info2; session_info2.session_info(dependencies=True)
34
+ ```
35
+
36
+ Alternatively, run this in a console:
37
+
38
+ ```python
39
+ >>> import session_info2; print(session_info2.session_info(dependencies=True)._repr_mimebundle_()["text/markdown"])
40
+ ```
41
+ render: python
42
+ placeholder: |
43
+ anndata 0.11.3
44
+ ---- ----
45
+ charset-normalizer 3.4.1
46
+ coverage 7.7.0
47
+ psutil 7.0.0
48
+ dask 2024.7.1
49
+ jaraco.context 5.3.0
50
+ numcodecs 0.15.1
51
+ jaraco.functools 4.0.1
52
+ Jinja2 3.1.6
53
+ sphinxcontrib-jsmath 1.0.1
54
+ sphinxcontrib-htmlhelp 2.1.0
55
+ toolz 1.0.0
56
+ session-info2 0.1.2
57
+ PyYAML 6.0.2
58
+ llvmlite 0.44.0
59
+ scipy 1.15.2
60
+ pandas 2.2.3
61
+ sphinxcontrib-devhelp 2.0.0
62
+ h5py 3.13.0
63
+ tblib 3.0.0
64
+ setuptools-scm 8.2.0
65
+ more-itertools 10.3.0
66
+ msgpack 1.1.0
67
+ sparse 0.15.5
68
+ wrapt 1.17.2
69
+ jaraco.collections 5.1.0
70
+ numba 0.61.0
71
+ pyarrow 19.0.1
72
+ pytz 2025.1
73
+ MarkupSafe 3.0.2
74
+ crc32c 2.7.1
75
+ sphinxcontrib-qthelp 2.0.0
76
+ sphinxcontrib-serializinghtml 2.0.0
77
+ zarr 2.18.4
78
+ asciitree 0.3.3
79
+ six 1.17.0
80
+ sphinxcontrib-applehelp 2.0.0
81
+ numpy 2.1.3
82
+ cloudpickle 3.1.1
83
+ sphinxcontrib-bibtex 2.6.3
84
+ natsort 8.4.0
85
+ jaraco.text 3.12.1
86
+ setuptools 76.1.0
87
+ Deprecated 1.2.18
88
+ packaging 24.2
89
+ python-dateutil 2.9.0.post0
90
+ ---- ----
91
+ Python 3.13.2 | packaged by conda-forge | (main, Feb 17 2025, 14:10:22) [GCC 13.3.0]
92
+ OS Linux-6.11.0-109019-tuxedo-x86_64-with-glibc2.39
93
+ Updated 2025-03-18 15:47
@@ -0,0 +1,5 @@
1
+ blank_issues_enabled: false
2
+ contact_links:
3
+ - name: Scverse Community Forum
4
+ url: https://discourse.scverse.org/
5
+ about: If you have questions about “How to do X”, please ask them here.
@@ -0,0 +1,11 @@
1
+ name: Feature request
2
+ description: Propose a new feature for biotapy
3
+ labels: enhancement
4
+ body:
5
+ - type: textarea
6
+ id: description
7
+ attributes:
8
+ label: Description of feature
9
+ description: Please describe your suggestion for a new feature. It might help to describe a problem or use case, plus any alternatives that you have considered.
10
+ validations:
11
+ required: true
@@ -0,0 +1,12 @@
1
+ version: 2
2
+ updates:
3
+ - package-ecosystem: github-actions
4
+ directory: /
5
+ schedule:
6
+ interval: weekly
7
+ cooldown:
8
+ default-days: 7
9
+ groups:
10
+ actions-deps:
11
+ patterns:
12
+ - "*"
@@ -0,0 +1,13 @@
1
+ ## Task
2
+ Phase / task id from `.knowledge/roadmap/`:
3
+
4
+ ## Checklist (rules.md)
5
+ - [ ] Task is in the active phase (R1.1); nothing outside it changed (R1.4)
6
+ - [ ] Reused a library call, or explained why none fits (R2.1): `<call>`
7
+ - [ ] No new parameter, abstraction or dependency without a test or approval (R2.3, R9.1)
8
+ - [ ] Docstring (R equivalent, Guide, example) and docs page updated (R8.2)
9
+ - [ ] Knowledge concepts updated, or confirmed unaffected (R12.1)
10
+ - [ ] Compiled engine only: asv benchmark >= 5x attached (R10.3)
11
+
12
+ ## Verification
13
+ Commands run and their result (R14):
@@ -0,0 +1,30 @@
1
+ name: Check Build
2
+
3
+ on:
4
+ push:
5
+ branches: [master]
6
+ pull_request:
7
+ branches: [master]
8
+
9
+ concurrency:
10
+ group: ${{ github.workflow }}-${{ github.ref }}
11
+ cancel-in-progress: true
12
+
13
+ permissions:
14
+ contents: read
15
+
16
+ jobs:
17
+ package:
18
+ runs-on: ubuntu-latest
19
+ steps:
20
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
21
+ with:
22
+ filter: blob:none
23
+ fetch-depth: 0
24
+ persist-credentials: false
25
+ - name: Install uv
26
+ uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0
27
+ - name: Build package
28
+ run: uv build
29
+ - name: Check package
30
+ run: uvx twine check --strict dist/*.whl
@@ -0,0 +1,33 @@
1
+ name: Release
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+
7
+ # Use "trusted publishing", see https://docs.pypi.org/trusted-publishers/
8
+ permissions: {}
9
+
10
+ jobs:
11
+ release:
12
+ name: Upload release to PyPI
13
+ runs-on: ubuntu-latest
14
+ environment:
15
+ name: pypi
16
+ url: https://pypi.org/p/biotapy
17
+ permissions:
18
+ contents: read
19
+ id-token: write # IMPORTANT: this permission is mandatory for trusted publishing
20
+ steps:
21
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
22
+ with:
23
+ filter: blob:none
24
+ fetch-depth: 0
25
+ persist-credentials: false
26
+ - name: Install uv
27
+ uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0
28
+ with:
29
+ enable-cache: false
30
+ - name: Build package
31
+ run: uv build
32
+ - name: Publish package distributions to PyPI
33
+ uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
@@ -0,0 +1,168 @@
1
+ name: Test
2
+
3
+ on:
4
+ push:
5
+ branches: [master]
6
+ pull_request:
7
+ branches: [master]
8
+ schedule:
9
+ - cron: "0 5 1,15 * *"
10
+
11
+ concurrency:
12
+ group: ${{ github.workflow }}-${{ github.ref }}
13
+ cancel-in-progress: true
14
+
15
+ permissions:
16
+ contents: read
17
+
18
+ jobs:
19
+ # Get the test environment from hatch as defined in pyproject.toml.
20
+ # This ensures that the pyproject.toml is the single point of truth for test definitions and the same tests are
21
+ # run locally and on continuous integration.
22
+ # Check [[tool.hatch.envs.hatch-test.matrix]] in pyproject.toml and https://hatch.pypa.io/latest/environment/ for
23
+ # more details.
24
+ get-environments:
25
+ runs-on: ubuntu-slim
26
+ outputs:
27
+ envs: ${{ steps.get-envs.outputs.envs }}
28
+ steps:
29
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
30
+ with:
31
+ filter: blob:none
32
+ fetch-depth: 0
33
+ persist-credentials: false
34
+ - name: Install uv
35
+ uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0
36
+ - name: Get test environments
37
+ id: get-envs
38
+ run: |
39
+ ENVS_JSON=$(uvx hatch env show --json | jq -c 'to_entries
40
+ | map(
41
+ select(.key | startswith("hatch-test"))
42
+ | {
43
+ name: .key,
44
+ label: (if (.key | contains("pre")) then .key + " (PRE-RELEASE DEPENDENCIES)" else .key end),
45
+ python: .value.python
46
+ }
47
+ )')
48
+ echo "envs=${ENVS_JSON}" | tee $GITHUB_OUTPUT
49
+
50
+ # Run tests through hatch. Spawns a separate runner for each environment defined in the hatch matrix obtained above.
51
+ test:
52
+ needs: get-environments
53
+ permissions:
54
+ id-token: write # for codecov OIDC
55
+ contents: read
56
+
57
+ strategy:
58
+ fail-fast: false
59
+ matrix:
60
+ os: [ubuntu-latest, macos-latest, windows-latest]
61
+ env: ${{ fromJSON(needs.get-environments.outputs.envs) }}
62
+
63
+ name: ${{ matrix.env.label }} (${{ matrix.os }})
64
+ runs-on: ${{ matrix.os }}
65
+ defaults:
66
+ run:
67
+ shell: bash # steps use POSIX syntax; Windows defaults to PowerShell
68
+ continue-on-error: ${{ contains(matrix.env.name, 'pre') }} # make "all-green" pass even if pre-release job fails
69
+
70
+ steps:
71
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
72
+ with:
73
+ filter: blob:none
74
+ fetch-depth: 0
75
+ persist-credentials: false
76
+ - name: Install uv
77
+ uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0
78
+ with:
79
+ python-version: ${{ matrix.env.python }}
80
+ - name: create hatch environment
81
+ run: uvx hatch env create ${{ matrix.env.name }}
82
+ - name: list all all installed package versions
83
+ run: uvx hatch run ${{ matrix.env.name }}:uv pip list
84
+ - name: run tests using hatch
85
+ env:
86
+ MPLBACKEND: agg
87
+ PLATFORM: ${{ matrix.os }}
88
+ DISPLAY: :42
89
+ run: uvx hatch run ${{ matrix.env.name }}:run-cov -v --color=yes -n auto
90
+ - name: generate coverage report
91
+ run: |
92
+ # See https://coverage.readthedocs.io/page/config.html#run-patch
93
+ test -f .coverage || uvx hatch run ${{ matrix.env.name }}:cov-combine
94
+ uvx hatch run ${{ matrix.env.name }}:cov-report # report visibly
95
+ uvx hatch run ${{ matrix.env.name }}:coverage xml # create report for upload
96
+ - name: Upload coverage
97
+ uses: codecov/codecov-action@303a32d7a59b442fa8d48b6a1cc6825c09c847a5 # v7.1.1
98
+ with:
99
+ fail_ci_if_error: true
100
+ use_oidc: true
101
+
102
+ # The rules.md gate: ruff (lint, format, size limits, tree-import bans), mypy --strict,
103
+ # import-linter layers, pyproject-fmt, zizmor. Same hooks as local prek (rules.md R14).
104
+ lint:
105
+ runs-on: ubuntu-latest
106
+ steps:
107
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
108
+ with:
109
+ filter: blob:none
110
+ fetch-depth: 0
111
+ persist-credentials: false
112
+ - name: Install uv
113
+ uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0
114
+ - name: Run all prek hooks
115
+ run: uvx prek run --all-files --show-diff-on-failure
116
+
117
+ # Optional dependencies must be imported lazily (decisions/optional-heavy-dependencies):
118
+ # importing with runtime dependencies only catches a module-level import of an extra.
119
+ import-without-extras:
120
+ runs-on: ubuntu-latest
121
+ steps:
122
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
123
+ with:
124
+ filter: blob:none
125
+ fetch-depth: 0
126
+ persist-credentials: false
127
+ - name: Install uv
128
+ uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0
129
+ - name: Import every biotapy module with runtime dependencies only
130
+ run: >-
131
+ uv run --no-dev python -c "import importlib, pkgutil, biotapy;
132
+ [importlib.import_module(m.name) for m in pkgutil.walk_packages(biotapy.__path__, 'biotapy.')]"
133
+
134
+ # Informational: lists knowledge concepts whose code changed in this PR (playbooks/maintain-knowledge).
135
+ # Flagged concepts go to the job summary; the job fails only when the script cannot diff (exit 2).
136
+ knowledge-touched:
137
+ if: github.event_name == 'pull_request'
138
+ runs-on: ubuntu-latest
139
+ steps:
140
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
141
+ with:
142
+ filter: blob:none
143
+ fetch-depth: 0
144
+ persist-credentials: false
145
+ - name: Report concepts covering changed code
146
+ env:
147
+ BASE: ${{ github.base_ref }}
148
+ run: |
149
+ status=0
150
+ bash scripts/knowledge_stale.sh --touched --against "origin/$BASE" > knowledge-report.txt || status=$?
151
+ { echo "### Knowledge concepts to review"; echo '```'; cat knowledge-report.txt; echo '```'; } >> "$GITHUB_STEP_SUMMARY"
152
+ test "$status" -le 1
153
+
154
+ # Check that all tests defined above pass. This makes it easy to set a single "required" test in branch
155
+ # protection instead of having to update it frequently. See https://github.com/re-actors/alls-green#why.
156
+ check:
157
+ name: Tests pass in all hatch environments
158
+ if: always()
159
+ needs:
160
+ - get-environments
161
+ - test
162
+ - lint
163
+ - import-without-extras
164
+ runs-on: ubuntu-latest
165
+ steps:
166
+ - uses: re-actors/alls-green@b5b5b37504aa4183270bd3d855c52a67f212be35 # v1.3.0
167
+ with:
168
+ jobs: ${{ toJSON(needs) }}
@@ -0,0 +1,23 @@
1
+ # Temp files
2
+ .DS_Store
3
+ *~
4
+ buck-out/
5
+
6
+ # Compiled files
7
+ .venv/
8
+ __pycache__/
9
+ .*cache/
10
+
11
+ # Distribution / packaging
12
+ /dist/
13
+ # Library: resolve fresh in CI and for users; no committed lockfile
14
+ /uv.lock
15
+
16
+ # Tests and coverage
17
+ /data/
18
+ /node_modules/
19
+ /.coverage*
20
+
21
+ # docs
22
+ /docs/generated/
23
+ /docs/_build/
@@ -0,0 +1,82 @@
1
+ ---
2
+ type: Contract
3
+ title: Data-model slots
4
+ description: Which AnnData/TreeData slot holds what, the exact result keys, the x_kind and provenance conventions, and which slots feature-changing operations drop.
5
+ tags: [data-model, api]
6
+ status: stable
7
+ paths: ["src/biotapy/_core/**", "src/biotapy/io/**", "src/biotapy/pp/**", "src/biotapy/tl/**"]
8
+ generated: { by: claude-code/claude-opus-5-5, at: 2026-09-26T08:21:10Z }
9
+ commit: b77a226
10
+ sources:
11
+ - id: spec
12
+ resource: ../../plan.md
13
+ title: Python Microbiome Toolkit development report
14
+ author: human:pedrocr83
15
+ - id: treedata
16
+ resource: https://github.com/YosefLab/treedata
17
+ title: treedata source (0.3.1)
18
+ - id: phyloseq-glom
19
+ resource: https://github.com/joey711/phyloseq/blob/master/R/transform_filter-methods.R
20
+ title: phyloseq tax_glom source
21
+ ---
22
+
23
+ # Statement
24
+
25
+ ## Slots
26
+ Extends the spec's data-model table with exact keys.[^spec]
27
+
28
+ | Slot | Holds | Keys |
29
+ |---|---|---|
30
+ | `X` | samples x features, `scipy.sparse.csr_matrix` | kind recorded in `uns["biotapy"]["x_kind"]` |
31
+ | `layers` | same-shape transforms of `X` | `relative`, `clr` |
32
+ | `obs` | sample metadata; `tl` per-sample results with `inplace=True` | `alpha_<metric>` (e.g. `alpha_shannon`) |
33
+ | `var` | taxonomy, one lowercase column per rank; sequences | ranks from `kingdom, phylum, class, order, family, genus, species`; `sequence` |
34
+ | `vart` | phylogeny as `networkx.DiGraph`, leaves = `var_names`, edge attribute `length` | `phylo` only |
35
+ | `obsm` | ordinations and embeddings | `X_pcoa`, `X_nmds`, `X_<plugin>` |
36
+ | `obsp` | sample-sample distance matrices | metric name: `braycurtis`, `jaccard`, `unweighted_unifrac`, `weighted_unifrac` |
37
+ | `uns["biotapy"]` | biotapy metadata, nothing else | `x_kind`, `provenance`, `pcoa` (eigenvalues, proportion explained) |
38
+
39
+ ## Conventions
40
+ 1. **Missing taxonomy** is `NaN`. Readers convert `""`, whitespace, `"NA"`, and
41
+ bare prefixes (`"g__"`) to `NaN`, strip `k__`-style prefixes, and map rank
42
+ aliases (`domain` -> `kingdom`) to the canonical lowercase names.
43
+ 2. **`x_kind`** is one of `counts`, `relative`, `rpk`, `cpm`, `abundance`.
44
+ Readers always set it. Missing key means `counts`. Functions that need raw
45
+ counts (rarefy, chao1) call `_core.require_counts` and raise otherwise.
46
+ 3. **Provenance** is `uns["biotapy"]["provenance"]`: a list of JSON strings
47
+ `{"step", "version", "params"}`, appended by `_core.add_provenance`.
48
+ JSON strings, not dicts, because h5ad cannot store a list of dicts.
49
+ 4. **Trees** are created only by `_core` ([tree-access](/contracts/tree-access.md))
50
+ with `label=None`, so TreeData adds no `tree` column to `var`.
51
+
52
+ ## Propagation
53
+ | Operation | Keeps | Drops |
54
+ |---|---|---|
55
+ | Feature-changing (`pp.filter_features`, `pp.tax_glom`, `pp.rarefy`) | `obs`, `var` rows kept, `vart` (pruned by TreeData), `uns["biotapy"]` | all `layers`, `obsm`, `obsp`, `varm`, `varp`, other `uns` keys |
56
+ | Sample-only (`pp.filter_samples`) | everything, subset by AnnData indexing | nothing |
57
+ | Layer-adding (`pp.relative`, `pp.clr`) | everything | nothing; adds one layer |
58
+
59
+ Feature-changing operations go through `_core.feature_subset`, the single place
60
+ that implements the "Drops" column.
61
+
62
+ ## Aggregation semantics (`tax_glom`)
63
+ Matches phyloseq:[^phyloseq-glom] features are grouped by the full lineage up
64
+ to the rank (not the rank value alone, so `uncultured` genera in different
65
+ families stay separate); the representative ("archetype") is the most abundant
66
+ feature, first on ties; ranks below the target become `NaN`. The kept tree is
67
+ the archetypes' subtree; TreeData keeps unary nodes, which leaves root-to-tip
68
+ path lengths, and therefore Faith PD and UniFrac, unchanged.[^treedata]
69
+
70
+ # Why
71
+ A slot whose meaning depends on which function wrote it cannot be trusted by
72
+ the next function. Dropping derived slots on feature changes prevents stale
73
+ distances or ordinations from being plotted against new data.
74
+
75
+ # Enforced by
76
+ - `tests/core/test_slots.py` (Phase 1, task 1.2) for `feature_subset` and provenance.
77
+ - `tests/datasets/test_toy.py` (task 1.3): h5td round-trip keeps every convention.
78
+ - Per-function tests assert the documented keys.
79
+
80
+ [^spec]: Python Microbiome Toolkit development report, section Data model
81
+ [^treedata]: treedata source (0.3.1)
82
+ [^phyloseq-glom]: phyloseq tax_glom source
@@ -0,0 +1,38 @@
1
+ ---
2
+ type: Contract
3
+ title: Engine parity
4
+ description: A compiled kernel is a drop-in behind an existing public function via `engine=`; the Python engine is the test oracle and both must match within a stated tolerance.
5
+ tags: [performance, testing]
6
+ status: stable
7
+ paths: ["src/biotapy/**/*.py", "rust/**", "benchmarks/**"]
8
+ generated: { by: claude-code/claude-opus-5-5, at: 2026-09-26T08:21:10Z }
9
+ commit: b77a226
10
+ sources:
11
+ - id: spec
12
+ resource: ../../plan.md
13
+ title: Python Microbiome Toolkit development report
14
+ author: human:pedrocr83
15
+ ---
16
+
17
+ # Statement
18
+ 1. An engine is selected by `engine: Literal["python", "numba", "rust"] = "python"`
19
+ (or a delegated library's own switch, exposed under the same name).
20
+ 2. Adding an engine never changes the public signature or return type.
21
+ 3. The `"python"` engine is never removed; it is the reference oracle.
22
+ 4. The same test function runs across every available engine
23
+ (`@pytest.mark.parametrize("engine", available_engines())`) with an explicit
24
+ `rtol`/`atol` written in the test.
25
+ 5. Merge requires an `asv` benchmark on the 5,000 x 50,000 sparse synthetic
26
+ dataset showing >= 5x speedup, plus a note on the function's docs page.
27
+ 6. An engine whose optional dependency is missing raises `ImportError` naming
28
+ the extra; it never silently falls back to another engine.
29
+
30
+ # Why
31
+ Without an oracle, a fast kernel that is subtly wrong ships unnoticed; without
32
+ the 5x bar, compiled code accumulates for marginal gains and raises the
33
+ contributor bar for nothing. See [python-first-compiled-last](/decisions/python-first-compiled-last.md).
34
+
35
+ # Enforced by
36
+ - Parametrized engine tests (first instance: Phase 1 perf track, if any).
37
+ - PR template checkbox "benchmark attached" (Phase 0, task 0.6).
38
+ - Not otherwise automated: reviewers must check the asv result.
@@ -0,0 +1,73 @@
1
+ ---
2
+ type: Contract
3
+ title: Public function shape
4
+ description: One task = one public function `verb(data, required, *, options) -> result`, fully typed, keyword-only options, seeded randomness, NumPy docstring with a parseable R-equivalent line.
5
+ tags: [api, conventions, docs]
6
+ status: stable
7
+ paths: ["src/biotapy/**/*.py"]
8
+ generated: { by: claude-code/claude-opus-5-5, at: 2026-09-26T08:21:10Z }
9
+ commit: b77a226
10
+ sources:
11
+ - id: spec
12
+ resource: ../../plan.md
13
+ title: Python Microbiome Toolkit development report
14
+ author: human:pedrocr83
15
+ ---
16
+
17
+ # Statement
18
+
19
+ Every public function in `io`, `datasets`, `pp`, `tl`, `fn`, `da`, `ml`, `pl`:
20
+
21
+ 1. **Signature**: `verb(data, <required args>, *, <options>) -> <result>`.
22
+ - `data` is annotated with the widest type that works: `AnnData` when no tree
23
+ is needed, `TreeData` when `vart` is read, `MuData` for multi-modal.
24
+ - Everything after the required arguments is keyword-only (`*`).
25
+ - No `**kwargs` pass-through, except a documented `plot_kwargs` in `pl`.
26
+ 2. **Return and mutation**: per [pure-by-default](/decisions/pure-by-default.md).
27
+ 3. **Randomness**: any stochastic function takes
28
+ `seed: int | np.random.Generator | None = None`, converted once with
29
+ `biotapy._core.as_generator(seed)`. No global RNG state is read or set.
30
+ 4. **Types**: full hints on parameters and return; no `Any`, no untyped
31
+ `dict`/`list` in public signatures; `Literal[...]` for string options.
32
+ 5. **Validation**: inputs are checked at the public boundary with
33
+ `biotapy._core` validators; errors are `ValueError`/`KeyError`/`TypeError`
34
+ naming the offending argument. Private helpers do not re-validate.
35
+ 6. **Docstring** (NumPy style, enforced by ruff `D` rules):
36
+
37
+ ```text
38
+ One-line summary ending with a period.
39
+
40
+ Parameters / Returns sections.
41
+
42
+ Notes
43
+ -----
44
+ R equivalent: ``phyloseq::tax_glom``, ``mia::agglomerateByRank``
45
+ Guide: :doc:`/guide/aggregation`
46
+
47
+ Examples
48
+ --------
49
+ >>> import biotapy as bt
50
+ >>> tdata = bt.datasets.toy()
51
+ >>> bt.pp.tax_glom(tdata, "phylum").n_vars
52
+ 3
53
+ ```
54
+
55
+ - Exactly one line starting `R equivalent:`; comma-separated
56
+ ``pkg::fn`` items, or `none`. The "Coming from R" page is generated from it.
57
+ - Examples use `bt.datasets.toy()` (built in memory, no download) and run
58
+ under doctest in CI.
59
+
60
+ # Why
61
+ - Keyword-only options let parameters be added without breaking callers.
62
+ - A seeded generator is the only way stochastic results are reproducible and testable.
63
+ - A parseable `R equivalent:` line keeps the migration table generated, never hand-written.
64
+
65
+ # Enforced by
66
+ - ruff `D`, `PLR0913`, `PLR0917` and mypy strict (Phase 0, task 0.4).
67
+ - `tests/test_docstrings.py` (Phase 1, task 1.19) parses every public function's `R equivalent:` line.
68
+ - Doctests in CI (`--doctest-modules` in pytest config, Phase 0 task 0.4).
69
+ - Purity: every `pp`/`tl` test asserts the input is unchanged.
70
+
71
+ # Binds
72
+ - [module-boundaries](/contracts/module-boundaries.md)
73
+ - [data-model-slots](/contracts/data-model-slots.md)