datalad-worktree 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datalad_worktree-0.4.0/.githooks/pre-commit +25 -0
- datalad_worktree-0.4.0/.github/workflows/publish.yml +33 -0
- datalad_worktree-0.4.0/.github/workflows/tests.yml +44 -0
- datalad_worktree-0.4.0/.gitignore +35 -0
- datalad_worktree-0.4.0/CHANGELOG.md +39 -0
- datalad_worktree-0.4.0/CLAUDE.md +47 -0
- datalad_worktree-0.4.0/CONTRIBUTING.md +51 -0
- datalad_worktree-0.4.0/PKG-INFO +105 -0
- datalad_worktree-0.4.0/README.md +94 -0
- datalad_worktree-0.4.0/docs/cli.md +147 -0
- datalad_worktree-0.4.0/docs/design.md +33 -0
- datalad_worktree-0.4.0/pyproject.toml +48 -0
- datalad_worktree-0.4.0/src/datalad_worktree/__init__.py +14 -0
- datalad_worktree-0.4.0/src/datalad_worktree/__main__.py +6 -0
- datalad_worktree-0.4.0/src/datalad_worktree/add.py +685 -0
- datalad_worktree-0.4.0/src/datalad_worktree/cli.py +433 -0
- datalad_worktree-0.4.0/src/datalad_worktree/container.py +254 -0
- datalad_worktree-0.4.0/src/datalad_worktree/core.py +195 -0
- datalad_worktree-0.4.0/src/datalad_worktree/delete.py +410 -0
- datalad_worktree-0.4.0/src/datalad_worktree/discovery.py +294 -0
- datalad_worktree-0.4.0/src/datalad_worktree/dl_command.py +684 -0
- datalad_worktree-0.4.0/src/datalad_worktree/fetch.py +433 -0
- datalad_worktree-0.4.0/src/datalad_worktree/list_cmd.py +167 -0
- datalad_worktree-0.4.0/src/datalad_worktree/mtimes.py +457 -0
- datalad_worktree-0.4.0/tests/__init__.py +0 -0
- datalad_worktree-0.4.0/tests/conftest.py +116 -0
- datalad_worktree-0.4.0/tests/fake_container_runtime.py +83 -0
- datalad_worktree-0.4.0/tests/test_add.py +392 -0
- datalad_worktree-0.4.0/tests/test_cli.py +55 -0
- datalad_worktree-0.4.0/tests/test_container.py +149 -0
- datalad_worktree-0.4.0/tests/test_containers_run.py +218 -0
- datalad_worktree-0.4.0/tests/test_delete.py +381 -0
- datalad_worktree-0.4.0/tests/test_dl_command.py +113 -0
- datalad_worktree-0.4.0/tests/test_fetch.py +357 -0
- datalad_worktree-0.4.0/tests/test_list.py +95 -0
- datalad_worktree-0.4.0/tests/test_mtimes.py +301 -0
- datalad_worktree-0.4.0/uv.lock +948 -0
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
#!/bin/sh
|
|
2
|
+
# Run the same lint CI runs, before the commit rather than after the push.
|
|
3
|
+
#
|
|
4
|
+
# Enable with: git config core.hooksPath .githooks
|
|
5
|
+
#
|
|
6
|
+
# NOTE: git hooks do not fire for `jj commit` -- jujutsu runs no hooks at all.
|
|
7
|
+
# If you commit with jj, use `jj lint` / `jj fix` instead; see CONTRIBUTING.md.
|
|
8
|
+
|
|
9
|
+
set -e
|
|
10
|
+
|
|
11
|
+
if ! command -v uv >/dev/null 2>&1; then
|
|
12
|
+
echo "pre-commit: uv not on PATH, skipping ruff" >&2
|
|
13
|
+
exit 0
|
|
14
|
+
fi
|
|
15
|
+
|
|
16
|
+
# Whole tree, not just staged files: ruff on this repo takes well under a
|
|
17
|
+
# second, and a per-file check would miss an unused import left behind in a
|
|
18
|
+
# file this commit does not touch.
|
|
19
|
+
if ! uv run --dev ruff check .; then
|
|
20
|
+
echo >&2
|
|
21
|
+
echo "pre-commit: ruff found problems (this is what CI's lint job checks)." >&2
|
|
22
|
+
echo " fix them: uv run --dev ruff check --fix ." >&2
|
|
23
|
+
echo " commit anyway: git commit --no-verify" >&2
|
|
24
|
+
exit 1
|
|
25
|
+
fi
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
build:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
steps:
|
|
11
|
+
- uses: actions/checkout@v7
|
|
12
|
+
- uses: astral-sh/setup-uv@v10.1.0
|
|
13
|
+
- name: Build package
|
|
14
|
+
run: uv build
|
|
15
|
+
- uses: actions/upload-artifact@v7
|
|
16
|
+
with:
|
|
17
|
+
name: dist
|
|
18
|
+
path: dist/
|
|
19
|
+
|
|
20
|
+
publish:
|
|
21
|
+
name: Upload release to PyPI
|
|
22
|
+
needs: build
|
|
23
|
+
runs-on: ubuntu-latest
|
|
24
|
+
environment: pypi
|
|
25
|
+
permissions:
|
|
26
|
+
id-token: write
|
|
27
|
+
steps:
|
|
28
|
+
- uses: actions/download-artifact@v8
|
|
29
|
+
with:
|
|
30
|
+
name: dist
|
|
31
|
+
path: dist/
|
|
32
|
+
- name: Publish package distributions to PyPI
|
|
33
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
name: Tests
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [master]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
fail-fast: false
|
|
13
|
+
matrix:
|
|
14
|
+
python-version: ["3.11", "3.12", "3.13"]
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v7
|
|
17
|
+
|
|
18
|
+
- name: Install git-annex
|
|
19
|
+
run: sudo apt-get update && sudo apt-get install -y git-annex
|
|
20
|
+
|
|
21
|
+
- name: Configure git identity for tests
|
|
22
|
+
run: |
|
|
23
|
+
git config --global user.email "test@example.com"
|
|
24
|
+
git config --global user.name "Test User"
|
|
25
|
+
|
|
26
|
+
- uses: astral-sh/setup-uv@v10.1.0
|
|
27
|
+
with:
|
|
28
|
+
python-version: ${{ matrix.python-version }}
|
|
29
|
+
|
|
30
|
+
- name: Install dependencies
|
|
31
|
+
run: uv sync --dev --locked
|
|
32
|
+
|
|
33
|
+
- name: Run tests
|
|
34
|
+
run: uv run pytest -v
|
|
35
|
+
|
|
36
|
+
lint:
|
|
37
|
+
runs-on: ubuntu-latest
|
|
38
|
+
steps:
|
|
39
|
+
- uses: actions/checkout@v7
|
|
40
|
+
- uses: astral-sh/setup-uv@v10.1.0
|
|
41
|
+
- name: Install dependencies
|
|
42
|
+
run: uv sync --dev --locked
|
|
43
|
+
- name: Run ruff
|
|
44
|
+
run: uv run ruff check .
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
*.so
|
|
6
|
+
|
|
7
|
+
# Distribution / packaging
|
|
8
|
+
dist/
|
|
9
|
+
build/
|
|
10
|
+
*.egg-info/
|
|
11
|
+
*.egg
|
|
12
|
+
|
|
13
|
+
# Virtual environments
|
|
14
|
+
.venv/
|
|
15
|
+
venv/
|
|
16
|
+
env/
|
|
17
|
+
|
|
18
|
+
# IDE
|
|
19
|
+
.idea/
|
|
20
|
+
.vscode/
|
|
21
|
+
*.swp
|
|
22
|
+
*.swo
|
|
23
|
+
*~
|
|
24
|
+
|
|
25
|
+
# OS
|
|
26
|
+
.DS_Store
|
|
27
|
+
Thumbs.db
|
|
28
|
+
|
|
29
|
+
# Testing
|
|
30
|
+
.pytest_cache/
|
|
31
|
+
.coverage
|
|
32
|
+
htmlcov/
|
|
33
|
+
|
|
34
|
+
# uv
|
|
35
|
+
.python-version
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
User-facing changes per release. Behaviour is documented in [docs/cli.md](docs/cli.md).
|
|
4
|
+
|
|
5
|
+
## 0.4.0 (2026-10-05)
|
|
6
|
+
|
|
7
|
+
### Breaking
|
|
8
|
+
|
|
9
|
+
- `remove` is renamed `delete`; `worktree` alone runs `list`.
|
|
10
|
+
- `add` takes `<branch> <worktree-path>`, in that order.
|
|
11
|
+
- `add` replaces an existing worktree by default; `-f` now only discards commits the main checkout lacks.
|
|
12
|
+
- `add` always starts the branch at each dataset's HEAD; `--no-create-branch` is gone.
|
|
13
|
+
- `delete` deletes the branch by default (`--keep-branch` opts out), discards uncommitted changes, and no longer prompts; `--delete-branch` and `-y` are gone.
|
|
14
|
+
- `--no-color` is gone; colour is off when output is not a terminal.
|
|
15
|
+
- Python API: `create_nested_worktrees(replace=)` is gone.
|
|
16
|
+
|
|
17
|
+
### Added
|
|
18
|
+
|
|
19
|
+
- `worktree fetch`: bring a worktree's results home, or new code and inputs into it.
|
|
20
|
+
- `add --follow-parent [<commit>]`: check subdatasets out at the commits their parent records.
|
|
21
|
+
- `add` configures container bind mounts (`--no-bindpaths` skips).
|
|
22
|
+
- `add` copies mtimes from the main checkout, Snakemake markers included (`--no-mtimes` skips).
|
|
23
|
+
- `delete -n`.
|
|
24
|
+
|
|
25
|
+
### Changed
|
|
26
|
+
|
|
27
|
+
- `add` and `delete` are all-or-nothing: every check runs first, and `-n` runs the same checks.
|
|
28
|
+
- `add` rolls back the worktrees it created if git fails partway.
|
|
29
|
+
- `list` groups by hierarchy and prunes worktrees deleted by other means.
|
|
30
|
+
- Pastel report colours, `delete` in pink.
|
|
31
|
+
|
|
32
|
+
### Fixed
|
|
33
|
+
|
|
34
|
+
- The main working tree is never deleted.
|
|
35
|
+
- Works with current DataLad (no deprecated `eval_results` import).
|
|
36
|
+
|
|
37
|
+
## 0.3.0 and earlier
|
|
38
|
+
|
|
39
|
+
See the git history.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# CLAUDE.md
|
|
2
|
+
|
|
3
|
+
Guidance for coding agents working on this repository. This file holds rules and pointers, not explanations. Each fact lives in exactly one place, so read it there:
|
|
4
|
+
|
|
5
|
+
| What | Where |
|
|
6
|
+
|---|---|
|
|
7
|
+
| What the tool is for, install | [README.md](README.md) |
|
|
8
|
+
| What each command and flag does | [docs/cli.md](docs/cli.md) |
|
|
9
|
+
| **Why**, across commands: the premise, containers, mtimes | [docs/design.md](docs/design.md) |
|
|
10
|
+
| **Why** a command or module is shaped the way it is, with the evidence | its module and function docstrings |
|
|
11
|
+
| Setup, test and lint commands, jj hooks | [CONTRIBUTING.md](CONTRIBUTING.md) |
|
|
12
|
+
| What changed in each release | [CHANGELOG.md](CHANGELOG.md) |
|
|
13
|
+
|
|
14
|
+
## Before changing behaviour
|
|
15
|
+
|
|
16
|
+
- **Read design.md and the docstrings of the code you are changing first.** Much of the code looks simplifiable and isn't: the obvious alternative was usually tried and broke something real, and the docstring next to it records what. If your change contradicts design.md or such a docstring, raise that with the maintainer instead of quietly working around it.
|
|
17
|
+
- **Some guards are pinned by tests that fail when the guard is removed.** Don't weaken these to get a test passing:
|
|
18
|
+
- the `mtimes.py` safety rules and their `tests/test_mtimes.py` guards;
|
|
19
|
+
- `_worktree_kind()` in `delete.py` and its tests;
|
|
20
|
+
- `test_fails_without_bindpaths`.
|
|
21
|
+
- **Ask for a design decision instead of picking one.** Changing a default, adding a flag, or changing what gets refused is the maintainer's call. Propose the options and trade-offs, then wait.
|
|
22
|
+
|
|
23
|
+
## Making a change
|
|
24
|
+
|
|
25
|
+
- **User-facing changes reach both front-ends.** `cli.py` (argparse) and `dl_command.py` (the DataLad `Interface` classes) each declare every flag. A new outcome is a `WorktreeResult`, rendered in `cli.py:_render_report()` and mapped in `dl_command.py`.
|
|
26
|
+
- **All git access goes through `subprocess.run(..., capture_output=True, text=True)`.** No gitpython, and no runtime dependencies beyond the standard library. DataLad stays optional.
|
|
27
|
+
- **Resolve datasets against each other with `is_git_repo_root()`, not `is_git_repo()`.** The latter walks up the tree and is true at an empty submodule mount point.
|
|
28
|
+
- **Style:** `from __future__ import annotations`, full type hints, `logger = logging.getLogger(__name__)` per module. `uv run --dev ruff check .` must be clean.
|
|
29
|
+
|
|
30
|
+
## Tests
|
|
31
|
+
|
|
32
|
+
- **Use real repositories, never mocked git.** Tests build DataLad hierarchies in `tmp_path`, so `datalad` is a dev dependency. The full suite takes about 7 minutes and needs Linux. While iterating, run only the affected module.
|
|
33
|
+
- **Every bug fix gets a regression test that fails without the fix.** Show it: revert the fix, watch the test fail, restore it, and say in the PR that you did.
|
|
34
|
+
- **Assert invariants, not pinned values.** That means orderings, equality between trees, and paths left untouched. The generators also yield container and mtime reports, so filter reports by result type instead of counting all of them.
|
|
35
|
+
- **Stamp and check mtimes with `follow_symlinks=False`.** Annexed files are symlinks, so the default touches the shared annex object instead and silently makes staleness tests vacuous.
|
|
36
|
+
- **Some tests need extra capabilities.** `tests/test_containers_run.py` needs `datalad-container` and unprivileged user and mount namespaces, and skips itself without them. Don't turn those skips into failures.
|
|
37
|
+
|
|
38
|
+
## Docs
|
|
39
|
+
|
|
40
|
+
- **Update docs in the same commit as the change:** behaviour in `docs/cli.md`, rationale with its evidence in the docstring of the code it explains, or in `docs/design.md` when it spans commands, and README only when a headline claim changes.
|
|
41
|
+
- **Never put rationale in this file.** Copies drift, so point to where the explanation lives instead.
|
|
42
|
+
|
|
43
|
+
## Commits and PRs
|
|
44
|
+
|
|
45
|
+
- **One concern per commit.** A refactor, a fix, a dependency change and a doc edit are separate commits.
|
|
46
|
+
- **Commit subjects use a prefix:** `BF:` (fix), `ENH:` (feature), `DOC:`, `CI:`, `NF:`, or `API:`. The body explains why, and references the issue (`Closes #N`).
|
|
47
|
+
- **History is linear.** The maintainer uses jj and stacks PRs; rebase, and never merge `master` into a branch.
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Thanks for considering a contribution to `datalad-worktree`.
|
|
4
|
+
|
|
5
|
+
## Setup
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
git clone https://github.com/just-meng/datalad-worktree.git
|
|
9
|
+
cd datalad-worktree
|
|
10
|
+
uv sync --dev
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Running tests and lint
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
uv run pytest
|
|
17
|
+
uv run ruff check .
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
Worktrees of annexed repos and `datalad containers-run` don't work on Windows, so the full test suite needs Linux (or WSL). `tests/test_containers_run.py` additionally needs unprivileged user/mount namespaces and skips itself where those aren't available.
|
|
21
|
+
|
|
22
|
+
### Running lint before you commit
|
|
23
|
+
|
|
24
|
+
CI's `lint` job is `uv run ruff check .`, and finding out from CI costs a round trip. A hook that runs it locally lives in `.githooks/`; enable it with:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
git config core.hooksPath .githooks
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
It checks the whole tree (ruff takes well under a second here), and `git commit --no-verify` bypasses it.
|
|
31
|
+
|
|
32
|
+
**If you commit with [jujutsu](https://jj-vcs.github.io/), that hook never fires** — jj runs no hooks at all, `pre-commit` included, and there is no hook mechanism in its config. Two jj-native equivalents; both are repo-local config, so they have to be set per clone:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
# `jj lint` -- exactly what CI runs
|
|
36
|
+
jj config set --repo aliases.lint \
|
|
37
|
+
'["util", "exec", "--", "uv", "run", "--dev", "ruff", "check", "."]'
|
|
38
|
+
|
|
39
|
+
# `jj fix` -- rewrite the commit with ruff's fixes applied
|
|
40
|
+
jj config set --repo fix.tools.ruff.command \
|
|
41
|
+
'["uv", "run", "--dev", "ruff", "check", "--fix", "--quiet", "--stdin-filename=$path", "-"]'
|
|
42
|
+
jj config set --repo fix.tools.ruff.patterns '["glob:**/*.py"]'
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`jj fix` works because ruff reads a file on stdin and writes the fixed version to stdout, which is the interface `jj fix` expects. It rewrites the working-copy commit by default, or `jj fix -s <rev>` a particular one — handy when CI goes red on a commit in the middle of a stack, since descendants are rebased for you.
|
|
46
|
+
|
|
47
|
+
## Submitting changes
|
|
48
|
+
|
|
49
|
+
Open a pull request against `master`. Keep commits focused -- one logical change per commit, with a message explaining *why*, not just *what*. `uv run ruff check .` should pass clean.
|
|
50
|
+
|
|
51
|
+
Why the code is shaped the way it is: [docs/design.md](docs/design.md) for what spans commands, and the docstrings of the code you are changing for the rest; read both before you change it. Coding agents: see [CLAUDE.md](CLAUDE.md).
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: datalad-worktree
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Create nested git worktrees for DataLad datasets
|
|
5
|
+
Author-email: Jiameng Wu <jiameng.wu@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
|
+
Provides-Extra: datalad
|
|
9
|
+
Requires-Dist: datalad; extra == 'datalad'
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
|
|
12
|
+
# datalad-worktree
|
|
13
|
+
|
|
14
|
+
Nested [git worktrees](https://git-scm.com/docs/git-worktree) for [DataLad](https://www.datalad.org/) dataset hierarchies: create one for a run, ship its results home, dispose of it — or simply repeat.
|
|
15
|
+
|
|
16
|
+
## What it is for
|
|
17
|
+
|
|
18
|
+
DataLad records provenance, Snakemake automates pipelines. Combining the two gives you [an automated computational workflow with staleness detection and full provenance](https://blog.datalad.org/posts/snakemake-datalad-worktree/). A long run then wants a checkout of its own, so that development can continue while it computes — which is exactly what `git worktree` is for.
|
|
19
|
+
|
|
20
|
+
Except a DataLad superdataset is not one repository. It needs a worktree per dataset, wired together so submodule mount points line up, and two things break when you do that:
|
|
21
|
+
|
|
22
|
+
- **`datalad containers-run` cannot read annexed files in a worktree.** `.git` is a link into the main repository, so annex object symlinks resolve outside the worktree, where the container cannot see them — `FileNotFoundError` on every annexed input ([datalad-container#288](https://github.com/datalad/datalad-container/issues/288)).
|
|
23
|
+
- **A fresh worktree looks arbitrarily stale.** Git records content, not timestamps, so for a make-style pipeline the mtime *ordering* that **is** the up-to-date state is gone — and systematically inverted, since the superdataset is checked out before its subdatasets. Every `code/` file ends up newer than every output derived from it: "all your code changed". Bringing results back has the mirror problem — git moves only what it rewrites, so a run that reproduces byte-identical output produces no commit at all, nothing moves, and that output stays stale forever, rerunning on every invocation.
|
|
24
|
+
|
|
25
|
+
`datalad-worktree` fixes both, in one command over the whole hierarchy:
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
cd /data/my-project
|
|
29
|
+
worktree add runs /tmp/worktrees/runs # one worktree per dataset, mtimes preserved,
|
|
30
|
+
# containers configured to reach the annex
|
|
31
|
+
|
|
32
|
+
cd /tmp/worktrees/runs
|
|
33
|
+
snakemake code/Snakefile -c 1 -k # the long run — keep developing in main meanwhile
|
|
34
|
+
|
|
35
|
+
cd /data/my-project
|
|
36
|
+
worktree fetch runs # results home, mtimes included
|
|
37
|
+
worktree delete runs # dispose of it — or overwrite with `worktree add`
|
|
38
|
+
# in the same location next time
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
`add` creates a worktree for the superdataset and every *installed* subdataset, nested the same way; uninstalled ones are skipped:
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
/tmp/worktrees/runs/ <- superdataset worktree (branch: runs)
|
|
45
|
+
├── inputs/ <- subdataset worktree
|
|
46
|
+
├── code/ <- subdataset worktree
|
|
47
|
+
└── results/
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Each is on branch `runs`, starting from that dataset's current state, even if `runs` already exists.
|
|
51
|
+
|
|
52
|
+
`fetch` brings the worktree's commits back into each dataset's main checkout, then copies their mtimes across. `delete` removes the worktree from every dataset along with its branch, and refuses when the worktree contains unfetched commits.
|
|
53
|
+
|
|
54
|
+
## Highlights
|
|
55
|
+
|
|
56
|
+
- **mtimes are preserved in both directions** — creating a worktree and fetching from one — so staleness detection keeps working and finished work is not recomputed. A file inherits a timestamp only when its content is identical on both sides, matched by git's content hash rather than by filename. Snakemake's untracked `.snakemake_timestamp` markers travel too, so an output marked up to date with `snakemake --touch` stays up to date.
|
|
57
|
+
- **`datalad containers-run` works inside the worktree**, via bind-mount configuration written per worktree, so the machine-specific paths stay out of the main checkout and out of history. One empty placeholder is committed on the worktree branch, which is what keeps a run record made there rerunnable elsewhere.
|
|
58
|
+
- **Reproduce any recorded state.** `worktree add [branch] [path] --follow-parent <commit-or-tag>` checks every dataset out exactly as that superdataset commit recorded it, including which subdatasets existed then — to rerun a `datalad run` record from a clean slate, say. Without a commit, subdatasets follow what the superdataset records now.
|
|
59
|
+
- **Results ship home without a clean tree.** Where your side has no commits of its own, `fetch` fast-forwards: nothing is rewritten, so unrelated work in progress is left alone. Where both sides have moved, it merges. It also runs the other way: from inside a worktree, `worktree fetch` brings new code and inputs in.
|
|
60
|
+
- **Safe by default.** Creation and deletion are all-or-nothing: if any dataset would fail, none is touched. Deletion never touches the main working tree or a directory git does not call a worktree; it refuses a worktree with unfetched commits, unless forced, and discards uncommitted changes, as `add` does. Every command that changes something takes `--dry-run`.
|
|
61
|
+
- **No dependencies.** Python standard library plus `git`. DataLad itself is optional — it only adds the `datalad worktree-*` commands.
|
|
62
|
+
|
|
63
|
+
## Installation
|
|
64
|
+
|
|
65
|
+
As a DataLad extension:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
uv tool install datalad \
|
|
69
|
+
--with datalad-worktree@git+https://github.com/just-meng/datalad-worktree.git
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
As a standalone CLI tool:
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
git clone https://github.com/just-meng/datalad-worktree.git
|
|
76
|
+
cd datalad-worktree
|
|
77
|
+
uv sync # --dev for the test suite
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Commands
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
worktree add <branch> <worktree-path> # create nested worktrees
|
|
84
|
+
worktree fetch [target] # bring commits and mtimes into the checkout you stand in
|
|
85
|
+
worktree delete <target> # remove worktrees, deepest-first
|
|
86
|
+
worktree list # show worktrees across the hierarchy (also the default)
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Installed as a DataLad extension — that is, into the same environment as DataLad, as above — each one is also available as `datalad worktree-add`, `worktree-list`, `worktree-delete`, `worktree-fetch`. Installed standalone, only the `worktree` command exists.
|
|
90
|
+
|
|
91
|
+
Flags and behaviour: [docs/cli.md](docs/cli.md). Why it works the way it does: [docs/design.md](docs/design.md).
|
|
92
|
+
|
|
93
|
+
## Requirements
|
|
94
|
+
|
|
95
|
+
- Python >= 3.11
|
|
96
|
+
- git on PATH (>= 2.38 for `fetch` to merge diverged datasets rather than refuse)
|
|
97
|
+
- [DataLad](https://www.datalad.org/), optional
|
|
98
|
+
|
|
99
|
+
## Contributing
|
|
100
|
+
|
|
101
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
102
|
+
|
|
103
|
+
## License
|
|
104
|
+
|
|
105
|
+
MIT
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# datalad-worktree
|
|
2
|
+
|
|
3
|
+
Nested [git worktrees](https://git-scm.com/docs/git-worktree) for [DataLad](https://www.datalad.org/) dataset hierarchies: create one for a run, ship its results home, dispose of it — or simply repeat.
|
|
4
|
+
|
|
5
|
+
## What it is for
|
|
6
|
+
|
|
7
|
+
DataLad records provenance, Snakemake automates pipelines. Combining the two gives you [an automated computational workflow with staleness detection and full provenance](https://blog.datalad.org/posts/snakemake-datalad-worktree/). A long run then wants a checkout of its own, so that development can continue while it computes — which is exactly what `git worktree` is for.
|
|
8
|
+
|
|
9
|
+
Except a DataLad superdataset is not one repository. It needs a worktree per dataset, wired together so submodule mount points line up, and two things break when you do that:
|
|
10
|
+
|
|
11
|
+
- **`datalad containers-run` cannot read annexed files in a worktree.** `.git` is a link into the main repository, so annex object symlinks resolve outside the worktree, where the container cannot see them — `FileNotFoundError` on every annexed input ([datalad-container#288](https://github.com/datalad/datalad-container/issues/288)).
|
|
12
|
+
- **A fresh worktree looks arbitrarily stale.** Git records content, not timestamps, so for a make-style pipeline the mtime *ordering* that **is** the up-to-date state is gone — and systematically inverted, since the superdataset is checked out before its subdatasets. Every `code/` file ends up newer than every output derived from it: "all your code changed". Bringing results back has the mirror problem — git moves only what it rewrites, so a run that reproduces byte-identical output produces no commit at all, nothing moves, and that output stays stale forever, rerunning on every invocation.
|
|
13
|
+
|
|
14
|
+
`datalad-worktree` fixes both, in one command over the whole hierarchy:
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
cd /data/my-project
|
|
18
|
+
worktree add runs /tmp/worktrees/runs # one worktree per dataset, mtimes preserved,
|
|
19
|
+
# containers configured to reach the annex
|
|
20
|
+
|
|
21
|
+
cd /tmp/worktrees/runs
|
|
22
|
+
snakemake code/Snakefile -c 1 -k # the long run — keep developing in main meanwhile
|
|
23
|
+
|
|
24
|
+
cd /data/my-project
|
|
25
|
+
worktree fetch runs # results home, mtimes included
|
|
26
|
+
worktree delete runs # dispose of it — or overwrite with `worktree add`
|
|
27
|
+
# in the same location next time
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
`add` creates a worktree for the superdataset and every *installed* subdataset, nested the same way; uninstalled ones are skipped:
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
/tmp/worktrees/runs/ <- superdataset worktree (branch: runs)
|
|
34
|
+
├── inputs/ <- subdataset worktree
|
|
35
|
+
├── code/ <- subdataset worktree
|
|
36
|
+
└── results/
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Each is on branch `runs`, starting from that dataset's current state, even if `runs` already exists.
|
|
40
|
+
|
|
41
|
+
`fetch` brings the worktree's commits back into each dataset's main checkout, then copies their mtimes across. `delete` removes the worktree from every dataset along with its branch, and refuses when the worktree contains unfetched commits.
|
|
42
|
+
|
|
43
|
+
## Highlights
|
|
44
|
+
|
|
45
|
+
- **mtimes are preserved in both directions** — creating a worktree and fetching from one — so staleness detection keeps working and finished work is not recomputed. A file inherits a timestamp only when its content is identical on both sides, matched by git's content hash rather than by filename. Snakemake's untracked `.snakemake_timestamp` markers travel too, so an output marked up to date with `snakemake --touch` stays up to date.
|
|
46
|
+
- **`datalad containers-run` works inside the worktree**, via bind-mount configuration written per worktree, so the machine-specific paths stay out of the main checkout and out of history. One empty placeholder is committed on the worktree branch, which is what keeps a run record made there rerunnable elsewhere.
|
|
47
|
+
- **Reproduce any recorded state.** `worktree add [branch] [path] --follow-parent <commit-or-tag>` checks every dataset out exactly as that superdataset commit recorded it, including which subdatasets existed then — to rerun a `datalad run` record from a clean slate, say. Without a commit, subdatasets follow what the superdataset records now.
|
|
48
|
+
- **Results ship home without a clean tree.** Where your side has no commits of its own, `fetch` fast-forwards: nothing is rewritten, so unrelated work in progress is left alone. Where both sides have moved, it merges. It also runs the other way: from inside a worktree, `worktree fetch` brings new code and inputs in.
|
|
49
|
+
- **Safe by default.** Creation and deletion are all-or-nothing: if any dataset would fail, none is touched. Deletion never touches the main working tree or a directory git does not call a worktree; it refuses a worktree with unfetched commits, unless forced, and discards uncommitted changes, as `add` does. Every command that changes something takes `--dry-run`.
|
|
50
|
+
- **No dependencies.** Python standard library plus `git`. DataLad itself is optional — it only adds the `datalad worktree-*` commands.
|
|
51
|
+
|
|
52
|
+
## Installation
|
|
53
|
+
|
|
54
|
+
As a DataLad extension:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
uv tool install datalad \
|
|
58
|
+
--with datalad-worktree@git+https://github.com/just-meng/datalad-worktree.git
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
As a standalone CLI tool:
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
git clone https://github.com/just-meng/datalad-worktree.git
|
|
65
|
+
cd datalad-worktree
|
|
66
|
+
uv sync # --dev for the test suite
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
## Commands
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
worktree add <branch> <worktree-path> # create nested worktrees
|
|
73
|
+
worktree fetch [target] # bring commits and mtimes into the checkout you stand in
|
|
74
|
+
worktree delete <target> # remove worktrees, deepest-first
|
|
75
|
+
worktree list # show worktrees across the hierarchy (also the default)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Installed as a DataLad extension — that is, into the same environment as DataLad, as above — each one is also available as `datalad worktree-add`, `worktree-list`, `worktree-delete`, `worktree-fetch`. Installed standalone, only the `worktree` command exists.
|
|
79
|
+
|
|
80
|
+
Flags and behaviour: [docs/cli.md](docs/cli.md). Why it works the way it does: [docs/design.md](docs/design.md).
|
|
81
|
+
|
|
82
|
+
## Requirements
|
|
83
|
+
|
|
84
|
+
- Python >= 3.11
|
|
85
|
+
- git on PATH (>= 2.38 for `fetch` to merge diverged datasets rather than refuse)
|
|
86
|
+
- [DataLad](https://www.datalad.org/), optional
|
|
87
|
+
|
|
88
|
+
## Contributing
|
|
89
|
+
|
|
90
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
91
|
+
|
|
92
|
+
## License
|
|
93
|
+
|
|
94
|
+
MIT
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
# CLI reference
|
|
2
|
+
|
|
3
|
+
```bash
|
|
4
|
+
worktree add runs /tmp/worktrees/runs # standalone CLI
|
|
5
|
+
datalad worktree-add runs /tmp/worktrees/runs # DataLad extension, same arguments
|
|
6
|
+
python -m datalad_worktree add runs /tmp/worktrees/runs
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Commands run from the superdataset root, or take `-d <path>` to name it.
|
|
10
|
+
|
|
11
|
+
## `worktree add`
|
|
12
|
+
|
|
13
|
+
```
|
|
14
|
+
worktree add <branch> <worktree-path> [options]
|
|
15
|
+
|
|
16
|
+
<branch> branch to check out in every worktree
|
|
17
|
+
<worktree-path> path for the superdataset worktree
|
|
18
|
+
|
|
19
|
+
-n, --dry-run show what would be done, refusals included
|
|
20
|
+
-f, --force replace an existing worktree even if it holds
|
|
21
|
+
commits the main checkout lacks, discarding them
|
|
22
|
+
--follow-parent [<commit>]
|
|
23
|
+
check each subdataset out at the commit its parent
|
|
24
|
+
records; with a commit or tag, the superdataset too
|
|
25
|
+
--no-bindpaths don't configure container bind mounts
|
|
26
|
+
--no-mtimes don't copy mtimes (Snakemake markers included)
|
|
27
|
+
from the main checkout
|
|
28
|
+
-d, --dataset <path> superdataset root (default: current directory)
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Creates a worktree for the superdataset and every installed subdataset, all on `branch`.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
worktree add runs /tmp/worktrees/runs # replaces the worktree if one is already there
|
|
35
|
+
worktree add -f runs /tmp/worktrees/runs # ... even if it holds commits never fetched
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
### Existing branches & worktrees
|
|
39
|
+
|
|
40
|
+
`branch` always starts at each dataset's HEAD. It is created where it is missing, and reset where it exists. A worktree already at `<worktree-path>` is deleted, branch included, and created afresh. Uncommitted changes in it are discarded.
|
|
41
|
+
|
|
42
|
+
#### Pre-flight
|
|
43
|
+
|
|
44
|
+
Before changing anything, `add` checks every dataset. If any check fails, it refuses and touches nothing, not even an existing worktree. It refuses when:
|
|
45
|
+
|
|
46
|
+
- an existing worktree or branch holds commits the main checkout lacks, unless `-f`;
|
|
47
|
+
- a directory at `<worktree-path>` is not one git knows as a worktree;
|
|
48
|
+
- `branch` is checked out in a worktree at another path;
|
|
49
|
+
- with `--follow-parent`, the commit can't be resolved, or a subdataset lacks the commit recorded for it.
|
|
50
|
+
|
|
51
|
+
Only once every check has passed is an existing worktree deleted and the new ones created. If git still fails partway, for a reason no check can foresee (a stale lock, a full disk), `add` stops and deletes the worktrees it created. A replaced worktree stays deleted: it held nothing the main checkout lacks, or `-f` said to discard it.
|
|
52
|
+
|
|
53
|
+
#### Dry run
|
|
54
|
+
|
|
55
|
+
`-n` runs the pre-flight and stops, changing nothing. It reports the refusals, or what the real run would do: the worktree it would replace, and for each dataset where its worktree would go and whether `branch` would be new or reset.
|
|
56
|
+
|
|
57
|
+
### `--follow-parent`
|
|
58
|
+
|
|
59
|
+
By default each dataset checks out `branch` on its own, so a subdataset lands on its own branch tip. That may not be the commit the superdataset records for it. `--follow-parent` checks each subdataset out at the commit its parent records instead. All subdatasets and only those recorded by the target commit are created in the worktrees.
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
worktree add runs /tmp/wt --follow-parent # as the superdataset records it now
|
|
63
|
+
worktree add rerun /tmp/wt --follow-parent v1.0 # as commit or tag v1.0 recorded it
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
### Containers and mtimes
|
|
67
|
+
|
|
68
|
+
The last two steps run over all the worktrees at once:
|
|
69
|
+
|
|
70
|
+
- **Container bind mounts** are configured for containers registered in the superdataset. Containers registered in a subdataset are not configured. `--no-bindpaths` skips this step.
|
|
71
|
+
- **mtimes are copied from the main checkout,** Snakemake's `.snakemake_timestamp` markers included. `--no-mtimes` skips this step.
|
|
72
|
+
|
|
73
|
+
## `worktree fetch`
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
worktree fetch [target] [options]
|
|
77
|
+
|
|
78
|
+
[target] worktree path or branch name to fetch from
|
|
79
|
+
(default: the checkout this worktree came from)
|
|
80
|
+
|
|
81
|
+
-n, --dry-run show what would be merged, refusals included
|
|
82
|
+
--no-mtimes don't copy mtimes (Snakemake markers included)
|
|
83
|
+
from the target
|
|
84
|
+
-d, --dataset <path> checkout to fetch into (default: current directory)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Brings `target`'s commits into the checkout you are standing in, then copies mtimes from `target`.
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
cd /data/my-project
|
|
91
|
+
worktree fetch runs # results home
|
|
92
|
+
|
|
93
|
+
cd /tmp/worktrees/runs
|
|
94
|
+
worktree fetch # new code and inputs in
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
- No `target` from a main checkout is an error.
|
|
98
|
+
- How commits come in:
|
|
99
|
+
- Where your side has no commits of its own, the fetch fast-forwards.
|
|
100
|
+
- Where both sides have commits, it merges (git >= 2.38; older git refuses).
|
|
101
|
+
- Where the merge would conflict, it refuses and names the paths. The usual case is both sides having moved the same subdataset.
|
|
102
|
+
- Uncommitted changes block it only where the fetch would overwrite them, and those paths are named.
|
|
103
|
+
- A dataset the worktree never changed is skipped.
|
|
104
|
+
- A detached HEAD in any of the worktree's datasets refuses the whole fetch.
|
|
105
|
+
|
|
106
|
+
## `worktree delete`
|
|
107
|
+
|
|
108
|
+
```
|
|
109
|
+
worktree delete <target> [options]
|
|
110
|
+
|
|
111
|
+
<target> worktree path or branch name
|
|
112
|
+
|
|
113
|
+
-n, --dry-run show what would be deleted, refusals included
|
|
114
|
+
--keep-branch keep the branch, and with it any commits the
|
|
115
|
+
main checkout lacks
|
|
116
|
+
-f, --force delete despite commits the main checkout
|
|
117
|
+
lacks, discarding them
|
|
118
|
+
-d, --dataset <path> superdataset root (default: current directory)
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
Deletes the worktree in every dataset, deepest first. It does not ask for confirmation, so use `-n` to preview.
|
|
122
|
+
|
|
123
|
+
- The branch is deleted too, unless `--keep-branch`. Uncommitted changes are discarded, as `add` discards them.
|
|
124
|
+
- Before deleting anything, `delete` checks every worktree. Unless `-f`, it refuses and deletes nothing if any worktree has commits the main checkout lacks. With `--keep-branch` these are kept on the branch, so only a worktree with a detached HEAD is refused for them.
|
|
125
|
+
- `-n` runs the same checks and stops, so it never promises a deletion the real run refuses.
|
|
126
|
+
- The main working tree is never deleted.
|
|
127
|
+
|
|
128
|
+
## `worktree list`
|
|
129
|
+
|
|
130
|
+
```
|
|
131
|
+
worktree list [options]
|
|
132
|
+
|
|
133
|
+
-d, --dataset <path> superdataset root (default: current directory)
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
`worktree` with no subcommand runs `list`, in the standalone CLI only. The listing shows datasets that have worktrees beyond the main checkout, grouped by the hierarchy each belongs to, with the main checkout's group first:
|
|
137
|
+
|
|
138
|
+
```
|
|
139
|
+
master
|
|
140
|
+
. /mnt/Data/et_psychedelics/processed/2p
|
|
141
|
+
runs
|
|
142
|
+
. /mnt/Data/worktrees/2p-runs
|
|
143
|
+
code /mnt/Data/worktrees/2p-runs/code (detached)
|
|
144
|
+
inputs/raw /mnt/Data/worktrees/2p-runs/inputs/raw
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
A worktree directory deleted by other means (`rm -rf`) is pruned first, so it doesn't appear.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Design notes
|
|
2
|
+
|
|
3
|
+
## The premise: worktrees are ephemeral
|
|
4
|
+
|
|
5
|
+
A worktree exists for one run: create it, run the pipeline, fetch the results home, throw it away. Every other decision follows from this.
|
|
6
|
+
|
|
7
|
+
- **The main checkout is the only lasting place.** Results come home through `fetch`, and a new worktree starts from the main checkout's state, mtimes included both ways. Otherwise Snakemake reruns, on one side, work the other side already did.
|
|
8
|
+
- **Creating is re-creating.** `add` replaces an existing worktree and starts every branch at HEAD. Going back to a particular state is possible with `--follow-parent <branch>`.
|
|
9
|
+
- **Only unfetched commits are protected.** Uncommitted changes are discarded by `add` and `delete` alike. `-f` means "throw the commits away".
|
|
10
|
+
- **All-or-nothing.** `add` and `delete` check every dataset before touching any, and `-n` runs those same checks and stops.
|
|
11
|
+
|
|
12
|
+
## Containers
|
|
13
|
+
|
|
14
|
+
- **Per-worktree config.** `add` writes the bind-mount option and a `cmdexec` carrying `{{bindpaths}}` with `git config --worktree`, so machine-specific paths stay out of the main checkout and out of history.
|
|
15
|
+
- **One empty `bindpaths` is committed** to `.datalad/config`. `datalad run` records `{{bindpaths}}` unexpanded, so a record made in a worktree only reruns elsewhere if the name resolves there too, to nothing.
|
|
16
|
+
- **Only the superdataset is configured** (#27). A container registered in a subdataset gets no bind paths.
|
|
17
|
+
|
|
18
|
+
## mtimes
|
|
19
|
+
|
|
20
|
+
Snakemake decides what to rerun by mtime, but git carries content, not timestamps:
|
|
21
|
+
|
|
22
|
+
- **On `add`,** the checkout stamps every file "now", superdataset first, so every `code/` file ends up newer than the outputs derived from it.
|
|
23
|
+
- **On `fetch`,** a merge moves only what it rewrites. That misses the grandparent directory a `directory()` output is judged by.
|
|
24
|
+
- **On `datalad run` with identical outputs,** the local mtimes are correct, but since no content is shipped home upon `fetch`, the job is stale in the main checkout.
|
|
25
|
+
|
|
26
|
+
So mtimes are copied from the other working tree: main → worktree on `add`, worktree → main on `fetch`. It covers directories, for `directory()` outputs, and Snakemake's gitignored `.snakemake_timestamp` markers (#39), which Snakemake reads instead of the directory and `snakemake --touch` stamps alone.
|
|
27
|
+
|
|
28
|
+
What keeps it safe:
|
|
29
|
+
|
|
30
|
+
- **Match on blob OID, not path,** so only byte-identical content inherits a timestamp. Skip paths dirty on either side.
|
|
31
|
+
- **Never follow symlinks.** Annexed files point into an object store shared with the main repository.
|
|
32
|
+
- **Skip unlocked annexed files.** Restamping one forces a re-hash through git-annex's clean filter: stamping 10099 symlinks cost the next `git status` 0.11 s, stamping 28 unlocked files (3.71 GB) cost it **152 s**.
|
|
33
|
+
- **No mtimes from commit dates** (the `git-restore-mtime` approach). jj squash and rebase rewrite these repositories routinely, so every file would look new.
|