flexlock 0.8.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
flexlock/git_utils.py ADDED
@@ -0,0 +1,196 @@
1
+ """Source code versioning utilities for FlexLock."""
2
+
3
+ import fnmatch
4
+ import os
5
+ import shutil
6
+ import uuid
7
+ import warnings
8
+ from pathlib import Path
9
+ from contextlib import contextmanager
10
+ from git.repo import Repo as GitRepo
11
+
12
+ from .exceptions import FlexLockSnapshotError
13
+
14
+
15
+ @contextmanager
16
+ def shadow_index(repo: GitRepo):
17
+ """Context manager for Git Plumbing operations without touching user index."""
18
+ git_dir = Path(repo.git_dir)
19
+ temp_index = git_dir / f"index_shadow_{uuid.uuid4().hex}"
20
+
21
+ # Clone current index to temp file for speed
22
+ try:
23
+ if (git_dir / "index").exists():
24
+ shutil.copy2(git_dir / "index", temp_index)
25
+ except Exception:
26
+ pass
27
+
28
+ env = os.environ.copy()
29
+ env["GIT_INDEX_FILE"] = str(temp_index)
30
+
31
+ try:
32
+ yield env
33
+ finally:
34
+ if temp_index.exists():
35
+ temp_index.unlink()
36
+
37
+
38
+ def sanitize_ref_name(name: str) -> str:
39
+ """Sanitize a string to be a valid git ref name."""
40
+ invalid_chars = [" ", "~", "^", ":", "?", "*", "[", "\\", "..", "@{", "//"]
41
+ for char in invalid_chars:
42
+ name = name.replace(char, "_")
43
+ return name
44
+
45
+
46
+ def create_shadow_snapshot(
47
+ repo_path: str = ".",
48
+ ignore_patterns: list | None = None,
49
+ ref_name: str | None = None,
50
+ ) -> dict:
51
+ """
52
+ Creates a Shadow Commit.
53
+ Returns: {commit_hash, tree_hash, is_dirty}
54
+ """
55
+ repo = GitRepo(repo_path, search_parent_directories=True)
56
+ ignore_patterns = ignore_patterns or []
57
+
58
+ with shadow_index(repo) as shadow_env:
59
+ git = repo.git
60
+
61
+ # 1. Stage everything (Modified + Untracked) into Shadow Index
62
+ git.add("--all", env=shadow_env)
63
+
64
+ # 2. Remove ignored patterns from Shadow Index
65
+ if ignore_patterns:
66
+ try:
67
+ git.rm(
68
+ "--cached",
69
+ "-r",
70
+ "--ignore-unmatch",
71
+ *ignore_patterns,
72
+ env=shadow_env,
73
+ )
74
+ except Exception:
75
+ pass
76
+
77
+ # 3. Write Tree (This is the content fingerprint)
78
+ tree_hash = git.write_tree(env=shadow_env)
79
+
80
+ # 4. Create Shadow Commit (Lineage)
81
+ parent = repo.head.commit.hexsha
82
+ msg = f"FlexLock Shadow: {parent[:7]} + Changes"
83
+ shadow_commit = git.commit_tree(
84
+ tree_hash, "-p", parent, "-m", msg, env=shadow_env
85
+ )
86
+
87
+ # 5. Save Ref (Prevent Garbage Collection)
88
+ ref_name = f"refs/flexlock/runs/{ref_name or shadow_commit}"
89
+ git.update_ref(sanitize_ref_name(ref_name), shadow_commit)
90
+
91
+ return {
92
+ "commit": shadow_commit,
93
+ "tree": tree_hash, # <--- The key for Equality Checks
94
+ "is_dirty": repo.is_dirty(untracked_files=True),
95
+ }
96
+
97
+
98
+ def create_shadow_tree(
99
+ repo_path: str = ".",
100
+ include: list | None = None,
101
+ exclude: list | None = None,
102
+ ) -> dict:
103
+ """Compute a content tree hash for a working tree without side effects.
104
+
105
+ Unlike :func:`create_shadow_snapshot`, this stages into a throwaway shadow
106
+ index and runs **``write-tree`` only** — it creates no commit object and no
107
+ ``refs/flexlock/runs/*`` ref. It is therefore safe to call on every
108
+ fingerprint check (smart-run) without accumulating objects/refs in ``.git``.
109
+
110
+ Args:
111
+ repo_path: Path inside the repository.
112
+ include: Optional pathspec(s); when given, only these paths are staged
113
+ (the fingerprint is restricted to the "relevant" subtree, so an
114
+ include-match becomes plain tree-hash equality). Defaults to all
115
+ tracked + untracked files.
116
+ exclude: Optional pathspec(s) removed from the staged index.
117
+
118
+ Returns:
119
+ dict: ``{"tree": <hash>, "is_dirty": <bool>}``.
120
+ """
121
+ repo = GitRepo(repo_path, search_parent_directories=True)
122
+ include = include or None
123
+ exclude = exclude or []
124
+
125
+ with shadow_index(repo) as shadow_env:
126
+ git = repo.git
127
+
128
+ # 1. Stage into the shadow index. Restrict to `include` when provided so
129
+ # the tree hash only reflects the relevant subtree.
130
+ if include:
131
+ git.add("--", *include, env=shadow_env)
132
+ else:
133
+ git.add("--all", env=shadow_env)
134
+
135
+ # 2. Drop excluded patterns from the shadow index.
136
+ if exclude:
137
+ try:
138
+ git.rm(
139
+ "--cached", "-r", "--ignore-unmatch", *exclude, env=shadow_env
140
+ )
141
+ except Exception:
142
+ pass
143
+
144
+ # 3. Write the tree — the content fingerprint. No commit, no ref.
145
+ tree_hash = git.write_tree(env=shadow_env)
146
+
147
+ return {
148
+ "tree": tree_hash,
149
+ "is_dirty": repo.is_dirty(untracked_files=True),
150
+ }
151
+
152
+
153
+ def get_git_tree_hash(path: str = ".") -> str:
154
+ """
155
+ Gets the current git tree hash for a repository without creating a new commit.
156
+ This represents the content fingerprint of the repository.
157
+
158
+ Args:
159
+ path (str): The path to the git repository.
160
+
161
+ Returns:
162
+ str: The tree hash.
163
+
164
+ Raises:
165
+ FlexLockSnapshotError: if ``path`` is not a usable git repository.
166
+ """
167
+ try:
168
+ repo = GitRepo(path, search_parent_directories=True)
169
+ # Get the tree hash of the current commit
170
+ return repo.head.commit.tree.hexsha
171
+ except Exception as e:
172
+ raise FlexLockSnapshotError(
173
+ f"Could not get git tree hash for {path!r}: {e}"
174
+ ) from e
175
+
176
+
177
+ def get_git_commit(path: str = ".") -> str:
178
+ """
179
+ Gets the current commit hash for a git repository without creating a new commit.
180
+
181
+ Args:
182
+ path (str): The path to the git repository.
183
+
184
+ Returns:
185
+ str: The commit hash.
186
+
187
+ Raises:
188
+ FlexLockSnapshotError: if ``path`` is not a usable git repository.
189
+ """
190
+ try:
191
+ repo = GitRepo(path, search_parent_directories=True)
192
+ return repo.head.commit.hexsha
193
+ except Exception as e:
194
+ raise FlexLockSnapshotError(
195
+ f"Could not get git commit for {path!r}: {e}"
196
+ ) from e
flexlock/index.py ADDED
@@ -0,0 +1,300 @@
1
+ """Project-wide fingerprint index — makes cache lookups O(1) and sweep items
2
+ first-class.
3
+
4
+ The index is a **derived cache**: ``run.lock`` (and the task DB) stay
5
+ authoritative and the index can always be deleted and rebuilt with
6
+ ``flexlock reindex``. It maps a run :mod:`fingerprint` to *where* a completed
7
+ run lives — either a ``run.lock`` directory (serial/single runs) or a
8
+ ``(task_db, task_id)`` pair (sweep tasks) — so a config first run as a sweep
9
+ task cache-hits when re-run serially, and vice-versa.
10
+
11
+ Only ``status='done'`` rows are ever served, so failed or interrupted runs are
12
+ never returned as cache hits.
13
+
14
+ Location resolution (see :func:`resolve_index_path`):
15
+ 1. ``$FLEXLOCK_INDEX`` if set;
16
+ 2. the nearest existing ``.flexlock/index.db`` walking up from ``base``;
17
+ 3. otherwise ``<base>/.flexlock/index.db``.
18
+
19
+ Writers on the read and write paths must resolve to the *same* file for a hit
20
+ to land — a project wrapper should set ``search_dirs`` and the index root to
21
+ agree (or set ``$FLEXLOCK_INDEX``).
22
+ """
23
+
24
+ import os
25
+ import sqlite3
26
+ import tempfile
27
+ import time
28
+ from dataclasses import dataclass
29
+ from pathlib import Path
30
+ from typing import Optional
31
+
32
+ from loguru import logger
33
+
34
+ INDEX_DIRNAME = ".flexlock"
35
+ INDEX_FILENAME = "index.db"
36
+
37
+ LOCATION_RUN_LOCK = "run_lock"
38
+ LOCATION_TASK = "task"
39
+
40
+ STATUS_DONE = "done"
41
+
42
+ _SCHEMA = """
43
+ CREATE TABLE IF NOT EXISTS runs (
44
+ fingerprint TEXT PRIMARY KEY,
45
+ status TEXT NOT NULL,
46
+ location_kind TEXT NOT NULL,
47
+ run_lock_path TEXT,
48
+ task_db_path TEXT,
49
+ task_id TEXT,
50
+ save_dir TEXT,
51
+ ts REAL
52
+ )
53
+ """
54
+
55
+
56
+ @dataclass
57
+ class IndexRow:
58
+ fingerprint: str
59
+ status: str
60
+ location_kind: str
61
+ run_lock_path: Optional[str]
62
+ task_db_path: Optional[str]
63
+ task_id: Optional[str]
64
+ save_dir: Optional[str]
65
+ ts: Optional[float]
66
+
67
+
68
+ # ── location resolution ──
69
+
70
+
71
+ def _shared_roots() -> set:
72
+ """Directories that must never host an auto-discovered project index.
73
+
74
+ Walking up to a stray ``.flexlock`` in ``$HOME``, the system temp dir, or
75
+ the filesystem root would silently capture unrelated runs across projects,
76
+ so these are skipped as index homes (an explicit ``$FLEXLOCK_INDEX`` still
77
+ wins if the user really wants a global index).
78
+ """
79
+ roots = {Path(tempfile.gettempdir()).resolve()}
80
+ try:
81
+ roots.add(Path.home().resolve())
82
+ except Exception:
83
+ pass
84
+ cur = Path(".").resolve()
85
+ roots.add(Path(cur.anchor)) # filesystem root
86
+ return roots
87
+
88
+
89
+ def _index_disabled_for(index_path: Path) -> bool:
90
+ """True when the index would sit directly in a shared root with no explicit
91
+ ``$FLEXLOCK_INDEX`` — i.e. the run isn't in a project tree, so auto-indexing
92
+ is skipped to avoid a global index capturing unrelated runs (the glob
93
+ fallback still applies)."""
94
+ if os.environ.get("FLEXLOCK_INDEX"):
95
+ return False
96
+ host = index_path.parent.parent # <host>/.flexlock/index.db -> <host>
97
+ try:
98
+ return host.resolve() in _shared_roots()
99
+ except Exception:
100
+ return False
101
+
102
+
103
+ def resolve_index_path(base, create_parent: bool = False) -> Path:
104
+ """Resolve the index db path for a run/search rooted at ``base``."""
105
+ env = os.environ.get("FLEXLOCK_INDEX")
106
+ if env:
107
+ p = Path(env)
108
+ else:
109
+ p = None
110
+ cur = Path(base).resolve()
111
+ skip = _shared_roots()
112
+ # Walk up looking for an existing index to share, but never latch onto
113
+ # a stray index in a shared root (home/tmp/fs-root).
114
+ for candidate_dir in [cur, *cur.parents]:
115
+ if candidate_dir in skip:
116
+ continue
117
+ candidate = candidate_dir / INDEX_DIRNAME / INDEX_FILENAME
118
+ if candidate.exists():
119
+ p = candidate
120
+ break
121
+ if p is None:
122
+ p = cur / INDEX_DIRNAME / INDEX_FILENAME
123
+ if create_parent:
124
+ p.parent.mkdir(parents=True, exist_ok=True)
125
+ return p
126
+
127
+
128
+ # ── connection ──
129
+
130
+
131
+ def _connect(index_path: Path) -> sqlite3.Connection:
132
+ index_path.parent.mkdir(parents=True, exist_ok=True)
133
+ conn = sqlite3.connect(str(index_path), timeout=15.0)
134
+ conn.execute("PRAGMA journal_mode=WAL")
135
+ conn.execute("PRAGMA busy_timeout=15000")
136
+ conn.execute(_SCHEMA)
137
+ return conn
138
+
139
+
140
+ # ── writes ──
141
+
142
+
143
+ def upsert(
144
+ index_path: Path,
145
+ fingerprint: str,
146
+ location_kind: str,
147
+ save_dir: str,
148
+ run_lock_path: Optional[str] = None,
149
+ task_db_path: Optional[str] = None,
150
+ task_id: Optional[str] = None,
151
+ status: str = STATUS_DONE,
152
+ ) -> None:
153
+ """Insert-or-replace a row keyed by fingerprint. Idempotent."""
154
+ try:
155
+ conn = _connect(Path(index_path))
156
+ with conn:
157
+ conn.execute(
158
+ "INSERT OR REPLACE INTO runs "
159
+ "(fingerprint, status, location_kind, run_lock_path, task_db_path, "
160
+ " task_id, save_dir, ts) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
161
+ (
162
+ fingerprint,
163
+ status,
164
+ location_kind,
165
+ run_lock_path,
166
+ task_db_path,
167
+ task_id,
168
+ save_dir,
169
+ time.time(),
170
+ ),
171
+ )
172
+ conn.close()
173
+ except sqlite3.Error as e:
174
+ # The index is a derived cache; a write failure must never break a run.
175
+ logger.warning(f"Fingerprint index upsert failed ({index_path}): {e}")
176
+
177
+
178
+ def record_run_lock(save_dir, fingerprint: str) -> None:
179
+ """Record a completed serial/single run (``run.lock`` location)."""
180
+ save_dir = Path(save_dir)
181
+ index_path = resolve_index_path(save_dir.parent)
182
+ if _index_disabled_for(index_path):
183
+ return
184
+ index_path.parent.mkdir(parents=True, exist_ok=True)
185
+ upsert(
186
+ index_path,
187
+ fingerprint=fingerprint,
188
+ location_kind=LOCATION_RUN_LOCK,
189
+ save_dir=str(save_dir),
190
+ run_lock_path=str(save_dir / "run.lock"),
191
+ )
192
+
193
+
194
+ def record_task(save_dir, task_db_path, task_id: str, fingerprint: str) -> None:
195
+ """Record a completed sweep task (``(task_db, task_id)`` location)."""
196
+ save_dir = Path(save_dir)
197
+ index_path = resolve_index_path(save_dir.parent)
198
+ if _index_disabled_for(index_path):
199
+ return
200
+ index_path.parent.mkdir(parents=True, exist_ok=True)
201
+ upsert(
202
+ index_path,
203
+ fingerprint=fingerprint,
204
+ location_kind=LOCATION_TASK,
205
+ save_dir=str(save_dir),
206
+ task_db_path=str(task_db_path),
207
+ task_id=task_id,
208
+ )
209
+
210
+
211
+ def prune(index_path: Path, fingerprint: str) -> None:
212
+ try:
213
+ conn = _connect(Path(index_path))
214
+ with conn:
215
+ conn.execute("DELETE FROM runs WHERE fingerprint=?", (fingerprint,))
216
+ conn.close()
217
+ except sqlite3.Error as e:
218
+ logger.warning(f"Fingerprint index prune failed ({index_path}): {e}")
219
+
220
+
221
+ # ── reads ──
222
+
223
+
224
+ def lookup(base, fingerprint: str) -> Optional[IndexRow]:
225
+ """Look up a done row by fingerprint, resolving the index from ``base``."""
226
+ index_path = resolve_index_path(base)
227
+ if _index_disabled_for(index_path) or not index_path.exists():
228
+ return None
229
+ try:
230
+ conn = _connect(index_path)
231
+ cur = conn.execute(
232
+ "SELECT fingerprint, status, location_kind, run_lock_path, "
233
+ "task_db_path, task_id, save_dir, ts FROM runs "
234
+ "WHERE fingerprint=? AND status=?",
235
+ (fingerprint, STATUS_DONE),
236
+ )
237
+ row = cur.fetchone()
238
+ conn.close()
239
+ except sqlite3.Error as e:
240
+ logger.warning(f"Fingerprint index lookup failed ({index_path}): {e}")
241
+ return None
242
+ if row is None:
243
+ return None
244
+ return IndexRow(*row)
245
+
246
+
247
+ def verify_and_resolve(base, row: IndexRow) -> Optional[Path]:
248
+ """Confirm the row's pointed-to run still exists and is complete.
249
+
250
+ Returns the run's ``save_dir`` on success. On a stale pointer (deleted or
251
+ incomplete) the row is pruned and ``None`` is returned, so the index
252
+ self-heals to a miss.
253
+ """
254
+ from .run_record import RunRecord
255
+
256
+ save_dir = Path(row.save_dir) if row.save_dir else None
257
+
258
+ if row.location_kind == LOCATION_RUN_LOCK:
259
+ if save_dir and RunRecord(save_dir).is_complete:
260
+ return save_dir
261
+ elif row.location_kind == LOCATION_TASK:
262
+ # Task is complete iff its results.json is present (the worker writes it
263
+ # via RunRecord.mark_complete just before recording the index row).
264
+ if save_dir and RunRecord(save_dir).is_complete_task():
265
+ return save_dir
266
+
267
+ # Stale pointer — prune and report a miss.
268
+ prune(resolve_index_path(base), row.fingerprint)
269
+ return None
270
+
271
+
272
+ def reindex(root) -> int:
273
+ """Backfill the index by walking ``**/run.lock`` under ``root``.
274
+
275
+ Uses the ``fingerprint`` stored in each ``run.lock`` (written at snapshot
276
+ time). Runs from before fingerprints were stored are skipped. Returns the
277
+ number of rows written.
278
+ """
279
+ import yaml
280
+
281
+ root = Path(root)
282
+ count = 0
283
+ for lock_file in root.glob("**/run.lock"):
284
+ run_dir = lock_file.parent
285
+ try:
286
+ data = yaml.safe_load(lock_file.read_text())
287
+ except Exception as e:
288
+ logger.debug(f"reindex: skipping {lock_file}: {e}")
289
+ continue
290
+ if not isinstance(data, dict):
291
+ continue
292
+ fp = data.get("fingerprint")
293
+ if not fp:
294
+ continue
295
+ if not (run_dir / "run.complete").exists():
296
+ continue
297
+ record_run_lock(run_dir, fp)
298
+ count += 1
299
+ logger.info(f"reindex: recorded {count} run(s) under {root}")
300
+ return count
flexlock/load_stage.py ADDED
@@ -0,0 +1,60 @@
1
+ """Utility for loading data from a previous FlexLock stage."""
2
+
3
+ from pathlib import Path
4
+ import yaml
5
+
6
+
7
+ def load_stage_from_path(path: str) -> dict:
8
+ """
9
+ Loads a stage from a given path and returns its flattened data, including
10
+ all its ancestors.
11
+
12
+ Args:
13
+ path (str): The path to the directory of the previous stage.
14
+
15
+ Returns:
16
+ dict: A flattened, deduplicated dictionary of all ancestor runs.
17
+ """
18
+ all_stages = {}
19
+ _load_and_flatten_recursively(Path(path).as_posix(), path, all_stages)
20
+ return all_stages
21
+
22
+
23
+ def _load_and_flatten_recursively(
24
+ stage_key: str, stage_path_str: str, all_stages: dict
25
+ ):
26
+ """
27
+ Recursively loads a stage and its ancestors, adding them to the all_stages dict.
28
+ """
29
+ # Use the canonical path as the key to prevent duplicates
30
+ canonical_key = Path(stage_path_str).resolve().name
31
+ if canonical_key in all_stages:
32
+ return
33
+
34
+ stage_path = Path(stage_path_str)
35
+ lock_file = stage_path / "run.lock"
36
+
37
+ if not lock_file.exists():
38
+ raise FileNotFoundError(
39
+ f"run.lock not found in previous stage '{stage_key}': {lock_file}"
40
+ )
41
+
42
+ with open(lock_file, "r") as f:
43
+ stage_data = yaml.safe_load(f)
44
+
45
+ # Recurse into nested stages first (depth-first)
46
+ # Support both "lineage" (new) and "prevs" (legacy)
47
+ lineage = stage_data.get("lineage") or stage_data.get("prevs")
48
+ if lineage:
49
+ for nested_key, nested_data in lineage.items():
50
+ nested_path = nested_data.get("path") or nested_data.get("config", {}).get("save_dir")
51
+ if not nested_path:
52
+ raise ValueError(
53
+ f"Could not find path in nested stage '{nested_key}' from '{stage_key}'"
54
+ )
55
+ _load_and_flatten_recursively(nested_key, nested_path, all_stages)
56
+
57
+ # Add the current stage's data (without its own lineage) to the dict
58
+ stage_data.pop("lineage", None)
59
+ stage_data.pop("prevs", None)
60
+ all_stages[canonical_key] = stage_data