tdd-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tdd_cli-0.1.0.dist-info/METADATA +391 -0
- tdd_cli-0.1.0.dist-info/RECORD +24 -0
- tdd_cli-0.1.0.dist-info/WHEEL +4 -0
- tdd_cli-0.1.0.dist-info/entry_points.txt +2 -0
- tdd_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
- tddcli/__init__.py +6 -0
- tddcli/adapters/__init__.py +48 -0
- tddcli/adapters/base.py +119 -0
- tddcli/adapters/pytest_adapter.py +179 -0
- tddcli/adapters/vitest_adapter.py +170 -0
- tddcli/advance.py +423 -0
- tddcli/cli.py +1043 -0
- tddcli/config.py +255 -0
- tddcli/contract.py +237 -0
- tddcli/envelope.py +96 -0
- tddcli/fleet.py +128 -0
- tddcli/gitutil.py +138 -0
- tddcli/identity.py +82 -0
- tddcli/leases.py +118 -0
- tddcli/ledger.py +433 -0
- tddcli/machine.py +390 -0
- tddcli/render.py +275 -0
- tddcli/snapshot.py +90 -0
- tddcli/staging.py +130 -0
tddcli/fleet.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Fleet view — every agent's progress on this repository, in one summary.
|
|
2
|
+
|
|
3
|
+
The ledger is one SQLite database per repository, shared by all worktrees, so the
|
|
4
|
+
data already exists in one place; this module only reads it. Read-only is
|
|
5
|
+
structural, not conventional: the database is opened with SQLite's `mode=ro` URI,
|
|
6
|
+
so the command cannot create, migrate, or mutate the ledger that live agents are
|
|
7
|
+
writing mid-run. That is what makes it safe to run — from any worktree, on any
|
|
8
|
+
branch — while runs are in flight, even if this code's schema constant were ever
|
|
9
|
+
to drift from the one on disk.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import sqlite3
|
|
16
|
+
from datetime import datetime, timezone
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
from . import leases
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def open_readonly(path: Path) -> sqlite3.Connection | None:
|
|
23
|
+
"""None when no ledger exists yet — `mode=ro` also refuses to create one."""
|
|
24
|
+
if not path.is_file():
|
|
25
|
+
return None
|
|
26
|
+
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
27
|
+
conn.row_factory = sqlite3.Row
|
|
28
|
+
return conn
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _age_s(iso: str | None) -> float | None:
|
|
32
|
+
if not iso:
|
|
33
|
+
return None
|
|
34
|
+
stamp = datetime.fromisoformat(iso)
|
|
35
|
+
if stamp.tzinfo is None:
|
|
36
|
+
stamp = stamp.replace(tzinfo=timezone.utc)
|
|
37
|
+
return round((datetime.now(timezone.utc) - stamp).total_seconds(), 1)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _runs(conn: sqlite3.Connection) -> list[dict]:
|
|
41
|
+
rows = conn.execute(
|
|
42
|
+
"SELECT r.id, r.worktree_path, r.executor_model, r.started_at,"
|
|
43
|
+
" p.plan_path, p.declared_cycles"
|
|
44
|
+
" FROM run r JOIN plan_contract p ON p.id = r.plan_contract_id"
|
|
45
|
+
" WHERE r.ended_at IS NULL ORDER BY r.id"
|
|
46
|
+
).fetchall()
|
|
47
|
+
out = []
|
|
48
|
+
for row in rows:
|
|
49
|
+
cycle = conn.execute(
|
|
50
|
+
"SELECT ordinal, phase, title FROM cycle"
|
|
51
|
+
" WHERE run_id = ? AND closed_at IS NULL ORDER BY ordinal LIMIT 1",
|
|
52
|
+
(row["id"],),
|
|
53
|
+
).fetchone()
|
|
54
|
+
last = conn.execute(
|
|
55
|
+
"SELECT MAX(started_at) AS at FROM invocation WHERE run_id = ?",
|
|
56
|
+
(row["id"],),
|
|
57
|
+
).fetchone()
|
|
58
|
+
out.append(
|
|
59
|
+
{
|
|
60
|
+
"run_id": row["id"],
|
|
61
|
+
"worktree": row["worktree_path"],
|
|
62
|
+
"plan": row["plan_path"],
|
|
63
|
+
"executor": row["executor_model"],
|
|
64
|
+
"started_at": row["started_at"],
|
|
65
|
+
"cycle": cycle["ordinal"] if cycle else None,
|
|
66
|
+
"of": len(json.loads(row["declared_cycles"])) or None,
|
|
67
|
+
"phase": cycle["phase"] if cycle else None,
|
|
68
|
+
"title": cycle["title"] if cycle else None,
|
|
69
|
+
# Staleness signal for a wedged agent: age of the newest suite
|
|
70
|
+
# invocation, falling back to run start when none has landed yet.
|
|
71
|
+
"last_activity_age_s": _age_s(last["at"] or row["started_at"]),
|
|
72
|
+
}
|
|
73
|
+
)
|
|
74
|
+
return out
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _claims(conn: sqlite3.Connection) -> list[dict]:
|
|
78
|
+
rows = conn.execute("SELECT * FROM baseline_claim ORDER BY id").fetchall()
|
|
79
|
+
return [
|
|
80
|
+
{
|
|
81
|
+
"worktree": r["worktree_path"],
|
|
82
|
+
"hostname": r["hostname"],
|
|
83
|
+
"projects_done": r["projects_done"],
|
|
84
|
+
"projects_total": r["projects_total"],
|
|
85
|
+
"current_project": r["current_project"],
|
|
86
|
+
"elapsed_s": _age_s(r["started_at"]),
|
|
87
|
+
}
|
|
88
|
+
for r in rows
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def summarise(ledger_db: Path) -> dict:
|
|
93
|
+
conn = open_readonly(ledger_db)
|
|
94
|
+
if conn is None:
|
|
95
|
+
return {"runs": [], "collecting": [], "suites": leases.snapshot()}
|
|
96
|
+
try:
|
|
97
|
+
return {
|
|
98
|
+
"runs": _runs(conn),
|
|
99
|
+
"collecting": _claims(conn),
|
|
100
|
+
"suites": leases.snapshot(),
|
|
101
|
+
}
|
|
102
|
+
finally:
|
|
103
|
+
conn.close()
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def render(summary: dict) -> str:
|
|
107
|
+
lines = []
|
|
108
|
+
for r in summary["runs"]:
|
|
109
|
+
cycle = f"cycle {r['cycle']}/{r['of']}" if r["cycle"] else "between cycles"
|
|
110
|
+
title = f" ({r['title']})" if r.get("title") else ""
|
|
111
|
+
lines.append(
|
|
112
|
+
f"{r['worktree']} {r['plan']} {cycle}{title} {r['phase'] or '-'}"
|
|
113
|
+
f" last activity {r['last_activity_age_s']}s ago"
|
|
114
|
+
)
|
|
115
|
+
for c in summary["collecting"]:
|
|
116
|
+
lines.append(
|
|
117
|
+
f"{c['worktree']} collecting baseline"
|
|
118
|
+
f" {c['projects_done']}/{c['projects_total']}"
|
|
119
|
+
f" (current: {c['current_project'] or '-'}) — {c['elapsed_s']}s elapsed"
|
|
120
|
+
)
|
|
121
|
+
s = summary["suites"]
|
|
122
|
+
lines.append(
|
|
123
|
+
f"suites executing now: {s['active']}"
|
|
124
|
+
f" — {s['workers_each']} worker(s) each of {s['total_cores']} cores"
|
|
125
|
+
)
|
|
126
|
+
if not summary["runs"] and not summary["collecting"]:
|
|
127
|
+
lines.insert(0, "no active runs")
|
|
128
|
+
return "\n".join(lines) + "\n"
|
tddcli/gitutil.py
ADDED
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""Git access. Every call is explicit about its worktree; nothing is resolved from cwd."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import subprocess
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class GitError(RuntimeError):
|
|
11
|
+
pass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def git(worktree: Path, *args: str, check: bool = True) -> str:
|
|
15
|
+
proc = subprocess.run(
|
|
16
|
+
["git", "-C", str(worktree), *args],
|
|
17
|
+
capture_output=True,
|
|
18
|
+
text=True,
|
|
19
|
+
)
|
|
20
|
+
if check and proc.returncode != 0:
|
|
21
|
+
raise GitError(f"git {' '.join(args)} failed: {proc.stderr.strip()}")
|
|
22
|
+
return proc.stdout
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def worktree_root(start: Path) -> Path:
|
|
26
|
+
out = git(start, "rev-parse", "--show-toplevel").strip()
|
|
27
|
+
if not out:
|
|
28
|
+
raise GitError(f"not a git worktree: {start}")
|
|
29
|
+
return Path(out).resolve()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def repo_identity(worktree: Path) -> Path:
|
|
33
|
+
"""The canonical repository path — the *common* git dir, shared by all worktrees.
|
|
34
|
+
|
|
35
|
+
R13.3: the ledger is keyed by this, not by the worktree, so pruning a worktree
|
|
36
|
+
never orphans its runs.
|
|
37
|
+
"""
|
|
38
|
+
common = git(worktree, "rev-parse", "--path-format=absolute", "--git-common-dir").strip()
|
|
39
|
+
return Path(common).resolve().parent
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def head(worktree: Path) -> str:
|
|
43
|
+
return git(worktree, "rev-parse", "HEAD").strip()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def blob_sha_at_head(worktree: Path, rel_path: str) -> tuple[str, str]:
|
|
47
|
+
"""(blob_sha, commit_sha) for a path as committed — never the working-tree copy."""
|
|
48
|
+
commit = head(worktree)
|
|
49
|
+
out = git(worktree, "rev-parse", f"HEAD:{rel_path}", check=False).strip()
|
|
50
|
+
if not out or " " in out:
|
|
51
|
+
raise GitError(f"{rel_path} is not committed at HEAD")
|
|
52
|
+
return out, commit
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def show_at_head(worktree: Path, rel_path: str) -> str:
|
|
56
|
+
return git(worktree, "show", f"HEAD:{rel_path}")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def tracked_at_head(worktree: Path, paths: list[str]) -> set[str]:
|
|
60
|
+
"""Which of `paths` exist in the HEAD commit — so a path absent here is a new file."""
|
|
61
|
+
if not paths:
|
|
62
|
+
return set()
|
|
63
|
+
out = git(worktree, "ls-tree", "-r", "--name-only", "-z", "HEAD", "--", *paths)
|
|
64
|
+
return {p for p in out.split("\0") if p}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def status_porcelain(worktree: Path) -> list[tuple[str, str]]:
|
|
68
|
+
out = git(worktree, "status", "--porcelain=v1", "-uall")
|
|
69
|
+
entries = []
|
|
70
|
+
for line in out.splitlines():
|
|
71
|
+
if not line.strip():
|
|
72
|
+
continue
|
|
73
|
+
entries.append((line[:2], line[3:].strip()))
|
|
74
|
+
return entries
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def is_dirty(worktree: Path) -> bool:
|
|
78
|
+
return bool(status_porcelain(worktree))
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def dirty_paths(worktree: Path) -> set[str]:
|
|
82
|
+
return {path for _, path in status_porcelain(worktree)}
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def changed_paths(worktree: Path) -> set[str]:
|
|
86
|
+
"""Tracked modifications plus untracked files, relative to the worktree root."""
|
|
87
|
+
return dirty_paths(worktree)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def diff_text(worktree: Path) -> str:
|
|
91
|
+
return git(worktree, "diff")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def tree_hash(worktree: Path, roots: list[str]) -> str:
|
|
95
|
+
"""Hash of tracked content under the given roots, plus untracked file contents.
|
|
96
|
+
|
|
97
|
+
Backs `no_change_since_last_run` (§6) and the refactor-phase skip (§6.1).
|
|
98
|
+
"""
|
|
99
|
+
h = hashlib.sha256()
|
|
100
|
+
for root in sorted(roots):
|
|
101
|
+
h.update(root.encode())
|
|
102
|
+
h.update(git(worktree, "ls-files", "-s", "--", root).encode())
|
|
103
|
+
diff = git(worktree, "diff", "--", root)
|
|
104
|
+
h.update(diff.encode())
|
|
105
|
+
untracked = git(worktree, "ls-files", "-o", "--exclude-standard", "--", root)
|
|
106
|
+
for rel in sorted(untracked.split()):
|
|
107
|
+
h.update(rel.encode())
|
|
108
|
+
p = worktree / rel
|
|
109
|
+
if p.is_file():
|
|
110
|
+
h.update(p.read_bytes())
|
|
111
|
+
return h.hexdigest()
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def add(worktree: Path, paths: list[str]) -> None:
|
|
115
|
+
if paths:
|
|
116
|
+
git(worktree, "add", "--", *paths)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def reset_index(worktree: Path) -> None:
|
|
120
|
+
git(worktree, "reset", "-q")
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def commit(worktree: Path, message: str, trailers: dict[str, str]) -> str:
|
|
124
|
+
body = message
|
|
125
|
+
if trailers:
|
|
126
|
+
body += "\n\n" + "\n".join(f"{k}: {v}" for k, v in trailers.items())
|
|
127
|
+
git(worktree, "commit", "-q", "-m", body)
|
|
128
|
+
return head(worktree)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def checkout_paths(worktree: Path, paths: list[str]) -> None:
|
|
132
|
+
if paths:
|
|
133
|
+
git(worktree, "checkout", "--", *paths)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def staged_paths(worktree: Path) -> list[str]:
|
|
137
|
+
out = git(worktree, "diff", "--cached", "--name-only")
|
|
138
|
+
return [p for p in out.splitlines() if p.strip()]
|
tddcli/identity.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Executor identity resolution (§5.1).
|
|
2
|
+
|
|
3
|
+
The harness exposes a session id but not the model, so the model is read from the
|
|
4
|
+
session transcript. Agents never supply identity by any path (R5.2) — the human
|
|
5
|
+
fallback exists for hosts where the transcript is unavailable.
|
|
6
|
+
|
|
7
|
+
Known limit: an in-process subagent inherits its parent's CLAUDE_CODE_SESSION_ID and
|
|
8
|
+
writes no transcript under TRANSCRIPT_ROOT, so a run it starts resolves to the
|
|
9
|
+
parent's model. Model comparisons require executors in top-level sessions.
|
|
10
|
+
|
|
11
|
+
All harness coupling lives in this one module (R5.1): a transcript format change
|
|
12
|
+
breaks here and nowhere else.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
TRANSCRIPT_ROOT = Path.home() / ".claude" / "projects"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class Executor:
|
|
27
|
+
model: str
|
|
28
|
+
session: str | None
|
|
29
|
+
source: str # transcript | human | unknown
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _slug(path: Path) -> str:
|
|
33
|
+
return str(path).replace(os.sep, "-")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _find_transcript(session_id: str, project_path: Path | None) -> Path | None:
|
|
37
|
+
candidates: list[Path] = []
|
|
38
|
+
if project_path is not None:
|
|
39
|
+
scoped = TRANSCRIPT_ROOT / _slug(project_path) / f"{session_id}.jsonl"
|
|
40
|
+
candidates.append(scoped)
|
|
41
|
+
if TRANSCRIPT_ROOT.is_dir():
|
|
42
|
+
candidates.extend(TRANSCRIPT_ROOT.glob(f"*/{session_id}.jsonl"))
|
|
43
|
+
for c in candidates:
|
|
44
|
+
if c.is_file():
|
|
45
|
+
return c
|
|
46
|
+
return None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _model_from_transcript(path: Path) -> str | None:
|
|
50
|
+
"""Last model wins — a session may switch models mid-run."""
|
|
51
|
+
found = None
|
|
52
|
+
try:
|
|
53
|
+
with path.open() as fh:
|
|
54
|
+
for line in fh:
|
|
55
|
+
line = line.strip()
|
|
56
|
+
if not line or '"model"' not in line:
|
|
57
|
+
continue
|
|
58
|
+
try:
|
|
59
|
+
rec = json.loads(line)
|
|
60
|
+
except json.JSONDecodeError:
|
|
61
|
+
continue
|
|
62
|
+
model = rec.get("model") or (rec.get("message") or {}).get("model")
|
|
63
|
+
if isinstance(model, str) and model:
|
|
64
|
+
found = model
|
|
65
|
+
except OSError:
|
|
66
|
+
return None
|
|
67
|
+
return found
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def resolve(project_path: Path | None = None, human_label: str | None = None) -> Executor:
|
|
71
|
+
session = os.environ.get("CLAUDE_CODE_SESSION_ID")
|
|
72
|
+
if session:
|
|
73
|
+
transcript = _find_transcript(session, project_path)
|
|
74
|
+
if transcript is not None:
|
|
75
|
+
model = _model_from_transcript(transcript)
|
|
76
|
+
if model:
|
|
77
|
+
return Executor(model=model, session=session, source="transcript")
|
|
78
|
+
|
|
79
|
+
if human_label:
|
|
80
|
+
return Executor(model=human_label, session=session, source="human")
|
|
81
|
+
|
|
82
|
+
return Executor(model="unknown", session=session, source="unknown")
|
tddcli/leases.py
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Machine-wide test-worker budget.
|
|
2
|
+
|
|
3
|
+
Several agents run tdd-cli concurrently on one machine, each in its own worktree.
|
|
4
|
+
With no coordination each had to pin its suite to `-n 1` — the only setting that
|
|
5
|
+
never oversubscribes the box — which serialises every suite even when the agent is
|
|
6
|
+
alone. This module gives each in-flight suite invocation an even share of the
|
|
7
|
+
machine's cores instead: a lease file per invocation in a directory shared across
|
|
8
|
+
worktrees, `workers = max(1, cores // live_leases)`.
|
|
9
|
+
|
|
10
|
+
Deliberate properties:
|
|
11
|
+
|
|
12
|
+
* The split is computed once, at lease acquisition. An agent that arrives mid-run
|
|
13
|
+
gets the smaller share immediately; the earlier agent's share corrects on its
|
|
14
|
+
next invocation. Suites are short relative to a run, so the imbalance is
|
|
15
|
+
transient and never oversubscribes by more than one suite's worth.
|
|
16
|
+
* No lock file. Each lease is its own uniquely-named file, created before
|
|
17
|
+
counting, so two simultaneous arrivals each see the other. The remaining race
|
|
18
|
+
(counting before the other's create lands) costs one transiently generous
|
|
19
|
+
split, not a corrupted state.
|
|
20
|
+
* Stale leases must not throttle a machine forever: a lease whose pid is dead is
|
|
21
|
+
swept, and so is one older than STALE_AFTER_S — a suite invocation cannot
|
|
22
|
+
legitimately outlive run_command's timeout, so twice that bounds a live lease.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import contextlib
|
|
28
|
+
import json
|
|
29
|
+
import os
|
|
30
|
+
import time
|
|
31
|
+
import uuid
|
|
32
|
+
from collections.abc import Iterator
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
|
|
35
|
+
LEASE_DIR_ENV = "TDD_LEASE_DIR"
|
|
36
|
+
CORE_BUDGET_ENV = "TDD_CORE_BUDGET"
|
|
37
|
+
|
|
38
|
+
#: run_command's timeout is 1800s; no live suite invocation can be older than that.
|
|
39
|
+
STALE_AFTER_S = 3600
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def lease_dir() -> Path:
|
|
43
|
+
env = os.environ.get(LEASE_DIR_ENV)
|
|
44
|
+
if env:
|
|
45
|
+
return Path(env)
|
|
46
|
+
return Path.home() / ".cache" / "tdd-cli" / "leases"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _total_cores() -> int:
|
|
50
|
+
"""CORE_BUDGET lets an operator reserve headroom (agents themselves need CPU)."""
|
|
51
|
+
with contextlib.suppress(ValueError):
|
|
52
|
+
budget = int(os.environ.get(CORE_BUDGET_ENV, "0"))
|
|
53
|
+
if budget > 0:
|
|
54
|
+
return budget
|
|
55
|
+
return os.cpu_count() or 1
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _pid_alive(pid: int) -> bool:
|
|
59
|
+
try:
|
|
60
|
+
os.kill(pid, 0)
|
|
61
|
+
except ProcessLookupError:
|
|
62
|
+
return False
|
|
63
|
+
except PermissionError:
|
|
64
|
+
return True # exists, owned by someone else
|
|
65
|
+
return True
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _is_live(path: Path) -> bool:
|
|
69
|
+
try:
|
|
70
|
+
if time.time() - path.stat().st_mtime > STALE_AFTER_S:
|
|
71
|
+
return False
|
|
72
|
+
payload = json.loads(path.read_text())
|
|
73
|
+
pid = int(payload["pid"])
|
|
74
|
+
except (OSError, ValueError, KeyError, json.JSONDecodeError):
|
|
75
|
+
return False
|
|
76
|
+
return _pid_alive(pid)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _live_count(directory: Path) -> int:
|
|
80
|
+
"""Count live leases, sweeping the rest so a crash never throttles the machine."""
|
|
81
|
+
live = 0
|
|
82
|
+
for path in directory.glob("*.json"):
|
|
83
|
+
if _is_live(path):
|
|
84
|
+
live += 1
|
|
85
|
+
else:
|
|
86
|
+
with contextlib.suppress(OSError):
|
|
87
|
+
path.unlink()
|
|
88
|
+
return live
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def snapshot() -> dict:
|
|
92
|
+
"""Observe the budget without participating in it — counts live leases but
|
|
93
|
+
sweeps nothing and takes nothing, so a fleet view never perturbs the split."""
|
|
94
|
+
directory = lease_dir()
|
|
95
|
+
total = _total_cores()
|
|
96
|
+
live = 0
|
|
97
|
+
if directory.is_dir():
|
|
98
|
+
live = sum(1 for path in directory.glob("*.json") if _is_live(path))
|
|
99
|
+
return {
|
|
100
|
+
"active": live,
|
|
101
|
+
"total_cores": total,
|
|
102
|
+
"workers_each": max(1, total // max(1, live)),
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@contextlib.contextmanager
|
|
107
|
+
def worker_lease(total_cores: int | None = None) -> Iterator[int]:
|
|
108
|
+
"""Hold a lease for one suite invocation; yields the worker count to use."""
|
|
109
|
+
directory = lease_dir()
|
|
110
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
111
|
+
total = total_cores or _total_cores()
|
|
112
|
+
mine = directory / f"{os.getpid()}-{uuid.uuid4().hex}.json"
|
|
113
|
+
mine.write_text(json.dumps({"pid": os.getpid(), "started_at": time.time()}))
|
|
114
|
+
try:
|
|
115
|
+
yield max(1, total // max(1, _live_count(directory)))
|
|
116
|
+
finally:
|
|
117
|
+
with contextlib.suppress(OSError):
|
|
118
|
+
mine.unlink()
|