tdd-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tddcli/cli.py ADDED
@@ -0,0 +1,1043 @@
1
+ """Command surface (§8). Transport-agnostic by construction (R13.8).
2
+
3
+ No command accepts a phase, a cycle number, or executor identity (R8.3).
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import argparse
9
+ import json
10
+ import os
11
+ import socket
12
+ import sqlite3
13
+ import sys
14
+ import time
15
+ from datetime import datetime, timezone
16
+ from pathlib import Path
17
+
18
+ from . import (
19
+ __version__,
20
+ adapters,
21
+ fleet,
22
+ gitutil,
23
+ identity,
24
+ render,
25
+ snapshot,
26
+ )
27
+ from . import (
28
+ config as config_mod,
29
+ )
30
+ from . import (
31
+ contract as contract_mod,
32
+ )
33
+ from .adapters.base import FAILED, NOT_COLLECTED
34
+ from .advance import advance as do_advance
35
+ from .envelope import Envelope, NextAction, Verb, failure, heartbeat
36
+ from .ledger import Ledger, LedgerVersionError, ledger_path, now
37
+ from .machine import CLOSED, SKIPPED, Engine
38
+
39
+ BLOCKER_KINDS = {
40
+ "regression", "target_unfixable", "bad_red", "plan_defect", "tooling",
41
+ "context_exhausted",
42
+ # Failing, but not caused by this run — a flake, or something the baseline missed.
43
+ # Distinct from `regression`, which records a defect the run introduced.
44
+ "pre_existing_failure",
45
+ }
46
+
47
+
48
+ def _worktree() -> Path:
49
+ return gitutil.worktree_root(Path.cwd())
50
+
51
+
52
+ def _context(require_run: bool = True):
53
+ worktree = _worktree()
54
+ cfg = config_mod.load(worktree)
55
+ ledger = Ledger(gitutil.repo_identity(worktree))
56
+ run = ledger.active_run(str(worktree))
57
+ if require_run and run is None:
58
+ raise SystemExit(
59
+ failure("no active run in this worktree; `tdd run start --plan <path>`").emit()
60
+ )
61
+ return worktree, cfg, ledger, run
62
+
63
+
64
+ def _engine(worktree, cfg, ledger, run) -> Engine:
65
+ return Engine(ledger, cfg, worktree, run)
66
+
67
+
68
+ def _claim_elapsed_s(claim: dict) -> float:
69
+ started = datetime.fromisoformat(claim["started_at"])
70
+ if started.tzinfo is None:
71
+ started = started.replace(tzinfo=timezone.utc)
72
+ return round((datetime.now(timezone.utc) - started).total_seconds(), 2)
73
+
74
+
75
+ def _collecting_envelope(claim: dict) -> Envelope:
76
+ """Shared by `cmd_progress` (JSON and bare) and `cmd_status`: a claim with no run
77
+ row yet is an in-flight baseline, not "never started". `status` is
78
+ documented as the agent's machine view; agents polled `progress` because
79
+ `status` gave them nothing — routing them to the human command to learn machine
80
+ state was the actual defect."""
81
+ return Envelope(
82
+ result={
83
+ "status": "collecting_baseline",
84
+ "projects_done": claim["projects_done"],
85
+ "projects_total": claim["projects_total"],
86
+ "current_project": claim["current_project"],
87
+ "elapsed_s": _claim_elapsed_s(claim),
88
+ },
89
+ next_action=NextAction(
90
+ Verb.AWAIT_BASELINE, "A baseline is being collected; poll `tdd progress` again.",
91
+ ),
92
+ )
93
+
94
+
95
+ # -- commands ------------------------------------------------------------
96
+
97
+
98
+ PYTHON_PROJECT_MARKERS = ("pyproject.toml", "setup.cfg", "setup.py", "pytest.ini")
99
+ VITEST_CONFIGS = ("vitest.config.ts", "vitest.config.js", "vitest.config.mts", "vitest.config.mjs")
100
+
101
+
102
+ def _detect_adapter(directory: Path) -> tuple[str, str] | None:
103
+ """(adapter, test_paths) for a directory, or None when nothing matches.
104
+
105
+ A `package.json` alone is not vitest evidence — a jest or mocha project has
106
+ one too, and a wrong guess here writes a config that fails on first use.
107
+ """
108
+ if any((directory / marker).is_file() for marker in PYTHON_PROJECT_MARKERS):
109
+ return "pytest", '["tests/"]'
110
+ if any((directory / c).is_file() for c in VITEST_CONFIGS):
111
+ return "vitest", '["**/__tests__/**", "**/*.test.ts", "**/*.test.tsx"]'
112
+ pkg = directory / "package.json"
113
+ if pkg.is_file():
114
+ try:
115
+ deps = json.loads(pkg.read_text())
116
+ except (json.JSONDecodeError, OSError):
117
+ return None
118
+ combined = {**deps.get("dependencies", {}), **deps.get("devDependencies", {})}
119
+ if "vitest" in combined:
120
+ return "vitest", '["**/__tests__/**", "**/*.test.ts", "**/*.test.tsx"]'
121
+ return None
122
+
123
+
124
+ def _project_key(name: str) -> str:
125
+ """A bare TOML key: letters, digits, underscore, hyphen."""
126
+ cleaned = "".join(c if c.isalnum() or c in "_-" else "-" for c in name)
127
+ return cleaned or "app"
128
+
129
+
130
+ def cmd_init(args) -> Envelope:
131
+ worktree = _worktree()
132
+ path = worktree / config_mod.CONFIG_NAME
133
+ if path.exists() and not args.force:
134
+ return failure(f"{path} already exists; pass --force to overwrite")
135
+
136
+ detected: list[tuple[str, str, str, str]] = [] # (name, root, adapter, test_paths)
137
+ unmatched: list[str] = []
138
+
139
+ # A single-project repo is the common case: the worktree root is the project.
140
+ root_hit = _detect_adapter(worktree)
141
+ if root_hit is not None:
142
+ detected.append((_project_key(worktree.name), ".", *root_hit))
143
+
144
+ for child in sorted(p for p in worktree.iterdir() if p.is_dir()):
145
+ if child.name.startswith(".") or child.name == "node_modules":
146
+ continue
147
+ hit = _detect_adapter(child)
148
+ if hit is not None:
149
+ detected.append((child.name, child.name, *hit))
150
+ elif (child / "package.json").is_file():
151
+ unmatched.append(child.name)
152
+
153
+ lines = ["# Generated by `tdd init` — review before use (roots are declared, never inferred).", ""]
154
+ for name, root, adapter, tests in detected:
155
+ lines += [
156
+ f"[project.{name}]",
157
+ f'root = "{root}"',
158
+ f'adapter = "{adapter}"',
159
+ f"test_paths = {tests}",
160
+ "lint = []",
161
+ "typecheck = []",
162
+ "",
163
+ ]
164
+ path.write_text("\n".join(lines))
165
+ return Envelope(
166
+ result={
167
+ "written": str(path),
168
+ "detected": [d[0] for d in detected],
169
+ "unmatched": unmatched or None,
170
+ },
171
+ next_action=NextAction(
172
+ Verb.CONFIRM_CYCLE_APPLICABLE,
173
+ "Review tdd.toml: confirm roots, add lint/typecheck commands and artifact edges."
174
+ + (
175
+ f" No supported adapter detected for: {', '.join(unmatched)} —"
176
+ " declare one manually (third-party adapters register under the"
177
+ " `tddcli.adapters` entry-point group)."
178
+ if unmatched else ""
179
+ ),
180
+ ),
181
+ )
182
+
183
+
184
+ LEGACY_ARTIFACTS = (".pytest_report.json", ".tdd-state.json")
185
+ SKIP_DIRS = {"node_modules", ".venv", "venv", "__pycache__", ".git"}
186
+
187
+
188
+ def _legacy_artifacts(worktree: Path) -> list[Path]:
189
+ """Find stale artifacts in *this* worktree only.
190
+
191
+ Nested worktrees under `.claude/worktrees/` are separate checkouts with their own
192
+ in-flight work; scanning into them reports another branch's live files as this
193
+ worktree's problem.
194
+ """
195
+ found: list[Path] = []
196
+
197
+ def walk(directory: Path, depth: int = 0) -> None:
198
+ if depth > 8:
199
+ return
200
+ try:
201
+ entries = list(directory.iterdir())
202
+ except OSError:
203
+ return
204
+ for entry in entries:
205
+ if entry.is_dir():
206
+ if entry.name in SKIP_DIRS:
207
+ continue
208
+ # A nested checkout owns its own state.
209
+ if (entry / ".git").exists():
210
+ continue
211
+ walk(entry, depth + 1)
212
+ elif entry.name in LEGACY_ARTIFACTS:
213
+ found.append(entry)
214
+
215
+ walk(worktree)
216
+ return sorted(found)
217
+
218
+
219
+ def cmd_doctor(args) -> Envelope:
220
+ worktree = _worktree()
221
+ checks: list[dict] = []
222
+
223
+ def check(name, ok, detail="", project=None):
224
+ entry = {"check": name, "ok": bool(ok), "detail": detail}
225
+ if project is not None:
226
+ entry["project"] = project
227
+ checks.append(entry)
228
+
229
+ check("worktree resolvable", True, str(worktree))
230
+ try:
231
+ cfg = config_mod.load(worktree)
232
+ check("tdd.toml valid", True, f"{len(cfg.projects)} projects")
233
+ except config_mod.ConfigError as exc:
234
+ check("tdd.toml valid", False, str(exc))
235
+ return Envelope(ok=False, error="configuration invalid", result={"checks": checks})
236
+
237
+ repo = gitutil.repo_identity(worktree)
238
+ ledger = Ledger(repo)
239
+ check("ledger reachable", True, str(ledger.path))
240
+ check("ledger outside worktree", not str(ledger.path).startswith(str(worktree)), str(ledger.path))
241
+
242
+ projects: dict[str, dict] = {}
243
+ for name, project in cfg.projects.items():
244
+ before = len(checks)
245
+ root = worktree / project.root
246
+ check("root exists", root.is_dir(), str(root), project=name)
247
+ check("adapter known", project.adapter in adapters.available(), project.adapter, project=name)
248
+ check("test_paths declared", bool(project.test_paths), project=name)
249
+ if project.adapter == "pytest":
250
+ # The probe runs in the project's own environment (uv, poetry, pipenv,
251
+ # pdm or the active venv) — hardcoding `uv run` here failed the check
252
+ # on any non-uv project even with the plugin installed.
253
+ probe = adapters.build(project, worktree).plugin_probe_cmd()
254
+ code, out, err = adapters.base.run_command(probe, root)
255
+ check("pytest-json-report installed", code == 0, (err or "")[:200], project=name)
256
+
257
+ # Run before `collectable()` so this actionable message wins over
258
+ # vitest's stack trace: a git worktree does not inherit `node_modules`
259
+ # (it isn't tracked), and `npx vitest` fails with a wall of noise that
260
+ # doesn't say why.
261
+ if project.adapter == "vitest" and not (root / "node_modules").is_dir():
262
+ check(
263
+ "node_modules present", False,
264
+ "git worktrees do not inherit `node_modules` (it isn't tracked)."
265
+ " Symlink it from the main checkout, e.g."
266
+ f" `ln -s <main-checkout>/{project.root}/node_modules {root}/node_modules`.",
267
+ project=name,
268
+ )
269
+
270
+ # Whole-suite `--collect-only`/`vitest list` (§10) — a single, cheap
271
+ # probe (0.04s on a broken project) that attributes a collection failure
272
+ # to its project. Nothing shells out to doctor today, so this is the only
273
+ # place a `ModuleNotFoundError` like the real `pyyaml` incident surfaces.
274
+ adapter = adapters.build(project, worktree)
275
+ gate = adapter.collectable()
276
+ check("collectable", gate.ok, gate.output, project=name)
277
+
278
+ projects[name] = {"ok": all(c["ok"] for c in checks[before:])}
279
+
280
+ for art in cfg.artifacts.values():
281
+ check(f"artifact {art.name}: has check or regenerate", bool(art.check or art.regenerate))
282
+
283
+ stale = _legacy_artifacts(worktree)
284
+ check("no legacy state artifacts", not stale, ", ".join(str(s) for s in stale[:5]))
285
+ check("worktree clean", not gitutil.is_dirty(worktree))
286
+
287
+ ok = all(c["ok"] for c in checks)
288
+ return Envelope(
289
+ ok=ok,
290
+ result={"checks": checks, "projects": projects, "healthy": ok},
291
+ next_action=NextAction(
292
+ Verb.CONFIRM_CYCLE_APPLICABLE if ok else Verb.RESOLVE_BLOCKER,
293
+ "Environment is ready." if ok else "Resolve the failing checks above.",
294
+ ),
295
+ )
296
+
297
+
298
+ def cmd_plan_register(args) -> Envelope:
299
+ worktree = _worktree()
300
+ cfg = config_mod.load(worktree)
301
+ ledger = Ledger(gitutil.repo_identity(worktree))
302
+ rel = str(Path(args.plan))
303
+ try:
304
+ parsed = contract_mod.register(worktree, rel, cfg)
305
+ except contract_mod.ContractError as exc:
306
+ # R7.10 — malformed front-matter is a planning defect and must surface.
307
+ return failure(f"malformed plan contract: {exc}", plan=rel)
308
+
309
+ if parsed.status == "undeclared" and not args.allow_undeclared:
310
+ return failure(
311
+ f"{rel} has no front-matter contract. Add one, or pass --allow-undeclared"
312
+ " (fidelity metrics will be unavailable).",
313
+ plan=rel,
314
+ )
315
+
316
+ existing = ledger.one(
317
+ "SELECT * FROM plan_contract WHERE plan_path = ? AND git_blob_sha IS ?",
318
+ (rel, parsed.blob_sha),
319
+ )
320
+ contract_id = existing["id"] if existing else ledger.insert(
321
+ "plan_contract",
322
+ plan_path=rel,
323
+ git_blob_sha=parsed.blob_sha,
324
+ git_commit=parsed.commit_sha,
325
+ status=parsed.status,
326
+ declared_cycles=contract_mod.cycles_to_json(parsed.cycles),
327
+ annotation_keys=json.dumps(parsed.annotation_keys),
328
+ registered_at=now(),
329
+ )
330
+ return Envelope(
331
+ result={
332
+ "contract_id": contract_id,
333
+ "status": parsed.status,
334
+ "blob": parsed.blob_sha,
335
+ "cycles": len(parsed.cycles),
336
+ "kinds": {k: sum(1 for c in parsed.cycles if c.kind == k)
337
+ for k in {c.kind for c in parsed.cycles}},
338
+ "reused": bool(existing),
339
+ },
340
+ next_action=NextAction(
341
+ Verb.CONFIRM_CYCLE_APPLICABLE, f"Contract registered. `tdd run start --plan {rel}`."
342
+ ),
343
+ )
344
+
345
+
346
+ def _probe_projects(cfg, worktree, ledger, on_progress):
347
+ """Probe every project's baseline (R9.5a): run + collect, timing each, emitting a
348
+ `baseline_captured` heartbeat, and calling `on_progress(done, name)` — extracted
349
+ from `cmd_run_start`, which carried claiming, timing, heartbeating and progress
350
+ updates inline past the point of legibility. Returns `{name: (verdict, collection)}`.
351
+ """
352
+ probes = {}
353
+ for done, (name, project) in enumerate(cfg.projects.items(), start=1):
354
+ adapter = adapters.build(project, worktree)
355
+ started = time.monotonic()
356
+ verdict, collection = adapter.run(None), adapter.collect()
357
+ elapsed = time.monotonic() - started
358
+ probes[name] = (verdict, collection)
359
+ heartbeat(
360
+ event="baseline_captured", project=name,
361
+ test_count=len(collection.tests), elapsed_s=round(elapsed, 2),
362
+ )
363
+ on_progress(done, name)
364
+ return probes
365
+
366
+
367
+ def cmd_run_start(args) -> Envelope:
368
+ worktree = _worktree()
369
+ cfg = config_mod.load(worktree)
370
+ ledger = Ledger(gitutil.repo_identity(worktree))
371
+
372
+ active = ledger.active_run(str(worktree))
373
+ if active is not None:
374
+ return failure(
375
+ "a run is already active in this worktree",
376
+ reason="run_already_active", run_id=active["id"], started_at=active["started_at"],
377
+ )
378
+
379
+ rel = str(Path(args.plan))
380
+ contract_row = ledger.one(
381
+ "SELECT * FROM plan_contract WHERE plan_path = ? ORDER BY id DESC LIMIT 1", (rel,)
382
+ )
383
+ if contract_row is None:
384
+ return failure(f"{rel} is not registered; run `tdd plan register {rel}` first")
385
+
386
+ # R7.11 — the plan blob is the contract; drift must surface.
387
+ blob_changed = False
388
+ try:
389
+ current_blob, _ = gitutil.blob_sha_at_head(worktree, rel)
390
+ blob_changed = bool(
391
+ contract_row["git_blob_sha"] and current_blob != contract_row["git_blob_sha"]
392
+ )
393
+ except gitutil.GitError:
394
+ pass
395
+
396
+ dirty = sorted(gitutil.dirty_paths(worktree))
397
+ if dirty and not args.allow_dirty:
398
+ return failure(
399
+ "working tree is dirty; commit first or pass --allow-dirty"
400
+ " (pre-existing changes are then excluded from authorship forever)",
401
+ dirty=dirty,
402
+ )
403
+
404
+ if contract_row["status"] == "undeclared" and not args.allow_undeclared:
405
+ return failure("contract is undeclared; pass --allow-undeclared")
406
+
407
+ # Claim the worktree before probing: two `run start` calls against
408
+ # one worktree must not both pass the baseline window. `Ledger.claim`'s `UNIQUE`
409
+ # insert is the lock — do not read-then-write, which is the race this
410
+ # closes. A claim whose owner is gone (e.g. a `SIGKILL`ed `run start`)
411
+ # is reclaimed rather than obeyed, or one dead process bricks the worktree
412
+ # forever; `active_claim` only computes staleness, it never deletes.
413
+ existing = ledger.active_claim(str(worktree))
414
+ if existing is not None and existing["stale"]:
415
+ ledger.release_claim(str(worktree))
416
+
417
+ try:
418
+ ledger.claim(
419
+ str(worktree), hostname=socket.gethostname(), pid=os.getpid(),
420
+ projects_total=len(cfg.projects),
421
+ )
422
+ except sqlite3.IntegrityError:
423
+ return failure(
424
+ "a baseline is already being collected in this worktree; do not re-run"
425
+ " `run start` — poll `tdd progress` instead, which reports"
426
+ " `collecting_baseline` with per-project counters until it finishes",
427
+ reason="baseline_in_progress",
428
+ )
429
+
430
+ try:
431
+ # Probe every project before the run exists (R9.5a). A baseline is subtracted
432
+ # from every later failure set, so an untrustworthy one is worse than none: it
433
+ # reports pre-existing failures as regressions for the life of the run.
434
+ # Refusing here also leaves no half-started run behind to block the next
435
+ # attempt — and must release the claim too, or the retry it invites is itself
436
+ # refused.
437
+ probes = _probe_projects(
438
+ cfg, worktree, ledger,
439
+ on_progress=lambda done, name: ledger.update_claim(
440
+ str(worktree), projects_done=done, current_project=name,
441
+ ),
442
+ )
443
+ for name, (verdict, collection) in probes.items():
444
+ if not collection.tests and collection.failed_files:
445
+ sample = sorted(collection.failed_files)[0]
446
+ return failure(
447
+ f"{name}: no test could be collected — {len(collection.failed_files)}"
448
+ f" file(s) failed to collect, starting with {sample}. The baseline"
449
+ " would record no failures and every pre-existing failure would then"
450
+ " read as a regression. Fix the environment (dependencies"
451
+ " installed?) and retry.",
452
+ project=name, failed_files=sorted(collection.failed_files),
453
+ )
454
+ if collection.tests and not verdict.passed and not verdict.failed:
455
+ return failure(
456
+ f"{name}: the suite collected {len(collection.tests)} test(s) but"
457
+ " the baseline run executed no tests, so it observed nothing. Check"
458
+ " `test_command` in tdd.toml and retry.",
459
+ project=name, collected=len(collection.tests),
460
+ )
461
+
462
+ executor = identity.resolve(worktree, args.executor)
463
+ run_id = ledger.insert(
464
+ "run",
465
+ plan_contract_id=contract_row["id"],
466
+ executor_model=executor.model,
467
+ executor_session=executor.session,
468
+ executor_source=executor.source,
469
+ worktree_path=str(worktree),
470
+ started_at=now(),
471
+ allow_dirty=int(bool(args.allow_dirty)),
472
+ preexisting_dirty=json.dumps(dirty),
473
+ config_sha=config_mod.config_sha(worktree),
474
+ )
475
+ run = ledger.one("SELECT * FROM run WHERE id = ?", (run_id,))
476
+ if blob_changed:
477
+ ledger.event(run_id, None, "plan_blob_changed", rel)
478
+
479
+ # Baselines and the collection snapshot, per project (R9.5, R8.9) — from the
480
+ # probe above, so the suite is not run twice.
481
+ for name, (verdict, collection) in probes.items():
482
+ ledger.insert(
483
+ "baseline", run_id=run_id, project=name,
484
+ failing=json.dumps(sorted(verdict.failed)), captured_at=now(),
485
+ )
486
+ ledger.insert(
487
+ "collection_snapshot", run_id=run_id, project=name,
488
+ tests=json.dumps(sorted(collection.tests)),
489
+ failed_files=json.dumps(collection.failed_files), captured_at=now(),
490
+ )
491
+
492
+ engine = Engine(ledger, cfg, worktree, run)
493
+ engine.check_artifacts(None)
494
+ first = engine.declared[0] if engine.declared else None
495
+ if first is None:
496
+ return failure("contract declares no cycles")
497
+ cycle = engine.open_cycle(first.ordinal)
498
+
499
+ verb, opening = engine.opening_action(cycle)
500
+ detail = f"Run {run_id} started ({executor.model}, via {executor.source}). {opening}"
501
+ return Envelope(
502
+ run=engine.run_state(cycle),
503
+ result={
504
+ "baselines": {
505
+ n: len(v) for n, v in ledger.baselines(run_id).items()
506
+ },
507
+ "executor_source": executor.source,
508
+ },
509
+ next_action=NextAction(verb, detail),
510
+ )
511
+ finally:
512
+ ledger.release_claim(str(worktree))
513
+
514
+
515
+ def cmd_status(args) -> Envelope:
516
+ worktree, cfg, ledger, run = _context(require_run=False)
517
+ if run is None:
518
+ claim = ledger.active_claim(str(worktree))
519
+ if claim is not None:
520
+ return _collecting_envelope(claim)
521
+ return Envelope(
522
+ result={"active": False},
523
+ next_action=NextAction(
524
+ Verb.CONFIRM_CYCLE_APPLICABLE, "No active run. `tdd run start --plan <path>`."
525
+ ),
526
+ )
527
+ engine = _engine(worktree, cfg, ledger, run)
528
+ cycle = ledger.open_cycle(run["id"])
529
+ if cycle is None:
530
+ return Envelope(
531
+ run={"id": run["id"], "phase": CLOSED},
532
+ next_action=NextAction(Verb.COMPLETE, "Run complete."),
533
+ )
534
+ attempts = len(ledger.invocations(cycle["id"], cycle["phase"]))
535
+ return Envelope(
536
+ run=engine.run_state(cycle),
537
+ result={
538
+ "attempts_in_phase": attempts,
539
+ "targets": json.loads(cycle["target_tests"]),
540
+ "sensitivity_open": ledger.open_sensitivity(cycle["id"]) is not None,
541
+ },
542
+ next_action=NextAction(
543
+ Verb.REFACTOR_OR_ADVANCE, "Run `tdd advance` to evaluate the current phase."
544
+ ),
545
+ )
546
+
547
+
548
+ def cmd_advance(args) -> Envelope:
549
+ worktree, cfg, ledger, run = _context()
550
+ engine = _engine(worktree, cfg, ledger, run)
551
+ cycle = ledger.open_cycle(run["id"])
552
+ if cycle is None:
553
+ return Envelope(
554
+ run={"id": run["id"], "phase": CLOSED},
555
+ next_action=NextAction(Verb.COMPLETE, "All cycles complete."),
556
+ )
557
+ return do_advance(engine, cycle, retry=args.retry)
558
+
559
+
560
+ def cmd_cycle_skip(args) -> Envelope:
561
+ worktree, cfg, ledger, run = _context()
562
+ engine = _engine(worktree, cfg, ledger, run)
563
+ cycle = ledger.open_cycle(run["id"])
564
+ if cycle is None:
565
+ return failure("no open cycle")
566
+ ledger.update(
567
+ "cycle", cycle["id"], phase=SKIPPED, closed_at=now(), skip_reason=args.reason
568
+ )
569
+ ledger.insert(
570
+ "transition", cycle_id=cycle["id"], from_phase=cycle["phase"],
571
+ to_phase=SKIPPED, at=now(),
572
+ )
573
+ nxt_declared = next(
574
+ (c for c in engine.declared if c.ordinal > cycle["ordinal"]), None
575
+ )
576
+ if nxt_declared is None:
577
+ ledger.update("run", run["id"], ended_at=now(), outcome="complete")
578
+ return Envelope(
579
+ run={"id": run["id"], "cycle": cycle["ordinal"], "phase": SKIPPED},
580
+ next_action=NextAction(Verb.COMPLETE, "Final cycle skipped; run complete."),
581
+ )
582
+ nxt = engine.open_cycle(nxt_declared.ordinal)
583
+ verb, opening = engine.opening_action(nxt)
584
+ return Envelope(
585
+ run=engine.run_state(nxt),
586
+ result={"skipped": cycle["ordinal"], "reason": args.reason},
587
+ next_action=NextAction(
588
+ verb, f"Cycle {cycle['ordinal']} skipped. {opening}"
589
+ ),
590
+ )
591
+
592
+
593
+ def cmd_annotate(args) -> Envelope:
594
+ worktree, cfg, ledger, run = _context()
595
+ cycle = ledger.open_cycle(run["id"])
596
+ ledger.insert(
597
+ "annotation", run_id=run["id"], cycle_id=cycle["id"] if cycle else None,
598
+ key=args.key, value=args.value, at=now(),
599
+ )
600
+ return Envelope(
601
+ run={"id": run["id"], "cycle": cycle["ordinal"] if cycle else None},
602
+ result={"key": args.key},
603
+ next_action=NextAction(Verb.REFACTOR_OR_ADVANCE, "Annotation recorded. `tdd advance`."),
604
+ )
605
+
606
+
607
+ def cmd_blocker(args) -> Envelope:
608
+ worktree, cfg, ledger, run = _context()
609
+ if args.kind not in BLOCKER_KINDS:
610
+ return failure(f"unknown blocker kind {args.kind!r}; use one of {sorted(BLOCKER_KINDS)}")
611
+ cycle = ledger.open_cycle(run["id"])
612
+ ledger.insert(
613
+ "blocker", run_id=run["id"], cycle_id=cycle["id"] if cycle else None,
614
+ kind=args.kind, detail=args.detail, at=now(),
615
+ )
616
+ # R8.7 — a blocked run is not live, so the stop hook must release.
617
+ ledger.update("run", run["id"], ended_at=now(), outcome="blocked")
618
+ return Envelope(
619
+ run={"id": run["id"], "cycle": cycle["ordinal"] if cycle else None, "phase": "BLOCKED"},
620
+ result={"kind": args.kind, "detail": args.detail},
621
+ next_action=NextAction(
622
+ Verb.BLOCKED,
623
+ f"Run blocked ({args.kind}). A human can resume with"
624
+ " `tdd resume --unblock --note ...`.",
625
+ ),
626
+ )
627
+
628
+
629
+ def _accept_failures_into_baseline(ledger: Ledger, run_id: int) -> dict[str, list[str]]:
630
+ """Fold the failures the last close sweep saw into the baseline (R9.5b).
631
+
632
+ A run whose baseline missed a failure cannot otherwise recover: unblocking returns
633
+ it to the phase it blocked in, and the next sweep finds the same failure. Only a
634
+ human reaches this, only by asking, and what was accepted is recorded.
635
+ """
636
+ latest = ledger.all(
637
+ "SELECT project, other_failures FROM invocation WHERE id IN ("
638
+ " SELECT MAX(id) FROM invocation WHERE run_id = ? AND phase_at = 'CLOSE_SWEEP'"
639
+ " GROUP BY project)",
640
+ (run_id,),
641
+ )
642
+ rows = {
643
+ r["project"]: r
644
+ for r in ledger.all("SELECT * FROM baseline WHERE run_id = ?", (run_id,))
645
+ }
646
+ accepted: dict[str, list[str]] = {}
647
+ for sweep in latest:
648
+ row = rows.get(sweep["project"])
649
+ if row is None:
650
+ continue
651
+ known = set(json.loads(row["failing"]))
652
+ new = sorted(set(json.loads(sweep["other_failures"])) - known)
653
+ if not new:
654
+ continue
655
+ ledger.update("baseline", row["id"], failing=json.dumps(sorted(known | set(new))))
656
+ accepted[sweep["project"]] = new
657
+ if accepted:
658
+ ledger.event(run_id, None, "baseline_amended", json.dumps(accepted))
659
+ return accepted
660
+
661
+
662
+ def cmd_resume(args) -> Envelope:
663
+ worktree = _worktree()
664
+ cfg = config_mod.load(worktree)
665
+ ledger = Ledger(gitutil.repo_identity(worktree))
666
+ run = ledger.active_run(str(worktree))
667
+ accepted: dict[str, list[str]] = {}
668
+
669
+ if args.accept_failures and not args.unblock:
670
+ return failure("--accept-failures applies to --unblock")
671
+
672
+ if args.unblock:
673
+ if run is not None:
674
+ return failure("run is already live; --unblock applies to a blocked run")
675
+ blocked = ledger.one(
676
+ "SELECT * FROM run WHERE worktree_path = ? AND outcome = 'blocked'"
677
+ " ORDER BY id DESC LIMIT 1",
678
+ (str(worktree),),
679
+ )
680
+ if blocked is None:
681
+ return failure("no blocked run to unblock in this worktree")
682
+ if not args.note:
683
+ return failure("--unblock requires --note describing the intervention")
684
+ ledger.update("run", blocked["id"], ended_at=None, outcome=None)
685
+ ledger.insert(
686
+ "human_intervention", run_id=blocked["id"], note=args.note, at=now()
687
+ )
688
+ if args.accept_failures:
689
+ accepted = _accept_failures_into_baseline(ledger, blocked["id"])
690
+ run = ledger.one("SELECT * FROM run WHERE id = ?", (blocked["id"],))
691
+
692
+ if run is None:
693
+ return failure("no active run in this worktree")
694
+
695
+ engine = _engine(worktree, cfg, ledger, run)
696
+ cycle = ledger.open_cycle(run["id"])
697
+ if cycle is None:
698
+ return Envelope(
699
+ run={"id": run["id"], "phase": CLOSED},
700
+ next_action=NextAction(Verb.COMPLETE, "Run complete."),
701
+ )
702
+ result = {"resumed": True}
703
+ if accepted:
704
+ result["accepted_into_baseline"] = accepted
705
+ return Envelope(
706
+ run=engine.run_state(cycle),
707
+ result=result,
708
+ next_action=NextAction(
709
+ Verb.REFACTOR_OR_ADVANCE,
710
+ f"Resumed at cycle {cycle['ordinal']}, phase {cycle['phase']}. Run `tdd advance`.",
711
+ ),
712
+ )
713
+
714
+
715
+ def cmd_sensitivity(args) -> Envelope:
716
+ worktree, cfg, ledger, run = _context()
717
+ cycle = ledger.open_cycle(run["id"])
718
+ if cycle is None:
719
+ return failure("no open cycle")
720
+
721
+ if args.step == "begin":
722
+ if ledger.open_sensitivity(cycle["id"]) is not None:
723
+ return failure("a sensitivity check is already open")
724
+ check_id = ledger.insert(
725
+ "sensitivity_check",
726
+ cycle_id=cycle["id"],
727
+ reference_diff=snapshot.capture(worktree, cfg),
728
+ reference_untracked=snapshot.fingerprint(worktree, cfg),
729
+ opened_at=now(),
730
+ )
731
+ return Envelope(
732
+ run={"id": run["id"], "cycle": cycle["ordinal"]},
733
+ result={"check_id": check_id},
734
+ next_action=NextAction(
735
+ Verb.RUN_SENSITIVITY_CHECK,
736
+ "Reference state recorded. Mutate the behaviour under test, then"
737
+ " `tdd sensitivity check`.",
738
+ ),
739
+ )
740
+
741
+ open_check = ledger.open_sensitivity(cycle["id"])
742
+ if open_check is None:
743
+ return failure("no open sensitivity check; run `tdd sensitivity begin` first")
744
+ engine = _engine(worktree, cfg, ledger, run)
745
+ targets = json.loads(cycle["target_tests"])
746
+ projects = json.loads(cycle["projects"])
747
+
748
+ if args.step == "check":
749
+ outcomes, _, _, failure_text = engine.run_projects(
750
+ projects, targets, cycle, "SENSITIVITY", False
751
+ )
752
+ # A mutation that breaks collection also proves the test depends on the code.
753
+ bites = bool(outcomes) and all(
754
+ o in (FAILED, NOT_COLLECTED) for o in outcomes.values()
755
+ )
756
+ ledger.update(
757
+ "sensitivity_check", open_check["id"],
758
+ mutation_diff=gitutil.diff_text(worktree)[:20000],
759
+ observed_failure=failure_text[:4000],
760
+ )
761
+ if not bites:
762
+ return Envelope(
763
+ ok=False,
764
+ error="the mutation did not make the target fail — the test pins nothing",
765
+ run={"id": run["id"], "cycle": cycle["ordinal"]},
766
+ result={"outcomes": outcomes},
767
+ next_action=NextAction(
768
+ Verb.RUN_SENSITIVITY_CHECK,
769
+ "Strengthen the mutation or the assertion, then check again.",
770
+ ),
771
+ )
772
+ return Envelope(
773
+ run={"id": run["id"], "cycle": cycle["ordinal"]},
774
+ result={"outcomes": outcomes, "observed_failure": failure_text[:800]},
775
+ next_action=NextAction(
776
+ Verb.RUN_SENSITIVITY_CHECK,
777
+ "The test fails under mutation. Restore with `tdd sensitivity end`.",
778
+ ),
779
+ )
780
+
781
+ # end — restore and verify byte-identical (R8.5)
782
+ to_restore = snapshot.restore(worktree, cfg, open_check["reference_diff"])
783
+ restored_ok = (
784
+ snapshot.fingerprint(worktree, cfg) == open_check["reference_untracked"]
785
+ )
786
+ ledger.update(
787
+ "sensitivity_check", open_check["id"],
788
+ restored_ok=int(restored_ok), closed_at=now(),
789
+ )
790
+ if not restored_ok:
791
+ ledger.event(run["id"], cycle["id"], "restore_mismatch", json.dumps(to_restore))
792
+ return Envelope(
793
+ ok=False,
794
+ error="restore is not byte-identical to the reference state",
795
+ run={"id": run["id"], "cycle": cycle["ordinal"]},
796
+ result={"restored": to_restore},
797
+ next_action=NextAction(
798
+ Verb.RESOLVE_BLOCKER,
799
+ "The working tree does not match the pre-mutation state. Restore it by"
800
+ " hand before continuing.",
801
+ ),
802
+ )
803
+ return Envelope(
804
+ run={"id": run["id"], "cycle": cycle["ordinal"]},
805
+ result={"restored": to_restore, "restored_ok": True},
806
+ next_action=NextAction(Verb.REFACTOR_OR_ADVANCE, "Restored and verified. `tdd advance`."),
807
+ )
808
+
809
+
810
+ def cmd_target(args) -> Envelope:
811
+ worktree, cfg, ledger, run = _context()
812
+ cycle = ledger.open_cycle(run["id"])
813
+ if cycle is None:
814
+ return failure("no open cycle")
815
+ ledger.update("cycle", cycle["id"], target_tests=json.dumps([args.test]))
816
+ ledger.event(run["id"], cycle["id"], "target_named_by_agent", args.test)
817
+ return Envelope(
818
+ run={"id": run["id"], "cycle": cycle["ordinal"]},
819
+ result={"target": args.test},
820
+ next_action=NextAction(Verb.REFACTOR_OR_ADVANCE, "Target set. `tdd advance`."),
821
+ )
822
+
823
+
824
+ def cmd_log_render(args) -> Envelope:
825
+ worktree, cfg, ledger, run = _context(require_run=False)
826
+ if run is None:
827
+ run = ledger.one(
828
+ "SELECT * FROM run WHERE worktree_path = ? ORDER BY id DESC LIMIT 1",
829
+ (str(worktree),),
830
+ )
831
+ if run is None:
832
+ return failure("no runs recorded for this worktree")
833
+ text = render.friction_log(ledger, run)
834
+ if args.out:
835
+ # R9.15 — a relative --out is worktree-relative, never cwd-relative: the
836
+ # command is run from wherever the agent happens to be standing, and
837
+ # `tasks/friction-logs/` means the one at the root of the repo.
838
+ out = Path(args.out)
839
+ if not out.is_absolute():
840
+ out = worktree / out
841
+ out.parent.mkdir(parents=True, exist_ok=True)
842
+ out.write_text(text)
843
+ written = str(out.relative_to(worktree)) if not Path(args.out).is_absolute() else str(out)
844
+ return Envelope(
845
+ result={"written": written, "path": str(out)},
846
+ next_action=NextAction(Verb.COMPLETE, f"Friction log written to {out}."),
847
+ )
848
+ sys.stdout.write(text)
849
+ return Envelope(
850
+ result={"rendered": True},
851
+ next_action=NextAction(Verb.COMPLETE, "Rendered."),
852
+ silent=True,
853
+ )
854
+
855
+
856
+ def cmd_progress(args) -> Envelope:
857
+ """Human-readable progress. `status` remains the agent's machine view."""
858
+ worktree, cfg, ledger, run = _context(require_run=False)
859
+ if run is None:
860
+ run = ledger.one(
861
+ "SELECT * FROM run WHERE worktree_path = ? ORDER BY id DESC LIMIT 1",
862
+ (str(worktree),),
863
+ )
864
+ if run is None:
865
+ # A baseline can take minutes; a claim with no run row yet is in-flight, not
866
+ # "never started". `ok: true` — a polling agent must not see
867
+ # repeated exit-1, the signal that caused the re-runs in the first place.
868
+ # Leaving the human form saying "no runs recorded" while JSON says
869
+ # "collecting" would be the same ambiguity in a new place.
870
+ claim = ledger.active_claim(str(worktree))
871
+ if claim is not None:
872
+ envelope = _collecting_envelope(claim)
873
+ if args.json:
874
+ return envelope
875
+ result = envelope.result
876
+ current = f" (current: {result['current_project']})" if result["current_project"] else ""
877
+ sys.stdout.write(
878
+ f"collecting baseline: {result['projects_done']}/{result['projects_total']}"
879
+ f" projects{current} — {result['elapsed_s']}s elapsed\n"
880
+ )
881
+ envelope.silent = True
882
+ return envelope
883
+ return failure("no runs recorded for this worktree")
884
+ if args.json:
885
+ engine = _engine(worktree, cfg, ledger, run)
886
+ cycle = ledger.open_cycle(run["id"])
887
+ return Envelope(
888
+ run=engine.run_state(cycle) if cycle else {"id": run["id"], "phase": CLOSED},
889
+ result=render.metrics(ledger, str(worktree)),
890
+ next_action=NextAction(Verb.COMPLETE, "Progress reported."),
891
+ )
892
+ sys.stdout.write(render.progress(ledger, run))
893
+ return Envelope(
894
+ result={"rendered": True},
895
+ next_action=NextAction(Verb.COMPLETE, "Progress rendered."),
896
+ silent=True,
897
+ )
898
+
899
+
900
+ def cmd_fleet(args) -> Envelope:
901
+ """Every worktree's active run against this repository, plus in-flight
902
+ baselines and currently executing suites. Deliberately does not use
903
+ `_context`: no tdd.toml, active run, or existing ledger is required, and the
904
+ ledger is opened read-only so live agents cannot be perturbed."""
905
+ worktree = _worktree()
906
+ summary = fleet.summarise(ledger_path(gitutil.repo_identity(worktree)))
907
+ if args.json:
908
+ return Envelope(
909
+ result=summary, next_action=NextAction(Verb.COMPLETE, "Fleet reported.")
910
+ )
911
+ sys.stdout.write(fleet.render(summary))
912
+ return Envelope(
913
+ result=summary,
914
+ next_action=NextAction(Verb.COMPLETE, "Fleet rendered."),
915
+ silent=True,
916
+ )
917
+
918
+
919
+ def cmd_metrics(args) -> Envelope:
920
+ worktree, cfg, ledger, run = _context(require_run=False)
921
+ return Envelope(
922
+ result=render.metrics(ledger, str(worktree)),
923
+ next_action=NextAction(Verb.COMPLETE, "Metrics computed."),
924
+ )
925
+
926
+
927
+ # -- parser --------------------------------------------------------------
928
+
929
+
930
+ def build_parser() -> argparse.ArgumentParser:
931
+ p = argparse.ArgumentParser(prog="tdd", description=__doc__)
932
+ p.add_argument("--version", action="version", version=f"tdd-cli {__version__}")
933
+ sub = p.add_subparsers(dest="command", required=True)
934
+
935
+ s = sub.add_parser("init", help="scaffold tdd.toml for review")
936
+ s.add_argument("--force", action="store_true")
937
+ s.set_defaults(fn=cmd_init)
938
+
939
+ s = sub.add_parser("doctor", help="environment preflight")
940
+ s.set_defaults(fn=cmd_doctor)
941
+
942
+ plan = sub.add_parser("plan", help="plan contracts").add_subparsers(
943
+ dest="plan_command", required=True
944
+ )
945
+ s = plan.add_parser("register")
946
+ s.add_argument("plan")
947
+ s.add_argument("--allow-undeclared", action="store_true")
948
+ s.set_defaults(fn=cmd_plan_register)
949
+
950
+ run_p = sub.add_parser("run", help="runs").add_subparsers(dest="run_command", required=True)
951
+ s = run_p.add_parser("start")
952
+ s.add_argument("--plan", required=True)
953
+ s.add_argument("--executor", help="human-supplied label; agents must not use this")
954
+ s.add_argument("--allow-dirty", action="store_true")
955
+ s.add_argument("--allow-undeclared", action="store_true")
956
+ s.set_defaults(fn=cmd_run_start)
957
+
958
+ s = sub.add_parser("status")
959
+ s.set_defaults(fn=cmd_status)
960
+
961
+ s = sub.add_parser("advance", help="the only command that changes phase")
962
+ s.add_argument("--retry", action="store_true", help="re-run an unchanged tree")
963
+ s.set_defaults(fn=cmd_advance)
964
+
965
+ cyc = sub.add_parser("cycle").add_subparsers(dest="cycle_command", required=True)
966
+ s = cyc.add_parser("skip")
967
+ s.add_argument("--reason", required=True)
968
+ s.set_defaults(fn=cmd_cycle_skip)
969
+
970
+ s = sub.add_parser("annotate")
971
+ s.add_argument("--key", required=True)
972
+ s.add_argument("--value", required=True)
973
+ s.set_defaults(fn=cmd_annotate)
974
+
975
+ s = sub.add_parser("blocker")
976
+ s.add_argument("--kind", required=True)
977
+ s.add_argument("--detail", required=True)
978
+ s.set_defaults(fn=cmd_blocker)
979
+
980
+ s = sub.add_parser("resume")
981
+ s.add_argument("--unblock", action="store_true")
982
+ s.add_argument("--note")
983
+ s.add_argument(
984
+ "--accept-failures",
985
+ action="store_true",
986
+ help="fold the failures the last close sweep saw into the baseline, so a run"
987
+ " whose baseline missed them can proceed; recorded as baseline_amended",
988
+ )
989
+ s.set_defaults(fn=cmd_resume)
990
+
991
+ s = sub.add_parser("sensitivity")
992
+ s.add_argument("step", choices=["begin", "check", "end"])
993
+ s.set_defaults(fn=cmd_sensitivity)
994
+
995
+ s = sub.add_parser("target", help="name the target test when several new tests appeared")
996
+ s.add_argument("test")
997
+ s.set_defaults(fn=cmd_target)
998
+
999
+ log = sub.add_parser("log").add_subparsers(dest="log_command", required=True)
1000
+ s = log.add_parser("render")
1001
+ s.add_argument(
1002
+ "--out",
1003
+ help="write here instead of stdout; a relative path is resolved from the"
1004
+ " worktree root, not the current directory",
1005
+ )
1006
+ s.set_defaults(fn=cmd_log_render)
1007
+
1008
+ s = sub.add_parser("progress", help="human-readable plan progress")
1009
+ s.add_argument("--json", action="store_true", help="machine output instead")
1010
+ s.set_defaults(fn=cmd_progress)
1011
+
1012
+ s = sub.add_parser(
1013
+ "fleet", help="all active runs on this repository, across every worktree"
1014
+ )
1015
+ s.add_argument("--json", action="store_true", help="machine output instead")
1016
+ s.set_defaults(fn=cmd_fleet)
1017
+
1018
+ s = sub.add_parser("metrics")
1019
+ s.set_defaults(fn=cmd_metrics)
1020
+ return p
1021
+
1022
+
1023
+ def main(argv: list[str] | None = None) -> int:
1024
+ if os.name == "nt":
1025
+ # Worker leases, process-liveness checks, and cache paths are POSIX-only.
1026
+ # Failing here, loudly, beats corrupting a lease directory ten minutes in.
1027
+ return failure(
1028
+ "tdd-cli does not support Windows: worker leases and process-liveness"
1029
+ " checks are POSIX-only. Run it under WSL instead.",
1030
+ reason="unsupported_platform",
1031
+ ).emit()
1032
+ try:
1033
+ args = build_parser().parse_args(argv)
1034
+ envelope = args.fn(args)
1035
+ except (config_mod.ConfigError, gitutil.GitError, LedgerVersionError) as exc:
1036
+ envelope = failure(str(exc))
1037
+ except SystemExit as exc:
1038
+ return int(exc.code or 0)
1039
+ return envelope.emit()
1040
+
1041
+
1042
+ if __name__ == "__main__":
1043
+ raise SystemExit(main())