tdd-cli 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/CHANGELOG.md +30 -0
  2. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/PKG-INFO +2 -2
  3. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/pyproject.toml +1 -1
  4. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/__init__.py +1 -1
  5. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/adapters/base.py +29 -1
  6. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/adapters/pytest_adapter.py +5 -2
  7. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/adapters/vitest_adapter.py +5 -4
  8. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/cli.py +89 -8
  9. tdd_cli-0.4.0/tests/test_doctor_blockers.py +173 -0
  10. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_failure_clipping.py +1 -1
  11. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_project_commands.py +1 -1
  12. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_project_env.py +3 -3
  13. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_suite_overrides.py +10 -10
  14. tdd_cli-0.4.0/tests/test_timing_visibility.py +149 -0
  15. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_worker_leases.py +1 -1
  16. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/.gitignore +0 -0
  17. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/LICENSE +0 -0
  18. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/README.md +0 -0
  19. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/SECURITY.md +0 -0
  20. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/claude-code-hooks/README.md +0 -0
  21. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/claude-code-hooks/bash_hook.py +0 -0
  22. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/claude-code-hooks/stop_hook.py +0 -0
  23. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/plan.md +0 -0
  24. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/skills/tdd-drive/README.md +0 -0
  25. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/skills/tdd-drive/SKILL.md +0 -0
  26. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/skills/tdd-handoff/README.md +0 -0
  27. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
  28. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/adapters/__init__.py +0 -0
  29. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/advance.py +0 -0
  30. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/config.py +0 -0
  31. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/contract.py +0 -0
  32. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/envelope.py +0 -0
  33. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/fleet.py +0 -0
  34. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/gitutil.py +0 -0
  35. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/identity.py +0 -0
  36. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/leases.py +0 -0
  37. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/ledger.py +0 -0
  38. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/machine.py +0 -0
  39. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/render.py +0 -0
  40. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/snapshot.py +0 -0
  41. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/staging.py +0 -0
  42. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/conftest.py +0 -0
  43. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_artifact_regeneration.py +0 -0
  44. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_baseline_integrity.py +0 -0
  45. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_config_and_staging.py +0 -0
  46. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_config_drift.py +0 -0
  47. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_contract.py +0 -0
  48. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_doctor_attribution.py +0 -0
  49. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_end_to_end.py +0 -0
  50. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_example_plan.py +0 -0
  51. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_fleet.py +0 -0
  52. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_heartbeat.py +0 -0
  53. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_init_detection.py +0 -0
  54. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_pin_cycles.py +0 -0
  55. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_progress.py +0 -0
  56. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_python_env_managers.py +0 -0
  57. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_refactor_cycles.py +0 -0
  58. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_release_surface.py +0 -0
  59. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_run_claim.py +0 -0
  60. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_single_project_repo.py +0 -0
  61. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_snapshot_and_identity.py +0 -0
  62. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_stub_hint.py +0 -0
  63. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_target_validation.py +0 -0
  64. {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_vitest_adapter.py +0 -0
@@ -4,6 +4,36 @@ All notable changes to this project are documented here.
4
4
  The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
5
5
  and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
+ ## [0.4.0] - 2026-08-16
8
+
9
+ ### Added
10
+
11
+ - `baseline_captured` reports `run_s` and `collect_s` alongside `elapsed_s`. The
12
+ suite run and the per-file collection have unrelated cost models — one scales
13
+ with tests, the other with files — so a single total could not say which was
14
+ slow, and answering that meant measuring projects by hand outside the tool.
15
+ - `TDD_TIMING=1` emits a `command_timing` line per subprocess on stderr
16
+ (`label`, `command`, `cwd`, `duration_ms`, `exit_code`), covering every
17
+ subprocess the tool spawns: suite runs, per-file collection, lint/typecheck
18
+ gates, doctor probes and artifact hooks. Off by default — the per-file loop
19
+ would otherwise emit one line per test file on every invocation. `label` is
20
+ one of `suite`, `collect`, `gate`, `doctor`; an unlabelled row comes from a
21
+ third-party adapter, since every built-in call site names itself (R8.4).
22
+
23
+ ### Fixed
24
+
25
+ - `tdd doctor` no longer emits a blocker it cannot explain. Every failing check
26
+ now carries a `detail` naming what to fix, enforced by the checklist recorder
27
+ so a check added later inherits the guarantee. Previously `worktree clean`
28
+ failed with `detail: ""`, leaving an agent with `resolve_blocker` and nothing
29
+ to resolve — it re-ran doctor and read the identical output.
30
+ - `worktree clean` is scoped to dirt a run would actually read: a declared
31
+ project root, a declared artifact path, or `tdd.toml`. Build residue is
32
+ excluded via `config.is_ignored`, so doctor's own `uv run` / `vitest list`
33
+ probes (`.venv`, `node_modules`, caches) can no longer be what makes doctor
34
+ fail. Unrelated dirt is reported in the passing check's `detail` rather than
35
+ blocking the run.
36
+
7
37
  ## [0.3.0] - 2026-08-10
8
38
 
9
39
  ### Added
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -47,7 +47,7 @@ packages = ["src/tddcli"]
47
47
  include = ["src", "tests", "examples", "README.md", "LICENSE", "CHANGELOG.md", "SECURITY.md"]
48
48
 
49
49
  [dependency-groups]
50
- dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5"]
50
+ dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5", "zizmor>=1.29"]
51
51
 
52
52
  [tool.pytest.ini_options]
53
53
  testpaths = ["tests"]
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.3.0"
6
+ __version__ = "0.4.0"
@@ -4,11 +4,13 @@ from __future__ import annotations
4
4
 
5
5
  import os
6
6
  import subprocess
7
+ import time
7
8
  from collections import Counter
8
9
  from dataclasses import dataclass, field
9
10
  from pathlib import Path
10
11
 
11
12
  from .. import leases
13
+ from ..envelope import heartbeat
12
14
 
13
15
  NOT_FOUND = "not_found"
14
16
  NOT_COLLECTED = "not_collected"
@@ -78,10 +80,26 @@ def clip_failure(text: str, limit: int = 1500) -> str:
78
80
  return f"{text[:head]}\n… [clipped] …\n{text[-tail:]}"
79
81
 
80
82
 
83
+ #: Opt-in per-command timing. Off by default: the per-file collect loop would emit
84
+ #: one line per test file on every invocation, drowning the heartbeats that exist
85
+ #: to make a slow baseline legible.
86
+ TIMING_ENV = "TDD_TIMING"
87
+
88
+
81
89
  def run_command(
82
90
  command: str, cwd: Path, timeout: int = 1800,
83
91
  extra_env: dict[str, str] | None = None,
92
+ label: str | None = None,
84
93
  ) -> tuple[int, str, str]:
94
+ """Every subprocess the tool spawns passes through here, which makes it the one
95
+ place worth timing: suite runs, per-file collection, lint and typecheck gates,
96
+ doctor probes, artifact hooks.
97
+
98
+ `label` is what makes the rows groupable. This function sees a command string
99
+ and a cwd — not which project or phase asked for it — so an unlabelled timing
100
+ is readable by a human and useless to a query.
101
+ """
102
+ started = time.monotonic()
85
103
  proc = subprocess.run(
86
104
  command,
87
105
  shell=True,
@@ -91,6 +109,15 @@ def run_command(
91
109
  timeout=timeout,
92
110
  env=None if extra_env is None else {**os.environ, **extra_env},
93
111
  )
112
+ if os.environ.get(TIMING_ENV):
113
+ heartbeat(
114
+ event="command_timing",
115
+ label=label,
116
+ command=command,
117
+ cwd=str(cwd),
118
+ duration_ms=int((time.monotonic() - started) * 1000),
119
+ exit_code=proc.returncode,
120
+ )
94
121
  return proc.returncode, proc.stdout, proc.stderr
95
122
 
96
123
 
@@ -130,6 +157,7 @@ class Adapter:
130
157
  command.replace("{workers}", str(workers)),
131
158
  self.root,
132
159
  extra_env={"TDD_WORKERS": str(workers), **(extra_env or {})},
160
+ label="suite",
133
161
  )
134
162
 
135
163
  def _test_cmd(self) -> str:
@@ -184,7 +212,7 @@ class Adapter:
184
212
  def _gate(self, commands: list[str]) -> GateResult:
185
213
  chunks = []
186
214
  for cmd in commands:
187
- code, out, err = run_command(cmd, self.root)
215
+ code, out, err = run_command(cmd, self.root, label="gate")
188
216
  if code != 0:
189
217
  chunks.append(f"$ {cmd}\n{out}\n{err}".strip())
190
218
  return GateResult(ok=not chunks, output="\n\n".join(chunks)[:4000])
@@ -199,7 +199,7 @@ class PytestAdapter(Adapter):
199
199
  ]
200
200
  for cmd, env in probes:
201
201
  code, out, err = run_command(
202
- f"{cmd} --collect-only -q", self.root, extra_env=env
202
+ f"{cmd} --collect-only -q", self.root, extra_env=env, label="doctor"
203
203
  )
204
204
  if code != 0:
205
205
  chunks.append(out.strip())
@@ -215,7 +215,9 @@ class PytestAdapter(Adapter):
215
215
  if not self.project.overrides:
216
216
  return GateResult(ok=True)
217
217
  probe = f"{self._test_cmd().replace('{workers}', '0')} --collect-only -q"
218
- code, out, err = run_command(probe, self.root, extra_env=self._suite_env(None))
218
+ code, out, err = run_command(
219
+ probe, self.root, extra_env=self._suite_env(None), label="doctor"
220
+ )
219
221
  reached = sorted({
220
222
  f for f in (
221
223
  line.split("::", 1)[0]
@@ -243,6 +245,7 @@ class PytestAdapter(Adapter):
243
245
  f"{base} --collect-only -q {shlex.quote(str(rel))}",
244
246
  self.root,
245
247
  extra_env=env,
248
+ label="collect",
246
249
  )
247
250
  if code != 0:
248
251
  result.failed_files[str(rel)] = (err or out).strip()[:800]
@@ -177,7 +177,7 @@ class VitestAdapter(Adapter):
177
177
  """
178
178
  chunks = []
179
179
  code, out, err = run_command(
180
- self._collect_cmd(), self.root, extra_env=self._suite_env(None)
180
+ self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
181
181
  )
182
182
  if code != 0:
183
183
  chunks.append((err or out).strip())
@@ -190,7 +190,7 @@ class VitestAdapter(Adapter):
190
190
  )
191
191
  continue
192
192
  code, out, err = run_command(
193
- ov.collect_command, self.root, extra_env=self._suite_env(ov)
193
+ ov.collect_command, self.root, extra_env=self._suite_env(ov), label="doctor"
194
194
  )
195
195
  if code != 0:
196
196
  chunks.append((err or out).strip())
@@ -204,7 +204,7 @@ class VitestAdapter(Adapter):
204
204
  if not self.project.overrides:
205
205
  return GateResult(ok=True)
206
206
  code, out, err = run_command(
207
- self._collect_cmd(), self.root, extra_env=self._suite_env(None)
207
+ self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
208
208
  )
209
209
  reached = sorted({
210
210
  f for f in (
@@ -237,7 +237,8 @@ class VitestAdapter(Adapter):
237
237
  base = ov.collect_command if ov else self._collect_cmd()
238
238
  env = self._suite_env(ov)
239
239
  code, out, err = run_command(
240
- f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env
240
+ f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env,
241
+ label="collect",
241
242
  )
242
243
 
243
244
  payload = _extract_json(out)
@@ -13,6 +13,7 @@ import socket
13
13
  import sqlite3
14
14
  import sys
15
15
  import time
16
+ from collections.abc import Callable
16
17
  from datetime import datetime, timezone
17
18
  from pathlib import Path
18
19
 
@@ -217,16 +218,61 @@ def _legacy_artifacts(worktree: Path) -> list[Path]:
217
218
  return sorted(found)
218
219
 
219
220
 
220
- def cmd_doctor(args) -> Envelope:
221
- worktree = _worktree()
221
+ def _doctor_checklist() -> tuple[list[dict], Callable]:
222
+ """A checks list and its recorder, which refuses a blocker it cannot explain.
223
+
224
+ `resolve_blocker` with an empty `detail` is unfalsifiable: an agent is told to
225
+ fix something and given nothing to fix. It re-runs doctor, reads the identical
226
+ output, and loops. Enforcing the detail here means a check added later inherits
227
+ the guarantee instead of relying on its author to remember.
228
+ """
222
229
  checks: list[dict] = []
223
230
 
224
231
  def check(name, ok, detail="", project=None):
232
+ if not ok and not str(detail).strip():
233
+ raise AssertionError(
234
+ f"doctor check {name!r} would fail silently: a failing check must"
235
+ " name what to fix"
236
+ )
225
237
  entry = {"check": name, "ok": bool(ok), "detail": detail}
226
238
  if project is not None:
227
239
  entry["project"] = project
228
240
  checks.append(entry)
229
241
 
242
+ return checks, check
243
+
244
+
245
+ def _blocks_the_loop(rel_path: str, cfg) -> bool:
246
+ """Whether dirt at `rel_path` is somewhere a run would actually read it."""
247
+ if rel_path == config_mod.CONFIG_NAME:
248
+ return True
249
+ if cfg.owning_project(rel_path) is not None:
250
+ return True
251
+ return any(art.owns(rel_path) for art in cfg.artifacts.values())
252
+
253
+
254
+ def _cleanliness_detail(blocking: list[str], unrelated: list[str]) -> str:
255
+ def listed(paths: list[str]) -> str:
256
+ head = ", ".join(paths[:5])
257
+ return head if len(paths) <= 5 else f"{head} (+{len(paths) - 5} more)"
258
+
259
+ if blocking:
260
+ return (
261
+ f"uncommitted changes a run would observe: {listed(blocking)}."
262
+ " Commit, stash, or gitignore them before `tdd run start`."
263
+ )
264
+ if unrelated:
265
+ return (
266
+ f"clean where a run reads; {len(unrelated)} unrelated path(s) left as-is:"
267
+ f" {listed(unrelated)}"
268
+ )
269
+ return ""
270
+
271
+
272
+ def cmd_doctor(args) -> Envelope:
273
+ worktree = _worktree()
274
+ checks, check = _doctor_checklist()
275
+
230
276
  check("worktree resolvable", True, str(worktree))
231
277
  try:
232
278
  cfg = config_mod.load(worktree)
@@ -246,13 +292,20 @@ def cmd_doctor(args) -> Envelope:
246
292
  root = worktree / project.root
247
293
  check("root exists", root.is_dir(), str(root), project=name)
248
294
  check("adapter known", project.adapter in adapters.available(), project.adapter, project=name)
249
- check("test_paths declared", bool(project.test_paths), project=name)
295
+ declared = bool(project.test_paths)
296
+ check(
297
+ "test_paths declared", declared,
298
+ "" if declared
299
+ else f"add `test_paths` to [project.{name}] in tdd.toml — without it no"
300
+ " suite can be discovered for this project",
301
+ project=name,
302
+ )
250
303
  if project.adapter == "pytest":
251
304
  # The probe runs in the project's own environment (uv, poetry, pipenv,
252
305
  # pdm or the active venv) — hardcoding `uv run` here failed the check
253
306
  # on any non-uv project even with the plugin installed.
254
307
  probe = adapters.build(project, worktree).plugin_probe_cmd()
255
- code, out, err = adapters.base.run_command(probe, root)
308
+ code, out, err = adapters.base.run_command(probe, root, label="doctor")
256
309
  check("pytest-json-report installed", code == 0, (err or "")[:200], project=name)
257
310
 
258
311
  # Run before `collectable()` so this actionable message wins over
@@ -289,11 +342,33 @@ def cmd_doctor(args) -> Envelope:
289
342
  projects[name] = {"ok": all(c["ok"] for c in checks[before:])}
290
343
 
291
344
  for art in cfg.artifacts.values():
292
- check(f"artifact {art.name}: has check or regenerate", bool(art.check or art.regenerate))
345
+ # One evaluation feeds both `ok` and the detail: evaluating the condition
346
+ # twice lets them disagree, and a passing check that still says "add a hook"
347
+ # is the same misdirection as a failing check that says nothing.
348
+ has_hook = bool(art.check or art.regenerate)
349
+ check(
350
+ f"artifact {art.name}: has check or regenerate", has_hook,
351
+ "" if has_hook
352
+ else f"add `check` or `regenerate` to [artifact.{art.name}] in tdd.toml —"
353
+ " freshness cannot be verified without one",
354
+ )
293
355
 
294
356
  stale = _legacy_artifacts(worktree)
295
- check("no legacy state artifacts", not stale, ", ".join(str(s) for s in stale[:5]))
296
- check("worktree clean", not gitutil.is_dirty(worktree))
357
+ check(
358
+ "no legacy state artifacts", not stale,
359
+ f"delete these pre-ledger state files: {', '.join(str(s) for s in stale[:5])}"
360
+ if stale else "",
361
+ )
362
+
363
+ # Only dirt a run would read can corrupt one. Blocking on everything else
364
+ # stopped agents on unrelated notes and editor settings, and — because the
365
+ # check named no path — gave them nothing to act on but a re-run. `is_ignored`
366
+ # also excludes doctor's own probe residue (`.venv`, `node_modules`, caches),
367
+ # so running doctor can no longer be what makes doctor fail.
368
+ dirt = sorted(p for p in gitutil.dirty_paths(worktree) if not cfg.is_ignored(p))
369
+ blocking = [p for p in dirt if _blocks_the_loop(p, cfg)]
370
+ unrelated = [p for p in dirt if p not in set(blocking)]
371
+ check("worktree clean", not blocking, _cleanliness_detail(blocking, unrelated))
297
372
 
298
373
  ok = all(c["ok"] for c in checks)
299
374
  return Envelope(
@@ -364,12 +439,18 @@ def _probe_projects(cfg, worktree, ledger, on_progress):
364
439
  for done, (name, project) in enumerate(cfg.projects.items(), start=1):
365
440
  adapter = adapters.build(project, worktree)
366
441
  started = time.monotonic()
367
- verdict, collection = adapter.run(None), adapter.collect()
442
+ verdict = adapter.run(None)
443
+ ran = time.monotonic()
444
+ collection = adapter.collect()
368
445
  elapsed = time.monotonic() - started
369
446
  probes[name] = (verdict, collection)
447
+ # Split, not just totalled: `run` and `collect` have unrelated cost models
448
+ # — one scales with tests, the other with files — and a single number sends
449
+ # whoever asks "why was that slow?" out of the tool to measure by hand.
370
450
  heartbeat(
371
451
  event="baseline_captured", project=name,
372
452
  test_count=len(collection.tests), elapsed_s=round(elapsed, 2),
453
+ run_s=round(ran - started, 2), collect_s=round(elapsed - (ran - started), 2),
373
454
  )
374
455
  on_progress(done, name)
375
456
  return probes
@@ -0,0 +1,173 @@
1
+ """Every blocker `tdd doctor` emits must name what to fix.
2
+
3
+ The incident: doctor returned `resolve_blocker` / "Resolve the failing checks above"
4
+ with a single failing check — `{"check": "worktree clean", "ok": false, "detail": ""}`.
5
+ The dirt was an unrelated planning note and a `.claude/settings.json` edit. The agent
6
+ could not see either, guessed, re-ran doctor, and got the identical opaque failure.
7
+
8
+ Two invariants close that loop:
9
+
10
+ 1. A failing check always carries a detail. `check()` raises rather than record an
11
+ unfalsifiable blocker, so the class cannot reappear in a check added later.
12
+ 2. Only dirt the loop would actually observe blocks — a declared project root, a
13
+ declared artifact, or `tdd.toml`. Everything else is reported, not enforced
14
+ (the PRD's "tree clean enough", §8.1).
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from conftest import git, run_cli
20
+
21
+
22
+ def _check(out: dict, name: str) -> dict:
23
+ matches = [c for c in out["result"]["checks"] if c["check"] == name]
24
+ assert len(matches) == 1, out["result"]["checks"]
25
+ return matches[0]
26
+
27
+
28
+ def test_dirt_inside_a_project_root_blocks_and_names_the_path(repo):
29
+ (repo / "backend" / "app" / "scratch.py").write_text("x = 1\n")
30
+
31
+ clean = _check(run_cli(repo, "doctor"), "worktree clean")
32
+ assert clean["ok"] is False
33
+ assert "backend/app/scratch.py" in clean["detail"]
34
+
35
+
36
+ def test_a_modified_tdd_toml_blocks(repo):
37
+ (repo / "tdd.toml").write_text((repo / "tdd.toml").read_text() + "\n")
38
+
39
+ clean = _check(run_cli(repo, "doctor"), "worktree clean")
40
+ assert clean["ok"] is False
41
+ assert "tdd.toml" in clean["detail"]
42
+
43
+
44
+ def test_dirt_outside_every_declared_root_does_not_block(repo):
45
+ """The incident's own shape: a planning note and an editor setting. Neither is
46
+ readable by any adapter, so neither can corrupt a baseline."""
47
+ (repo / "tasks").mkdir()
48
+ (repo / "tasks" / "fix-406-proxy-ws-forwarding.md").write_text("# plan\n")
49
+ (repo / ".claude").mkdir()
50
+ (repo / ".claude" / "settings.json").write_text("{}\n")
51
+
52
+ clean = _check(run_cli(repo, "doctor"), "worktree clean")
53
+ assert clean["ok"] is True, clean
54
+ # Reported, not enforced — a human still wants to know the tree is not pristine.
55
+ assert "tasks/fix-406-proxy-ws-forwarding.md" in clean["detail"]
56
+
57
+
58
+ def test_doctors_own_probe_residue_does_not_block(repo):
59
+ """Doctor shells out to `uv run` and `vitest list`, which write `.venv`,
60
+ `node_modules` and cache directories into the tree it then grades. Without this,
61
+ running doctor is what makes doctor fail — and running it again re-does it."""
62
+ (repo / "backend" / ".venv" / "bin").mkdir(parents=True)
63
+ (repo / "backend" / ".venv" / "bin" / "python").write_text("")
64
+ (repo / "backend" / "__pycache__").mkdir()
65
+ (repo / "backend" / "__pycache__" / "app.pyc").write_text("")
66
+
67
+ clean = _check(run_cli(repo, "doctor"), "worktree clean")
68
+ assert clean["ok"] is True, clean
69
+
70
+
71
+ def test_blocking_dirt_is_reported_even_when_unrelated_dirt_exists(repo):
72
+ (repo / "backend" / "app" / "scratch.py").write_text("x = 1\n")
73
+ (repo / "notes.md").write_text("# notes\n")
74
+
75
+ clean = _check(run_cli(repo, "doctor"), "worktree clean")
76
+ assert clean["ok"] is False
77
+ assert "backend/app/scratch.py" in clean["detail"]
78
+
79
+
80
+ def test_no_failing_check_is_ever_emitted_without_a_detail(repo_broken):
81
+ """The structural half. `repo_broken` fails collection; dirty the tree and break
82
+ the artifact wiring too, so several failure paths run in one invocation."""
83
+ (repo_broken / "verify" / "tests" / "test_extra.py").write_text("\n")
84
+ (repo_broken / "tdd.toml").write_text(
85
+ (repo_broken / "tdd.toml").read_text()
86
+ + '\n[artifact.openapi]\npath = "openapi.json"\nproduced_by = "backend"\n'
87
+ )
88
+
89
+ out = run_cli(repo_broken, "doctor")
90
+ failing = [c for c in out["result"]["checks"] if not c["ok"]]
91
+ assert failing, out["result"]["checks"]
92
+ for c in failing:
93
+ assert c["detail"].strip(), c
94
+
95
+
96
+ def test_a_check_cannot_fail_without_a_detail():
97
+ """Enforced at the helper, so a check added later inherits the guarantee rather
98
+ than relying on its author to remember."""
99
+ import pytest
100
+
101
+ from tddcli.cli import _doctor_checklist
102
+
103
+ checks, check = _doctor_checklist()
104
+ check("fine", False, "here is what to do")
105
+ with pytest.raises(AssertionError, match="silent"):
106
+ check("silent", False)
107
+ assert [c["check"] for c in checks] == ["fine"]
108
+
109
+
110
+ def _with_artifact(repo, hook: str = 'regenerate = "true"') -> None:
111
+ (repo / "schema").mkdir(exist_ok=True)
112
+ (repo / "schema" / "openapi.json").write_text("{}\n")
113
+ (repo / "tdd.toml").write_text(
114
+ (repo / "tdd.toml").read_text()
115
+ + f'\n[artifact.openapi]\npath = "schema/openapi.json"\n'
116
+ f'produced_by = "backend"\n{hook}\n'
117
+ )
118
+ git(repo, "add", "-A")
119
+ git(repo, "commit", "-q", "-m", "declare openapi artifact")
120
+
121
+
122
+ def test_dirt_at_a_declared_artifact_path_blocks(repo):
123
+ """An artifact sits outside every project root but is still read by a run —
124
+ `run start` verifies its freshness. Uncommitted drift there is exactly the
125
+ staleness the artifact edge exists to catch."""
126
+ _with_artifact(repo)
127
+ (repo / "schema" / "openapi.json").write_text('{"drift": true}\n')
128
+
129
+ clean = _check(run_cli(repo, "doctor"), "worktree clean")
130
+ assert clean["ok"] is False
131
+ assert "schema/openapi.json" in clean["detail"]
132
+
133
+
134
+ def test_an_artifact_needs_only_one_of_check_or_regenerate(repo):
135
+ """Either hook alone makes freshness verifiable — requiring both would fail
136
+ every artifact that only knows how to rebuild itself."""
137
+ _with_artifact(repo, hook='regenerate = "true"')
138
+
139
+ assert _check(run_cli(repo, "doctor"), "artifact openapi: has check or regenerate")["ok"] is True
140
+
141
+ _check_only = repo / "tdd.toml"
142
+ _check_only.write_text(_check_only.read_text().replace('regenerate = "true"', 'check = "true"'))
143
+ git(repo, "add", "-A")
144
+ git(repo, "commit", "-q", "-m", "swap regenerate for check")
145
+
146
+ assert _check(run_cli(repo, "doctor"), "artifact openapi: has check or regenerate")["ok"] is True
147
+
148
+
149
+ def test_a_long_dirt_list_is_truncated_but_the_fifth_path_is_not(repo):
150
+ """The detail is read by an agent, so it stays bounded — but the boundary must
151
+ not eat a path it had room for."""
152
+ for i in range(5):
153
+ (repo / "backend" / "app" / f"f{i}.py").write_text("x = 1\n")
154
+
155
+ detail = _check(run_cli(repo, "doctor"), "worktree clean")["detail"]
156
+ assert "more)" not in detail, detail
157
+ assert "backend/app/f4.py" in detail
158
+
159
+ (repo / "backend" / "app" / "f5.py").write_text("x = 1\n")
160
+ detail = _check(run_cli(repo, "doctor"), "worktree clean")["detail"]
161
+ assert "(+1 more)" in detail, detail
162
+
163
+
164
+ def test_test_paths_omission_says_what_to_add(repo):
165
+ (repo / "tdd.toml").write_text(
166
+ '[project.backend]\nroot = "backend"\nadapter = "pytest"\ntest_paths = []\n'
167
+ )
168
+ git(repo, "add", "-A")
169
+ git(repo, "commit", "-q", "-m", "drop test_paths")
170
+
171
+ declared = _check(run_cli(repo, "doctor"), "test_paths declared")
172
+ assert declared["ok"] is False
173
+ assert "test_paths" in declared["detail"]
@@ -46,7 +46,7 @@ def test_pytest_target_failure_keeps_the_error_at_the_tail(tmp_path, monkeypatch
46
46
  adapter = _pytest_adapter(tmp_path)
47
47
  longrepr = ("connector frame\n" * 300) + "ConnectionRefusedError: [Errno 61]"
48
48
 
49
- def fake(command, cwd, timeout=1800, extra_env=None):
49
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
50
50
  marker = "--json-report-file="
51
51
  path = command.split(marker, 1)[1].split(" --", 1)[0]
52
52
  Path(path.strip("'\"")).write_text(json.dumps({
@@ -62,7 +62,7 @@ def test_only_reporting_flags_are_appended(tmp_path, monkeypatch):
62
62
  adapter = adapters.build(project, tmp_path)
63
63
  seen = {}
64
64
 
65
- def fake_run(command, cwd, timeout=1800, extra_env=None):
65
+ def fake_run(command, cwd, timeout=1800, extra_env=None, label=None):
66
66
  seen["command"] = command
67
67
  return 1, "", "no report"
68
68
 
@@ -54,7 +54,7 @@ def test_default_suite_runs_with_the_project_env_expanded(tmp_path, monkeypatch)
54
54
  adapter = adapters.build(project, tmp_path)
55
55
  seen: list = []
56
56
 
57
- def fake(command, cwd, timeout=1800, extra_env=None):
57
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
58
58
  seen.append((command, extra_env))
59
59
  marker = "--json-report-file="
60
60
  path = command.split(marker, 1)[1].split(" --", 1)[0]
@@ -95,7 +95,7 @@ def test_pytest_collection_of_default_files_carries_the_project_env(
95
95
  adapter = adapters.build(project, tmp_path)
96
96
  seen: list = []
97
97
 
98
- def fake(command, cwd, timeout=1800, extra_env=None):
98
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
99
99
  seen.append((command, extra_env))
100
100
  return 0, "tests/test_a.py::test_a\n", ""
101
101
 
@@ -114,7 +114,7 @@ def test_vitest_default_suite_runs_with_the_project_env(tmp_path, monkeypatch):
114
114
  adapter = adapters.build(project, tmp_path)
115
115
  seen: list = []
116
116
 
117
- def fake(command, cwd, timeout=1800, extra_env=None):
117
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
118
118
  seen.append((command, extra_env))
119
119
  return 0, json.dumps({"testResults": []}), ""
120
120
 
@@ -110,7 +110,7 @@ def _fake_pytest_run(reports_by_prefix: dict[str, dict], seen: list):
110
110
  """A run_command double that answers each suite command with its own report,
111
111
  keyed by command prefix, writing the JSON where the real plugin would."""
112
112
 
113
- def fake(command, cwd, timeout=1800, extra_env=None):
113
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
114
114
  seen.append((command, extra_env))
115
115
  for prefix, report in reports_by_prefix.items():
116
116
  if command.startswith(prefix):
@@ -195,7 +195,7 @@ def test_pytest_broken_override_suite_is_a_loud_error_not_a_silent_gap(
195
195
  project = project_with(tmp_path, 'test_command = "pytest tests"\n' + OVERRIDE_BLOCK)
196
196
  adapter = adapters.build(project, tmp_path)
197
197
 
198
- def fake(command, cwd, timeout=1800, extra_env=None):
198
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
199
199
  if command.startswith("pytest contract"):
200
200
  return 4, "", "ERROR: file or directory not found: contract"
201
201
  marker = "--json-report-file="
@@ -222,7 +222,7 @@ def test_pytest_collection_routes_override_files_to_the_override_command(
222
222
  adapter = adapters.build(project, tmp_path)
223
223
  seen: list = []
224
224
 
225
- def fake(command, cwd, timeout=1800, extra_env=None):
225
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
226
226
  seen.append((command, extra_env))
227
227
  name = "test_api.py::test_ping" if "contract" in command else "test_a.py::test_a"
228
228
  return 0, name, ""
@@ -272,7 +272,7 @@ def test_vitest_run_finds_a_target_that_only_the_override_config_reaches(
272
272
  project = project_with(tmp_path, VITEST_OVERRIDE, adapter="vitest")
273
273
  adapter = adapters.build(project, tmp_path)
274
274
 
275
- def fake(command, cwd, timeout=1800, extra_env=None):
275
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
276
276
  if "--config vitest.contract.config.ts" in command:
277
277
  report = _vitest_report(
278
278
  str(tmp_path / "backend" / "contract" / "api.contract.test.ts"),
@@ -331,7 +331,7 @@ def test_vitest_collection_routes_override_files_to_the_override_command(
331
331
  adapter = adapters.build(project, tmp_path)
332
332
  seen: list = []
333
333
 
334
- def fake(command, cwd, timeout=1800, extra_env=None):
334
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
335
335
  seen.append(command)
336
336
  return 0, "contract/api.contract.test.ts > pings the api", ""
337
337
 
@@ -479,7 +479,7 @@ def test_vitest_duplicate_test_id_across_suites_is_a_loud_error(
479
479
  ]
480
480
  }
481
481
 
482
- def fake(command, cwd, timeout=1800, extra_env=None):
482
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
483
483
  if "--config" in command:
484
484
  return 0, json.dumps(result("passed")), ""
485
485
  return 1, json.dumps(result("failed")), ""
@@ -500,7 +500,7 @@ def test_pytest_isolation_probe_flags_default_reach_into_override_files(
500
500
  adapter = adapters.build(project, tmp_path)
501
501
  seen: list = []
502
502
 
503
- def fake(command, cwd, timeout=1800, extra_env=None):
503
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
504
504
  seen.append(command)
505
505
  return 0, (
506
506
  "tests/test_a.py::test_a\n"
@@ -528,7 +528,7 @@ def test_pytest_isolation_probe_passes_when_the_default_suite_is_scoped(
528
528
  monkeypatch.setattr(
529
529
  adapters.pytest_adapter,
530
530
  "run_command",
531
- lambda command, cwd, timeout=1800, extra_env=None: (
531
+ lambda command, cwd, timeout=1800, extra_env=None, label=None: (
532
532
  0, "tests/test_a.py::test_a\n", ""
533
533
  ),
534
534
  )
@@ -550,7 +550,7 @@ def test_vitest_isolation_probe_flags_default_reach_into_override_files(
550
550
  adapter = adapters.build(project, tmp_path)
551
551
  seen: list = []
552
552
 
553
- def fake(command, cwd, timeout=1800, extra_env=None):
553
+ def fake(command, cwd, timeout=1800, extra_env=None, label=None):
554
554
  seen.append(command)
555
555
  return 0, (
556
556
  "src/__tests__/a.test.ts > adds\n"
@@ -570,7 +570,7 @@ def test_isolation_probe_is_free_when_a_project_declares_no_overrides(
570
570
  project = project_with(tmp_path, 'test_command = "pytest tests"\n')
571
571
  adapter = adapters.build(project, tmp_path)
572
572
 
573
- def explode(command, cwd, timeout=1800, extra_env=None):
573
+ def explode(command, cwd, timeout=1800, extra_env=None, label=None):
574
574
  raise AssertionError("no probe should run without overrides")
575
575
 
576
576
  monkeypatch.setattr(adapters.pytest_adapter, "run_command", explode)
@@ -0,0 +1,149 @@
1
+ """Where a slow command actually spent its time.
2
+
3
+ A `run start` took 23 minutes and reported one number per project — `elapsed_s`,
4
+ covering `run()` and `collect()` together. Answering "which of the two?" required
5
+ measuring the projects by hand, outside the tool, in a different checkout; the
6
+ warm numbers that produced did not match the cold ones and the first diagnosis
7
+ drawn from them was wrong.
8
+
9
+ The tool already holds both timings at the moment it discards them. Two seams
10
+ make them visible:
11
+
12
+ * `baseline_captured` reports `run_s` and `collect_s` separately, so the split is
13
+ in the output the agent already prints.
14
+ * `run_command` — the single choke point every subprocess passes through — emits a
15
+ `command_timing` line under `TDD_TIMING=1`, which attributes cost per command:
16
+ per-file collection, gates, doctor probes, artifact hooks. Off by default,
17
+ because the per-file loop would otherwise emit one line per test file on every
18
+ invocation.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import json
24
+
25
+ from conftest import run_cli, write_plan
26
+ from tddcli.adapters import base
27
+
28
+ PLAN = """---
29
+ cycles:
30
+ - n: 1
31
+ project: backend
32
+ title: "adding two numbers"
33
+ test: "tests/test_add.py::test_add_two_numbers"
34
+ stub_expected: ["app/calc.py"]
35
+ commit_red: "test: adding two numbers"
36
+ commit_green: "feat: add()"
37
+ ---
38
+
39
+ # Plan
40
+ """
41
+
42
+
43
+ def _lines(stderr: str, event: str) -> list[dict]:
44
+ found = []
45
+ for line in stderr.splitlines():
46
+ try:
47
+ payload = json.loads(line.strip())
48
+ except json.JSONDecodeError:
49
+ continue
50
+ if payload.get("event") == event:
51
+ found.append(payload)
52
+ return found
53
+
54
+
55
+ def _start(repo):
56
+ plan = write_plan(repo, PLAN)
57
+ assert run_cli(repo, "plan", "register", plan)["ok"]
58
+ return run_cli(repo, "run", "start", "--plan", plan)
59
+
60
+
61
+ def test_baseline_reports_run_and_collect_separately(repo, capsys):
62
+ out = _start(repo)
63
+ assert out["ok"], out
64
+
65
+ backend = next(
66
+ line for line in _lines(capsys.readouterr().err, "baseline_captured")
67
+ if line["project"] == "backend"
68
+ )
69
+ assert isinstance(backend["run_s"], (int, float)), backend
70
+ assert isinstance(backend["collect_s"], (int, float)), backend
71
+
72
+
73
+ def test_the_split_still_adds_up_to_the_reported_total(repo, capsys):
74
+ """`elapsed_s` stays, so existing consumers keep working — and it must remain
75
+ the sum of the parts, or the split is describing a different thing."""
76
+ out = _start(repo)
77
+ assert out["ok"], out
78
+
79
+ backend = next(
80
+ line for line in _lines(capsys.readouterr().err, "baseline_captured")
81
+ if line["project"] == "backend"
82
+ )
83
+ assert backend["run_s"] + backend["collect_s"] <= backend["elapsed_s"] + 0.05
84
+ assert backend["elapsed_s"] - (backend["run_s"] + backend["collect_s"]) < 0.5
85
+
86
+
87
+ def test_command_timing_is_silent_unless_asked_for(repo, capsys, monkeypatch):
88
+ monkeypatch.delenv(base.TIMING_ENV, raising=False)
89
+ assert _start(repo)["ok"]
90
+ assert _lines(capsys.readouterr().err, "command_timing") == []
91
+
92
+
93
+ def test_command_timing_names_the_command_and_its_cost(repo, capsys, monkeypatch):
94
+ monkeypatch.setenv(base.TIMING_ENV, "1")
95
+ assert _start(repo)["ok"]
96
+
97
+ timings = _lines(capsys.readouterr().err, "command_timing")
98
+ assert timings, "TDD_TIMING=1 produced no command_timing lines"
99
+ for entry in timings:
100
+ assert isinstance(entry["duration_ms"], int)
101
+ assert entry["command"]
102
+ assert isinstance(entry["exit_code"], int)
103
+
104
+
105
+ def test_a_timed_command_is_attributed_to_its_caller(repo, capsys, monkeypatch):
106
+ """Without a label the rows are readable but not groupable: `run_command` sees
107
+ a command string and a cwd, not which project or phase asked for it."""
108
+ monkeypatch.setenv(base.TIMING_ENV, "1")
109
+ assert _start(repo)["ok"]
110
+
111
+ timings = _lines(capsys.readouterr().err, "command_timing")
112
+ labels = {entry.get("label") for entry in timings}
113
+ assert "suite" in labels, labels
114
+ assert "collect" in labels, labels
115
+
116
+
117
+ def test_the_per_file_collect_loop_is_attributed_per_file(repo, capsys, monkeypatch):
118
+ """The loop's cost is per file, so its timing has to be too — a single total
119
+ cannot say which file is slow."""
120
+ monkeypatch.setenv(base.TIMING_ENV, "1")
121
+ assert _start(repo)["ok"]
122
+
123
+ collects = [
124
+ entry for entry in _lines(capsys.readouterr().err, "command_timing")
125
+ if entry.get("label") == "collect"
126
+ ]
127
+ assert collects, "no collect timings"
128
+ assert any("test_smoke.py" in entry["command"] for entry in collects), collects
129
+
130
+
131
+ def test_doctor_probes_are_labelled_as_doctor(repo, capsys, monkeypatch):
132
+ """Doctor's probes reach `run_command` from three places — the reporter check
133
+ in `cmd_doctor`, `collectable()` and `override_isolation()` — and both adapter
134
+ methods are called from nowhere else. Unlabelled they arrive as `label: null`,
135
+ indistinguishable from a third-party adapter's unlabelled call."""
136
+ monkeypatch.setenv(base.TIMING_ENV, "1")
137
+ assert run_cli(repo, "doctor")["ok"]
138
+
139
+ timings = _lines(capsys.readouterr().err, "command_timing")
140
+ assert timings, "TDD_TIMING=1 produced no command_timing lines for doctor"
141
+ assert {entry.get("label") for entry in timings} == {"doctor"}, timings
142
+
143
+
144
+ def test_timing_does_not_disturb_the_command_result(repo, monkeypatch):
145
+ """The wrapper returns exactly what the subprocess returned."""
146
+ monkeypatch.setenv(base.TIMING_ENV, "1")
147
+ code, out, err = base.run_command("echo hello", repo)
148
+ assert code == 0
149
+ assert out.strip() == "hello"
@@ -127,7 +127,7 @@ def test_live_foreign_lease_counts(lease_dir):
127
127
  def recorded(monkeypatch):
128
128
  calls: list[dict] = []
129
129
 
130
- def stub(command, cwd, timeout=1800, extra_env=None):
130
+ def stub(command, cwd, timeout=1800, extra_env=None, label=None):
131
131
  calls.append({"command": command, "cwd": cwd, "extra_env": extra_env})
132
132
  return 1, "", ""
133
133
 
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes