tdd-cli 0.3.0__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/CHANGELOG.md +45 -0
  2. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/PKG-INFO +2 -2
  3. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/pyproject.toml +1 -1
  4. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/__init__.py +1 -1
  5. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/adapters/base.py +71 -1
  6. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/adapters/pytest_adapter.py +42 -13
  7. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/adapters/vitest_adapter.py +42 -8
  8. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/cli.py +89 -8
  9. tdd_cli-0.4.1/tests/test_batch_collection.py +233 -0
  10. tdd_cli-0.4.1/tests/test_doctor_blockers.py +173 -0
  11. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_failure_clipping.py +1 -1
  12. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_project_commands.py +1 -1
  13. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_project_env.py +3 -3
  14. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_suite_overrides.py +17 -11
  15. tdd_cli-0.4.1/tests/test_timing_visibility.py +158 -0
  16. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_worker_leases.py +1 -1
  17. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/.gitignore +0 -0
  18. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/LICENSE +0 -0
  19. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/README.md +0 -0
  20. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/SECURITY.md +0 -0
  21. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/README.md +0 -0
  22. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/bash_hook.py +0 -0
  23. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/stop_hook.py +0 -0
  24. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/plan.md +0 -0
  25. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/skills/tdd-drive/README.md +0 -0
  26. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/skills/tdd-drive/SKILL.md +0 -0
  27. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/skills/tdd-handoff/README.md +0 -0
  28. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/skills/tdd-handoff/SKILL.md +0 -0
  29. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/adapters/__init__.py +0 -0
  30. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/advance.py +0 -0
  31. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/config.py +0 -0
  32. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/contract.py +0 -0
  33. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/envelope.py +0 -0
  34. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/fleet.py +0 -0
  35. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/gitutil.py +0 -0
  36. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/identity.py +0 -0
  37. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/leases.py +0 -0
  38. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/ledger.py +0 -0
  39. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/machine.py +0 -0
  40. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/render.py +0 -0
  41. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/snapshot.py +0 -0
  42. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/staging.py +0 -0
  43. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/conftest.py +0 -0
  44. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_artifact_regeneration.py +0 -0
  45. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_baseline_integrity.py +0 -0
  46. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_config_and_staging.py +0 -0
  47. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_config_drift.py +0 -0
  48. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_contract.py +0 -0
  49. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_doctor_attribution.py +0 -0
  50. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_end_to_end.py +0 -0
  51. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_example_plan.py +0 -0
  52. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_fleet.py +0 -0
  53. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_heartbeat.py +0 -0
  54. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_init_detection.py +0 -0
  55. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_pin_cycles.py +0 -0
  56. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_progress.py +0 -0
  57. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_python_env_managers.py +0 -0
  58. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_refactor_cycles.py +0 -0
  59. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_release_surface.py +0 -0
  60. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_run_claim.py +0 -0
  61. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_single_project_repo.py +0 -0
  62. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_snapshot_and_identity.py +0 -0
  63. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_stub_hint.py +0 -0
  64. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_target_validation.py +0 -0
  65. {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_vitest_adapter.py +0 -0
@@ -4,6 +4,51 @@ All notable changes to this project are documented here.
4
4
  The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
5
5
  and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
+ ## [0.4.1] - 2026-08-16
8
+
9
+ ### Changed
10
+
11
+ - `collect()` runs one invocation per declared suite instead of one per test
12
+ file, falling back to the per-file loop for anything a batch did not account
13
+ for. Per-file collection was **77% of a real `run start`** — 313 subprocesses
14
+ costing 402s, against 117s to actually run every test — because cost scaled
15
+ with file count at a ~1.08s floor per invocation (the environment manager
16
+ resolving plus the runner booting). Measured 38.8x faster on a 60-file
17
+ project, with an identical collected set. R10.3's guarantee is unchanged: a
18
+ file that fails to collect is still attributed to itself and cannot destroy
19
+ the set, and a file the batch never reports is still collected individually,
20
+ so the set can only match or improve on the old one (#27).
21
+
22
+ ## [0.4.0] - 2026-08-16
23
+
24
+ ### Added
25
+
26
+ - `baseline_captured` reports `run_s` and `collect_s` alongside `elapsed_s`. The
27
+ suite run and the per-file collection have unrelated cost models — one scales
28
+ with tests, the other with files — so a single total could not say which was
29
+ slow, and answering that meant measuring projects by hand outside the tool.
30
+ - `TDD_TIMING=1` emits a `command_timing` line per subprocess on stderr
31
+ (`label`, `command`, `cwd`, `duration_ms`, `exit_code`), covering every
32
+ subprocess the tool spawns: suite runs, per-file collection, lint/typecheck
33
+ gates, doctor probes and artifact hooks. Off by default — the per-file loop
34
+ would otherwise emit one line per test file on every invocation. `label` is
35
+ one of `suite`, `collect`, `gate`, `doctor`; an unlabelled row comes from a
36
+ third-party adapter, since every built-in call site names itself (R8.4).
37
+
38
+ ### Fixed
39
+
40
+ - `tdd doctor` no longer emits a blocker it cannot explain. Every failing check
41
+ now carries a `detail` naming what to fix, enforced by the checklist recorder
42
+ so a check added later inherits the guarantee. Previously `worktree clean`
43
+ failed with `detail: ""`, leaving an agent with `resolve_blocker` and nothing
44
+ to resolve — it re-ran doctor and read the identical output.
45
+ - `worktree clean` is scoped to dirt a run would actually read: a declared
46
+ project root, a declared artifact path, or `tdd.toml`. Build residue is
47
+ excluded via `config.is_ignored`, so doctor's own `uv run` / `vitest list`
48
+ probes (`.venv`, `node_modules`, caches) can no longer be what makes doctor
49
+ fail. Unrelated dirt is reported in the passing check's `detail` rather than
50
+ blocking the run.
51
+
7
52
  ## [0.3.0] - 2026-08-10
8
53
 
9
54
  ### Added
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.3.0
3
+ Version: 0.4.1
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -47,7 +47,7 @@ packages = ["src/tddcli"]
47
47
  include = ["src", "tests", "examples", "README.md", "LICENSE", "CHANGELOG.md", "SECURITY.md"]
48
48
 
49
49
  [dependency-groups]
50
- dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5"]
50
+ dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5", "zizmor>=1.29"]
51
51
 
52
52
  [tool.pytest.ini_options]
53
53
  testpaths = ["tests"]
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.3.0"
6
+ __version__ = "0.4.1"
@@ -4,11 +4,13 @@ from __future__ import annotations
4
4
 
5
5
  import os
6
6
  import subprocess
7
+ import time
7
8
  from collections import Counter
8
9
  from dataclasses import dataclass, field
9
10
  from pathlib import Path
10
11
 
11
12
  from .. import leases
13
+ from ..envelope import heartbeat
12
14
 
13
15
  NOT_FOUND = "not_found"
14
16
  NOT_COLLECTED = "not_collected"
@@ -78,10 +80,26 @@ def clip_failure(text: str, limit: int = 1500) -> str:
78
80
  return f"{text[:head]}\n… [clipped] …\n{text[-tail:]}"
79
81
 
80
82
 
83
+ #: Opt-in per-command timing. Off by default: the per-file collect loop would emit
84
+ #: one line per test file on every invocation, drowning the heartbeats that exist
85
+ #: to make a slow baseline legible.
86
+ TIMING_ENV = "TDD_TIMING"
87
+
88
+
81
89
  def run_command(
82
90
  command: str, cwd: Path, timeout: int = 1800,
83
91
  extra_env: dict[str, str] | None = None,
92
+ label: str | None = None,
84
93
  ) -> tuple[int, str, str]:
94
+ """Every subprocess the tool spawns passes through here, which makes it the one
95
+ place worth timing: suite runs, per-file collection, lint and typecheck gates,
96
+ doctor probes, artifact hooks.
97
+
98
+ `label` is what makes the rows groupable. This function sees a command string
99
+ and a cwd — not which project or phase asked for it — so an unlabelled timing
100
+ is readable by a human and useless to a query.
101
+ """
102
+ started = time.monotonic()
85
103
  proc = subprocess.run(
86
104
  command,
87
105
  shell=True,
@@ -91,6 +109,15 @@ def run_command(
91
109
  timeout=timeout,
92
110
  env=None if extra_env is None else {**os.environ, **extra_env},
93
111
  )
112
+ if os.environ.get(TIMING_ENV):
113
+ heartbeat(
114
+ event="command_timing",
115
+ label=label,
116
+ command=command,
117
+ cwd=str(cwd),
118
+ duration_ms=int((time.monotonic() - started) * 1000),
119
+ exit_code=proc.returncode,
120
+ )
94
121
  return proc.returncode, proc.stdout, proc.stderr
95
122
 
96
123
 
@@ -130,6 +157,7 @@ class Adapter:
130
157
  command.replace("{workers}", str(workers)),
131
158
  self.root,
132
159
  extra_env={"TDD_WORKERS": str(workers), **(extra_env or {})},
160
+ label="suite",
133
161
  )
134
162
 
135
163
  def _test_cmd(self) -> str:
@@ -161,6 +189,48 @@ class Adapter:
161
189
  return "a body that fails loudly, never working logic"
162
190
 
163
191
  def collect(self) -> Collection:
192
+ """Enumerate the project's tests: one invocation per declared suite, with
193
+ the per-file loop kept for what that cannot account for (issue #27).
194
+
195
+ Per file, collection cost scaled with *file count* rather than test count —
196
+ measured at 77% of a whole `run start`, against a floor of ~1.08s per
197
+ invocation that is the environment manager resolving plus the runner
198
+ booting, not collection work.
199
+
200
+ The batch and the loop do not discover the same way: a batch uses the
201
+ runner's own config, the loop walks `test_paths`. So the loop still runs for
202
+ every file the batch did not report — whether the batch failed, returned
203
+ nothing, or simply never mentioned that file. R10.3's guarantee is
204
+ unchanged: one uncollectable file is attributed to itself, and cannot
205
+ destroy the set. What changes is that the healthy case no longer pays for
206
+ the broken one.
207
+ """
208
+ result = Collection()
209
+ unaccounted = {str(p.relative_to(self.root)) for p in self._test_files()}
210
+ for command, env in self._collect_invocations():
211
+ batch = self._collect_batch(command, env)
212
+ if batch is None:
213
+ continue
214
+ tests, files = batch
215
+ result.tests |= tests
216
+ unaccounted -= files
217
+ return self._collect_per_file(unaccounted, result)
218
+
219
+ def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
220
+ """One collection command per declared suite: the default plus each
221
+ override's (R7.13), so an override's files are enumerated with its own
222
+ command and env."""
223
+ raise NotImplementedError
224
+
225
+ def _collect_batch(
226
+ self, command: str, env: dict[str, str] | None
227
+ ) -> tuple[set[str], set[str]] | None:
228
+ """`(test ids, files seen)` for one whole-suite invocation, or None when it
229
+ produced nothing usable — the signal to leave its files to the loop."""
230
+ raise NotImplementedError
231
+
232
+ def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
233
+ """R10.3's loop, now reached only for files no batch accounted for."""
164
234
  raise NotImplementedError
165
235
 
166
236
  def collectable(self) -> GateResult:
@@ -184,7 +254,7 @@ class Adapter:
184
254
  def _gate(self, commands: list[str]) -> GateResult:
185
255
  chunks = []
186
256
  for cmd in commands:
187
- code, out, err = run_command(cmd, self.root)
257
+ code, out, err = run_command(cmd, self.root, label="gate")
188
258
  if code != 0:
189
259
  chunks.append(f"$ {cmd}\n{out}\n{err}".strip())
190
260
  return GateResult(ok=not chunks, output="\n\n".join(chunks)[:4000])
@@ -199,7 +199,7 @@ class PytestAdapter(Adapter):
199
199
  ]
200
200
  for cmd, env in probes:
201
201
  code, out, err = run_command(
202
- f"{cmd} --collect-only -q", self.root, extra_env=env
202
+ f"{cmd} --collect-only -q", self.root, extra_env=env, label="doctor"
203
203
  )
204
204
  if code != 0:
205
205
  chunks.append(out.strip())
@@ -215,7 +215,9 @@ class PytestAdapter(Adapter):
215
215
  if not self.project.overrides:
216
216
  return GateResult(ok=True)
217
217
  probe = f"{self._test_cmd().replace('{workers}', '0')} --collect-only -q"
218
- code, out, err = run_command(probe, self.root, extra_env=self._suite_env(None))
218
+ code, out, err = run_command(
219
+ probe, self.root, extra_env=self._suite_env(None), label="doctor"
220
+ )
219
221
  reached = sorted({
220
222
  f for f in (
221
223
  line.split("::", 1)[0]
@@ -233,22 +235,49 @@ class PytestAdapter(Adapter):
233
235
  " cannot reach them (e.g. `pytest tests/`)."
234
236
  ))
235
237
 
236
- def collect(self) -> Collection:
238
+ def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
239
+ """An override without a `collect_command` collects with its `test_command`
240
+ — `--collect-only` composes with any run command."""
241
+ return [(self._collect_cmd(), self._suite_env(None))] + [
242
+ (ov.collect_command or ov.test_command, self._suite_env(ov))
243
+ for ov in self.project.overrides
244
+ ]
245
+
246
+ def _nodeids(self, out: str) -> tuple[set[str], set[str]]:
247
+ tests: set[str] = set()
248
+ files: set[str] = set()
249
+ for line in out.splitlines():
250
+ line = line.strip()
251
+ if "::" in line and not line.startswith(("=", "-", "no tests")):
252
+ tests.add(self.qualify(line))
253
+ files.add(line.split("::", 1)[0])
254
+ return tests, files
255
+
256
+ def _collect_batch(
257
+ self, command: str, env: dict[str, str] | None
258
+ ) -> tuple[set[str], set[str]] | None:
259
+ code, out, err = run_command(
260
+ f"{command} --collect-only -q", self.root, extra_env=env, label="collect"
261
+ )
262
+ if code != 0:
263
+ return None
264
+ # An empty result needs no special case: it accounts for no files, so every
265
+ # file falls to the loop exactly as a failure would.
266
+ return self._nodeids(out)
267
+
268
+ def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
237
269
  """Per file (R10.3) — one uncollectable module must not destroy the whole set."""
238
- result = Collection()
239
- for path in self._test_files():
240
- rel = path.relative_to(self.root)
241
- base, env = self._collect_cmd_for(str(rel))
270
+ for rel in sorted(rels):
271
+ base, env = self._collect_cmd_for(rel)
242
272
  code, out, err = run_command(
243
- f"{base} --collect-only -q {shlex.quote(str(rel))}",
273
+ f"{base} --collect-only -q {shlex.quote(rel)}",
244
274
  self.root,
245
275
  extra_env=env,
276
+ label="collect",
246
277
  )
247
278
  if code != 0:
248
- result.failed_files[str(rel)] = (err or out).strip()[:800]
279
+ result.failed_files[rel] = (err or out).strip()[:800]
249
280
  continue
250
- for line in out.splitlines():
251
- line = line.strip()
252
- if "::" in line and not line.startswith(("=", "-", "no tests")):
253
- result.tests.add(self.qualify(line))
281
+ tests, _ = self._nodeids(out)
282
+ result.tests |= tests
254
283
  return result
@@ -177,7 +177,7 @@ class VitestAdapter(Adapter):
177
177
  """
178
178
  chunks = []
179
179
  code, out, err = run_command(
180
- self._collect_cmd(), self.root, extra_env=self._suite_env(None)
180
+ self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
181
181
  )
182
182
  if code != 0:
183
183
  chunks.append((err or out).strip())
@@ -190,7 +190,7 @@ class VitestAdapter(Adapter):
190
190
  )
191
191
  continue
192
192
  code, out, err = run_command(
193
- ov.collect_command, self.root, extra_env=self._suite_env(ov)
193
+ ov.collect_command, self.root, extra_env=self._suite_env(ov), label="doctor"
194
194
  )
195
195
  if code != 0:
196
196
  chunks.append((err or out).strip())
@@ -204,7 +204,7 @@ class VitestAdapter(Adapter):
204
204
  if not self.project.overrides:
205
205
  return GateResult(ok=True)
206
206
  code, out, err = run_command(
207
- self._collect_cmd(), self.root, extra_env=self._suite_env(None)
207
+ self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
208
208
  )
209
209
  reached = sorted({
210
210
  f for f in (
@@ -223,10 +223,43 @@ class VitestAdapter(Adapter):
223
223
  " config (test.exclude) or scope its include globs."
224
224
  ))
225
225
 
226
- def collect(self) -> Collection:
227
- result = Collection()
228
- for path in self._test_files():
229
- rel = path.relative_to(self.root)
226
+ def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
227
+ """An override with no `collect_command` gets no batch — `vitest list` knows
228
+ nothing of its config, and its files fall to the loop, which records the
229
+ missing `collect_command` against each of them as before."""
230
+ return [(self._collect_cmd(), self._suite_env(None))] + [
231
+ (ov.collect_command, self._suite_env(ov))
232
+ for ov in self.project.overrides if ov.collect_command
233
+ ]
234
+
235
+ def _collect_batch(
236
+ self, command: str, env: dict[str, str] | None
237
+ ) -> tuple[set[str], set[str]] | None:
238
+ """Unlike `_parse_list_output`, which pins every id to the one file it was
239
+ given, a whole-suite listing must read the file from each line — the same
240
+ `file > describe > name` shape, one file per line rather than one per run."""
241
+ code, out, err = run_command(command, self.root, extra_env=env, label="collect")
242
+ if code != 0:
243
+ return None
244
+ tests: set[str] = set()
245
+ files: set[str] = set()
246
+ for line in out.splitlines():
247
+ line = line.strip()
248
+ if " > " not in line:
249
+ continue
250
+ rel, _, remainder = line.partition(" > ")
251
+ full_name = " ".join(part.strip() for part in remainder.split(" > "))
252
+ if rel and full_name:
253
+ tests.add(self.qualify(f"{rel.strip()} > {full_name}"))
254
+ files.add(rel.strip())
255
+ # An empty result needs no special case: it accounts for no files, so every
256
+ # file falls to the loop exactly as a failure would.
257
+ return tests, files
258
+
259
+ def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
260
+ for rel_str in sorted(rels):
261
+ path = self.root / rel_str
262
+ rel = Path(rel_str)
230
263
  ov = self.project.override_for(str(rel))
231
264
  if ov is not None and not ov.collect_command:
232
265
  result.failed_files[str(rel)] = (
@@ -237,7 +270,8 @@ class VitestAdapter(Adapter):
237
270
  base = ov.collect_command if ov else self._collect_cmd()
238
271
  env = self._suite_env(ov)
239
272
  code, out, err = run_command(
240
- f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env
273
+ f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env,
274
+ label="collect",
241
275
  )
242
276
 
243
277
  payload = _extract_json(out)
@@ -13,6 +13,7 @@ import socket
13
13
  import sqlite3
14
14
  import sys
15
15
  import time
16
+ from collections.abc import Callable
16
17
  from datetime import datetime, timezone
17
18
  from pathlib import Path
18
19
 
@@ -217,16 +218,61 @@ def _legacy_artifacts(worktree: Path) -> list[Path]:
217
218
  return sorted(found)
218
219
 
219
220
 
220
- def cmd_doctor(args) -> Envelope:
221
- worktree = _worktree()
221
+ def _doctor_checklist() -> tuple[list[dict], Callable]:
222
+ """A checks list and its recorder, which refuses a blocker it cannot explain.
223
+
224
+ `resolve_blocker` with an empty `detail` is unfalsifiable: an agent is told to
225
+ fix something and given nothing to fix. It re-runs doctor, reads the identical
226
+ output, and loops. Enforcing the detail here means a check added later inherits
227
+ the guarantee instead of relying on its author to remember.
228
+ """
222
229
  checks: list[dict] = []
223
230
 
224
231
  def check(name, ok, detail="", project=None):
232
+ if not ok and not str(detail).strip():
233
+ raise AssertionError(
234
+ f"doctor check {name!r} would fail silently: a failing check must"
235
+ " name what to fix"
236
+ )
225
237
  entry = {"check": name, "ok": bool(ok), "detail": detail}
226
238
  if project is not None:
227
239
  entry["project"] = project
228
240
  checks.append(entry)
229
241
 
242
+ return checks, check
243
+
244
+
245
+ def _blocks_the_loop(rel_path: str, cfg) -> bool:
246
+ """Whether dirt at `rel_path` is somewhere a run would actually read it."""
247
+ if rel_path == config_mod.CONFIG_NAME:
248
+ return True
249
+ if cfg.owning_project(rel_path) is not None:
250
+ return True
251
+ return any(art.owns(rel_path) for art in cfg.artifacts.values())
252
+
253
+
254
+ def _cleanliness_detail(blocking: list[str], unrelated: list[str]) -> str:
255
+ def listed(paths: list[str]) -> str:
256
+ head = ", ".join(paths[:5])
257
+ return head if len(paths) <= 5 else f"{head} (+{len(paths) - 5} more)"
258
+
259
+ if blocking:
260
+ return (
261
+ f"uncommitted changes a run would observe: {listed(blocking)}."
262
+ " Commit, stash, or gitignore them before `tdd run start`."
263
+ )
264
+ if unrelated:
265
+ return (
266
+ f"clean where a run reads; {len(unrelated)} unrelated path(s) left as-is:"
267
+ f" {listed(unrelated)}"
268
+ )
269
+ return ""
270
+
271
+
272
+ def cmd_doctor(args) -> Envelope:
273
+ worktree = _worktree()
274
+ checks, check = _doctor_checklist()
275
+
230
276
  check("worktree resolvable", True, str(worktree))
231
277
  try:
232
278
  cfg = config_mod.load(worktree)
@@ -246,13 +292,20 @@ def cmd_doctor(args) -> Envelope:
246
292
  root = worktree / project.root
247
293
  check("root exists", root.is_dir(), str(root), project=name)
248
294
  check("adapter known", project.adapter in adapters.available(), project.adapter, project=name)
249
- check("test_paths declared", bool(project.test_paths), project=name)
295
+ declared = bool(project.test_paths)
296
+ check(
297
+ "test_paths declared", declared,
298
+ "" if declared
299
+ else f"add `test_paths` to [project.{name}] in tdd.toml — without it no"
300
+ " suite can be discovered for this project",
301
+ project=name,
302
+ )
250
303
  if project.adapter == "pytest":
251
304
  # The probe runs in the project's own environment (uv, poetry, pipenv,
252
305
  # pdm or the active venv) — hardcoding `uv run` here failed the check
253
306
  # on any non-uv project even with the plugin installed.
254
307
  probe = adapters.build(project, worktree).plugin_probe_cmd()
255
- code, out, err = adapters.base.run_command(probe, root)
308
+ code, out, err = adapters.base.run_command(probe, root, label="doctor")
256
309
  check("pytest-json-report installed", code == 0, (err or "")[:200], project=name)
257
310
 
258
311
  # Run before `collectable()` so this actionable message wins over
@@ -289,11 +342,33 @@ def cmd_doctor(args) -> Envelope:
289
342
  projects[name] = {"ok": all(c["ok"] for c in checks[before:])}
290
343
 
291
344
  for art in cfg.artifacts.values():
292
- check(f"artifact {art.name}: has check or regenerate", bool(art.check or art.regenerate))
345
+ # One evaluation feeds both `ok` and the detail: evaluating the condition
346
+ # twice lets them disagree, and a passing check that still says "add a hook"
347
+ # is the same misdirection as a failing check that says nothing.
348
+ has_hook = bool(art.check or art.regenerate)
349
+ check(
350
+ f"artifact {art.name}: has check or regenerate", has_hook,
351
+ "" if has_hook
352
+ else f"add `check` or `regenerate` to [artifact.{art.name}] in tdd.toml —"
353
+ " freshness cannot be verified without one",
354
+ )
293
355
 
294
356
  stale = _legacy_artifacts(worktree)
295
- check("no legacy state artifacts", not stale, ", ".join(str(s) for s in stale[:5]))
296
- check("worktree clean", not gitutil.is_dirty(worktree))
357
+ check(
358
+ "no legacy state artifacts", not stale,
359
+ f"delete these pre-ledger state files: {', '.join(str(s) for s in stale[:5])}"
360
+ if stale else "",
361
+ )
362
+
363
+ # Only dirt a run would read can corrupt one. Blocking on everything else
364
+ # stopped agents on unrelated notes and editor settings, and — because the
365
+ # check named no path — gave them nothing to act on but a re-run. `is_ignored`
366
+ # also excludes doctor's own probe residue (`.venv`, `node_modules`, caches),
367
+ # so running doctor can no longer be what makes doctor fail.
368
+ dirt = sorted(p for p in gitutil.dirty_paths(worktree) if not cfg.is_ignored(p))
369
+ blocking = [p for p in dirt if _blocks_the_loop(p, cfg)]
370
+ unrelated = [p for p in dirt if p not in set(blocking)]
371
+ check("worktree clean", not blocking, _cleanliness_detail(blocking, unrelated))
297
372
 
298
373
  ok = all(c["ok"] for c in checks)
299
374
  return Envelope(
@@ -364,12 +439,18 @@ def _probe_projects(cfg, worktree, ledger, on_progress):
364
439
  for done, (name, project) in enumerate(cfg.projects.items(), start=1):
365
440
  adapter = adapters.build(project, worktree)
366
441
  started = time.monotonic()
367
- verdict, collection = adapter.run(None), adapter.collect()
442
+ verdict = adapter.run(None)
443
+ ran = time.monotonic()
444
+ collection = adapter.collect()
368
445
  elapsed = time.monotonic() - started
369
446
  probes[name] = (verdict, collection)
447
+ # Split, not just totalled: `run` and `collect` have unrelated cost models
448
+ # — one scales with tests, the other with files — and a single number sends
449
+ # whoever asks "why was that slow?" out of the tool to measure by hand.
370
450
  heartbeat(
371
451
  event="baseline_captured", project=name,
372
452
  test_count=len(collection.tests), elapsed_s=round(elapsed, 2),
453
+ run_s=round(ran - started, 2), collect_s=round(elapsed - (ran - started), 2),
373
454
  )
374
455
  on_progress(done, name)
375
456
  return probes