tdd-cli 0.2.1__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/CHANGELOG.md +56 -1
  2. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/PKG-INFO +2 -2
  3. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/pyproject.toml +1 -1
  4. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/__init__.py +1 -1
  5. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/adapters/base.py +53 -9
  6. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/adapters/pytest_adapter.py +14 -8
  7. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/adapters/vitest_adapter.py +21 -11
  8. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/cli.py +107 -8
  9. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/config.py +13 -0
  10. tdd_cli-0.4.0/tests/test_doctor_blockers.py +173 -0
  11. tdd_cli-0.4.0/tests/test_failure_clipping.py +64 -0
  12. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_project_commands.py +1 -1
  13. tdd_cli-0.4.0/tests/test_project_env.py +124 -0
  14. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_suite_overrides.py +15 -15
  15. tdd_cli-0.4.0/tests/test_target_validation.py +55 -0
  16. tdd_cli-0.4.0/tests/test_timing_visibility.py +149 -0
  17. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_vitest_adapter.py +30 -2
  18. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_worker_leases.py +1 -1
  19. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/.gitignore +0 -0
  20. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/LICENSE +0 -0
  21. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/README.md +0 -0
  22. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/SECURITY.md +0 -0
  23. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/claude-code-hooks/README.md +0 -0
  24. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/claude-code-hooks/bash_hook.py +0 -0
  25. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/claude-code-hooks/stop_hook.py +0 -0
  26. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/plan.md +0 -0
  27. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/skills/tdd-drive/README.md +0 -0
  28. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/skills/tdd-drive/SKILL.md +0 -0
  29. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/skills/tdd-handoff/README.md +0 -0
  30. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
  31. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/adapters/__init__.py +0 -0
  32. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/advance.py +0 -0
  33. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/contract.py +0 -0
  34. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/envelope.py +0 -0
  35. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/fleet.py +0 -0
  36. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/gitutil.py +0 -0
  37. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/identity.py +0 -0
  38. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/leases.py +0 -0
  39. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/ledger.py +0 -0
  40. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/machine.py +0 -0
  41. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/render.py +0 -0
  42. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/snapshot.py +0 -0
  43. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/staging.py +0 -0
  44. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/conftest.py +0 -0
  45. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_artifact_regeneration.py +0 -0
  46. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_baseline_integrity.py +0 -0
  47. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_config_and_staging.py +0 -0
  48. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_config_drift.py +0 -0
  49. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_contract.py +0 -0
  50. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_doctor_attribution.py +0 -0
  51. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_end_to_end.py +0 -0
  52. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_example_plan.py +0 -0
  53. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_fleet.py +0 -0
  54. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_heartbeat.py +0 -0
  55. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_init_detection.py +0 -0
  56. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_pin_cycles.py +0 -0
  57. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_progress.py +0 -0
  58. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_python_env_managers.py +0 -0
  59. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_refactor_cycles.py +0 -0
  60. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_release_surface.py +0 -0
  61. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_run_claim.py +0 -0
  62. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_single_project_repo.py +0 -0
  63. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_snapshot_and_identity.py +0 -0
  64. {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_stub_hint.py +0 -0
@@ -4,7 +4,62 @@ All notable changes to this project are documented here.
4
4
  The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
5
5
  and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
- ## [Unreleased]
7
+ ## [0.4.0] - 2026-08-16
8
+
9
+ ### Added
10
+
11
+ - `baseline_captured` reports `run_s` and `collect_s` alongside `elapsed_s`. The
12
+ suite run and the per-file collection have unrelated cost models — one scales
13
+ with tests, the other with files — so a single total could not say which was
14
+ slow, and answering that meant measuring projects by hand outside the tool.
15
+ - `TDD_TIMING=1` emits a `command_timing` line per subprocess on stderr
16
+ (`label`, `command`, `cwd`, `duration_ms`, `exit_code`), covering every
17
+ subprocess the tool spawns: suite runs, per-file collection, lint/typecheck
18
+ gates, doctor probes and artifact hooks. Off by default — the per-file loop
19
+ would otherwise emit one line per test file on every invocation. `label` is
20
+ one of `suite`, `collect`, `gate`, `doctor`; an unlabelled row comes from a
21
+ third-party adapter, since every built-in call site names itself (R8.4).
22
+
23
+ ### Fixed
24
+
25
+ - `tdd doctor` no longer emits a blocker it cannot explain. Every failing check
26
+ now carries a `detail` naming what to fix, enforced by the checklist recorder
27
+ so a check added later inherits the guarantee. Previously `worktree clean`
28
+ failed with `detail: ""`, leaving an agent with `resolve_blocker` and nothing
29
+ to resolve — it re-ran doctor and read the identical output.
30
+ - `worktree clean` is scoped to dirt a run would actually read: a declared
31
+ project root, a declared artifact path, or `tdd.toml`. Build residue is
32
+ excluded via `config.is_ignored`, so doctor's own `uv run` / `vitest list`
33
+ probes (`.venv`, `node_modules`, caches) can no longer be what makes doctor
34
+ fail. Unrelated dirt is reported in the passing check's `detail` rather than
35
+ blocking the run.
36
+
37
+ ## [0.3.0] - 2026-08-10
38
+
39
+ ### Added
40
+
41
+ - `env` on `[project.<name>]`: environment for the default suite's runs and
42
+ collection, with the same semantics as an override's `env` (`${VAR}` expands
43
+ from the environment at invocation). An override's `env` layers on top for
44
+ its own suite. Previously only override suites could declare environment,
45
+ leaving a default suite that reads an endpoint from a variable with no
46
+ registry-level way to receive it (#16).
47
+
48
+ ### Fixed
49
+
50
+ - vitest test ids are project-root-relative (`frontend::app/x.test.tsx > name`),
51
+ matching pytest nodeids and the form plan declarations qualify to — they were
52
+ worktree-relative (`frontend::frontend/app/...`), so a declared vitest target
53
+ could never match a verdict: standard cycles limped through on R8.9 adoption
54
+ (a spurious `declared_test_mismatch` per cycle) and pin cycles deadlocked in
55
+ `AWAITING_PIN`, since a pre-existing test is never adoptable (#21).
56
+ - `tdd target` refuses a name that is not a collected test in the cycle's
57
+ projects, suggesting the closest collected ids — previously any string was
58
+ recorded as the target and failed later, misattributed, as `not_found` (#15).
59
+ - Failure text (`target_failure`, uncollected-suite messages) is clipped keeping
60
+ both ends instead of truncated from the head: Python puts the actual error at
61
+ the tail of a traceback, so a head-only cut on a deep stack delivered
62
+ framework frames and cut exactly the line that says what went wrong (#17).
8
63
 
9
64
  ## [0.2.1] - 2026-08-10
10
65
 
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.2.1
3
+ Version: 0.4.0
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -47,7 +47,7 @@ packages = ["src/tddcli"]
47
47
  include = ["src", "tests", "examples", "README.md", "LICENSE", "CHANGELOG.md", "SECURITY.md"]
48
48
 
49
49
  [dependency-groups]
50
- dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5"]
50
+ dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5", "zizmor>=1.29"]
51
51
 
52
52
  [tool.pytest.ini_options]
53
53
  testpaths = ["tests"]
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.2.1"
6
+ __version__ = "0.4.0"
@@ -4,11 +4,13 @@ from __future__ import annotations
4
4
 
5
5
  import os
6
6
  import subprocess
7
+ import time
7
8
  from collections import Counter
8
9
  from dataclasses import dataclass, field
9
10
  from pathlib import Path
10
11
 
11
12
  from .. import leases
13
+ from ..envelope import heartbeat
12
14
 
13
15
  NOT_FOUND = "not_found"
14
16
  NOT_COLLECTED = "not_collected"
@@ -64,10 +66,40 @@ def _overlap_error(overlap: list[str]) -> str:
64
66
  )
65
67
 
66
68
 
69
+ def clip_failure(text: str, limit: int = 1500) -> str:
70
+ """Clip failure text to `limit`, keeping both ends. Python puts the actual
71
+ error at the tail of a traceback, so a head-only cut on a deep stack
72
+ (async frameworks, ORMs, HTTP clients) delivered framework frames and cut
73
+ exactly the line that says what went wrong — forcing a re-run outside tdd
74
+ to see an error the tool already had. The head is kept too: for a plain
75
+ assertion failure the first line carries the assertion itself."""
76
+ if len(text) <= limit:
77
+ return text
78
+ head = limit // 5
79
+ tail = limit - head
80
+ return f"{text[:head]}\n… [clipped] …\n{text[-tail:]}"
81
+
82
+
83
+ #: Opt-in per-command timing. Off by default: the per-file collect loop would emit
84
+ #: one line per test file on every invocation, drowning the heartbeats that exist
85
+ #: to make a slow baseline legible.
86
+ TIMING_ENV = "TDD_TIMING"
87
+
88
+
67
89
  def run_command(
68
90
  command: str, cwd: Path, timeout: int = 1800,
69
91
  extra_env: dict[str, str] | None = None,
92
+ label: str | None = None,
70
93
  ) -> tuple[int, str, str]:
94
+ """Every subprocess the tool spawns passes through here, which makes it the one
95
+ place worth timing: suite runs, per-file collection, lint and typecheck gates,
96
+ doctor probes, artifact hooks.
97
+
98
+ `label` is what makes the rows groupable. This function sees a command string
99
+ and a cwd — not which project or phase asked for it — so an unlabelled timing
100
+ is readable by a human and useless to a query.
101
+ """
102
+ started = time.monotonic()
71
103
  proc = subprocess.run(
72
104
  command,
73
105
  shell=True,
@@ -77,6 +109,15 @@ def run_command(
77
109
  timeout=timeout,
78
110
  env=None if extra_env is None else {**os.environ, **extra_env},
79
111
  )
112
+ if os.environ.get(TIMING_ENV):
113
+ heartbeat(
114
+ event="command_timing",
115
+ label=label,
116
+ command=command,
117
+ cwd=str(cwd),
118
+ duration_ms=int((time.monotonic() - started) * 1000),
119
+ exit_code=proc.returncode,
120
+ )
80
121
  return proc.returncode, proc.stdout, proc.stderr
81
122
 
82
123
 
@@ -116,6 +157,7 @@ class Adapter:
116
157
  command.replace("{workers}", str(workers)),
117
158
  self.root,
118
159
  extra_env={"TDD_WORKERS": str(workers), **(extra_env or {})},
160
+ label="suite",
119
161
  )
120
162
 
121
163
  def _test_cmd(self) -> str:
@@ -128,17 +170,19 @@ class Adapter:
128
170
  alternate runner config is still observed — without widening the default
129
171
  config, which is exactly the workaround that breaks CI.
130
172
  """
131
- return [(self._test_cmd(), None)] + [
132
- (ov.test_command, self._override_env(ov)) for ov in self.project.overrides
173
+ return [(self._test_cmd(), self._suite_env(None))] + [
174
+ (ov.test_command, self._suite_env(ov)) for ov in self.project.overrides
133
175
  ]
134
176
 
135
- @staticmethod
136
- def _override_env(override) -> dict[str, str] | None:
137
- """`${VAR}` references resolve from the environment at invocation time, so a
138
- port assigned by the harness need not be hard-coded in the reviewed file."""
139
- if override is None or not override.env:
177
+ def _suite_env(self, override) -> dict[str, str] | None:
178
+ """The environment for one suite invocation: the project's `env`, with the
179
+ owning override's layered on top (None means the default suite). `${VAR}`
180
+ references resolve from the environment at invocation time, so a port
181
+ assigned per checkout need not be hard-coded in the reviewed file."""
182
+ merged = {**self.project.env, **(override.env if override else {})}
183
+ if not merged:
140
184
  return None
141
- return {k: os.path.expandvars(v) for k, v in override.env.items()}
185
+ return {k: os.path.expandvars(v) for k, v in merged.items()}
142
186
 
143
187
  def stub_hint(self) -> str:
144
188
  """The language idiom for a stub body, quoted into the create_stub directive."""
@@ -168,7 +212,7 @@ class Adapter:
168
212
  def _gate(self, commands: list[str]) -> GateResult:
169
213
  chunks = []
170
214
  for cmd in commands:
171
- code, out, err = run_command(cmd, self.root)
215
+ code, out, err = run_command(cmd, self.root, label="gate")
172
216
  if code != 0:
173
217
  chunks.append(f"$ {cmd}\n{out}\n{err}".strip())
174
218
  return GateResult(ok=not chunks, output="\n\n".join(chunks)[:4000])
@@ -27,6 +27,7 @@ from .base import (
27
27
  Verdict,
28
28
  _overlap_error,
29
29
  _suite_overlap,
30
+ clip_failure,
30
31
  run_command,
31
32
  )
32
33
 
@@ -139,12 +140,14 @@ class PytestAdapter(Adapter):
139
140
  if hit is not None:
140
141
  verdict.target_outcome = PASSED if hit["outcome"] == "passed" else FAILED
141
142
  call = hit.get("call") or hit.get("setup") or {}
142
- verdict.target_failure = str(call.get("longrepr", ""))[:1500]
143
+ verdict.target_failure = clip_failure(str(call.get("longrepr", "")))
143
144
  else:
144
145
  target_file = native.split("::", 1)[0]
145
146
  if any(c == target_file or c.startswith(target_file) for c in uncollectable):
146
147
  verdict.target_outcome = NOT_COLLECTED
147
- verdict.target_failure = self._collector_error(collectors, target_file)[:1500]
148
+ verdict.target_failure = clip_failure(
149
+ self._collector_error(collectors, target_file)
150
+ )
148
151
  else:
149
152
  verdict.target_outcome = NOT_FOUND
150
153
  return verdict
@@ -162,8 +165,8 @@ class PytestAdapter(Adapter):
162
165
  `test_command` — pytest's `--collect-only` composes with any run command."""
163
166
  ov = self.project.override_for(rel)
164
167
  if ov is None:
165
- return self._collect_cmd(), None
166
- return ov.collect_command or ov.test_command, self._override_env(ov)
168
+ return self._collect_cmd(), self._suite_env(None)
169
+ return ov.collect_command or ov.test_command, self._suite_env(ov)
167
170
 
168
171
  def _test_files(self) -> list[Path]:
169
172
  found: list[Path] = []
@@ -190,13 +193,13 @@ class PytestAdapter(Adapter):
190
193
  loses the real error and the failure surfaces unattributed.
191
194
  """
192
195
  chunks = []
193
- probes = [(self._collect_cmd(), None)] + [
194
- (ov.collect_command or ov.test_command, self._override_env(ov))
196
+ probes = [(self._collect_cmd(), self._suite_env(None))] + [
197
+ (ov.collect_command or ov.test_command, self._suite_env(ov))
195
198
  for ov in self.project.overrides
196
199
  ]
197
200
  for cmd, env in probes:
198
201
  code, out, err = run_command(
199
- f"{cmd} --collect-only -q", self.root, extra_env=env
202
+ f"{cmd} --collect-only -q", self.root, extra_env=env, label="doctor"
200
203
  )
201
204
  if code != 0:
202
205
  chunks.append(out.strip())
@@ -212,7 +215,9 @@ class PytestAdapter(Adapter):
212
215
  if not self.project.overrides:
213
216
  return GateResult(ok=True)
214
217
  probe = f"{self._test_cmd().replace('{workers}', '0')} --collect-only -q"
215
- code, out, err = run_command(probe, self.root)
218
+ code, out, err = run_command(
219
+ probe, self.root, extra_env=self._suite_env(None), label="doctor"
220
+ )
216
221
  reached = sorted({
217
222
  f for f in (
218
223
  line.split("::", 1)[0]
@@ -240,6 +245,7 @@ class PytestAdapter(Adapter):
240
245
  f"{base} --collect-only -q {shlex.quote(str(rel))}",
241
246
  self.root,
242
247
  extra_env=env,
248
+ label="collect",
243
249
  )
244
250
  if code != 0:
245
251
  result.failed_files[str(rel)] = (err or out).strip()[:800]
@@ -1,8 +1,11 @@
1
1
  """vitest adapter.
2
2
 
3
- Test ids are `<worktree-relative file> > <fullName>`, where fullName is the
4
- space-joined ancestorTitles plus the test title. vitest may prefix its JSON with
5
- non-JSON lines, so the payload is located rather than assumed (R10.2).
3
+ Test ids are `<project-root-relative file> > <fullName>`, where fullName is the
4
+ space-joined ancestorTitles plus the test title. Root-relative matches the pytest
5
+ adapter's nodeids and — decisively — `Engine._qualify`, which strips the project
6
+ root from plan declarations; a worktree-relative id here can never equal a
7
+ declared target. vitest may prefix its JSON with non-JSON lines, so the payload
8
+ is located rather than assumed (R10.2).
6
9
  """
7
10
 
8
11
  from __future__ import annotations
@@ -23,6 +26,7 @@ from .base import (
23
26
  Verdict,
24
27
  _overlap_error,
25
28
  _suite_overlap,
29
+ clip_failure,
26
30
  run_command,
27
31
  )
28
32
 
@@ -46,7 +50,7 @@ class VitestAdapter(Adapter):
46
50
  def _id_for(self, suite_path: str, full_name: str) -> str:
47
51
  abs_path = Path(suite_path)
48
52
  try:
49
- rel = os.path.relpath(abs_path, self.worktree)
53
+ rel = os.path.relpath(abs_path, self.root)
50
54
  except ValueError:
51
55
  rel = suite_path
52
56
  return self.qualify(f"{rel} > {full_name}")
@@ -92,7 +96,7 @@ class VitestAdapter(Adapter):
92
96
  suite_path = suite.get("name", "")
93
97
  assertions = suite.get("assertionResults", [])
94
98
  if not assertions and suite.get("status") == "failed":
95
- failed_suites[suite_path] = str(suite.get("message", ""))[:1500]
99
+ failed_suites[suite_path] = clip_failure(str(suite.get("message", "")))
96
100
  for t in assertions:
97
101
  qualified = self._id_for(suite_path, t["fullName"])
98
102
  if t["status"] == "passed":
@@ -113,7 +117,8 @@ class VitestAdapter(Adapter):
113
117
  for t in suite.get("assertionResults", []):
114
118
  if self._id_for(suite.get("name", ""), t["fullName"]) == target:
115
119
  verdict.target_failure = "\n".join(
116
- m[:600] for m in t.get("failureMessages", [])[:3]
120
+ clip_failure(m, 600)
121
+ for m in t.get("failureMessages", [])[:3]
117
122
  )
118
123
  return verdict
119
124
 
@@ -171,7 +176,9 @@ class VitestAdapter(Adapter):
171
176
  suite — against a live backend — just to enumerate it.
172
177
  """
173
178
  chunks = []
174
- code, out, err = run_command(self._collect_cmd(), self.root)
179
+ code, out, err = run_command(
180
+ self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
181
+ )
175
182
  if code != 0:
176
183
  chunks.append((err or out).strip())
177
184
  for ov in self.project.overrides:
@@ -183,7 +190,7 @@ class VitestAdapter(Adapter):
183
190
  )
184
191
  continue
185
192
  code, out, err = run_command(
186
- ov.collect_command, self.root, extra_env=self._override_env(ov)
193
+ ov.collect_command, self.root, extra_env=self._suite_env(ov), label="doctor"
187
194
  )
188
195
  if code != 0:
189
196
  chunks.append((err or out).strip())
@@ -196,7 +203,9 @@ class VitestAdapter(Adapter):
196
203
  execute against whatever the tests need live)."""
197
204
  if not self.project.overrides:
198
205
  return GateResult(ok=True)
199
- code, out, err = run_command(self._collect_cmd(), self.root)
206
+ code, out, err = run_command(
207
+ self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
208
+ )
200
209
  reached = sorted({
201
210
  f for f in (
202
211
  line.strip().partition(" > ")[0]
@@ -226,9 +235,10 @@ class VitestAdapter(Adapter):
226
235
  )
227
236
  continue
228
237
  base = ov.collect_command if ov else self._collect_cmd()
229
- env = self._override_env(ov)
238
+ env = self._suite_env(ov)
230
239
  code, out, err = run_command(
231
- f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env
240
+ f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env,
241
+ label="collect",
232
242
  )
233
243
 
234
244
  payload = _extract_json(out)
@@ -6,12 +6,14 @@ No command accepts a phase, a cycle number, or executor identity (R8.3).
6
6
  from __future__ import annotations
7
7
 
8
8
  import argparse
9
+ import difflib
9
10
  import json
10
11
  import os
11
12
  import socket
12
13
  import sqlite3
13
14
  import sys
14
15
  import time
16
+ from collections.abc import Callable
15
17
  from datetime import datetime, timezone
16
18
  from pathlib import Path
17
19
 
@@ -216,16 +218,61 @@ def _legacy_artifacts(worktree: Path) -> list[Path]:
216
218
  return sorted(found)
217
219
 
218
220
 
219
- def cmd_doctor(args) -> Envelope:
220
- worktree = _worktree()
221
+ def _doctor_checklist() -> tuple[list[dict], Callable]:
222
+ """A checks list and its recorder, which refuses a blocker it cannot explain.
223
+
224
+ `resolve_blocker` with an empty `detail` is unfalsifiable: an agent is told to
225
+ fix something and given nothing to fix. It re-runs doctor, reads the identical
226
+ output, and loops. Enforcing the detail here means a check added later inherits
227
+ the guarantee instead of relying on its author to remember.
228
+ """
221
229
  checks: list[dict] = []
222
230
 
223
231
  def check(name, ok, detail="", project=None):
232
+ if not ok and not str(detail).strip():
233
+ raise AssertionError(
234
+ f"doctor check {name!r} would fail silently: a failing check must"
235
+ " name what to fix"
236
+ )
224
237
  entry = {"check": name, "ok": bool(ok), "detail": detail}
225
238
  if project is not None:
226
239
  entry["project"] = project
227
240
  checks.append(entry)
228
241
 
242
+ return checks, check
243
+
244
+
245
+ def _blocks_the_loop(rel_path: str, cfg) -> bool:
246
+ """Whether dirt at `rel_path` is somewhere a run would actually read it."""
247
+ if rel_path == config_mod.CONFIG_NAME:
248
+ return True
249
+ if cfg.owning_project(rel_path) is not None:
250
+ return True
251
+ return any(art.owns(rel_path) for art in cfg.artifacts.values())
252
+
253
+
254
+ def _cleanliness_detail(blocking: list[str], unrelated: list[str]) -> str:
255
+ def listed(paths: list[str]) -> str:
256
+ head = ", ".join(paths[:5])
257
+ return head if len(paths) <= 5 else f"{head} (+{len(paths) - 5} more)"
258
+
259
+ if blocking:
260
+ return (
261
+ f"uncommitted changes a run would observe: {listed(blocking)}."
262
+ " Commit, stash, or gitignore them before `tdd run start`."
263
+ )
264
+ if unrelated:
265
+ return (
266
+ f"clean where a run reads; {len(unrelated)} unrelated path(s) left as-is:"
267
+ f" {listed(unrelated)}"
268
+ )
269
+ return ""
270
+
271
+
272
+ def cmd_doctor(args) -> Envelope:
273
+ worktree = _worktree()
274
+ checks, check = _doctor_checklist()
275
+
229
276
  check("worktree resolvable", True, str(worktree))
230
277
  try:
231
278
  cfg = config_mod.load(worktree)
@@ -245,13 +292,20 @@ def cmd_doctor(args) -> Envelope:
245
292
  root = worktree / project.root
246
293
  check("root exists", root.is_dir(), str(root), project=name)
247
294
  check("adapter known", project.adapter in adapters.available(), project.adapter, project=name)
248
- check("test_paths declared", bool(project.test_paths), project=name)
295
+ declared = bool(project.test_paths)
296
+ check(
297
+ "test_paths declared", declared,
298
+ "" if declared
299
+ else f"add `test_paths` to [project.{name}] in tdd.toml — without it no"
300
+ " suite can be discovered for this project",
301
+ project=name,
302
+ )
249
303
  if project.adapter == "pytest":
250
304
  # The probe runs in the project's own environment (uv, poetry, pipenv,
251
305
  # pdm or the active venv) — hardcoding `uv run` here failed the check
252
306
  # on any non-uv project even with the plugin installed.
253
307
  probe = adapters.build(project, worktree).plugin_probe_cmd()
254
- code, out, err = adapters.base.run_command(probe, root)
308
+ code, out, err = adapters.base.run_command(probe, root, label="doctor")
255
309
  check("pytest-json-report installed", code == 0, (err or "")[:200], project=name)
256
310
 
257
311
  # Run before `collectable()` so this actionable message wins over
@@ -288,11 +342,33 @@ def cmd_doctor(args) -> Envelope:
288
342
  projects[name] = {"ok": all(c["ok"] for c in checks[before:])}
289
343
 
290
344
  for art in cfg.artifacts.values():
291
- check(f"artifact {art.name}: has check or regenerate", bool(art.check or art.regenerate))
345
+ # One evaluation feeds both `ok` and the detail: evaluating the condition
346
+ # twice lets them disagree, and a passing check that still says "add a hook"
347
+ # is the same misdirection as a failing check that says nothing.
348
+ has_hook = bool(art.check or art.regenerate)
349
+ check(
350
+ f"artifact {art.name}: has check or regenerate", has_hook,
351
+ "" if has_hook
352
+ else f"add `check` or `regenerate` to [artifact.{art.name}] in tdd.toml —"
353
+ " freshness cannot be verified without one",
354
+ )
292
355
 
293
356
  stale = _legacy_artifacts(worktree)
294
- check("no legacy state artifacts", not stale, ", ".join(str(s) for s in stale[:5]))
295
- check("worktree clean", not gitutil.is_dirty(worktree))
357
+ check(
358
+ "no legacy state artifacts", not stale,
359
+ f"delete these pre-ledger state files: {', '.join(str(s) for s in stale[:5])}"
360
+ if stale else "",
361
+ )
362
+
363
+ # Only dirt a run would read can corrupt one. Blocking on everything else
364
+ # stopped agents on unrelated notes and editor settings, and — because the
365
+ # check named no path — gave them nothing to act on but a re-run. `is_ignored`
366
+ # also excludes doctor's own probe residue (`.venv`, `node_modules`, caches),
367
+ # so running doctor can no longer be what makes doctor fail.
368
+ dirt = sorted(p for p in gitutil.dirty_paths(worktree) if not cfg.is_ignored(p))
369
+ blocking = [p for p in dirt if _blocks_the_loop(p, cfg)]
370
+ unrelated = [p for p in dirt if p not in set(blocking)]
371
+ check("worktree clean", not blocking, _cleanliness_detail(blocking, unrelated))
296
372
 
297
373
  ok = all(c["ok"] for c in checks)
298
374
  return Envelope(
@@ -363,12 +439,18 @@ def _probe_projects(cfg, worktree, ledger, on_progress):
363
439
  for done, (name, project) in enumerate(cfg.projects.items(), start=1):
364
440
  adapter = adapters.build(project, worktree)
365
441
  started = time.monotonic()
366
- verdict, collection = adapter.run(None), adapter.collect()
442
+ verdict = adapter.run(None)
443
+ ran = time.monotonic()
444
+ collection = adapter.collect()
367
445
  elapsed = time.monotonic() - started
368
446
  probes[name] = (verdict, collection)
447
+ # Split, not just totalled: `run` and `collect` have unrelated cost models
448
+ # — one scales with tests, the other with files — and a single number sends
449
+ # whoever asks "why was that slow?" out of the tool to measure by hand.
369
450
  heartbeat(
370
451
  event="baseline_captured", project=name,
371
452
  test_count=len(collection.tests), elapsed_s=round(elapsed, 2),
453
+ run_s=round(ran - started, 2), collect_s=round(elapsed - (ran - started), 2),
372
454
  )
373
455
  on_progress(done, name)
374
456
  return probes
@@ -822,6 +904,23 @@ def cmd_target(args) -> Envelope:
822
904
  cycle = ledger.open_cycle(run["id"])
823
905
  if cycle is None:
824
906
  return failure("no open cycle")
907
+
908
+ # The target must be grounded in observed collection, the same way phase is
909
+ # grounded in observed execution (#15): recording free text deferred a typo —
910
+ # or a speculative `tdd target env` — to the next suite run, where it
911
+ # surfaced as `not_found` against a test that never existed.
912
+ known: set[str] = set()
913
+ for name in json.loads(cycle["projects"]):
914
+ adapter = adapters.build(cfg.project(name), worktree)
915
+ known |= adapter.collect().tests
916
+ if args.test not in known:
917
+ close = difflib.get_close_matches(args.test, sorted(known), n=3, cutoff=0.6)
918
+ hint = f" Closest collected ids: {', '.join(close)}." if close else ""
919
+ return failure(
920
+ f"{args.test} is not a collected test in this cycle's projects;"
921
+ f" the target was not changed.{hint}"
922
+ )
923
+
825
924
  ledger.update("cycle", cycle["id"], target_tests=json.dumps([args.test]))
826
925
  ledger.event(run["id"], cycle["id"], "target_named_by_agent", args.test)
827
926
  return Envelope(
@@ -88,6 +88,11 @@ class Project:
88
88
  #: Per-file collection. Must not be parallelised: collection is cheap and xdist
89
89
  #: adds startup cost per file.
90
90
  collect_command: str | None = None
91
+ #: Environment for the default suite's runs and collection, same semantics as
92
+ #: an override's `env`: `${VAR}` references expand from the environment at
93
+ #: invocation, so per-checkout values (a database port) stay out of the
94
+ #: reviewed file. An override's `env` layers on top for its own suite.
95
+ env: dict[str, str] = field(default_factory=dict)
91
96
  #: Alternate suites for files the default command cannot reach (R7.13).
92
97
  overrides: list[Override] = field(default_factory=list)
93
98
 
@@ -290,6 +295,13 @@ def load(worktree: Path) -> Config:
290
295
  raise ConfigError(f"project {name!r} has no root")
291
296
  if "adapter" not in body:
292
297
  raise ConfigError(f"project {name!r} has no adapter")
298
+ env = body.get("env", {})
299
+ if not isinstance(env, dict) or not all(
300
+ isinstance(v, str) for v in env.values()
301
+ ):
302
+ raise ConfigError(
303
+ f"project {name!r}: env must be a table of string values"
304
+ )
293
305
  projects[name] = Project(
294
306
  name=name,
295
307
  root=body["root"].rstrip("/"),
@@ -300,6 +312,7 @@ def load(worktree: Path) -> Config:
300
312
  in_close_sweep=body.get("in_close_sweep", True),
301
313
  test_command=body.get("test_command"),
302
314
  collect_command=body.get("collect_command"),
315
+ env=env,
303
316
  overrides=_load_overrides(name, body.get("override", [])),
304
317
  )
305
318