tdd-cli 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/CHANGELOG.md +30 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/PKG-INFO +2 -2
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/pyproject.toml +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/adapters/base.py +29 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/adapters/pytest_adapter.py +5 -2
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/adapters/vitest_adapter.py +5 -4
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/cli.py +89 -8
- tdd_cli-0.4.0/tests/test_doctor_blockers.py +173 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_failure_clipping.py +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_project_commands.py +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_project_env.py +3 -3
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_suite_overrides.py +10 -10
- tdd_cli-0.4.0/tests/test_timing_visibility.py +149 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_worker_leases.py +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/.gitignore +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/LICENSE +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/README.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/SECURITY.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/plan.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/advance.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/config.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/contract.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/identity.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/ledger.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/machine.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/render.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/src/tddcli/staging.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/conftest.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_artifact_regeneration.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_baseline_integrity.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_config_and_staging.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_contract.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_heartbeat.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_release_surface.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_snapshot_and_identity.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.0}/tests/test_vitest_adapter.py +0 -0
|
@@ -4,6 +4,36 @@ All notable changes to this project are documented here.
|
|
|
4
4
|
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
5
5
|
and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [0.4.0] - 2026-08-16
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- `baseline_captured` reports `run_s` and `collect_s` alongside `elapsed_s`. The
|
|
12
|
+
suite run and the per-file collection have unrelated cost models — one scales
|
|
13
|
+
with tests, the other with files — so a single total could not say which was
|
|
14
|
+
slow, and answering that meant measuring projects by hand outside the tool.
|
|
15
|
+
- `TDD_TIMING=1` emits a `command_timing` line per subprocess on stderr
|
|
16
|
+
(`label`, `command`, `cwd`, `duration_ms`, `exit_code`), covering every
|
|
17
|
+
subprocess the tool spawns: suite runs, per-file collection, lint/typecheck
|
|
18
|
+
gates, doctor probes and artifact hooks. Off by default — the per-file loop
|
|
19
|
+
would otherwise emit one line per test file on every invocation. `label` is
|
|
20
|
+
one of `suite`, `collect`, `gate`, `doctor`; an unlabelled row comes from a
|
|
21
|
+
third-party adapter, since every built-in call site names itself (R8.4).
|
|
22
|
+
|
|
23
|
+
### Fixed
|
|
24
|
+
|
|
25
|
+
- `tdd doctor` no longer emits a blocker it cannot explain. Every failing check
|
|
26
|
+
now carries a `detail` naming what to fix, enforced by the checklist recorder
|
|
27
|
+
so a check added later inherits the guarantee. Previously `worktree clean`
|
|
28
|
+
failed with `detail: ""`, leaving an agent with `resolve_blocker` and nothing
|
|
29
|
+
to resolve — it re-ran doctor and read the identical output.
|
|
30
|
+
- `worktree clean` is scoped to dirt a run would actually read: a declared
|
|
31
|
+
project root, a declared artifact path, or `tdd.toml`. Build residue is
|
|
32
|
+
excluded via `config.is_ignored`, so doctor's own `uv run` / `vitest list`
|
|
33
|
+
probes (`.venv`, `node_modules`, caches) can no longer be what makes doctor
|
|
34
|
+
fail. Unrelated dirt is reported in the passing check's `detail` rather than
|
|
35
|
+
blocking the run.
|
|
36
|
+
|
|
7
37
|
## [0.3.0] - 2026-08-10
|
|
8
38
|
|
|
9
39
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -47,7 +47,7 @@ packages = ["src/tddcli"]
|
|
|
47
47
|
include = ["src", "tests", "examples", "README.md", "LICENSE", "CHANGELOG.md", "SECURITY.md"]
|
|
48
48
|
|
|
49
49
|
[dependency-groups]
|
|
50
|
-
dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5"]
|
|
50
|
+
dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5", "zizmor>=1.29"]
|
|
51
51
|
|
|
52
52
|
[tool.pytest.ini_options]
|
|
53
53
|
testpaths = ["tests"]
|
|
@@ -4,11 +4,13 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import os
|
|
6
6
|
import subprocess
|
|
7
|
+
import time
|
|
7
8
|
from collections import Counter
|
|
8
9
|
from dataclasses import dataclass, field
|
|
9
10
|
from pathlib import Path
|
|
10
11
|
|
|
11
12
|
from .. import leases
|
|
13
|
+
from ..envelope import heartbeat
|
|
12
14
|
|
|
13
15
|
NOT_FOUND = "not_found"
|
|
14
16
|
NOT_COLLECTED = "not_collected"
|
|
@@ -78,10 +80,26 @@ def clip_failure(text: str, limit: int = 1500) -> str:
|
|
|
78
80
|
return f"{text[:head]}\n… [clipped] …\n{text[-tail:]}"
|
|
79
81
|
|
|
80
82
|
|
|
83
|
+
#: Opt-in per-command timing. Off by default: the per-file collect loop would emit
|
|
84
|
+
#: one line per test file on every invocation, drowning the heartbeats that exist
|
|
85
|
+
#: to make a slow baseline legible.
|
|
86
|
+
TIMING_ENV = "TDD_TIMING"
|
|
87
|
+
|
|
88
|
+
|
|
81
89
|
def run_command(
|
|
82
90
|
command: str, cwd: Path, timeout: int = 1800,
|
|
83
91
|
extra_env: dict[str, str] | None = None,
|
|
92
|
+
label: str | None = None,
|
|
84
93
|
) -> tuple[int, str, str]:
|
|
94
|
+
"""Every subprocess the tool spawns passes through here, which makes it the one
|
|
95
|
+
place worth timing: suite runs, per-file collection, lint and typecheck gates,
|
|
96
|
+
doctor probes, artifact hooks.
|
|
97
|
+
|
|
98
|
+
`label` is what makes the rows groupable. This function sees a command string
|
|
99
|
+
and a cwd — not which project or phase asked for it — so an unlabelled timing
|
|
100
|
+
is readable by a human and useless to a query.
|
|
101
|
+
"""
|
|
102
|
+
started = time.monotonic()
|
|
85
103
|
proc = subprocess.run(
|
|
86
104
|
command,
|
|
87
105
|
shell=True,
|
|
@@ -91,6 +109,15 @@ def run_command(
|
|
|
91
109
|
timeout=timeout,
|
|
92
110
|
env=None if extra_env is None else {**os.environ, **extra_env},
|
|
93
111
|
)
|
|
112
|
+
if os.environ.get(TIMING_ENV):
|
|
113
|
+
heartbeat(
|
|
114
|
+
event="command_timing",
|
|
115
|
+
label=label,
|
|
116
|
+
command=command,
|
|
117
|
+
cwd=str(cwd),
|
|
118
|
+
duration_ms=int((time.monotonic() - started) * 1000),
|
|
119
|
+
exit_code=proc.returncode,
|
|
120
|
+
)
|
|
94
121
|
return proc.returncode, proc.stdout, proc.stderr
|
|
95
122
|
|
|
96
123
|
|
|
@@ -130,6 +157,7 @@ class Adapter:
|
|
|
130
157
|
command.replace("{workers}", str(workers)),
|
|
131
158
|
self.root,
|
|
132
159
|
extra_env={"TDD_WORKERS": str(workers), **(extra_env or {})},
|
|
160
|
+
label="suite",
|
|
133
161
|
)
|
|
134
162
|
|
|
135
163
|
def _test_cmd(self) -> str:
|
|
@@ -184,7 +212,7 @@ class Adapter:
|
|
|
184
212
|
def _gate(self, commands: list[str]) -> GateResult:
|
|
185
213
|
chunks = []
|
|
186
214
|
for cmd in commands:
|
|
187
|
-
code, out, err = run_command(cmd, self.root)
|
|
215
|
+
code, out, err = run_command(cmd, self.root, label="gate")
|
|
188
216
|
if code != 0:
|
|
189
217
|
chunks.append(f"$ {cmd}\n{out}\n{err}".strip())
|
|
190
218
|
return GateResult(ok=not chunks, output="\n\n".join(chunks)[:4000])
|
|
@@ -199,7 +199,7 @@ class PytestAdapter(Adapter):
|
|
|
199
199
|
]
|
|
200
200
|
for cmd, env in probes:
|
|
201
201
|
code, out, err = run_command(
|
|
202
|
-
f"{cmd} --collect-only -q", self.root, extra_env=env
|
|
202
|
+
f"{cmd} --collect-only -q", self.root, extra_env=env, label="doctor"
|
|
203
203
|
)
|
|
204
204
|
if code != 0:
|
|
205
205
|
chunks.append(out.strip())
|
|
@@ -215,7 +215,9 @@ class PytestAdapter(Adapter):
|
|
|
215
215
|
if not self.project.overrides:
|
|
216
216
|
return GateResult(ok=True)
|
|
217
217
|
probe = f"{self._test_cmd().replace('{workers}', '0')} --collect-only -q"
|
|
218
|
-
code, out, err = run_command(
|
|
218
|
+
code, out, err = run_command(
|
|
219
|
+
probe, self.root, extra_env=self._suite_env(None), label="doctor"
|
|
220
|
+
)
|
|
219
221
|
reached = sorted({
|
|
220
222
|
f for f in (
|
|
221
223
|
line.split("::", 1)[0]
|
|
@@ -243,6 +245,7 @@ class PytestAdapter(Adapter):
|
|
|
243
245
|
f"{base} --collect-only -q {shlex.quote(str(rel))}",
|
|
244
246
|
self.root,
|
|
245
247
|
extra_env=env,
|
|
248
|
+
label="collect",
|
|
246
249
|
)
|
|
247
250
|
if code != 0:
|
|
248
251
|
result.failed_files[str(rel)] = (err or out).strip()[:800]
|
|
@@ -177,7 +177,7 @@ class VitestAdapter(Adapter):
|
|
|
177
177
|
"""
|
|
178
178
|
chunks = []
|
|
179
179
|
code, out, err = run_command(
|
|
180
|
-
self._collect_cmd(), self.root, extra_env=self._suite_env(None)
|
|
180
|
+
self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
|
|
181
181
|
)
|
|
182
182
|
if code != 0:
|
|
183
183
|
chunks.append((err or out).strip())
|
|
@@ -190,7 +190,7 @@ class VitestAdapter(Adapter):
|
|
|
190
190
|
)
|
|
191
191
|
continue
|
|
192
192
|
code, out, err = run_command(
|
|
193
|
-
ov.collect_command, self.root, extra_env=self._suite_env(ov)
|
|
193
|
+
ov.collect_command, self.root, extra_env=self._suite_env(ov), label="doctor"
|
|
194
194
|
)
|
|
195
195
|
if code != 0:
|
|
196
196
|
chunks.append((err or out).strip())
|
|
@@ -204,7 +204,7 @@ class VitestAdapter(Adapter):
|
|
|
204
204
|
if not self.project.overrides:
|
|
205
205
|
return GateResult(ok=True)
|
|
206
206
|
code, out, err = run_command(
|
|
207
|
-
self._collect_cmd(), self.root, extra_env=self._suite_env(None)
|
|
207
|
+
self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
|
|
208
208
|
)
|
|
209
209
|
reached = sorted({
|
|
210
210
|
f for f in (
|
|
@@ -237,7 +237,8 @@ class VitestAdapter(Adapter):
|
|
|
237
237
|
base = ov.collect_command if ov else self._collect_cmd()
|
|
238
238
|
env = self._suite_env(ov)
|
|
239
239
|
code, out, err = run_command(
|
|
240
|
-
f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env
|
|
240
|
+
f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env,
|
|
241
|
+
label="collect",
|
|
241
242
|
)
|
|
242
243
|
|
|
243
244
|
payload = _extract_json(out)
|
|
@@ -13,6 +13,7 @@ import socket
|
|
|
13
13
|
import sqlite3
|
|
14
14
|
import sys
|
|
15
15
|
import time
|
|
16
|
+
from collections.abc import Callable
|
|
16
17
|
from datetime import datetime, timezone
|
|
17
18
|
from pathlib import Path
|
|
18
19
|
|
|
@@ -217,16 +218,61 @@ def _legacy_artifacts(worktree: Path) -> list[Path]:
|
|
|
217
218
|
return sorted(found)
|
|
218
219
|
|
|
219
220
|
|
|
220
|
-
def
|
|
221
|
-
|
|
221
|
+
def _doctor_checklist() -> tuple[list[dict], Callable]:
|
|
222
|
+
"""A checks list and its recorder, which refuses a blocker it cannot explain.
|
|
223
|
+
|
|
224
|
+
`resolve_blocker` with an empty `detail` is unfalsifiable: an agent is told to
|
|
225
|
+
fix something and given nothing to fix. It re-runs doctor, reads the identical
|
|
226
|
+
output, and loops. Enforcing the detail here means a check added later inherits
|
|
227
|
+
the guarantee instead of relying on its author to remember.
|
|
228
|
+
"""
|
|
222
229
|
checks: list[dict] = []
|
|
223
230
|
|
|
224
231
|
def check(name, ok, detail="", project=None):
|
|
232
|
+
if not ok and not str(detail).strip():
|
|
233
|
+
raise AssertionError(
|
|
234
|
+
f"doctor check {name!r} would fail silently: a failing check must"
|
|
235
|
+
" name what to fix"
|
|
236
|
+
)
|
|
225
237
|
entry = {"check": name, "ok": bool(ok), "detail": detail}
|
|
226
238
|
if project is not None:
|
|
227
239
|
entry["project"] = project
|
|
228
240
|
checks.append(entry)
|
|
229
241
|
|
|
242
|
+
return checks, check
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _blocks_the_loop(rel_path: str, cfg) -> bool:
|
|
246
|
+
"""Whether dirt at `rel_path` is somewhere a run would actually read it."""
|
|
247
|
+
if rel_path == config_mod.CONFIG_NAME:
|
|
248
|
+
return True
|
|
249
|
+
if cfg.owning_project(rel_path) is not None:
|
|
250
|
+
return True
|
|
251
|
+
return any(art.owns(rel_path) for art in cfg.artifacts.values())
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _cleanliness_detail(blocking: list[str], unrelated: list[str]) -> str:
|
|
255
|
+
def listed(paths: list[str]) -> str:
|
|
256
|
+
head = ", ".join(paths[:5])
|
|
257
|
+
return head if len(paths) <= 5 else f"{head} (+{len(paths) - 5} more)"
|
|
258
|
+
|
|
259
|
+
if blocking:
|
|
260
|
+
return (
|
|
261
|
+
f"uncommitted changes a run would observe: {listed(blocking)}."
|
|
262
|
+
" Commit, stash, or gitignore them before `tdd run start`."
|
|
263
|
+
)
|
|
264
|
+
if unrelated:
|
|
265
|
+
return (
|
|
266
|
+
f"clean where a run reads; {len(unrelated)} unrelated path(s) left as-is:"
|
|
267
|
+
f" {listed(unrelated)}"
|
|
268
|
+
)
|
|
269
|
+
return ""
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def cmd_doctor(args) -> Envelope:
|
|
273
|
+
worktree = _worktree()
|
|
274
|
+
checks, check = _doctor_checklist()
|
|
275
|
+
|
|
230
276
|
check("worktree resolvable", True, str(worktree))
|
|
231
277
|
try:
|
|
232
278
|
cfg = config_mod.load(worktree)
|
|
@@ -246,13 +292,20 @@ def cmd_doctor(args) -> Envelope:
|
|
|
246
292
|
root = worktree / project.root
|
|
247
293
|
check("root exists", root.is_dir(), str(root), project=name)
|
|
248
294
|
check("adapter known", project.adapter in adapters.available(), project.adapter, project=name)
|
|
249
|
-
|
|
295
|
+
declared = bool(project.test_paths)
|
|
296
|
+
check(
|
|
297
|
+
"test_paths declared", declared,
|
|
298
|
+
"" if declared
|
|
299
|
+
else f"add `test_paths` to [project.{name}] in tdd.toml — without it no"
|
|
300
|
+
" suite can be discovered for this project",
|
|
301
|
+
project=name,
|
|
302
|
+
)
|
|
250
303
|
if project.adapter == "pytest":
|
|
251
304
|
# The probe runs in the project's own environment (uv, poetry, pipenv,
|
|
252
305
|
# pdm or the active venv) — hardcoding `uv run` here failed the check
|
|
253
306
|
# on any non-uv project even with the plugin installed.
|
|
254
307
|
probe = adapters.build(project, worktree).plugin_probe_cmd()
|
|
255
|
-
code, out, err = adapters.base.run_command(probe, root)
|
|
308
|
+
code, out, err = adapters.base.run_command(probe, root, label="doctor")
|
|
256
309
|
check("pytest-json-report installed", code == 0, (err or "")[:200], project=name)
|
|
257
310
|
|
|
258
311
|
# Run before `collectable()` so this actionable message wins over
|
|
@@ -289,11 +342,33 @@ def cmd_doctor(args) -> Envelope:
|
|
|
289
342
|
projects[name] = {"ok": all(c["ok"] for c in checks[before:])}
|
|
290
343
|
|
|
291
344
|
for art in cfg.artifacts.values():
|
|
292
|
-
|
|
345
|
+
# One evaluation feeds both `ok` and the detail: evaluating the condition
|
|
346
|
+
# twice lets them disagree, and a passing check that still says "add a hook"
|
|
347
|
+
# is the same misdirection as a failing check that says nothing.
|
|
348
|
+
has_hook = bool(art.check or art.regenerate)
|
|
349
|
+
check(
|
|
350
|
+
f"artifact {art.name}: has check or regenerate", has_hook,
|
|
351
|
+
"" if has_hook
|
|
352
|
+
else f"add `check` or `regenerate` to [artifact.{art.name}] in tdd.toml —"
|
|
353
|
+
" freshness cannot be verified without one",
|
|
354
|
+
)
|
|
293
355
|
|
|
294
356
|
stale = _legacy_artifacts(worktree)
|
|
295
|
-
check(
|
|
296
|
-
|
|
357
|
+
check(
|
|
358
|
+
"no legacy state artifacts", not stale,
|
|
359
|
+
f"delete these pre-ledger state files: {', '.join(str(s) for s in stale[:5])}"
|
|
360
|
+
if stale else "",
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
# Only dirt a run would read can corrupt one. Blocking on everything else
|
|
364
|
+
# stopped agents on unrelated notes and editor settings, and — because the
|
|
365
|
+
# check named no path — gave them nothing to act on but a re-run. `is_ignored`
|
|
366
|
+
# also excludes doctor's own probe residue (`.venv`, `node_modules`, caches),
|
|
367
|
+
# so running doctor can no longer be what makes doctor fail.
|
|
368
|
+
dirt = sorted(p for p in gitutil.dirty_paths(worktree) if not cfg.is_ignored(p))
|
|
369
|
+
blocking = [p for p in dirt if _blocks_the_loop(p, cfg)]
|
|
370
|
+
unrelated = [p for p in dirt if p not in set(blocking)]
|
|
371
|
+
check("worktree clean", not blocking, _cleanliness_detail(blocking, unrelated))
|
|
297
372
|
|
|
298
373
|
ok = all(c["ok"] for c in checks)
|
|
299
374
|
return Envelope(
|
|
@@ -364,12 +439,18 @@ def _probe_projects(cfg, worktree, ledger, on_progress):
|
|
|
364
439
|
for done, (name, project) in enumerate(cfg.projects.items(), start=1):
|
|
365
440
|
adapter = adapters.build(project, worktree)
|
|
366
441
|
started = time.monotonic()
|
|
367
|
-
verdict
|
|
442
|
+
verdict = adapter.run(None)
|
|
443
|
+
ran = time.monotonic()
|
|
444
|
+
collection = adapter.collect()
|
|
368
445
|
elapsed = time.monotonic() - started
|
|
369
446
|
probes[name] = (verdict, collection)
|
|
447
|
+
# Split, not just totalled: `run` and `collect` have unrelated cost models
|
|
448
|
+
# — one scales with tests, the other with files — and a single number sends
|
|
449
|
+
# whoever asks "why was that slow?" out of the tool to measure by hand.
|
|
370
450
|
heartbeat(
|
|
371
451
|
event="baseline_captured", project=name,
|
|
372
452
|
test_count=len(collection.tests), elapsed_s=round(elapsed, 2),
|
|
453
|
+
run_s=round(ran - started, 2), collect_s=round(elapsed - (ran - started), 2),
|
|
373
454
|
)
|
|
374
455
|
on_progress(done, name)
|
|
375
456
|
return probes
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""Every blocker `tdd doctor` emits must name what to fix.
|
|
2
|
+
|
|
3
|
+
The incident: doctor returned `resolve_blocker` / "Resolve the failing checks above"
|
|
4
|
+
with a single failing check — `{"check": "worktree clean", "ok": false, "detail": ""}`.
|
|
5
|
+
The dirt was an unrelated planning note and a `.claude/settings.json` edit. The agent
|
|
6
|
+
could not see either, guessed, re-ran doctor, and got the identical opaque failure.
|
|
7
|
+
|
|
8
|
+
Two invariants close that loop:
|
|
9
|
+
|
|
10
|
+
1. A failing check always carries a detail. `check()` raises rather than record an
|
|
11
|
+
unfalsifiable blocker, so the class cannot reappear in a check added later.
|
|
12
|
+
2. Only dirt the loop would actually observe blocks — a declared project root, a
|
|
13
|
+
declared artifact, or `tdd.toml`. Everything else is reported, not enforced
|
|
14
|
+
(the PRD's "tree clean enough", §8.1).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from conftest import git, run_cli
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _check(out: dict, name: str) -> dict:
|
|
23
|
+
matches = [c for c in out["result"]["checks"] if c["check"] == name]
|
|
24
|
+
assert len(matches) == 1, out["result"]["checks"]
|
|
25
|
+
return matches[0]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def test_dirt_inside_a_project_root_blocks_and_names_the_path(repo):
|
|
29
|
+
(repo / "backend" / "app" / "scratch.py").write_text("x = 1\n")
|
|
30
|
+
|
|
31
|
+
clean = _check(run_cli(repo, "doctor"), "worktree clean")
|
|
32
|
+
assert clean["ok"] is False
|
|
33
|
+
assert "backend/app/scratch.py" in clean["detail"]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_a_modified_tdd_toml_blocks(repo):
|
|
37
|
+
(repo / "tdd.toml").write_text((repo / "tdd.toml").read_text() + "\n")
|
|
38
|
+
|
|
39
|
+
clean = _check(run_cli(repo, "doctor"), "worktree clean")
|
|
40
|
+
assert clean["ok"] is False
|
|
41
|
+
assert "tdd.toml" in clean["detail"]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_dirt_outside_every_declared_root_does_not_block(repo):
|
|
45
|
+
"""The incident's own shape: a planning note and an editor setting. Neither is
|
|
46
|
+
readable by any adapter, so neither can corrupt a baseline."""
|
|
47
|
+
(repo / "tasks").mkdir()
|
|
48
|
+
(repo / "tasks" / "fix-406-proxy-ws-forwarding.md").write_text("# plan\n")
|
|
49
|
+
(repo / ".claude").mkdir()
|
|
50
|
+
(repo / ".claude" / "settings.json").write_text("{}\n")
|
|
51
|
+
|
|
52
|
+
clean = _check(run_cli(repo, "doctor"), "worktree clean")
|
|
53
|
+
assert clean["ok"] is True, clean
|
|
54
|
+
# Reported, not enforced — a human still wants to know the tree is not pristine.
|
|
55
|
+
assert "tasks/fix-406-proxy-ws-forwarding.md" in clean["detail"]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_doctors_own_probe_residue_does_not_block(repo):
|
|
59
|
+
"""Doctor shells out to `uv run` and `vitest list`, which write `.venv`,
|
|
60
|
+
`node_modules` and cache directories into the tree it then grades. Without this,
|
|
61
|
+
running doctor is what makes doctor fail — and running it again re-does it."""
|
|
62
|
+
(repo / "backend" / ".venv" / "bin").mkdir(parents=True)
|
|
63
|
+
(repo / "backend" / ".venv" / "bin" / "python").write_text("")
|
|
64
|
+
(repo / "backend" / "__pycache__").mkdir()
|
|
65
|
+
(repo / "backend" / "__pycache__" / "app.pyc").write_text("")
|
|
66
|
+
|
|
67
|
+
clean = _check(run_cli(repo, "doctor"), "worktree clean")
|
|
68
|
+
assert clean["ok"] is True, clean
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_blocking_dirt_is_reported_even_when_unrelated_dirt_exists(repo):
|
|
72
|
+
(repo / "backend" / "app" / "scratch.py").write_text("x = 1\n")
|
|
73
|
+
(repo / "notes.md").write_text("# notes\n")
|
|
74
|
+
|
|
75
|
+
clean = _check(run_cli(repo, "doctor"), "worktree clean")
|
|
76
|
+
assert clean["ok"] is False
|
|
77
|
+
assert "backend/app/scratch.py" in clean["detail"]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def test_no_failing_check_is_ever_emitted_without_a_detail(repo_broken):
|
|
81
|
+
"""The structural half. `repo_broken` fails collection; dirty the tree and break
|
|
82
|
+
the artifact wiring too, so several failure paths run in one invocation."""
|
|
83
|
+
(repo_broken / "verify" / "tests" / "test_extra.py").write_text("\n")
|
|
84
|
+
(repo_broken / "tdd.toml").write_text(
|
|
85
|
+
(repo_broken / "tdd.toml").read_text()
|
|
86
|
+
+ '\n[artifact.openapi]\npath = "openapi.json"\nproduced_by = "backend"\n'
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
out = run_cli(repo_broken, "doctor")
|
|
90
|
+
failing = [c for c in out["result"]["checks"] if not c["ok"]]
|
|
91
|
+
assert failing, out["result"]["checks"]
|
|
92
|
+
for c in failing:
|
|
93
|
+
assert c["detail"].strip(), c
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def test_a_check_cannot_fail_without_a_detail():
|
|
97
|
+
"""Enforced at the helper, so a check added later inherits the guarantee rather
|
|
98
|
+
than relying on its author to remember."""
|
|
99
|
+
import pytest
|
|
100
|
+
|
|
101
|
+
from tddcli.cli import _doctor_checklist
|
|
102
|
+
|
|
103
|
+
checks, check = _doctor_checklist()
|
|
104
|
+
check("fine", False, "here is what to do")
|
|
105
|
+
with pytest.raises(AssertionError, match="silent"):
|
|
106
|
+
check("silent", False)
|
|
107
|
+
assert [c["check"] for c in checks] == ["fine"]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _with_artifact(repo, hook: str = 'regenerate = "true"') -> None:
|
|
111
|
+
(repo / "schema").mkdir(exist_ok=True)
|
|
112
|
+
(repo / "schema" / "openapi.json").write_text("{}\n")
|
|
113
|
+
(repo / "tdd.toml").write_text(
|
|
114
|
+
(repo / "tdd.toml").read_text()
|
|
115
|
+
+ f'\n[artifact.openapi]\npath = "schema/openapi.json"\n'
|
|
116
|
+
f'produced_by = "backend"\n{hook}\n'
|
|
117
|
+
)
|
|
118
|
+
git(repo, "add", "-A")
|
|
119
|
+
git(repo, "commit", "-q", "-m", "declare openapi artifact")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_dirt_at_a_declared_artifact_path_blocks(repo):
|
|
123
|
+
"""An artifact sits outside every project root but is still read by a run —
|
|
124
|
+
`run start` verifies its freshness. Uncommitted drift there is exactly the
|
|
125
|
+
staleness the artifact edge exists to catch."""
|
|
126
|
+
_with_artifact(repo)
|
|
127
|
+
(repo / "schema" / "openapi.json").write_text('{"drift": true}\n')
|
|
128
|
+
|
|
129
|
+
clean = _check(run_cli(repo, "doctor"), "worktree clean")
|
|
130
|
+
assert clean["ok"] is False
|
|
131
|
+
assert "schema/openapi.json" in clean["detail"]
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def test_an_artifact_needs_only_one_of_check_or_regenerate(repo):
|
|
135
|
+
"""Either hook alone makes freshness verifiable — requiring both would fail
|
|
136
|
+
every artifact that only knows how to rebuild itself."""
|
|
137
|
+
_with_artifact(repo, hook='regenerate = "true"')
|
|
138
|
+
|
|
139
|
+
assert _check(run_cli(repo, "doctor"), "artifact openapi: has check or regenerate")["ok"] is True
|
|
140
|
+
|
|
141
|
+
_check_only = repo / "tdd.toml"
|
|
142
|
+
_check_only.write_text(_check_only.read_text().replace('regenerate = "true"', 'check = "true"'))
|
|
143
|
+
git(repo, "add", "-A")
|
|
144
|
+
git(repo, "commit", "-q", "-m", "swap regenerate for check")
|
|
145
|
+
|
|
146
|
+
assert _check(run_cli(repo, "doctor"), "artifact openapi: has check or regenerate")["ok"] is True
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def test_a_long_dirt_list_is_truncated_but_the_fifth_path_is_not(repo):
|
|
150
|
+
"""The detail is read by an agent, so it stays bounded — but the boundary must
|
|
151
|
+
not eat a path it had room for."""
|
|
152
|
+
for i in range(5):
|
|
153
|
+
(repo / "backend" / "app" / f"f{i}.py").write_text("x = 1\n")
|
|
154
|
+
|
|
155
|
+
detail = _check(run_cli(repo, "doctor"), "worktree clean")["detail"]
|
|
156
|
+
assert "more)" not in detail, detail
|
|
157
|
+
assert "backend/app/f4.py" in detail
|
|
158
|
+
|
|
159
|
+
(repo / "backend" / "app" / "f5.py").write_text("x = 1\n")
|
|
160
|
+
detail = _check(run_cli(repo, "doctor"), "worktree clean")["detail"]
|
|
161
|
+
assert "(+1 more)" in detail, detail
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def test_test_paths_omission_says_what_to_add(repo):
|
|
165
|
+
(repo / "tdd.toml").write_text(
|
|
166
|
+
'[project.backend]\nroot = "backend"\nadapter = "pytest"\ntest_paths = []\n'
|
|
167
|
+
)
|
|
168
|
+
git(repo, "add", "-A")
|
|
169
|
+
git(repo, "commit", "-q", "-m", "drop test_paths")
|
|
170
|
+
|
|
171
|
+
declared = _check(run_cli(repo, "doctor"), "test_paths declared")
|
|
172
|
+
assert declared["ok"] is False
|
|
173
|
+
assert "test_paths" in declared["detail"]
|
|
@@ -46,7 +46,7 @@ def test_pytest_target_failure_keeps_the_error_at_the_tail(tmp_path, monkeypatch
|
|
|
46
46
|
adapter = _pytest_adapter(tmp_path)
|
|
47
47
|
longrepr = ("connector frame\n" * 300) + "ConnectionRefusedError: [Errno 61]"
|
|
48
48
|
|
|
49
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
49
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
50
50
|
marker = "--json-report-file="
|
|
51
51
|
path = command.split(marker, 1)[1].split(" --", 1)[0]
|
|
52
52
|
Path(path.strip("'\"")).write_text(json.dumps({
|
|
@@ -62,7 +62,7 @@ def test_only_reporting_flags_are_appended(tmp_path, monkeypatch):
|
|
|
62
62
|
adapter = adapters.build(project, tmp_path)
|
|
63
63
|
seen = {}
|
|
64
64
|
|
|
65
|
-
def fake_run(command, cwd, timeout=1800, extra_env=None):
|
|
65
|
+
def fake_run(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
66
66
|
seen["command"] = command
|
|
67
67
|
return 1, "", "no report"
|
|
68
68
|
|
|
@@ -54,7 +54,7 @@ def test_default_suite_runs_with_the_project_env_expanded(tmp_path, monkeypatch)
|
|
|
54
54
|
adapter = adapters.build(project, tmp_path)
|
|
55
55
|
seen: list = []
|
|
56
56
|
|
|
57
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
57
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
58
58
|
seen.append((command, extra_env))
|
|
59
59
|
marker = "--json-report-file="
|
|
60
60
|
path = command.split(marker, 1)[1].split(" --", 1)[0]
|
|
@@ -95,7 +95,7 @@ def test_pytest_collection_of_default_files_carries_the_project_env(
|
|
|
95
95
|
adapter = adapters.build(project, tmp_path)
|
|
96
96
|
seen: list = []
|
|
97
97
|
|
|
98
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
98
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
99
99
|
seen.append((command, extra_env))
|
|
100
100
|
return 0, "tests/test_a.py::test_a\n", ""
|
|
101
101
|
|
|
@@ -114,7 +114,7 @@ def test_vitest_default_suite_runs_with_the_project_env(tmp_path, monkeypatch):
|
|
|
114
114
|
adapter = adapters.build(project, tmp_path)
|
|
115
115
|
seen: list = []
|
|
116
116
|
|
|
117
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
117
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
118
118
|
seen.append((command, extra_env))
|
|
119
119
|
return 0, json.dumps({"testResults": []}), ""
|
|
120
120
|
|
|
@@ -110,7 +110,7 @@ def _fake_pytest_run(reports_by_prefix: dict[str, dict], seen: list):
|
|
|
110
110
|
"""A run_command double that answers each suite command with its own report,
|
|
111
111
|
keyed by command prefix, writing the JSON where the real plugin would."""
|
|
112
112
|
|
|
113
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
113
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
114
114
|
seen.append((command, extra_env))
|
|
115
115
|
for prefix, report in reports_by_prefix.items():
|
|
116
116
|
if command.startswith(prefix):
|
|
@@ -195,7 +195,7 @@ def test_pytest_broken_override_suite_is_a_loud_error_not_a_silent_gap(
|
|
|
195
195
|
project = project_with(tmp_path, 'test_command = "pytest tests"\n' + OVERRIDE_BLOCK)
|
|
196
196
|
adapter = adapters.build(project, tmp_path)
|
|
197
197
|
|
|
198
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
198
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
199
199
|
if command.startswith("pytest contract"):
|
|
200
200
|
return 4, "", "ERROR: file or directory not found: contract"
|
|
201
201
|
marker = "--json-report-file="
|
|
@@ -222,7 +222,7 @@ def test_pytest_collection_routes_override_files_to_the_override_command(
|
|
|
222
222
|
adapter = adapters.build(project, tmp_path)
|
|
223
223
|
seen: list = []
|
|
224
224
|
|
|
225
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
225
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
226
226
|
seen.append((command, extra_env))
|
|
227
227
|
name = "test_api.py::test_ping" if "contract" in command else "test_a.py::test_a"
|
|
228
228
|
return 0, name, ""
|
|
@@ -272,7 +272,7 @@ def test_vitest_run_finds_a_target_that_only_the_override_config_reaches(
|
|
|
272
272
|
project = project_with(tmp_path, VITEST_OVERRIDE, adapter="vitest")
|
|
273
273
|
adapter = adapters.build(project, tmp_path)
|
|
274
274
|
|
|
275
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
275
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
276
276
|
if "--config vitest.contract.config.ts" in command:
|
|
277
277
|
report = _vitest_report(
|
|
278
278
|
str(tmp_path / "backend" / "contract" / "api.contract.test.ts"),
|
|
@@ -331,7 +331,7 @@ def test_vitest_collection_routes_override_files_to_the_override_command(
|
|
|
331
331
|
adapter = adapters.build(project, tmp_path)
|
|
332
332
|
seen: list = []
|
|
333
333
|
|
|
334
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
334
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
335
335
|
seen.append(command)
|
|
336
336
|
return 0, "contract/api.contract.test.ts > pings the api", ""
|
|
337
337
|
|
|
@@ -479,7 +479,7 @@ def test_vitest_duplicate_test_id_across_suites_is_a_loud_error(
|
|
|
479
479
|
]
|
|
480
480
|
}
|
|
481
481
|
|
|
482
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
482
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
483
483
|
if "--config" in command:
|
|
484
484
|
return 0, json.dumps(result("passed")), ""
|
|
485
485
|
return 1, json.dumps(result("failed")), ""
|
|
@@ -500,7 +500,7 @@ def test_pytest_isolation_probe_flags_default_reach_into_override_files(
|
|
|
500
500
|
adapter = adapters.build(project, tmp_path)
|
|
501
501
|
seen: list = []
|
|
502
502
|
|
|
503
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
503
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
504
504
|
seen.append(command)
|
|
505
505
|
return 0, (
|
|
506
506
|
"tests/test_a.py::test_a\n"
|
|
@@ -528,7 +528,7 @@ def test_pytest_isolation_probe_passes_when_the_default_suite_is_scoped(
|
|
|
528
528
|
monkeypatch.setattr(
|
|
529
529
|
adapters.pytest_adapter,
|
|
530
530
|
"run_command",
|
|
531
|
-
lambda command, cwd, timeout=1800, extra_env=None: (
|
|
531
|
+
lambda command, cwd, timeout=1800, extra_env=None, label=None: (
|
|
532
532
|
0, "tests/test_a.py::test_a\n", ""
|
|
533
533
|
),
|
|
534
534
|
)
|
|
@@ -550,7 +550,7 @@ def test_vitest_isolation_probe_flags_default_reach_into_override_files(
|
|
|
550
550
|
adapter = adapters.build(project, tmp_path)
|
|
551
551
|
seen: list = []
|
|
552
552
|
|
|
553
|
-
def fake(command, cwd, timeout=1800, extra_env=None):
|
|
553
|
+
def fake(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
554
554
|
seen.append(command)
|
|
555
555
|
return 0, (
|
|
556
556
|
"src/__tests__/a.test.ts > adds\n"
|
|
@@ -570,7 +570,7 @@ def test_isolation_probe_is_free_when_a_project_declares_no_overrides(
|
|
|
570
570
|
project = project_with(tmp_path, 'test_command = "pytest tests"\n')
|
|
571
571
|
adapter = adapters.build(project, tmp_path)
|
|
572
572
|
|
|
573
|
-
def explode(command, cwd, timeout=1800, extra_env=None):
|
|
573
|
+
def explode(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
574
574
|
raise AssertionError("no probe should run without overrides")
|
|
575
575
|
|
|
576
576
|
monkeypatch.setattr(adapters.pytest_adapter, "run_command", explode)
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Where a slow command actually spent its time.
|
|
2
|
+
|
|
3
|
+
A `run start` took 23 minutes and reported one number per project — `elapsed_s`,
|
|
4
|
+
covering `run()` and `collect()` together. Answering "which of the two?" required
|
|
5
|
+
measuring the projects by hand, outside the tool, in a different checkout; the
|
|
6
|
+
warm numbers that produced did not match the cold ones and the first diagnosis
|
|
7
|
+
drawn from them was wrong.
|
|
8
|
+
|
|
9
|
+
The tool already holds both timings at the moment it discards them. Two seams
|
|
10
|
+
make them visible:
|
|
11
|
+
|
|
12
|
+
* `baseline_captured` reports `run_s` and `collect_s` separately, so the split is
|
|
13
|
+
in the output the agent already prints.
|
|
14
|
+
* `run_command` — the single choke point every subprocess passes through — emits a
|
|
15
|
+
`command_timing` line under `TDD_TIMING=1`, which attributes cost per command:
|
|
16
|
+
per-file collection, gates, doctor probes, artifact hooks. Off by default,
|
|
17
|
+
because the per-file loop would otherwise emit one line per test file on every
|
|
18
|
+
invocation.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import json
|
|
24
|
+
|
|
25
|
+
from conftest import run_cli, write_plan
|
|
26
|
+
from tddcli.adapters import base
|
|
27
|
+
|
|
28
|
+
PLAN = """---
|
|
29
|
+
cycles:
|
|
30
|
+
- n: 1
|
|
31
|
+
project: backend
|
|
32
|
+
title: "adding two numbers"
|
|
33
|
+
test: "tests/test_add.py::test_add_two_numbers"
|
|
34
|
+
stub_expected: ["app/calc.py"]
|
|
35
|
+
commit_red: "test: adding two numbers"
|
|
36
|
+
commit_green: "feat: add()"
|
|
37
|
+
---
|
|
38
|
+
|
|
39
|
+
# Plan
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _lines(stderr: str, event: str) -> list[dict]:
|
|
44
|
+
found = []
|
|
45
|
+
for line in stderr.splitlines():
|
|
46
|
+
try:
|
|
47
|
+
payload = json.loads(line.strip())
|
|
48
|
+
except json.JSONDecodeError:
|
|
49
|
+
continue
|
|
50
|
+
if payload.get("event") == event:
|
|
51
|
+
found.append(payload)
|
|
52
|
+
return found
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _start(repo):
|
|
56
|
+
plan = write_plan(repo, PLAN)
|
|
57
|
+
assert run_cli(repo, "plan", "register", plan)["ok"]
|
|
58
|
+
return run_cli(repo, "run", "start", "--plan", plan)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def test_baseline_reports_run_and_collect_separately(repo, capsys):
|
|
62
|
+
out = _start(repo)
|
|
63
|
+
assert out["ok"], out
|
|
64
|
+
|
|
65
|
+
backend = next(
|
|
66
|
+
line for line in _lines(capsys.readouterr().err, "baseline_captured")
|
|
67
|
+
if line["project"] == "backend"
|
|
68
|
+
)
|
|
69
|
+
assert isinstance(backend["run_s"], (int, float)), backend
|
|
70
|
+
assert isinstance(backend["collect_s"], (int, float)), backend
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def test_the_split_still_adds_up_to_the_reported_total(repo, capsys):
|
|
74
|
+
"""`elapsed_s` stays, so existing consumers keep working — and it must remain
|
|
75
|
+
the sum of the parts, or the split is describing a different thing."""
|
|
76
|
+
out = _start(repo)
|
|
77
|
+
assert out["ok"], out
|
|
78
|
+
|
|
79
|
+
backend = next(
|
|
80
|
+
line for line in _lines(capsys.readouterr().err, "baseline_captured")
|
|
81
|
+
if line["project"] == "backend"
|
|
82
|
+
)
|
|
83
|
+
assert backend["run_s"] + backend["collect_s"] <= backend["elapsed_s"] + 0.05
|
|
84
|
+
assert backend["elapsed_s"] - (backend["run_s"] + backend["collect_s"]) < 0.5
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def test_command_timing_is_silent_unless_asked_for(repo, capsys, monkeypatch):
|
|
88
|
+
monkeypatch.delenv(base.TIMING_ENV, raising=False)
|
|
89
|
+
assert _start(repo)["ok"]
|
|
90
|
+
assert _lines(capsys.readouterr().err, "command_timing") == []
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_command_timing_names_the_command_and_its_cost(repo, capsys, monkeypatch):
|
|
94
|
+
monkeypatch.setenv(base.TIMING_ENV, "1")
|
|
95
|
+
assert _start(repo)["ok"]
|
|
96
|
+
|
|
97
|
+
timings = _lines(capsys.readouterr().err, "command_timing")
|
|
98
|
+
assert timings, "TDD_TIMING=1 produced no command_timing lines"
|
|
99
|
+
for entry in timings:
|
|
100
|
+
assert isinstance(entry["duration_ms"], int)
|
|
101
|
+
assert entry["command"]
|
|
102
|
+
assert isinstance(entry["exit_code"], int)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def test_a_timed_command_is_attributed_to_its_caller(repo, capsys, monkeypatch):
|
|
106
|
+
"""Without a label the rows are readable but not groupable: `run_command` sees
|
|
107
|
+
a command string and a cwd, not which project or phase asked for it."""
|
|
108
|
+
monkeypatch.setenv(base.TIMING_ENV, "1")
|
|
109
|
+
assert _start(repo)["ok"]
|
|
110
|
+
|
|
111
|
+
timings = _lines(capsys.readouterr().err, "command_timing")
|
|
112
|
+
labels = {entry.get("label") for entry in timings}
|
|
113
|
+
assert "suite" in labels, labels
|
|
114
|
+
assert "collect" in labels, labels
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def test_the_per_file_collect_loop_is_attributed_per_file(repo, capsys, monkeypatch):
|
|
118
|
+
"""The loop's cost is per file, so its timing has to be too — a single total
|
|
119
|
+
cannot say which file is slow."""
|
|
120
|
+
monkeypatch.setenv(base.TIMING_ENV, "1")
|
|
121
|
+
assert _start(repo)["ok"]
|
|
122
|
+
|
|
123
|
+
collects = [
|
|
124
|
+
entry for entry in _lines(capsys.readouterr().err, "command_timing")
|
|
125
|
+
if entry.get("label") == "collect"
|
|
126
|
+
]
|
|
127
|
+
assert collects, "no collect timings"
|
|
128
|
+
assert any("test_smoke.py" in entry["command"] for entry in collects), collects
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def test_doctor_probes_are_labelled_as_doctor(repo, capsys, monkeypatch):
|
|
132
|
+
"""Doctor's probes reach `run_command` from three places — the reporter check
|
|
133
|
+
in `cmd_doctor`, `collectable()` and `override_isolation()` — and both adapter
|
|
134
|
+
methods are called from nowhere else. Unlabelled they arrive as `label: null`,
|
|
135
|
+
indistinguishable from a third-party adapter's unlabelled call."""
|
|
136
|
+
monkeypatch.setenv(base.TIMING_ENV, "1")
|
|
137
|
+
assert run_cli(repo, "doctor")["ok"]
|
|
138
|
+
|
|
139
|
+
timings = _lines(capsys.readouterr().err, "command_timing")
|
|
140
|
+
assert timings, "TDD_TIMING=1 produced no command_timing lines for doctor"
|
|
141
|
+
assert {entry.get("label") for entry in timings} == {"doctor"}, timings
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def test_timing_does_not_disturb_the_command_result(repo, monkeypatch):
|
|
145
|
+
"""The wrapper returns exactly what the subprocess returned."""
|
|
146
|
+
monkeypatch.setenv(base.TIMING_ENV, "1")
|
|
147
|
+
code, out, err = base.run_command("echo hello", repo)
|
|
148
|
+
assert code == 0
|
|
149
|
+
assert out.strip() == "hello"
|
|
@@ -127,7 +127,7 @@ def test_live_foreign_lease_counts(lease_dir):
|
|
|
127
127
|
def recorded(monkeypatch):
|
|
128
128
|
calls: list[dict] = []
|
|
129
129
|
|
|
130
|
-
def stub(command, cwd, timeout=1800, extra_env=None):
|
|
130
|
+
def stub(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
131
131
|
calls.append({"command": command, "cwd": cwd, "extra_env": extra_env})
|
|
132
132
|
return 1, "", ""
|
|
133
133
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|