tdd-cli 0.3.0__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/CHANGELOG.md +45 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/PKG-INFO +2 -2
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/pyproject.toml +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/adapters/base.py +71 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/adapters/pytest_adapter.py +42 -13
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/adapters/vitest_adapter.py +42 -8
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/cli.py +89 -8
- tdd_cli-0.4.1/tests/test_batch_collection.py +233 -0
- tdd_cli-0.4.1/tests/test_doctor_blockers.py +173 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_failure_clipping.py +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_project_commands.py +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_project_env.py +3 -3
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_suite_overrides.py +17 -11
- tdd_cli-0.4.1/tests/test_timing_visibility.py +158 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_worker_leases.py +1 -1
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/.gitignore +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/LICENSE +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/README.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/SECURITY.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/plan.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/advance.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/config.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/contract.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/identity.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/ledger.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/machine.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/render.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/src/tddcli/staging.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/conftest.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_artifact_regeneration.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_baseline_integrity.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_config_and_staging.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_contract.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_fleet.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_heartbeat.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_progress.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_release_surface.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_snapshot_and_identity.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.3.0 → tdd_cli-0.4.1}/tests/test_vitest_adapter.py +0 -0
|
@@ -4,6 +4,51 @@ All notable changes to this project are documented here.
|
|
|
4
4
|
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
5
5
|
and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [0.4.1] - 2026-08-16
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- `collect()` runs one invocation per declared suite instead of one per test
|
|
12
|
+
file, falling back to the per-file loop for anything a batch did not account
|
|
13
|
+
for. Per-file collection was **77% of a real `run start`** — 313 subprocesses
|
|
14
|
+
costing 402s, against 117s to actually run every test — because cost scaled
|
|
15
|
+
with file count at a ~1.08s floor per invocation (the environment manager
|
|
16
|
+
resolving plus the runner booting). Measured 38.8x faster on a 60-file
|
|
17
|
+
project, with an identical collected set. R10.3's guarantee is unchanged: a
|
|
18
|
+
file that fails to collect is still attributed to itself and cannot destroy
|
|
19
|
+
the set, and a file the batch never reports is still collected individually,
|
|
20
|
+
so the set can only match or improve on the old one (#27).
|
|
21
|
+
|
|
22
|
+
## [0.4.0] - 2026-08-16
|
|
23
|
+
|
|
24
|
+
### Added
|
|
25
|
+
|
|
26
|
+
- `baseline_captured` reports `run_s` and `collect_s` alongside `elapsed_s`. The
|
|
27
|
+
suite run and the per-file collection have unrelated cost models — one scales
|
|
28
|
+
with tests, the other with files — so a single total could not say which was
|
|
29
|
+
slow, and answering that meant measuring projects by hand outside the tool.
|
|
30
|
+
- `TDD_TIMING=1` emits a `command_timing` line per subprocess on stderr
|
|
31
|
+
(`label`, `command`, `cwd`, `duration_ms`, `exit_code`), covering every
|
|
32
|
+
subprocess the tool spawns: suite runs, per-file collection, lint/typecheck
|
|
33
|
+
gates, doctor probes and artifact hooks. Off by default — the per-file loop
|
|
34
|
+
would otherwise emit one line per test file on every invocation. `label` is
|
|
35
|
+
one of `suite`, `collect`, `gate`, `doctor`; an unlabelled row comes from a
|
|
36
|
+
third-party adapter, since every built-in call site names itself (R8.4).
|
|
37
|
+
|
|
38
|
+
### Fixed
|
|
39
|
+
|
|
40
|
+
- `tdd doctor` no longer emits a blocker it cannot explain. Every failing check
|
|
41
|
+
now carries a `detail` naming what to fix, enforced by the checklist recorder
|
|
42
|
+
so a check added later inherits the guarantee. Previously `worktree clean`
|
|
43
|
+
failed with `detail: ""`, leaving an agent with `resolve_blocker` and nothing
|
|
44
|
+
to resolve — it re-ran doctor and read the identical output.
|
|
45
|
+
- `worktree clean` is scoped to dirt a run would actually read: a declared
|
|
46
|
+
project root, a declared artifact path, or `tdd.toml`. Build residue is
|
|
47
|
+
excluded via `config.is_ignored`, so doctor's own `uv run` / `vitest list`
|
|
48
|
+
probes (`.venv`, `node_modules`, caches) can no longer be what makes doctor
|
|
49
|
+
fail. Unrelated dirt is reported in the passing check's `detail` rather than
|
|
50
|
+
blocking the run.
|
|
51
|
+
|
|
7
52
|
## [0.3.0] - 2026-08-10
|
|
8
53
|
|
|
9
54
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -47,7 +47,7 @@ packages = ["src/tddcli"]
|
|
|
47
47
|
include = ["src", "tests", "examples", "README.md", "LICENSE", "CHANGELOG.md", "SECURITY.md"]
|
|
48
48
|
|
|
49
49
|
[dependency-groups]
|
|
50
|
-
dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5"]
|
|
50
|
+
dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5", "zizmor>=1.29"]
|
|
51
51
|
|
|
52
52
|
[tool.pytest.ini_options]
|
|
53
53
|
testpaths = ["tests"]
|
|
@@ -4,11 +4,13 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import os
|
|
6
6
|
import subprocess
|
|
7
|
+
import time
|
|
7
8
|
from collections import Counter
|
|
8
9
|
from dataclasses import dataclass, field
|
|
9
10
|
from pathlib import Path
|
|
10
11
|
|
|
11
12
|
from .. import leases
|
|
13
|
+
from ..envelope import heartbeat
|
|
12
14
|
|
|
13
15
|
NOT_FOUND = "not_found"
|
|
14
16
|
NOT_COLLECTED = "not_collected"
|
|
@@ -78,10 +80,26 @@ def clip_failure(text: str, limit: int = 1500) -> str:
|
|
|
78
80
|
return f"{text[:head]}\n… [clipped] …\n{text[-tail:]}"
|
|
79
81
|
|
|
80
82
|
|
|
83
|
+
#: Opt-in per-command timing. Off by default: the per-file collect loop would emit
|
|
84
|
+
#: one line per test file on every invocation, drowning the heartbeats that exist
|
|
85
|
+
#: to make a slow baseline legible.
|
|
86
|
+
TIMING_ENV = "TDD_TIMING"
|
|
87
|
+
|
|
88
|
+
|
|
81
89
|
def run_command(
|
|
82
90
|
command: str, cwd: Path, timeout: int = 1800,
|
|
83
91
|
extra_env: dict[str, str] | None = None,
|
|
92
|
+
label: str | None = None,
|
|
84
93
|
) -> tuple[int, str, str]:
|
|
94
|
+
"""Every subprocess the tool spawns passes through here, which makes it the one
|
|
95
|
+
place worth timing: suite runs, per-file collection, lint and typecheck gates,
|
|
96
|
+
doctor probes, artifact hooks.
|
|
97
|
+
|
|
98
|
+
`label` is what makes the rows groupable. This function sees a command string
|
|
99
|
+
and a cwd — not which project or phase asked for it — so an unlabelled timing
|
|
100
|
+
is readable by a human and useless to a query.
|
|
101
|
+
"""
|
|
102
|
+
started = time.monotonic()
|
|
85
103
|
proc = subprocess.run(
|
|
86
104
|
command,
|
|
87
105
|
shell=True,
|
|
@@ -91,6 +109,15 @@ def run_command(
|
|
|
91
109
|
timeout=timeout,
|
|
92
110
|
env=None if extra_env is None else {**os.environ, **extra_env},
|
|
93
111
|
)
|
|
112
|
+
if os.environ.get(TIMING_ENV):
|
|
113
|
+
heartbeat(
|
|
114
|
+
event="command_timing",
|
|
115
|
+
label=label,
|
|
116
|
+
command=command,
|
|
117
|
+
cwd=str(cwd),
|
|
118
|
+
duration_ms=int((time.monotonic() - started) * 1000),
|
|
119
|
+
exit_code=proc.returncode,
|
|
120
|
+
)
|
|
94
121
|
return proc.returncode, proc.stdout, proc.stderr
|
|
95
122
|
|
|
96
123
|
|
|
@@ -130,6 +157,7 @@ class Adapter:
|
|
|
130
157
|
command.replace("{workers}", str(workers)),
|
|
131
158
|
self.root,
|
|
132
159
|
extra_env={"TDD_WORKERS": str(workers), **(extra_env or {})},
|
|
160
|
+
label="suite",
|
|
133
161
|
)
|
|
134
162
|
|
|
135
163
|
def _test_cmd(self) -> str:
|
|
@@ -161,6 +189,48 @@ class Adapter:
|
|
|
161
189
|
return "a body that fails loudly, never working logic"
|
|
162
190
|
|
|
163
191
|
def collect(self) -> Collection:
|
|
192
|
+
"""Enumerate the project's tests: one invocation per declared suite, with
|
|
193
|
+
the per-file loop kept for what that cannot account for (issue #27).
|
|
194
|
+
|
|
195
|
+
Per file, collection cost scaled with *file count* rather than test count —
|
|
196
|
+
measured at 77% of a whole `run start`, against a floor of ~1.08s per
|
|
197
|
+
invocation that is the environment manager resolving plus the runner
|
|
198
|
+
booting, not collection work.
|
|
199
|
+
|
|
200
|
+
The batch and the loop do not discover the same way: a batch uses the
|
|
201
|
+
runner's own config, the loop walks `test_paths`. So the loop still runs for
|
|
202
|
+
every file the batch did not report — whether the batch failed, returned
|
|
203
|
+
nothing, or simply never mentioned that file. R10.3's guarantee is
|
|
204
|
+
unchanged: one uncollectable file is attributed to itself, and cannot
|
|
205
|
+
destroy the set. What changes is that the healthy case no longer pays for
|
|
206
|
+
the broken one.
|
|
207
|
+
"""
|
|
208
|
+
result = Collection()
|
|
209
|
+
unaccounted = {str(p.relative_to(self.root)) for p in self._test_files()}
|
|
210
|
+
for command, env in self._collect_invocations():
|
|
211
|
+
batch = self._collect_batch(command, env)
|
|
212
|
+
if batch is None:
|
|
213
|
+
continue
|
|
214
|
+
tests, files = batch
|
|
215
|
+
result.tests |= tests
|
|
216
|
+
unaccounted -= files
|
|
217
|
+
return self._collect_per_file(unaccounted, result)
|
|
218
|
+
|
|
219
|
+
def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
|
|
220
|
+
"""One collection command per declared suite: the default plus each
|
|
221
|
+
override's (R7.13), so an override's files are enumerated with its own
|
|
222
|
+
command and env."""
|
|
223
|
+
raise NotImplementedError
|
|
224
|
+
|
|
225
|
+
def _collect_batch(
|
|
226
|
+
self, command: str, env: dict[str, str] | None
|
|
227
|
+
) -> tuple[set[str], set[str]] | None:
|
|
228
|
+
"""`(test ids, files seen)` for one whole-suite invocation, or None when it
|
|
229
|
+
produced nothing usable — the signal to leave its files to the loop."""
|
|
230
|
+
raise NotImplementedError
|
|
231
|
+
|
|
232
|
+
def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
|
|
233
|
+
"""R10.3's loop, now reached only for files no batch accounted for."""
|
|
164
234
|
raise NotImplementedError
|
|
165
235
|
|
|
166
236
|
def collectable(self) -> GateResult:
|
|
@@ -184,7 +254,7 @@ class Adapter:
|
|
|
184
254
|
def _gate(self, commands: list[str]) -> GateResult:
|
|
185
255
|
chunks = []
|
|
186
256
|
for cmd in commands:
|
|
187
|
-
code, out, err = run_command(cmd, self.root)
|
|
257
|
+
code, out, err = run_command(cmd, self.root, label="gate")
|
|
188
258
|
if code != 0:
|
|
189
259
|
chunks.append(f"$ {cmd}\n{out}\n{err}".strip())
|
|
190
260
|
return GateResult(ok=not chunks, output="\n\n".join(chunks)[:4000])
|
|
@@ -199,7 +199,7 @@ class PytestAdapter(Adapter):
|
|
|
199
199
|
]
|
|
200
200
|
for cmd, env in probes:
|
|
201
201
|
code, out, err = run_command(
|
|
202
|
-
f"{cmd} --collect-only -q", self.root, extra_env=env
|
|
202
|
+
f"{cmd} --collect-only -q", self.root, extra_env=env, label="doctor"
|
|
203
203
|
)
|
|
204
204
|
if code != 0:
|
|
205
205
|
chunks.append(out.strip())
|
|
@@ -215,7 +215,9 @@ class PytestAdapter(Adapter):
|
|
|
215
215
|
if not self.project.overrides:
|
|
216
216
|
return GateResult(ok=True)
|
|
217
217
|
probe = f"{self._test_cmd().replace('{workers}', '0')} --collect-only -q"
|
|
218
|
-
code, out, err = run_command(
|
|
218
|
+
code, out, err = run_command(
|
|
219
|
+
probe, self.root, extra_env=self._suite_env(None), label="doctor"
|
|
220
|
+
)
|
|
219
221
|
reached = sorted({
|
|
220
222
|
f for f in (
|
|
221
223
|
line.split("::", 1)[0]
|
|
@@ -233,22 +235,49 @@ class PytestAdapter(Adapter):
|
|
|
233
235
|
" cannot reach them (e.g. `pytest tests/`)."
|
|
234
236
|
))
|
|
235
237
|
|
|
236
|
-
def
|
|
238
|
+
def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
|
|
239
|
+
"""An override without a `collect_command` collects with its `test_command`
|
|
240
|
+
— `--collect-only` composes with any run command."""
|
|
241
|
+
return [(self._collect_cmd(), self._suite_env(None))] + [
|
|
242
|
+
(ov.collect_command or ov.test_command, self._suite_env(ov))
|
|
243
|
+
for ov in self.project.overrides
|
|
244
|
+
]
|
|
245
|
+
|
|
246
|
+
def _nodeids(self, out: str) -> tuple[set[str], set[str]]:
|
|
247
|
+
tests: set[str] = set()
|
|
248
|
+
files: set[str] = set()
|
|
249
|
+
for line in out.splitlines():
|
|
250
|
+
line = line.strip()
|
|
251
|
+
if "::" in line and not line.startswith(("=", "-", "no tests")):
|
|
252
|
+
tests.add(self.qualify(line))
|
|
253
|
+
files.add(line.split("::", 1)[0])
|
|
254
|
+
return tests, files
|
|
255
|
+
|
|
256
|
+
def _collect_batch(
|
|
257
|
+
self, command: str, env: dict[str, str] | None
|
|
258
|
+
) -> tuple[set[str], set[str]] | None:
|
|
259
|
+
code, out, err = run_command(
|
|
260
|
+
f"{command} --collect-only -q", self.root, extra_env=env, label="collect"
|
|
261
|
+
)
|
|
262
|
+
if code != 0:
|
|
263
|
+
return None
|
|
264
|
+
# An empty result needs no special case: it accounts for no files, so every
|
|
265
|
+
# file falls to the loop exactly as a failure would.
|
|
266
|
+
return self._nodeids(out)
|
|
267
|
+
|
|
268
|
+
def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
|
|
237
269
|
"""Per file (R10.3) — one uncollectable module must not destroy the whole set."""
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
rel = path.relative_to(self.root)
|
|
241
|
-
base, env = self._collect_cmd_for(str(rel))
|
|
270
|
+
for rel in sorted(rels):
|
|
271
|
+
base, env = self._collect_cmd_for(rel)
|
|
242
272
|
code, out, err = run_command(
|
|
243
|
-
f"{base} --collect-only -q {shlex.quote(
|
|
273
|
+
f"{base} --collect-only -q {shlex.quote(rel)}",
|
|
244
274
|
self.root,
|
|
245
275
|
extra_env=env,
|
|
276
|
+
label="collect",
|
|
246
277
|
)
|
|
247
278
|
if code != 0:
|
|
248
|
-
result.failed_files[
|
|
279
|
+
result.failed_files[rel] = (err or out).strip()[:800]
|
|
249
280
|
continue
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
if "::" in line and not line.startswith(("=", "-", "no tests")):
|
|
253
|
-
result.tests.add(self.qualify(line))
|
|
281
|
+
tests, _ = self._nodeids(out)
|
|
282
|
+
result.tests |= tests
|
|
254
283
|
return result
|
|
@@ -177,7 +177,7 @@ class VitestAdapter(Adapter):
|
|
|
177
177
|
"""
|
|
178
178
|
chunks = []
|
|
179
179
|
code, out, err = run_command(
|
|
180
|
-
self._collect_cmd(), self.root, extra_env=self._suite_env(None)
|
|
180
|
+
self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
|
|
181
181
|
)
|
|
182
182
|
if code != 0:
|
|
183
183
|
chunks.append((err or out).strip())
|
|
@@ -190,7 +190,7 @@ class VitestAdapter(Adapter):
|
|
|
190
190
|
)
|
|
191
191
|
continue
|
|
192
192
|
code, out, err = run_command(
|
|
193
|
-
ov.collect_command, self.root, extra_env=self._suite_env(ov)
|
|
193
|
+
ov.collect_command, self.root, extra_env=self._suite_env(ov), label="doctor"
|
|
194
194
|
)
|
|
195
195
|
if code != 0:
|
|
196
196
|
chunks.append((err or out).strip())
|
|
@@ -204,7 +204,7 @@ class VitestAdapter(Adapter):
|
|
|
204
204
|
if not self.project.overrides:
|
|
205
205
|
return GateResult(ok=True)
|
|
206
206
|
code, out, err = run_command(
|
|
207
|
-
self._collect_cmd(), self.root, extra_env=self._suite_env(None)
|
|
207
|
+
self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
|
|
208
208
|
)
|
|
209
209
|
reached = sorted({
|
|
210
210
|
f for f in (
|
|
@@ -223,10 +223,43 @@ class VitestAdapter(Adapter):
|
|
|
223
223
|
" config (test.exclude) or scope its include globs."
|
|
224
224
|
))
|
|
225
225
|
|
|
226
|
-
def
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
226
|
+
def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
|
|
227
|
+
"""An override with no `collect_command` gets no batch — `vitest list` knows
|
|
228
|
+
nothing of its config, and its files fall to the loop, which records the
|
|
229
|
+
missing `collect_command` against each of them as before."""
|
|
230
|
+
return [(self._collect_cmd(), self._suite_env(None))] + [
|
|
231
|
+
(ov.collect_command, self._suite_env(ov))
|
|
232
|
+
for ov in self.project.overrides if ov.collect_command
|
|
233
|
+
]
|
|
234
|
+
|
|
235
|
+
def _collect_batch(
|
|
236
|
+
self, command: str, env: dict[str, str] | None
|
|
237
|
+
) -> tuple[set[str], set[str]] | None:
|
|
238
|
+
"""Unlike `_parse_list_output`, which pins every id to the one file it was
|
|
239
|
+
given, a whole-suite listing must read the file from each line — the same
|
|
240
|
+
`file > describe > name` shape, one file per line rather than one per run."""
|
|
241
|
+
code, out, err = run_command(command, self.root, extra_env=env, label="collect")
|
|
242
|
+
if code != 0:
|
|
243
|
+
return None
|
|
244
|
+
tests: set[str] = set()
|
|
245
|
+
files: set[str] = set()
|
|
246
|
+
for line in out.splitlines():
|
|
247
|
+
line = line.strip()
|
|
248
|
+
if " > " not in line:
|
|
249
|
+
continue
|
|
250
|
+
rel, _, remainder = line.partition(" > ")
|
|
251
|
+
full_name = " ".join(part.strip() for part in remainder.split(" > "))
|
|
252
|
+
if rel and full_name:
|
|
253
|
+
tests.add(self.qualify(f"{rel.strip()} > {full_name}"))
|
|
254
|
+
files.add(rel.strip())
|
|
255
|
+
# An empty result needs no special case: it accounts for no files, so every
|
|
256
|
+
# file falls to the loop exactly as a failure would.
|
|
257
|
+
return tests, files
|
|
258
|
+
|
|
259
|
+
def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
|
|
260
|
+
for rel_str in sorted(rels):
|
|
261
|
+
path = self.root / rel_str
|
|
262
|
+
rel = Path(rel_str)
|
|
230
263
|
ov = self.project.override_for(str(rel))
|
|
231
264
|
if ov is not None and not ov.collect_command:
|
|
232
265
|
result.failed_files[str(rel)] = (
|
|
@@ -237,7 +270,8 @@ class VitestAdapter(Adapter):
|
|
|
237
270
|
base = ov.collect_command if ov else self._collect_cmd()
|
|
238
271
|
env = self._suite_env(ov)
|
|
239
272
|
code, out, err = run_command(
|
|
240
|
-
f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env
|
|
273
|
+
f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env,
|
|
274
|
+
label="collect",
|
|
241
275
|
)
|
|
242
276
|
|
|
243
277
|
payload = _extract_json(out)
|
|
@@ -13,6 +13,7 @@ import socket
|
|
|
13
13
|
import sqlite3
|
|
14
14
|
import sys
|
|
15
15
|
import time
|
|
16
|
+
from collections.abc import Callable
|
|
16
17
|
from datetime import datetime, timezone
|
|
17
18
|
from pathlib import Path
|
|
18
19
|
|
|
@@ -217,16 +218,61 @@ def _legacy_artifacts(worktree: Path) -> list[Path]:
|
|
|
217
218
|
return sorted(found)
|
|
218
219
|
|
|
219
220
|
|
|
220
|
-
def
|
|
221
|
-
|
|
221
|
+
def _doctor_checklist() -> tuple[list[dict], Callable]:
|
|
222
|
+
"""A checks list and its recorder, which refuses a blocker it cannot explain.
|
|
223
|
+
|
|
224
|
+
`resolve_blocker` with an empty `detail` is unfalsifiable: an agent is told to
|
|
225
|
+
fix something and given nothing to fix. It re-runs doctor, reads the identical
|
|
226
|
+
output, and loops. Enforcing the detail here means a check added later inherits
|
|
227
|
+
the guarantee instead of relying on its author to remember.
|
|
228
|
+
"""
|
|
222
229
|
checks: list[dict] = []
|
|
223
230
|
|
|
224
231
|
def check(name, ok, detail="", project=None):
|
|
232
|
+
if not ok and not str(detail).strip():
|
|
233
|
+
raise AssertionError(
|
|
234
|
+
f"doctor check {name!r} would fail silently: a failing check must"
|
|
235
|
+
" name what to fix"
|
|
236
|
+
)
|
|
225
237
|
entry = {"check": name, "ok": bool(ok), "detail": detail}
|
|
226
238
|
if project is not None:
|
|
227
239
|
entry["project"] = project
|
|
228
240
|
checks.append(entry)
|
|
229
241
|
|
|
242
|
+
return checks, check
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _blocks_the_loop(rel_path: str, cfg) -> bool:
|
|
246
|
+
"""Whether dirt at `rel_path` is somewhere a run would actually read it."""
|
|
247
|
+
if rel_path == config_mod.CONFIG_NAME:
|
|
248
|
+
return True
|
|
249
|
+
if cfg.owning_project(rel_path) is not None:
|
|
250
|
+
return True
|
|
251
|
+
return any(art.owns(rel_path) for art in cfg.artifacts.values())
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _cleanliness_detail(blocking: list[str], unrelated: list[str]) -> str:
|
|
255
|
+
def listed(paths: list[str]) -> str:
|
|
256
|
+
head = ", ".join(paths[:5])
|
|
257
|
+
return head if len(paths) <= 5 else f"{head} (+{len(paths) - 5} more)"
|
|
258
|
+
|
|
259
|
+
if blocking:
|
|
260
|
+
return (
|
|
261
|
+
f"uncommitted changes a run would observe: {listed(blocking)}."
|
|
262
|
+
" Commit, stash, or gitignore them before `tdd run start`."
|
|
263
|
+
)
|
|
264
|
+
if unrelated:
|
|
265
|
+
return (
|
|
266
|
+
f"clean where a run reads; {len(unrelated)} unrelated path(s) left as-is:"
|
|
267
|
+
f" {listed(unrelated)}"
|
|
268
|
+
)
|
|
269
|
+
return ""
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def cmd_doctor(args) -> Envelope:
|
|
273
|
+
worktree = _worktree()
|
|
274
|
+
checks, check = _doctor_checklist()
|
|
275
|
+
|
|
230
276
|
check("worktree resolvable", True, str(worktree))
|
|
231
277
|
try:
|
|
232
278
|
cfg = config_mod.load(worktree)
|
|
@@ -246,13 +292,20 @@ def cmd_doctor(args) -> Envelope:
|
|
|
246
292
|
root = worktree / project.root
|
|
247
293
|
check("root exists", root.is_dir(), str(root), project=name)
|
|
248
294
|
check("adapter known", project.adapter in adapters.available(), project.adapter, project=name)
|
|
249
|
-
|
|
295
|
+
declared = bool(project.test_paths)
|
|
296
|
+
check(
|
|
297
|
+
"test_paths declared", declared,
|
|
298
|
+
"" if declared
|
|
299
|
+
else f"add `test_paths` to [project.{name}] in tdd.toml — without it no"
|
|
300
|
+
" suite can be discovered for this project",
|
|
301
|
+
project=name,
|
|
302
|
+
)
|
|
250
303
|
if project.adapter == "pytest":
|
|
251
304
|
# The probe runs in the project's own environment (uv, poetry, pipenv,
|
|
252
305
|
# pdm or the active venv) — hardcoding `uv run` here failed the check
|
|
253
306
|
# on any non-uv project even with the plugin installed.
|
|
254
307
|
probe = adapters.build(project, worktree).plugin_probe_cmd()
|
|
255
|
-
code, out, err = adapters.base.run_command(probe, root)
|
|
308
|
+
code, out, err = adapters.base.run_command(probe, root, label="doctor")
|
|
256
309
|
check("pytest-json-report installed", code == 0, (err or "")[:200], project=name)
|
|
257
310
|
|
|
258
311
|
# Run before `collectable()` so this actionable message wins over
|
|
@@ -289,11 +342,33 @@ def cmd_doctor(args) -> Envelope:
|
|
|
289
342
|
projects[name] = {"ok": all(c["ok"] for c in checks[before:])}
|
|
290
343
|
|
|
291
344
|
for art in cfg.artifacts.values():
|
|
292
|
-
|
|
345
|
+
# One evaluation feeds both `ok` and the detail: evaluating the condition
|
|
346
|
+
# twice lets them disagree, and a passing check that still says "add a hook"
|
|
347
|
+
# is the same misdirection as a failing check that says nothing.
|
|
348
|
+
has_hook = bool(art.check or art.regenerate)
|
|
349
|
+
check(
|
|
350
|
+
f"artifact {art.name}: has check or regenerate", has_hook,
|
|
351
|
+
"" if has_hook
|
|
352
|
+
else f"add `check` or `regenerate` to [artifact.{art.name}] in tdd.toml —"
|
|
353
|
+
" freshness cannot be verified without one",
|
|
354
|
+
)
|
|
293
355
|
|
|
294
356
|
stale = _legacy_artifacts(worktree)
|
|
295
|
-
check(
|
|
296
|
-
|
|
357
|
+
check(
|
|
358
|
+
"no legacy state artifacts", not stale,
|
|
359
|
+
f"delete these pre-ledger state files: {', '.join(str(s) for s in stale[:5])}"
|
|
360
|
+
if stale else "",
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
# Only dirt a run would read can corrupt one. Blocking on everything else
|
|
364
|
+
# stopped agents on unrelated notes and editor settings, and — because the
|
|
365
|
+
# check named no path — gave them nothing to act on but a re-run. `is_ignored`
|
|
366
|
+
# also excludes doctor's own probe residue (`.venv`, `node_modules`, caches),
|
|
367
|
+
# so running doctor can no longer be what makes doctor fail.
|
|
368
|
+
dirt = sorted(p for p in gitutil.dirty_paths(worktree) if not cfg.is_ignored(p))
|
|
369
|
+
blocking = [p for p in dirt if _blocks_the_loop(p, cfg)]
|
|
370
|
+
unrelated = [p for p in dirt if p not in set(blocking)]
|
|
371
|
+
check("worktree clean", not blocking, _cleanliness_detail(blocking, unrelated))
|
|
297
372
|
|
|
298
373
|
ok = all(c["ok"] for c in checks)
|
|
299
374
|
return Envelope(
|
|
@@ -364,12 +439,18 @@ def _probe_projects(cfg, worktree, ledger, on_progress):
|
|
|
364
439
|
for done, (name, project) in enumerate(cfg.projects.items(), start=1):
|
|
365
440
|
adapter = adapters.build(project, worktree)
|
|
366
441
|
started = time.monotonic()
|
|
367
|
-
verdict
|
|
442
|
+
verdict = adapter.run(None)
|
|
443
|
+
ran = time.monotonic()
|
|
444
|
+
collection = adapter.collect()
|
|
368
445
|
elapsed = time.monotonic() - started
|
|
369
446
|
probes[name] = (verdict, collection)
|
|
447
|
+
# Split, not just totalled: `run` and `collect` have unrelated cost models
|
|
448
|
+
# — one scales with tests, the other with files — and a single number sends
|
|
449
|
+
# whoever asks "why was that slow?" out of the tool to measure by hand.
|
|
370
450
|
heartbeat(
|
|
371
451
|
event="baseline_captured", project=name,
|
|
372
452
|
test_count=len(collection.tests), elapsed_s=round(elapsed, 2),
|
|
453
|
+
run_s=round(ran - started, 2), collect_s=round(elapsed - (ran - started), 2),
|
|
373
454
|
)
|
|
374
455
|
on_progress(done, name)
|
|
375
456
|
return probes
|