tdd-cli 0.2.1__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/CHANGELOG.md +56 -1
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/PKG-INFO +2 -2
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/pyproject.toml +1 -1
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/adapters/base.py +53 -9
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/adapters/pytest_adapter.py +14 -8
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/adapters/vitest_adapter.py +21 -11
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/cli.py +107 -8
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/config.py +13 -0
- tdd_cli-0.4.0/tests/test_doctor_blockers.py +173 -0
- tdd_cli-0.4.0/tests/test_failure_clipping.py +64 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_project_commands.py +1 -1
- tdd_cli-0.4.0/tests/test_project_env.py +124 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_suite_overrides.py +15 -15
- tdd_cli-0.4.0/tests/test_target_validation.py +55 -0
- tdd_cli-0.4.0/tests/test_timing_visibility.py +149 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_vitest_adapter.py +30 -2
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_worker_leases.py +1 -1
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/.gitignore +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/LICENSE +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/README.md +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/SECURITY.md +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/plan.md +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/advance.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/contract.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/identity.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/ledger.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/machine.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/render.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/src/tddcli/staging.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/conftest.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_artifact_regeneration.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_baseline_integrity.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_config_and_staging.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_contract.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_heartbeat.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_release_surface.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_snapshot_and_identity.py +0 -0
- {tdd_cli-0.2.1 → tdd_cli-0.4.0}/tests/test_stub_hint.py +0 -0
|
@@ -4,7 +4,62 @@ All notable changes to this project are documented here.
|
|
|
4
4
|
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
5
5
|
and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
-
## [
|
|
7
|
+
## [0.4.0] - 2026-08-16
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- `baseline_captured` reports `run_s` and `collect_s` alongside `elapsed_s`. The
|
|
12
|
+
suite run and the per-file collection have unrelated cost models — one scales
|
|
13
|
+
with tests, the other with files — so a single total could not say which was
|
|
14
|
+
slow, and answering that meant measuring projects by hand outside the tool.
|
|
15
|
+
- `TDD_TIMING=1` emits a `command_timing` line per subprocess on stderr
|
|
16
|
+
(`label`, `command`, `cwd`, `duration_ms`, `exit_code`), covering every
|
|
17
|
+
subprocess the tool spawns: suite runs, per-file collection, lint/typecheck
|
|
18
|
+
gates, doctor probes and artifact hooks. Off by default — the per-file loop
|
|
19
|
+
would otherwise emit one line per test file on every invocation. `label` is
|
|
20
|
+
one of `suite`, `collect`, `gate`, `doctor`; an unlabelled row comes from a
|
|
21
|
+
third-party adapter, since every built-in call site names itself (R8.4).
|
|
22
|
+
|
|
23
|
+
### Fixed
|
|
24
|
+
|
|
25
|
+
- `tdd doctor` no longer emits a blocker it cannot explain. Every failing check
|
|
26
|
+
now carries a `detail` naming what to fix, enforced by the checklist recorder
|
|
27
|
+
so a check added later inherits the guarantee. Previously `worktree clean`
|
|
28
|
+
failed with `detail: ""`, leaving an agent with `resolve_blocker` and nothing
|
|
29
|
+
to resolve — it re-ran doctor and read the identical output.
|
|
30
|
+
- `worktree clean` is scoped to dirt a run would actually read: a declared
|
|
31
|
+
project root, a declared artifact path, or `tdd.toml`. Build residue is
|
|
32
|
+
excluded via `config.is_ignored`, so doctor's own `uv run` / `vitest list`
|
|
33
|
+
probes (`.venv`, `node_modules`, caches) can no longer be what makes doctor
|
|
34
|
+
fail. Unrelated dirt is reported in the passing check's `detail` rather than
|
|
35
|
+
blocking the run.
|
|
36
|
+
|
|
37
|
+
## [0.3.0] - 2026-08-10
|
|
38
|
+
|
|
39
|
+
### Added
|
|
40
|
+
|
|
41
|
+
- `env` on `[project.<name>]`: environment for the default suite's runs and
|
|
42
|
+
collection, with the same semantics as an override's `env` (`${VAR}` expands
|
|
43
|
+
from the environment at invocation). An override's `env` layers on top for
|
|
44
|
+
its own suite. Previously only override suites could declare environment,
|
|
45
|
+
leaving a default suite that reads an endpoint from a variable with no
|
|
46
|
+
registry-level way to receive it (#16).
|
|
47
|
+
|
|
48
|
+
### Fixed
|
|
49
|
+
|
|
50
|
+
- vitest test ids are project-root-relative (`frontend::app/x.test.tsx > name`),
|
|
51
|
+
matching pytest nodeids and the form plan declarations qualify to — they were
|
|
52
|
+
worktree-relative (`frontend::frontend/app/...`), so a declared vitest target
|
|
53
|
+
could never match a verdict: standard cycles limped through on R8.9 adoption
|
|
54
|
+
(a spurious `declared_test_mismatch` per cycle) and pin cycles deadlocked in
|
|
55
|
+
`AWAITING_PIN`, since a pre-existing test is never adoptable (#21).
|
|
56
|
+
- `tdd target` refuses a name that is not a collected test in the cycle's
|
|
57
|
+
projects, suggesting the closest collected ids — previously any string was
|
|
58
|
+
recorded as the target and failed later, misattributed, as `not_found` (#15).
|
|
59
|
+
- Failure text (`target_failure`, uncollected-suite messages) is clipped keeping
|
|
60
|
+
both ends instead of truncated from the head: Python puts the actual error at
|
|
61
|
+
the tail of a traceback, so a head-only cut on a deep stack delivered
|
|
62
|
+
framework frames and cut exactly the line that says what went wrong (#17).
|
|
8
63
|
|
|
9
64
|
## [0.2.1] - 2026-08-10
|
|
10
65
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -47,7 +47,7 @@ packages = ["src/tddcli"]
|
|
|
47
47
|
include = ["src", "tests", "examples", "README.md", "LICENSE", "CHANGELOG.md", "SECURITY.md"]
|
|
48
48
|
|
|
49
49
|
[dependency-groups]
|
|
50
|
-
dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5"]
|
|
50
|
+
dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5", "zizmor>=1.29"]
|
|
51
51
|
|
|
52
52
|
[tool.pytest.ini_options]
|
|
53
53
|
testpaths = ["tests"]
|
|
@@ -4,11 +4,13 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import os
|
|
6
6
|
import subprocess
|
|
7
|
+
import time
|
|
7
8
|
from collections import Counter
|
|
8
9
|
from dataclasses import dataclass, field
|
|
9
10
|
from pathlib import Path
|
|
10
11
|
|
|
11
12
|
from .. import leases
|
|
13
|
+
from ..envelope import heartbeat
|
|
12
14
|
|
|
13
15
|
NOT_FOUND = "not_found"
|
|
14
16
|
NOT_COLLECTED = "not_collected"
|
|
@@ -64,10 +66,40 @@ def _overlap_error(overlap: list[str]) -> str:
|
|
|
64
66
|
)
|
|
65
67
|
|
|
66
68
|
|
|
69
|
+
def clip_failure(text: str, limit: int = 1500) -> str:
|
|
70
|
+
"""Clip failure text to `limit`, keeping both ends. Python puts the actual
|
|
71
|
+
error at the tail of a traceback, so a head-only cut on a deep stack
|
|
72
|
+
(async frameworks, ORMs, HTTP clients) delivered framework frames and cut
|
|
73
|
+
exactly the line that says what went wrong — forcing a re-run outside tdd
|
|
74
|
+
to see an error the tool already had. The head is kept too: for a plain
|
|
75
|
+
assertion failure the first line carries the assertion itself."""
|
|
76
|
+
if len(text) <= limit:
|
|
77
|
+
return text
|
|
78
|
+
head = limit // 5
|
|
79
|
+
tail = limit - head
|
|
80
|
+
return f"{text[:head]}\n… [clipped] …\n{text[-tail:]}"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
#: Opt-in per-command timing. Off by default: the per-file collect loop would emit
|
|
84
|
+
#: one line per test file on every invocation, drowning the heartbeats that exist
|
|
85
|
+
#: to make a slow baseline legible.
|
|
86
|
+
TIMING_ENV = "TDD_TIMING"
|
|
87
|
+
|
|
88
|
+
|
|
67
89
|
def run_command(
|
|
68
90
|
command: str, cwd: Path, timeout: int = 1800,
|
|
69
91
|
extra_env: dict[str, str] | None = None,
|
|
92
|
+
label: str | None = None,
|
|
70
93
|
) -> tuple[int, str, str]:
|
|
94
|
+
"""Every subprocess the tool spawns passes through here, which makes it the one
|
|
95
|
+
place worth timing: suite runs, per-file collection, lint and typecheck gates,
|
|
96
|
+
doctor probes, artifact hooks.
|
|
97
|
+
|
|
98
|
+
`label` is what makes the rows groupable. This function sees a command string
|
|
99
|
+
and a cwd — not which project or phase asked for it — so an unlabelled timing
|
|
100
|
+
is readable by a human and useless to a query.
|
|
101
|
+
"""
|
|
102
|
+
started = time.monotonic()
|
|
71
103
|
proc = subprocess.run(
|
|
72
104
|
command,
|
|
73
105
|
shell=True,
|
|
@@ -77,6 +109,15 @@ def run_command(
|
|
|
77
109
|
timeout=timeout,
|
|
78
110
|
env=None if extra_env is None else {**os.environ, **extra_env},
|
|
79
111
|
)
|
|
112
|
+
if os.environ.get(TIMING_ENV):
|
|
113
|
+
heartbeat(
|
|
114
|
+
event="command_timing",
|
|
115
|
+
label=label,
|
|
116
|
+
command=command,
|
|
117
|
+
cwd=str(cwd),
|
|
118
|
+
duration_ms=int((time.monotonic() - started) * 1000),
|
|
119
|
+
exit_code=proc.returncode,
|
|
120
|
+
)
|
|
80
121
|
return proc.returncode, proc.stdout, proc.stderr
|
|
81
122
|
|
|
82
123
|
|
|
@@ -116,6 +157,7 @@ class Adapter:
|
|
|
116
157
|
command.replace("{workers}", str(workers)),
|
|
117
158
|
self.root,
|
|
118
159
|
extra_env={"TDD_WORKERS": str(workers), **(extra_env or {})},
|
|
160
|
+
label="suite",
|
|
119
161
|
)
|
|
120
162
|
|
|
121
163
|
def _test_cmd(self) -> str:
|
|
@@ -128,17 +170,19 @@ class Adapter:
|
|
|
128
170
|
alternate runner config is still observed — without widening the default
|
|
129
171
|
config, which is exactly the workaround that breaks CI.
|
|
130
172
|
"""
|
|
131
|
-
return [(self._test_cmd(), None)] + [
|
|
132
|
-
(ov.test_command, self.
|
|
173
|
+
return [(self._test_cmd(), self._suite_env(None))] + [
|
|
174
|
+
(ov.test_command, self._suite_env(ov)) for ov in self.project.overrides
|
|
133
175
|
]
|
|
134
176
|
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
177
|
+
def _suite_env(self, override) -> dict[str, str] | None:
|
|
178
|
+
"""The environment for one suite invocation: the project's `env`, with the
|
|
179
|
+
owning override's layered on top (None means the default suite). `${VAR}`
|
|
180
|
+
references resolve from the environment at invocation time, so a port
|
|
181
|
+
assigned per checkout need not be hard-coded in the reviewed file."""
|
|
182
|
+
merged = {**self.project.env, **(override.env if override else {})}
|
|
183
|
+
if not merged:
|
|
140
184
|
return None
|
|
141
|
-
return {k: os.path.expandvars(v) for k, v in
|
|
185
|
+
return {k: os.path.expandvars(v) for k, v in merged.items()}
|
|
142
186
|
|
|
143
187
|
def stub_hint(self) -> str:
|
|
144
188
|
"""The language idiom for a stub body, quoted into the create_stub directive."""
|
|
@@ -168,7 +212,7 @@ class Adapter:
|
|
|
168
212
|
def _gate(self, commands: list[str]) -> GateResult:
|
|
169
213
|
chunks = []
|
|
170
214
|
for cmd in commands:
|
|
171
|
-
code, out, err = run_command(cmd, self.root)
|
|
215
|
+
code, out, err = run_command(cmd, self.root, label="gate")
|
|
172
216
|
if code != 0:
|
|
173
217
|
chunks.append(f"$ {cmd}\n{out}\n{err}".strip())
|
|
174
218
|
return GateResult(ok=not chunks, output="\n\n".join(chunks)[:4000])
|
|
@@ -27,6 +27,7 @@ from .base import (
|
|
|
27
27
|
Verdict,
|
|
28
28
|
_overlap_error,
|
|
29
29
|
_suite_overlap,
|
|
30
|
+
clip_failure,
|
|
30
31
|
run_command,
|
|
31
32
|
)
|
|
32
33
|
|
|
@@ -139,12 +140,14 @@ class PytestAdapter(Adapter):
|
|
|
139
140
|
if hit is not None:
|
|
140
141
|
verdict.target_outcome = PASSED if hit["outcome"] == "passed" else FAILED
|
|
141
142
|
call = hit.get("call") or hit.get("setup") or {}
|
|
142
|
-
verdict.target_failure = str(call.get("longrepr", ""))
|
|
143
|
+
verdict.target_failure = clip_failure(str(call.get("longrepr", "")))
|
|
143
144
|
else:
|
|
144
145
|
target_file = native.split("::", 1)[0]
|
|
145
146
|
if any(c == target_file or c.startswith(target_file) for c in uncollectable):
|
|
146
147
|
verdict.target_outcome = NOT_COLLECTED
|
|
147
|
-
verdict.target_failure =
|
|
148
|
+
verdict.target_failure = clip_failure(
|
|
149
|
+
self._collector_error(collectors, target_file)
|
|
150
|
+
)
|
|
148
151
|
else:
|
|
149
152
|
verdict.target_outcome = NOT_FOUND
|
|
150
153
|
return verdict
|
|
@@ -162,8 +165,8 @@ class PytestAdapter(Adapter):
|
|
|
162
165
|
`test_command` — pytest's `--collect-only` composes with any run command."""
|
|
163
166
|
ov = self.project.override_for(rel)
|
|
164
167
|
if ov is None:
|
|
165
|
-
return self._collect_cmd(), None
|
|
166
|
-
return ov.collect_command or ov.test_command, self.
|
|
168
|
+
return self._collect_cmd(), self._suite_env(None)
|
|
169
|
+
return ov.collect_command or ov.test_command, self._suite_env(ov)
|
|
167
170
|
|
|
168
171
|
def _test_files(self) -> list[Path]:
|
|
169
172
|
found: list[Path] = []
|
|
@@ -190,13 +193,13 @@ class PytestAdapter(Adapter):
|
|
|
190
193
|
loses the real error and the failure surfaces unattributed.
|
|
191
194
|
"""
|
|
192
195
|
chunks = []
|
|
193
|
-
probes = [(self._collect_cmd(), None)] + [
|
|
194
|
-
(ov.collect_command or ov.test_command, self.
|
|
196
|
+
probes = [(self._collect_cmd(), self._suite_env(None))] + [
|
|
197
|
+
(ov.collect_command or ov.test_command, self._suite_env(ov))
|
|
195
198
|
for ov in self.project.overrides
|
|
196
199
|
]
|
|
197
200
|
for cmd, env in probes:
|
|
198
201
|
code, out, err = run_command(
|
|
199
|
-
f"{cmd} --collect-only -q", self.root, extra_env=env
|
|
202
|
+
f"{cmd} --collect-only -q", self.root, extra_env=env, label="doctor"
|
|
200
203
|
)
|
|
201
204
|
if code != 0:
|
|
202
205
|
chunks.append(out.strip())
|
|
@@ -212,7 +215,9 @@ class PytestAdapter(Adapter):
|
|
|
212
215
|
if not self.project.overrides:
|
|
213
216
|
return GateResult(ok=True)
|
|
214
217
|
probe = f"{self._test_cmd().replace('{workers}', '0')} --collect-only -q"
|
|
215
|
-
code, out, err = run_command(
|
|
218
|
+
code, out, err = run_command(
|
|
219
|
+
probe, self.root, extra_env=self._suite_env(None), label="doctor"
|
|
220
|
+
)
|
|
216
221
|
reached = sorted({
|
|
217
222
|
f for f in (
|
|
218
223
|
line.split("::", 1)[0]
|
|
@@ -240,6 +245,7 @@ class PytestAdapter(Adapter):
|
|
|
240
245
|
f"{base} --collect-only -q {shlex.quote(str(rel))}",
|
|
241
246
|
self.root,
|
|
242
247
|
extra_env=env,
|
|
248
|
+
label="collect",
|
|
243
249
|
)
|
|
244
250
|
if code != 0:
|
|
245
251
|
result.failed_files[str(rel)] = (err or out).strip()[:800]
|
|
@@ -1,8 +1,11 @@
|
|
|
1
1
|
"""vitest adapter.
|
|
2
2
|
|
|
3
|
-
Test ids are `<
|
|
4
|
-
space-joined ancestorTitles plus the test title.
|
|
5
|
-
|
|
3
|
+
Test ids are `<project-root-relative file> > <fullName>`, where fullName is the
|
|
4
|
+
space-joined ancestorTitles plus the test title. Root-relative matches the pytest
|
|
5
|
+
adapter's nodeids and — decisively — `Engine._qualify`, which strips the project
|
|
6
|
+
root from plan declarations; a worktree-relative id here can never equal a
|
|
7
|
+
declared target. vitest may prefix its JSON with non-JSON lines, so the payload
|
|
8
|
+
is located rather than assumed (R10.2).
|
|
6
9
|
"""
|
|
7
10
|
|
|
8
11
|
from __future__ import annotations
|
|
@@ -23,6 +26,7 @@ from .base import (
|
|
|
23
26
|
Verdict,
|
|
24
27
|
_overlap_error,
|
|
25
28
|
_suite_overlap,
|
|
29
|
+
clip_failure,
|
|
26
30
|
run_command,
|
|
27
31
|
)
|
|
28
32
|
|
|
@@ -46,7 +50,7 @@ class VitestAdapter(Adapter):
|
|
|
46
50
|
def _id_for(self, suite_path: str, full_name: str) -> str:
|
|
47
51
|
abs_path = Path(suite_path)
|
|
48
52
|
try:
|
|
49
|
-
rel = os.path.relpath(abs_path, self.
|
|
53
|
+
rel = os.path.relpath(abs_path, self.root)
|
|
50
54
|
except ValueError:
|
|
51
55
|
rel = suite_path
|
|
52
56
|
return self.qualify(f"{rel} > {full_name}")
|
|
@@ -92,7 +96,7 @@ class VitestAdapter(Adapter):
|
|
|
92
96
|
suite_path = suite.get("name", "")
|
|
93
97
|
assertions = suite.get("assertionResults", [])
|
|
94
98
|
if not assertions and suite.get("status") == "failed":
|
|
95
|
-
failed_suites[suite_path] = str(suite.get("message", ""))
|
|
99
|
+
failed_suites[suite_path] = clip_failure(str(suite.get("message", "")))
|
|
96
100
|
for t in assertions:
|
|
97
101
|
qualified = self._id_for(suite_path, t["fullName"])
|
|
98
102
|
if t["status"] == "passed":
|
|
@@ -113,7 +117,8 @@ class VitestAdapter(Adapter):
|
|
|
113
117
|
for t in suite.get("assertionResults", []):
|
|
114
118
|
if self._id_for(suite.get("name", ""), t["fullName"]) == target:
|
|
115
119
|
verdict.target_failure = "\n".join(
|
|
116
|
-
m
|
|
120
|
+
clip_failure(m, 600)
|
|
121
|
+
for m in t.get("failureMessages", [])[:3]
|
|
117
122
|
)
|
|
118
123
|
return verdict
|
|
119
124
|
|
|
@@ -171,7 +176,9 @@ class VitestAdapter(Adapter):
|
|
|
171
176
|
suite — against a live backend — just to enumerate it.
|
|
172
177
|
"""
|
|
173
178
|
chunks = []
|
|
174
|
-
code, out, err = run_command(
|
|
179
|
+
code, out, err = run_command(
|
|
180
|
+
self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
|
|
181
|
+
)
|
|
175
182
|
if code != 0:
|
|
176
183
|
chunks.append((err or out).strip())
|
|
177
184
|
for ov in self.project.overrides:
|
|
@@ -183,7 +190,7 @@ class VitestAdapter(Adapter):
|
|
|
183
190
|
)
|
|
184
191
|
continue
|
|
185
192
|
code, out, err = run_command(
|
|
186
|
-
ov.collect_command, self.root, extra_env=self.
|
|
193
|
+
ov.collect_command, self.root, extra_env=self._suite_env(ov), label="doctor"
|
|
187
194
|
)
|
|
188
195
|
if code != 0:
|
|
189
196
|
chunks.append((err or out).strip())
|
|
@@ -196,7 +203,9 @@ class VitestAdapter(Adapter):
|
|
|
196
203
|
execute against whatever the tests need live)."""
|
|
197
204
|
if not self.project.overrides:
|
|
198
205
|
return GateResult(ok=True)
|
|
199
|
-
code, out, err = run_command(
|
|
206
|
+
code, out, err = run_command(
|
|
207
|
+
self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
|
|
208
|
+
)
|
|
200
209
|
reached = sorted({
|
|
201
210
|
f for f in (
|
|
202
211
|
line.strip().partition(" > ")[0]
|
|
@@ -226,9 +235,10 @@ class VitestAdapter(Adapter):
|
|
|
226
235
|
)
|
|
227
236
|
continue
|
|
228
237
|
base = ov.collect_command if ov else self._collect_cmd()
|
|
229
|
-
env = self.
|
|
238
|
+
env = self._suite_env(ov)
|
|
230
239
|
code, out, err = run_command(
|
|
231
|
-
f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env
|
|
240
|
+
f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env,
|
|
241
|
+
label="collect",
|
|
232
242
|
)
|
|
233
243
|
|
|
234
244
|
payload = _extract_json(out)
|
|
@@ -6,12 +6,14 @@ No command accepts a phase, a cycle number, or executor identity (R8.3).
|
|
|
6
6
|
from __future__ import annotations
|
|
7
7
|
|
|
8
8
|
import argparse
|
|
9
|
+
import difflib
|
|
9
10
|
import json
|
|
10
11
|
import os
|
|
11
12
|
import socket
|
|
12
13
|
import sqlite3
|
|
13
14
|
import sys
|
|
14
15
|
import time
|
|
16
|
+
from collections.abc import Callable
|
|
15
17
|
from datetime import datetime, timezone
|
|
16
18
|
from pathlib import Path
|
|
17
19
|
|
|
@@ -216,16 +218,61 @@ def _legacy_artifacts(worktree: Path) -> list[Path]:
|
|
|
216
218
|
return sorted(found)
|
|
217
219
|
|
|
218
220
|
|
|
219
|
-
def
|
|
220
|
-
|
|
221
|
+
def _doctor_checklist() -> tuple[list[dict], Callable]:
|
|
222
|
+
"""A checks list and its recorder, which refuses a blocker it cannot explain.
|
|
223
|
+
|
|
224
|
+
`resolve_blocker` with an empty `detail` is unfalsifiable: an agent is told to
|
|
225
|
+
fix something and given nothing to fix. It re-runs doctor, reads the identical
|
|
226
|
+
output, and loops. Enforcing the detail here means a check added later inherits
|
|
227
|
+
the guarantee instead of relying on its author to remember.
|
|
228
|
+
"""
|
|
221
229
|
checks: list[dict] = []
|
|
222
230
|
|
|
223
231
|
def check(name, ok, detail="", project=None):
|
|
232
|
+
if not ok and not str(detail).strip():
|
|
233
|
+
raise AssertionError(
|
|
234
|
+
f"doctor check {name!r} would fail silently: a failing check must"
|
|
235
|
+
" name what to fix"
|
|
236
|
+
)
|
|
224
237
|
entry = {"check": name, "ok": bool(ok), "detail": detail}
|
|
225
238
|
if project is not None:
|
|
226
239
|
entry["project"] = project
|
|
227
240
|
checks.append(entry)
|
|
228
241
|
|
|
242
|
+
return checks, check
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _blocks_the_loop(rel_path: str, cfg) -> bool:
|
|
246
|
+
"""Whether dirt at `rel_path` is somewhere a run would actually read it."""
|
|
247
|
+
if rel_path == config_mod.CONFIG_NAME:
|
|
248
|
+
return True
|
|
249
|
+
if cfg.owning_project(rel_path) is not None:
|
|
250
|
+
return True
|
|
251
|
+
return any(art.owns(rel_path) for art in cfg.artifacts.values())
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _cleanliness_detail(blocking: list[str], unrelated: list[str]) -> str:
|
|
255
|
+
def listed(paths: list[str]) -> str:
|
|
256
|
+
head = ", ".join(paths[:5])
|
|
257
|
+
return head if len(paths) <= 5 else f"{head} (+{len(paths) - 5} more)"
|
|
258
|
+
|
|
259
|
+
if blocking:
|
|
260
|
+
return (
|
|
261
|
+
f"uncommitted changes a run would observe: {listed(blocking)}."
|
|
262
|
+
" Commit, stash, or gitignore them before `tdd run start`."
|
|
263
|
+
)
|
|
264
|
+
if unrelated:
|
|
265
|
+
return (
|
|
266
|
+
f"clean where a run reads; {len(unrelated)} unrelated path(s) left as-is:"
|
|
267
|
+
f" {listed(unrelated)}"
|
|
268
|
+
)
|
|
269
|
+
return ""
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def cmd_doctor(args) -> Envelope:
|
|
273
|
+
worktree = _worktree()
|
|
274
|
+
checks, check = _doctor_checklist()
|
|
275
|
+
|
|
229
276
|
check("worktree resolvable", True, str(worktree))
|
|
230
277
|
try:
|
|
231
278
|
cfg = config_mod.load(worktree)
|
|
@@ -245,13 +292,20 @@ def cmd_doctor(args) -> Envelope:
|
|
|
245
292
|
root = worktree / project.root
|
|
246
293
|
check("root exists", root.is_dir(), str(root), project=name)
|
|
247
294
|
check("adapter known", project.adapter in adapters.available(), project.adapter, project=name)
|
|
248
|
-
|
|
295
|
+
declared = bool(project.test_paths)
|
|
296
|
+
check(
|
|
297
|
+
"test_paths declared", declared,
|
|
298
|
+
"" if declared
|
|
299
|
+
else f"add `test_paths` to [project.{name}] in tdd.toml — without it no"
|
|
300
|
+
" suite can be discovered for this project",
|
|
301
|
+
project=name,
|
|
302
|
+
)
|
|
249
303
|
if project.adapter == "pytest":
|
|
250
304
|
# The probe runs in the project's own environment (uv, poetry, pipenv,
|
|
251
305
|
# pdm or the active venv) — hardcoding `uv run` here failed the check
|
|
252
306
|
# on any non-uv project even with the plugin installed.
|
|
253
307
|
probe = adapters.build(project, worktree).plugin_probe_cmd()
|
|
254
|
-
code, out, err = adapters.base.run_command(probe, root)
|
|
308
|
+
code, out, err = adapters.base.run_command(probe, root, label="doctor")
|
|
255
309
|
check("pytest-json-report installed", code == 0, (err or "")[:200], project=name)
|
|
256
310
|
|
|
257
311
|
# Run before `collectable()` so this actionable message wins over
|
|
@@ -288,11 +342,33 @@ def cmd_doctor(args) -> Envelope:
|
|
|
288
342
|
projects[name] = {"ok": all(c["ok"] for c in checks[before:])}
|
|
289
343
|
|
|
290
344
|
for art in cfg.artifacts.values():
|
|
291
|
-
|
|
345
|
+
# One evaluation feeds both `ok` and the detail: evaluating the condition
|
|
346
|
+
# twice lets them disagree, and a passing check that still says "add a hook"
|
|
347
|
+
# is the same misdirection as a failing check that says nothing.
|
|
348
|
+
has_hook = bool(art.check or art.regenerate)
|
|
349
|
+
check(
|
|
350
|
+
f"artifact {art.name}: has check or regenerate", has_hook,
|
|
351
|
+
"" if has_hook
|
|
352
|
+
else f"add `check` or `regenerate` to [artifact.{art.name}] in tdd.toml —"
|
|
353
|
+
" freshness cannot be verified without one",
|
|
354
|
+
)
|
|
292
355
|
|
|
293
356
|
stale = _legacy_artifacts(worktree)
|
|
294
|
-
check(
|
|
295
|
-
|
|
357
|
+
check(
|
|
358
|
+
"no legacy state artifacts", not stale,
|
|
359
|
+
f"delete these pre-ledger state files: {', '.join(str(s) for s in stale[:5])}"
|
|
360
|
+
if stale else "",
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
# Only dirt a run would read can corrupt one. Blocking on everything else
|
|
364
|
+
# stopped agents on unrelated notes and editor settings, and — because the
|
|
365
|
+
# check named no path — gave them nothing to act on but a re-run. `is_ignored`
|
|
366
|
+
# also excludes doctor's own probe residue (`.venv`, `node_modules`, caches),
|
|
367
|
+
# so running doctor can no longer be what makes doctor fail.
|
|
368
|
+
dirt = sorted(p for p in gitutil.dirty_paths(worktree) if not cfg.is_ignored(p))
|
|
369
|
+
blocking = [p for p in dirt if _blocks_the_loop(p, cfg)]
|
|
370
|
+
unrelated = [p for p in dirt if p not in set(blocking)]
|
|
371
|
+
check("worktree clean", not blocking, _cleanliness_detail(blocking, unrelated))
|
|
296
372
|
|
|
297
373
|
ok = all(c["ok"] for c in checks)
|
|
298
374
|
return Envelope(
|
|
@@ -363,12 +439,18 @@ def _probe_projects(cfg, worktree, ledger, on_progress):
|
|
|
363
439
|
for done, (name, project) in enumerate(cfg.projects.items(), start=1):
|
|
364
440
|
adapter = adapters.build(project, worktree)
|
|
365
441
|
started = time.monotonic()
|
|
366
|
-
verdict
|
|
442
|
+
verdict = adapter.run(None)
|
|
443
|
+
ran = time.monotonic()
|
|
444
|
+
collection = adapter.collect()
|
|
367
445
|
elapsed = time.monotonic() - started
|
|
368
446
|
probes[name] = (verdict, collection)
|
|
447
|
+
# Split, not just totalled: `run` and `collect` have unrelated cost models
|
|
448
|
+
# — one scales with tests, the other with files — and a single number sends
|
|
449
|
+
# whoever asks "why was that slow?" out of the tool to measure by hand.
|
|
369
450
|
heartbeat(
|
|
370
451
|
event="baseline_captured", project=name,
|
|
371
452
|
test_count=len(collection.tests), elapsed_s=round(elapsed, 2),
|
|
453
|
+
run_s=round(ran - started, 2), collect_s=round(elapsed - (ran - started), 2),
|
|
372
454
|
)
|
|
373
455
|
on_progress(done, name)
|
|
374
456
|
return probes
|
|
@@ -822,6 +904,23 @@ def cmd_target(args) -> Envelope:
|
|
|
822
904
|
cycle = ledger.open_cycle(run["id"])
|
|
823
905
|
if cycle is None:
|
|
824
906
|
return failure("no open cycle")
|
|
907
|
+
|
|
908
|
+
# The target must be grounded in observed collection, the same way phase is
|
|
909
|
+
# grounded in observed execution (#15): recording free text deferred a typo —
|
|
910
|
+
# or a speculative `tdd target env` — to the next suite run, where it
|
|
911
|
+
# surfaced as `not_found` against a test that never existed.
|
|
912
|
+
known: set[str] = set()
|
|
913
|
+
for name in json.loads(cycle["projects"]):
|
|
914
|
+
adapter = adapters.build(cfg.project(name), worktree)
|
|
915
|
+
known |= adapter.collect().tests
|
|
916
|
+
if args.test not in known:
|
|
917
|
+
close = difflib.get_close_matches(args.test, sorted(known), n=3, cutoff=0.6)
|
|
918
|
+
hint = f" Closest collected ids: {', '.join(close)}." if close else ""
|
|
919
|
+
return failure(
|
|
920
|
+
f"{args.test} is not a collected test in this cycle's projects;"
|
|
921
|
+
f" the target was not changed.{hint}"
|
|
922
|
+
)
|
|
923
|
+
|
|
825
924
|
ledger.update("cycle", cycle["id"], target_tests=json.dumps([args.test]))
|
|
826
925
|
ledger.event(run["id"], cycle["id"], "target_named_by_agent", args.test)
|
|
827
926
|
return Envelope(
|
|
@@ -88,6 +88,11 @@ class Project:
|
|
|
88
88
|
#: Per-file collection. Must not be parallelised: collection is cheap and xdist
|
|
89
89
|
#: adds startup cost per file.
|
|
90
90
|
collect_command: str | None = None
|
|
91
|
+
#: Environment for the default suite's runs and collection, same semantics as
|
|
92
|
+
#: an override's `env`: `${VAR}` references expand from the environment at
|
|
93
|
+
#: invocation, so per-checkout values (a database port) stay out of the
|
|
94
|
+
#: reviewed file. An override's `env` layers on top for its own suite.
|
|
95
|
+
env: dict[str, str] = field(default_factory=dict)
|
|
91
96
|
#: Alternate suites for files the default command cannot reach (R7.13).
|
|
92
97
|
overrides: list[Override] = field(default_factory=list)
|
|
93
98
|
|
|
@@ -290,6 +295,13 @@ def load(worktree: Path) -> Config:
|
|
|
290
295
|
raise ConfigError(f"project {name!r} has no root")
|
|
291
296
|
if "adapter" not in body:
|
|
292
297
|
raise ConfigError(f"project {name!r} has no adapter")
|
|
298
|
+
env = body.get("env", {})
|
|
299
|
+
if not isinstance(env, dict) or not all(
|
|
300
|
+
isinstance(v, str) for v in env.values()
|
|
301
|
+
):
|
|
302
|
+
raise ConfigError(
|
|
303
|
+
f"project {name!r}: env must be a table of string values"
|
|
304
|
+
)
|
|
293
305
|
projects[name] = Project(
|
|
294
306
|
name=name,
|
|
295
307
|
root=body["root"].rstrip("/"),
|
|
@@ -300,6 +312,7 @@ def load(worktree: Path) -> Config:
|
|
|
300
312
|
in_close_sweep=body.get("in_close_sweep", True),
|
|
301
313
|
test_command=body.get("test_command"),
|
|
302
314
|
collect_command=body.get("collect_command"),
|
|
315
|
+
env=env,
|
|
303
316
|
overrides=_load_overrides(name, body.get("override", [])),
|
|
304
317
|
)
|
|
305
318
|
|