tdd-cli 0.4.0__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/CHANGELOG.md +15 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/PKG-INFO +1 -1
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/adapters/base.py +42 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/adapters/pytest_adapter.py +37 -11
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/adapters/vitest_adapter.py +37 -4
- tdd_cli-0.4.1/tests/test_batch_collection.py +233 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_suite_overrides.py +7 -1
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_timing_visibility.py +14 -5
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/.gitignore +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/LICENSE +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/README.md +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/SECURITY.md +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/plan.md +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/pyproject.toml +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/advance.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/cli.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/config.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/contract.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/identity.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/ledger.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/machine.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/render.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/staging.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/conftest.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_artifact_regeneration.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_baseline_integrity.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_config_and_staging.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_contract.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_doctor_blockers.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_failure_clipping.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_fleet.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_heartbeat.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_progress.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_project_commands.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_project_env.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_release_surface.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_snapshot_and_identity.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_vitest_adapter.py +0 -0
- {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_worker_leases.py +0 -0
|
@@ -4,6 +4,21 @@ All notable changes to this project are documented here.
|
|
|
4
4
|
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
5
5
|
and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [0.4.1] - 2026-08-16
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- `collect()` runs one invocation per declared suite instead of one per test
|
|
12
|
+
file, falling back to the per-file loop for anything a batch did not account
|
|
13
|
+
for. Per-file collection was **77% of a real `run start`** — 313 subprocesses
|
|
14
|
+
costing 402s, against 117s to actually run every test — because cost scaled
|
|
15
|
+
with file count at a ~1.08s floor per invocation (the environment manager
|
|
16
|
+
resolving plus the runner booting). Measured 38.8x faster on a 60-file
|
|
17
|
+
project, with an identical collected set. R10.3's guarantee is unchanged: a
|
|
18
|
+
file that fails to collect is still attributed to itself and cannot destroy
|
|
19
|
+
the set, and a file the batch never reports is still collected individually,
|
|
20
|
+
so the set can only match or improve on the old one (#27).
|
|
21
|
+
|
|
7
22
|
## [0.4.0] - 2026-08-16
|
|
8
23
|
|
|
9
24
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -189,6 +189,48 @@ class Adapter:
|
|
|
189
189
|
return "a body that fails loudly, never working logic"
|
|
190
190
|
|
|
191
191
|
def collect(self) -> Collection:
|
|
192
|
+
"""Enumerate the project's tests: one invocation per declared suite, with
|
|
193
|
+
the per-file loop kept for what that cannot account for (issue #27).
|
|
194
|
+
|
|
195
|
+
Per file, collection cost scaled with *file count* rather than test count —
|
|
196
|
+
measured at 77% of a whole `run start`, against a floor of ~1.08s per
|
|
197
|
+
invocation that is the environment manager resolving plus the runner
|
|
198
|
+
booting, not collection work.
|
|
199
|
+
|
|
200
|
+
The batch and the loop do not discover the same way: a batch uses the
|
|
201
|
+
runner's own config, the loop walks `test_paths`. So the loop still runs for
|
|
202
|
+
every file the batch did not report — whether the batch failed, returned
|
|
203
|
+
nothing, or simply never mentioned that file. R10.3's guarantee is
|
|
204
|
+
unchanged: one uncollectable file is attributed to itself, and cannot
|
|
205
|
+
destroy the set. What changes is that the healthy case no longer pays for
|
|
206
|
+
the broken one.
|
|
207
|
+
"""
|
|
208
|
+
result = Collection()
|
|
209
|
+
unaccounted = {str(p.relative_to(self.root)) for p in self._test_files()}
|
|
210
|
+
for command, env in self._collect_invocations():
|
|
211
|
+
batch = self._collect_batch(command, env)
|
|
212
|
+
if batch is None:
|
|
213
|
+
continue
|
|
214
|
+
tests, files = batch
|
|
215
|
+
result.tests |= tests
|
|
216
|
+
unaccounted -= files
|
|
217
|
+
return self._collect_per_file(unaccounted, result)
|
|
218
|
+
|
|
219
|
+
def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
|
|
220
|
+
"""One collection command per declared suite: the default plus each
|
|
221
|
+
override's (R7.13), so an override's files are enumerated with its own
|
|
222
|
+
command and env."""
|
|
223
|
+
raise NotImplementedError
|
|
224
|
+
|
|
225
|
+
def _collect_batch(
|
|
226
|
+
self, command: str, env: dict[str, str] | None
|
|
227
|
+
) -> tuple[set[str], set[str]] | None:
|
|
228
|
+
"""`(test ids, files seen)` for one whole-suite invocation, or None when it
|
|
229
|
+
produced nothing usable — the signal to leave its files to the loop."""
|
|
230
|
+
raise NotImplementedError
|
|
231
|
+
|
|
232
|
+
def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
|
|
233
|
+
"""R10.3's loop, now reached only for files no batch accounted for."""
|
|
192
234
|
raise NotImplementedError
|
|
193
235
|
|
|
194
236
|
def collectable(self) -> GateResult:
|
|
@@ -235,23 +235,49 @@ class PytestAdapter(Adapter):
|
|
|
235
235
|
" cannot reach them (e.g. `pytest tests/`)."
|
|
236
236
|
))
|
|
237
237
|
|
|
238
|
-
def
|
|
238
|
+
def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
|
|
239
|
+
"""An override without a `collect_command` collects with its `test_command`
|
|
240
|
+
— `--collect-only` composes with any run command."""
|
|
241
|
+
return [(self._collect_cmd(), self._suite_env(None))] + [
|
|
242
|
+
(ov.collect_command or ov.test_command, self._suite_env(ov))
|
|
243
|
+
for ov in self.project.overrides
|
|
244
|
+
]
|
|
245
|
+
|
|
246
|
+
def _nodeids(self, out: str) -> tuple[set[str], set[str]]:
|
|
247
|
+
tests: set[str] = set()
|
|
248
|
+
files: set[str] = set()
|
|
249
|
+
for line in out.splitlines():
|
|
250
|
+
line = line.strip()
|
|
251
|
+
if "::" in line and not line.startswith(("=", "-", "no tests")):
|
|
252
|
+
tests.add(self.qualify(line))
|
|
253
|
+
files.add(line.split("::", 1)[0])
|
|
254
|
+
return tests, files
|
|
255
|
+
|
|
256
|
+
def _collect_batch(
|
|
257
|
+
self, command: str, env: dict[str, str] | None
|
|
258
|
+
) -> tuple[set[str], set[str]] | None:
|
|
259
|
+
code, out, err = run_command(
|
|
260
|
+
f"{command} --collect-only -q", self.root, extra_env=env, label="collect"
|
|
261
|
+
)
|
|
262
|
+
if code != 0:
|
|
263
|
+
return None
|
|
264
|
+
# An empty result needs no special case: it accounts for no files, so every
|
|
265
|
+
# file falls to the loop exactly as a failure would.
|
|
266
|
+
return self._nodeids(out)
|
|
267
|
+
|
|
268
|
+
def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
|
|
239
269
|
"""Per file (R10.3) — one uncollectable module must not destroy the whole set."""
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
rel = path.relative_to(self.root)
|
|
243
|
-
base, env = self._collect_cmd_for(str(rel))
|
|
270
|
+
for rel in sorted(rels):
|
|
271
|
+
base, env = self._collect_cmd_for(rel)
|
|
244
272
|
code, out, err = run_command(
|
|
245
|
-
f"{base} --collect-only -q {shlex.quote(
|
|
273
|
+
f"{base} --collect-only -q {shlex.quote(rel)}",
|
|
246
274
|
self.root,
|
|
247
275
|
extra_env=env,
|
|
248
276
|
label="collect",
|
|
249
277
|
)
|
|
250
278
|
if code != 0:
|
|
251
|
-
result.failed_files[
|
|
279
|
+
result.failed_files[rel] = (err or out).strip()[:800]
|
|
252
280
|
continue
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
if "::" in line and not line.startswith(("=", "-", "no tests")):
|
|
256
|
-
result.tests.add(self.qualify(line))
|
|
281
|
+
tests, _ = self._nodeids(out)
|
|
282
|
+
result.tests |= tests
|
|
257
283
|
return result
|
|
@@ -223,10 +223,43 @@ class VitestAdapter(Adapter):
|
|
|
223
223
|
" config (test.exclude) or scope its include globs."
|
|
224
224
|
))
|
|
225
225
|
|
|
226
|
-
def
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
226
|
+
def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
|
|
227
|
+
"""An override with no `collect_command` gets no batch — `vitest list` knows
|
|
228
|
+
nothing of its config, and its files fall to the loop, which records the
|
|
229
|
+
missing `collect_command` against each of them as before."""
|
|
230
|
+
return [(self._collect_cmd(), self._suite_env(None))] + [
|
|
231
|
+
(ov.collect_command, self._suite_env(ov))
|
|
232
|
+
for ov in self.project.overrides if ov.collect_command
|
|
233
|
+
]
|
|
234
|
+
|
|
235
|
+
def _collect_batch(
|
|
236
|
+
self, command: str, env: dict[str, str] | None
|
|
237
|
+
) -> tuple[set[str], set[str]] | None:
|
|
238
|
+
"""Unlike `_parse_list_output`, which pins every id to the one file it was
|
|
239
|
+
given, a whole-suite listing must read the file from each line — the same
|
|
240
|
+
`file > describe > name` shape, one file per line rather than one per run."""
|
|
241
|
+
code, out, err = run_command(command, self.root, extra_env=env, label="collect")
|
|
242
|
+
if code != 0:
|
|
243
|
+
return None
|
|
244
|
+
tests: set[str] = set()
|
|
245
|
+
files: set[str] = set()
|
|
246
|
+
for line in out.splitlines():
|
|
247
|
+
line = line.strip()
|
|
248
|
+
if " > " not in line:
|
|
249
|
+
continue
|
|
250
|
+
rel, _, remainder = line.partition(" > ")
|
|
251
|
+
full_name = " ".join(part.strip() for part in remainder.split(" > "))
|
|
252
|
+
if rel and full_name:
|
|
253
|
+
tests.add(self.qualify(f"{rel.strip()} > {full_name}"))
|
|
254
|
+
files.add(rel.strip())
|
|
255
|
+
# An empty result needs no special case: it accounts for no files, so every
|
|
256
|
+
# file falls to the loop exactly as a failure would.
|
|
257
|
+
return tests, files
|
|
258
|
+
|
|
259
|
+
def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
|
|
260
|
+
for rel_str in sorted(rels):
|
|
261
|
+
path = self.root / rel_str
|
|
262
|
+
rel = Path(rel_str)
|
|
230
263
|
ov = self.project.override_for(str(rel))
|
|
231
264
|
if ov is not None and not ov.collect_command:
|
|
232
265
|
result.failed_files[str(rel)] = (
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""Collection asks the runner once, not once per file (issue #27).
|
|
2
|
+
|
|
3
|
+
Measured on a real repo: 313 per-file subprocesses cost 402s, 77% of a whole
|
|
4
|
+
`run start`, while *running* all the tests cost 117s. The floor per invocation
|
|
5
|
+
was 1.08s — the environment manager resolving plus the runner booting, paid
|
|
6
|
+
again per file. A single file's tests enumerate in milliseconds.
|
|
7
|
+
|
|
8
|
+
Both adapters already had a whole-suite probe (`collectable()`); collection now
|
|
9
|
+
uses that shape and keeps the per-file loop for the case it exists to serve.
|
|
10
|
+
|
|
11
|
+
The rule that makes this safe: **a file the batch does not account for gets
|
|
12
|
+
exactly the old per-file treatment.** Batch fails, batch is empty, batch skips a
|
|
13
|
+
file the registry declares — each falls through to the loop, so the collected set
|
|
14
|
+
can only match or improve on the old one, never silently shrink (R10.3/R10.4).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from conftest import run_cli, write_plan
|
|
22
|
+
from tddcli import adapters
|
|
23
|
+
from tddcli import config as config_mod
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _adapter(repo: Path, project: str = "backend"):
|
|
27
|
+
return adapters.build(config_mod.load(repo).project(project), repo)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _counting(monkeypatch, module=None):
|
|
31
|
+
"""Wrap the real `run_command`, recording every command it is asked to run."""
|
|
32
|
+
seen: list[str] = []
|
|
33
|
+
real = adapters.base.run_command
|
|
34
|
+
|
|
35
|
+
def wrapper(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
36
|
+
seen.append(command)
|
|
37
|
+
return real(command, cwd, timeout=timeout, extra_env=extra_env, label=label)
|
|
38
|
+
|
|
39
|
+
monkeypatch.setattr(module or adapters.base, "run_command", wrapper)
|
|
40
|
+
return seen
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _write_tests(repo: Path, n: int) -> None:
|
|
44
|
+
for i in range(n):
|
|
45
|
+
(repo / "backend" / "tests" / f"test_gen{i}.py").write_text(
|
|
46
|
+
f"def test_gen{i}():\n assert True\n"
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_a_healthy_project_is_collected_in_one_invocation(repo, monkeypatch):
|
|
51
|
+
"""The whole point: cost stops scaling with file count."""
|
|
52
|
+
_write_tests(repo, 6)
|
|
53
|
+
seen = _counting(monkeypatch, adapters.pytest_adapter)
|
|
54
|
+
|
|
55
|
+
collected = _adapter(repo).collect()
|
|
56
|
+
|
|
57
|
+
assert len(seen) == 1, seen
|
|
58
|
+
assert len(collected.tests) == 7, collected.tests # 6 generated + test_smoke
|
|
59
|
+
assert collected.failed_files == {}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_the_batch_finds_the_same_tests_the_per_file_loop_would(repo, monkeypatch):
|
|
63
|
+
"""Equivalence is the whole risk of this change: batch enumeration uses the
|
|
64
|
+
runner's own discovery, the per-file loop uses `test_paths` globs."""
|
|
65
|
+
_write_tests(repo, 4)
|
|
66
|
+
adapter = _adapter(repo)
|
|
67
|
+
|
|
68
|
+
batched = adapter.collect().tests
|
|
69
|
+
per_file = adapter._collect_per_file(
|
|
70
|
+
{str(p.relative_to(adapter.root)) for p in adapter._test_files()},
|
|
71
|
+
adapters.base.Collection(),
|
|
72
|
+
).tests
|
|
73
|
+
|
|
74
|
+
assert batched == per_file, batched ^ per_file
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def test_a_file_the_batch_never_reported_is_collected_individually(repo, monkeypatch):
|
|
78
|
+
"""A file matching `test_paths` that the runner's config excludes must not
|
|
79
|
+
vanish from the set — a quietly smaller baseline is worse than a slow one."""
|
|
80
|
+
_write_tests(repo, 3)
|
|
81
|
+
adapter = _adapter(repo)
|
|
82
|
+
real = adapters.base.run_command
|
|
83
|
+
seen: list[str] = []
|
|
84
|
+
|
|
85
|
+
def hide_one(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
86
|
+
seen.append(command)
|
|
87
|
+
code, out, err = real(command, cwd, timeout=timeout, extra_env=extra_env, label=label)
|
|
88
|
+
if "test_gen1.py" not in command: # the batch "forgets" this file
|
|
89
|
+
out = "\n".join(
|
|
90
|
+
line for line in out.splitlines() if "test_gen1.py" not in line
|
|
91
|
+
)
|
|
92
|
+
return code, out, err
|
|
93
|
+
|
|
94
|
+
monkeypatch.setattr(adapters.pytest_adapter, "run_command", hide_one)
|
|
95
|
+
collected = adapter.collect()
|
|
96
|
+
|
|
97
|
+
assert any("test_gen1.py" in t for t in collected.tests), collected.tests
|
|
98
|
+
# Exactly one rescue invocation, naming that file — not a whole re-sweep.
|
|
99
|
+
per_file = [c for c in seen if "test_gen1.py" in c]
|
|
100
|
+
assert len(per_file) == 1, seen
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def test_a_failing_batch_falls_back_to_per_file_attribution(repo_broken):
|
|
104
|
+
"""R10.3's purpose, preserved: one uncollectable module must not destroy the
|
|
105
|
+
set, and the failure must name the file it came from."""
|
|
106
|
+
adapter = _adapter(repo_broken, "verify")
|
|
107
|
+
collected = adapter.collect()
|
|
108
|
+
|
|
109
|
+
assert collected.failed_files, "a broken module must be attributed"
|
|
110
|
+
assert any("test_v.py" in name for name in collected.failed_files), collected.failed_files
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def test_one_broken_file_does_not_lose_its_healthy_neighbours(repo_broken):
|
|
114
|
+
"""The batch fails wholesale, so the fallback has to recover everything else."""
|
|
115
|
+
(repo_broken / "verify" / "tests" / "test_ok.py").write_text(
|
|
116
|
+
"def test_ok():\n assert True\n"
|
|
117
|
+
)
|
|
118
|
+
collected = _adapter(repo_broken, "verify").collect()
|
|
119
|
+
|
|
120
|
+
assert any("test_ok.py" in t for t in collected.tests), collected.tests
|
|
121
|
+
assert any("test_v.py" in f for f in collected.failed_files), collected.failed_files
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def test_override_files_are_collected_by_their_own_suite(repo, monkeypatch):
|
|
125
|
+
"""R7.13: an override's files are enumerated with the override's command and
|
|
126
|
+
env, so batching must stay one invocation *per declared suite*."""
|
|
127
|
+
(repo / "backend" / "contract").mkdir()
|
|
128
|
+
(repo / "backend" / "contract" / "test_api.py").write_text(
|
|
129
|
+
"def test_ping():\n assert True\n"
|
|
130
|
+
)
|
|
131
|
+
(repo / "tdd.toml").write_text(
|
|
132
|
+
"[project.backend]\n"
|
|
133
|
+
'root = "backend"\n'
|
|
134
|
+
'adapter = "pytest"\n'
|
|
135
|
+
'test_paths = ["tests/"]\n'
|
|
136
|
+
'test_command = "pytest tests"\n'
|
|
137
|
+
"[[project.backend.override]]\n"
|
|
138
|
+
'pattern = "contract/"\n'
|
|
139
|
+
'test_command = "pytest contract"\n'
|
|
140
|
+
)
|
|
141
|
+
seen = _counting(monkeypatch, adapters.pytest_adapter)
|
|
142
|
+
|
|
143
|
+
collected = _adapter(repo).collect()
|
|
144
|
+
|
|
145
|
+
assert len(seen) == 2, seen # default suite + override suite, no per-file
|
|
146
|
+
assert any("contract/test_api.py" in t for t in collected.tests), collected.tests
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def test_partial_output_from_a_failed_batch_is_not_trusted(repo, monkeypatch):
|
|
150
|
+
"""A runner that aborts mid-collection still prints what it reached. Accepting
|
|
151
|
+
that would take a file's *incomplete* test list as final — and reconciliation
|
|
152
|
+
could not save it, because the file was mentioned. Every file is re-collected
|
|
153
|
+
instead, which is the old cost only in the case that was already broken."""
|
|
154
|
+
_write_tests(repo, 3)
|
|
155
|
+
adapter = _adapter(repo)
|
|
156
|
+
real = adapters.base.run_command
|
|
157
|
+
seen: list[str] = []
|
|
158
|
+
|
|
159
|
+
def failing_batch(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
160
|
+
seen.append(command)
|
|
161
|
+
if "test_gen" not in command and "test_smoke" not in command:
|
|
162
|
+
# whole-suite invocation: aborts, having reached one file
|
|
163
|
+
return 1, "tests/test_gen0.py::test_gen0\n", "INTERNALERROR"
|
|
164
|
+
return real(command, cwd, timeout=timeout, extra_env=extra_env, label=label)
|
|
165
|
+
|
|
166
|
+
monkeypatch.setattr(adapters.pytest_adapter, "run_command", failing_batch)
|
|
167
|
+
collected = adapter.collect()
|
|
168
|
+
|
|
169
|
+
assert any("test_gen0.py" in c for c in seen[1:]), seen
|
|
170
|
+
assert len(collected.tests) == 4, collected.tests # 3 generated + test_smoke
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def test_vitest_partial_output_from_a_failed_batch_is_not_trusted(repo_multi, monkeypatch):
|
|
174
|
+
"""Same rule for the vitest adapter, whose listing has no exit-code-free way to
|
|
175
|
+
tell a complete run from an aborted one."""
|
|
176
|
+
(repo_multi / "frontend" / "a.test.ts").write_text("")
|
|
177
|
+
(repo_multi / "frontend" / "b.test.ts").write_text("")
|
|
178
|
+
adapter = _adapter(repo_multi, "frontend")
|
|
179
|
+
seen: list[str] = []
|
|
180
|
+
|
|
181
|
+
def failing_batch(command, cwd, timeout=1800, extra_env=None, label=None):
|
|
182
|
+
seen.append(command)
|
|
183
|
+
if command.endswith("list"): # whole-suite invocation
|
|
184
|
+
return 1, "a.test.ts > alpha\n", "crashed"
|
|
185
|
+
return 0, f"{command.rsplit(' ', 1)[1]} > rescued", ""
|
|
186
|
+
|
|
187
|
+
monkeypatch.setattr(adapters.vitest_adapter, "run_command", failing_batch)
|
|
188
|
+
collected = adapter.collect()
|
|
189
|
+
|
|
190
|
+
assert collected.tests == {
|
|
191
|
+
"frontend::a.test.ts > rescued", "frontend::b.test.ts > rescued",
|
|
192
|
+
}, collected.tests
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def test_vitest_batch_attributes_each_id_to_its_own_file(repo_multi, monkeypatch):
|
|
196
|
+
"""`_parse_list_output` pinned every id to one path, which is right per-file
|
|
197
|
+
and wrong for a whole-suite listing."""
|
|
198
|
+
(repo_multi / "frontend" / "a.test.ts").write_text(
|
|
199
|
+
"import {test, expect} from 'vitest'\ntest('alpha', () => expect(1).toBe(1))\n"
|
|
200
|
+
)
|
|
201
|
+
(repo_multi / "frontend" / "b.test.ts").write_text(
|
|
202
|
+
"import {test, expect} from 'vitest'\ntest('beta', () => expect(1).toBe(1))\n"
|
|
203
|
+
)
|
|
204
|
+
adapter = _adapter(repo_multi, "frontend")
|
|
205
|
+
monkeypatch.setattr(
|
|
206
|
+
adapters.vitest_adapter, "run_command",
|
|
207
|
+
lambda command, cwd, timeout=1800, extra_env=None, label=None: (
|
|
208
|
+
0, "a.test.ts > alpha\nb.test.ts > beta\n", ""
|
|
209
|
+
),
|
|
210
|
+
)
|
|
211
|
+
collected = adapter.collect()
|
|
212
|
+
|
|
213
|
+
assert collected.tests == {"frontend::a.test.ts > alpha", "frontend::b.test.ts > beta"}
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def test_run_start_still_reports_the_same_baseline(repo, monkeypatch):
|
|
217
|
+
"""End to end, through the command that pays for this."""
|
|
218
|
+
_write_tests(repo, 3)
|
|
219
|
+
plan = write_plan(repo, """---
|
|
220
|
+
cycles:
|
|
221
|
+
- n: 1
|
|
222
|
+
project: backend
|
|
223
|
+
title: "adding two numbers"
|
|
224
|
+
test: "tests/test_add.py::test_add_two_numbers"
|
|
225
|
+
commit_red: "test: adding"
|
|
226
|
+
commit_green: "feat: add"
|
|
227
|
+
---
|
|
228
|
+
# Plan
|
|
229
|
+
""")
|
|
230
|
+
assert run_cli(repo, "plan", "register", plan)["ok"]
|
|
231
|
+
out = run_cli(repo, "run", "start", "--plan", plan)
|
|
232
|
+
assert out["ok"], out
|
|
233
|
+
assert out["result"]["baselines"] == {"backend": 0}, out["result"]
|
|
@@ -340,7 +340,13 @@ def test_vitest_collection_routes_override_files_to_the_override_command(
|
|
|
340
340
|
assert collection.tests == {
|
|
341
341
|
"backend::contract/api.contract.test.ts > pings the api"
|
|
342
342
|
}
|
|
343
|
-
|
|
343
|
+
# R7.13's requirement is that the override's own command enumerates its files —
|
|
344
|
+
# not that it does so one file at a time. Collection batches per declared suite
|
|
345
|
+
# (issue #27), so the override's command is one of the invocations rather than
|
|
346
|
+
# the first, and carries no file argument.
|
|
347
|
+
assert any(
|
|
348
|
+
c.startswith("npx vitest list --config vitest.contract.config.ts") for c in seen
|
|
349
|
+
), seen
|
|
344
350
|
|
|
345
351
|
# With every suite listable the gate passes: the probe must not manufacture a
|
|
346
352
|
# failure out of a healthy override.
|
|
@@ -23,6 +23,8 @@ from __future__ import annotations
|
|
|
23
23
|
import json
|
|
24
24
|
|
|
25
25
|
from conftest import run_cli, write_plan
|
|
26
|
+
from tddcli import adapters
|
|
27
|
+
from tddcli import config as config_mod
|
|
26
28
|
from tddcli.adapters import base
|
|
27
29
|
|
|
28
30
|
PLAN = """---
|
|
@@ -114,18 +116,25 @@ def test_a_timed_command_is_attributed_to_its_caller(repo, capsys, monkeypatch):
|
|
|
114
116
|
assert "collect" in labels, labels
|
|
115
117
|
|
|
116
118
|
|
|
117
|
-
def
|
|
118
|
-
"""The loop's cost is per file, so its timing has to be too — a single
|
|
119
|
-
cannot say which file is slow.
|
|
119
|
+
def test_the_per_file_collect_fallback_is_attributed_per_file(repo_broken, capsys, monkeypatch):
|
|
120
|
+
"""The fallback loop's cost is per file, so its timing has to be too — a single
|
|
121
|
+
total cannot say which file is slow.
|
|
122
|
+
|
|
123
|
+
Collection batches per suite now (issue #27), so the loop runs only where a
|
|
124
|
+
batch could not account for a file — which is exactly where per-file
|
|
125
|
+
attribution earns its keep. `repo_broken`'s uncollectable module fails the
|
|
126
|
+
batch and drops every file in that project to the loop."""
|
|
120
127
|
monkeypatch.setenv(base.TIMING_ENV, "1")
|
|
121
|
-
|
|
128
|
+
adapters.build(
|
|
129
|
+
config_mod.load(repo_broken).project("verify"), repo_broken
|
|
130
|
+
).collect()
|
|
122
131
|
|
|
123
132
|
collects = [
|
|
124
133
|
entry for entry in _lines(capsys.readouterr().err, "command_timing")
|
|
125
134
|
if entry.get("label") == "collect"
|
|
126
135
|
]
|
|
127
136
|
assert collects, "no collect timings"
|
|
128
|
-
assert any("
|
|
137
|
+
assert any("test_v.py" in entry["command"] for entry in collects), collects
|
|
129
138
|
|
|
130
139
|
|
|
131
140
|
def test_doctor_probes_are_labelled_as_doctor(repo, capsys, monkeypatch):
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|