tdd-cli 0.4.0__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/CHANGELOG.md +15 -0
  2. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/PKG-INFO +1 -1
  3. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/__init__.py +1 -1
  4. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/adapters/base.py +42 -0
  5. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/adapters/pytest_adapter.py +37 -11
  6. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/adapters/vitest_adapter.py +37 -4
  7. tdd_cli-0.4.1/tests/test_batch_collection.py +233 -0
  8. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_suite_overrides.py +7 -1
  9. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_timing_visibility.py +14 -5
  10. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/.gitignore +0 -0
  11. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/LICENSE +0 -0
  12. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/README.md +0 -0
  13. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/SECURITY.md +0 -0
  14. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/README.md +0 -0
  15. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/bash_hook.py +0 -0
  16. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/claude-code-hooks/stop_hook.py +0 -0
  17. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/plan.md +0 -0
  18. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/skills/tdd-drive/README.md +0 -0
  19. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/skills/tdd-drive/SKILL.md +0 -0
  20. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/skills/tdd-handoff/README.md +0 -0
  21. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/examples/skills/tdd-handoff/SKILL.md +0 -0
  22. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/pyproject.toml +0 -0
  23. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/adapters/__init__.py +0 -0
  24. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/advance.py +0 -0
  25. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/cli.py +0 -0
  26. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/config.py +0 -0
  27. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/contract.py +0 -0
  28. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/envelope.py +0 -0
  29. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/fleet.py +0 -0
  30. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/gitutil.py +0 -0
  31. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/identity.py +0 -0
  32. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/leases.py +0 -0
  33. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/ledger.py +0 -0
  34. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/machine.py +0 -0
  35. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/render.py +0 -0
  36. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/snapshot.py +0 -0
  37. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/src/tddcli/staging.py +0 -0
  38. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/conftest.py +0 -0
  39. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_artifact_regeneration.py +0 -0
  40. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_baseline_integrity.py +0 -0
  41. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_config_and_staging.py +0 -0
  42. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_config_drift.py +0 -0
  43. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_contract.py +0 -0
  44. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_doctor_attribution.py +0 -0
  45. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_doctor_blockers.py +0 -0
  46. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_end_to_end.py +0 -0
  47. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_example_plan.py +0 -0
  48. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_failure_clipping.py +0 -0
  49. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_fleet.py +0 -0
  50. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_heartbeat.py +0 -0
  51. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_init_detection.py +0 -0
  52. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_pin_cycles.py +0 -0
  53. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_progress.py +0 -0
  54. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_project_commands.py +0 -0
  55. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_project_env.py +0 -0
  56. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_python_env_managers.py +0 -0
  57. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_refactor_cycles.py +0 -0
  58. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_release_surface.py +0 -0
  59. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_run_claim.py +0 -0
  60. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_single_project_repo.py +0 -0
  61. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_snapshot_and_identity.py +0 -0
  62. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_stub_hint.py +0 -0
  63. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_target_validation.py +0 -0
  64. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_vitest_adapter.py +0 -0
  65. {tdd_cli-0.4.0 → tdd_cli-0.4.1}/tests/test_worker_leases.py +0 -0
@@ -4,6 +4,21 @@ All notable changes to this project are documented here.
4
4
  The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
5
5
  and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
+ ## [0.4.1] - 2026-08-16
8
+
9
+ ### Changed
10
+
11
+ - `collect()` runs one invocation per declared suite instead of one per test
12
+ file, falling back to the per-file loop for anything a batch did not account
13
+ for. Per-file collection was **77% of a real `run start`** — 313 subprocesses
14
+ costing 402s, against 117s to actually run every test — because cost scaled
15
+ with file count at a ~1.08s floor per invocation (the environment manager
16
+ resolving plus the runner booting). Measured 38.8x faster on a 60-file
17
+ project, with an identical collected set. R10.3's guarantee is unchanged: a
18
+ file that fails to collect is still attributed to itself and cannot destroy
19
+ the set, and a file the batch never reports is still collected individually,
20
+ so the set can only match or improve on the old one (#27).
21
+
7
22
  ## [0.4.0] - 2026-08-16
8
23
 
9
24
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.4.0
3
+ Version: 0.4.1
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.4.0"
6
+ __version__ = "0.4.1"
@@ -189,6 +189,48 @@ class Adapter:
189
189
  return "a body that fails loudly, never working logic"
190
190
 
191
191
  def collect(self) -> Collection:
192
+ """Enumerate the project's tests: one invocation per declared suite, with
193
+ the per-file loop kept for what that cannot account for (issue #27).
194
+
195
+ Per file, collection cost scaled with *file count* rather than test count —
196
+ measured at 77% of a whole `run start`, against a floor of ~1.08s per
197
+ invocation that is the environment manager resolving plus the runner
198
+ booting, not collection work.
199
+
200
+ The batch and the loop do not discover the same way: a batch uses the
201
+ runner's own config, the loop walks `test_paths`. So the loop still runs for
202
+ every file the batch did not report — whether the batch failed, returned
203
+ nothing, or simply never mentioned that file. R10.3's guarantee is
204
+ unchanged: one uncollectable file is attributed to itself, and cannot
205
+ destroy the set. What changes is that the healthy case no longer pays for
206
+ the broken one.
207
+ """
208
+ result = Collection()
209
+ unaccounted = {str(p.relative_to(self.root)) for p in self._test_files()}
210
+ for command, env in self._collect_invocations():
211
+ batch = self._collect_batch(command, env)
212
+ if batch is None:
213
+ continue
214
+ tests, files = batch
215
+ result.tests |= tests
216
+ unaccounted -= files
217
+ return self._collect_per_file(unaccounted, result)
218
+
219
+ def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
220
+ """One collection command per declared suite: the default plus each
221
+ override's (R7.13), so an override's files are enumerated with its own
222
+ command and env."""
223
+ raise NotImplementedError
224
+
225
+ def _collect_batch(
226
+ self, command: str, env: dict[str, str] | None
227
+ ) -> tuple[set[str], set[str]] | None:
228
+ """`(test ids, files seen)` for one whole-suite invocation, or None when it
229
+ produced nothing usable — the signal to leave its files to the loop."""
230
+ raise NotImplementedError
231
+
232
+ def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
233
+ """R10.3's loop, now reached only for files no batch accounted for."""
192
234
  raise NotImplementedError
193
235
 
194
236
  def collectable(self) -> GateResult:
@@ -235,23 +235,49 @@ class PytestAdapter(Adapter):
235
235
  " cannot reach them (e.g. `pytest tests/`)."
236
236
  ))
237
237
 
238
- def collect(self) -> Collection:
238
+ def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
239
+ """An override without a `collect_command` collects with its `test_command`
240
+ — `--collect-only` composes with any run command."""
241
+ return [(self._collect_cmd(), self._suite_env(None))] + [
242
+ (ov.collect_command or ov.test_command, self._suite_env(ov))
243
+ for ov in self.project.overrides
244
+ ]
245
+
246
+ def _nodeids(self, out: str) -> tuple[set[str], set[str]]:
247
+ tests: set[str] = set()
248
+ files: set[str] = set()
249
+ for line in out.splitlines():
250
+ line = line.strip()
251
+ if "::" in line and not line.startswith(("=", "-", "no tests")):
252
+ tests.add(self.qualify(line))
253
+ files.add(line.split("::", 1)[0])
254
+ return tests, files
255
+
256
+ def _collect_batch(
257
+ self, command: str, env: dict[str, str] | None
258
+ ) -> tuple[set[str], set[str]] | None:
259
+ code, out, err = run_command(
260
+ f"{command} --collect-only -q", self.root, extra_env=env, label="collect"
261
+ )
262
+ if code != 0:
263
+ return None
264
+ # An empty result needs no special case: it accounts for no files, so every
265
+ # file falls to the loop exactly as a failure would.
266
+ return self._nodeids(out)
267
+
268
+ def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
239
269
  """Per file (R10.3) — one uncollectable module must not destroy the whole set."""
240
- result = Collection()
241
- for path in self._test_files():
242
- rel = path.relative_to(self.root)
243
- base, env = self._collect_cmd_for(str(rel))
270
+ for rel in sorted(rels):
271
+ base, env = self._collect_cmd_for(rel)
244
272
  code, out, err = run_command(
245
- f"{base} --collect-only -q {shlex.quote(str(rel))}",
273
+ f"{base} --collect-only -q {shlex.quote(rel)}",
246
274
  self.root,
247
275
  extra_env=env,
248
276
  label="collect",
249
277
  )
250
278
  if code != 0:
251
- result.failed_files[str(rel)] = (err or out).strip()[:800]
279
+ result.failed_files[rel] = (err or out).strip()[:800]
252
280
  continue
253
- for line in out.splitlines():
254
- line = line.strip()
255
- if "::" in line and not line.startswith(("=", "-", "no tests")):
256
- result.tests.add(self.qualify(line))
281
+ tests, _ = self._nodeids(out)
282
+ result.tests |= tests
257
283
  return result
@@ -223,10 +223,43 @@ class VitestAdapter(Adapter):
223
223
  " config (test.exclude) or scope its include globs."
224
224
  ))
225
225
 
226
- def collect(self) -> Collection:
227
- result = Collection()
228
- for path in self._test_files():
229
- rel = path.relative_to(self.root)
226
+ def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
227
+ """An override with no `collect_command` gets no batch — `vitest list` knows
228
+ nothing of its config, and its files fall to the loop, which records the
229
+ missing `collect_command` against each of them as before."""
230
+ return [(self._collect_cmd(), self._suite_env(None))] + [
231
+ (ov.collect_command, self._suite_env(ov))
232
+ for ov in self.project.overrides if ov.collect_command
233
+ ]
234
+
235
+ def _collect_batch(
236
+ self, command: str, env: dict[str, str] | None
237
+ ) -> tuple[set[str], set[str]] | None:
238
+ """Unlike `_parse_list_output`, which pins every id to the one file it was
239
+ given, a whole-suite listing must read the file from each line — the same
240
+ `file > describe > name` shape, one file per line rather than one per run."""
241
+ code, out, err = run_command(command, self.root, extra_env=env, label="collect")
242
+ if code != 0:
243
+ return None
244
+ tests: set[str] = set()
245
+ files: set[str] = set()
246
+ for line in out.splitlines():
247
+ line = line.strip()
248
+ if " > " not in line:
249
+ continue
250
+ rel, _, remainder = line.partition(" > ")
251
+ full_name = " ".join(part.strip() for part in remainder.split(" > "))
252
+ if rel and full_name:
253
+ tests.add(self.qualify(f"{rel.strip()} > {full_name}"))
254
+ files.add(rel.strip())
255
+ # An empty result needs no special case: it accounts for no files, so every
256
+ # file falls to the loop exactly as a failure would.
257
+ return tests, files
258
+
259
+ def _collect_per_file(self, rels: set[str], result: Collection) -> Collection:
260
+ for rel_str in sorted(rels):
261
+ path = self.root / rel_str
262
+ rel = Path(rel_str)
230
263
  ov = self.project.override_for(str(rel))
231
264
  if ov is not None and not ov.collect_command:
232
265
  result.failed_files[str(rel)] = (
@@ -0,0 +1,233 @@
1
+ """Collection asks the runner once, not once per file (issue #27).
2
+
3
+ Measured on a real repo: 313 per-file subprocesses cost 402s, 77% of a whole
4
+ `run start`, while *running* all the tests cost 117s. The floor per invocation
5
+ was 1.08s — the environment manager resolving plus the runner booting, paid
6
+ again per file. A single file's tests enumerate in milliseconds.
7
+
8
+ Both adapters already had a whole-suite probe (`collectable()`); collection now
9
+ uses that shape and keeps the per-file loop for the case it exists to serve.
10
+
11
+ The rule that makes this safe: **a file the batch does not account for gets
12
+ exactly the old per-file treatment.** Batch fails, batch is empty, batch skips a
13
+ file the registry declares — each falls through to the loop, so the collected set
14
+ can only match or improve on the old one, never silently shrink (R10.3/R10.4).
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from pathlib import Path
20
+
21
+ from conftest import run_cli, write_plan
22
+ from tddcli import adapters
23
+ from tddcli import config as config_mod
24
+
25
+
26
+ def _adapter(repo: Path, project: str = "backend"):
27
+ return adapters.build(config_mod.load(repo).project(project), repo)
28
+
29
+
30
+ def _counting(monkeypatch, module=None):
31
+ """Wrap the real `run_command`, recording every command it is asked to run."""
32
+ seen: list[str] = []
33
+ real = adapters.base.run_command
34
+
35
+ def wrapper(command, cwd, timeout=1800, extra_env=None, label=None):
36
+ seen.append(command)
37
+ return real(command, cwd, timeout=timeout, extra_env=extra_env, label=label)
38
+
39
+ monkeypatch.setattr(module or adapters.base, "run_command", wrapper)
40
+ return seen
41
+
42
+
43
+ def _write_tests(repo: Path, n: int) -> None:
44
+ for i in range(n):
45
+ (repo / "backend" / "tests" / f"test_gen{i}.py").write_text(
46
+ f"def test_gen{i}():\n assert True\n"
47
+ )
48
+
49
+
50
+ def test_a_healthy_project_is_collected_in_one_invocation(repo, monkeypatch):
51
+ """The whole point: cost stops scaling with file count."""
52
+ _write_tests(repo, 6)
53
+ seen = _counting(monkeypatch, adapters.pytest_adapter)
54
+
55
+ collected = _adapter(repo).collect()
56
+
57
+ assert len(seen) == 1, seen
58
+ assert len(collected.tests) == 7, collected.tests # 6 generated + test_smoke
59
+ assert collected.failed_files == {}
60
+
61
+
62
+ def test_the_batch_finds_the_same_tests_the_per_file_loop_would(repo, monkeypatch):
63
+ """Equivalence is the whole risk of this change: batch enumeration uses the
64
+ runner's own discovery, the per-file loop uses `test_paths` globs."""
65
+ _write_tests(repo, 4)
66
+ adapter = _adapter(repo)
67
+
68
+ batched = adapter.collect().tests
69
+ per_file = adapter._collect_per_file(
70
+ {str(p.relative_to(adapter.root)) for p in adapter._test_files()},
71
+ adapters.base.Collection(),
72
+ ).tests
73
+
74
+ assert batched == per_file, batched ^ per_file
75
+
76
+
77
+ def test_a_file_the_batch_never_reported_is_collected_individually(repo, monkeypatch):
78
+ """A file matching `test_paths` that the runner's config excludes must not
79
+ vanish from the set — a quietly smaller baseline is worse than a slow one."""
80
+ _write_tests(repo, 3)
81
+ adapter = _adapter(repo)
82
+ real = adapters.base.run_command
83
+ seen: list[str] = []
84
+
85
+ def hide_one(command, cwd, timeout=1800, extra_env=None, label=None):
86
+ seen.append(command)
87
+ code, out, err = real(command, cwd, timeout=timeout, extra_env=extra_env, label=label)
88
+ if "test_gen1.py" not in command: # the batch "forgets" this file
89
+ out = "\n".join(
90
+ line for line in out.splitlines() if "test_gen1.py" not in line
91
+ )
92
+ return code, out, err
93
+
94
+ monkeypatch.setattr(adapters.pytest_adapter, "run_command", hide_one)
95
+ collected = adapter.collect()
96
+
97
+ assert any("test_gen1.py" in t for t in collected.tests), collected.tests
98
+ # Exactly one rescue invocation, naming that file — not a whole re-sweep.
99
+ per_file = [c for c in seen if "test_gen1.py" in c]
100
+ assert len(per_file) == 1, seen
101
+
102
+
103
+ def test_a_failing_batch_falls_back_to_per_file_attribution(repo_broken):
104
+ """R10.3's purpose, preserved: one uncollectable module must not destroy the
105
+ set, and the failure must name the file it came from."""
106
+ adapter = _adapter(repo_broken, "verify")
107
+ collected = adapter.collect()
108
+
109
+ assert collected.failed_files, "a broken module must be attributed"
110
+ assert any("test_v.py" in name for name in collected.failed_files), collected.failed_files
111
+
112
+
113
+ def test_one_broken_file_does_not_lose_its_healthy_neighbours(repo_broken):
114
+ """The batch fails wholesale, so the fallback has to recover everything else."""
115
+ (repo_broken / "verify" / "tests" / "test_ok.py").write_text(
116
+ "def test_ok():\n assert True\n"
117
+ )
118
+ collected = _adapter(repo_broken, "verify").collect()
119
+
120
+ assert any("test_ok.py" in t for t in collected.tests), collected.tests
121
+ assert any("test_v.py" in f for f in collected.failed_files), collected.failed_files
122
+
123
+
124
+ def test_override_files_are_collected_by_their_own_suite(repo, monkeypatch):
125
+ """R7.13: an override's files are enumerated with the override's command and
126
+ env, so batching must stay one invocation *per declared suite*."""
127
+ (repo / "backend" / "contract").mkdir()
128
+ (repo / "backend" / "contract" / "test_api.py").write_text(
129
+ "def test_ping():\n assert True\n"
130
+ )
131
+ (repo / "tdd.toml").write_text(
132
+ "[project.backend]\n"
133
+ 'root = "backend"\n'
134
+ 'adapter = "pytest"\n'
135
+ 'test_paths = ["tests/"]\n'
136
+ 'test_command = "pytest tests"\n'
137
+ "[[project.backend.override]]\n"
138
+ 'pattern = "contract/"\n'
139
+ 'test_command = "pytest contract"\n'
140
+ )
141
+ seen = _counting(monkeypatch, adapters.pytest_adapter)
142
+
143
+ collected = _adapter(repo).collect()
144
+
145
+ assert len(seen) == 2, seen # default suite + override suite, no per-file
146
+ assert any("contract/test_api.py" in t for t in collected.tests), collected.tests
147
+
148
+
149
+ def test_partial_output_from_a_failed_batch_is_not_trusted(repo, monkeypatch):
150
+ """A runner that aborts mid-collection still prints what it reached. Accepting
151
+ that would take a file's *incomplete* test list as final — and reconciliation
152
+ could not save it, because the file was mentioned. Every file is re-collected
153
+ instead, which is the old cost only in the case that was already broken."""
154
+ _write_tests(repo, 3)
155
+ adapter = _adapter(repo)
156
+ real = adapters.base.run_command
157
+ seen: list[str] = []
158
+
159
+ def failing_batch(command, cwd, timeout=1800, extra_env=None, label=None):
160
+ seen.append(command)
161
+ if "test_gen" not in command and "test_smoke" not in command:
162
+ # whole-suite invocation: aborts, having reached one file
163
+ return 1, "tests/test_gen0.py::test_gen0\n", "INTERNALERROR"
164
+ return real(command, cwd, timeout=timeout, extra_env=extra_env, label=label)
165
+
166
+ monkeypatch.setattr(adapters.pytest_adapter, "run_command", failing_batch)
167
+ collected = adapter.collect()
168
+
169
+ assert any("test_gen0.py" in c for c in seen[1:]), seen
170
+ assert len(collected.tests) == 4, collected.tests # 3 generated + test_smoke
171
+
172
+
173
+ def test_vitest_partial_output_from_a_failed_batch_is_not_trusted(repo_multi, monkeypatch):
174
+ """Same rule for the vitest adapter, whose listing has no exit-code-free way to
175
+ tell a complete run from an aborted one."""
176
+ (repo_multi / "frontend" / "a.test.ts").write_text("")
177
+ (repo_multi / "frontend" / "b.test.ts").write_text("")
178
+ adapter = _adapter(repo_multi, "frontend")
179
+ seen: list[str] = []
180
+
181
+ def failing_batch(command, cwd, timeout=1800, extra_env=None, label=None):
182
+ seen.append(command)
183
+ if command.endswith("list"): # whole-suite invocation
184
+ return 1, "a.test.ts > alpha\n", "crashed"
185
+ return 0, f"{command.rsplit(' ', 1)[1]} > rescued", ""
186
+
187
+ monkeypatch.setattr(adapters.vitest_adapter, "run_command", failing_batch)
188
+ collected = adapter.collect()
189
+
190
+ assert collected.tests == {
191
+ "frontend::a.test.ts > rescued", "frontend::b.test.ts > rescued",
192
+ }, collected.tests
193
+
194
+
195
+ def test_vitest_batch_attributes_each_id_to_its_own_file(repo_multi, monkeypatch):
196
+ """`_parse_list_output` pinned every id to one path, which is right per-file
197
+ and wrong for a whole-suite listing."""
198
+ (repo_multi / "frontend" / "a.test.ts").write_text(
199
+ "import {test, expect} from 'vitest'\ntest('alpha', () => expect(1).toBe(1))\n"
200
+ )
201
+ (repo_multi / "frontend" / "b.test.ts").write_text(
202
+ "import {test, expect} from 'vitest'\ntest('beta', () => expect(1).toBe(1))\n"
203
+ )
204
+ adapter = _adapter(repo_multi, "frontend")
205
+ monkeypatch.setattr(
206
+ adapters.vitest_adapter, "run_command",
207
+ lambda command, cwd, timeout=1800, extra_env=None, label=None: (
208
+ 0, "a.test.ts > alpha\nb.test.ts > beta\n", ""
209
+ ),
210
+ )
211
+ collected = adapter.collect()
212
+
213
+ assert collected.tests == {"frontend::a.test.ts > alpha", "frontend::b.test.ts > beta"}
214
+
215
+
216
+ def test_run_start_still_reports_the_same_baseline(repo, monkeypatch):
217
+ """End to end, through the command that pays for this."""
218
+ _write_tests(repo, 3)
219
+ plan = write_plan(repo, """---
220
+ cycles:
221
+ - n: 1
222
+ project: backend
223
+ title: "adding two numbers"
224
+ test: "tests/test_add.py::test_add_two_numbers"
225
+ commit_red: "test: adding"
226
+ commit_green: "feat: add"
227
+ ---
228
+ # Plan
229
+ """)
230
+ assert run_cli(repo, "plan", "register", plan)["ok"]
231
+ out = run_cli(repo, "run", "start", "--plan", plan)
232
+ assert out["ok"], out
233
+ assert out["result"]["baselines"] == {"backend": 0}, out["result"]
@@ -340,7 +340,13 @@ def test_vitest_collection_routes_override_files_to_the_override_command(
340
340
  assert collection.tests == {
341
341
  "backend::contract/api.contract.test.ts > pings the api"
342
342
  }
343
- assert seen[0].startswith("npx vitest list --config vitest.contract.config.ts ")
343
+ # R7.13's requirement is that the override's own command enumerates its files —
344
+ # not that it does so one file at a time. Collection batches per declared suite
345
+ # (issue #27), so the override's command is one of the invocations rather than
346
+ # the first, and carries no file argument.
347
+ assert any(
348
+ c.startswith("npx vitest list --config vitest.contract.config.ts") for c in seen
349
+ ), seen
344
350
 
345
351
  # With every suite listable the gate passes: the probe must not manufacture a
346
352
  # failure out of a healthy override.
@@ -23,6 +23,8 @@ from __future__ import annotations
23
23
  import json
24
24
 
25
25
  from conftest import run_cli, write_plan
26
+ from tddcli import adapters
27
+ from tddcli import config as config_mod
26
28
  from tddcli.adapters import base
27
29
 
28
30
  PLAN = """---
@@ -114,18 +116,25 @@ def test_a_timed_command_is_attributed_to_its_caller(repo, capsys, monkeypatch):
114
116
  assert "collect" in labels, labels
115
117
 
116
118
 
117
- def test_the_per_file_collect_loop_is_attributed_per_file(repo, capsys, monkeypatch):
118
- """The loop's cost is per file, so its timing has to be too — a single total
119
- cannot say which file is slow."""
119
+ def test_the_per_file_collect_fallback_is_attributed_per_file(repo_broken, capsys, monkeypatch):
120
+ """The fallback loop's cost is per file, so its timing has to be too — a single
121
+ total cannot say which file is slow.
122
+
123
+ Collection batches per suite now (issue #27), so the loop runs only where a
124
+ batch could not account for a file — which is exactly where per-file
125
+ attribution earns its keep. `repo_broken`'s uncollectable module fails the
126
+ batch and drops every file in that project to the loop."""
120
127
  monkeypatch.setenv(base.TIMING_ENV, "1")
121
- assert _start(repo)["ok"]
128
+ adapters.build(
129
+ config_mod.load(repo_broken).project("verify"), repo_broken
130
+ ).collect()
122
131
 
123
132
  collects = [
124
133
  entry for entry in _lines(capsys.readouterr().err, "command_timing")
125
134
  if entry.get("label") == "collect"
126
135
  ]
127
136
  assert collects, "no collect timings"
128
- assert any("test_smoke.py" in entry["command"] for entry in collects), collects
137
+ assert any("test_v.py" in entry["command"] for entry in collects), collects
129
138
 
130
139
 
131
140
  def test_doctor_probes_are_labelled_as_doctor(repo, capsys, monkeypatch):
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes