tdd-cli 0.7.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/CHANGELOG.md +52 -0
  2. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/PKG-INFO +94 -5
  3. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/README.md +93 -4
  4. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/__init__.py +1 -1
  5. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/base.py +8 -0
  6. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/vitest_adapter.py +62 -32
  7. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/advance.py +51 -5
  8. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/cli.py +107 -27
  9. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/config.py +13 -0
  10. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/contract.py +15 -0
  11. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/ledger.py +79 -3
  12. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/machine.py +33 -5
  13. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/render.py +6 -0
  14. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/staging.py +6 -0
  15. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/conftest.py +3 -0
  16. tdd_cli-0.8.0/tests/test_ancillary_files.py +72 -0
  17. tdd_cli-0.8.0/tests/test_artifact_regeneration.py +166 -0
  18. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_baseline_integrity.py +266 -1
  19. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_config_and_staging.py +15 -0
  20. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_contract.py +120 -0
  21. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_heartbeat.py +49 -0
  22. tdd_cli-0.8.0/tests/test_id_normalisation.py +127 -0
  23. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_release_surface.py +15 -0
  24. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_snapshot_and_identity.py +45 -0
  25. tdd_cli-0.8.0/tests/test_undeclared_close_gate.py +106 -0
  26. tdd_cli-0.8.0/tests/test_undeclared_dedup.py +168 -0
  27. tdd_cli-0.7.0/tests/test_artifact_regeneration.py +0 -76
  28. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/.gitignore +0 -0
  29. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/LICENSE +0 -0
  30. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/SECURITY.md +0 -0
  31. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/README.md +0 -0
  32. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/bash_hook.py +0 -0
  33. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/stop_hook.py +0 -0
  34. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/plan.md +0 -0
  35. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/skills/tdd-drive/README.md +0 -0
  36. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/skills/tdd-drive/SKILL.md +0 -0
  37. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/skills/tdd-handoff/README.md +0 -0
  38. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
  39. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/pyproject.toml +0 -0
  40. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/__init__.py +0 -0
  41. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/exec_adapter.py +0 -0
  42. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/gradle_adapter.py +0 -0
  43. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/pytest_adapter.py +0 -0
  44. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/xctest_adapter.py +0 -0
  45. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/envelope.py +0 -0
  46. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/fleet.py +0 -0
  47. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/gitutil.py +0 -0
  48. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/identity.py +0 -0
  49. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/leases.py +0 -0
  50. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/snapshot.py +0 -0
  51. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_batch_collection.py +0 -0
  52. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_concurrent_advance.py +0 -0
  53. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_config_drift.py +0 -0
  54. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_doctor_attribution.py +0 -0
  55. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_doctor_blockers.py +0 -0
  56. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_end_to_end.py +0 -0
  57. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_example_plan.py +0 -0
  58. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_exec_adapter.py +0 -0
  59. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_failure_clipping.py +0 -0
  60. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_fleet.py +0 -0
  61. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_gradle_adapter.py +0 -0
  62. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_init_detection.py +0 -0
  63. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_named_leases_and_timeout.py +0 -0
  64. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_pin_cycles.py +0 -0
  65. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_progress.py +0 -0
  66. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_project_commands.py +0 -0
  67. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_project_env.py +0 -0
  68. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_python_env_managers.py +0 -0
  69. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_refactor_cycles.py +0 -0
  70. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_run_claim.py +0 -0
  71. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_single_project_repo.py +0 -0
  72. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_stub_hint.py +0 -0
  73. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_suite_overrides.py +0 -0
  74. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_target_validation.py +0 -0
  75. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_timing_visibility.py +0 -0
  76. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_vitest_adapter.py +0 -0
  77. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_worker_leases.py +0 -0
  78. {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_xctest_adapter.py +0 -0
@@ -6,6 +6,58 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.8.0] - 2026-08-28
10
+
11
+ ### Added
12
+
13
+ - **`--reuse-baselines` caches baseline probes by content hash.** `run start`
14
+ can skip re-probing a project whose tree is unchanged: a probe result is
15
+ cached keyed by `(project, tree_hash, config_sha)` — where `tree_hash` folds in
16
+ every upstream producer root — and an identical rerun emits `baseline_reused`
17
+ instead of `baseline_captured`, reusing the cached failing set and collection
18
+ snapshot rather than re-running the suite. Provenance is recorded on the
19
+ baseline row (`source`), and `--reuse-max-age` re-probes any entry older than
20
+ the given age. Off by default; the cache stays empty unless the flag is passed
21
+ (#45, #59).
22
+
23
+ - **`--baseline-jobs` parallelizes baseline probing.** `run start` probes each
24
+ project's baseline under a bounded `ThreadPoolExecutor` when `--baseline-jobs`
25
+ is greater than 1 (default 1, must be >= 1). The `baseline_captured` heartbeat
26
+ survives the pool, and a worker probe that raises becomes an attributed failure
27
+ that aborts cleanly and frees the worktree rather than wedging it (#46, #62).
28
+
29
+ - **Plan-level `ancillary_files`.** A top-level front-matter key declaring
30
+ cross-project or companion paths a plan touches (README, generated fixtures,
31
+ sibling-project files). Declared ancillary paths are bucketed into their own
32
+ staging bucket, committed with the cycle, and fire no `undeclared_file_touched`
33
+ event. Validated as a list of strings at registration and persisted to the
34
+ ledger via a v5→v6 migration (#70, #80).
35
+
36
+ - **Run-close gate on undeclared touched paths.** `run close` now blocks when a
37
+ path previously flagged as `undeclared_file_touched` is still dirty in the
38
+ worktree, so undeclared changes cannot slip through at the end of a run. A
39
+ flagged path that was since committed does not block; one that has vanished is
40
+ reported via a new `undeclared_file_dropped` event rather than blocking (#69,
41
+ #81).
42
+
43
+ - **Reserved per-cycle `meta:` passthrough.** A cycle may carry an authored
44
+ `meta:` mapping in the plan front-matter; it round-trips through storage
45
+ unchanged and is available for plan-time metadata. A non-mapping `meta:`
46
+ hard-fails registration with a `ContractError` (#58, #65).
47
+
48
+ ### Fixed
49
+
50
+ - `undeclared_file_touched` is deduplicated within a cycle, so a path touched
51
+ across multiple phases no longer floods the cycle with repeated events (#55,
52
+ #61).
53
+ - vitest test ids are normalised on the describe/test separator before matching,
54
+ so a formatting-only difference between a declared target and the observed
55
+ verdict is no longer reported as a spurious `declared_test_mismatch` (#57,
56
+ #63).
57
+ - The `stale_artifact` event is suppressed when the tool auto-regenerates the
58
+ artifact and commits it, so a successful regeneration no longer also emits a
59
+ staleness warning (#64).
60
+
9
61
  ## [0.7.0] - 2026-08-23
10
62
 
11
63
  ### Fixed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.7.0
3
+ Version: 0.8.0
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -227,6 +227,8 @@ cycles:
227
227
  stub_expected: ["app/exception_map.py"]
228
228
  commit_red: "test: unmapped exception is not swallowed"
229
229
  commit_green: "feat: domain exception map skeleton"
230
+ meta: # optional authored-at-plan-time metadata; opaque to the tool
231
+ covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
230
232
  - n: 8
231
233
  project: backend
232
234
  pin_cycle: true # characterisation; passes on arrival by design
@@ -238,9 +240,51 @@ cycles:
238
240
  - "backend::tests/test_openapi.py::test_upload_body_schema"
239
241
  - "frontend::services/__tests__/upload.test.ts > matches contract"
240
242
  annotation_keys: ["literal_detail_handlers_kept"]
243
+ ancillary_files:
244
+ - frontend/src/api/registerClient.ts # type-break from regenerated client
245
+ - docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
241
246
  ---
242
247
  ```
243
248
 
249
+ **Top-level keys:**
250
+
251
+ `annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
252
+ gate checks that every key is present before the plan can be marked complete.
253
+
254
+ `ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
255
+ touch outside any registered project root (cross-project ripples, companion documents).
256
+ Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
257
+ matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
258
+ and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
259
+ not on the list still fires `undeclared_file_touched` exactly as today. This is a
260
+ plan-level list only; per-cycle overrides are a planned follow-up.
261
+
262
+ ### Run-close gate for undeclared file touches
263
+
264
+ When the last declared cycle closes, the tool gathers every path that appeared in any
265
+ `undeclared_file_touched` event across the run and checks the worktree:
266
+
267
+ - **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
268
+ the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
269
+ commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
270
+ and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
271
+ - **Committed during the run** → clean; the run completes normally.
272
+ - **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
273
+ emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
274
+ makes a silent drop visible without blocking, since the file is already gone.
275
+
276
+ The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
277
+ so they are never seen by the gate.
278
+
279
+ **Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
280
+ `stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
281
+ `commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
282
+ `meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
283
+ but its *contents* are opaque to the tool — any key/value pairs are accepted and
284
+ round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
285
+ authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
286
+ needs to read back. Other unknown per-cycle keys are silently ignored.
287
+
244
288
  Absent front-matter is legitimate — the run proceeds as `undeclared` with
245
289
  `--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
246
290
  hard-fails registration: it is almost always a defect in the planning process, and that
@@ -418,6 +462,50 @@ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
418
462
  Use this when a cycle may edit files outside the predicted reachable set and you want every
419
463
  project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
420
464
 
465
+ ### Reusing baselines across runs (R9.5e)
466
+
467
+ On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
468
+ wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
469
+ when nothing in the project changed:
470
+
471
+ tdd run start --plan tasks/plan.md --reuse-baselines
472
+
473
+ The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
474
+ If the key matches a previous entry, the cached failing set and collection snapshot are used —
475
+ no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
476
+ `source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
477
+ projects were skipped. Without the flag, the cache is neither read nor written.
478
+
479
+ A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
480
+ (see below) — reuse is loud by design, never silent.
481
+
482
+ To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
483
+
484
+ tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
485
+
486
+ Entries older than the given threshold are ignored and re-probed fresh.
487
+
488
+ ### Parallel baseline probing (R9.5f)
489
+
490
+ By default `run start` probes one project at a time. On a repo with many independent projects the
491
+ wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
492
+ projects concurrently:
493
+
494
+ tdd run start --plan tasks/plan.md --baseline-jobs 4
495
+
496
+ Each probe is independent — one adapter instance per project — so concurrency does not affect
497
+ which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
498
+ heartbeat is still emitted per project as each probe completes.
499
+
500
+ The default is `--baseline-jobs 1` (serial). Raise it deliberately:
501
+
502
+ - **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
503
+ - **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
504
+ or declare a `lease` name — leased suites serialize automatically even inside the pool.
505
+
506
+ If any probe fails, `run start` returns a failure attributed to that project and no run row is
507
+ created, so the worktree is immediately retryable.
508
+
421
509
  ### When a sweep reaches an un-baselined project (R9.5d)
422
510
 
423
511
  If an edit during a run touches a file owned by an artifact that was outside the predicted
@@ -472,10 +560,11 @@ The CLI cannot compel an agent — only the harness can.
472
560
  artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
473
561
  `advance` refuses an unchanged tree unless `--retry`.
474
562
 
475
- **Recorded, never blocked:** non-stub writes during RED, undeclared file touches, scope
476
- divergence, extra attempts. Prevention rules with edge cases produce false denials, and a
477
- blocked agent improvises around them — putting it right back in the reporting path the
478
- tool exists to keep it out of.
563
+ **Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
564
+ scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
565
+ and a blocked agent improvises around them — putting it right back in the reporting path
566
+ the tool exists to keep it out of. Undeclared file touches are an exception at run close:
567
+ see the run-close gate above.
479
568
 
480
569
  **Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
481
570
  stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
@@ -199,6 +199,8 @@ cycles:
199
199
  stub_expected: ["app/exception_map.py"]
200
200
  commit_red: "test: unmapped exception is not swallowed"
201
201
  commit_green: "feat: domain exception map skeleton"
202
+ meta: # optional authored-at-plan-time metadata; opaque to the tool
203
+ covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
202
204
  - n: 8
203
205
  project: backend
204
206
  pin_cycle: true # characterisation; passes on arrival by design
@@ -210,9 +212,51 @@ cycles:
210
212
  - "backend::tests/test_openapi.py::test_upload_body_schema"
211
213
  - "frontend::services/__tests__/upload.test.ts > matches contract"
212
214
  annotation_keys: ["literal_detail_handlers_kept"]
215
+ ancillary_files:
216
+ - frontend/src/api/registerClient.ts # type-break from regenerated client
217
+ - docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
213
218
  ---
214
219
  ```
215
220
 
221
+ **Top-level keys:**
222
+
223
+ `annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
224
+ gate checks that every key is present before the plan can be marked complete.
225
+
226
+ `ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
227
+ touch outside any registered project root (cross-project ripples, companion documents).
228
+ Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
229
+ matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
230
+ and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
231
+ not on the list still fires `undeclared_file_touched` exactly as today. This is a
232
+ plan-level list only; per-cycle overrides are a planned follow-up.
233
+
234
+ ### Run-close gate for undeclared file touches
235
+
236
+ When the last declared cycle closes, the tool gathers every path that appeared in any
237
+ `undeclared_file_touched` event across the run and checks the worktree:
238
+
239
+ - **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
240
+ the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
241
+ commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
242
+ and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
243
+ - **Committed during the run** → clean; the run completes normally.
244
+ - **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
245
+ emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
246
+ makes a silent drop visible without blocking, since the file is already gone.
247
+
248
+ The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
249
+ so they are never seen by the gate.
250
+
251
+ **Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
252
+ `stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
253
+ `commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
254
+ `meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
255
+ but its *contents* are opaque to the tool — any key/value pairs are accepted and
256
+ round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
257
+ authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
258
+ needs to read back. Other unknown per-cycle keys are silently ignored.
259
+
216
260
  Absent front-matter is legitimate — the run proceeds as `undeclared` with
217
261
  `--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
218
262
  hard-fails registration: it is almost always a defect in the planning process, and that
@@ -390,6 +434,50 @@ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
390
434
  Use this when a cycle may edit files outside the predicted reachable set and you want every
391
435
  project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
392
436
 
437
+ ### Reusing baselines across runs (R9.5e)
438
+
439
+ On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
440
+ wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
441
+ when nothing in the project changed:
442
+
443
+ tdd run start --plan tasks/plan.md --reuse-baselines
444
+
445
+ The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
446
+ If the key matches a previous entry, the cached failing set and collection snapshot are used —
447
+ no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
448
+ `source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
449
+ projects were skipped. Without the flag, the cache is neither read nor written.
450
+
451
+ A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
452
+ (see below) — reuse is loud by design, never silent.
453
+
454
+ To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
455
+
456
+ tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
457
+
458
+ Entries older than the given threshold are ignored and re-probed fresh.
459
+
460
+ ### Parallel baseline probing (R9.5f)
461
+
462
+ By default `run start` probes one project at a time. On a repo with many independent projects the
463
+ wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
464
+ projects concurrently:
465
+
466
+ tdd run start --plan tasks/plan.md --baseline-jobs 4
467
+
468
+ Each probe is independent — one adapter instance per project — so concurrency does not affect
469
+ which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
470
+ heartbeat is still emitted per project as each probe completes.
471
+
472
+ The default is `--baseline-jobs 1` (serial). Raise it deliberately:
473
+
474
+ - **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
475
+ - **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
476
+ or declare a `lease` name — leased suites serialize automatically even inside the pool.
477
+
478
+ If any probe fails, `run start` returns a failure attributed to that project and no run row is
479
+ created, so the worktree is immediately retryable.
480
+
393
481
  ### When a sweep reaches an un-baselined project (R9.5d)
394
482
 
395
483
  If an edit during a run touches a file owned by an artifact that was outside the predicted
@@ -444,10 +532,11 @@ The CLI cannot compel an agent — only the harness can.
444
532
  artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
445
533
  `advance` refuses an unchanged tree unless `--retry`.
446
534
 
447
- **Recorded, never blocked:** non-stub writes during RED, undeclared file touches, scope
448
- divergence, extra attempts. Prevention rules with edge cases produce false denials, and a
449
- blocked agent improvises around them — putting it right back in the reporting path the
450
- tool exists to keep it out of.
535
+ **Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
536
+ scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
537
+ and a blocked agent improvises around them — putting it right back in the reporting path
538
+ the tool exists to keep it out of. Undeclared file touches are an exception at run close:
539
+ see the run-close gate above.
451
540
 
452
541
  **Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
453
542
  stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.7.0"
6
+ __version__ = "0.8.0"
@@ -139,6 +139,14 @@ class Adapter:
139
139
  prefix = f"{self.project.name}::"
140
140
  return qualified[len(prefix) :] if qualified.startswith(prefix) else qualified
141
141
 
142
+ def normalise_id(self, test_id: str) -> str:
143
+ """Return the canonical form of a declared target id for matching against collected ids.
144
+
145
+ The default is an identity — subclasses override when the runner's collected
146
+ ids differ from a natural human spelling (e.g. vitest's describe/test separator).
147
+ """
148
+ return test_id
149
+
142
150
  def run(self, target: str | None = None) -> Verdict:
143
151
  raise NotImplementedError
144
152
 
@@ -45,7 +45,25 @@ class VitestAdapter(Adapter):
45
45
  name = "vitest"
46
46
 
47
47
  def stub_hint(self) -> str:
48
- return "`throw new Error(\"not implemented\")` in every body"
48
+ return '`throw new Error("not implemented")` in every body'
49
+
50
+ def normalise_id(self, test_id: str) -> str:
51
+ """Canonicalise the describe/test separator for target matching.
52
+
53
+ vitest's `fullName` joins ancestor titles and the test title with a space,
54
+ so collected ids look like `frontend::a.test.ts > someHelper formats a value`.
55
+ A planner naturally writes ` > ` at each nesting level. Treating both as
56
+ equivalent prevents a formatting-only difference from producing NOT_FOUND.
57
+
58
+ The structural ` > ` between the file and the name is preserved; only the
59
+ ` > ` separators inside the name part are collapsed to a space.
60
+ """
61
+ raw = self.strip(test_id)
62
+ file_part, sep, remainder = raw.partition(" > ")
63
+ if not sep:
64
+ return test_id
65
+ name = " ".join(part.strip() for part in remainder.split(" > "))
66
+ return self.qualify(f"{file_part} > {name}")
49
67
 
50
68
  def _id_for(self, suite_path: str, full_name: str) -> str:
51
69
  abs_path = Path(suite_path)
@@ -72,18 +90,18 @@ class VitestAdapter(Adapter):
72
90
  code, out, err = self._run_suite(f"{base} --reporter=json", extra_env)
73
91
  report = _extract_json(out)
74
92
  if report is None:
75
- verdict.error = (
76
- f"`{base}` produced no JSON output: {(err or out)[:500]}"
77
- )
93
+ verdict.error = f"`{base}` produced no JSON output: {(err or out)[:500]}"
78
94
  return verdict
79
95
  verdict.duration_ms += int(report.get("duration") or 0)
80
96
  results = report.get("testResults", [])
81
97
  suites.extend(results)
82
- suite_ids.append({
83
- self._id_for(s.get("name", ""), t["fullName"])
84
- for s in results
85
- for t in s.get("assertionResults", [])
86
- })
98
+ suite_ids.append(
99
+ {
100
+ self._id_for(s.get("name", ""), t["fullName"])
101
+ for s in results
102
+ for t in s.get("assertionResults", [])
103
+ }
104
+ )
87
105
 
88
106
  overlap = _suite_overlap(suite_ids)
89
107
  if overlap:
@@ -108,17 +126,23 @@ class VitestAdapter(Adapter):
108
126
  verdict.target_outcome = NOT_FOUND
109
127
  return verdict
110
128
 
111
- if target in verdict.passed:
129
+ ntarget = self.normalise_id(target)
130
+ passed_norm = {self.normalise_id(t): t for t in verdict.passed}
131
+ failed_norm = {self.normalise_id(t): t for t in verdict.failed}
132
+
133
+ if ntarget in passed_norm:
112
134
  verdict.target_outcome = PASSED
113
135
  return verdict
114
- if target in verdict.failed:
136
+ if ntarget in failed_norm:
115
137
  verdict.target_outcome = FAILED
116
138
  for suite in suites:
117
139
  for t in suite.get("assertionResults", []):
118
- if self._id_for(suite.get("name", ""), t["fullName"]) == target:
140
+ if (
141
+ self.normalise_id(self._id_for(suite.get("name", ""), t["fullName"]))
142
+ == ntarget
143
+ ):
119
144
  verdict.target_failure = "\n".join(
120
- clip_failure(m, 600)
121
- for m in t.get("failureMessages", [])[:3]
145
+ clip_failure(m, 600) for m in t.get("failureMessages", [])[:3]
122
146
  )
123
147
  return verdict
124
148
 
@@ -206,22 +230,26 @@ class VitestAdapter(Adapter):
206
230
  code, out, err = run_command(
207
231
  self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
208
232
  )
209
- reached = sorted({
210
- f for f in (
211
- line.strip().partition(" > ")[0]
212
- for line in out.splitlines()
213
- if " > " in line
214
- )
215
- if self.project.override_for(f)
216
- })
233
+ reached = sorted(
234
+ {
235
+ f
236
+ for f in (
237
+ line.strip().partition(" > ")[0] for line in out.splitlines() if " > " in line
238
+ )
239
+ if self.project.override_for(f)
240
+ }
241
+ )
217
242
  if not reached:
218
243
  return GateResult(ok=True)
219
- return GateResult(ok=False, output=(
220
- "the default config's discovery reaches files an override owns, so"
221
- " suite runs would observe them without the override's command/env:"
222
- f" {', '.join(reached[:5])}. Exclude them from the default vitest"
223
- " config (test.exclude) or scope its include globs."
224
- ))
244
+ return GateResult(
245
+ ok=False,
246
+ output=(
247
+ "the default config's discovery reaches files an override owns, so"
248
+ " suite runs would observe them without the override's command/env:"
249
+ f" {', '.join(reached[:5])}. Exclude them from the default vitest"
250
+ " config (test.exclude) or scope its include globs."
251
+ ),
252
+ )
225
253
 
226
254
  def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
227
255
  """An override with no `collect_command` gets no batch — `vitest list` knows
@@ -229,7 +257,8 @@ class VitestAdapter(Adapter):
229
257
  missing `collect_command` against each of them as before."""
230
258
  return [(self._collect_cmd(), self._suite_env(None))] + [
231
259
  (ov.collect_command, self._suite_env(ov))
232
- for ov in self.project.overrides if ov.collect_command
260
+ for ov in self.project.overrides
261
+ if ov.collect_command
233
262
  ]
234
263
 
235
264
  def _collect_batch(
@@ -270,7 +299,9 @@ class VitestAdapter(Adapter):
270
299
  base = ov.collect_command if ov else self._collect_cmd()
271
300
  env = self._suite_env(ov)
272
301
  code, out, err = run_command(
273
- f"{base} {shlex.quote(str(rel))}", self.root, extra_env=env,
302
+ f"{base} {shlex.quote(str(rel))}",
303
+ self.root,
304
+ extra_env=env,
274
305
  label="collect",
275
306
  )
276
307
 
@@ -290,7 +321,6 @@ class VitestAdapter(Adapter):
290
321
  # Zero tests from a file that exists is a tooling failure, not an
291
322
  # empty file — record it rather than silently collecting nothing.
292
323
  result.failed_files[str(rel)] = (
293
- f"no tests parsed from `{base}` (exit {code}): "
294
- + (err or out).strip()[:600]
324
+ f"no tests parsed from `{base}` (exit {code}): " + (err or out).strip()[:600]
295
325
  )
296
326
  return result
@@ -12,6 +12,7 @@ from . import adapters, gitutil, staging
12
12
  from . import config as config_mod
13
13
  from .adapters.base import FAILED, NOT_COLLECTED, NOT_FOUND, PASSED
14
14
  from .envelope import Envelope, NextAction, Verb
15
+ from .ledger import now
15
16
  from .machine import (
16
17
  AWAITING_IMPL,
17
18
  AWAITING_PIN,
@@ -58,6 +59,16 @@ def _stub_directive_issued(engine: Engine, cycle) -> bool:
58
59
  ) is not None
59
60
 
60
61
 
62
+ def _last_outside_emitted(engine: Engine, cycle) -> str | None:
63
+ row = engine.ledger.one(
64
+ "SELECT detail FROM integrity_event"
65
+ " WHERE cycle_id = ? AND kind = 'undeclared_file_touched'"
66
+ " ORDER BY id DESC LIMIT 1",
67
+ (cycle["id"],),
68
+ )
69
+ return row["detail"] if row else None
70
+
71
+
61
72
  def _sanctioned_stubs(engine: Engine, cycle, implementation: list[str]) -> list[str]:
62
73
  """The files that answer a `create_stub` directive the tool itself issued.
63
74
 
@@ -74,7 +85,8 @@ def _sanctioned_stubs(engine: Engine, cycle, implementation: list[str]) -> list[
74
85
  def _stage_and_commit(engine: Engine, cycle, phase: str, declared) -> tuple[str | None, list[str], object]:
75
86
  changed = engine.authored_changes(cycle)
76
87
  classification = staging.classify(
77
- engine.config, changed, json.loads(cycle["projects"]), declared, engine.excluded
88
+ engine.config, changed, json.loads(cycle["projects"]), declared, engine.excluded,
89
+ ancillary=set(engine.ancillary_files),
78
90
  )
79
91
  if phase == staging.RED:
80
92
  adopted = _sanctioned_stubs(engine, cycle, classification.implementation)
@@ -89,10 +101,12 @@ def _stage_and_commit(engine: Engine, cycle, phase: str, declared) -> tuple[str
89
101
  json.dumps(classification.implementation),
90
102
  )
91
103
  if classification.outside:
92
- engine.ledger.event(
93
- engine.run["id"], cycle["id"], "undeclared_file_touched",
94
- json.dumps(classification.outside),
95
- )
104
+ _detail = json.dumps(classification.outside)
105
+ if _last_outside_emitted(engine, cycle) != _detail:
106
+ engine.ledger.event(
107
+ engine.run["id"], cycle["id"], "undeclared_file_touched",
108
+ _detail,
109
+ )
96
110
  paths = staging.paths_for_phase(phase, classification)
97
111
  message = staging.default_message(phase, declared, cycle["ordinal"])
98
112
  sha, staged = staging.commit(
@@ -335,6 +349,38 @@ def _handle_refactor(engine: Engine, cycle, retried: bool) -> Envelope:
335
349
 
336
350
  nxt = engine.close_cycle(cycle)
337
351
  if nxt is None:
352
+ blocking = engine.close_undeclared_gate(cycle)
353
+ if blocking:
354
+ engine.ledger.insert(
355
+ "blocker",
356
+ run_id=engine.run["id"],
357
+ cycle_id=cycle["id"],
358
+ kind="undeclared_file_uncommitted",
359
+ detail=json.dumps(blocking),
360
+ at=now(),
361
+ )
362
+ engine.ledger.update("run", engine.run["id"], outcome="blocked")
363
+ return Envelope(
364
+ run={
365
+ "id": engine.run["id"],
366
+ "plan": engine.contract_row["plan_path"],
367
+ "cycle": cycle["ordinal"],
368
+ "of": len(engine.declared),
369
+ "phase": "BLOCKED",
370
+ "executor": engine.run["executor_model"],
371
+ },
372
+ result={
373
+ "kind": "undeclared_file_uncommitted",
374
+ "paths": blocking,
375
+ "commit": sha,
376
+ },
377
+ next_action=NextAction(
378
+ Verb.BLOCKED,
379
+ f"Run reached its last cycle but {len(blocking)} flagged file(s) are"
380
+ f" uncommitted: {blocking}. Commit them, or"
381
+ " `tdd resume --unblock --note ...` to discard.",
382
+ ),
383
+ )
338
384
  return Envelope(
339
385
  run={
340
386
  "id": engine.run["id"],