tdd-cli 0.7.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/CHANGELOG.md +52 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/PKG-INFO +94 -5
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/README.md +93 -4
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/base.py +8 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/vitest_adapter.py +62 -32
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/advance.py +51 -5
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/cli.py +107 -27
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/config.py +13 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/contract.py +15 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/ledger.py +79 -3
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/machine.py +33 -5
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/render.py +6 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/staging.py +6 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/conftest.py +3 -0
- tdd_cli-0.8.0/tests/test_ancillary_files.py +72 -0
- tdd_cli-0.8.0/tests/test_artifact_regeneration.py +166 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_baseline_integrity.py +266 -1
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_config_and_staging.py +15 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_contract.py +120 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_heartbeat.py +49 -0
- tdd_cli-0.8.0/tests/test_id_normalisation.py +127 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_release_surface.py +15 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_snapshot_and_identity.py +45 -0
- tdd_cli-0.8.0/tests/test_undeclared_close_gate.py +106 -0
- tdd_cli-0.8.0/tests/test_undeclared_dedup.py +168 -0
- tdd_cli-0.7.0/tests/test_artifact_regeneration.py +0 -76
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/.gitignore +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/LICENSE +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/SECURITY.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/plan.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/pyproject.toml +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/exec_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/gradle_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/pytest_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/adapters/xctest_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/identity.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_batch_collection.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_concurrent_advance.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_doctor_blockers.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_exec_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_failure_clipping.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_gradle_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_named_leases_and_timeout.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_project_commands.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_project_env.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_suite_overrides.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_timing_visibility.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_vitest_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_worker_leases.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.8.0}/tests/test_xctest_adapter.py +0 -0
|
@@ -6,6 +6,58 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.8.0] - 2026-08-28
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **`--reuse-baselines` caches baseline probes by content hash.** `run start`
|
|
14
|
+
can skip re-probing a project whose tree is unchanged: a probe result is
|
|
15
|
+
cached keyed by `(project, tree_hash, config_sha)` — where `tree_hash` folds in
|
|
16
|
+
every upstream producer root — and an identical rerun emits `baseline_reused`
|
|
17
|
+
instead of `baseline_captured`, reusing the cached failing set and collection
|
|
18
|
+
snapshot rather than re-running the suite. Provenance is recorded on the
|
|
19
|
+
baseline row (`source`), and `--reuse-max-age` re-probes any entry older than
|
|
20
|
+
the given age. Off by default; the cache stays empty unless the flag is passed
|
|
21
|
+
(#45, #59).
|
|
22
|
+
|
|
23
|
+
- **`--baseline-jobs` parallelizes baseline probing.** `run start` probes each
|
|
24
|
+
project's baseline under a bounded `ThreadPoolExecutor` when `--baseline-jobs`
|
|
25
|
+
is greater than 1 (default 1, must be >= 1). The `baseline_captured` heartbeat
|
|
26
|
+
survives the pool, and a worker probe that raises becomes an attributed failure
|
|
27
|
+
that aborts cleanly and frees the worktree rather than wedging it (#46, #62).
|
|
28
|
+
|
|
29
|
+
- **Plan-level `ancillary_files`.** A top-level front-matter key declaring
|
|
30
|
+
cross-project or companion paths a plan touches (README, generated fixtures,
|
|
31
|
+
sibling-project files). Declared ancillary paths are bucketed into their own
|
|
32
|
+
staging bucket, committed with the cycle, and fire no `undeclared_file_touched`
|
|
33
|
+
event. Validated as a list of strings at registration and persisted to the
|
|
34
|
+
ledger via a v5→v6 migration (#70, #80).
|
|
35
|
+
|
|
36
|
+
- **Run-close gate on undeclared touched paths.** `run close` now blocks when a
|
|
37
|
+
path previously flagged as `undeclared_file_touched` is still dirty in the
|
|
38
|
+
worktree, so undeclared changes cannot slip through at the end of a run. A
|
|
39
|
+
flagged path that was since committed does not block; one that has vanished is
|
|
40
|
+
reported via a new `undeclared_file_dropped` event rather than blocking (#69,
|
|
41
|
+
#81).
|
|
42
|
+
|
|
43
|
+
- **Reserved per-cycle `meta:` passthrough.** A cycle may carry an authored
|
|
44
|
+
`meta:` mapping in the plan front-matter; it round-trips through storage
|
|
45
|
+
unchanged and is available for plan-time metadata. A non-mapping `meta:`
|
|
46
|
+
hard-fails registration with a `ContractError` (#58, #65).
|
|
47
|
+
|
|
48
|
+
### Fixed
|
|
49
|
+
|
|
50
|
+
- `undeclared_file_touched` is deduplicated within a cycle, so a path touched
|
|
51
|
+
across multiple phases no longer floods the cycle with repeated events (#55,
|
|
52
|
+
#61).
|
|
53
|
+
- vitest test ids are normalised on the describe/test separator before matching,
|
|
54
|
+
so a formatting-only difference between a declared target and the observed
|
|
55
|
+
verdict is no longer reported as a spurious `declared_test_mismatch` (#57,
|
|
56
|
+
#63).
|
|
57
|
+
- The `stale_artifact` event is suppressed when the tool auto-regenerates the
|
|
58
|
+
artifact and commits it, so a successful regeneration no longer also emits a
|
|
59
|
+
staleness warning (#64).
|
|
60
|
+
|
|
9
61
|
## [0.7.0] - 2026-08-23
|
|
10
62
|
|
|
11
63
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -227,6 +227,8 @@ cycles:
|
|
|
227
227
|
stub_expected: ["app/exception_map.py"]
|
|
228
228
|
commit_red: "test: unmapped exception is not swallowed"
|
|
229
229
|
commit_green: "feat: domain exception map skeleton"
|
|
230
|
+
meta: # optional authored-at-plan-time metadata; opaque to the tool
|
|
231
|
+
covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
|
|
230
232
|
- n: 8
|
|
231
233
|
project: backend
|
|
232
234
|
pin_cycle: true # characterisation; passes on arrival by design
|
|
@@ -238,9 +240,51 @@ cycles:
|
|
|
238
240
|
- "backend::tests/test_openapi.py::test_upload_body_schema"
|
|
239
241
|
- "frontend::services/__tests__/upload.test.ts > matches contract"
|
|
240
242
|
annotation_keys: ["literal_detail_handlers_kept"]
|
|
243
|
+
ancillary_files:
|
|
244
|
+
- frontend/src/api/registerClient.ts # type-break from regenerated client
|
|
245
|
+
- docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
|
|
241
246
|
---
|
|
242
247
|
```
|
|
243
248
|
|
|
249
|
+
**Top-level keys:**
|
|
250
|
+
|
|
251
|
+
`annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
|
|
252
|
+
gate checks that every key is present before the plan can be marked complete.
|
|
253
|
+
|
|
254
|
+
`ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
|
|
255
|
+
touch outside any registered project root (cross-project ripples, companion documents).
|
|
256
|
+
Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
|
|
257
|
+
matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
|
|
258
|
+
and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
|
|
259
|
+
not on the list still fires `undeclared_file_touched` exactly as today. This is a
|
|
260
|
+
plan-level list only; per-cycle overrides are a planned follow-up.
|
|
261
|
+
|
|
262
|
+
### Run-close gate for undeclared file touches
|
|
263
|
+
|
|
264
|
+
When the last declared cycle closes, the tool gathers every path that appeared in any
|
|
265
|
+
`undeclared_file_touched` event across the run and checks the worktree:
|
|
266
|
+
|
|
267
|
+
- **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
|
|
268
|
+
the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
|
|
269
|
+
commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
|
|
270
|
+
and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
|
|
271
|
+
- **Committed during the run** → clean; the run completes normally.
|
|
272
|
+
- **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
|
|
273
|
+
emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
|
|
274
|
+
makes a silent drop visible without blocking, since the file is already gone.
|
|
275
|
+
|
|
276
|
+
The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
|
|
277
|
+
so they are never seen by the gate.
|
|
278
|
+
|
|
279
|
+
**Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
|
|
280
|
+
`stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
|
|
281
|
+
`commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
|
|
282
|
+
`meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
|
|
283
|
+
but its *contents* are opaque to the tool — any key/value pairs are accepted and
|
|
284
|
+
round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
|
|
285
|
+
authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
|
|
286
|
+
needs to read back. Other unknown per-cycle keys are silently ignored.
|
|
287
|
+
|
|
244
288
|
Absent front-matter is legitimate — the run proceeds as `undeclared` with
|
|
245
289
|
`--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
|
|
246
290
|
hard-fails registration: it is almost always a defect in the planning process, and that
|
|
@@ -418,6 +462,50 @@ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
|
|
|
418
462
|
Use this when a cycle may edit files outside the predicted reachable set and you want every
|
|
419
463
|
project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
|
|
420
464
|
|
|
465
|
+
### Reusing baselines across runs (R9.5e)
|
|
466
|
+
|
|
467
|
+
On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
|
|
468
|
+
wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
|
|
469
|
+
when nothing in the project changed:
|
|
470
|
+
|
|
471
|
+
tdd run start --plan tasks/plan.md --reuse-baselines
|
|
472
|
+
|
|
473
|
+
The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
|
|
474
|
+
If the key matches a previous entry, the cached failing set and collection snapshot are used —
|
|
475
|
+
no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
|
|
476
|
+
`source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
|
|
477
|
+
projects were skipped. Without the flag, the cache is neither read nor written.
|
|
478
|
+
|
|
479
|
+
A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
|
|
480
|
+
(see below) — reuse is loud by design, never silent.
|
|
481
|
+
|
|
482
|
+
To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
|
|
483
|
+
|
|
484
|
+
tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
|
|
485
|
+
|
|
486
|
+
Entries older than the given threshold are ignored and re-probed fresh.
|
|
487
|
+
|
|
488
|
+
### Parallel baseline probing (R9.5f)
|
|
489
|
+
|
|
490
|
+
By default `run start` probes one project at a time. On a repo with many independent projects the
|
|
491
|
+
wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
|
|
492
|
+
projects concurrently:
|
|
493
|
+
|
|
494
|
+
tdd run start --plan tasks/plan.md --baseline-jobs 4
|
|
495
|
+
|
|
496
|
+
Each probe is independent — one adapter instance per project — so concurrency does not affect
|
|
497
|
+
which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
|
|
498
|
+
heartbeat is still emitted per project as each probe completes.
|
|
499
|
+
|
|
500
|
+
The default is `--baseline-jobs 1` (serial). Raise it deliberately:
|
|
501
|
+
|
|
502
|
+
- **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
|
|
503
|
+
- **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
|
|
504
|
+
or declare a `lease` name — leased suites serialize automatically even inside the pool.
|
|
505
|
+
|
|
506
|
+
If any probe fails, `run start` returns a failure attributed to that project and no run row is
|
|
507
|
+
created, so the worktree is immediately retryable.
|
|
508
|
+
|
|
421
509
|
### When a sweep reaches an un-baselined project (R9.5d)
|
|
422
510
|
|
|
423
511
|
If an edit during a run touches a file owned by an artifact that was outside the predicted
|
|
@@ -472,10 +560,11 @@ The CLI cannot compel an agent — only the harness can.
|
|
|
472
560
|
artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
|
|
473
561
|
`advance` refuses an unchanged tree unless `--retry`.
|
|
474
562
|
|
|
475
|
-
**Recorded, never blocked:** non-stub writes during RED, undeclared file touches,
|
|
476
|
-
divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
477
|
-
blocked agent improvises around them — putting it right back in the reporting path
|
|
478
|
-
tool exists to keep it out of.
|
|
563
|
+
**Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
|
|
564
|
+
scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
565
|
+
and a blocked agent improvises around them — putting it right back in the reporting path
|
|
566
|
+
the tool exists to keep it out of. Undeclared file touches are an exception at run close:
|
|
567
|
+
see the run-close gate above.
|
|
479
568
|
|
|
480
569
|
**Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
|
|
481
570
|
stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
|
|
@@ -199,6 +199,8 @@ cycles:
|
|
|
199
199
|
stub_expected: ["app/exception_map.py"]
|
|
200
200
|
commit_red: "test: unmapped exception is not swallowed"
|
|
201
201
|
commit_green: "feat: domain exception map skeleton"
|
|
202
|
+
meta: # optional authored-at-plan-time metadata; opaque to the tool
|
|
203
|
+
covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
|
|
202
204
|
- n: 8
|
|
203
205
|
project: backend
|
|
204
206
|
pin_cycle: true # characterisation; passes on arrival by design
|
|
@@ -210,9 +212,51 @@ cycles:
|
|
|
210
212
|
- "backend::tests/test_openapi.py::test_upload_body_schema"
|
|
211
213
|
- "frontend::services/__tests__/upload.test.ts > matches contract"
|
|
212
214
|
annotation_keys: ["literal_detail_handlers_kept"]
|
|
215
|
+
ancillary_files:
|
|
216
|
+
- frontend/src/api/registerClient.ts # type-break from regenerated client
|
|
217
|
+
- docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
|
|
213
218
|
---
|
|
214
219
|
```
|
|
215
220
|
|
|
221
|
+
**Top-level keys:**
|
|
222
|
+
|
|
223
|
+
`annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
|
|
224
|
+
gate checks that every key is present before the plan can be marked complete.
|
|
225
|
+
|
|
226
|
+
`ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
|
|
227
|
+
touch outside any registered project root (cross-project ripples, companion documents).
|
|
228
|
+
Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
|
|
229
|
+
matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
|
|
230
|
+
and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
|
|
231
|
+
not on the list still fires `undeclared_file_touched` exactly as today. This is a
|
|
232
|
+
plan-level list only; per-cycle overrides are a planned follow-up.
|
|
233
|
+
|
|
234
|
+
### Run-close gate for undeclared file touches
|
|
235
|
+
|
|
236
|
+
When the last declared cycle closes, the tool gathers every path that appeared in any
|
|
237
|
+
`undeclared_file_touched` event across the run and checks the worktree:
|
|
238
|
+
|
|
239
|
+
- **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
|
|
240
|
+
the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
|
|
241
|
+
commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
|
|
242
|
+
and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
|
|
243
|
+
- **Committed during the run** → clean; the run completes normally.
|
|
244
|
+
- **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
|
|
245
|
+
emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
|
|
246
|
+
makes a silent drop visible without blocking, since the file is already gone.
|
|
247
|
+
|
|
248
|
+
The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
|
|
249
|
+
so they are never seen by the gate.
|
|
250
|
+
|
|
251
|
+
**Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
|
|
252
|
+
`stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
|
|
253
|
+
`commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
|
|
254
|
+
`meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
|
|
255
|
+
but its *contents* are opaque to the tool — any key/value pairs are accepted and
|
|
256
|
+
round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
|
|
257
|
+
authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
|
|
258
|
+
needs to read back. Other unknown per-cycle keys are silently ignored.
|
|
259
|
+
|
|
216
260
|
Absent front-matter is legitimate — the run proceeds as `undeclared` with
|
|
217
261
|
`--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
|
|
218
262
|
hard-fails registration: it is almost always a defect in the planning process, and that
|
|
@@ -390,6 +434,50 @@ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
|
|
|
390
434
|
Use this when a cycle may edit files outside the predicted reachable set and you want every
|
|
391
435
|
project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
|
|
392
436
|
|
|
437
|
+
### Reusing baselines across runs (R9.5e)
|
|
438
|
+
|
|
439
|
+
On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
|
|
440
|
+
wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
|
|
441
|
+
when nothing in the project changed:
|
|
442
|
+
|
|
443
|
+
tdd run start --plan tasks/plan.md --reuse-baselines
|
|
444
|
+
|
|
445
|
+
The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
|
|
446
|
+
If the key matches a previous entry, the cached failing set and collection snapshot are used —
|
|
447
|
+
no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
|
|
448
|
+
`source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
|
|
449
|
+
projects were skipped. Without the flag, the cache is neither read nor written.
|
|
450
|
+
|
|
451
|
+
A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
|
|
452
|
+
(see below) — reuse is loud by design, never silent.
|
|
453
|
+
|
|
454
|
+
To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
|
|
455
|
+
|
|
456
|
+
tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
|
|
457
|
+
|
|
458
|
+
Entries older than the given threshold are ignored and re-probed fresh.
|
|
459
|
+
|
|
460
|
+
### Parallel baseline probing (R9.5f)
|
|
461
|
+
|
|
462
|
+
By default `run start` probes one project at a time. On a repo with many independent projects the
|
|
463
|
+
wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
|
|
464
|
+
projects concurrently:
|
|
465
|
+
|
|
466
|
+
tdd run start --plan tasks/plan.md --baseline-jobs 4
|
|
467
|
+
|
|
468
|
+
Each probe is independent — one adapter instance per project — so concurrency does not affect
|
|
469
|
+
which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
|
|
470
|
+
heartbeat is still emitted per project as each probe completes.
|
|
471
|
+
|
|
472
|
+
The default is `--baseline-jobs 1` (serial). Raise it deliberately:
|
|
473
|
+
|
|
474
|
+
- **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
|
|
475
|
+
- **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
|
|
476
|
+
or declare a `lease` name — leased suites serialize automatically even inside the pool.
|
|
477
|
+
|
|
478
|
+
If any probe fails, `run start` returns a failure attributed to that project and no run row is
|
|
479
|
+
created, so the worktree is immediately retryable.
|
|
480
|
+
|
|
393
481
|
### When a sweep reaches an un-baselined project (R9.5d)
|
|
394
482
|
|
|
395
483
|
If an edit during a run touches a file owned by an artifact that was outside the predicted
|
|
@@ -444,10 +532,11 @@ The CLI cannot compel an agent — only the harness can.
|
|
|
444
532
|
artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
|
|
445
533
|
`advance` refuses an unchanged tree unless `--retry`.
|
|
446
534
|
|
|
447
|
-
**Recorded, never blocked:** non-stub writes during RED, undeclared file touches,
|
|
448
|
-
divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
449
|
-
blocked agent improvises around them — putting it right back in the reporting path
|
|
450
|
-
tool exists to keep it out of.
|
|
535
|
+
**Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
|
|
536
|
+
scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
537
|
+
and a blocked agent improvises around them — putting it right back in the reporting path
|
|
538
|
+
the tool exists to keep it out of. Undeclared file touches are an exception at run close:
|
|
539
|
+
see the run-close gate above.
|
|
451
540
|
|
|
452
541
|
**Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
|
|
453
542
|
stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
|
|
@@ -139,6 +139,14 @@ class Adapter:
|
|
|
139
139
|
prefix = f"{self.project.name}::"
|
|
140
140
|
return qualified[len(prefix) :] if qualified.startswith(prefix) else qualified
|
|
141
141
|
|
|
142
|
+
def normalise_id(self, test_id: str) -> str:
|
|
143
|
+
"""Return the canonical form of a declared target id for matching against collected ids.
|
|
144
|
+
|
|
145
|
+
The default is an identity — subclasses override when the runner's collected
|
|
146
|
+
ids differ from a natural human spelling (e.g. vitest's describe/test separator).
|
|
147
|
+
"""
|
|
148
|
+
return test_id
|
|
149
|
+
|
|
142
150
|
def run(self, target: str | None = None) -> Verdict:
|
|
143
151
|
raise NotImplementedError
|
|
144
152
|
|
|
@@ -45,7 +45,25 @@ class VitestAdapter(Adapter):
|
|
|
45
45
|
name = "vitest"
|
|
46
46
|
|
|
47
47
|
def stub_hint(self) -> str:
|
|
48
|
-
return
|
|
48
|
+
return '`throw new Error("not implemented")` in every body'
|
|
49
|
+
|
|
50
|
+
def normalise_id(self, test_id: str) -> str:
|
|
51
|
+
"""Canonicalise the describe/test separator for target matching.
|
|
52
|
+
|
|
53
|
+
vitest's `fullName` joins ancestor titles and the test title with a space,
|
|
54
|
+
so collected ids look like `frontend::a.test.ts > someHelper formats a value`.
|
|
55
|
+
A planner naturally writes ` > ` at each nesting level. Treating both as
|
|
56
|
+
equivalent prevents a formatting-only difference from producing NOT_FOUND.
|
|
57
|
+
|
|
58
|
+
The structural ` > ` between the file and the name is preserved; only the
|
|
59
|
+
` > ` separators inside the name part are collapsed to a space.
|
|
60
|
+
"""
|
|
61
|
+
raw = self.strip(test_id)
|
|
62
|
+
file_part, sep, remainder = raw.partition(" > ")
|
|
63
|
+
if not sep:
|
|
64
|
+
return test_id
|
|
65
|
+
name = " ".join(part.strip() for part in remainder.split(" > "))
|
|
66
|
+
return self.qualify(f"{file_part} > {name}")
|
|
49
67
|
|
|
50
68
|
def _id_for(self, suite_path: str, full_name: str) -> str:
|
|
51
69
|
abs_path = Path(suite_path)
|
|
@@ -72,18 +90,18 @@ class VitestAdapter(Adapter):
|
|
|
72
90
|
code, out, err = self._run_suite(f"{base} --reporter=json", extra_env)
|
|
73
91
|
report = _extract_json(out)
|
|
74
92
|
if report is None:
|
|
75
|
-
verdict.error = (
|
|
76
|
-
f"`{base}` produced no JSON output: {(err or out)[:500]}"
|
|
77
|
-
)
|
|
93
|
+
verdict.error = f"`{base}` produced no JSON output: {(err or out)[:500]}"
|
|
78
94
|
return verdict
|
|
79
95
|
verdict.duration_ms += int(report.get("duration") or 0)
|
|
80
96
|
results = report.get("testResults", [])
|
|
81
97
|
suites.extend(results)
|
|
82
|
-
suite_ids.append(
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
98
|
+
suite_ids.append(
|
|
99
|
+
{
|
|
100
|
+
self._id_for(s.get("name", ""), t["fullName"])
|
|
101
|
+
for s in results
|
|
102
|
+
for t in s.get("assertionResults", [])
|
|
103
|
+
}
|
|
104
|
+
)
|
|
87
105
|
|
|
88
106
|
overlap = _suite_overlap(suite_ids)
|
|
89
107
|
if overlap:
|
|
@@ -108,17 +126,23 @@ class VitestAdapter(Adapter):
|
|
|
108
126
|
verdict.target_outcome = NOT_FOUND
|
|
109
127
|
return verdict
|
|
110
128
|
|
|
111
|
-
|
|
129
|
+
ntarget = self.normalise_id(target)
|
|
130
|
+
passed_norm = {self.normalise_id(t): t for t in verdict.passed}
|
|
131
|
+
failed_norm = {self.normalise_id(t): t for t in verdict.failed}
|
|
132
|
+
|
|
133
|
+
if ntarget in passed_norm:
|
|
112
134
|
verdict.target_outcome = PASSED
|
|
113
135
|
return verdict
|
|
114
|
-
if
|
|
136
|
+
if ntarget in failed_norm:
|
|
115
137
|
verdict.target_outcome = FAILED
|
|
116
138
|
for suite in suites:
|
|
117
139
|
for t in suite.get("assertionResults", []):
|
|
118
|
-
if
|
|
140
|
+
if (
|
|
141
|
+
self.normalise_id(self._id_for(suite.get("name", ""), t["fullName"]))
|
|
142
|
+
== ntarget
|
|
143
|
+
):
|
|
119
144
|
verdict.target_failure = "\n".join(
|
|
120
|
-
clip_failure(m, 600)
|
|
121
|
-
for m in t.get("failureMessages", [])[:3]
|
|
145
|
+
clip_failure(m, 600) for m in t.get("failureMessages", [])[:3]
|
|
122
146
|
)
|
|
123
147
|
return verdict
|
|
124
148
|
|
|
@@ -206,22 +230,26 @@ class VitestAdapter(Adapter):
|
|
|
206
230
|
code, out, err = run_command(
|
|
207
231
|
self._collect_cmd(), self.root, extra_env=self._suite_env(None), label="doctor"
|
|
208
232
|
)
|
|
209
|
-
reached = sorted(
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
for
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
233
|
+
reached = sorted(
|
|
234
|
+
{
|
|
235
|
+
f
|
|
236
|
+
for f in (
|
|
237
|
+
line.strip().partition(" > ")[0] for line in out.splitlines() if " > " in line
|
|
238
|
+
)
|
|
239
|
+
if self.project.override_for(f)
|
|
240
|
+
}
|
|
241
|
+
)
|
|
217
242
|
if not reached:
|
|
218
243
|
return GateResult(ok=True)
|
|
219
|
-
return GateResult(
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
244
|
+
return GateResult(
|
|
245
|
+
ok=False,
|
|
246
|
+
output=(
|
|
247
|
+
"the default config's discovery reaches files an override owns, so"
|
|
248
|
+
" suite runs would observe them without the override's command/env:"
|
|
249
|
+
f" {', '.join(reached[:5])}. Exclude them from the default vitest"
|
|
250
|
+
" config (test.exclude) or scope its include globs."
|
|
251
|
+
),
|
|
252
|
+
)
|
|
225
253
|
|
|
226
254
|
def _collect_invocations(self) -> list[tuple[str, dict[str, str] | None]]:
|
|
227
255
|
"""An override with no `collect_command` gets no batch — `vitest list` knows
|
|
@@ -229,7 +257,8 @@ class VitestAdapter(Adapter):
|
|
|
229
257
|
missing `collect_command` against each of them as before."""
|
|
230
258
|
return [(self._collect_cmd(), self._suite_env(None))] + [
|
|
231
259
|
(ov.collect_command, self._suite_env(ov))
|
|
232
|
-
for ov in self.project.overrides
|
|
260
|
+
for ov in self.project.overrides
|
|
261
|
+
if ov.collect_command
|
|
233
262
|
]
|
|
234
263
|
|
|
235
264
|
def _collect_batch(
|
|
@@ -270,7 +299,9 @@ class VitestAdapter(Adapter):
|
|
|
270
299
|
base = ov.collect_command if ov else self._collect_cmd()
|
|
271
300
|
env = self._suite_env(ov)
|
|
272
301
|
code, out, err = run_command(
|
|
273
|
-
f"{base} {shlex.quote(str(rel))}",
|
|
302
|
+
f"{base} {shlex.quote(str(rel))}",
|
|
303
|
+
self.root,
|
|
304
|
+
extra_env=env,
|
|
274
305
|
label="collect",
|
|
275
306
|
)
|
|
276
307
|
|
|
@@ -290,7 +321,6 @@ class VitestAdapter(Adapter):
|
|
|
290
321
|
# Zero tests from a file that exists is a tooling failure, not an
|
|
291
322
|
# empty file — record it rather than silently collecting nothing.
|
|
292
323
|
result.failed_files[str(rel)] = (
|
|
293
|
-
f"no tests parsed from `{base}` (exit {code}): "
|
|
294
|
-
+ (err or out).strip()[:600]
|
|
324
|
+
f"no tests parsed from `{base}` (exit {code}): " + (err or out).strip()[:600]
|
|
295
325
|
)
|
|
296
326
|
return result
|
|
@@ -12,6 +12,7 @@ from . import adapters, gitutil, staging
|
|
|
12
12
|
from . import config as config_mod
|
|
13
13
|
from .adapters.base import FAILED, NOT_COLLECTED, NOT_FOUND, PASSED
|
|
14
14
|
from .envelope import Envelope, NextAction, Verb
|
|
15
|
+
from .ledger import now
|
|
15
16
|
from .machine import (
|
|
16
17
|
AWAITING_IMPL,
|
|
17
18
|
AWAITING_PIN,
|
|
@@ -58,6 +59,16 @@ def _stub_directive_issued(engine: Engine, cycle) -> bool:
|
|
|
58
59
|
) is not None
|
|
59
60
|
|
|
60
61
|
|
|
62
|
+
def _last_outside_emitted(engine: Engine, cycle) -> str | None:
|
|
63
|
+
row = engine.ledger.one(
|
|
64
|
+
"SELECT detail FROM integrity_event"
|
|
65
|
+
" WHERE cycle_id = ? AND kind = 'undeclared_file_touched'"
|
|
66
|
+
" ORDER BY id DESC LIMIT 1",
|
|
67
|
+
(cycle["id"],),
|
|
68
|
+
)
|
|
69
|
+
return row["detail"] if row else None
|
|
70
|
+
|
|
71
|
+
|
|
61
72
|
def _sanctioned_stubs(engine: Engine, cycle, implementation: list[str]) -> list[str]:
|
|
62
73
|
"""The files that answer a `create_stub` directive the tool itself issued.
|
|
63
74
|
|
|
@@ -74,7 +85,8 @@ def _sanctioned_stubs(engine: Engine, cycle, implementation: list[str]) -> list[
|
|
|
74
85
|
def _stage_and_commit(engine: Engine, cycle, phase: str, declared) -> tuple[str | None, list[str], object]:
|
|
75
86
|
changed = engine.authored_changes(cycle)
|
|
76
87
|
classification = staging.classify(
|
|
77
|
-
engine.config, changed, json.loads(cycle["projects"]), declared, engine.excluded
|
|
88
|
+
engine.config, changed, json.loads(cycle["projects"]), declared, engine.excluded,
|
|
89
|
+
ancillary=set(engine.ancillary_files),
|
|
78
90
|
)
|
|
79
91
|
if phase == staging.RED:
|
|
80
92
|
adopted = _sanctioned_stubs(engine, cycle, classification.implementation)
|
|
@@ -89,10 +101,12 @@ def _stage_and_commit(engine: Engine, cycle, phase: str, declared) -> tuple[str
|
|
|
89
101
|
json.dumps(classification.implementation),
|
|
90
102
|
)
|
|
91
103
|
if classification.outside:
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
104
|
+
_detail = json.dumps(classification.outside)
|
|
105
|
+
if _last_outside_emitted(engine, cycle) != _detail:
|
|
106
|
+
engine.ledger.event(
|
|
107
|
+
engine.run["id"], cycle["id"], "undeclared_file_touched",
|
|
108
|
+
_detail,
|
|
109
|
+
)
|
|
96
110
|
paths = staging.paths_for_phase(phase, classification)
|
|
97
111
|
message = staging.default_message(phase, declared, cycle["ordinal"])
|
|
98
112
|
sha, staged = staging.commit(
|
|
@@ -335,6 +349,38 @@ def _handle_refactor(engine: Engine, cycle, retried: bool) -> Envelope:
|
|
|
335
349
|
|
|
336
350
|
nxt = engine.close_cycle(cycle)
|
|
337
351
|
if nxt is None:
|
|
352
|
+
blocking = engine.close_undeclared_gate(cycle)
|
|
353
|
+
if blocking:
|
|
354
|
+
engine.ledger.insert(
|
|
355
|
+
"blocker",
|
|
356
|
+
run_id=engine.run["id"],
|
|
357
|
+
cycle_id=cycle["id"],
|
|
358
|
+
kind="undeclared_file_uncommitted",
|
|
359
|
+
detail=json.dumps(blocking),
|
|
360
|
+
at=now(),
|
|
361
|
+
)
|
|
362
|
+
engine.ledger.update("run", engine.run["id"], outcome="blocked")
|
|
363
|
+
return Envelope(
|
|
364
|
+
run={
|
|
365
|
+
"id": engine.run["id"],
|
|
366
|
+
"plan": engine.contract_row["plan_path"],
|
|
367
|
+
"cycle": cycle["ordinal"],
|
|
368
|
+
"of": len(engine.declared),
|
|
369
|
+
"phase": "BLOCKED",
|
|
370
|
+
"executor": engine.run["executor_model"],
|
|
371
|
+
},
|
|
372
|
+
result={
|
|
373
|
+
"kind": "undeclared_file_uncommitted",
|
|
374
|
+
"paths": blocking,
|
|
375
|
+
"commit": sha,
|
|
376
|
+
},
|
|
377
|
+
next_action=NextAction(
|
|
378
|
+
Verb.BLOCKED,
|
|
379
|
+
f"Run reached its last cycle but {len(blocking)} flagged file(s) are"
|
|
380
|
+
f" uncommitted: {blocking}. Commit them, or"
|
|
381
|
+
" `tdd resume --unblock --note ...` to discard.",
|
|
382
|
+
),
|
|
383
|
+
)
|
|
338
384
|
return Envelope(
|
|
339
385
|
run={
|
|
340
386
|
"id": engine.run["id"],
|