tdd-cli 0.7.0__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/CHANGELOG.md +126 -0
  2. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/PKG-INFO +169 -5
  3. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/README.md +168 -4
  4. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/skills/tdd-drive/SKILL.md +9 -2
  5. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/__init__.py +1 -1
  6. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/base.py +17 -0
  7. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/exec_adapter.py +3 -0
  8. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/gradle_adapter.py +13 -1
  9. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/pytest_adapter.py +20 -1
  10. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/vitest_adapter.py +78 -32
  11. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/xctest_adapter.py +19 -1
  12. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/advance.py +152 -29
  13. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/cli.py +255 -33
  14. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/config.py +29 -0
  15. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/contract.py +15 -0
  16. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/identity.py +16 -4
  17. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/ledger.py +106 -3
  18. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/machine.py +33 -5
  19. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/render.py +33 -2
  20. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/staging.py +6 -0
  21. tdd_cli-0.9.0/src/tddcli/target_lint.py +61 -0
  22. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/conftest.py +12 -0
  23. tdd_cli-0.9.0/tests/test_advance_adoption.py +126 -0
  24. tdd_cli-0.9.0/tests/test_ancillary_files.py +72 -0
  25. tdd_cli-0.9.0/tests/test_artifact_regeneration.py +166 -0
  26. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_baseline_integrity.py +266 -1
  27. tdd_cli-0.9.0/tests/test_baseline_sanity.py +229 -0
  28. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_config_and_staging.py +37 -0
  29. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_contract.py +120 -0
  30. tdd_cli-0.9.0/tests/test_evidence_extraction.py +226 -0
  31. tdd_cli-0.9.0/tests/test_executor_attribution.py +133 -0
  32. tdd_cli-0.9.0/tests/test_executor_notes.py +209 -0
  33. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_heartbeat.py +56 -6
  34. tdd_cli-0.9.0/tests/test_id_normalisation.py +127 -0
  35. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_release_surface.py +15 -0
  36. tdd_cli-0.9.0/tests/test_sensitivity_evidence.py +157 -0
  37. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_snapshot_and_identity.py +92 -0
  38. tdd_cli-0.9.0/tests/test_target_lint.py +202 -0
  39. tdd_cli-0.9.0/tests/test_undeclared_close_gate.py +106 -0
  40. tdd_cli-0.9.0/tests/test_undeclared_dedup.py +168 -0
  41. tdd_cli-0.7.0/tests/test_artifact_regeneration.py +0 -76
  42. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/.gitignore +0 -0
  43. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/LICENSE +0 -0
  44. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/SECURITY.md +0 -0
  45. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/README.md +0 -0
  46. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/bash_hook.py +0 -0
  47. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/stop_hook.py +0 -0
  48. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/plan.md +0 -0
  49. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/skills/tdd-drive/README.md +0 -0
  50. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/skills/tdd-handoff/README.md +0 -0
  51. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
  52. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/pyproject.toml +0 -0
  53. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/__init__.py +0 -0
  54. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/envelope.py +0 -0
  55. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/fleet.py +0 -0
  56. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/gitutil.py +0 -0
  57. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/leases.py +0 -0
  58. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/snapshot.py +0 -0
  59. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_batch_collection.py +0 -0
  60. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_concurrent_advance.py +0 -0
  61. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_config_drift.py +0 -0
  62. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_doctor_attribution.py +0 -0
  63. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_doctor_blockers.py +0 -0
  64. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_end_to_end.py +0 -0
  65. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_example_plan.py +0 -0
  66. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_exec_adapter.py +0 -0
  67. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_failure_clipping.py +0 -0
  68. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_fleet.py +0 -0
  69. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_gradle_adapter.py +0 -0
  70. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_init_detection.py +0 -0
  71. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_named_leases_and_timeout.py +0 -0
  72. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_pin_cycles.py +0 -0
  73. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_progress.py +0 -0
  74. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_project_commands.py +0 -0
  75. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_project_env.py +0 -0
  76. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_python_env_managers.py +0 -0
  77. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_refactor_cycles.py +0 -0
  78. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_run_claim.py +0 -0
  79. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_single_project_repo.py +0 -0
  80. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_stub_hint.py +0 -0
  81. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_suite_overrides.py +0 -0
  82. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_target_validation.py +0 -0
  83. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_timing_visibility.py +0 -0
  84. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_vitest_adapter.py +0 -0
  85. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_worker_leases.py +0 -0
  86. {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_xctest_adapter.py +0 -0
@@ -6,6 +6,132 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.9.0] - 2026-09-03
10
+
11
+ ### Added
12
+
13
+ - **`tdd note` — an executor-narrative channel.** `tdd note "<text>"` attaches
14
+ a phase-stamped note to the current cycle, or a run-level note once the run
15
+ has ended (it falls back to the latest run, so no active run is required).
16
+ Cycle notes render in the friction log as blockquote claims under their cycle;
17
+ run-level notes render under a new `## Executor narrative` section, which is
18
+ omitted when there are none. Envelopes that carry an integrity event now nudge
19
+ for a note while the reason is fresh (silenced once the cycle has one), and
20
+ the terminal `COMPLETE` envelope — via `advance` or a final-cycle skip — asks
21
+ for a closing narrative before rendering. Stored in a new `note` table via a
22
+ v7→v8 ledger migration (#77, #94).
23
+
24
+ - **Baseline sanity gate.** `run start` refuses a baseline whose failing ratio
25
+ exceeds a threshold (default 0.5, suites under 10 collected tests exempt) with
26
+ `reason: "baseline_implausible"`, on the grounds that a mostly-red suite means
27
+ the environment is broken rather than the code. A per-project
28
+ `baseline_max_failure_ratio` in `tdd.toml` overrides the default, and
29
+ `--accept-baseline` bypasses the gate and logs a `baseline_accepted` integrity
30
+ event (#67, #84).
31
+
32
+ - **Standing-failure delta.** A non-empty baseline is now compared against the
33
+ previous run's baseline for the same worktree and a `baseline_standing_delta`
34
+ event partitions the failing set into new versus inherited failures, so a
35
+ pre-existing red test is distinguishable from one that broke since the last
36
+ run (#67, #84).
37
+
38
+ - **Per-project `health_command`.** An optional command in `tdd.toml` that is
39
+ probed before baseline capture; if it fails, `run start` refuses with
40
+ `reason: "services_unreachable"` instead of recording a baseline against a
41
+ down dependency (#67, #84).
42
+
43
+ - **Declared-target lint.** `plan register` and `run start` now lint every
44
+ declared target against the adapter's id grammar (pytest `::`, vitest ` > `,
45
+ gradle `Class/method`, xctest `Bundle/Class/testMethod`) and against project
46
+ roots: a target path that duplicates a non-`.` project root prefix is refused
47
+ unless the nested path genuinely exists in the worktree. Findings are returned
48
+ under `reason: "target_lint"`; `run start` re-lints the stored contract
49
+ against the current config before claiming a baseline (#71, #88).
50
+
51
+ - **Per-adapter sensitivity evidence line.** Each adapter now extracts a
52
+ `target_evidence` line from a failing target — the first `E` line for pytest
53
+ (skipping the xdist worker header), the first `: error:` line in the test's
54
+ window for xctest, `failureMessages[0]` for vitest, the junit failure message
55
+ for gradle, and the last non-empty output line for exec. The sensitivity check
56
+ persists it as `sensitivity_check.evidence_line` (v6→v7 migration) and the
57
+ friction log's observed snippet prefers it over the raw first line, rendering
58
+ `<no assertion line captured>` when empty and keeping the tail of over-long
59
+ lines. Legacy rows with no stored evidence keep the first-line fallback (#68,
60
+ #90).
61
+
62
+ - **Diagnosable executor attribution.** `TDD_EXECUTOR_MODEL` lets a harness
63
+ declare the executor identity (`source: declared`), taking precedence over
64
+ transcript detection. When identity cannot be resolved, `Executor.reason`
65
+ says why — session id unset, transcript not found, or transcript without a
66
+ model record — and `run start` logs an `executor_unknown` event and surfaces
67
+ `executor_warning` in its envelope. `tdd doctor` gains an informational
68
+ executor-identity check (#74, #92).
69
+
70
+ ### Changed
71
+
72
+ - **Target adoption is evaluated in the same `advance`.** When a cycle's
73
+ declared target is missing and exactly one new test appears, `advance`
74
+ adopts it (logging `declared_test_mismatch`) and judges RED — or drives the
75
+ sensitivity check when it passed — from the suite run that already happened,
76
+ instead of asking for a re-run. A declared id that differs
77
+ from a single same-file candidate only by separator normalisation is
78
+ disambiguated and adopted without asking; two same-file candidates still
79
+ require an explicit `tdd target` (#72, #91).
80
+
81
+ - CI now tests on Python 3.11 and 3.14 only (#87).
82
+
83
+ ## [0.8.0] - 2026-08-28
84
+
85
+ ### Added
86
+
87
+ - **`--reuse-baselines` caches baseline probes by content hash.** `run start`
88
+ can skip re-probing a project whose tree is unchanged: a probe result is
89
+ cached keyed by `(project, tree_hash, config_sha)` — where `tree_hash` folds in
90
+ every upstream producer root — and an identical rerun emits `baseline_reused`
91
+ instead of `baseline_captured`, reusing the cached failing set and collection
92
+ snapshot rather than re-running the suite. Provenance is recorded on the
93
+ baseline row (`source`), and `--reuse-max-age` re-probes any entry older than
94
+ the given age. Off by default; the cache stays empty unless the flag is passed
95
+ (#45, #59).
96
+
97
+ - **`--baseline-jobs` parallelizes baseline probing.** `run start` probes each
98
+ project's baseline under a bounded `ThreadPoolExecutor` when `--baseline-jobs`
99
+ is greater than 1 (default 1, must be >= 1). The `baseline_captured` heartbeat
100
+ survives the pool, and a worker probe that raises becomes an attributed failure
101
+ that aborts cleanly and frees the worktree rather than wedging it (#46, #62).
102
+
103
+ - **Plan-level `ancillary_files`.** A top-level front-matter key declaring
104
+ cross-project or companion paths a plan touches (README, generated fixtures,
105
+ sibling-project files). Declared ancillary paths are bucketed into their own
106
+ staging bucket, committed with the cycle, and fire no `undeclared_file_touched`
107
+ event. Validated as a list of strings at registration and persisted to the
108
+ ledger via a v5→v6 migration (#70, #80).
109
+
110
+ - **Run-close gate on undeclared touched paths.** `run close` now blocks when a
111
+ path previously flagged as `undeclared_file_touched` is still dirty in the
112
+ worktree, so undeclared changes cannot slip through at the end of a run. A
113
+ flagged path that was since committed does not block; one that has vanished is
114
+ reported via a new `undeclared_file_dropped` event rather than blocking (#69,
115
+ #81).
116
+
117
+ - **Reserved per-cycle `meta:` passthrough.** A cycle may carry an authored
118
+ `meta:` mapping in the plan front-matter; it round-trips through storage
119
+ unchanged and is available for plan-time metadata. A non-mapping `meta:`
120
+ hard-fails registration with a `ContractError` (#58, #65).
121
+
122
+ ### Fixed
123
+
124
+ - `undeclared_file_touched` is deduplicated within a cycle, so a path touched
125
+ across multiple phases no longer floods the cycle with repeated events (#55,
126
+ #61).
127
+ - vitest test ids are normalised on the describe/test separator before matching,
128
+ so a formatting-only difference between a declared target and the observed
129
+ verdict is no longer reported as a spurious `declared_test_mismatch` (#57,
130
+ #63).
131
+ - The `stale_artifact` event is suppressed when the tool auto-regenerates the
132
+ artifact and commits it, so a successful regeneration no longer also emits a
133
+ staleness warning (#64).
134
+
9
135
  ## [0.7.0] - 2026-08-23
10
136
 
11
137
  ### Fixed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.7.0
3
+ Version: 0.9.0
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -227,6 +227,8 @@ cycles:
227
227
  stub_expected: ["app/exception_map.py"]
228
228
  commit_red: "test: unmapped exception is not swallowed"
229
229
  commit_green: "feat: domain exception map skeleton"
230
+ meta: # optional authored-at-plan-time metadata; opaque to the tool
231
+ covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
230
232
  - n: 8
231
233
  project: backend
232
234
  pin_cycle: true # characterisation; passes on arrival by design
@@ -238,9 +240,51 @@ cycles:
238
240
  - "backend::tests/test_openapi.py::test_upload_body_schema"
239
241
  - "frontend::services/__tests__/upload.test.ts > matches contract"
240
242
  annotation_keys: ["literal_detail_handlers_kept"]
243
+ ancillary_files:
244
+ - frontend/src/api/registerClient.ts # type-break from regenerated client
245
+ - docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
241
246
  ---
242
247
  ```
243
248
 
249
+ **Top-level keys:**
250
+
251
+ `annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
252
+ gate checks that every key is present before the plan can be marked complete.
253
+
254
+ `ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
255
+ touch outside any registered project root (cross-project ripples, companion documents).
256
+ Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
257
+ matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
258
+ and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
259
+ not on the list still fires `undeclared_file_touched` exactly as today. This is a
260
+ plan-level list only; per-cycle overrides are a planned follow-up.
261
+
262
+ ### Run-close gate for undeclared file touches
263
+
264
+ When the last declared cycle closes, the tool gathers every path that appeared in any
265
+ `undeclared_file_touched` event across the run and checks the worktree:
266
+
267
+ - **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
268
+ the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
269
+ commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
270
+ and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
271
+ - **Committed during the run** → clean; the run completes normally.
272
+ - **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
273
+ emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
274
+ makes a silent drop visible without blocking, since the file is already gone.
275
+
276
+ The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
277
+ so they are never seen by the gate.
278
+
279
+ **Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
280
+ `stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
281
+ `commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
282
+ `meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
283
+ but its *contents* are opaque to the tool — any key/value pairs are accepted and
284
+ round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
285
+ authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
286
+ needs to read back. Other unknown per-cycle keys are silently ignored.
287
+
244
288
  Absent front-matter is legitimate — the run proceeds as `undeclared` with
245
289
  `--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
246
290
  hard-fails registration: it is almost always a defect in the planning process, and that
@@ -275,6 +319,13 @@ rendered, never written. Judgement enters in exactly two ways:
275
319
  `commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
276
320
  `plan_defect` is the one that matters most: it records where the plan and the codebase
277
321
  disagreed, which is precisely what the next plan needs to know.
322
+ - **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
323
+ moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
324
+ notes are written after the run ends and attach to the run. Both render in the friction
325
+ log as visually distinct blockquotes (claims, not measurements). Use notes to record
326
+ *why* something happened — a plan assumption that was wrong, an integrity event that the
327
+ telemetry already captures but cannot explain. Notes are unverified by design; an auditor
328
+ compares claims against reality.
278
329
  - **Per run, as prose appended below the rendered document.** Legitimate and expected —
279
330
  post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
280
331
  is unverified: an auditor should trust the projected sections and read appended
@@ -397,6 +448,7 @@ and is never reclassified as a pin.
397
448
  | `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
398
449
  | `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
399
450
  | `tdd annotate --key --value` | attach judgement to the current cycle |
451
+ | `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
400
452
  | `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
401
453
  | `tdd resume [--unblock --note]` | reconstruct position; human intervention |
402
454
  | `tdd log render [--out]` | project the ledger into a friction log |
@@ -418,6 +470,50 @@ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
418
470
  Use this when a cycle may edit files outside the predicted reachable set and you want every
419
471
  project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
420
472
 
473
+ ### Reusing baselines across runs (R9.5e)
474
+
475
+ On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
476
+ wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
477
+ when nothing in the project changed:
478
+
479
+ tdd run start --plan tasks/plan.md --reuse-baselines
480
+
481
+ The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
482
+ If the key matches a previous entry, the cached failing set and collection snapshot are used —
483
+ no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
484
+ `source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
485
+ projects were skipped. Without the flag, the cache is neither read nor written.
486
+
487
+ A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
488
+ (see below) — reuse is loud by design, never silent.
489
+
490
+ To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
491
+
492
+ tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
493
+
494
+ Entries older than the given threshold are ignored and re-probed fresh.
495
+
496
+ ### Parallel baseline probing (R9.5f)
497
+
498
+ By default `run start` probes one project at a time. On a repo with many independent projects the
499
+ wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
500
+ projects concurrently:
501
+
502
+ tdd run start --plan tasks/plan.md --baseline-jobs 4
503
+
504
+ Each probe is independent — one adapter instance per project — so concurrency does not affect
505
+ which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
506
+ heartbeat is still emitted per project as each probe completes.
507
+
508
+ The default is `--baseline-jobs 1` (serial). Raise it deliberately:
509
+
510
+ - **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
511
+ - **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
512
+ or declare a `lease` name — leased suites serialize automatically even inside the pool.
513
+
514
+ If any probe fails, `run start` returns a failure attributed to that project and no run row is
515
+ created, so the worktree is immediately retryable.
516
+
421
517
  ### When a sweep reaches an un-baselined project (R9.5d)
422
518
 
423
519
  If an edit during a run touches a file owned by an artifact that was outside the predicted
@@ -432,6 +528,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
432
528
  failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
433
529
  with the project properly baselined.
434
530
 
531
+ ### Baseline sanity gate (R9.5g)
532
+
533
+ `run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
534
+ environment, not the code, is broken. A project is flagged when it collected at least 10 tests
535
+ (`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
536
+
537
+ error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
538
+
539
+ The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
540
+ not small pre-existing failure sets (baseline subtraction handles those). A project collecting
541
+ fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
542
+ not a broken stack.
543
+
544
+ Pass `--accept-baseline` to override the refusal and proceed anyway:
545
+
546
+ tdd run start --plan tasks/plan.md --accept-baseline
547
+
548
+ An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
549
+ `tdd metrics` and the friction log) so the override is never silent. The gate runs on every
550
+ baseline, including reused ones.
551
+
552
+ To raise the threshold for a project whose legitimate baseline is inherently noisy, set
553
+ `baseline_max_failure_ratio` in `tdd.toml`:
554
+
555
+ ```toml
556
+ [project.legacy]
557
+ root = "legacy"
558
+ adapter = "pytest"
559
+ baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
560
+ ```
561
+
562
+ The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
563
+
564
+ ### Standing-failure delta
565
+
566
+ When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
567
+ event partitioning the current standing failures into:
568
+
569
+ - **new** — absent from the previous run's baseline for this project (growing problem)
570
+ - **inherited** — also present in the previous baseline (stable background noise)
571
+ - **resolved** — in the previous baseline but absent now (fixed between runs)
572
+
573
+ The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
574
+ baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
575
+ nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
576
+
577
+ ### Environment-dependent suites
578
+
579
+ Declare a `health_command` on a project to probe reachability before `run start` captures its
580
+ baseline:
581
+
582
+ ```toml
583
+ [project.backend]
584
+ root = "backend"
585
+ adapter = "pytest"
586
+ health_command = "curl -fsS http://localhost:8080/healthz"
587
+ ```
588
+
589
+ If the command exits non-zero, `run start` refuses immediately with `reason:
590
+ "services_unreachable"` — before claiming the worktree, before probing, before writing anything
591
+ to the ledger. The error names the project and its exit code so the agent can surface the exact
592
+ diagnosis rather than recording a ~1000-failure baseline and continuing.
593
+
594
+ There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
595
+ The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
596
+ is the marker that a project requires live services; no separate boolean is needed.
597
+
435
598
  ## Running a long baseline
436
599
 
437
600
  `run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
@@ -472,10 +635,11 @@ The CLI cannot compel an agent — only the harness can.
472
635
  artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
473
636
  `advance` refuses an unchanged tree unless `--retry`.
474
637
 
475
- **Recorded, never blocked:** non-stub writes during RED, undeclared file touches, scope
476
- divergence, extra attempts. Prevention rules with edge cases produce false denials, and a
477
- blocked agent improvises around them — putting it right back in the reporting path the
478
- tool exists to keep it out of.
638
+ **Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
639
+ scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
640
+ and a blocked agent improvises around them — putting it right back in the reporting path
641
+ the tool exists to keep it out of. Undeclared file touches are an exception at run close:
642
+ see the run-close gate above.
479
643
 
480
644
  **Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
481
645
  stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
@@ -199,6 +199,8 @@ cycles:
199
199
  stub_expected: ["app/exception_map.py"]
200
200
  commit_red: "test: unmapped exception is not swallowed"
201
201
  commit_green: "feat: domain exception map skeleton"
202
+ meta: # optional authored-at-plan-time metadata; opaque to the tool
203
+ covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
202
204
  - n: 8
203
205
  project: backend
204
206
  pin_cycle: true # characterisation; passes on arrival by design
@@ -210,9 +212,51 @@ cycles:
210
212
  - "backend::tests/test_openapi.py::test_upload_body_schema"
211
213
  - "frontend::services/__tests__/upload.test.ts > matches contract"
212
214
  annotation_keys: ["literal_detail_handlers_kept"]
215
+ ancillary_files:
216
+ - frontend/src/api/registerClient.ts # type-break from regenerated client
217
+ - docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
213
218
  ---
214
219
  ```
215
220
 
221
+ **Top-level keys:**
222
+
223
+ `annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
224
+ gate checks that every key is present before the plan can be marked complete.
225
+
226
+ `ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
227
+ touch outside any registered project root (cross-project ripples, companion documents).
228
+ Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
229
+ matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
230
+ and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
231
+ not on the list still fires `undeclared_file_touched` exactly as today. This is a
232
+ plan-level list only; per-cycle overrides are a planned follow-up.
233
+
234
+ ### Run-close gate for undeclared file touches
235
+
236
+ When the last declared cycle closes, the tool gathers every path that appeared in any
237
+ `undeclared_file_touched` event across the run and checks the worktree:
238
+
239
+ - **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
240
+ the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
241
+ commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
242
+ and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
243
+ - **Committed during the run** → clean; the run completes normally.
244
+ - **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
245
+ emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
246
+ makes a silent drop visible without blocking, since the file is already gone.
247
+
248
+ The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
249
+ so they are never seen by the gate.
250
+
251
+ **Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
252
+ `stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
253
+ `commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
254
+ `meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
255
+ but its *contents* are opaque to the tool — any key/value pairs are accepted and
256
+ round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
257
+ authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
258
+ needs to read back. Other unknown per-cycle keys are silently ignored.
259
+
216
260
  Absent front-matter is legitimate — the run proceeds as `undeclared` with
217
261
  `--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
218
262
  hard-fails registration: it is almost always a defect in the planning process, and that
@@ -247,6 +291,13 @@ rendered, never written. Judgement enters in exactly two ways:
247
291
  `commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
248
292
  `plan_defect` is the one that matters most: it records where the plan and the codebase
249
293
  disagreed, which is precisely what the next plan needs to know.
294
+ - **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
295
+ moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
296
+ notes are written after the run ends and attach to the run. Both render in the friction
297
+ log as visually distinct blockquotes (claims, not measurements). Use notes to record
298
+ *why* something happened — a plan assumption that was wrong, an integrity event that the
299
+ telemetry already captures but cannot explain. Notes are unverified by design; an auditor
300
+ compares claims against reality.
250
301
  - **Per run, as prose appended below the rendered document.** Legitimate and expected —
251
302
  post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
252
303
  is unverified: an auditor should trust the projected sections and read appended
@@ -369,6 +420,7 @@ and is never reclassified as a pin.
369
420
  | `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
370
421
  | `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
371
422
  | `tdd annotate --key --value` | attach judgement to the current cycle |
423
+ | `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
372
424
  | `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
373
425
  | `tdd resume [--unblock --note]` | reconstruct position; human intervention |
374
426
  | `tdd log render [--out]` | project the ledger into a friction log |
@@ -390,6 +442,50 @@ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
390
442
  Use this when a cycle may edit files outside the predicted reachable set and you want every
391
443
  project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
392
444
 
445
+ ### Reusing baselines across runs (R9.5e)
446
+
447
+ On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
448
+ wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
449
+ when nothing in the project changed:
450
+
451
+ tdd run start --plan tasks/plan.md --reuse-baselines
452
+
453
+ The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
454
+ If the key matches a previous entry, the cached failing set and collection snapshot are used —
455
+ no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
456
+ `source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
457
+ projects were skipped. Without the flag, the cache is neither read nor written.
458
+
459
+ A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
460
+ (see below) — reuse is loud by design, never silent.
461
+
462
+ To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
463
+
464
+ tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
465
+
466
+ Entries older than the given threshold are ignored and re-probed fresh.
467
+
468
+ ### Parallel baseline probing (R9.5f)
469
+
470
+ By default `run start` probes one project at a time. On a repo with many independent projects the
471
+ wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
472
+ projects concurrently:
473
+
474
+ tdd run start --plan tasks/plan.md --baseline-jobs 4
475
+
476
+ Each probe is independent — one adapter instance per project — so concurrency does not affect
477
+ which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
478
+ heartbeat is still emitted per project as each probe completes.
479
+
480
+ The default is `--baseline-jobs 1` (serial). Raise it deliberately:
481
+
482
+ - **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
483
+ - **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
484
+ or declare a `lease` name — leased suites serialize automatically even inside the pool.
485
+
486
+ If any probe fails, `run start` returns a failure attributed to that project and no run row is
487
+ created, so the worktree is immediately retryable.
488
+
393
489
  ### When a sweep reaches an un-baselined project (R9.5d)
394
490
 
395
491
  If an edit during a run touches a file owned by an artifact that was outside the predicted
@@ -404,6 +500,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
404
500
  failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
405
501
  with the project properly baselined.
406
502
 
503
+ ### Baseline sanity gate (R9.5g)
504
+
505
+ `run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
506
+ environment, not the code, is broken. A project is flagged when it collected at least 10 tests
507
+ (`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
508
+
509
+ error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
510
+
511
+ The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
512
+ not small pre-existing failure sets (baseline subtraction handles those). A project collecting
513
+ fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
514
+ not a broken stack.
515
+
516
+ Pass `--accept-baseline` to override the refusal and proceed anyway:
517
+
518
+ tdd run start --plan tasks/plan.md --accept-baseline
519
+
520
+ An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
521
+ `tdd metrics` and the friction log) so the override is never silent. The gate runs on every
522
+ baseline, including reused ones.
523
+
524
+ To raise the threshold for a project whose legitimate baseline is inherently noisy, set
525
+ `baseline_max_failure_ratio` in `tdd.toml`:
526
+
527
+ ```toml
528
+ [project.legacy]
529
+ root = "legacy"
530
+ adapter = "pytest"
531
+ baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
532
+ ```
533
+
534
+ The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
535
+
536
+ ### Standing-failure delta
537
+
538
+ When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
539
+ event partitioning the current standing failures into:
540
+
541
+ - **new** — absent from the previous run's baseline for this project (growing problem)
542
+ - **inherited** — also present in the previous baseline (stable background noise)
543
+ - **resolved** — in the previous baseline but absent now (fixed between runs)
544
+
545
+ The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
546
+ baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
547
+ nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
548
+
549
+ ### Environment-dependent suites
550
+
551
+ Declare a `health_command` on a project to probe reachability before `run start` captures its
552
+ baseline:
553
+
554
+ ```toml
555
+ [project.backend]
556
+ root = "backend"
557
+ adapter = "pytest"
558
+ health_command = "curl -fsS http://localhost:8080/healthz"
559
+ ```
560
+
561
+ If the command exits non-zero, `run start` refuses immediately with `reason:
562
+ "services_unreachable"` — before claiming the worktree, before probing, before writing anything
563
+ to the ledger. The error names the project and its exit code so the agent can surface the exact
564
+ diagnosis rather than recording a ~1000-failure baseline and continuing.
565
+
566
+ There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
567
+ The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
568
+ is the marker that a project requires live services; no separate boolean is needed.
569
+
407
570
  ## Running a long baseline
408
571
 
409
572
  `run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
@@ -444,10 +607,11 @@ The CLI cannot compel an agent — only the harness can.
444
607
  artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
445
608
  `advance` refuses an unchanged tree unless `--retry`.
446
609
 
447
- **Recorded, never blocked:** non-stub writes during RED, undeclared file touches, scope
448
- divergence, extra attempts. Prevention rules with edge cases produce false denials, and a
449
- blocked agent improvises around them — putting it right back in the reporting path the
450
- tool exists to keep it out of.
610
+ **Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
611
+ scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
612
+ and a blocked agent improvises around them — putting it right back in the reporting path
613
+ the tool exists to keep it out of. Undeclared file touches are an exception at run close:
614
+ see the run-close gate above.
451
615
 
452
616
  **Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
453
617
  stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
@@ -74,9 +74,16 @@ end of the run.
74
74
  - The cycle absorbed undeclared scope, or surfaced follow-up work:
75
75
  `--key unplanned_change` / `--key new_work_raised`.
76
76
 
77
+ To capture *why* something happened — a plan assumption that proved wrong, the reason an
78
+ integrity event fired — use `tdd note "<text>"` at the moment you know. A note written
79
+ during an open cycle is stamped with that cycle and phase and appears in the friction log
80
+ as a blockquote alongside the cycle's telemetry. After the run ends, `tdd note` attaches
81
+ at run level and renders in a dedicated **Executor narrative** section. Notes are unverified
82
+ by design; write them as claims, not measurements.
83
+
77
84
  Narrative that spans cycles or happened after the run (CI failures, patterns) goes as
78
- markdown appended below the rendered friction log after `tdd log render` — never into a
79
- cycle annotation it doesn't belong to.
85
+ `tdd note` after the run ends, or as markdown appended below the rendered friction log
86
+ after `tdd log render` — never into a cycle annotation it doesn't belong to.
80
87
 
81
88
  ## When a suite run changes nothing
82
89
 
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.7.0"
6
+ __version__ = "0.9.0"
@@ -25,6 +25,7 @@ class Verdict:
25
25
  target: str | None = None
26
26
  target_outcome: str = NOT_FOUND
27
27
  target_failure: str = ""
28
+ target_evidence: str = ""
28
29
  passed: list[str] = field(default_factory=list)
29
30
  failed: list[str] = field(default_factory=list)
30
31
  duration_ms: int = 0
@@ -139,6 +140,14 @@ class Adapter:
139
140
  prefix = f"{self.project.name}::"
140
141
  return qualified[len(prefix) :] if qualified.startswith(prefix) else qualified
141
142
 
143
+ def normalise_id(self, test_id: str) -> str:
144
+ """Return the canonical form of a declared target id for matching against collected ids.
145
+
146
+ The default is an identity — subclasses override when the runner's collected
147
+ ids differ from a natural human spelling (e.g. vitest's describe/test separator).
148
+ """
149
+ return test_id
150
+
142
151
  def run(self, target: str | None = None) -> Verdict:
143
152
  raise NotImplementedError
144
153
 
@@ -206,6 +215,14 @@ class Adapter:
206
215
  return None
207
216
  return {k: os.path.expandvars(v) for k, v in merged.items()}
208
217
 
218
+ def lint_target_id(self, native: str) -> str | None:
219
+ """Return a problem message when `native` can never match a collected id, else None."""
220
+ return None
221
+
222
+ def target_path(self, native: str) -> str | None:
223
+ """Return the file-path portion of `native`, or None for non-path-bearing ids."""
224
+ return None
225
+
209
226
  def stub_hint(self) -> str:
210
227
  """The language idiom for a stub body, quoted into the create_stub directive."""
211
228
  return "a body that fails loudly, never working logic"
@@ -168,6 +168,9 @@ class ExecAdapter(Adapter):
168
168
  if target == qualified:
169
169
  verdict.target_outcome = FAILED
170
170
  verdict.target_failure = clip_failure(combined)
171
+ verdict.target_evidence = next(
172
+ (ln for ln in reversed(combined.splitlines()) if ln.strip()), ""
173
+ )
171
174
 
172
175
  verdict.duration_ms = int((time.monotonic() - started) * 1000)
173
176
  return verdict