tdd-cli 0.7.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/CHANGELOG.md +126 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/PKG-INFO +169 -5
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/README.md +168 -4
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/skills/tdd-drive/SKILL.md +9 -2
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/base.py +17 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/exec_adapter.py +3 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/gradle_adapter.py +13 -1
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/pytest_adapter.py +20 -1
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/vitest_adapter.py +78 -32
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/xctest_adapter.py +19 -1
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/advance.py +152 -29
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/cli.py +255 -33
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/config.py +29 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/contract.py +15 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/identity.py +16 -4
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/ledger.py +106 -3
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/machine.py +33 -5
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/render.py +33 -2
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/staging.py +6 -0
- tdd_cli-0.9.0/src/tddcli/target_lint.py +61 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/conftest.py +12 -0
- tdd_cli-0.9.0/tests/test_advance_adoption.py +126 -0
- tdd_cli-0.9.0/tests/test_ancillary_files.py +72 -0
- tdd_cli-0.9.0/tests/test_artifact_regeneration.py +166 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_baseline_integrity.py +266 -1
- tdd_cli-0.9.0/tests/test_baseline_sanity.py +229 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_config_and_staging.py +37 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_contract.py +120 -0
- tdd_cli-0.9.0/tests/test_evidence_extraction.py +226 -0
- tdd_cli-0.9.0/tests/test_executor_attribution.py +133 -0
- tdd_cli-0.9.0/tests/test_executor_notes.py +209 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_heartbeat.py +56 -6
- tdd_cli-0.9.0/tests/test_id_normalisation.py +127 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_release_surface.py +15 -0
- tdd_cli-0.9.0/tests/test_sensitivity_evidence.py +157 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_snapshot_and_identity.py +92 -0
- tdd_cli-0.9.0/tests/test_target_lint.py +202 -0
- tdd_cli-0.9.0/tests/test_undeclared_close_gate.py +106 -0
- tdd_cli-0.9.0/tests/test_undeclared_dedup.py +168 -0
- tdd_cli-0.7.0/tests/test_artifact_regeneration.py +0 -76
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/.gitignore +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/LICENSE +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/SECURITY.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/plan.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/pyproject.toml +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_batch_collection.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_concurrent_advance.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_doctor_blockers.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_exec_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_failure_clipping.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_gradle_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_named_leases_and_timeout.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_project_commands.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_project_env.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_suite_overrides.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_timing_visibility.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_vitest_adapter.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_worker_leases.py +0 -0
- {tdd_cli-0.7.0 → tdd_cli-0.9.0}/tests/test_xctest_adapter.py +0 -0
|
@@ -6,6 +6,132 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.9.0] - 2026-09-03
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **`tdd note` — an executor-narrative channel.** `tdd note "<text>"` attaches
|
|
14
|
+
a phase-stamped note to the current cycle, or a run-level note once the run
|
|
15
|
+
has ended (it falls back to the latest run, so no active run is required).
|
|
16
|
+
Cycle notes render in the friction log as blockquote claims under their cycle;
|
|
17
|
+
run-level notes render under a new `## Executor narrative` section, which is
|
|
18
|
+
omitted when there are none. Envelopes that carry an integrity event now nudge
|
|
19
|
+
for a note while the reason is fresh (silenced once the cycle has one), and
|
|
20
|
+
the terminal `COMPLETE` envelope — via `advance` or a final-cycle skip — asks
|
|
21
|
+
for a closing narrative before rendering. Stored in a new `note` table via a
|
|
22
|
+
v7→v8 ledger migration (#77, #94).
|
|
23
|
+
|
|
24
|
+
- **Baseline sanity gate.** `run start` refuses a baseline whose failing ratio
|
|
25
|
+
exceeds a threshold (default 0.5, suites under 10 collected tests exempt) with
|
|
26
|
+
`reason: "baseline_implausible"`, on the grounds that a mostly-red suite means
|
|
27
|
+
the environment is broken rather than the code. A per-project
|
|
28
|
+
`baseline_max_failure_ratio` in `tdd.toml` overrides the default, and
|
|
29
|
+
`--accept-baseline` bypasses the gate and logs a `baseline_accepted` integrity
|
|
30
|
+
event (#67, #84).
|
|
31
|
+
|
|
32
|
+
- **Standing-failure delta.** A non-empty baseline is now compared against the
|
|
33
|
+
previous run's baseline for the same worktree and a `baseline_standing_delta`
|
|
34
|
+
event partitions the failing set into new versus inherited failures, so a
|
|
35
|
+
pre-existing red test is distinguishable from one that broke since the last
|
|
36
|
+
run (#67, #84).
|
|
37
|
+
|
|
38
|
+
- **Per-project `health_command`.** An optional command in `tdd.toml` that is
|
|
39
|
+
probed before baseline capture; if it fails, `run start` refuses with
|
|
40
|
+
`reason: "services_unreachable"` instead of recording a baseline against a
|
|
41
|
+
down dependency (#67, #84).
|
|
42
|
+
|
|
43
|
+
- **Declared-target lint.** `plan register` and `run start` now lint every
|
|
44
|
+
declared target against the adapter's id grammar (pytest `::`, vitest ` > `,
|
|
45
|
+
gradle `Class/method`, xctest `Bundle/Class/testMethod`) and against project
|
|
46
|
+
roots: a target path that duplicates a non-`.` project root prefix is refused
|
|
47
|
+
unless the nested path genuinely exists in the worktree. Findings are returned
|
|
48
|
+
under `reason: "target_lint"`; `run start` re-lints the stored contract
|
|
49
|
+
against the current config before claiming a baseline (#71, #88).
|
|
50
|
+
|
|
51
|
+
- **Per-adapter sensitivity evidence line.** Each adapter now extracts a
|
|
52
|
+
`target_evidence` line from a failing target — the first `E` line for pytest
|
|
53
|
+
(skipping the xdist worker header), the first `: error:` line in the test's
|
|
54
|
+
window for xctest, `failureMessages[0]` for vitest, the junit failure message
|
|
55
|
+
for gradle, and the last non-empty output line for exec. The sensitivity check
|
|
56
|
+
persists it as `sensitivity_check.evidence_line` (v6→v7 migration) and the
|
|
57
|
+
friction log's observed snippet prefers it over the raw first line, rendering
|
|
58
|
+
`<no assertion line captured>` when empty and keeping the tail of over-long
|
|
59
|
+
lines. Legacy rows with no stored evidence keep the first-line fallback (#68,
|
|
60
|
+
#90).
|
|
61
|
+
|
|
62
|
+
- **Diagnosable executor attribution.** `TDD_EXECUTOR_MODEL` lets a harness
|
|
63
|
+
declare the executor identity (`source: declared`), taking precedence over
|
|
64
|
+
transcript detection. When identity cannot be resolved, `Executor.reason`
|
|
65
|
+
says why — session id unset, transcript not found, or transcript without a
|
|
66
|
+
model record — and `run start` logs an `executor_unknown` event and surfaces
|
|
67
|
+
`executor_warning` in its envelope. `tdd doctor` gains an informational
|
|
68
|
+
executor-identity check (#74, #92).
|
|
69
|
+
|
|
70
|
+
### Changed
|
|
71
|
+
|
|
72
|
+
- **Target adoption is evaluated in the same `advance`.** When a cycle's
|
|
73
|
+
declared target is missing and exactly one new test appears, `advance`
|
|
74
|
+
adopts it (logging `declared_test_mismatch`) and judges RED — or drives the
|
|
75
|
+
sensitivity check when it passed — from the suite run that already happened,
|
|
76
|
+
instead of asking for a re-run. A declared id that differs
|
|
77
|
+
from a single same-file candidate only by separator normalisation is
|
|
78
|
+
disambiguated and adopted without asking; two same-file candidates still
|
|
79
|
+
require an explicit `tdd target` (#72, #91).
|
|
80
|
+
|
|
81
|
+
- CI now tests on Python 3.11 and 3.14 only (#87).
|
|
82
|
+
|
|
83
|
+
## [0.8.0] - 2026-08-28
|
|
84
|
+
|
|
85
|
+
### Added
|
|
86
|
+
|
|
87
|
+
- **`--reuse-baselines` caches baseline probes by content hash.** `run start`
|
|
88
|
+
can skip re-probing a project whose tree is unchanged: a probe result is
|
|
89
|
+
cached keyed by `(project, tree_hash, config_sha)` — where `tree_hash` folds in
|
|
90
|
+
every upstream producer root — and an identical rerun emits `baseline_reused`
|
|
91
|
+
instead of `baseline_captured`, reusing the cached failing set and collection
|
|
92
|
+
snapshot rather than re-running the suite. Provenance is recorded on the
|
|
93
|
+
baseline row (`source`), and `--reuse-max-age` re-probes any entry older than
|
|
94
|
+
the given age. Off by default; the cache stays empty unless the flag is passed
|
|
95
|
+
(#45, #59).
|
|
96
|
+
|
|
97
|
+
- **`--baseline-jobs` parallelizes baseline probing.** `run start` probes each
|
|
98
|
+
project's baseline under a bounded `ThreadPoolExecutor` when `--baseline-jobs`
|
|
99
|
+
is greater than 1 (default 1, must be >= 1). The `baseline_captured` heartbeat
|
|
100
|
+
survives the pool, and a worker probe that raises becomes an attributed failure
|
|
101
|
+
that aborts cleanly and frees the worktree rather than wedging it (#46, #62).
|
|
102
|
+
|
|
103
|
+
- **Plan-level `ancillary_files`.** A top-level front-matter key declaring
|
|
104
|
+
cross-project or companion paths a plan touches (README, generated fixtures,
|
|
105
|
+
sibling-project files). Declared ancillary paths are bucketed into their own
|
|
106
|
+
staging bucket, committed with the cycle, and fire no `undeclared_file_touched`
|
|
107
|
+
event. Validated as a list of strings at registration and persisted to the
|
|
108
|
+
ledger via a v5→v6 migration (#70, #80).
|
|
109
|
+
|
|
110
|
+
- **Run-close gate on undeclared touched paths.** `run close` now blocks when a
|
|
111
|
+
path previously flagged as `undeclared_file_touched` is still dirty in the
|
|
112
|
+
worktree, so undeclared changes cannot slip through at the end of a run. A
|
|
113
|
+
flagged path that was since committed does not block; one that has vanished is
|
|
114
|
+
reported via a new `undeclared_file_dropped` event rather than blocking (#69,
|
|
115
|
+
#81).
|
|
116
|
+
|
|
117
|
+
- **Reserved per-cycle `meta:` passthrough.** A cycle may carry an authored
|
|
118
|
+
`meta:` mapping in the plan front-matter; it round-trips through storage
|
|
119
|
+
unchanged and is available for plan-time metadata. A non-mapping `meta:`
|
|
120
|
+
hard-fails registration with a `ContractError` (#58, #65).
|
|
121
|
+
|
|
122
|
+
### Fixed
|
|
123
|
+
|
|
124
|
+
- `undeclared_file_touched` is deduplicated within a cycle, so a path touched
|
|
125
|
+
across multiple phases no longer floods the cycle with repeated events (#55,
|
|
126
|
+
#61).
|
|
127
|
+
- vitest test ids are normalised on the describe/test separator before matching,
|
|
128
|
+
so a formatting-only difference between a declared target and the observed
|
|
129
|
+
verdict is no longer reported as a spurious `declared_test_mismatch` (#57,
|
|
130
|
+
#63).
|
|
131
|
+
- The `stale_artifact` event is suppressed when the tool auto-regenerates the
|
|
132
|
+
artifact and commits it, so a successful regeneration no longer also emits a
|
|
133
|
+
staleness warning (#64).
|
|
134
|
+
|
|
9
135
|
## [0.7.0] - 2026-08-23
|
|
10
136
|
|
|
11
137
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -227,6 +227,8 @@ cycles:
|
|
|
227
227
|
stub_expected: ["app/exception_map.py"]
|
|
228
228
|
commit_red: "test: unmapped exception is not swallowed"
|
|
229
229
|
commit_green: "feat: domain exception map skeleton"
|
|
230
|
+
meta: # optional authored-at-plan-time metadata; opaque to the tool
|
|
231
|
+
covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
|
|
230
232
|
- n: 8
|
|
231
233
|
project: backend
|
|
232
234
|
pin_cycle: true # characterisation; passes on arrival by design
|
|
@@ -238,9 +240,51 @@ cycles:
|
|
|
238
240
|
- "backend::tests/test_openapi.py::test_upload_body_schema"
|
|
239
241
|
- "frontend::services/__tests__/upload.test.ts > matches contract"
|
|
240
242
|
annotation_keys: ["literal_detail_handlers_kept"]
|
|
243
|
+
ancillary_files:
|
|
244
|
+
- frontend/src/api/registerClient.ts # type-break from regenerated client
|
|
245
|
+
- docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
|
|
241
246
|
---
|
|
242
247
|
```
|
|
243
248
|
|
|
249
|
+
**Top-level keys:**
|
|
250
|
+
|
|
251
|
+
`annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
|
|
252
|
+
gate checks that every key is present before the plan can be marked complete.
|
|
253
|
+
|
|
254
|
+
`ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
|
|
255
|
+
touch outside any registered project root (cross-project ripples, companion documents).
|
|
256
|
+
Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
|
|
257
|
+
matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
|
|
258
|
+
and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
|
|
259
|
+
not on the list still fires `undeclared_file_touched` exactly as today. This is a
|
|
260
|
+
plan-level list only; per-cycle overrides are a planned follow-up.
|
|
261
|
+
|
|
262
|
+
### Run-close gate for undeclared file touches
|
|
263
|
+
|
|
264
|
+
When the last declared cycle closes, the tool gathers every path that appeared in any
|
|
265
|
+
`undeclared_file_touched` event across the run and checks the worktree:
|
|
266
|
+
|
|
267
|
+
- **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
|
|
268
|
+
the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
|
|
269
|
+
commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
|
|
270
|
+
and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
|
|
271
|
+
- **Committed during the run** → clean; the run completes normally.
|
|
272
|
+
- **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
|
|
273
|
+
emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
|
|
274
|
+
makes a silent drop visible without blocking, since the file is already gone.
|
|
275
|
+
|
|
276
|
+
The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
|
|
277
|
+
so they are never seen by the gate.
|
|
278
|
+
|
|
279
|
+
**Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
|
|
280
|
+
`stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
|
|
281
|
+
`commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
|
|
282
|
+
`meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
|
|
283
|
+
but its *contents* are opaque to the tool — any key/value pairs are accepted and
|
|
284
|
+
round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
|
|
285
|
+
authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
|
|
286
|
+
needs to read back. Other unknown per-cycle keys are silently ignored.
|
|
287
|
+
|
|
244
288
|
Absent front-matter is legitimate — the run proceeds as `undeclared` with
|
|
245
289
|
`--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
|
|
246
290
|
hard-fails registration: it is almost always a defect in the planning process, and that
|
|
@@ -275,6 +319,13 @@ rendered, never written. Judgement enters in exactly two ways:
|
|
|
275
319
|
`commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
|
|
276
320
|
`plan_defect` is the one that matters most: it records where the plan and the codebase
|
|
277
321
|
disagreed, which is precisely what the next plan needs to know.
|
|
322
|
+
- **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
|
|
323
|
+
moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
|
|
324
|
+
notes are written after the run ends and attach to the run. Both render in the friction
|
|
325
|
+
log as visually distinct blockquotes (claims, not measurements). Use notes to record
|
|
326
|
+
*why* something happened — a plan assumption that was wrong, an integrity event that the
|
|
327
|
+
telemetry already captures but cannot explain. Notes are unverified by design; an auditor
|
|
328
|
+
compares claims against reality.
|
|
278
329
|
- **Per run, as prose appended below the rendered document.** Legitimate and expected —
|
|
279
330
|
post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
|
|
280
331
|
is unverified: an auditor should trust the projected sections and read appended
|
|
@@ -397,6 +448,7 @@ and is never reclassified as a pin.
|
|
|
397
448
|
| `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
|
|
398
449
|
| `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
|
|
399
450
|
| `tdd annotate --key --value` | attach judgement to the current cycle |
|
|
451
|
+
| `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
|
|
400
452
|
| `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
|
|
401
453
|
| `tdd resume [--unblock --note]` | reconstruct position; human intervention |
|
|
402
454
|
| `tdd log render [--out]` | project the ledger into a friction log |
|
|
@@ -418,6 +470,50 @@ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
|
|
|
418
470
|
Use this when a cycle may edit files outside the predicted reachable set and you want every
|
|
419
471
|
project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
|
|
420
472
|
|
|
473
|
+
### Reusing baselines across runs (R9.5e)
|
|
474
|
+
|
|
475
|
+
On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
|
|
476
|
+
wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
|
|
477
|
+
when nothing in the project changed:
|
|
478
|
+
|
|
479
|
+
tdd run start --plan tasks/plan.md --reuse-baselines
|
|
480
|
+
|
|
481
|
+
The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
|
|
482
|
+
If the key matches a previous entry, the cached failing set and collection snapshot are used —
|
|
483
|
+
no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
|
|
484
|
+
`source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
|
|
485
|
+
projects were skipped. Without the flag, the cache is neither read nor written.
|
|
486
|
+
|
|
487
|
+
A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
|
|
488
|
+
(see below) — reuse is loud by design, never silent.
|
|
489
|
+
|
|
490
|
+
To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
|
|
491
|
+
|
|
492
|
+
tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
|
|
493
|
+
|
|
494
|
+
Entries older than the given threshold are ignored and re-probed fresh.
|
|
495
|
+
|
|
496
|
+
### Parallel baseline probing (R9.5f)
|
|
497
|
+
|
|
498
|
+
By default `run start` probes one project at a time. On a repo with many independent projects the
|
|
499
|
+
wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
|
|
500
|
+
projects concurrently:
|
|
501
|
+
|
|
502
|
+
tdd run start --plan tasks/plan.md --baseline-jobs 4
|
|
503
|
+
|
|
504
|
+
Each probe is independent — one adapter instance per project — so concurrency does not affect
|
|
505
|
+
which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
|
|
506
|
+
heartbeat is still emitted per project as each probe completes.
|
|
507
|
+
|
|
508
|
+
The default is `--baseline-jobs 1` (serial). Raise it deliberately:
|
|
509
|
+
|
|
510
|
+
- **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
|
|
511
|
+
- **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
|
|
512
|
+
or declare a `lease` name — leased suites serialize automatically even inside the pool.
|
|
513
|
+
|
|
514
|
+
If any probe fails, `run start` returns a failure attributed to that project and no run row is
|
|
515
|
+
created, so the worktree is immediately retryable.
|
|
516
|
+
|
|
421
517
|
### When a sweep reaches an un-baselined project (R9.5d)
|
|
422
518
|
|
|
423
519
|
If an edit during a run touches a file owned by an artifact that was outside the predicted
|
|
@@ -432,6 +528,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
|
|
|
432
528
|
failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
|
|
433
529
|
with the project properly baselined.
|
|
434
530
|
|
|
531
|
+
### Baseline sanity gate (R9.5g)
|
|
532
|
+
|
|
533
|
+
`run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
|
|
534
|
+
environment, not the code, is broken. A project is flagged when it collected at least 10 tests
|
|
535
|
+
(`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
|
|
536
|
+
|
|
537
|
+
error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
|
|
538
|
+
|
|
539
|
+
The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
|
|
540
|
+
not small pre-existing failure sets (baseline subtraction handles those). A project collecting
|
|
541
|
+
fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
|
|
542
|
+
not a broken stack.
|
|
543
|
+
|
|
544
|
+
Pass `--accept-baseline` to override the refusal and proceed anyway:
|
|
545
|
+
|
|
546
|
+
tdd run start --plan tasks/plan.md --accept-baseline
|
|
547
|
+
|
|
548
|
+
An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
|
|
549
|
+
`tdd metrics` and the friction log) so the override is never silent. The gate runs on every
|
|
550
|
+
baseline, including reused ones.
|
|
551
|
+
|
|
552
|
+
To raise the threshold for a project whose legitimate baseline is inherently noisy, set
|
|
553
|
+
`baseline_max_failure_ratio` in `tdd.toml`:
|
|
554
|
+
|
|
555
|
+
```toml
|
|
556
|
+
[project.legacy]
|
|
557
|
+
root = "legacy"
|
|
558
|
+
adapter = "pytest"
|
|
559
|
+
baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
|
|
560
|
+
```
|
|
561
|
+
|
|
562
|
+
The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
|
|
563
|
+
|
|
564
|
+
### Standing-failure delta
|
|
565
|
+
|
|
566
|
+
When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
|
|
567
|
+
event partitioning the current standing failures into:
|
|
568
|
+
|
|
569
|
+
- **new** — absent from the previous run's baseline for this project (growing problem)
|
|
570
|
+
- **inherited** — also present in the previous baseline (stable background noise)
|
|
571
|
+
- **resolved** — in the previous baseline but absent now (fixed between runs)
|
|
572
|
+
|
|
573
|
+
The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
|
|
574
|
+
baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
|
|
575
|
+
nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
|
|
576
|
+
|
|
577
|
+
### Environment-dependent suites
|
|
578
|
+
|
|
579
|
+
Declare a `health_command` on a project to probe reachability before `run start` captures its
|
|
580
|
+
baseline:
|
|
581
|
+
|
|
582
|
+
```toml
|
|
583
|
+
[project.backend]
|
|
584
|
+
root = "backend"
|
|
585
|
+
adapter = "pytest"
|
|
586
|
+
health_command = "curl -fsS http://localhost:8080/healthz"
|
|
587
|
+
```
|
|
588
|
+
|
|
589
|
+
If the command exits non-zero, `run start` refuses immediately with `reason:
|
|
590
|
+
"services_unreachable"` — before claiming the worktree, before probing, before writing anything
|
|
591
|
+
to the ledger. The error names the project and its exit code so the agent can surface the exact
|
|
592
|
+
diagnosis rather than recording a ~1000-failure baseline and continuing.
|
|
593
|
+
|
|
594
|
+
There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
|
|
595
|
+
The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
|
|
596
|
+
is the marker that a project requires live services; no separate boolean is needed.
|
|
597
|
+
|
|
435
598
|
## Running a long baseline
|
|
436
599
|
|
|
437
600
|
`run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
|
|
@@ -472,10 +635,11 @@ The CLI cannot compel an agent — only the harness can.
|
|
|
472
635
|
artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
|
|
473
636
|
`advance` refuses an unchanged tree unless `--retry`.
|
|
474
637
|
|
|
475
|
-
**Recorded, never blocked:** non-stub writes during RED, undeclared file touches,
|
|
476
|
-
divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
477
|
-
blocked agent improvises around them — putting it right back in the reporting path
|
|
478
|
-
tool exists to keep it out of.
|
|
638
|
+
**Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
|
|
639
|
+
scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
640
|
+
and a blocked agent improvises around them — putting it right back in the reporting path
|
|
641
|
+
the tool exists to keep it out of. Undeclared file touches are an exception at run close:
|
|
642
|
+
see the run-close gate above.
|
|
479
643
|
|
|
480
644
|
**Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
|
|
481
645
|
stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
|
|
@@ -199,6 +199,8 @@ cycles:
|
|
|
199
199
|
stub_expected: ["app/exception_map.py"]
|
|
200
200
|
commit_red: "test: unmapped exception is not swallowed"
|
|
201
201
|
commit_green: "feat: domain exception map skeleton"
|
|
202
|
+
meta: # optional authored-at-plan-time metadata; opaque to the tool
|
|
203
|
+
covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
|
|
202
204
|
- n: 8
|
|
203
205
|
project: backend
|
|
204
206
|
pin_cycle: true # characterisation; passes on arrival by design
|
|
@@ -210,9 +212,51 @@ cycles:
|
|
|
210
212
|
- "backend::tests/test_openapi.py::test_upload_body_schema"
|
|
211
213
|
- "frontend::services/__tests__/upload.test.ts > matches contract"
|
|
212
214
|
annotation_keys: ["literal_detail_handlers_kept"]
|
|
215
|
+
ancillary_files:
|
|
216
|
+
- frontend/src/api/registerClient.ts # type-break from regenerated client
|
|
217
|
+
- docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
|
|
213
218
|
---
|
|
214
219
|
```
|
|
215
220
|
|
|
221
|
+
**Top-level keys:**
|
|
222
|
+
|
|
223
|
+
`annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
|
|
224
|
+
gate checks that every key is present before the plan can be marked complete.
|
|
225
|
+
|
|
226
|
+
`ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
|
|
227
|
+
touch outside any registered project root (cross-project ripples, companion documents).
|
|
228
|
+
Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
|
|
229
|
+
matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
|
|
230
|
+
and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
|
|
231
|
+
not on the list still fires `undeclared_file_touched` exactly as today. This is a
|
|
232
|
+
plan-level list only; per-cycle overrides are a planned follow-up.
|
|
233
|
+
|
|
234
|
+
### Run-close gate for undeclared file touches
|
|
235
|
+
|
|
236
|
+
When the last declared cycle closes, the tool gathers every path that appeared in any
|
|
237
|
+
`undeclared_file_touched` event across the run and checks the worktree:
|
|
238
|
+
|
|
239
|
+
- **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
|
|
240
|
+
the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
|
|
241
|
+
commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
|
|
242
|
+
and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
|
|
243
|
+
- **Committed during the run** → clean; the run completes normally.
|
|
244
|
+
- **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
|
|
245
|
+
emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
|
|
246
|
+
makes a silent drop visible without blocking, since the file is already gone.
|
|
247
|
+
|
|
248
|
+
The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
|
|
249
|
+
so they are never seen by the gate.
|
|
250
|
+
|
|
251
|
+
**Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
|
|
252
|
+
`stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
|
|
253
|
+
`commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
|
|
254
|
+
`meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
|
|
255
|
+
but its *contents* are opaque to the tool — any key/value pairs are accepted and
|
|
256
|
+
round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
|
|
257
|
+
authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
|
|
258
|
+
needs to read back. Other unknown per-cycle keys are silently ignored.
|
|
259
|
+
|
|
216
260
|
Absent front-matter is legitimate — the run proceeds as `undeclared` with
|
|
217
261
|
`--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
|
|
218
262
|
hard-fails registration: it is almost always a defect in the planning process, and that
|
|
@@ -247,6 +291,13 @@ rendered, never written. Judgement enters in exactly two ways:
|
|
|
247
291
|
`commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
|
|
248
292
|
`plan_defect` is the one that matters most: it records where the plan and the codebase
|
|
249
293
|
disagreed, which is precisely what the next plan needs to know.
|
|
294
|
+
- **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
|
|
295
|
+
moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
|
|
296
|
+
notes are written after the run ends and attach to the run. Both render in the friction
|
|
297
|
+
log as visually distinct blockquotes (claims, not measurements). Use notes to record
|
|
298
|
+
*why* something happened — a plan assumption that was wrong, an integrity event that the
|
|
299
|
+
telemetry already captures but cannot explain. Notes are unverified by design; an auditor
|
|
300
|
+
compares claims against reality.
|
|
250
301
|
- **Per run, as prose appended below the rendered document.** Legitimate and expected —
|
|
251
302
|
post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
|
|
252
303
|
is unverified: an auditor should trust the projected sections and read appended
|
|
@@ -369,6 +420,7 @@ and is never reclassified as a pin.
|
|
|
369
420
|
| `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
|
|
370
421
|
| `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
|
|
371
422
|
| `tdd annotate --key --value` | attach judgement to the current cycle |
|
|
423
|
+
| `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
|
|
372
424
|
| `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
|
|
373
425
|
| `tdd resume [--unblock --note]` | reconstruct position; human intervention |
|
|
374
426
|
| `tdd log render [--out]` | project the ledger into a friction log |
|
|
@@ -390,6 +442,50 @@ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
|
|
|
390
442
|
Use this when a cycle may edit files outside the predicted reachable set and you want every
|
|
391
443
|
project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
|
|
392
444
|
|
|
445
|
+
### Reusing baselines across runs (R9.5e)
|
|
446
|
+
|
|
447
|
+
On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
|
|
448
|
+
wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
|
|
449
|
+
when nothing in the project changed:
|
|
450
|
+
|
|
451
|
+
tdd run start --plan tasks/plan.md --reuse-baselines
|
|
452
|
+
|
|
453
|
+
The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
|
|
454
|
+
If the key matches a previous entry, the cached failing set and collection snapshot are used —
|
|
455
|
+
no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
|
|
456
|
+
`source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
|
|
457
|
+
projects were skipped. Without the flag, the cache is neither read nor written.
|
|
458
|
+
|
|
459
|
+
A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
|
|
460
|
+
(see below) — reuse is loud by design, never silent.
|
|
461
|
+
|
|
462
|
+
To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
|
|
463
|
+
|
|
464
|
+
tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
|
|
465
|
+
|
|
466
|
+
Entries older than the given threshold are ignored and re-probed fresh.
|
|
467
|
+
|
|
468
|
+
### Parallel baseline probing (R9.5f)
|
|
469
|
+
|
|
470
|
+
By default `run start` probes one project at a time. On a repo with many independent projects the
|
|
471
|
+
wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
|
|
472
|
+
projects concurrently:
|
|
473
|
+
|
|
474
|
+
tdd run start --plan tasks/plan.md --baseline-jobs 4
|
|
475
|
+
|
|
476
|
+
Each probe is independent — one adapter instance per project — so concurrency does not affect
|
|
477
|
+
which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
|
|
478
|
+
heartbeat is still emitted per project as each probe completes.
|
|
479
|
+
|
|
480
|
+
The default is `--baseline-jobs 1` (serial). Raise it deliberately:
|
|
481
|
+
|
|
482
|
+
- **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
|
|
483
|
+
- **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
|
|
484
|
+
or declare a `lease` name — leased suites serialize automatically even inside the pool.
|
|
485
|
+
|
|
486
|
+
If any probe fails, `run start` returns a failure attributed to that project and no run row is
|
|
487
|
+
created, so the worktree is immediately retryable.
|
|
488
|
+
|
|
393
489
|
### When a sweep reaches an un-baselined project (R9.5d)
|
|
394
490
|
|
|
395
491
|
If an edit during a run touches a file owned by an artifact that was outside the predicted
|
|
@@ -404,6 +500,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
|
|
|
404
500
|
failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
|
|
405
501
|
with the project properly baselined.
|
|
406
502
|
|
|
503
|
+
### Baseline sanity gate (R9.5g)
|
|
504
|
+
|
|
505
|
+
`run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
|
|
506
|
+
environment, not the code, is broken. A project is flagged when it collected at least 10 tests
|
|
507
|
+
(`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
|
|
508
|
+
|
|
509
|
+
error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
|
|
510
|
+
|
|
511
|
+
The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
|
|
512
|
+
not small pre-existing failure sets (baseline subtraction handles those). A project collecting
|
|
513
|
+
fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
|
|
514
|
+
not a broken stack.
|
|
515
|
+
|
|
516
|
+
Pass `--accept-baseline` to override the refusal and proceed anyway:
|
|
517
|
+
|
|
518
|
+
tdd run start --plan tasks/plan.md --accept-baseline
|
|
519
|
+
|
|
520
|
+
An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
|
|
521
|
+
`tdd metrics` and the friction log) so the override is never silent. The gate runs on every
|
|
522
|
+
baseline, including reused ones.
|
|
523
|
+
|
|
524
|
+
To raise the threshold for a project whose legitimate baseline is inherently noisy, set
|
|
525
|
+
`baseline_max_failure_ratio` in `tdd.toml`:
|
|
526
|
+
|
|
527
|
+
```toml
|
|
528
|
+
[project.legacy]
|
|
529
|
+
root = "legacy"
|
|
530
|
+
adapter = "pytest"
|
|
531
|
+
baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
|
|
532
|
+
```
|
|
533
|
+
|
|
534
|
+
The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
|
|
535
|
+
|
|
536
|
+
### Standing-failure delta
|
|
537
|
+
|
|
538
|
+
When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
|
|
539
|
+
event partitioning the current standing failures into:
|
|
540
|
+
|
|
541
|
+
- **new** — absent from the previous run's baseline for this project (growing problem)
|
|
542
|
+
- **inherited** — also present in the previous baseline (stable background noise)
|
|
543
|
+
- **resolved** — in the previous baseline but absent now (fixed between runs)
|
|
544
|
+
|
|
545
|
+
The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
|
|
546
|
+
baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
|
|
547
|
+
nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
|
|
548
|
+
|
|
549
|
+
### Environment-dependent suites
|
|
550
|
+
|
|
551
|
+
Declare a `health_command` on a project to probe reachability before `run start` captures its
|
|
552
|
+
baseline:
|
|
553
|
+
|
|
554
|
+
```toml
|
|
555
|
+
[project.backend]
|
|
556
|
+
root = "backend"
|
|
557
|
+
adapter = "pytest"
|
|
558
|
+
health_command = "curl -fsS http://localhost:8080/healthz"
|
|
559
|
+
```
|
|
560
|
+
|
|
561
|
+
If the command exits non-zero, `run start` refuses immediately with `reason:
|
|
562
|
+
"services_unreachable"` — before claiming the worktree, before probing, before writing anything
|
|
563
|
+
to the ledger. The error names the project and its exit code so the agent can surface the exact
|
|
564
|
+
diagnosis rather than recording a ~1000-failure baseline and continuing.
|
|
565
|
+
|
|
566
|
+
There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
|
|
567
|
+
The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
|
|
568
|
+
is the marker that a project requires live services; no separate boolean is needed.
|
|
569
|
+
|
|
407
570
|
## Running a long baseline
|
|
408
571
|
|
|
409
572
|
`run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
|
|
@@ -444,10 +607,11 @@ The CLI cannot compel an agent — only the harness can.
|
|
|
444
607
|
artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
|
|
445
608
|
`advance` refuses an unchanged tree unless `--retry`.
|
|
446
609
|
|
|
447
|
-
**Recorded, never blocked:** non-stub writes during RED, undeclared file touches,
|
|
448
|
-
divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
449
|
-
blocked agent improvises around them — putting it right back in the reporting path
|
|
450
|
-
tool exists to keep it out of.
|
|
610
|
+
**Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
|
|
611
|
+
scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
612
|
+
and a blocked agent improvises around them — putting it right back in the reporting path
|
|
613
|
+
the tool exists to keep it out of. Undeclared file touches are an exception at run close:
|
|
614
|
+
see the run-close gate above.
|
|
451
615
|
|
|
452
616
|
**Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
|
|
453
617
|
stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
|
|
@@ -74,9 +74,16 @@ end of the run.
|
|
|
74
74
|
- The cycle absorbed undeclared scope, or surfaced follow-up work:
|
|
75
75
|
`--key unplanned_change` / `--key new_work_raised`.
|
|
76
76
|
|
|
77
|
+
To capture *why* something happened — a plan assumption that proved wrong, the reason an
|
|
78
|
+
integrity event fired — use `tdd note "<text>"` at the moment you know. A note written
|
|
79
|
+
during an open cycle is stamped with that cycle and phase and appears in the friction log
|
|
80
|
+
as a blockquote alongside the cycle's telemetry. After the run ends, `tdd note` attaches
|
|
81
|
+
at run level and renders in a dedicated **Executor narrative** section. Notes are unverified
|
|
82
|
+
by design; write them as claims, not measurements.
|
|
83
|
+
|
|
77
84
|
Narrative that spans cycles or happened after the run (CI failures, patterns) goes as
|
|
78
|
-
|
|
79
|
-
cycle annotation it doesn't belong to.
|
|
85
|
+
`tdd note` after the run ends, or as markdown appended below the rendered friction log
|
|
86
|
+
after `tdd log render` — never into a cycle annotation it doesn't belong to.
|
|
80
87
|
|
|
81
88
|
## When a suite run changes nothing
|
|
82
89
|
|
|
@@ -25,6 +25,7 @@ class Verdict:
|
|
|
25
25
|
target: str | None = None
|
|
26
26
|
target_outcome: str = NOT_FOUND
|
|
27
27
|
target_failure: str = ""
|
|
28
|
+
target_evidence: str = ""
|
|
28
29
|
passed: list[str] = field(default_factory=list)
|
|
29
30
|
failed: list[str] = field(default_factory=list)
|
|
30
31
|
duration_ms: int = 0
|
|
@@ -139,6 +140,14 @@ class Adapter:
|
|
|
139
140
|
prefix = f"{self.project.name}::"
|
|
140
141
|
return qualified[len(prefix) :] if qualified.startswith(prefix) else qualified
|
|
141
142
|
|
|
143
|
+
def normalise_id(self, test_id: str) -> str:
|
|
144
|
+
"""Return the canonical form of a declared target id for matching against collected ids.
|
|
145
|
+
|
|
146
|
+
The default is an identity — subclasses override when the runner's collected
|
|
147
|
+
ids differ from a natural human spelling (e.g. vitest's describe/test separator).
|
|
148
|
+
"""
|
|
149
|
+
return test_id
|
|
150
|
+
|
|
142
151
|
def run(self, target: str | None = None) -> Verdict:
|
|
143
152
|
raise NotImplementedError
|
|
144
153
|
|
|
@@ -206,6 +215,14 @@ class Adapter:
|
|
|
206
215
|
return None
|
|
207
216
|
return {k: os.path.expandvars(v) for k, v in merged.items()}
|
|
208
217
|
|
|
218
|
+
def lint_target_id(self, native: str) -> str | None:
|
|
219
|
+
"""Return a problem message when `native` can never match a collected id, else None."""
|
|
220
|
+
return None
|
|
221
|
+
|
|
222
|
+
def target_path(self, native: str) -> str | None:
|
|
223
|
+
"""Return the file-path portion of `native`, or None for non-path-bearing ids."""
|
|
224
|
+
return None
|
|
225
|
+
|
|
209
226
|
def stub_hint(self) -> str:
|
|
210
227
|
"""The language idiom for a stub body, quoted into the create_stub directive."""
|
|
211
228
|
return "a body that fails loudly, never working logic"
|
|
@@ -168,6 +168,9 @@ class ExecAdapter(Adapter):
|
|
|
168
168
|
if target == qualified:
|
|
169
169
|
verdict.target_outcome = FAILED
|
|
170
170
|
verdict.target_failure = clip_failure(combined)
|
|
171
|
+
verdict.target_evidence = next(
|
|
172
|
+
(ln for ln in reversed(combined.splitlines()) if ln.strip()), ""
|
|
173
|
+
)
|
|
171
174
|
|
|
172
175
|
verdict.duration_ms = int((time.monotonic() - started) * 1000)
|
|
173
176
|
return verdict
|