tdd-cli 0.6.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/CHANGELOG.md +78 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/PKG-INFO +159 -7
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/README.md +158 -6
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/adapters/base.py +8 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/adapters/vitest_adapter.py +62 -32
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/adapters/xctest_adapter.py +11 -11
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/advance.py +61 -5
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/cli.py +171 -40
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/config.py +42 -7
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/contract.py +15 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/ledger.py +141 -22
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/machine.py +53 -10
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/render.py +6 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/staging.py +6 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/conftest.py +91 -1
- tdd_cli-0.8.0/tests/test_ancillary_files.py +72 -0
- tdd_cli-0.8.0/tests/test_artifact_regeneration.py +166 -0
- tdd_cli-0.8.0/tests/test_baseline_integrity.py +586 -0
- tdd_cli-0.8.0/tests/test_concurrent_advance.py +178 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_config_and_staging.py +118 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_contract.py +120 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_heartbeat.py +69 -1
- tdd_cli-0.8.0/tests/test_id_normalisation.py +127 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_release_surface.py +15 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_snapshot_and_identity.py +45 -0
- tdd_cli-0.8.0/tests/test_undeclared_close_gate.py +106 -0
- tdd_cli-0.8.0/tests/test_undeclared_dedup.py +168 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_xctest_adapter.py +40 -49
- tdd_cli-0.6.0/tests/test_artifact_regeneration.py +0 -76
- tdd_cli-0.6.0/tests/test_baseline_integrity.py +0 -148
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/.gitignore +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/LICENSE +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/SECURITY.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/examples/plan.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/pyproject.toml +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/adapters/exec_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/adapters/gradle_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/adapters/pytest_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/identity.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_batch_collection.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_doctor_blockers.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_exec_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_failure_clipping.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_gradle_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_named_leases_and_timeout.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_project_commands.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_project_env.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_suite_overrides.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_timing_visibility.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_vitest_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.8.0}/tests/test_worker_leases.py +0 -0
|
@@ -4,6 +4,84 @@ All notable changes to this project are documented here.
|
|
|
4
4
|
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
5
5
|
and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.8.0] - 2026-08-28
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **`--reuse-baselines` caches baseline probes by content hash.** `run start`
|
|
14
|
+
can skip re-probing a project whose tree is unchanged: a probe result is
|
|
15
|
+
cached keyed by `(project, tree_hash, config_sha)` — where `tree_hash` folds in
|
|
16
|
+
every upstream producer root — and an identical rerun emits `baseline_reused`
|
|
17
|
+
instead of `baseline_captured`, reusing the cached failing set and collection
|
|
18
|
+
snapshot rather than re-running the suite. Provenance is recorded on the
|
|
19
|
+
baseline row (`source`), and `--reuse-max-age` re-probes any entry older than
|
|
20
|
+
the given age. Off by default; the cache stays empty unless the flag is passed
|
|
21
|
+
(#45, #59).
|
|
22
|
+
|
|
23
|
+
- **`--baseline-jobs` parallelizes baseline probing.** `run start` probes each
|
|
24
|
+
project's baseline under a bounded `ThreadPoolExecutor` when `--baseline-jobs`
|
|
25
|
+
is greater than 1 (default 1, must be >= 1). The `baseline_captured` heartbeat
|
|
26
|
+
survives the pool, and a worker probe that raises becomes an attributed failure
|
|
27
|
+
that aborts cleanly and frees the worktree rather than wedging it (#46, #62).
|
|
28
|
+
|
|
29
|
+
- **Plan-level `ancillary_files`.** A top-level front-matter key declaring
|
|
30
|
+
cross-project or companion paths a plan touches (README, generated fixtures,
|
|
31
|
+
sibling-project files). Declared ancillary paths are bucketed into their own
|
|
32
|
+
staging bucket, committed with the cycle, and fire no `undeclared_file_touched`
|
|
33
|
+
event. Validated as a list of strings at registration and persisted to the
|
|
34
|
+
ledger via a v5→v6 migration (#70, #80).
|
|
35
|
+
|
|
36
|
+
- **Run-close gate on undeclared touched paths.** `run close` now blocks when a
|
|
37
|
+
path previously flagged as `undeclared_file_touched` is still dirty in the
|
|
38
|
+
worktree, so undeclared changes cannot slip through at the end of a run. A
|
|
39
|
+
flagged path that was since committed does not block; one that has vanished is
|
|
40
|
+
reported via a new `undeclared_file_dropped` event rather than blocking (#69,
|
|
41
|
+
#81).
|
|
42
|
+
|
|
43
|
+
- **Reserved per-cycle `meta:` passthrough.** A cycle may carry an authored
|
|
44
|
+
`meta:` mapping in the plan front-matter; it round-trips through storage
|
|
45
|
+
unchanged and is available for plan-time metadata. A non-mapping `meta:`
|
|
46
|
+
hard-fails registration with a `ContractError` (#58, #65).
|
|
47
|
+
|
|
48
|
+
### Fixed
|
|
49
|
+
|
|
50
|
+
- `undeclared_file_touched` is deduplicated within a cycle, so a path touched
|
|
51
|
+
across multiple phases no longer floods the cycle with repeated events (#55,
|
|
52
|
+
#61).
|
|
53
|
+
- vitest test ids are normalised on the describe/test separator before matching,
|
|
54
|
+
so a formatting-only difference between a declared target and the observed
|
|
55
|
+
verdict is no longer reported as a spurious `declared_test_mismatch` (#57,
|
|
56
|
+
#63).
|
|
57
|
+
- The `stale_artifact` event is suppressed when the tool auto-regenerates the
|
|
58
|
+
artifact and commits it, so a successful regeneration no longer also emits a
|
|
59
|
+
staleness warning (#64).
|
|
60
|
+
|
|
61
|
+
## [0.7.0] - 2026-08-23
|
|
62
|
+
|
|
63
|
+
### Fixed
|
|
64
|
+
|
|
65
|
+
- **Concurrent `tdd advance` no longer corrupts a run.** Two `advance` processes
|
|
66
|
+
racing on the same worktree could both close the same cycle row. Because
|
|
67
|
+
`close_cycle` unconditionally opened the next ordinal, a double-close forked the
|
|
68
|
+
run into two parallel cycle chains; every remaining cycle ran twice, the run
|
|
69
|
+
"completed" with one chain's last row permanently open, and the doubling was
|
|
70
|
+
invisible from the agent's perspective. `close_cycle` now re-reads `closed_at`
|
|
71
|
+
before acting; if the row is already closed it returns the currently-open cycle
|
|
72
|
+
without transitioning or opening anything. `open_cycle` returns the existing open
|
|
73
|
+
row for an ordinal rather than inserting a duplicate.
|
|
74
|
+
|
|
75
|
+
### Added
|
|
76
|
+
|
|
77
|
+
- **Per-worktree advance claim.** `tdd advance` now acquires a `advance_claim` row
|
|
78
|
+
before dispatching. A second concurrent `advance` is refused immediately with
|
|
79
|
+
`ok: false`, `reason: "advance_in_flight"`, and metadata (`pid`, `started_at`,
|
|
80
|
+
`elapsed_s`) that lets the agent confirm the holder is still alive. The claim is
|
|
81
|
+
released in a `finally` so a raising handler cannot wedge the worktree; a claim
|
|
82
|
+
held by a dead pid is reclaimed automatically on the next call. Schema version
|
|
83
|
+
bumped to 3.
|
|
84
|
+
|
|
7
85
|
## [0.6.0] - 2026-08-19
|
|
8
86
|
|
|
9
87
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -227,6 +227,8 @@ cycles:
|
|
|
227
227
|
stub_expected: ["app/exception_map.py"]
|
|
228
228
|
commit_red: "test: unmapped exception is not swallowed"
|
|
229
229
|
commit_green: "feat: domain exception map skeleton"
|
|
230
|
+
meta: # optional authored-at-plan-time metadata; opaque to the tool
|
|
231
|
+
covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
|
|
230
232
|
- n: 8
|
|
231
233
|
project: backend
|
|
232
234
|
pin_cycle: true # characterisation; passes on arrival by design
|
|
@@ -238,9 +240,51 @@ cycles:
|
|
|
238
240
|
- "backend::tests/test_openapi.py::test_upload_body_schema"
|
|
239
241
|
- "frontend::services/__tests__/upload.test.ts > matches contract"
|
|
240
242
|
annotation_keys: ["literal_detail_handlers_kept"]
|
|
243
|
+
ancillary_files:
|
|
244
|
+
- frontend/src/api/registerClient.ts # type-break from regenerated client
|
|
245
|
+
- docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
|
|
241
246
|
---
|
|
242
247
|
```
|
|
243
248
|
|
|
249
|
+
**Top-level keys:**
|
|
250
|
+
|
|
251
|
+
`annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
|
|
252
|
+
gate checks that every key is present before the plan can be marked complete.
|
|
253
|
+
|
|
254
|
+
`ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
|
|
255
|
+
touch outside any registered project root (cross-project ripples, companion documents).
|
|
256
|
+
Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
|
|
257
|
+
matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
|
|
258
|
+
and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
|
|
259
|
+
not on the list still fires `undeclared_file_touched` exactly as today. This is a
|
|
260
|
+
plan-level list only; per-cycle overrides are a planned follow-up.
|
|
261
|
+
|
|
262
|
+
### Run-close gate for undeclared file touches
|
|
263
|
+
|
|
264
|
+
When the last declared cycle closes, the tool gathers every path that appeared in any
|
|
265
|
+
`undeclared_file_touched` event across the run and checks the worktree:
|
|
266
|
+
|
|
267
|
+
- **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
|
|
268
|
+
the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
|
|
269
|
+
commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
|
|
270
|
+
and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
|
|
271
|
+
- **Committed during the run** → clean; the run completes normally.
|
|
272
|
+
- **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
|
|
273
|
+
emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
|
|
274
|
+
makes a silent drop visible without blocking, since the file is already gone.
|
|
275
|
+
|
|
276
|
+
The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
|
|
277
|
+
so they are never seen by the gate.
|
|
278
|
+
|
|
279
|
+
**Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
|
|
280
|
+
`stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
|
|
281
|
+
`commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
|
|
282
|
+
`meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
|
|
283
|
+
but its *contents* are opaque to the tool — any key/value pairs are accepted and
|
|
284
|
+
round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
|
|
285
|
+
authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
|
|
286
|
+
needs to read back. Other unknown per-cycle keys are silently ignored.
|
|
287
|
+
|
|
244
288
|
Absent front-matter is legitimate — the run proceeds as `undeclared` with
|
|
245
289
|
`--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
|
|
246
290
|
hard-fails registration: it is almost always a defect in the planning process, and that
|
|
@@ -314,6 +358,34 @@ test_paths = ["src/test/"]
|
|
|
314
358
|
test_command = "./gradlew testDebugUnitTest"
|
|
315
359
|
```
|
|
316
360
|
|
|
361
|
+
The `xctest` adapter drives Swift and Objective-C projects through `xcodebuild`. Test ids
|
|
362
|
+
use xcodebuild's own `-only-testing:` format — `BundleName/ClassName/testMethodName` — so
|
|
363
|
+
a single XCTest can be driven through RED → GREEN without any id translation. Collection
|
|
364
|
+
prefers `xcodebuild test -enumerate-tests` (Xcode 16+) and falls back to grepping
|
|
365
|
+
`class Foo: XCTestCase` / `func testBar()` out of the Swift sources under `test_paths`,
|
|
366
|
+
deriving the bundle name from `-scheme`. As with gradle, a *build* failure maps to
|
|
367
|
+
`not_collected` rather than `failed`, so a stub referencing a missing symbol is not
|
|
368
|
+
mistaken for RED.
|
|
369
|
+
|
|
370
|
+
`test_command` is required here — the adapter cannot guess the scheme, destination, or
|
|
371
|
+
derived-data path. It appends `-only-testing:` for targeted runs and `-enumerate-tests`
|
|
372
|
+
for collection, and changes nothing else. Simulator runs are serial in practice: give
|
|
373
|
+
them a `lease` so two projects don't drive the same simulator at once.
|
|
374
|
+
|
|
375
|
+
```toml
|
|
376
|
+
[project.native-ios]
|
|
377
|
+
root = "native-ios"
|
|
378
|
+
adapter = "xctest"
|
|
379
|
+
test_paths = ["AppTests/"]
|
|
380
|
+
test_command = """xcodebuild test \
|
|
381
|
+
-project App.xcodeproj \
|
|
382
|
+
-scheme AppTests \
|
|
383
|
+
-destination 'platform=iOS Simulator,name=App-Unit' \
|
|
384
|
+
-derivedDataPath /tmp/app-unit-dd"""
|
|
385
|
+
lease = "ios-simulator"
|
|
386
|
+
timeout = 900
|
|
387
|
+
```
|
|
388
|
+
|
|
317
389
|
Third-party adapters register under the
|
|
318
390
|
`tddcli.adapters` entry-point group:
|
|
319
391
|
|
|
@@ -375,14 +447,93 @@ and is never reclassified as a pin.
|
|
|
375
447
|
| `tdd metrics` | fidelity, attempts, violations, interventions |
|
|
376
448
|
| `tdd fleet [--json]` | all active runs across every worktree; read-only |
|
|
377
449
|
|
|
450
|
+
## Scoped baseline capture (R9.5c)
|
|
451
|
+
|
|
452
|
+
`run start` probes only the projects the plan can actually reach: the declared cycle projects
|
|
453
|
+
plus the transitive `consumed_by` closure of artifacts whose producer is in that set. Projects
|
|
454
|
+
outside the reachable set never run during the plan, so their baseline is never subtracted from
|
|
455
|
+
anything — probing them is pure overhead. A `baseline_scoped` integrity event records which
|
|
456
|
+
projects were skipped, so the scoping is auditable.
|
|
457
|
+
|
|
458
|
+
Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
|
|
459
|
+
|
|
460
|
+
tdd run start --plan tasks/plan.md --baseline-all
|
|
461
|
+
|
|
462
|
+
Use this when a cycle may edit files outside the predicted reachable set and you want every
|
|
463
|
+
project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
|
|
464
|
+
|
|
465
|
+
### Reusing baselines across runs (R9.5e)
|
|
466
|
+
|
|
467
|
+
On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
|
|
468
|
+
wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
|
|
469
|
+
when nothing in the project changed:
|
|
470
|
+
|
|
471
|
+
tdd run start --plan tasks/plan.md --reuse-baselines
|
|
472
|
+
|
|
473
|
+
The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
|
|
474
|
+
If the key matches a previous entry, the cached failing set and collection snapshot are used —
|
|
475
|
+
no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
|
|
476
|
+
`source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
|
|
477
|
+
projects were skipped. Without the flag, the cache is neither read nor written.
|
|
478
|
+
|
|
479
|
+
A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
|
|
480
|
+
(see below) — reuse is loud by design, never silent.
|
|
481
|
+
|
|
482
|
+
To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
|
|
483
|
+
|
|
484
|
+
tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
|
|
485
|
+
|
|
486
|
+
Entries older than the given threshold are ignored and re-probed fresh.
|
|
487
|
+
|
|
488
|
+
### Parallel baseline probing (R9.5f)
|
|
489
|
+
|
|
490
|
+
By default `run start` probes one project at a time. On a repo with many independent projects the
|
|
491
|
+
wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
|
|
492
|
+
projects concurrently:
|
|
493
|
+
|
|
494
|
+
tdd run start --plan tasks/plan.md --baseline-jobs 4
|
|
495
|
+
|
|
496
|
+
Each probe is independent — one adapter instance per project — so concurrency does not affect
|
|
497
|
+
which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
|
|
498
|
+
heartbeat is still emitted per project as each probe completes.
|
|
499
|
+
|
|
500
|
+
The default is `--baseline-jobs 1` (serial). Raise it deliberately:
|
|
501
|
+
|
|
502
|
+
- **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
|
|
503
|
+
- **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
|
|
504
|
+
or declare a `lease` name — leased suites serialize automatically even inside the pool.
|
|
505
|
+
|
|
506
|
+
If any probe fails, `run start` returns a failure attributed to that project and no run row is
|
|
507
|
+
created, so the worktree is immediately retryable.
|
|
508
|
+
|
|
509
|
+
### When a sweep reaches an un-baselined project (R9.5d)
|
|
510
|
+
|
|
511
|
+
If an edit during a run touches a file owned by an artifact that was outside the predicted
|
|
512
|
+
reachable set, the close sweep may pull in a project that was never baselined. Its failures are
|
|
513
|
+
unattributable — no baseline exists to subtract — so the sweep replies `resolve_blocker` with
|
|
514
|
+
kind `no_baseline_for_project` rather than mislabelling them as regressions. Recovery:
|
|
515
|
+
|
|
516
|
+
tdd blocker --kind no_baseline_for_project --detail "svc pulled in unexpectedly"
|
|
517
|
+
tdd resume --unblock --accept-failures --note "folding svc sweep failures into baseline"
|
|
518
|
+
|
|
519
|
+
`--accept-failures` inserts a fresh baseline row for the un-baselined project (recording those
|
|
520
|
+
failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
|
|
521
|
+
with the project properly baselined.
|
|
522
|
+
|
|
378
523
|
## Running a long baseline
|
|
379
524
|
|
|
380
|
-
`run start` probes
|
|
525
|
+
`run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
|
|
381
526
|
that can take minutes — well past an agent harness's default Bash timeout. If the command
|
|
382
527
|
appears to hang or time out, **do not re-run it**: the probe is still making progress in the
|
|
383
528
|
background, and a second `run start` against the same worktree is refused with
|
|
384
529
|
`reason: "baseline_in_progress"` — retrying on timeout just stacks refusals on top of a
|
|
385
|
-
baseline that was never stuck.
|
|
530
|
+
baseline that was never stuck.
|
|
531
|
+
|
|
532
|
+
`tdd advance` is similarly single-flight per worktree. A close sweep (artifact regeneration, full
|
|
533
|
+
suite, lint, typecheck) can run for several minutes. If an advance appears to hang, **do not
|
|
534
|
+
re-run it**: a second `advance` against the same worktree is refused with
|
|
535
|
+
`reason: "advance_in_flight"`. The refusal carries `pid`, `started_at`, and `elapsed_s` so the
|
|
536
|
+
agent can confirm the first process is still alive. Run `tdd status` to see the current run state. In order of preference:
|
|
386
537
|
|
|
387
538
|
1. **Background it.** Run `tdd run start` in the background if your harness supports it. The
|
|
388
539
|
heartbeat (`baseline_captured` / `project_completed` lines on stderr) lands in the task log
|
|
@@ -409,10 +560,11 @@ The CLI cannot compel an agent — only the harness can.
|
|
|
409
560
|
artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
|
|
410
561
|
`advance` refuses an unchanged tree unless `--retry`.
|
|
411
562
|
|
|
412
|
-
**Recorded, never blocked:** non-stub writes during RED, undeclared file touches,
|
|
413
|
-
divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
414
|
-
blocked agent improvises around them — putting it right back in the reporting path
|
|
415
|
-
tool exists to keep it out of.
|
|
563
|
+
**Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
|
|
564
|
+
scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
565
|
+
and a blocked agent improvises around them — putting it right back in the reporting path
|
|
566
|
+
the tool exists to keep it out of. Undeclared file touches are an exception at run close:
|
|
567
|
+
see the run-close gate above.
|
|
416
568
|
|
|
417
569
|
**Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
|
|
418
570
|
stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
|
|
@@ -199,6 +199,8 @@ cycles:
|
|
|
199
199
|
stub_expected: ["app/exception_map.py"]
|
|
200
200
|
commit_red: "test: unmapped exception is not swallowed"
|
|
201
201
|
commit_green: "feat: domain exception map skeleton"
|
|
202
|
+
meta: # optional authored-at-plan-time metadata; opaque to the tool
|
|
203
|
+
covers: ["B1", "B2"] # any mapping is valid; contents are tool-opaque
|
|
202
204
|
- n: 8
|
|
203
205
|
project: backend
|
|
204
206
|
pin_cycle: true # characterisation; passes on arrival by design
|
|
@@ -210,9 +212,51 @@ cycles:
|
|
|
210
212
|
- "backend::tests/test_openapi.py::test_upload_body_schema"
|
|
211
213
|
- "frontend::services/__tests__/upload.test.ts > matches contract"
|
|
212
214
|
annotation_keys: ["literal_detail_handlers_kept"]
|
|
215
|
+
ancillary_files:
|
|
216
|
+
- frontend/src/api/registerClient.ts # type-break from regenerated client
|
|
217
|
+
- docs/INVARIANTS.md # companion doc read at runtime by cycle 7's test
|
|
213
218
|
---
|
|
214
219
|
```
|
|
215
220
|
|
|
221
|
+
**Top-level keys:**
|
|
222
|
+
|
|
223
|
+
`annotation_keys` — a list of judgement-annotation keys the plan requires; the run close
|
|
224
|
+
gate checks that every key is present before the plan can be marked complete.
|
|
225
|
+
|
|
226
|
+
`ancillary_files` — a plan-level list of repo-root-relative paths the plan is known to
|
|
227
|
+
touch outside any registered project root (cross-project ripples, companion documents).
|
|
228
|
+
Paths are hash-frozen with the plan blob like all other front-matter. A changed path that
|
|
229
|
+
matches the list is classified as *declared* — no `undeclared_file_touched` event fires —
|
|
230
|
+
and is staged into the GREEN/REFACTOR phase commit alongside the cycle's own files. A path
|
|
231
|
+
not on the list still fires `undeclared_file_touched` exactly as today. This is a
|
|
232
|
+
plan-level list only; per-cycle overrides are a planned follow-up.
|
|
233
|
+
|
|
234
|
+
### Run-close gate for undeclared file touches
|
|
235
|
+
|
|
236
|
+
When the last declared cycle closes, the tool gathers every path that appeared in any
|
|
237
|
+
`undeclared_file_touched` event across the run and checks the worktree:
|
|
238
|
+
|
|
239
|
+
- **Still dirty/untracked** → a typed blocker `undeclared_file_uncommitted` is inserted,
|
|
240
|
+
the run outcome is set to `blocked`, and `next_action.verb` is `"blocked"`. Resolution:
|
|
241
|
+
commit the files, then `tdd resume --unblock --note "committed notes.md"`, or discard
|
|
242
|
+
and record a justification with `tdd resume --unblock --note "discarding scratch file"`.
|
|
243
|
+
- **Committed during the run** → clean; the run completes normally.
|
|
244
|
+
- **Vanished without being committed** → an `undeclared_file_dropped` integrity event is
|
|
245
|
+
emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
|
|
246
|
+
makes a silent drop visible without blocking, since the file is already gone.
|
|
247
|
+
|
|
248
|
+
The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
|
|
249
|
+
so they are never seen by the gate.
|
|
250
|
+
|
|
251
|
+
**Per-cycle keys:** `n` (ordinal), `project`/`projects`, `test`/`tests`, `title`, `files`,
|
|
252
|
+
`stub_expected`, `modifies_tests`, `commit_red`, `commit_green`, `commit_refactor`,
|
|
253
|
+
`commit_pin`, `pin_cycle`, `contract_cycle`, `refactor_cycle`, and `meta`.
|
|
254
|
+
`meta` is a reserved passthrough mapping: its *shape* is validated (must be a mapping),
|
|
255
|
+
but its *contents* are opaque to the tool — any key/value pairs are accepted and
|
|
256
|
+
round-tripped intact through `cycles_to_json`/`cycles_from_json`. Use it for
|
|
257
|
+
authored-at-plan-time metadata that external tooling (e.g. a behaviour-coverage checker)
|
|
258
|
+
needs to read back. Other unknown per-cycle keys are silently ignored.
|
|
259
|
+
|
|
216
260
|
Absent front-matter is legitimate — the run proceeds as `undeclared` with
|
|
217
261
|
`--allow-undeclared`, and fidelity metrics are unavailable. **Malformed** front-matter
|
|
218
262
|
hard-fails registration: it is almost always a defect in the planning process, and that
|
|
@@ -286,6 +330,34 @@ test_paths = ["src/test/"]
|
|
|
286
330
|
test_command = "./gradlew testDebugUnitTest"
|
|
287
331
|
```
|
|
288
332
|
|
|
333
|
+
The `xctest` adapter drives Swift and Objective-C projects through `xcodebuild`. Test ids
|
|
334
|
+
use xcodebuild's own `-only-testing:` format — `BundleName/ClassName/testMethodName` — so
|
|
335
|
+
a single XCTest can be driven through RED → GREEN without any id translation. Collection
|
|
336
|
+
prefers `xcodebuild test -enumerate-tests` (Xcode 16+) and falls back to grepping
|
|
337
|
+
`class Foo: XCTestCase` / `func testBar()` out of the Swift sources under `test_paths`,
|
|
338
|
+
deriving the bundle name from `-scheme`. As with gradle, a *build* failure maps to
|
|
339
|
+
`not_collected` rather than `failed`, so a stub referencing a missing symbol is not
|
|
340
|
+
mistaken for RED.
|
|
341
|
+
|
|
342
|
+
`test_command` is required here — the adapter cannot guess the scheme, destination, or
|
|
343
|
+
derived-data path. It appends `-only-testing:` for targeted runs and `-enumerate-tests`
|
|
344
|
+
for collection, and changes nothing else. Simulator runs are serial in practice: give
|
|
345
|
+
them a `lease` so two projects don't drive the same simulator at once.
|
|
346
|
+
|
|
347
|
+
```toml
|
|
348
|
+
[project.native-ios]
|
|
349
|
+
root = "native-ios"
|
|
350
|
+
adapter = "xctest"
|
|
351
|
+
test_paths = ["AppTests/"]
|
|
352
|
+
test_command = """xcodebuild test \
|
|
353
|
+
-project App.xcodeproj \
|
|
354
|
+
-scheme AppTests \
|
|
355
|
+
-destination 'platform=iOS Simulator,name=App-Unit' \
|
|
356
|
+
-derivedDataPath /tmp/app-unit-dd"""
|
|
357
|
+
lease = "ios-simulator"
|
|
358
|
+
timeout = 900
|
|
359
|
+
```
|
|
360
|
+
|
|
289
361
|
Third-party adapters register under the
|
|
290
362
|
`tddcli.adapters` entry-point group:
|
|
291
363
|
|
|
@@ -347,14 +419,93 @@ and is never reclassified as a pin.
|
|
|
347
419
|
| `tdd metrics` | fidelity, attempts, violations, interventions |
|
|
348
420
|
| `tdd fleet [--json]` | all active runs across every worktree; read-only |
|
|
349
421
|
|
|
422
|
+
## Scoped baseline capture (R9.5c)
|
|
423
|
+
|
|
424
|
+
`run start` probes only the projects the plan can actually reach: the declared cycle projects
|
|
425
|
+
plus the transitive `consumed_by` closure of artifacts whose producer is in that set. Projects
|
|
426
|
+
outside the reachable set never run during the plan, so their baseline is never subtracted from
|
|
427
|
+
anything — probing them is pure overhead. A `baseline_scoped` integrity event records which
|
|
428
|
+
projects were skipped, so the scoping is auditable.
|
|
429
|
+
|
|
430
|
+
Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
|
|
431
|
+
|
|
432
|
+
tdd run start --plan tasks/plan.md --baseline-all
|
|
433
|
+
|
|
434
|
+
Use this when a cycle may edit files outside the predicted reachable set and you want every
|
|
435
|
+
project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
|
|
436
|
+
|
|
437
|
+
### Reusing baselines across runs (R9.5e)
|
|
438
|
+
|
|
439
|
+
On repos where runs are frequent and most projects change rarely, re-probing unchanged suites
|
|
440
|
+
wastes time. Pass `--reuse-baselines` to cache probe results and skip re-probing on the next run
|
|
441
|
+
when nothing in the project changed:
|
|
442
|
+
|
|
443
|
+
tdd run start --plan tasks/plan.md --reuse-baselines
|
|
444
|
+
|
|
445
|
+
The cache key is `(project, tree_hash(project root ∪ upstream producer roots), config_sha)`.
|
|
446
|
+
If the key matches a previous entry, the cached failing set and collection snapshot are used —
|
|
447
|
+
no suite is run — and a `baseline_reused` heartbeat appears on stderr. The baseline row carries
|
|
448
|
+
`source = "reused"` for auditability, and a `baseline_reused` integrity event lists which
|
|
449
|
+
projects were skipped. Without the flag, the cache is neither read nor written.
|
|
450
|
+
|
|
451
|
+
A stale or wrong reused baseline is always recoverable via `resume --unblock --accept-failures`
|
|
452
|
+
(see below) — reuse is loud by design, never silent.
|
|
453
|
+
|
|
454
|
+
To limit how old a cached entry may be, pass `--reuse-max-age <seconds>`:
|
|
455
|
+
|
|
456
|
+
tdd run start --plan tasks/plan.md --reuse-baselines --reuse-max-age 3600
|
|
457
|
+
|
|
458
|
+
Entries older than the given threshold are ignored and re-probed fresh.
|
|
459
|
+
|
|
460
|
+
### Parallel baseline probing (R9.5f)
|
|
461
|
+
|
|
462
|
+
By default `run start` probes one project at a time. On a repo with many independent projects the
|
|
463
|
+
wall-clock cost is the *sum* of all suite runtimes. Pass `--baseline-jobs N` to probe up to N
|
|
464
|
+
projects concurrently:
|
|
465
|
+
|
|
466
|
+
tdd run start --plan tasks/plan.md --baseline-jobs 4
|
|
467
|
+
|
|
468
|
+
Each probe is independent — one adapter instance per project — so concurrency does not affect
|
|
469
|
+
which projects are probed, the refusal checks, or the baseline rows written. A `baseline_captured`
|
|
470
|
+
heartbeat is still emitted per project as each probe completes.
|
|
471
|
+
|
|
472
|
+
The default is `--baseline-jobs 1` (serial). Raise it deliberately:
|
|
473
|
+
|
|
474
|
+
- **I/O-bound suites** (network, DB, file-heavy) benefit most; CPU-bound suites less so.
|
|
475
|
+
- **Suites contending for global resources** (xctest simulators, fixed ports) should stay at 1,
|
|
476
|
+
or declare a `lease` name — leased suites serialize automatically even inside the pool.
|
|
477
|
+
|
|
478
|
+
If any probe fails, `run start` returns a failure attributed to that project and no run row is
|
|
479
|
+
created, so the worktree is immediately retryable.
|
|
480
|
+
|
|
481
|
+
### When a sweep reaches an un-baselined project (R9.5d)
|
|
482
|
+
|
|
483
|
+
If an edit during a run touches a file owned by an artifact that was outside the predicted
|
|
484
|
+
reachable set, the close sweep may pull in a project that was never baselined. Its failures are
|
|
485
|
+
unattributable — no baseline exists to subtract — so the sweep replies `resolve_blocker` with
|
|
486
|
+
kind `no_baseline_for_project` rather than mislabelling them as regressions. Recovery:
|
|
487
|
+
|
|
488
|
+
tdd blocker --kind no_baseline_for_project --detail "svc pulled in unexpectedly"
|
|
489
|
+
tdd resume --unblock --accept-failures --note "folding svc sweep failures into baseline"
|
|
490
|
+
|
|
491
|
+
`--accept-failures` inserts a fresh baseline row for the un-baselined project (recording those
|
|
492
|
+
failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
|
|
493
|
+
with the project properly baselined.
|
|
494
|
+
|
|
350
495
|
## Running a long baseline
|
|
351
496
|
|
|
352
|
-
`run start` probes
|
|
497
|
+
`run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
|
|
353
498
|
that can take minutes — well past an agent harness's default Bash timeout. If the command
|
|
354
499
|
appears to hang or time out, **do not re-run it**: the probe is still making progress in the
|
|
355
500
|
background, and a second `run start` against the same worktree is refused with
|
|
356
501
|
`reason: "baseline_in_progress"` — retrying on timeout just stacks refusals on top of a
|
|
357
|
-
baseline that was never stuck.
|
|
502
|
+
baseline that was never stuck.
|
|
503
|
+
|
|
504
|
+
`tdd advance` is similarly single-flight per worktree. A close sweep (artifact regeneration, full
|
|
505
|
+
suite, lint, typecheck) can run for several minutes. If an advance appears to hang, **do not
|
|
506
|
+
re-run it**: a second `advance` against the same worktree is refused with
|
|
507
|
+
`reason: "advance_in_flight"`. The refusal carries `pid`, `started_at`, and `elapsed_s` so the
|
|
508
|
+
agent can confirm the first process is still alive. Run `tdd status` to see the current run state. In order of preference:
|
|
358
509
|
|
|
359
510
|
1. **Background it.** Run `tdd run start` in the background if your harness supports it. The
|
|
360
511
|
heartbeat (`baseline_captured` / `project_completed` lines on stderr) lands in the task log
|
|
@@ -381,10 +532,11 @@ The CLI cannot compel an agent — only the harness can.
|
|
|
381
532
|
artifact; a passed-on-arrival cycle cannot close without a verified sensitivity check;
|
|
382
533
|
`advance` refuses an unchanged tree unless `--retry`.
|
|
383
534
|
|
|
384
|
-
**Recorded, never blocked:** non-stub writes during RED, undeclared file touches,
|
|
385
|
-
divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
386
|
-
blocked agent improvises around them — putting it right back in the reporting path
|
|
387
|
-
tool exists to keep it out of.
|
|
535
|
+
**Recorded, never blocked mid-run:** non-stub writes during RED, undeclared file touches,
|
|
536
|
+
scope divergence, extra attempts. Prevention rules with edge cases produce false denials,
|
|
537
|
+
and a blocked agent improvises around them — putting it right back in the reporting path
|
|
538
|
+
the tool exists to keep it out of. Undeclared file touches are an exception at run close:
|
|
539
|
+
see the run-close gate above.
|
|
388
540
|
|
|
389
541
|
**Delegated to hooks:** a Stop hook that queries `tdd status` and refuses to let an agent
|
|
390
542
|
stop while a run is live; a Bash hook redirecting bare `pytest`/`vitest` through `tdd advance`.
|
|
@@ -139,6 +139,14 @@ class Adapter:
|
|
|
139
139
|
prefix = f"{self.project.name}::"
|
|
140
140
|
return qualified[len(prefix) :] if qualified.startswith(prefix) else qualified
|
|
141
141
|
|
|
142
|
+
def normalise_id(self, test_id: str) -> str:
|
|
143
|
+
"""Return the canonical form of a declared target id for matching against collected ids.
|
|
144
|
+
|
|
145
|
+
The default is an identity — subclasses override when the runner's collected
|
|
146
|
+
ids differ from a natural human spelling (e.g. vitest's describe/test separator).
|
|
147
|
+
"""
|
|
148
|
+
return test_id
|
|
149
|
+
|
|
142
150
|
def run(self, target: str | None = None) -> Verdict:
|
|
143
151
|
raise NotImplementedError
|
|
144
152
|
|