tdd-cli 0.8.0__tar.gz → 0.10.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/CHANGELOG.md +87 -0
  2. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/PKG-INFO +94 -2
  3. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/README.md +93 -1
  4. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/skills/tdd-drive/SKILL.md +9 -2
  5. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/pyproject.toml +2 -1
  6. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/__init__.py +1 -1
  7. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/__init__.py +2 -0
  8. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/base.py +9 -0
  9. tdd_cli-0.10.0/src/tddcli/adapters/cargo_adapter.py +321 -0
  10. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/exec_adapter.py +3 -0
  11. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/gradle_adapter.py +13 -1
  12. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/pytest_adapter.py +20 -1
  13. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/vitest_adapter.py +17 -1
  14. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/xctest_adapter.py +19 -1
  15. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/advance.py +101 -24
  16. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/cli.py +148 -6
  17. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/config.py +16 -0
  18. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/identity.py +16 -4
  19. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/ledger.py +28 -1
  20. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/render.py +27 -2
  21. tdd_cli-0.10.0/src/tddcli/target_lint.py +61 -0
  22. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/conftest.py +9 -1
  23. tdd_cli-0.10.0/tests/test_advance_adoption.py +126 -0
  24. tdd_cli-0.10.0/tests/test_baseline_sanity.py +229 -0
  25. tdd_cli-0.10.0/tests/test_cargo_adapter.py +395 -0
  26. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_config_and_staging.py +22 -0
  27. tdd_cli-0.10.0/tests/test_evidence_extraction.py +226 -0
  28. tdd_cli-0.10.0/tests/test_executor_attribution.py +133 -0
  29. tdd_cli-0.10.0/tests/test_executor_notes.py +209 -0
  30. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_heartbeat.py +7 -6
  31. tdd_cli-0.10.0/tests/test_sensitivity_evidence.py +157 -0
  32. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_snapshot_and_identity.py +47 -0
  33. tdd_cli-0.10.0/tests/test_target_lint.py +202 -0
  34. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/.gitignore +0 -0
  35. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/LICENSE +0 -0
  36. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/SECURITY.md +0 -0
  37. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/claude-code-hooks/README.md +0 -0
  38. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/claude-code-hooks/bash_hook.py +0 -0
  39. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/claude-code-hooks/stop_hook.py +0 -0
  40. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/plan.md +0 -0
  41. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/skills/tdd-drive/README.md +0 -0
  42. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/skills/tdd-handoff/README.md +0 -0
  43. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
  44. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/contract.py +0 -0
  45. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/envelope.py +0 -0
  46. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/fleet.py +0 -0
  47. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/gitutil.py +0 -0
  48. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/leases.py +0 -0
  49. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/machine.py +0 -0
  50. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/snapshot.py +0 -0
  51. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/staging.py +0 -0
  52. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_ancillary_files.py +0 -0
  53. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_artifact_regeneration.py +0 -0
  54. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_baseline_integrity.py +0 -0
  55. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_batch_collection.py +0 -0
  56. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_concurrent_advance.py +0 -0
  57. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_config_drift.py +0 -0
  58. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_contract.py +0 -0
  59. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_doctor_attribution.py +0 -0
  60. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_doctor_blockers.py +0 -0
  61. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_end_to_end.py +0 -0
  62. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_example_plan.py +0 -0
  63. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_exec_adapter.py +0 -0
  64. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_failure_clipping.py +0 -0
  65. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_fleet.py +0 -0
  66. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_gradle_adapter.py +0 -0
  67. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_id_normalisation.py +0 -0
  68. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_init_detection.py +0 -0
  69. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_named_leases_and_timeout.py +0 -0
  70. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_pin_cycles.py +0 -0
  71. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_progress.py +0 -0
  72. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_project_commands.py +0 -0
  73. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_project_env.py +0 -0
  74. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_python_env_managers.py +0 -0
  75. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_refactor_cycles.py +0 -0
  76. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_release_surface.py +0 -0
  77. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_run_claim.py +0 -0
  78. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_single_project_repo.py +0 -0
  79. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_stub_hint.py +0 -0
  80. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_suite_overrides.py +0 -0
  81. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_target_validation.py +0 -0
  82. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_timing_visibility.py +0 -0
  83. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_undeclared_close_gate.py +0 -0
  84. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_undeclared_dedup.py +0 -0
  85. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_vitest_adapter.py +0 -0
  86. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_worker_leases.py +0 -0
  87. {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_xctest_adapter.py +0 -0
@@ -6,6 +6,93 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.10.0] - 2026-09-05
10
+
11
+ ### Added
12
+
13
+ - **cargo adapter** — Rust driver through `cargo test`. Ids are `<target>::<path>`
14
+ (`lib::…` for unit tests, `<tests-file-stem>::…` for integration tests) and compose
15
+ to `--lib`/`--test <name>` + `-- --exact <path>` for targeted runs. Verdicts are
16
+ parsed from cargo's per-test console lines, attributed to targets by the
17
+ `Running …` headers with stderr merged; doc-tests are excluded. A compile error maps
18
+ to `not_collected`, matching gradle/xctest. Collection is `cargo test --tests --
19
+ --list` with a per-file `--test <stem>` fallback; `tdd doctor`'s collectable gate is
20
+ `cargo test --no-run`. Target lint requires the `::` separator.
21
+
22
+ ## [0.9.0] - 2026-09-03
23
+
24
+ ### Added
25
+
26
+ - **`tdd note` — an executor-narrative channel.** `tdd note "<text>"` attaches
27
+ a phase-stamped note to the current cycle, or a run-level note once the run
28
+ has ended (it falls back to the latest run, so no active run is required).
29
+ Cycle notes render in the friction log as blockquote claims under their cycle;
30
+ run-level notes render under a new `## Executor narrative` section, which is
31
+ omitted when there are none. Envelopes that carry an integrity event now nudge
32
+ for a note while the reason is fresh (silenced once the cycle has one), and
33
+ the terminal `COMPLETE` envelope — via `advance` or a final-cycle skip — asks
34
+ for a closing narrative before rendering. Stored in a new `note` table via a
35
+ v7→v8 ledger migration (#77, #94).
36
+
37
+ - **Baseline sanity gate.** `run start` refuses a baseline whose failing ratio
38
+ exceeds a threshold (default 0.5, suites under 10 collected tests exempt) with
39
+ `reason: "baseline_implausible"`, on the grounds that a mostly-red suite means
40
+ the environment is broken rather than the code. A per-project
41
+ `baseline_max_failure_ratio` in `tdd.toml` overrides the default, and
42
+ `--accept-baseline` bypasses the gate and logs a `baseline_accepted` integrity
43
+ event (#67, #84).
44
+
45
+ - **Standing-failure delta.** A non-empty baseline is now compared against the
46
+ previous run's baseline for the same worktree and a `baseline_standing_delta`
47
+ event partitions the failing set into new versus inherited failures, so a
48
+ pre-existing red test is distinguishable from one that broke since the last
49
+ run (#67, #84).
50
+
51
+ - **Per-project `health_command`.** An optional command in `tdd.toml` that is
52
+ probed before baseline capture; if it fails, `run start` refuses with
53
+ `reason: "services_unreachable"` instead of recording a baseline against a
54
+ down dependency (#67, #84).
55
+
56
+ - **Declared-target lint.** `plan register` and `run start` now lint every
57
+ declared target against the adapter's id grammar (pytest `::`, vitest ` > `,
58
+ gradle `Class/method`, xctest `Bundle/Class/testMethod`) and against project
59
+ roots: a target path that duplicates a non-`.` project root prefix is refused
60
+ unless the nested path genuinely exists in the worktree. Findings are returned
61
+ under `reason: "target_lint"`; `run start` re-lints the stored contract
62
+ against the current config before claiming a baseline (#71, #88).
63
+
64
+ - **Per-adapter sensitivity evidence line.** Each adapter now extracts a
65
+ `target_evidence` line from a failing target — the first `E` line for pytest
66
+ (skipping the xdist worker header), the first `: error:` line in the test's
67
+ window for xctest, `failureMessages[0]` for vitest, the junit failure message
68
+ for gradle, and the last non-empty output line for exec. The sensitivity check
69
+ persists it as `sensitivity_check.evidence_line` (v6→v7 migration) and the
70
+ friction log's observed snippet prefers it over the raw first line, rendering
71
+ `<no assertion line captured>` when empty and keeping the tail of over-long
72
+ lines. Legacy rows with no stored evidence keep the first-line fallback (#68,
73
+ #90).
74
+
75
+ - **Diagnosable executor attribution.** `TDD_EXECUTOR_MODEL` lets a harness
76
+ declare the executor identity (`source: declared`), taking precedence over
77
+ transcript detection. When identity cannot be resolved, `Executor.reason`
78
+ says why — session id unset, transcript not found, or transcript without a
79
+ model record — and `run start` logs an `executor_unknown` event and surfaces
80
+ `executor_warning` in its envelope. `tdd doctor` gains an informational
81
+ executor-identity check (#74, #92).
82
+
83
+ ### Changed
84
+
85
+ - **Target adoption is evaluated in the same `advance`.** When a cycle's
86
+ declared target is missing and exactly one new test appears, `advance`
87
+ adopts it (logging `declared_test_mismatch`) and judges RED — or drives the
88
+ sensitivity check when it passed — from the suite run that already happened,
89
+ instead of asking for a re-run. A declared id that differs
90
+ from a single same-file candidate only by separator normalisation is
91
+ disambiguated and adopted without asking; two same-file candidates still
92
+ require an explicit `tdd target` (#72, #91).
93
+
94
+ - CI now tests on Python 3.11 and 3.14 only (#87).
95
+
9
96
  ## [0.8.0] - 2026-08-28
10
97
 
11
98
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.8.0
3
+ Version: 0.10.0
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -319,6 +319,13 @@ rendered, never written. Judgement enters in exactly two ways:
319
319
  `commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
320
320
  `plan_defect` is the one that matters most: it records where the plan and the codebase
321
321
  disagreed, which is precisely what the next plan needs to know.
322
+ - **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
323
+ moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
324
+ notes are written after the run ends and attach to the run. Both render in the friction
325
+ log as visually distinct blockquotes (claims, not measurements). Use notes to record
326
+ *why* something happened — a plan assumption that was wrong, an integrity event that the
327
+ telemetry already captures but cannot explain. Notes are unverified by design; an auditor
328
+ compares claims against reality.
322
329
  - **Per run, as prose appended below the rendered document.** Legitimate and expected —
323
330
  post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
324
331
  is unverified: an auditor should trust the projected sections and read appended
@@ -334,7 +341,7 @@ the next plan is written.
334
341
 
335
342
  ## Adapters
336
343
 
337
- `pytest`, `vitest`, `gradle`, `xctest`, and `exec` are built in. The pytest adapter runs
344
+ `pytest`, `vitest`, `gradle`, `xctest`, `cargo`, and `exec` are built in. The pytest adapter runs
338
345
  the suite through the project's own environment manager, detected from its marker files —
339
346
  `uv.lock`, `poetry.lock`, `Pipfile`, `pdm.lock`, or `[tool.poetry]` in `pyproject.toml` —
340
347
  checked at the project root first, then the worktree root (workspace layouts keep one
@@ -386,6 +393,23 @@ lease = "ios-simulator"
386
393
  timeout = 900
387
394
  ```
388
395
 
396
+ The `cargo` adapter drives Rust crates through `cargo test`. Test ids are
397
+ `<target>::<path>` — `lib::layout::tests::wraps_short` for a `#[cfg(test)]` unit test,
398
+ `roundtrip::renders_cover_only` for `tests/roundtrip.rs` — so a targeted run composes to
399
+ cargo's own selectors (`--lib` / `--test <name>` plus `-- --exact <path>`) with no
400
+ translation. Verdicts come from cargo's per-test console lines, attributed to targets by
401
+ the `Running …` headers (stderr is merged so the order survives); doc-tests are excluded
402
+ (`--tests`). A *compile* failure maps to `not_collected`, as for gradle and xctest: write a
403
+ compiling stub (`todo!()`), then observe the panic. Declare `tests/` as `test_paths`, not
404
+ `src/` — unit-test modules live inside production files and must not be staged as tests.
405
+
406
+ ```toml
407
+ [project.kernel]
408
+ root = "kernel"
409
+ adapter = "cargo"
410
+ test_paths = ["tests/"]
411
+ ```
412
+
389
413
  Third-party adapters register under the
390
414
  `tddcli.adapters` entry-point group:
391
415
 
@@ -441,6 +465,7 @@ and is never reclassified as a pin.
441
465
  | `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
442
466
  | `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
443
467
  | `tdd annotate --key --value` | attach judgement to the current cycle |
468
+ | `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
444
469
  | `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
445
470
  | `tdd resume [--unblock --note]` | reconstruct position; human intervention |
446
471
  | `tdd log render [--out]` | project the ledger into a friction log |
@@ -520,6 +545,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
520
545
  failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
521
546
  with the project properly baselined.
522
547
 
548
+ ### Baseline sanity gate (R9.5g)
549
+
550
+ `run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
551
+ environment, not the code, is broken. A project is flagged when it collected at least 10 tests
552
+ (`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
553
+
554
+ error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
555
+
556
+ The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
557
+ not small pre-existing failure sets (baseline subtraction handles those). A project collecting
558
+ fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
559
+ not a broken stack.
560
+
561
+ Pass `--accept-baseline` to override the refusal and proceed anyway:
562
+
563
+ tdd run start --plan tasks/plan.md --accept-baseline
564
+
565
+ An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
566
+ `tdd metrics` and the friction log) so the override is never silent. The gate runs on every
567
+ baseline, including reused ones.
568
+
569
+ To raise the threshold for a project whose legitimate baseline is inherently noisy, set
570
+ `baseline_max_failure_ratio` in `tdd.toml`:
571
+
572
+ ```toml
573
+ [project.legacy]
574
+ root = "legacy"
575
+ adapter = "pytest"
576
+ baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
577
+ ```
578
+
579
+ The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
580
+
581
+ ### Standing-failure delta
582
+
583
+ When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
584
+ event partitioning the current standing failures into:
585
+
586
+ - **new** — absent from the previous run's baseline for this project (growing problem)
587
+ - **inherited** — also present in the previous baseline (stable background noise)
588
+ - **resolved** — in the previous baseline but absent now (fixed between runs)
589
+
590
+ The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
591
+ baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
592
+ nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
593
+
594
+ ### Environment-dependent suites
595
+
596
+ Declare a `health_command` on a project to probe reachability before `run start` captures its
597
+ baseline:
598
+
599
+ ```toml
600
+ [project.backend]
601
+ root = "backend"
602
+ adapter = "pytest"
603
+ health_command = "curl -fsS http://localhost:8080/healthz"
604
+ ```
605
+
606
+ If the command exits non-zero, `run start` refuses immediately with `reason:
607
+ "services_unreachable"` — before claiming the worktree, before probing, before writing anything
608
+ to the ledger. The error names the project and its exit code so the agent can surface the exact
609
+ diagnosis rather than recording a ~1000-failure baseline and continuing.
610
+
611
+ There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
612
+ The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
613
+ is the marker that a project requires live services; no separate boolean is needed.
614
+
523
615
  ## Running a long baseline
524
616
 
525
617
  `run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
@@ -291,6 +291,13 @@ rendered, never written. Judgement enters in exactly two ways:
291
291
  `commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
292
292
  `plan_defect` is the one that matters most: it records where the plan and the codebase
293
293
  disagreed, which is precisely what the next plan needs to know.
294
+ - **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
295
+ moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
296
+ notes are written after the run ends and attach to the run. Both render in the friction
297
+ log as visually distinct blockquotes (claims, not measurements). Use notes to record
298
+ *why* something happened — a plan assumption that was wrong, an integrity event that the
299
+ telemetry already captures but cannot explain. Notes are unverified by design; an auditor
300
+ compares claims against reality.
294
301
  - **Per run, as prose appended below the rendered document.** Legitimate and expected —
295
302
  post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
296
303
  is unverified: an auditor should trust the projected sections and read appended
@@ -306,7 +313,7 @@ the next plan is written.
306
313
 
307
314
  ## Adapters
308
315
 
309
- `pytest`, `vitest`, `gradle`, `xctest`, and `exec` are built in. The pytest adapter runs
316
+ `pytest`, `vitest`, `gradle`, `xctest`, `cargo`, and `exec` are built in. The pytest adapter runs
310
317
  the suite through the project's own environment manager, detected from its marker files —
311
318
  `uv.lock`, `poetry.lock`, `Pipfile`, `pdm.lock`, or `[tool.poetry]` in `pyproject.toml` —
312
319
  checked at the project root first, then the worktree root (workspace layouts keep one
@@ -358,6 +365,23 @@ lease = "ios-simulator"
358
365
  timeout = 900
359
366
  ```
360
367
 
368
+ The `cargo` adapter drives Rust crates through `cargo test`. Test ids are
369
+ `<target>::<path>` — `lib::layout::tests::wraps_short` for a `#[cfg(test)]` unit test,
370
+ `roundtrip::renders_cover_only` for `tests/roundtrip.rs` — so a targeted run composes to
371
+ cargo's own selectors (`--lib` / `--test <name>` plus `-- --exact <path>`) with no
372
+ translation. Verdicts come from cargo's per-test console lines, attributed to targets by
373
+ the `Running …` headers (stderr is merged so the order survives); doc-tests are excluded
374
+ (`--tests`). A *compile* failure maps to `not_collected`, as for gradle and xctest: write a
375
+ compiling stub (`todo!()`), then observe the panic. Declare `tests/` as `test_paths`, not
376
+ `src/` — unit-test modules live inside production files and must not be staged as tests.
377
+
378
+ ```toml
379
+ [project.kernel]
380
+ root = "kernel"
381
+ adapter = "cargo"
382
+ test_paths = ["tests/"]
383
+ ```
384
+
361
385
  Third-party adapters register under the
362
386
  `tddcli.adapters` entry-point group:
363
387
 
@@ -413,6 +437,7 @@ and is never reclassified as a pin.
413
437
  | `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
414
438
  | `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
415
439
  | `tdd annotate --key --value` | attach judgement to the current cycle |
440
+ | `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
416
441
  | `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
417
442
  | `tdd resume [--unblock --note]` | reconstruct position; human intervention |
418
443
  | `tdd log render [--out]` | project the ledger into a friction log |
@@ -492,6 +517,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
492
517
  failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
493
518
  with the project properly baselined.
494
519
 
520
+ ### Baseline sanity gate (R9.5g)
521
+
522
+ `run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
523
+ environment, not the code, is broken. A project is flagged when it collected at least 10 tests
524
+ (`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
525
+
526
+ error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
527
+
528
+ The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
529
+ not small pre-existing failure sets (baseline subtraction handles those). A project collecting
530
+ fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
531
+ not a broken stack.
532
+
533
+ Pass `--accept-baseline` to override the refusal and proceed anyway:
534
+
535
+ tdd run start --plan tasks/plan.md --accept-baseline
536
+
537
+ An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
538
+ `tdd metrics` and the friction log) so the override is never silent. The gate runs on every
539
+ baseline, including reused ones.
540
+
541
+ To raise the threshold for a project whose legitimate baseline is inherently noisy, set
542
+ `baseline_max_failure_ratio` in `tdd.toml`:
543
+
544
+ ```toml
545
+ [project.legacy]
546
+ root = "legacy"
547
+ adapter = "pytest"
548
+ baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
549
+ ```
550
+
551
+ The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
552
+
553
+ ### Standing-failure delta
554
+
555
+ When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
556
+ event partitioning the current standing failures into:
557
+
558
+ - **new** — absent from the previous run's baseline for this project (growing problem)
559
+ - **inherited** — also present in the previous baseline (stable background noise)
560
+ - **resolved** — in the previous baseline but absent now (fixed between runs)
561
+
562
+ The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
563
+ baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
564
+ nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
565
+
566
+ ### Environment-dependent suites
567
+
568
+ Declare a `health_command` on a project to probe reachability before `run start` captures its
569
+ baseline:
570
+
571
+ ```toml
572
+ [project.backend]
573
+ root = "backend"
574
+ adapter = "pytest"
575
+ health_command = "curl -fsS http://localhost:8080/healthz"
576
+ ```
577
+
578
+ If the command exits non-zero, `run start` refuses immediately with `reason:
579
+ "services_unreachable"` — before claiming the worktree, before probing, before writing anything
580
+ to the ledger. The error names the project and its exit code so the agent can surface the exact
581
+ diagnosis rather than recording a ~1000-failure baseline and continuing.
582
+
583
+ There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
584
+ The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
585
+ is the marker that a project requires live services; no separate boolean is needed.
586
+
495
587
  ## Running a long baseline
496
588
 
497
589
  `run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
@@ -74,9 +74,16 @@ end of the run.
74
74
  - The cycle absorbed undeclared scope, or surfaced follow-up work:
75
75
  `--key unplanned_change` / `--key new_work_raised`.
76
76
 
77
+ To capture *why* something happened — a plan assumption that proved wrong, the reason an
78
+ integrity event fired — use `tdd note "<text>"` at the moment you know. A note written
79
+ during an open cycle is stamped with that cycle and phase and appears in the friction log
80
+ as a blockquote alongside the cycle's telemetry. After the run ends, `tdd note` attaches
81
+ at run level and renders in a dedicated **Executor narrative** section. Notes are unverified
82
+ by design; write them as claims, not measurements.
83
+
77
84
  Narrative that spans cycles or happened after the run (CI failures, patterns) goes as
78
- markdown appended below the rendered friction log after `tdd log render` — never into a
79
- cycle annotation it doesn't belong to.
85
+ `tdd note` after the run ends, or as markdown appended below the rendered friction log
86
+ after `tdd log render` — never into a cycle annotation it doesn't belong to.
80
87
 
81
88
  ## When a suite run changes nothing
82
89
 
@@ -47,10 +47,11 @@ packages = ["src/tddcli"]
47
47
  include = ["src", "tests", "examples", "README.md", "LICENSE", "CHANGELOG.md", "SECURITY.md"]
48
48
 
49
49
  [dependency-groups]
50
- dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5", "zizmor>=1.29"]
50
+ dev = ["pytest>=8.0", "pytest-json-report>=1.5", "pytest-xdist>=3.5", "ruff>=0.5", "zizmor>=1.29"]
51
51
 
52
52
  [tool.pytest.ini_options]
53
53
  testpaths = ["tests"]
54
+ addopts = ["-n", "auto"]
54
55
 
55
56
  [tool.ruff]
56
57
  line-length = 100
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.8.0"
6
+ __version__ = "0.10.0"
@@ -4,6 +4,7 @@ import importlib.metadata
4
4
  from pathlib import Path
5
5
 
6
6
  from .base import Adapter, Collection, GateResult, Verdict
7
+ from .cargo_adapter import CargoAdapter
7
8
  from .exec_adapter import ExecAdapter
8
9
  from .gradle_adapter import GradleAdapter
9
10
  from .pytest_adapter import PytestAdapter
@@ -11,6 +12,7 @@ from .vitest_adapter import VitestAdapter
11
12
  from .xctest_adapter import XCTestAdapter
12
13
 
13
14
  REGISTRY: dict[str, type[Adapter]] = {
15
+ "cargo": CargoAdapter,
14
16
  "exec": ExecAdapter,
15
17
  "gradle": GradleAdapter,
16
18
  "pytest": PytestAdapter,
@@ -25,6 +25,7 @@ class Verdict:
25
25
  target: str | None = None
26
26
  target_outcome: str = NOT_FOUND
27
27
  target_failure: str = ""
28
+ target_evidence: str = ""
28
29
  passed: list[str] = field(default_factory=list)
29
30
  failed: list[str] = field(default_factory=list)
30
31
  duration_ms: int = 0
@@ -214,6 +215,14 @@ class Adapter:
214
215
  return None
215
216
  return {k: os.path.expandvars(v) for k, v in merged.items()}
216
217
 
218
+ def lint_target_id(self, native: str) -> str | None:
219
+ """Return a problem message when `native` can never match a collected id, else None."""
220
+ return None
221
+
222
+ def target_path(self, native: str) -> str | None:
223
+ """Return the file-path portion of `native`, or None for non-path-bearing ids."""
224
+ return None
225
+
217
226
  def stub_hint(self) -> str:
218
227
  """The language idiom for a stub body, quoted into the create_stub directive."""
219
228
  return "a body that fails loudly, never working logic"