tdd-cli 0.8.0__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/CHANGELOG.md +74 -0
  2. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/PKG-INFO +76 -1
  3. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/README.md +75 -0
  4. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/examples/skills/tdd-drive/SKILL.md +9 -2
  5. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/__init__.py +1 -1
  6. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/adapters/base.py +9 -0
  7. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/adapters/exec_adapter.py +3 -0
  8. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/adapters/gradle_adapter.py +13 -1
  9. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/adapters/pytest_adapter.py +20 -1
  10. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/adapters/vitest_adapter.py +17 -1
  11. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/adapters/xctest_adapter.py +19 -1
  12. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/advance.py +101 -24
  13. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/cli.py +148 -6
  14. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/config.py +16 -0
  15. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/identity.py +16 -4
  16. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/ledger.py +28 -1
  17. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/render.py +27 -2
  18. tdd_cli-0.9.0/src/tddcli/target_lint.py +61 -0
  19. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/conftest.py +9 -0
  20. tdd_cli-0.9.0/tests/test_advance_adoption.py +126 -0
  21. tdd_cli-0.9.0/tests/test_baseline_sanity.py +229 -0
  22. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_config_and_staging.py +22 -0
  23. tdd_cli-0.9.0/tests/test_evidence_extraction.py +226 -0
  24. tdd_cli-0.9.0/tests/test_executor_attribution.py +133 -0
  25. tdd_cli-0.9.0/tests/test_executor_notes.py +209 -0
  26. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_heartbeat.py +7 -6
  27. tdd_cli-0.9.0/tests/test_sensitivity_evidence.py +157 -0
  28. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_snapshot_and_identity.py +47 -0
  29. tdd_cli-0.9.0/tests/test_target_lint.py +202 -0
  30. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/.gitignore +0 -0
  31. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/LICENSE +0 -0
  32. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/SECURITY.md +0 -0
  33. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/README.md +0 -0
  34. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/bash_hook.py +0 -0
  35. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/examples/claude-code-hooks/stop_hook.py +0 -0
  36. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/examples/plan.md +0 -0
  37. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/examples/skills/tdd-drive/README.md +0 -0
  38. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/examples/skills/tdd-handoff/README.md +0 -0
  39. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
  40. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/pyproject.toml +0 -0
  41. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/adapters/__init__.py +0 -0
  42. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/contract.py +0 -0
  43. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/envelope.py +0 -0
  44. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/fleet.py +0 -0
  45. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/gitutil.py +0 -0
  46. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/leases.py +0 -0
  47. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/machine.py +0 -0
  48. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/snapshot.py +0 -0
  49. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/src/tddcli/staging.py +0 -0
  50. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_ancillary_files.py +0 -0
  51. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_artifact_regeneration.py +0 -0
  52. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_baseline_integrity.py +0 -0
  53. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_batch_collection.py +0 -0
  54. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_concurrent_advance.py +0 -0
  55. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_config_drift.py +0 -0
  56. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_contract.py +0 -0
  57. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_doctor_attribution.py +0 -0
  58. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_doctor_blockers.py +0 -0
  59. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_end_to_end.py +0 -0
  60. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_example_plan.py +0 -0
  61. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_exec_adapter.py +0 -0
  62. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_failure_clipping.py +0 -0
  63. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_fleet.py +0 -0
  64. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_gradle_adapter.py +0 -0
  65. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_id_normalisation.py +0 -0
  66. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_init_detection.py +0 -0
  67. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_named_leases_and_timeout.py +0 -0
  68. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_pin_cycles.py +0 -0
  69. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_progress.py +0 -0
  70. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_project_commands.py +0 -0
  71. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_project_env.py +0 -0
  72. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_python_env_managers.py +0 -0
  73. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_refactor_cycles.py +0 -0
  74. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_release_surface.py +0 -0
  75. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_run_claim.py +0 -0
  76. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_single_project_repo.py +0 -0
  77. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_stub_hint.py +0 -0
  78. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_suite_overrides.py +0 -0
  79. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_target_validation.py +0 -0
  80. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_timing_visibility.py +0 -0
  81. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_undeclared_close_gate.py +0 -0
  82. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_undeclared_dedup.py +0 -0
  83. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_vitest_adapter.py +0 -0
  84. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_worker_leases.py +0 -0
  85. {tdd_cli-0.8.0 → tdd_cli-0.9.0}/tests/test_xctest_adapter.py +0 -0
@@ -6,6 +6,80 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.9.0] - 2026-09-03
10
+
11
+ ### Added
12
+
13
+ - **`tdd note` — an executor-narrative channel.** `tdd note "<text>"` attaches
14
+ a phase-stamped note to the current cycle, or a run-level note once the run
15
+ has ended (it falls back to the latest run, so no active run is required).
16
+ Cycle notes render in the friction log as blockquote claims under their cycle;
17
+ run-level notes render under a new `## Executor narrative` section, which is
18
+ omitted when there are none. Envelopes that carry an integrity event now nudge
19
+ for a note while the reason is fresh (silenced once the cycle has one), and
20
+ the terminal `COMPLETE` envelope — via `advance` or a final-cycle skip — asks
21
+ for a closing narrative before rendering. Stored in a new `note` table via a
22
+ v7→v8 ledger migration (#77, #94).
23
+
24
+ - **Baseline sanity gate.** `run start` refuses a baseline whose failing ratio
25
+ exceeds a threshold (default 0.5, suites under 10 collected tests exempt) with
26
+ `reason: "baseline_implausible"`, on the grounds that a mostly-red suite means
27
+ the environment is broken rather than the code. A per-project
28
+ `baseline_max_failure_ratio` in `tdd.toml` overrides the default, and
29
+ `--accept-baseline` bypasses the gate and logs a `baseline_accepted` integrity
30
+ event (#67, #84).
31
+
32
+ - **Standing-failure delta.** A non-empty baseline is now compared against the
33
+ previous run's baseline for the same worktree and a `baseline_standing_delta`
34
+ event partitions the failing set into new versus inherited failures, so a
35
+ pre-existing red test is distinguishable from one that broke since the last
36
+ run (#67, #84).
37
+
38
+ - **Per-project `health_command`.** An optional command in `tdd.toml` that is
39
+ probed before baseline capture; if it fails, `run start` refuses with
40
+ `reason: "services_unreachable"` instead of recording a baseline against a
41
+ down dependency (#67, #84).
42
+
43
+ - **Declared-target lint.** `plan register` and `run start` now lint every
44
+ declared target against the adapter's id grammar (pytest `::`, vitest ` > `,
45
+ gradle `Class/method`, xctest `Bundle/Class/testMethod`) and against project
46
+ roots: a target path that duplicates a non-`.` project root prefix is refused
47
+ unless the nested path genuinely exists in the worktree. Findings are returned
48
+ under `reason: "target_lint"`; `run start` re-lints the stored contract
49
+ against the current config before claiming a baseline (#71, #88).
50
+
51
+ - **Per-adapter sensitivity evidence line.** Each adapter now extracts a
52
+ `target_evidence` line from a failing target — the first `E` line for pytest
53
+ (skipping the xdist worker header), the first `: error:` line in the test's
54
+ window for xctest, `failureMessages[0]` for vitest, the junit failure message
55
+ for gradle, and the last non-empty output line for exec. The sensitivity check
56
+ persists it as `sensitivity_check.evidence_line` (v6→v7 migration) and the
57
+ friction log's observed snippet prefers it over the raw first line, rendering
58
+ `<no assertion line captured>` when empty and keeping the tail of over-long
59
+ lines. Legacy rows with no stored evidence keep the first-line fallback (#68,
60
+ #90).
61
+
62
+ - **Diagnosable executor attribution.** `TDD_EXECUTOR_MODEL` lets a harness
63
+ declare the executor identity (`source: declared`), taking precedence over
64
+ transcript detection. When identity cannot be resolved, `Executor.reason`
65
+ says why — session id unset, transcript not found, or transcript without a
66
+ model record — and `run start` logs an `executor_unknown` event and surfaces
67
+ `executor_warning` in its envelope. `tdd doctor` gains an informational
68
+ executor-identity check (#74, #92).
69
+
70
+ ### Changed
71
+
72
+ - **Target adoption is evaluated in the same `advance`.** When a cycle's
73
+ declared target is missing and exactly one new test appears, `advance`
74
+ adopts it (logging `declared_test_mismatch`) and judges RED — or drives the
75
+ sensitivity check when it passed — from the suite run that already happened,
76
+ instead of asking for a re-run. A declared id that differs
77
+ from a single same-file candidate only by separator normalisation is
78
+ disambiguated and adopted without asking; two same-file candidates still
79
+ require an explicit `tdd target` (#72, #91).
80
+
81
+ - CI now tests on Python 3.11 and 3.14 only (#87).
82
+
9
83
  ## [0.8.0] - 2026-08-28
10
84
 
11
85
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.8.0
3
+ Version: 0.9.0
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -319,6 +319,13 @@ rendered, never written. Judgement enters in exactly two ways:
319
319
  `commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
320
320
  `plan_defect` is the one that matters most: it records where the plan and the codebase
321
321
  disagreed, which is precisely what the next plan needs to know.
322
+ - **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
323
+ moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
324
+ notes are written after the run ends and attach to the run. Both render in the friction
325
+ log as visually distinct blockquotes (claims, not measurements). Use notes to record
326
+ *why* something happened — a plan assumption that was wrong, an integrity event that the
327
+ telemetry already captures but cannot explain. Notes are unverified by design; an auditor
328
+ compares claims against reality.
322
329
  - **Per run, as prose appended below the rendered document.** Legitimate and expected —
323
330
  post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
324
331
  is unverified: an auditor should trust the projected sections and read appended
@@ -441,6 +448,7 @@ and is never reclassified as a pin.
441
448
  | `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
442
449
  | `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
443
450
  | `tdd annotate --key --value` | attach judgement to the current cycle |
451
+ | `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
444
452
  | `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
445
453
  | `tdd resume [--unblock --note]` | reconstruct position; human intervention |
446
454
  | `tdd log render [--out]` | project the ledger into a friction log |
@@ -520,6 +528,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
520
528
  failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
521
529
  with the project properly baselined.
522
530
 
531
+ ### Baseline sanity gate (R9.5g)
532
+
533
+ `run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
534
+ environment, not the code, is broken. A project is flagged when it collected at least 10 tests
535
+ (`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
536
+
537
+ error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
538
+
539
+ The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
540
+ not small pre-existing failure sets (baseline subtraction handles those). A project collecting
541
+ fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
542
+ not a broken stack.
543
+
544
+ Pass `--accept-baseline` to override the refusal and proceed anyway:
545
+
546
+ tdd run start --plan tasks/plan.md --accept-baseline
547
+
548
+ An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
549
+ `tdd metrics` and the friction log) so the override is never silent. The gate runs on every
550
+ baseline, including reused ones.
551
+
552
+ To raise the threshold for a project whose legitimate baseline is inherently noisy, set
553
+ `baseline_max_failure_ratio` in `tdd.toml`:
554
+
555
+ ```toml
556
+ [project.legacy]
557
+ root = "legacy"
558
+ adapter = "pytest"
559
+ baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
560
+ ```
561
+
562
+ The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
563
+
564
+ ### Standing-failure delta
565
+
566
+ When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
567
+ event partitioning the current standing failures into:
568
+
569
+ - **new** — absent from the previous run's baseline for this project (growing problem)
570
+ - **inherited** — also present in the previous baseline (stable background noise)
571
+ - **resolved** — in the previous baseline but absent now (fixed between runs)
572
+
573
+ The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
574
+ baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
575
+ nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
576
+
577
+ ### Environment-dependent suites
578
+
579
+ Declare a `health_command` on a project to probe reachability before `run start` captures its
580
+ baseline:
581
+
582
+ ```toml
583
+ [project.backend]
584
+ root = "backend"
585
+ adapter = "pytest"
586
+ health_command = "curl -fsS http://localhost:8080/healthz"
587
+ ```
588
+
589
+ If the command exits non-zero, `run start` refuses immediately with `reason:
590
+ "services_unreachable"` — before claiming the worktree, before probing, before writing anything
591
+ to the ledger. The error names the project and its exit code so the agent can surface the exact
592
+ diagnosis rather than recording a ~1000-failure baseline and continuing.
593
+
594
+ There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
595
+ The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
596
+ is the marker that a project requires live services; no separate boolean is needed.
597
+
523
598
  ## Running a long baseline
524
599
 
525
600
  `run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
@@ -291,6 +291,13 @@ rendered, never written. Judgement enters in exactly two ways:
291
291
  `commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
292
292
  `plan_defect` is the one that matters most: it records where the plan and the codebase
293
293
  disagreed, which is precisely what the next plan needs to know.
294
+ - **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
295
+ moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
296
+ notes are written after the run ends and attach to the run. Both render in the friction
297
+ log as visually distinct blockquotes (claims, not measurements). Use notes to record
298
+ *why* something happened — a plan assumption that was wrong, an integrity event that the
299
+ telemetry already captures but cannot explain. Notes are unverified by design; an auditor
300
+ compares claims against reality.
294
301
  - **Per run, as prose appended below the rendered document.** Legitimate and expected —
295
302
  post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
296
303
  is unverified: an auditor should trust the projected sections and read appended
@@ -413,6 +420,7 @@ and is never reclassified as a pin.
413
420
  | `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
414
421
  | `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
415
422
  | `tdd annotate --key --value` | attach judgement to the current cycle |
423
+ | `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
416
424
  | `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
417
425
  | `tdd resume [--unblock --note]` | reconstruct position; human intervention |
418
426
  | `tdd log render [--out]` | project the ledger into a friction log |
@@ -492,6 +500,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
492
500
  failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
493
501
  with the project properly baselined.
494
502
 
503
+ ### Baseline sanity gate (R9.5g)
504
+
505
+ `run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
506
+ environment, not the code, is broken. A project is flagged when it collected at least 10 tests
507
+ (`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
508
+
509
+ error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
510
+
511
+ The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
512
+ not small pre-existing failure sets (baseline subtraction handles those). A project collecting
513
+ fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
514
+ not a broken stack.
515
+
516
+ Pass `--accept-baseline` to override the refusal and proceed anyway:
517
+
518
+ tdd run start --plan tasks/plan.md --accept-baseline
519
+
520
+ An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
521
+ `tdd metrics` and the friction log) so the override is never silent. The gate runs on every
522
+ baseline, including reused ones.
523
+
524
+ To raise the threshold for a project whose legitimate baseline is inherently noisy, set
525
+ `baseline_max_failure_ratio` in `tdd.toml`:
526
+
527
+ ```toml
528
+ [project.legacy]
529
+ root = "legacy"
530
+ adapter = "pytest"
531
+ baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
532
+ ```
533
+
534
+ The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
535
+
536
+ ### Standing-failure delta
537
+
538
+ When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
539
+ event partitioning the current standing failures into:
540
+
541
+ - **new** — absent from the previous run's baseline for this project (growing problem)
542
+ - **inherited** — also present in the previous baseline (stable background noise)
543
+ - **resolved** — in the previous baseline but absent now (fixed between runs)
544
+
545
+ The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
546
+ baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
547
+ nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
548
+
549
+ ### Environment-dependent suites
550
+
551
+ Declare a `health_command` on a project to probe reachability before `run start` captures its
552
+ baseline:
553
+
554
+ ```toml
555
+ [project.backend]
556
+ root = "backend"
557
+ adapter = "pytest"
558
+ health_command = "curl -fsS http://localhost:8080/healthz"
559
+ ```
560
+
561
+ If the command exits non-zero, `run start` refuses immediately with `reason:
562
+ "services_unreachable"` — before claiming the worktree, before probing, before writing anything
563
+ to the ledger. The error names the project and its exit code so the agent can surface the exact
564
+ diagnosis rather than recording a ~1000-failure baseline and continuing.
565
+
566
+ There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
567
+ The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
568
+ is the marker that a project requires live services; no separate boolean is needed.
569
+
495
570
  ## Running a long baseline
496
571
 
497
572
  `run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
@@ -74,9 +74,16 @@ end of the run.
74
74
  - The cycle absorbed undeclared scope, or surfaced follow-up work:
75
75
  `--key unplanned_change` / `--key new_work_raised`.
76
76
 
77
+ To capture *why* something happened — a plan assumption that proved wrong, the reason an
78
+ integrity event fired — use `tdd note "<text>"` at the moment you know. A note written
79
+ during an open cycle is stamped with that cycle and phase and appears in the friction log
80
+ as a blockquote alongside the cycle's telemetry. After the run ends, `tdd note` attaches
81
+ at run level and renders in a dedicated **Executor narrative** section. Notes are unverified
82
+ by design; write them as claims, not measurements.
83
+
77
84
  Narrative that spans cycles or happened after the run (CI failures, patterns) goes as
78
- markdown appended below the rendered friction log after `tdd log render` — never into a
79
- cycle annotation it doesn't belong to.
85
+ `tdd note` after the run ends, or as markdown appended below the rendered friction log
86
+ after `tdd log render` — never into a cycle annotation it doesn't belong to.
80
87
 
81
88
  ## When a suite run changes nothing
82
89
 
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.8.0"
6
+ __version__ = "0.9.0"
@@ -25,6 +25,7 @@ class Verdict:
25
25
  target: str | None = None
26
26
  target_outcome: str = NOT_FOUND
27
27
  target_failure: str = ""
28
+ target_evidence: str = ""
28
29
  passed: list[str] = field(default_factory=list)
29
30
  failed: list[str] = field(default_factory=list)
30
31
  duration_ms: int = 0
@@ -214,6 +215,14 @@ class Adapter:
214
215
  return None
215
216
  return {k: os.path.expandvars(v) for k, v in merged.items()}
216
217
 
218
+ def lint_target_id(self, native: str) -> str | None:
219
+ """Return a problem message when `native` can never match a collected id, else None."""
220
+ return None
221
+
222
+ def target_path(self, native: str) -> str | None:
223
+ """Return the file-path portion of `native`, or None for non-path-bearing ids."""
224
+ return None
225
+
217
226
  def stub_hint(self) -> str:
218
227
  """The language idiom for a stub body, quoted into the create_stub directive."""
219
228
  return "a body that fails loudly, never working logic"
@@ -168,6 +168,9 @@ class ExecAdapter(Adapter):
168
168
  if target == qualified:
169
169
  verdict.target_outcome = FAILED
170
170
  verdict.target_failure = clip_failure(combined)
171
+ verdict.target_evidence = next(
172
+ (ln for ln in reversed(combined.splitlines()) if ln.strip()), ""
173
+ )
171
174
 
172
175
  verdict.duration_ms = int((time.monotonic() - started) * 1000)
173
176
  return verdict
@@ -114,6 +114,14 @@ class GradleAdapter(Adapter):
114
114
  " assertion failure"
115
115
  )
116
116
 
117
+ def lint_target_id(self, native: str) -> str | None:
118
+ if "/" not in native:
119
+ return (
120
+ f"gradle target ids must contain '/' between class and method "
121
+ f"(got {native!r}); expected shape: com.example.ClassName/testMethodName"
122
+ )
123
+ return None
124
+
117
125
  # ------------------------------------------------------------------
118
126
  # Core command
119
127
  # ------------------------------------------------------------------
@@ -321,7 +329,11 @@ class GradleAdapter(Adapter):
321
329
  verdict.target_outcome = PASSED
322
330
  elif target in failed:
323
331
  verdict.target_outcome = FAILED
324
- verdict.target_failure = failures.get(target, "test failed")
332
+ failure_text = failures.get(target, "test failed")
333
+ verdict.target_failure = failure_text
334
+ verdict.target_evidence = next(
335
+ (ln for ln in failure_text.splitlines() if ln.strip()), ""
336
+ )
325
337
  # else NOT_FOUND (default): the suite ran but never produced this id
326
338
 
327
339
  return verdict
@@ -12,6 +12,7 @@ Two defects in the prior system are corrected here (R10.2):
12
12
  from __future__ import annotations
13
13
 
14
14
  import json
15
+ import re
15
16
  import shlex
16
17
  import tempfile
17
18
  from pathlib import Path
@@ -48,6 +49,14 @@ class PytestAdapter(Adapter):
48
49
  def stub_hint(self) -> str:
49
50
  return "`raise NotImplementedError` in every body"
50
51
 
52
+ def lint_target_id(self, native: str) -> str | None:
53
+ if "::" not in native:
54
+ return f"pytest target ids must contain '::' (got {native!r}); expected shape: path/to/test_file.py::test_name"
55
+ return None
56
+
57
+ def target_path(self, native: str) -> str | None:
58
+ return native.split("::", 1)[0]
59
+
51
60
  def _runner_prefix(self) -> str:
52
61
  """The project root is checked before the worktree root: a workspace keeps
53
62
  one lockfile at the top, but a member with its own marker owns its choice."""
@@ -140,7 +149,9 @@ class PytestAdapter(Adapter):
140
149
  if hit is not None:
141
150
  verdict.target_outcome = PASSED if hit["outcome"] == "passed" else FAILED
142
151
  call = hit.get("call") or hit.get("setup") or {}
143
- verdict.target_failure = clip_failure(str(call.get("longrepr", "")))
152
+ longrepr = str(call.get("longrepr", ""))
153
+ verdict.target_failure = clip_failure(longrepr)
154
+ verdict.target_evidence = self._evidence_line(longrepr)
144
155
  else:
145
156
  target_file = native.split("::", 1)[0]
146
157
  if any(c == target_file or c.startswith(target_file) for c in uncollectable):
@@ -152,6 +163,14 @@ class PytestAdapter(Adapter):
152
163
  verdict.target_outcome = NOT_FOUND
153
164
  return verdict
154
165
 
166
+ @staticmethod
167
+ def _evidence_line(longrepr: str) -> str:
168
+ for line in longrepr.splitlines():
169
+ m = re.match(r"^E\s+(.*)", line)
170
+ if m:
171
+ return m.group(1)
172
+ return ""
173
+
155
174
  @staticmethod
156
175
  def _collector_error(collectors: list[dict], target_file: str) -> str:
157
176
  for collector in collectors:
@@ -47,6 +47,17 @@ class VitestAdapter(Adapter):
47
47
  def stub_hint(self) -> str:
48
48
  return '`throw new Error("not implemented")` in every body'
49
49
 
50
+ def lint_target_id(self, native: str) -> str | None:
51
+ if " > " not in native:
52
+ return (
53
+ f"vitest target ids must contain ' > ' between the file and test name "
54
+ f"(got {native!r}); expected shape: <file> > <describe titles> <test title>"
55
+ )
56
+ return None
57
+
58
+ def target_path(self, native: str) -> str | None:
59
+ return native.split(" > ", 1)[0]
60
+
50
61
  def normalise_id(self, test_id: str) -> str:
51
62
  """Canonicalise the describe/test separator for target matching.
52
63
 
@@ -141,8 +152,13 @@ class VitestAdapter(Adapter):
141
152
  self.normalise_id(self._id_for(suite.get("name", ""), t["fullName"]))
142
153
  == ntarget
143
154
  ):
155
+ messages = t.get("failureMessages", [])
144
156
  verdict.target_failure = "\n".join(
145
- clip_failure(m, 600) for m in t.get("failureMessages", [])[:3]
157
+ clip_failure(m, 600) for m in messages[:3]
158
+ )
159
+ first_msg = messages[0] if messages else ""
160
+ verdict.target_evidence = next(
161
+ (ln for ln in first_msg.splitlines() if ln.strip()), ""
146
162
  )
147
163
  return verdict
148
164
 
@@ -94,6 +94,15 @@ class XCTestAdapter(Adapter):
94
94
  " compiles first, then observe the assertion failure"
95
95
  )
96
96
 
97
+ def lint_target_id(self, native: str) -> str | None:
98
+ parts = native.split("/")
99
+ if len(parts) != 3 or any(not p for p in parts):
100
+ return (
101
+ f"xctest target ids must be exactly three '/'-separated parts "
102
+ f"(got {native!r}); expected shape: Bundle/Class/testMethod"
103
+ )
104
+ return None
105
+
97
106
  # ------------------------------------------------------------------
98
107
  # Core command
99
108
  # ------------------------------------------------------------------
@@ -263,7 +272,9 @@ class XCTestAdapter(Adapter):
263
272
  verdict.target_outcome = PASSED
264
273
  elif target in verdict.failed:
265
274
  verdict.target_outcome = FAILED
266
- verdict.target_failure = self._failure_for(combined, self.strip(target))
275
+ window = self._failure_for(combined, self.strip(target))
276
+ verdict.target_failure = window
277
+ verdict.target_evidence = self._evidence_line(window)
267
278
  # else NOT_FOUND (default)
268
279
 
269
280
  return verdict
@@ -280,6 +291,13 @@ class XCTestAdapter(Adapter):
280
291
  if ": error:" in line or "** BUILD FAILED **" in line
281
292
  )
282
293
 
294
+ @staticmethod
295
+ def _evidence_line(window: str) -> str:
296
+ for line in window.splitlines():
297
+ if ": error:" in line:
298
+ return line
299
+ return ""
300
+
283
301
  def _failure_for(self, combined: str, native_id: str) -> str:
284
302
  """Capture the assertion lines between 'started' and 'failed' for one test.
285
303
 
@@ -22,13 +22,35 @@ from .machine import (
22
22
  Engine,
23
23
  )
24
24
 
25
+ NUDGE_KINDS = {"red_first_violation", "undeclared_file_touched", "implementation_during_red"}
26
+
27
+
28
+ def _note_nudge(engine: Engine, cycle) -> str:
29
+ if cycle is None:
30
+ return ""
31
+ events = engine.ledger.all(
32
+ "SELECT kind FROM integrity_event WHERE cycle_id = ? AND kind IN ({})".format(
33
+ ",".join("?" * len(NUDGE_KINDS))
34
+ ),
35
+ (cycle["id"], *NUDGE_KINDS),
36
+ )
37
+ if not events:
38
+ return ""
39
+ existing_note = engine.ledger.one(
40
+ "SELECT id FROM note WHERE cycle_id = ?", (cycle["id"],)
41
+ )
42
+ if existing_note is not None:
43
+ return ""
44
+ return ' An integrity event was recorded on this cycle — consider `tdd note "<why>"` while the reason is fresh.'
45
+
25
46
 
26
47
  def _reply(engine: Engine, cycle, verb: Verb, detail: str, **result) -> Envelope:
27
48
  fresh = engine.ledger.one("SELECT * FROM cycle WHERE id = ?", (cycle["id"],)) if cycle else None
49
+ nudge = _note_nudge(engine, fresh or cycle)
28
50
  return Envelope(
29
51
  run=engine.run_state(fresh or cycle),
30
52
  result=result,
31
- next_action=NextAction(verb, detail),
53
+ next_action=NextAction(verb, detail + nudge),
32
54
  )
33
55
 
34
56
 
@@ -117,6 +139,27 @@ def _stage_and_commit(engine: Engine, cycle, phase: str, declared) -> tuple[str
117
139
  return sha, staged, classification
118
140
 
119
141
 
142
+ def _disambiguate(candidates: list[str], declared: str, adapter) -> str | None:
143
+ norm_declared = adapter.normalise_id(declared)
144
+ matches = [c for c in candidates if adapter.normalise_id(c) == norm_declared]
145
+ if len(matches) == 1:
146
+ return matches[0]
147
+ declared_file = declared.split("::", 1)[-1].split("::")[0].split(" > ")[0]
148
+ same_file = [c for c in candidates if c.split("::", 1)[-1].split("::")[0].split(" > ")[0] == declared_file]
149
+ if len(same_file) == 1:
150
+ return same_file[0]
151
+ return None
152
+
153
+
154
+ def _outcome_from_verdicts(verdicts, test_id: str) -> str | None:
155
+ for v in verdicts:
156
+ if test_id in v.failed:
157
+ return FAILED
158
+ if test_id in v.passed:
159
+ return PASSED
160
+ return None
161
+
162
+
120
163
  # -- handlers ------------------------------------------------------------
121
164
 
122
165
 
@@ -127,7 +170,7 @@ def _handle_test_phase(engine: Engine, cycle, retried: bool, expect_pass: bool)
127
170
  targets = json.loads(cycle["target_tests"])
128
171
  phase = cycle["phase"]
129
172
 
130
- outcomes, others, _, failure = engine.run_projects(
173
+ outcomes, others, verdicts, failure = engine.run_projects(
131
174
  projects, targets, cycle, phase, retried
132
175
  )
133
176
 
@@ -144,29 +187,60 @@ def _handle_test_phase(engine: Engine, cycle, retried: bool, expect_pass: bool)
144
187
  "cycle", cycle["id"], target_tests=json.dumps(kept + candidates)
145
188
  )
146
189
  cycle = engine.ledger.one("SELECT * FROM cycle WHERE id = ?", (cycle["id"],))
190
+ adopted_outcome = _outcome_from_verdicts(verdicts, candidates[0])
191
+ if adopted_outcome is None:
192
+ return _reply(
193
+ engine, cycle, Verb.REFACTOR_OR_ADVANCE,
194
+ f"Adopted {candidates[0]} as the target (declared {missing[0]} was not"
195
+ " collected). Run `tdd advance` again to evaluate it.",
196
+ adopted=candidates,
197
+ )
198
+ targets = kept + candidates
199
+ outcomes = {candidates[0]: adopted_outcome}
200
+ others = [t for t in others if t != candidates[0]]
201
+ elif len(candidates) > 1:
202
+ owner = missing[0].split("::", 1)[0]
203
+ adapter = adapters.build(engine.config.project(owner), engine.worktree)
204
+ resolved = _disambiguate(candidates, missing[0], adapter)
205
+ if resolved is not None:
206
+ engine.ledger.event(
207
+ engine.run["id"], cycle["id"], "declared_test_mismatch",
208
+ json.dumps({"declared": missing, "adopted": [resolved], "all_candidates": candidates}),
209
+ )
210
+ kept = [t for t in targets if t not in missing]
211
+ engine.ledger.update(
212
+ "cycle", cycle["id"], target_tests=json.dumps(kept + [resolved])
213
+ )
214
+ cycle = engine.ledger.one("SELECT * FROM cycle WHERE id = ?", (cycle["id"],))
215
+ adopted_outcome = _outcome_from_verdicts(verdicts, resolved)
216
+ if adopted_outcome is None:
217
+ return _reply(
218
+ engine, cycle, Verb.REFACTOR_OR_ADVANCE,
219
+ f"Adopted {resolved} as the target (declared {missing[0]} was not"
220
+ " collected). Run `tdd advance` again to evaluate it.",
221
+ adopted=[resolved],
222
+ )
223
+ targets = kept + [resolved]
224
+ outcomes = {resolved: adopted_outcome}
225
+ others = [t for t in others if t != resolved]
226
+ else:
227
+ engine.ledger.event(
228
+ engine.run["id"], cycle["id"], "multiple_new_tests",
229
+ json.dumps(candidates),
230
+ )
231
+ return _reply(
232
+ engine, cycle, Verb.NAME_TARGET_TEST,
233
+ "Several new tests appeared; a cycle covers one behaviour. Name the"
234
+ " intended target with `tdd target <id>`.",
235
+ candidates=candidates,
236
+ )
237
+ else:
147
238
  return _reply(
148
- engine, cycle, Verb.REFACTOR_OR_ADVANCE,
149
- f"Adopted {candidates[0]} as the target (declared {missing[0]} was not"
150
- " collected). Run `tdd advance` again to evaluate it.",
151
- adopted=candidates,
152
- )
153
- if len(candidates) > 1:
154
- engine.ledger.event(
155
- engine.run["id"], cycle["id"], "multiple_new_tests",
156
- json.dumps(candidates),
157
- )
158
- return _reply(
159
- engine, cycle, Verb.NAME_TARGET_TEST,
160
- "Several new tests appeared; a cycle covers one behaviour. Name the"
161
- " intended target with `tdd target <id>`.",
162
- candidates=candidates,
239
+ engine, cycle, Verb.WRITE_TEST,
240
+ f"Target {missing[0]} was not collected and no new test was found."
241
+ " Write the failing test.",
242
+ missing=missing,
163
243
  )
164
- return _reply(
165
- engine, cycle, Verb.WRITE_TEST,
166
- f"Target {missing[0]} was not collected and no new test was found."
167
- " Write the failing test.",
168
- missing=missing,
169
- )
170
244
 
171
245
  not_collected = [t for t, o in outcomes.items() if o == NOT_COLLECTED]
172
246
  if not_collected:
@@ -392,7 +466,10 @@ def _handle_refactor(engine: Engine, cycle, retried: bool) -> Envelope:
392
466
  },
393
467
  result={"commit": sha, "regenerated": regenerated or None},
394
468
  next_action=NextAction(
395
- Verb.COMPLETE, "All declared cycles are complete. Run `tdd log render`."
469
+ Verb.COMPLETE,
470
+ "All declared cycles are complete."
471
+ " Before rendering, record a closing narrative with `tdd note \"<hardest cycle and why, plan inaccuracies, deviations>\"`."
472
+ " Then run `tdd log render`.",
396
473
  ),
397
474
  )
398
475
  verb, opening = engine.opening_action(nxt)