tdd-cli 0.8.0__tar.gz → 0.10.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/CHANGELOG.md +87 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/PKG-INFO +94 -2
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/README.md +93 -1
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/skills/tdd-drive/SKILL.md +9 -2
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/pyproject.toml +2 -1
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/__init__.py +2 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/base.py +9 -0
- tdd_cli-0.10.0/src/tddcli/adapters/cargo_adapter.py +321 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/exec_adapter.py +3 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/gradle_adapter.py +13 -1
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/pytest_adapter.py +20 -1
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/vitest_adapter.py +17 -1
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/adapters/xctest_adapter.py +19 -1
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/advance.py +101 -24
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/cli.py +148 -6
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/config.py +16 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/identity.py +16 -4
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/ledger.py +28 -1
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/render.py +27 -2
- tdd_cli-0.10.0/src/tddcli/target_lint.py +61 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/conftest.py +9 -1
- tdd_cli-0.10.0/tests/test_advance_adoption.py +126 -0
- tdd_cli-0.10.0/tests/test_baseline_sanity.py +229 -0
- tdd_cli-0.10.0/tests/test_cargo_adapter.py +395 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_config_and_staging.py +22 -0
- tdd_cli-0.10.0/tests/test_evidence_extraction.py +226 -0
- tdd_cli-0.10.0/tests/test_executor_attribution.py +133 -0
- tdd_cli-0.10.0/tests/test_executor_notes.py +209 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_heartbeat.py +7 -6
- tdd_cli-0.10.0/tests/test_sensitivity_evidence.py +157 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_snapshot_and_identity.py +47 -0
- tdd_cli-0.10.0/tests/test_target_lint.py +202 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/.gitignore +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/LICENSE +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/SECURITY.md +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/plan.md +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/contract.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/machine.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/src/tddcli/staging.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_ancillary_files.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_artifact_regeneration.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_baseline_integrity.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_batch_collection.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_concurrent_advance.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_contract.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_doctor_blockers.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_exec_adapter.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_failure_clipping.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_gradle_adapter.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_id_normalisation.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_named_leases_and_timeout.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_project_commands.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_project_env.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_release_surface.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_suite_overrides.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_timing_visibility.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_undeclared_close_gate.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_undeclared_dedup.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_vitest_adapter.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_worker_leases.py +0 -0
- {tdd_cli-0.8.0 → tdd_cli-0.10.0}/tests/test_xctest_adapter.py +0 -0
|
@@ -6,6 +6,93 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.10.0] - 2026-09-05
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **cargo adapter** — Rust driver through `cargo test`. Ids are `<target>::<path>`
|
|
14
|
+
(`lib::…` for unit tests, `<tests-file-stem>::…` for integration tests) and compose
|
|
15
|
+
to `--lib`/`--test <name>` + `-- --exact <path>` for targeted runs. Verdicts are
|
|
16
|
+
parsed from cargo's per-test console lines, attributed to targets by the
|
|
17
|
+
`Running …` headers with stderr merged; doc-tests are excluded. A compile error maps
|
|
18
|
+
to `not_collected`, matching gradle/xctest. Collection is `cargo test --tests --
|
|
19
|
+
--list` with a per-file `--test <stem>` fallback; `tdd doctor`'s collectable gate is
|
|
20
|
+
`cargo test --no-run`. Target lint requires the `::` separator.
|
|
21
|
+
|
|
22
|
+
## [0.9.0] - 2026-09-03
|
|
23
|
+
|
|
24
|
+
### Added
|
|
25
|
+
|
|
26
|
+
- **`tdd note` — an executor-narrative channel.** `tdd note "<text>"` attaches
|
|
27
|
+
a phase-stamped note to the current cycle, or a run-level note once the run
|
|
28
|
+
has ended (it falls back to the latest run, so no active run is required).
|
|
29
|
+
Cycle notes render in the friction log as blockquote claims under their cycle;
|
|
30
|
+
run-level notes render under a new `## Executor narrative` section, which is
|
|
31
|
+
omitted when there are none. Envelopes that carry an integrity event now nudge
|
|
32
|
+
for a note while the reason is fresh (silenced once the cycle has one), and
|
|
33
|
+
the terminal `COMPLETE` envelope — via `advance` or a final-cycle skip — asks
|
|
34
|
+
for a closing narrative before rendering. Stored in a new `note` table via a
|
|
35
|
+
v7→v8 ledger migration (#77, #94).
|
|
36
|
+
|
|
37
|
+
- **Baseline sanity gate.** `run start` refuses a baseline whose failing ratio
|
|
38
|
+
exceeds a threshold (default 0.5, suites under 10 collected tests exempt) with
|
|
39
|
+
`reason: "baseline_implausible"`, on the grounds that a mostly-red suite means
|
|
40
|
+
the environment is broken rather than the code. A per-project
|
|
41
|
+
`baseline_max_failure_ratio` in `tdd.toml` overrides the default, and
|
|
42
|
+
`--accept-baseline` bypasses the gate and logs a `baseline_accepted` integrity
|
|
43
|
+
event (#67, #84).
|
|
44
|
+
|
|
45
|
+
- **Standing-failure delta.** A non-empty baseline is now compared against the
|
|
46
|
+
previous run's baseline for the same worktree and a `baseline_standing_delta`
|
|
47
|
+
event partitions the failing set into new versus inherited failures, so a
|
|
48
|
+
pre-existing red test is distinguishable from one that broke since the last
|
|
49
|
+
run (#67, #84).
|
|
50
|
+
|
|
51
|
+
- **Per-project `health_command`.** An optional command in `tdd.toml` that is
|
|
52
|
+
probed before baseline capture; if it fails, `run start` refuses with
|
|
53
|
+
`reason: "services_unreachable"` instead of recording a baseline against a
|
|
54
|
+
down dependency (#67, #84).
|
|
55
|
+
|
|
56
|
+
- **Declared-target lint.** `plan register` and `run start` now lint every
|
|
57
|
+
declared target against the adapter's id grammar (pytest `::`, vitest ` > `,
|
|
58
|
+
gradle `Class/method`, xctest `Bundle/Class/testMethod`) and against project
|
|
59
|
+
roots: a target path that duplicates a non-`.` project root prefix is refused
|
|
60
|
+
unless the nested path genuinely exists in the worktree. Findings are returned
|
|
61
|
+
under `reason: "target_lint"`; `run start` re-lints the stored contract
|
|
62
|
+
against the current config before claiming a baseline (#71, #88).
|
|
63
|
+
|
|
64
|
+
- **Per-adapter sensitivity evidence line.** Each adapter now extracts a
|
|
65
|
+
`target_evidence` line from a failing target — the first `E` line for pytest
|
|
66
|
+
(skipping the xdist worker header), the first `: error:` line in the test's
|
|
67
|
+
window for xctest, `failureMessages[0]` for vitest, the junit failure message
|
|
68
|
+
for gradle, and the last non-empty output line for exec. The sensitivity check
|
|
69
|
+
persists it as `sensitivity_check.evidence_line` (v6→v7 migration) and the
|
|
70
|
+
friction log's observed snippet prefers it over the raw first line, rendering
|
|
71
|
+
`<no assertion line captured>` when empty and keeping the tail of over-long
|
|
72
|
+
lines. Legacy rows with no stored evidence keep the first-line fallback (#68,
|
|
73
|
+
#90).
|
|
74
|
+
|
|
75
|
+
- **Diagnosable executor attribution.** `TDD_EXECUTOR_MODEL` lets a harness
|
|
76
|
+
declare the executor identity (`source: declared`), taking precedence over
|
|
77
|
+
transcript detection. When identity cannot be resolved, `Executor.reason`
|
|
78
|
+
says why — session id unset, transcript not found, or transcript without a
|
|
79
|
+
model record — and `run start` logs an `executor_unknown` event and surfaces
|
|
80
|
+
`executor_warning` in its envelope. `tdd doctor` gains an informational
|
|
81
|
+
executor-identity check (#74, #92).
|
|
82
|
+
|
|
83
|
+
### Changed
|
|
84
|
+
|
|
85
|
+
- **Target adoption is evaluated in the same `advance`.** When a cycle's
|
|
86
|
+
declared target is missing and exactly one new test appears, `advance`
|
|
87
|
+
adopts it (logging `declared_test_mismatch`) and judges RED — or drives the
|
|
88
|
+
sensitivity check when it passed — from the suite run that already happened,
|
|
89
|
+
instead of asking for a re-run. A declared id that differs
|
|
90
|
+
from a single same-file candidate only by separator normalisation is
|
|
91
|
+
disambiguated and adopted without asking; two same-file candidates still
|
|
92
|
+
require an explicit `tdd target` (#72, #91).
|
|
93
|
+
|
|
94
|
+
- CI now tests on Python 3.11 and 3.14 only (#87).
|
|
95
|
+
|
|
9
96
|
## [0.8.0] - 2026-08-28
|
|
10
97
|
|
|
11
98
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.10.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -319,6 +319,13 @@ rendered, never written. Judgement enters in exactly two ways:
|
|
|
319
319
|
`commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
|
|
320
320
|
`plan_defect` is the one that matters most: it records where the plan and the codebase
|
|
321
321
|
disagreed, which is precisely what the next plan needs to know.
|
|
322
|
+
- **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
|
|
323
|
+
moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
|
|
324
|
+
notes are written after the run ends and attach to the run. Both render in the friction
|
|
325
|
+
log as visually distinct blockquotes (claims, not measurements). Use notes to record
|
|
326
|
+
*why* something happened — a plan assumption that was wrong, an integrity event that the
|
|
327
|
+
telemetry already captures but cannot explain. Notes are unverified by design; an auditor
|
|
328
|
+
compares claims against reality.
|
|
322
329
|
- **Per run, as prose appended below the rendered document.** Legitimate and expected —
|
|
323
330
|
post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
|
|
324
331
|
is unverified: an auditor should trust the projected sections and read appended
|
|
@@ -334,7 +341,7 @@ the next plan is written.
|
|
|
334
341
|
|
|
335
342
|
## Adapters
|
|
336
343
|
|
|
337
|
-
`pytest`, `vitest`, `gradle`, `xctest`, and `exec` are built in. The pytest adapter runs
|
|
344
|
+
`pytest`, `vitest`, `gradle`, `xctest`, `cargo`, and `exec` are built in. The pytest adapter runs
|
|
338
345
|
the suite through the project's own environment manager, detected from its marker files —
|
|
339
346
|
`uv.lock`, `poetry.lock`, `Pipfile`, `pdm.lock`, or `[tool.poetry]` in `pyproject.toml` —
|
|
340
347
|
checked at the project root first, then the worktree root (workspace layouts keep one
|
|
@@ -386,6 +393,23 @@ lease = "ios-simulator"
|
|
|
386
393
|
timeout = 900
|
|
387
394
|
```
|
|
388
395
|
|
|
396
|
+
The `cargo` adapter drives Rust crates through `cargo test`. Test ids are
|
|
397
|
+
`<target>::<path>` — `lib::layout::tests::wraps_short` for a `#[cfg(test)]` unit test,
|
|
398
|
+
`roundtrip::renders_cover_only` for `tests/roundtrip.rs` — so a targeted run composes to
|
|
399
|
+
cargo's own selectors (`--lib` / `--test <name>` plus `-- --exact <path>`) with no
|
|
400
|
+
translation. Verdicts come from cargo's per-test console lines, attributed to targets by
|
|
401
|
+
the `Running …` headers (stderr is merged so the order survives); doc-tests are excluded
|
|
402
|
+
(`--tests`). A *compile* failure maps to `not_collected`, as for gradle and xctest: write a
|
|
403
|
+
compiling stub (`todo!()`), then observe the panic. Declare `tests/` as `test_paths`, not
|
|
404
|
+
`src/` — unit-test modules live inside production files and must not be staged as tests.
|
|
405
|
+
|
|
406
|
+
```toml
|
|
407
|
+
[project.kernel]
|
|
408
|
+
root = "kernel"
|
|
409
|
+
adapter = "cargo"
|
|
410
|
+
test_paths = ["tests/"]
|
|
411
|
+
```
|
|
412
|
+
|
|
389
413
|
Third-party adapters register under the
|
|
390
414
|
`tddcli.adapters` entry-point group:
|
|
391
415
|
|
|
@@ -441,6 +465,7 @@ and is never reclassified as a pin.
|
|
|
441
465
|
| `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
|
|
442
466
|
| `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
|
|
443
467
|
| `tdd annotate --key --value` | attach judgement to the current cycle |
|
|
468
|
+
| `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
|
|
444
469
|
| `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
|
|
445
470
|
| `tdd resume [--unblock --note]` | reconstruct position; human intervention |
|
|
446
471
|
| `tdd log render [--out]` | project the ledger into a friction log |
|
|
@@ -520,6 +545,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
|
|
|
520
545
|
failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
|
|
521
546
|
with the project properly baselined.
|
|
522
547
|
|
|
548
|
+
### Baseline sanity gate (R9.5g)
|
|
549
|
+
|
|
550
|
+
`run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
|
|
551
|
+
environment, not the code, is broken. A project is flagged when it collected at least 10 tests
|
|
552
|
+
(`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
|
|
553
|
+
|
|
554
|
+
error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
|
|
555
|
+
|
|
556
|
+
The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
|
|
557
|
+
not small pre-existing failure sets (baseline subtraction handles those). A project collecting
|
|
558
|
+
fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
|
|
559
|
+
not a broken stack.
|
|
560
|
+
|
|
561
|
+
Pass `--accept-baseline` to override the refusal and proceed anyway:
|
|
562
|
+
|
|
563
|
+
tdd run start --plan tasks/plan.md --accept-baseline
|
|
564
|
+
|
|
565
|
+
An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
|
|
566
|
+
`tdd metrics` and the friction log) so the override is never silent. The gate runs on every
|
|
567
|
+
baseline, including reused ones.
|
|
568
|
+
|
|
569
|
+
To raise the threshold for a project whose legitimate baseline is inherently noisy, set
|
|
570
|
+
`baseline_max_failure_ratio` in `tdd.toml`:
|
|
571
|
+
|
|
572
|
+
```toml
|
|
573
|
+
[project.legacy]
|
|
574
|
+
root = "legacy"
|
|
575
|
+
adapter = "pytest"
|
|
576
|
+
baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
|
|
577
|
+
```
|
|
578
|
+
|
|
579
|
+
The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
|
|
580
|
+
|
|
581
|
+
### Standing-failure delta
|
|
582
|
+
|
|
583
|
+
When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
|
|
584
|
+
event partitioning the current standing failures into:
|
|
585
|
+
|
|
586
|
+
- **new** — absent from the previous run's baseline for this project (growing problem)
|
|
587
|
+
- **inherited** — also present in the previous baseline (stable background noise)
|
|
588
|
+
- **resolved** — in the previous baseline but absent now (fixed between runs)
|
|
589
|
+
|
|
590
|
+
The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
|
|
591
|
+
baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
|
|
592
|
+
nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
|
|
593
|
+
|
|
594
|
+
### Environment-dependent suites
|
|
595
|
+
|
|
596
|
+
Declare a `health_command` on a project to probe reachability before `run start` captures its
|
|
597
|
+
baseline:
|
|
598
|
+
|
|
599
|
+
```toml
|
|
600
|
+
[project.backend]
|
|
601
|
+
root = "backend"
|
|
602
|
+
adapter = "pytest"
|
|
603
|
+
health_command = "curl -fsS http://localhost:8080/healthz"
|
|
604
|
+
```
|
|
605
|
+
|
|
606
|
+
If the command exits non-zero, `run start` refuses immediately with `reason:
|
|
607
|
+
"services_unreachable"` — before claiming the worktree, before probing, before writing anything
|
|
608
|
+
to the ledger. The error names the project and its exit code so the agent can surface the exact
|
|
609
|
+
diagnosis rather than recording a ~1000-failure baseline and continuing.
|
|
610
|
+
|
|
611
|
+
There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
|
|
612
|
+
The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
|
|
613
|
+
is the marker that a project requires live services; no separate boolean is needed.
|
|
614
|
+
|
|
523
615
|
## Running a long baseline
|
|
524
616
|
|
|
525
617
|
`run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
|
|
@@ -291,6 +291,13 @@ rendered, never written. Judgement enters in exactly two ways:
|
|
|
291
291
|
`commit_shape_deviation`, `test_setup_smell`, `unplanned_change`, `new_work_raised`.
|
|
292
292
|
`plan_defect` is the one that matters most: it records where the plan and the codebase
|
|
293
293
|
disagreed, which is precisely what the next plan needs to know.
|
|
294
|
+
- **Per cycle or run, through `tdd note "<text>"`** — free-text narrative captured at the
|
|
295
|
+
moment the reason exists. Cycle notes are scoped to the open cycle and phase; run-level
|
|
296
|
+
notes are written after the run ends and attach to the run. Both render in the friction
|
|
297
|
+
log as visually distinct blockquotes (claims, not measurements). Use notes to record
|
|
298
|
+
*why* something happened — a plan assumption that was wrong, an integrity event that the
|
|
299
|
+
telemetry already captures but cannot explain. Notes are unverified by design; an auditor
|
|
300
|
+
compares claims against reality.
|
|
294
301
|
- **Per run, as prose appended below the rendered document.** Legitimate and expected —
|
|
295
302
|
post-run narrative (CI failures, patterns noticed) has no cycle to attach to. But it
|
|
296
303
|
is unverified: an auditor should trust the projected sections and read appended
|
|
@@ -306,7 +313,7 @@ the next plan is written.
|
|
|
306
313
|
|
|
307
314
|
## Adapters
|
|
308
315
|
|
|
309
|
-
`pytest`, `vitest`, `gradle`, `xctest`, and `exec` are built in. The pytest adapter runs
|
|
316
|
+
`pytest`, `vitest`, `gradle`, `xctest`, `cargo`, and `exec` are built in. The pytest adapter runs
|
|
310
317
|
the suite through the project's own environment manager, detected from its marker files —
|
|
311
318
|
`uv.lock`, `poetry.lock`, `Pipfile`, `pdm.lock`, or `[tool.poetry]` in `pyproject.toml` —
|
|
312
319
|
checked at the project root first, then the worktree root (workspace layouts keep one
|
|
@@ -358,6 +365,23 @@ lease = "ios-simulator"
|
|
|
358
365
|
timeout = 900
|
|
359
366
|
```
|
|
360
367
|
|
|
368
|
+
The `cargo` adapter drives Rust crates through `cargo test`. Test ids are
|
|
369
|
+
`<target>::<path>` — `lib::layout::tests::wraps_short` for a `#[cfg(test)]` unit test,
|
|
370
|
+
`roundtrip::renders_cover_only` for `tests/roundtrip.rs` — so a targeted run composes to
|
|
371
|
+
cargo's own selectors (`--lib` / `--test <name>` plus `-- --exact <path>`) with no
|
|
372
|
+
translation. Verdicts come from cargo's per-test console lines, attributed to targets by
|
|
373
|
+
the `Running …` headers (stderr is merged so the order survives); doc-tests are excluded
|
|
374
|
+
(`--tests`). A *compile* failure maps to `not_collected`, as for gradle and xctest: write a
|
|
375
|
+
compiling stub (`todo!()`), then observe the panic. Declare `tests/` as `test_paths`, not
|
|
376
|
+
`src/` — unit-test modules live inside production files and must not be staged as tests.
|
|
377
|
+
|
|
378
|
+
```toml
|
|
379
|
+
[project.kernel]
|
|
380
|
+
root = "kernel"
|
|
381
|
+
adapter = "cargo"
|
|
382
|
+
test_paths = ["tests/"]
|
|
383
|
+
```
|
|
384
|
+
|
|
361
385
|
Third-party adapters register under the
|
|
362
386
|
`tddcli.adapters` entry-point group:
|
|
363
387
|
|
|
@@ -413,6 +437,7 @@ and is never reclassified as a pin.
|
|
|
413
437
|
| `tdd cycle skip --reason` | sanctioned path for a cycle the plan got wrong |
|
|
414
438
|
| `tdd sensitivity begin\|check\|end` | prove a passing test can fail; verify restore |
|
|
415
439
|
| `tdd annotate --key --value` | attach judgement to the current cycle |
|
|
440
|
+
| `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
|
|
416
441
|
| `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
|
|
417
442
|
| `tdd resume [--unblock --note]` | reconstruct position; human intervention |
|
|
418
443
|
| `tdd log render [--out]` | project the ledger into a friction log |
|
|
@@ -492,6 +517,73 @@ kind `no_baseline_for_project` rather than mislabelling them as regressions. Rec
|
|
|
492
517
|
failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
|
|
493
518
|
with the project properly baselined.
|
|
494
519
|
|
|
520
|
+
### Baseline sanity gate (R9.5g)
|
|
521
|
+
|
|
522
|
+
`run start` refuses a baseline whose failure ratio is implausibly high — a signal that the
|
|
523
|
+
environment, not the code, is broken. A project is flagged when it collected at least 10 tests
|
|
524
|
+
(`BASELINE_MIN_COLLECTED`) and more than 50% of them failed (`BASELINE_MAX_FAILURE_RATIO_DEFAULT`):
|
|
525
|
+
|
|
526
|
+
error: baseline_implausible — backend: 1005/1010 failing (ratio 0.995)
|
|
527
|
+
|
|
528
|
+
The gate targets stack-down blowouts (all 1000 tests failing because a service is unreachable),
|
|
529
|
+
not small pre-existing failure sets (baseline subtraction handles those). A project collecting
|
|
530
|
+
fewer than 10 tests is exempt — a 2-of-3 red baseline is ordinary pre-existing-failure territory,
|
|
531
|
+
not a broken stack.
|
|
532
|
+
|
|
533
|
+
Pass `--accept-baseline` to override the refusal and proceed anyway:
|
|
534
|
+
|
|
535
|
+
tdd run start --plan tasks/plan.md --accept-baseline
|
|
536
|
+
|
|
537
|
+
An accepted implausible baseline records a `baseline_accepted` integrity event (visible in
|
|
538
|
+
`tdd metrics` and the friction log) so the override is never silent. The gate runs on every
|
|
539
|
+
baseline, including reused ones.
|
|
540
|
+
|
|
541
|
+
To raise the threshold for a project whose legitimate baseline is inherently noisy, set
|
|
542
|
+
`baseline_max_failure_ratio` in `tdd.toml`:
|
|
543
|
+
|
|
544
|
+
```toml
|
|
545
|
+
[project.legacy]
|
|
546
|
+
root = "legacy"
|
|
547
|
+
adapter = "pytest"
|
|
548
|
+
baseline_max_failure_ratio = 0.8 # allow up to 80% failing at baseline
|
|
549
|
+
```
|
|
550
|
+
|
|
551
|
+
The value must be in `(0, 1]`. Without it, the repo-wide default of 0.5 applies.
|
|
552
|
+
|
|
553
|
+
### Standing-failure delta
|
|
554
|
+
|
|
555
|
+
When a non-empty baseline is captured, `run start` emits a `baseline_standing_delta` integrity
|
|
556
|
+
event partitioning the current standing failures into:
|
|
557
|
+
|
|
558
|
+
- **new** — absent from the previous run's baseline for this project (growing problem)
|
|
559
|
+
- **inherited** — also present in the previous baseline (stable background noise)
|
|
560
|
+
- **resolved** — in the previous baseline but absent now (fixed between runs)
|
|
561
|
+
|
|
562
|
+
The event is visible in `tdd metrics` under `integrity_events`. A first run with no prior
|
|
563
|
+
baseline reports every failure as `new`. A fully-green baseline (zero failing tests) emits
|
|
564
|
+
nothing. Use this to spot a permanently-red set that baseline subtraction would otherwise hide.
|
|
565
|
+
|
|
566
|
+
### Environment-dependent suites
|
|
567
|
+
|
|
568
|
+
Declare a `health_command` on a project to probe reachability before `run start` captures its
|
|
569
|
+
baseline:
|
|
570
|
+
|
|
571
|
+
```toml
|
|
572
|
+
[project.backend]
|
|
573
|
+
root = "backend"
|
|
574
|
+
adapter = "pytest"
|
|
575
|
+
health_command = "curl -fsS http://localhost:8080/healthz"
|
|
576
|
+
```
|
|
577
|
+
|
|
578
|
+
If the command exits non-zero, `run start` refuses immediately with `reason:
|
|
579
|
+
"services_unreachable"` — before claiming the worktree, before probing, before writing anything
|
|
580
|
+
to the ledger. The error names the project and its exit code so the agent can surface the exact
|
|
581
|
+
diagnosis rather than recording a ~1000-failure baseline and continuing.
|
|
582
|
+
|
|
583
|
+
There is no override flag for `services_unreachable`: a down stack yields no believable baseline.
|
|
584
|
+
The resolution is to fix the stack or drop the project from this run. Presence of `health_command`
|
|
585
|
+
is the marker that a project requires live services; no separate boolean is needed.
|
|
586
|
+
|
|
495
587
|
## Running a long baseline
|
|
496
588
|
|
|
497
589
|
`run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
|
|
@@ -74,9 +74,16 @@ end of the run.
|
|
|
74
74
|
- The cycle absorbed undeclared scope, or surfaced follow-up work:
|
|
75
75
|
`--key unplanned_change` / `--key new_work_raised`.
|
|
76
76
|
|
|
77
|
+
To capture *why* something happened — a plan assumption that proved wrong, the reason an
|
|
78
|
+
integrity event fired — use `tdd note "<text>"` at the moment you know. A note written
|
|
79
|
+
during an open cycle is stamped with that cycle and phase and appears in the friction log
|
|
80
|
+
as a blockquote alongside the cycle's telemetry. After the run ends, `tdd note` attaches
|
|
81
|
+
at run level and renders in a dedicated **Executor narrative** section. Notes are unverified
|
|
82
|
+
by design; write them as claims, not measurements.
|
|
83
|
+
|
|
77
84
|
Narrative that spans cycles or happened after the run (CI failures, patterns) goes as
|
|
78
|
-
|
|
79
|
-
cycle annotation it doesn't belong to.
|
|
85
|
+
`tdd note` after the run ends, or as markdown appended below the rendered friction log
|
|
86
|
+
after `tdd log render` — never into a cycle annotation it doesn't belong to.
|
|
80
87
|
|
|
81
88
|
## When a suite run changes nothing
|
|
82
89
|
|
|
@@ -47,10 +47,11 @@ packages = ["src/tddcli"]
|
|
|
47
47
|
include = ["src", "tests", "examples", "README.md", "LICENSE", "CHANGELOG.md", "SECURITY.md"]
|
|
48
48
|
|
|
49
49
|
[dependency-groups]
|
|
50
|
-
dev = ["pytest>=8.0", "pytest-json-report>=1.5", "ruff>=0.5", "zizmor>=1.29"]
|
|
50
|
+
dev = ["pytest>=8.0", "pytest-json-report>=1.5", "pytest-xdist>=3.5", "ruff>=0.5", "zizmor>=1.29"]
|
|
51
51
|
|
|
52
52
|
[tool.pytest.ini_options]
|
|
53
53
|
testpaths = ["tests"]
|
|
54
|
+
addopts = ["-n", "auto"]
|
|
54
55
|
|
|
55
56
|
[tool.ruff]
|
|
56
57
|
line-length = 100
|
|
@@ -4,6 +4,7 @@ import importlib.metadata
|
|
|
4
4
|
from pathlib import Path
|
|
5
5
|
|
|
6
6
|
from .base import Adapter, Collection, GateResult, Verdict
|
|
7
|
+
from .cargo_adapter import CargoAdapter
|
|
7
8
|
from .exec_adapter import ExecAdapter
|
|
8
9
|
from .gradle_adapter import GradleAdapter
|
|
9
10
|
from .pytest_adapter import PytestAdapter
|
|
@@ -11,6 +12,7 @@ from .vitest_adapter import VitestAdapter
|
|
|
11
12
|
from .xctest_adapter import XCTestAdapter
|
|
12
13
|
|
|
13
14
|
REGISTRY: dict[str, type[Adapter]] = {
|
|
15
|
+
"cargo": CargoAdapter,
|
|
14
16
|
"exec": ExecAdapter,
|
|
15
17
|
"gradle": GradleAdapter,
|
|
16
18
|
"pytest": PytestAdapter,
|
|
@@ -25,6 +25,7 @@ class Verdict:
|
|
|
25
25
|
target: str | None = None
|
|
26
26
|
target_outcome: str = NOT_FOUND
|
|
27
27
|
target_failure: str = ""
|
|
28
|
+
target_evidence: str = ""
|
|
28
29
|
passed: list[str] = field(default_factory=list)
|
|
29
30
|
failed: list[str] = field(default_factory=list)
|
|
30
31
|
duration_ms: int = 0
|
|
@@ -214,6 +215,14 @@ class Adapter:
|
|
|
214
215
|
return None
|
|
215
216
|
return {k: os.path.expandvars(v) for k, v in merged.items()}
|
|
216
217
|
|
|
218
|
+
def lint_target_id(self, native: str) -> str | None:
|
|
219
|
+
"""Return a problem message when `native` can never match a collected id, else None."""
|
|
220
|
+
return None
|
|
221
|
+
|
|
222
|
+
def target_path(self, native: str) -> str | None:
|
|
223
|
+
"""Return the file-path portion of `native`, or None for non-path-bearing ids."""
|
|
224
|
+
return None
|
|
225
|
+
|
|
217
226
|
def stub_hint(self) -> str:
|
|
218
227
|
"""The language idiom for a stub body, quoted into the create_stub directive."""
|
|
219
228
|
return "a body that fails loudly, never working logic"
|