tdd-cli 0.12.2__tar.gz → 0.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/CHANGELOG.md +71 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/PKG-INFO +54 -6
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/README.md +53 -5
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/SECURITY.md +10 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/docs/PRD.md +61 -9
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/docs/harness-integration.md +2 -2
- tdd_cli-0.14.0/docs/split-runner.md +187 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/pyproject.toml +1 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/__init__.py +1 -1
- tdd_cli-0.14.0/src/tddcli/actor.py +163 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/adapters/base.py +3 -12
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/adapters/cargo_adapter.py +4 -2
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/adapters/exec_adapter.py +4 -3
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/adapters/gradle_adapter.py +6 -3
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/adapters/pytest_adapter.py +26 -7
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/adapters/vitest_adapter.py +34 -4
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/adapters/xctest_adapter.py +4 -2
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/advance.py +52 -3
- tdd_cli-0.14.0/src/tddcli/agentfs.py +53 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/cli.py +270 -12
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/docs.py +5 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/gitutil.py +44 -25
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/identity.py +66 -2
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/ledger.py +95 -8
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/machine.py +15 -5
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/render.py +122 -4
- tdd_cli-0.14.0/src/tddcli/runner.py +110 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/snapshot.py +4 -5
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/conftest.py +91 -0
- tdd_cli-0.14.0/tests/test_actor_seams.py +44 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_cargo_adapter.py +16 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_close_sweep_gates.py +39 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_docs_command.py +8 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_exec_adapter.py +1 -1
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_gradle_adapter.py +17 -1
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_heartbeat.py +5 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_late_baseline.py +5 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_project_commands.py +4 -2
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_release_surface.py +15 -0
- tdd_cli-0.14.0/tests/test_run_abandon.py +256 -0
- tdd_cli-0.14.0/tests/test_runner_import.py +148 -0
- tdd_cli-0.14.0/tests/test_split_client.py +94 -0
- tdd_cli-0.14.0/tests/test_split_config.py +58 -0
- tdd_cli-0.14.0/tests/test_split_doctor.py +61 -0
- tdd_cli-0.14.0/tests/test_split_identity.py +81 -0
- tdd_cli-0.14.0/tests/test_split_ledger.py +30 -0
- tdd_cli-0.14.0/tests/test_split_runner.py +105 -0
- tdd_cli-0.14.0/tests/test_sudo_actor.py +70 -0
- tdd_cli-0.14.0/tests/test_suite_scope.py +261 -0
- tdd_cli-0.14.0/tests/test_target_only_runs.py +160 -0
- tdd_cli-0.14.0/tests/test_time_report.py +130 -0
- tdd_cli-0.14.0/tests/test_tree_hash.py +141 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_xctest_adapter.py +16 -1
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/.gitignore +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/LICENSE +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/examples/plan.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/config.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/contract.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/plan_paths.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/staging.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/src/tddcli/target_lint.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_advance_adoption.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_ancillary_files.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_artifact_regeneration.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_baseline_integrity.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_baseline_sanity.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_batch_collection.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_concurrent_advance.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_config_and_staging.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_contract.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_doctor_blockers.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_evidence_extraction.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_executor_attribution.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_executor_notes.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_failure_clipping.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_id_normalisation.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_named_leases_and_timeout.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_plan_paths.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_project_env.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_pytest_xdist_group.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_sensitivity_evidence.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_snapshot_and_identity.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_suite_overrides.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_target_lint.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_timing_visibility.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_undeclared_close_gate.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_undeclared_dedup.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_vitest_adapter.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.14.0}/tests/test_worker_leases.py +0 -0
|
@@ -6,6 +6,77 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.14.0] - 2026-09-24
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **`tdd run abandon`: end a run that will not be finished** (#150). `tdd run abandon
|
|
14
|
+
--reason <text>` ends the worktree's live or blocked run with `outcome = abandoned`;
|
|
15
|
+
`--run <id>` does the same for a run whose worktree is gone. It records the reason and
|
|
16
|
+
who did it (account and executor) in a new `abandonment` table, counts as a human
|
|
17
|
+
intervention, and releases the worktree's claims. `tdd fleet` stops listing the run,
|
|
18
|
+
`tdd metrics` reports it with an `abandoned: {reason, by}` entry, and a new run can
|
|
19
|
+
start at the same path. The table is added on open; the ledger schema version is
|
|
20
|
+
unchanged.
|
|
21
|
+
|
|
22
|
+
## [0.13.0] - 2026-09-23
|
|
23
|
+
|
|
24
|
+
### Added
|
|
25
|
+
|
|
26
|
+
- **Where a run's time went** (#147). `tdd metrics` gains a per-run `time` object:
|
|
27
|
+
`wall_clock_s`, `suite_s`, `suite_share` and `by_phase` (runs and suite seconds for
|
|
28
|
+
every phase recorded). A live run's wall clock runs to now. The friction log gains a
|
|
29
|
+
`## Time` section with the same split, a per-phase table and a per-cycle table
|
|
30
|
+
(wall clock, suite time and suite runs). Nothing new is recorded: these are the
|
|
31
|
+
ledger's existing timestamps and `invocation.duration_ms`.
|
|
32
|
+
- **Split mode: a ledger the agent's account cannot open** (#142). With
|
|
33
|
+
`/etc/tdd-cli/runner.toml` in place, a separate runner account owns the ledger
|
|
34
|
+
in a mode-700 directory. The agent's `tdd` forwards every verb except `docs`
|
|
35
|
+
and `init` to it through `sudo`, and the runner executes every suite, gate,
|
|
36
|
+
hook and git command as the agent. Executor identity is recorded as `operator`
|
|
37
|
+
when the runner's config assigns the calling account a model, and `claimed`
|
|
38
|
+
otherwise. `tdd runner import <ledger>` brings an existing ledger under the
|
|
39
|
+
runner, marked `pre_split_import`, which `tdd metrics` reports. Setup and a
|
|
40
|
+
verification checklist: `tdd docs split`.
|
|
41
|
+
- `tdd doctor` reports `mode` (`single` or `split`). In split mode it probes,
|
|
42
|
+
live and as the agent, that the uid drop works, that the agent cannot read the
|
|
43
|
+
ledger, and that it cannot write the install.
|
|
44
|
+
|
|
45
|
+
### Changed
|
|
46
|
+
|
|
47
|
+
- **GREEN runs the target first** (#149). An `AWAITING_IMPL` advance whose target
|
|
48
|
+
still fails now says so after running only the target. A passing target is followed
|
|
49
|
+
by the whole-suite run that decides GREEN, as before. `tdd metrics` counts
|
|
50
|
+
`impl_attempts` from each advance's target-only run only, and still counts every
|
|
51
|
+
`AWAITING_IMPL` run in cycles recorded before this change.
|
|
52
|
+
- `tdd doctor` on a single-user machine no longer lets "ledger outside
|
|
53
|
+
worktree" stand for isolation: a `ledger isolation` notice says the ledger is
|
|
54
|
+
owned by the same uid that runs the agent. The README and SECURITY.md say the
|
|
55
|
+
same. Single-user installs behave exactly as before.
|
|
56
|
+
- **RED runs only the target test** (#148). `AWAITING_TEST`, `AWAITING_PIN` and
|
|
57
|
+
`tdd sensitivity check` now run the target alone: pytest runs the owning suite's
|
|
58
|
+
collect command with the node id, and vitest filters to the file and an anchored,
|
|
59
|
+
escaped `-t` name. A failure elsewhere no longer blocks RED. It is caught at
|
|
60
|
+
GREEN, which runs the whole suite for **every** adapter. A collection error in
|
|
61
|
+
the target's own file is still `not_collected`. A target adopted in place of the
|
|
62
|
+
declared id is run in the same advance. The ledger is now schema 12:
|
|
63
|
+
`invocation.others_observed` is `0` for a target-only run.
|
|
64
|
+
|
|
65
|
+
### Fixed
|
|
66
|
+
|
|
67
|
+
- **GREEN ran only the target on cargo, gradle, xctest and exec** (#153), so a
|
|
68
|
+
regression elsewhere reached the close sweep unseen, and the sweep could skip the
|
|
69
|
+
cycle's own project without the whole suite ever running. These adapters now
|
|
70
|
+
narrow only when asked for a target-only run.
|
|
71
|
+
|
|
72
|
+
- **The close sweep re-ran a tree that had just passed** (#146): `gitutil.tree_hash` hashed git
|
|
73
|
+
state (index entries plus the unstaged diff), so the same content hashed differently before
|
|
74
|
+
and after the GREEN commit, and the §6.1 skip of a cycle's own suites never fired. It now
|
|
75
|
+
hashes working-tree content through a throwaway index, independent of what is staged or
|
|
76
|
+
committed. A skipped close sweep still runs the cycle project's lint/typecheck gates; before
|
|
77
|
+
this fix, the skip would have dropped them too. `--reuse-baselines` cache entries written by
|
|
78
|
+
earlier versions no longer match and are re-probed once.
|
|
79
|
+
|
|
9
80
|
## [0.12.2] - 2026-09-17
|
|
10
81
|
|
|
11
82
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.14.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -48,8 +48,15 @@ stop silently mid-plan; and no record comparable across runs, plans, or models.
|
|
|
48
48
|
|
|
49
49
|
This tool removes the agent from the reporting path. It runs the suites itself, computes
|
|
50
50
|
every phase transition from what the tests observably did, and records the whole run in a
|
|
51
|
-
ledger the
|
|
52
|
-
|
|
51
|
+
ledger outside the worktree — which is what the friction logs and metrics at the end rest
|
|
52
|
+
on.
|
|
53
|
+
|
|
54
|
+
How far out of reach that ledger is depends on the machine. On a single-user install it
|
|
55
|
+
belongs to the same account that runs the agent: an agent that edits files in its checkout
|
|
56
|
+
cannot touch it, but one that goes looking for it can. In **split mode** a separate runner
|
|
57
|
+
account owns the ledger and runs the agent's suites *as the agent*, so the agent's account
|
|
58
|
+
cannot open it at all. `tdd doctor` says which mode a machine is in; `tdd docs split` sets
|
|
59
|
+
the second one up.
|
|
53
60
|
|
|
54
61
|
## Install
|
|
55
62
|
|
|
@@ -289,6 +296,26 @@ When the last declared cycle closes, the tool gathers every path that appeared i
|
|
|
289
296
|
emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
|
|
290
297
|
makes a silent drop visible without blocking, since the file is already gone.
|
|
291
298
|
|
|
299
|
+
### Abandoning a run
|
|
300
|
+
|
|
301
|
+
A run that will not be finished — stopped on purpose, or blocked for good — is ended with
|
|
302
|
+
`tdd run abandon --reason "<why>"` rather than `resume --unblock`. It takes the worktree's
|
|
303
|
+
live run, else its most recent blocked run. It sets the run's `outcome` to `abandoned`,
|
|
304
|
+
records the reason and who abandoned it (the account, and the executor identity as `run
|
|
305
|
+
start` resolves it; on a split runner the account is the calling agent's `SUDO_USER`),
|
|
306
|
+
records a `human_intervention`, and releases the worktree's baseline and advance claims.
|
|
307
|
+
`tdd fleet` stops listing the run, `tdd metrics` reports it as `abandoned` with an
|
|
308
|
+
`abandoned: {reason, by}` entry, and a new run can start at the same worktree path.
|
|
309
|
+
|
|
310
|
+
When the worktree is gone, abandon by id from any worktree of the repo: `tdd run abandon
|
|
311
|
+
--run <id> --reason "<why>"`. `--run` is accepted only for a run whose worktree no longer
|
|
312
|
+
exists or is the caller's own, so one agent cannot end another's live run.
|
|
313
|
+
|
|
314
|
+
Refusals, under `result.reason`, write nothing: `reason_required` (blank reason),
|
|
315
|
+
`no_run` (no live or blocked run here), `run_not_found`, `run_ended` (already complete or
|
|
316
|
+
abandoned), `worktree_exists` (`--run` names another worktree that still exists), and
|
|
317
|
+
`advance_in_flight` (a live `tdd advance` holds the worktree; a stale claim is released).
|
|
318
|
+
|
|
292
319
|
The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
|
|
293
320
|
so they are never seen by the gate.
|
|
294
321
|
|
|
@@ -323,7 +350,8 @@ cycle's RED path empirically, assigns cycle kinds, and authors the contract —
|
|
|
323
350
|
back to the **planning** process. It reports plan fidelity (declared vs delivered vs
|
|
324
351
|
skipped vs never-reached cycles, human interventions) and, per cycle: the target, suite
|
|
325
352
|
runs by phase, the first-run outcome against expectation, sensitivity checks, commits,
|
|
326
|
-
and integrity events.
|
|
353
|
+
and integrity events. A `## Time` section says where the run's time went: the wall clock,
|
|
354
|
+
suite time and its share, per phase and per cycle.
|
|
327
355
|
|
|
328
356
|
Every observable fact in it is projected from recorded events. The agent that did the
|
|
329
357
|
work cannot compose it — that is what makes it worth reading, and why the log is
|
|
@@ -348,7 +376,8 @@ rendered, never written. Judgement enters in exactly two ways:
|
|
|
348
376
|
narrative as the agent's opinion.
|
|
349
377
|
|
|
350
378
|
`tdd metrics` is the quantitative companion: attempts per cycle, RED-first violation
|
|
351
|
-
rate, fidelity, blockers, interventions
|
|
379
|
+
rate, fidelity, blockers, interventions, and suite time per phase against the run's wall
|
|
380
|
+
clock. Cross-plan aggregates are deliberately labelled
|
|
352
381
|
non-comparable — cycle difficulty varies too much — so compare runs of the same contract
|
|
353
382
|
only (e.g. the same plan executed by two models).
|
|
354
383
|
|
|
@@ -486,8 +515,9 @@ and is never reclassified as a pin.
|
|
|
486
515
|
| `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
|
|
487
516
|
| `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
|
|
488
517
|
| `tdd resume [--unblock --note]` | reconstruct position; human intervention |
|
|
518
|
+
| `tdd run abandon --reason <text> [--run <id>]` | end a run that will not be finished; human only |
|
|
489
519
|
| `tdd log render [--out]` | project the ledger into a friction log |
|
|
490
|
-
| `tdd metrics` | fidelity, attempts, violations, interventions |
|
|
520
|
+
| `tdd metrics` | fidelity, attempts, violations, interventions, time |
|
|
491
521
|
| `tdd fleet [--json]` | all active runs across every worktree; read-only |
|
|
492
522
|
|
|
493
523
|
## Scoped baseline capture (R9.5c)
|
|
@@ -670,6 +700,24 @@ One SQLite ledger **per repository**, in `~/.local/share/tdd-cli/` (override wit
|
|
|
670
700
|
from the current directory, never committed. `worktree_path` is a column, so concurrent runs
|
|
671
701
|
in separate worktrees are isolated without a pruned worktree orphaning its history.
|
|
672
702
|
|
|
703
|
+
That file is owned by whoever runs `tdd`. On a single-user machine that is the agent's own
|
|
704
|
+
account, so "outside the worktree" is all the separation there is, and `tdd doctor` reports
|
|
705
|
+
`mode: single` and says so. **Split mode** gives the ledger to a runner account the agent
|
|
706
|
+
cannot become:
|
|
707
|
+
|
|
708
|
+
- the agent's `tdd` forwards every verb except `docs` and `init` to the runner, through one
|
|
709
|
+
`sudoers` line;
|
|
710
|
+
- the runner keeps the ledger in a mode-700 directory of its own and ignores the caller's
|
|
711
|
+
`TDD_LEDGER_HOME`;
|
|
712
|
+
- suites, gates, hooks and git all run as the agent, in the agent's worktree;
|
|
713
|
+
- executor identity is `operator` when the runner's config assigns the calling account a
|
|
714
|
+
model, and `claimed` otherwise;
|
|
715
|
+
- `tdd runner import <ledger>` brings an existing ledger under the runner, marked as
|
|
716
|
+
pre-split history that `tdd metrics` reports.
|
|
717
|
+
|
|
718
|
+
`tdd docs split` ([docs/split-runner.md](docs/split-runner.md)) has the setup, the
|
|
719
|
+
verification checklist, and what split mode does not protect against.
|
|
720
|
+
|
|
673
721
|
## Enforcement boundary
|
|
674
722
|
|
|
675
723
|
The CLI cannot compel an agent — only the harness can.
|
|
@@ -20,8 +20,15 @@ stop silently mid-plan; and no record comparable across runs, plans, or models.
|
|
|
20
20
|
|
|
21
21
|
This tool removes the agent from the reporting path. It runs the suites itself, computes
|
|
22
22
|
every phase transition from what the tests observably did, and records the whole run in a
|
|
23
|
-
ledger the
|
|
24
|
-
|
|
23
|
+
ledger outside the worktree — which is what the friction logs and metrics at the end rest
|
|
24
|
+
on.
|
|
25
|
+
|
|
26
|
+
How far out of reach that ledger is depends on the machine. On a single-user install it
|
|
27
|
+
belongs to the same account that runs the agent: an agent that edits files in its checkout
|
|
28
|
+
cannot touch it, but one that goes looking for it can. In **split mode** a separate runner
|
|
29
|
+
account owns the ledger and runs the agent's suites *as the agent*, so the agent's account
|
|
30
|
+
cannot open it at all. `tdd doctor` says which mode a machine is in; `tdd docs split` sets
|
|
31
|
+
the second one up.
|
|
25
32
|
|
|
26
33
|
## Install
|
|
27
34
|
|
|
@@ -261,6 +268,26 @@ When the last declared cycle closes, the tool gathers every path that appeared i
|
|
|
261
268
|
emitted (visible in `tdd metrics` and the friction log) and the run completes. The event
|
|
262
269
|
makes a silent drop visible without blocking, since the file is already gone.
|
|
263
270
|
|
|
271
|
+
### Abandoning a run
|
|
272
|
+
|
|
273
|
+
A run that will not be finished — stopped on purpose, or blocked for good — is ended with
|
|
274
|
+
`tdd run abandon --reason "<why>"` rather than `resume --unblock`. It takes the worktree's
|
|
275
|
+
live run, else its most recent blocked run. It sets the run's `outcome` to `abandoned`,
|
|
276
|
+
records the reason and who abandoned it (the account, and the executor identity as `run
|
|
277
|
+
start` resolves it; on a split runner the account is the calling agent's `SUDO_USER`),
|
|
278
|
+
records a `human_intervention`, and releases the worktree's baseline and advance claims.
|
|
279
|
+
`tdd fleet` stops listing the run, `tdd metrics` reports it as `abandoned` with an
|
|
280
|
+
`abandoned: {reason, by}` entry, and a new run can start at the same worktree path.
|
|
281
|
+
|
|
282
|
+
When the worktree is gone, abandon by id from any worktree of the repo: `tdd run abandon
|
|
283
|
+
--run <id> --reason "<why>"`. `--run` is accepted only for a run whose worktree no longer
|
|
284
|
+
exists or is the caller's own, so one agent cannot end another's live run.
|
|
285
|
+
|
|
286
|
+
Refusals, under `result.reason`, write nothing: `reason_required` (blank reason),
|
|
287
|
+
`no_run` (no live or blocked run here), `run_not_found`, `run_ended` (already complete or
|
|
288
|
+
abandoned), `worktree_exists` (`--run` names another worktree that still exists), and
|
|
289
|
+
`advance_in_flight` (a live `tdd advance` holds the worktree; a stale claim is released).
|
|
290
|
+
|
|
264
291
|
The gate composes with `ancillary_files`: declared paths never fire `undeclared_file_touched`,
|
|
265
292
|
so they are never seen by the gate.
|
|
266
293
|
|
|
@@ -295,7 +322,8 @@ cycle's RED path empirically, assigns cycle kinds, and authors the contract —
|
|
|
295
322
|
back to the **planning** process. It reports plan fidelity (declared vs delivered vs
|
|
296
323
|
skipped vs never-reached cycles, human interventions) and, per cycle: the target, suite
|
|
297
324
|
runs by phase, the first-run outcome against expectation, sensitivity checks, commits,
|
|
298
|
-
and integrity events.
|
|
325
|
+
and integrity events. A `## Time` section says where the run's time went: the wall clock,
|
|
326
|
+
suite time and its share, per phase and per cycle.
|
|
299
327
|
|
|
300
328
|
Every observable fact in it is projected from recorded events. The agent that did the
|
|
301
329
|
work cannot compose it — that is what makes it worth reading, and why the log is
|
|
@@ -320,7 +348,8 @@ rendered, never written. Judgement enters in exactly two ways:
|
|
|
320
348
|
narrative as the agent's opinion.
|
|
321
349
|
|
|
322
350
|
`tdd metrics` is the quantitative companion: attempts per cycle, RED-first violation
|
|
323
|
-
rate, fidelity, blockers, interventions
|
|
351
|
+
rate, fidelity, blockers, interventions, and suite time per phase against the run's wall
|
|
352
|
+
clock. Cross-plan aggregates are deliberately labelled
|
|
324
353
|
non-comparable — cycle difficulty varies too much — so compare runs of the same contract
|
|
325
354
|
only (e.g. the same plan executed by two models).
|
|
326
355
|
|
|
@@ -458,8 +487,9 @@ and is never reclassified as a pin.
|
|
|
458
487
|
| `tdd note "<text>"` | attach a free-text narrative note to the current cycle or run |
|
|
459
488
|
| `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
|
|
460
489
|
| `tdd resume [--unblock --note]` | reconstruct position; human intervention |
|
|
490
|
+
| `tdd run abandon --reason <text> [--run <id>]` | end a run that will not be finished; human only |
|
|
461
491
|
| `tdd log render [--out]` | project the ledger into a friction log |
|
|
462
|
-
| `tdd metrics` | fidelity, attempts, violations, interventions |
|
|
492
|
+
| `tdd metrics` | fidelity, attempts, violations, interventions, time |
|
|
463
493
|
| `tdd fleet [--json]` | all active runs across every worktree; read-only |
|
|
464
494
|
|
|
465
495
|
## Scoped baseline capture (R9.5c)
|
|
@@ -642,6 +672,24 @@ One SQLite ledger **per repository**, in `~/.local/share/tdd-cli/` (override wit
|
|
|
642
672
|
from the current directory, never committed. `worktree_path` is a column, so concurrent runs
|
|
643
673
|
in separate worktrees are isolated without a pruned worktree orphaning its history.
|
|
644
674
|
|
|
675
|
+
That file is owned by whoever runs `tdd`. On a single-user machine that is the agent's own
|
|
676
|
+
account, so "outside the worktree" is all the separation there is, and `tdd doctor` reports
|
|
677
|
+
`mode: single` and says so. **Split mode** gives the ledger to a runner account the agent
|
|
678
|
+
cannot become:
|
|
679
|
+
|
|
680
|
+
- the agent's `tdd` forwards every verb except `docs` and `init` to the runner, through one
|
|
681
|
+
`sudoers` line;
|
|
682
|
+
- the runner keeps the ledger in a mode-700 directory of its own and ignores the caller's
|
|
683
|
+
`TDD_LEDGER_HOME`;
|
|
684
|
+
- suites, gates, hooks and git all run as the agent, in the agent's worktree;
|
|
685
|
+
- executor identity is `operator` when the runner's config assigns the calling account a
|
|
686
|
+
model, and `claimed` otherwise;
|
|
687
|
+
- `tdd runner import <ledger>` brings an existing ledger under the runner, marked as
|
|
688
|
+
pre-split history that `tdd metrics` reports.
|
|
689
|
+
|
|
690
|
+
`tdd docs split` ([docs/split-runner.md](docs/split-runner.md)) has the setup, the
|
|
691
|
+
verification checklist, and what split mode does not protect against.
|
|
692
|
+
|
|
645
693
|
## Enforcement boundary
|
|
646
694
|
|
|
647
695
|
The CLI cannot compel an agent — only the harness can.
|
|
@@ -16,6 +16,16 @@ Other properties worth knowing:
|
|
|
16
16
|
plain SQLite. It records test ids, file paths, diffs (for sensitivity
|
|
17
17
|
checks), and executor identity — treat it with the same sensitivity as the
|
|
18
18
|
repository itself.
|
|
19
|
+
- **On a single-user install the ledger is writable by the account that runs
|
|
20
|
+
the agent.** It has no triggers, hash chain or signature: an agent that looks
|
|
21
|
+
for it can edit it, and nothing records that it did. `tdd doctor` reports
|
|
22
|
+
this as `mode: single`. Where the record has to hold against the agent
|
|
23
|
+
itself, use split mode (`tdd docs split`): a separate runner account owns the
|
|
24
|
+
ledger, reached through one `sudoers` rule, and runs the agent's commands as
|
|
25
|
+
the agent. Split mode adds a privilege boundary to your machine; review the
|
|
26
|
+
two sudoers lines it needs as you would any other.
|
|
27
|
+
- Split mode does not make a test report trustworthy. Suites run as the agent,
|
|
28
|
+
and the runner parses what they write as data.
|
|
19
29
|
- `tdd fleet` opens the ledger read-only and never writes.
|
|
20
30
|
- Worker leases live in `~/.cache/tdd-cli/leases` and contain hostnames and
|
|
21
31
|
pids, nothing else.
|
|
@@ -134,6 +134,12 @@ visible at the moment it can still be fixed.
|
|
|
134
134
|
- **R5.2** Agents never supply executor identity by any path. Step 3 is a human affordance.
|
|
135
135
|
- **R5.3** Resolution requires the tool to run on the same host as the agent. Remote or CI
|
|
136
136
|
execution can set `TDD_EXECUTOR_MODEL` to declare the answer explicitly.
|
|
137
|
+
- **R5.3a** In split mode (R13.10) every source above is the agent's to write, and none of it
|
|
138
|
+
is in the runner's environment or home. The runner therefore records one of two things. If
|
|
139
|
+
its own config assigns the calling account (`SUDO_USER`) a model, that model with source
|
|
140
|
+
`operator`. Otherwise the ordinary resolution runs on the agent's side and its answer is
|
|
141
|
+
recorded with source `claimed`, so a comparison between models can exclude it. `unknown`
|
|
142
|
+
stays `unknown`. R5.2 holds across the boundary: no verb carries an identity.
|
|
137
143
|
|
|
138
144
|
### Cycle
|
|
139
145
|
| Field | Notes |
|
|
@@ -155,6 +161,7 @@ One execution of one project's test suite. **Never overwritten; append-only.**
|
|
|
155
161
|
| `target_failure_excerpt` | truncated |
|
|
156
162
|
| `total_passed`, `total_failed` | |
|
|
157
163
|
| `other_failures` | after baseline subtraction (§9.2) |
|
|
164
|
+
| `others_observed` | `1` when the run could see failures outside the target; `0` for a target-only run (R9.1a), whose `other_failures` is `[]` because nothing else ran, not because nothing else failed |
|
|
158
165
|
| `lint_outcome`, `typecheck_outcome` | per §9.4 |
|
|
159
166
|
| `duration_ms`, `started_at` | |
|
|
160
167
|
|
|
@@ -223,8 +230,8 @@ AWAITING_TEST ──advance──▶ AWAITING_IMPL ──advance──▶ AWAITI
|
|
|
223
230
|
|
|
224
231
|
| Phase | Agent's job | `advance` passes when |
|
|
225
232
|
|---|---|---|
|
|
226
|
-
| `AWAITING_TEST` | write exactly one failing test | target **fails
|
|
227
|
-
| `AWAITING_IMPL` | write the minimum implementation | target **passes**, no new failures elsewhere |
|
|
233
|
+
| `AWAITING_TEST` | write exactly one failing test | target **fails**; only the target runs (R9.1a) |
|
|
234
|
+
| `AWAITING_IMPL` | write the minimum implementation | target **passes**, no new failures elsewhere; the target runs first, and the whole suite only once it passes (R9.1b, R9.1e) |
|
|
228
235
|
| `AWAITING_REFACTOR` | refactor, or nothing | lint/typecheck clean and the close sweep green (R9.2). The cycle's own suites are **skipped** when the tree hash is unchanged since the GREEN commit — they just passed on an identical tree. The downstream sweep always runs, having not run yet. |
|
|
229
236
|
|
|
230
237
|
Outcomes that are not simple advancement:
|
|
@@ -343,8 +350,9 @@ Requirements:
|
|
|
343
350
|
- **R7.8** Generator tools that are not hand-edited during TDD — `codegen` being the motivating
|
|
344
351
|
case — are modelled as artifact regeneration commands, never as projects.
|
|
345
352
|
- **R7.12** A project declares its own suite command (`test_command`, and `collect_command` for
|
|
346
|
-
collection). Adapters append only reporting flags
|
|
347
|
-
|
|
353
|
+
collection). Adapters append only reporting flags — and, for a target-only run (R9.1a),
|
|
354
|
+
the selector that picks the target: parallelism, plugins and markers stay exactly as the
|
|
355
|
+
project declared, so the suite under TDD is the suite the team trusts.
|
|
348
356
|
- **R7.13** A project may declare **per-pattern suite overrides**: an alternate `test_command`
|
|
349
357
|
(plus optional `collect_command` and `env`) for files matching a root-relative pattern.
|
|
350
358
|
Collection and suite runs union the default suite with every override suite, so a cycle can
|
|
@@ -520,6 +528,7 @@ one has no move left but to re-run doctor and read the same output again.
|
|
|
520
528
|
| `tdd blocker --kind <k> --detail <text>` | record a typed blocker; set run `outcome = blocked`, releasing the stop hook (H1) |
|
|
521
529
|
| `tdd resume` | reconstruct position from the ledger and emit `next_action` |
|
|
522
530
|
| `tdd resume --unblock --note <text>` | **human only.** Reopen a blocked run, recording a `human_intervention` event with the note |
|
|
531
|
+
| `tdd run abandon --reason <text> [--run <id>]` | **human only.** End a live or blocked run that will not be finished: `outcome = abandoned`, an `abandonment` row (reason, account, executor), a `human_intervention` event, and the worktree's claims released. `--run` only for a run whose worktree is gone or is the caller's own |
|
|
523
532
|
|
|
524
533
|
- **R8.7** A blocked run is not live. H1 must permit the agent to stop, or a blocker traps it.
|
|
525
534
|
- **R8.8** `human_intervention` events are the input to interventions-per-run, the primary
|
|
@@ -567,9 +576,34 @@ For the passed-on-arrival case, which occurred in 4 of 8 executed cycles in the
|
|
|
567
576
|
## 9. Execution semantics
|
|
568
577
|
|
|
569
578
|
### 9.1 Regression scope
|
|
570
|
-
- **R9.1**
|
|
571
|
-
|
|
572
|
-
|
|
579
|
+
- **R9.1** A cycle's phases run in the **union of projects named by the cycle's targets**. For an
|
|
580
|
+
ordinary cycle that is one project; for a contract cycle (§9.3) it is all projects the cycle
|
|
581
|
+
spans. How much of each project's suite runs depends on the phase (R9.1a–R9.1c).
|
|
582
|
+
- **R9.1a** `AWAITING_TEST` and `AWAITING_PIN` run **only the target tests**, for every adapter
|
|
583
|
+
that can select a single test id. RED asks one question: does the named test fail? A collection
|
|
584
|
+
error in the target's own file is still caught, because it is reported for the target itself,
|
|
585
|
+
and it remains `not_collected` (§6.1). Failures elsewhere are **not observed** at RED: the
|
|
586
|
+
invocation records `others_observed = 0`, and RED is never blocked by failures outside the
|
|
587
|
+
cycle. A target adopted in place of the declared id (R8.9) is run in the same advance when
|
|
588
|
+
the target-only run did not execute it.
|
|
589
|
+
- **R9.1b** `AWAITING_IMPL` runs the **whole suite** of each project, for **every** adapter,
|
|
590
|
+
and reads the target outcome from that run. This is what makes R9.1a safe. A RED that breaks
|
|
591
|
+
a shared fixture or helper is caught at GREEN, and it is blamed on the same cycle, because the
|
|
592
|
+
GREEN invocation carries the cycle id. It is also what lets the close sweep skip the cycle's own
|
|
593
|
+
suites when the tree is unchanged since GREEN (§6.1).
|
|
594
|
+
- **R9.1c** `tdd sensitivity check` runs only the target tests. It asks whether the target fails
|
|
595
|
+
with the mutation in place, and nothing else.
|
|
596
|
+
- **R9.1d** Rationale: on a 33-second suite, running the whole suite at RED cost 16 minutes of a
|
|
597
|
+
72-minute run (#145), and the time grows with suite size. The guarantee that every phase sees
|
|
598
|
+
the whole suite is given up at RED only. GREEN keeps it, because a whole-suite run at GREEN is
|
|
599
|
+
what blames a regression on a cycle. Decided in #148.
|
|
600
|
+
- **R9.1e** Each `AWAITING_IMPL` advance runs **the targets alone first**. If any target does
|
|
601
|
+
not pass, the reply is `write_implementation` and nothing else runs: a still-failing target is
|
|
602
|
+
all a failed attempt has to report. Only when every target passes does the whole-suite run of
|
|
603
|
+
R9.1b follow, and it alone decides GREEN, so a passing GREEN proves exactly what R9.1b says. That
|
|
604
|
+
whole-suite run is the attempt's last invocation, which is the one the close sweep's skip
|
|
605
|
+
compares against (§6.1). A target that fails while something else also broke is reported first;
|
|
606
|
+
the other breakage is reported once the target passes. Decided in #149.
|
|
573
607
|
- **R9.2** On the transition out of `AWAITING_REFACTOR`, the close sweep runs: the cycle's own
|
|
574
608
|
projects, plus every project **downstream of an artifact the cycle modified** per the
|
|
575
609
|
`consumed_by` edges in §7.1, plus lint and typecheck for each. Projects with
|
|
@@ -645,6 +679,8 @@ For the passed-on-arrival case, which occurred in 4 of 8 executed cycles in the
|
|
|
645
679
|
wrong reused baseline is recoverable via `resume --unblock --accept-failures` (R9.5b),
|
|
646
680
|
which treats a reused baseline row as an ordinary row, for tests that fail at the start sha.
|
|
647
681
|
- **R9.6** Baseline failures are subtracted from `other_failures` in every subsequent invocation.
|
|
682
|
+
A target-only invocation (R9.1a) has nothing to subtract from: it records `others_observed = 0`,
|
|
683
|
+
and its empty `other_failures` must not be read as a clean run.
|
|
648
684
|
A baseline row is only ever captured from `run.start_sha`, never from the current tree.
|
|
649
685
|
- **R9.7** A baseline failure that starts passing is recorded, not ignored.
|
|
650
686
|
- **R9.5g** `run start` refuses a baseline whose failure ratio exceeds the implausibility threshold
|
|
@@ -835,14 +871,15 @@ Adapter.typecheck(project) -> GateResult
|
|
|
835
871
|
| Fact | Source |
|
|
836
872
|
|---|---|
|
|
837
873
|
| RED-first violation | invocation verdict in `AWAITING_TEST` |
|
|
838
|
-
| Implementation attempts | `COUNT(invocation)` where phase = `AWAITING_IMPL` |
|
|
874
|
+
| Implementation attempts | `COUNT(invocation)` where phase = `AWAITING_IMPL` and `others_observed = 0`: the target-only run every attempt starts with (R9.1e). A cycle recorded before R9.1e has none, and counts every `AWAITING_IMPL` invocation |
|
|
839
875
|
| Convergence vs thrash | trajectory of `other_failures` across attempts in a cycle |
|
|
840
876
|
| Implementation written during RED | staged-set classification at the RED commit (R9.14) — exact, both languages |
|
|
841
877
|
| Stub-only *content* during RED | Python: `ast` check that added function bodies are sentinels. TypeScript: line-count heuristic only. **The metric is labelled partial for TS rather than reported as uniform.** Secondary to R9.14 |
|
|
842
878
|
| Tests removed or weakened | `collect()` set diff + assertion-count diff, minus `modifies_tests` |
|
|
843
879
|
| Files outside declared blast radius | `git diff --name-only` vs contract |
|
|
844
880
|
| Commits per cycle | commit trailers (§13.3) |
|
|
845
|
-
| Suite duration
|
|
881
|
+
| Suite duration and wall clock | `invocation.duration_ms` summed per phase and per cycle; wall clock from `run.started_at`/`ended_at` (to now while live) and `cycle.opened_at`/`closed_at` — reported by `tdd metrics` (`time`) and the friction log's `## Time` section |
|
|
882
|
+
| Cost | not recorded |
|
|
846
883
|
| Plan fidelity | declared vs executed cycles and tests |
|
|
847
884
|
|
|
848
885
|
### 11.2 Contributed by the agent
|
|
@@ -905,6 +942,21 @@ depend on any of them being installed.
|
|
|
905
942
|
- **R13.5** Rationale: keying by worktree path orphans a run's entire history when the worktree is
|
|
906
943
|
pruned — a live condition in the motivating repository. A repo-level ledger also makes
|
|
907
944
|
cross-worktree metrics a `GROUP BY` rather than a merge.
|
|
945
|
+
- **R13.3a** The "per-user data directory" is the directory of whoever runs `tdd`. On a
|
|
946
|
+
single-user install that is the agent's own account: the ledger is outside the worktree and
|
|
947
|
+
no further out of reach than that, and `tdd doctor` must say so (`mode: single`, the
|
|
948
|
+
`ledger isolation` notice) rather than claim isolation.
|
|
949
|
+
- **R13.10** **Split mode.** A machine with `/etc/tdd-cli/runner.toml` has a *runner* account
|
|
950
|
+
that owns the ledger in a mode-700 directory and ignores the caller's `TDD_LEDGER_HOME`. Every
|
|
951
|
+
other account is a *client*: its `tdd` answers `docs` and `init` itself and forwards every
|
|
952
|
+
other verb, verbatim, to the runner through `sudo`. The runner executes every suite, gate,
|
|
953
|
+
hook and git command as the calling account and observes the results from outside; it never
|
|
954
|
+
acts as itself in the agent's territory, and with no caller it refuses to act. The config is
|
|
955
|
+
trusted only when owned by root or by the account reading it and not writable by others.
|
|
956
|
+
`tdd doctor` verifies, live and as the agent, that the uid drop works, that the agent cannot
|
|
957
|
+
read the ledger, and that it cannot write the install. `tdd runner import` brings an existing
|
|
958
|
+
ledger under the runner and marks it `pre_split_import`. Split mode is one machine's
|
|
959
|
+
arrangement; it adds no daemon, so R13.7 below stands.
|
|
908
960
|
|
|
909
961
|
### 13.3 Git integration
|
|
910
962
|
- **R13.6** Commits made during a run carry `TDD-Run`, `TDD-Cycle` and `TDD-Phase` trailers.
|
|
@@ -71,8 +71,8 @@ skill that duplicates any of it will fight the ledger.
|
|
|
71
71
|
|---|---|---|
|
|
72
72
|
| `write_test` | the cycle needs its test (or the test still fails to fail correctly) | write the declared target test — for a pin cycle, a characterisation test that passes on arrival — then `tdd advance` |
|
|
73
73
|
| `create_stub` | the target test cannot be collected: the module it imports does not exist | create the declared stub file(s) with no logic — the `detail` quotes the language's idiom — then `tdd advance` |
|
|
74
|
-
| `write_implementation` | RED is confirmed and committed | write the minimum code to pass the target, then `tdd advance` |
|
|
75
|
-
| `fix_regression` | tests outside the cycle are failing, or the close sweep / lint / typecheck gates failed | fix them without breaking the target, then `tdd advance` |
|
|
74
|
+
| `write_implementation` | RED is confirmed and committed, or the target still fails at GREEN (only the target ran) | write the minimum code to pass the target, then `tdd advance` |
|
|
75
|
+
| `fix_regression` | tests outside the cycle are failing at GREEN, or the close sweep / lint / typecheck gates failed | fix them without breaking the target, then `tdd advance` |
|
|
76
76
|
| `run_sensitivity_check` | a test passed where proof it *can* fail is required | `tdd sensitivity begin`, mutate the behaviour under test, `tdd sensitivity check`, then `tdd sensitivity end` |
|
|
77
77
|
| `name_target_test` | several new tests appeared; a cycle covers one behaviour | pick the intended one: `tdd target <id>` |
|
|
78
78
|
| `refactor_or_advance` | GREEN is committed, or the tool simply wants `tdd advance` next | refactor only if the plan calls for it, then `tdd advance` |
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
# Split mode: a ledger the agent cannot reach
|
|
2
|
+
|
|
3
|
+
On a single-user machine the ledger is a SQLite file owned by the account that runs
|
|
4
|
+
`tdd`, which is the account the agent runs as. The agent cannot reach it by editing its
|
|
5
|
+
worktree, but it can reach it by looking: `sqlite3 ~/.local/share/tdd-cli/*.sqlite3`
|
|
6
|
+
opens it, and nothing in the file says afterwards that anyone did. `tdd doctor` reports
|
|
7
|
+
this as `mode: single` and says so in its `ledger isolation` check.
|
|
8
|
+
|
|
9
|
+
Split mode moves the ledger, and the process that observes the tests, to a second
|
|
10
|
+
account:
|
|
11
|
+
|
|
12
|
+
- A **runner** account owns the ledger, in a directory only it can open (mode 700).
|
|
13
|
+
- The agent's `tdd` does nothing itself except `docs` and `init`. For every other verb
|
|
14
|
+
it asks the runner, through `sudo`.
|
|
15
|
+
- The runner runs every suite, gate, artifact hook and git command **as the agent**,
|
|
16
|
+
in the agent's worktree, and reads the results from outside. The agent's code never
|
|
17
|
+
runs as the account that owns the ledger.
|
|
18
|
+
|
|
19
|
+
The command surface is unchanged. No verb accepts a phase, a cycle number or an
|
|
20
|
+
identity on either side of the boundary.
|
|
21
|
+
|
|
22
|
+
This is one machine's arrangement. Each machine has its own runner and its own ledger.
|
|
23
|
+
|
|
24
|
+
## What it does and does not protect
|
|
25
|
+
|
|
26
|
+
It stops the agent's account from reading or writing the ledger, and from running code
|
|
27
|
+
as the account that can.
|
|
28
|
+
|
|
29
|
+
It does not stop:
|
|
30
|
+
|
|
31
|
+
- **A forged test report.** The suite runs as the agent, so a test can write whatever
|
|
32
|
+
report it likes. The runner parses reports as data and never executes them, but it
|
|
33
|
+
cannot know a passing report is honest. This is the same on a single-user machine.
|
|
34
|
+
- **A client-side downgrade.** An agent can point its own `tdd` at a config of its own
|
|
35
|
+
(`TDD_RUNNER_CONFIG`), or install a second copy of tdd-cli, and record a run in a
|
|
36
|
+
ledger it owns. Nothing is forged by that: the run is simply absent from the runner's
|
|
37
|
+
ledger, which is the only one to trust. Read runs from the runner's ledger.
|
|
38
|
+
|
|
39
|
+
## Setup
|
|
40
|
+
|
|
41
|
+
You need root for this. Examples use `tdd-runner` for the runner account and `agent1`
|
|
42
|
+
for the agent's.
|
|
43
|
+
|
|
44
|
+
### 1. The runner account
|
|
45
|
+
|
|
46
|
+
Create a system account with a home directory and no login shell you hand out.
|
|
47
|
+
On Linux: `useradd --system --create-home --shell /usr/sbin/nologin tdd-runner`.
|
|
48
|
+
On macOS: `sysadminctl -addUser tdd-runner -home /var/tdd-runner` (then hide it if you
|
|
49
|
+
wish).
|
|
50
|
+
|
|
51
|
+
### 2. Install tdd-cli where the agent cannot write
|
|
52
|
+
|
|
53
|
+
A runner whose code the agent can edit is the agent. Install for the runner as root:
|
|
54
|
+
|
|
55
|
+
python3 -m venv /opt/tdd-cli
|
|
56
|
+
/opt/tdd-cli/bin/pip install tdd-cli
|
|
57
|
+
|
|
58
|
+
It must be readable and executable by the agent's account: the runner runs small
|
|
59
|
+
helpers from this install *as the agent* (`python -m tddcli.agentfs`,
|
|
60
|
+
`python -m tddcli.identity`). It must not be writable by it.
|
|
61
|
+
|
|
62
|
+
The agent's own `tdd` can be this same binary or a separate install; both read the same
|
|
63
|
+
config and become a client.
|
|
64
|
+
|
|
65
|
+
### 3. sudoers
|
|
66
|
+
|
|
67
|
+
Two lines, in a file under `/etc/sudoers.d/` (edit with `visudo -f`):
|
|
68
|
+
|
|
69
|
+
agent1 ALL=(tdd-runner) NOPASSWD: /opt/tdd-cli/bin/tdd
|
|
70
|
+
tdd-runner ALL=(agent1) NOPASSWD:SETENV: ALL
|
|
71
|
+
|
|
72
|
+
The first lets the agent ask the runner, and only through that one binary. The second
|
|
73
|
+
lets the runner act as the agent. Add one pair per agent account.
|
|
74
|
+
|
|
75
|
+
Four things to get right:
|
|
76
|
+
|
|
77
|
+
- **`SETENV:` is required** on the second line. The runner passes the agent's own
|
|
78
|
+
environment to the agent's commands with `sudo -E`; without the tag sudo answers
|
|
79
|
+
"sorry, you are not allowed to preserve the environment" and `tdd doctor` fails its
|
|
80
|
+
`runs as the agent` check.
|
|
81
|
+
- **`NOPASSWD` is required on both.** `tdd` always passes `-n`, so a rule that would
|
|
82
|
+
prompt fails at once instead of hanging an unattended agent.
|
|
83
|
+
- **Leave `env_reset` on** (it is sudo's default). It is what strips the agent's
|
|
84
|
+
`TDD_LEDGER_HOME`, `TDD_RUNNER_CONFIG` and `PYTHONPATH` on the way to the runner.
|
|
85
|
+
- **Nothing to configure for `HOME` or the working directory.** The client passes `-H`,
|
|
86
|
+
because macOS sudo would otherwise leave `HOME` pointing at the agent's and the
|
|
87
|
+
runner's ledger home would follow it. sudo keeps the caller's working directory, which
|
|
88
|
+
is how the runner knows which worktree it was asked about.
|
|
89
|
+
|
|
90
|
+
### 4. The config file
|
|
91
|
+
|
|
92
|
+
`/etc/tdd-cli/runner.toml`, owned by root, mode 644:
|
|
93
|
+
|
|
94
|
+
[runner]
|
|
95
|
+
user = "tdd-runner"
|
|
96
|
+
command = "/opt/tdd-cli/bin/tdd" # the path the first sudoers line names
|
|
97
|
+
# sudo = "/usr/bin/sudo" # the default
|
|
98
|
+
# ledger_home = "/var/lib/tdd-cli" # default: ~tdd-runner/.local/share/tdd-cli
|
|
99
|
+
|
|
100
|
+
[executor] # optional, see "Executor identity"
|
|
101
|
+
agent1 = "claude-fable-5-1"
|
|
102
|
+
|
|
103
|
+
A process whose account is `user` is the runner. Every other process on the machine is
|
|
104
|
+
a client. With no file, the machine is single-user and nothing changes.
|
|
105
|
+
|
|
106
|
+
**The file is trusted only if it is owned by root or by the account reading it, and is
|
|
107
|
+
not group- or world-writable.** Anything else is refused. That is what stops an agent
|
|
108
|
+
handing the runner a config of its own: even if a sudoers rule let `TDD_RUNNER_CONFIG`
|
|
109
|
+
through, the file it named would belong to the agent, and the runner refuses it.
|
|
110
|
+
|
|
111
|
+
### 5. Let the runner read the worktree
|
|
112
|
+
|
|
113
|
+
The runner reads the worktree as itself: `tdd.toml`, the test files it globs for, JUnit
|
|
114
|
+
XML under `build/`. The agent's worktree, and the directories above it, must be readable
|
|
115
|
+
and traversable by the runner account (`chmod -R go+rX`, or a shared group). On macOS
|
|
116
|
+
that rules out `~/Desktop`, `~/Documents` and `~/Downloads`, which are mode 700.
|
|
117
|
+
|
|
118
|
+
The runner never writes there as itself. Commits, the friction log, sensitivity-check
|
|
119
|
+
restores and temporary probe worktrees are all made as the agent.
|
|
120
|
+
|
|
121
|
+
If a file is unreadable the runner says which, with `reason: "worktree_unreadable"`.
|
|
122
|
+
|
|
123
|
+
## Verify it
|
|
124
|
+
|
|
125
|
+
Run these as the agent's account, in a worktree:
|
|
126
|
+
|
|
127
|
+
1. `tdd doctor` reports `mode: "split"`, `healthy: true`, and these three checks `ok`:
|
|
128
|
+
- `runs as the agent`: `id -u`, asked through sudo, answered with the agent's uid.
|
|
129
|
+
- `ledger out of the agent's reach`: `test -r <ledger>`, as the agent, was refused.
|
|
130
|
+
- `install not writable by the agent`: `test -w <install>`, as the agent, was refused.
|
|
131
|
+
2. `cat ~tdd-runner/.local/share/tdd-cli/*.sqlite3` (or your `ledger_home`) is denied.
|
|
132
|
+
3. Start a run and close a cycle. `git log -1 --format='%an %cn'` shows the agent as
|
|
133
|
+
author and committer, and `ls -l` on the committed files shows the agent as owner.
|
|
134
|
+
4. `sudo -u tdd-runner /opt/tdd-cli/bin/tdd status`, run by hand from the agent's
|
|
135
|
+
account, works; the same command naming any other binary is denied.
|
|
136
|
+
|
|
137
|
+
The project's own test suite cannot do this for you. It runs without root, so it uses a
|
|
138
|
+
stand-in for sudo that does everything except change uid. The three doctor checks are
|
|
139
|
+
the only place the uid change itself is verified, and they run against your real sudo.
|
|
140
|
+
|
|
141
|
+
## Executor identity
|
|
142
|
+
|
|
143
|
+
`tdd run start` records which model did the work. Everything it reads on a single-user
|
|
144
|
+
machine (the session id, the transcript, `TDD_EXECUTOR_MODEL`) is the agent's to write,
|
|
145
|
+
so in split mode the runner trusts none of it:
|
|
146
|
+
|
|
147
|
+
- If the calling account is in `[executor]`, the run records that model with
|
|
148
|
+
`executor_source: "operator"`. One account per agent slot makes this exact.
|
|
149
|
+
- Otherwise the ordinary resolution runs on the agent's side and the run records what
|
|
150
|
+
it found with `executor_source: "claimed"`. Leave `claimed` runs out of any comparison
|
|
151
|
+
between models that has to be defended.
|
|
152
|
+
|
|
153
|
+
`--executor`, the human label, is a claim like any other in split mode.
|
|
154
|
+
|
|
155
|
+
## Existing ledgers
|
|
156
|
+
|
|
157
|
+
A ledger written before the split was reachable by the agent the whole time. The runner
|
|
158
|
+
can keep it, marked as such. From a root shell (so that sudo reports uid 0), or logged
|
|
159
|
+
in as the runner with no sudo at all:
|
|
160
|
+
|
|
161
|
+
sudo -u tdd-runner /opt/tdd-cli/bin/tdd runner import /home/agent1/.local/share/tdd-cli/<repo>.sqlite3
|
|
162
|
+
|
|
163
|
+
The runner must be able to read that file. It copies the ledger into the
|
|
164
|
+
runner's home under the same name, migrates it to the current schema, and records
|
|
165
|
+
`pre_split_import` with the source path, the time, and `last_run_id`. `tdd metrics`
|
|
166
|
+
reports that marker, so anyone comparing runs can see which predate the split.
|
|
167
|
+
|
|
168
|
+
It refuses when:
|
|
169
|
+
|
|
170
|
+
- the machine is not split, or the caller is not the runner (`not_split`);
|
|
171
|
+
- an agent's account is the caller (`operator_only`): an agent that could import could
|
|
172
|
+
hand the runner a history it wrote itself;
|
|
173
|
+
- the runner already holds a ledger of that name (`ledger_exists`). There is no merge.
|
|
174
|
+
The runner's copy may hold runs recorded out of the agent's reach, and a second import
|
|
175
|
+
would trade them for ones that were not.
|
|
176
|
+
|
|
177
|
+
Afterwards, remove the agent's copy so that nothing reads it by mistake.
|
|
178
|
+
|
|
179
|
+
## Fleet, leases and other details
|
|
180
|
+
|
|
181
|
+
- Worker leases move with the runner: they live under the runner's `~/.cache/tdd-cli`,
|
|
182
|
+
so one budget covers every agent account on the machine.
|
|
183
|
+
- `tdd fleet`, `tdd metrics` and `tdd log render` are forwarded like everything else
|
|
184
|
+
and read the runner's ledger.
|
|
185
|
+
- A runner invoked directly, with no sudo in between, acts for nobody: every verb
|
|
186
|
+
except `runner import` is refused with `reason: "no_agent"`. It never falls back to
|
|
187
|
+
running suites as itself.
|