tdd-cli 0.12.2__tar.gz → 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/CHANGELOG.md +58 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/PKG-INFO +33 -6
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/README.md +32 -5
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/SECURITY.md +10 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/docs/PRD.md +60 -9
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/docs/harness-integration.md +2 -2
- tdd_cli-0.13.0/docs/split-runner.md +187 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/pyproject.toml +1 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/__init__.py +1 -1
- tdd_cli-0.13.0/src/tddcli/actor.py +163 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/adapters/base.py +3 -12
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/adapters/cargo_adapter.py +4 -2
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/adapters/exec_adapter.py +4 -3
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/adapters/gradle_adapter.py +6 -3
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/adapters/pytest_adapter.py +26 -7
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/adapters/vitest_adapter.py +34 -4
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/adapters/xctest_adapter.py +4 -2
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/advance.py +52 -3
- tdd_cli-0.13.0/src/tddcli/agentfs.py +53 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/cli.py +194 -7
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/docs.py +5 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/gitutil.py +44 -25
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/identity.py +66 -2
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/ledger.py +85 -8
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/machine.py +15 -5
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/render.py +112 -4
- tdd_cli-0.13.0/src/tddcli/runner.py +110 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/snapshot.py +4 -5
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/conftest.py +91 -0
- tdd_cli-0.13.0/tests/test_actor_seams.py +44 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_cargo_adapter.py +16 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_close_sweep_gates.py +39 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_docs_command.py +8 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_exec_adapter.py +1 -1
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_gradle_adapter.py +17 -1
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_heartbeat.py +5 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_late_baseline.py +5 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_project_commands.py +4 -2
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_release_surface.py +15 -0
- tdd_cli-0.13.0/tests/test_runner_import.py +148 -0
- tdd_cli-0.13.0/tests/test_split_client.py +94 -0
- tdd_cli-0.13.0/tests/test_split_config.py +58 -0
- tdd_cli-0.13.0/tests/test_split_doctor.py +61 -0
- tdd_cli-0.13.0/tests/test_split_identity.py +81 -0
- tdd_cli-0.13.0/tests/test_split_ledger.py +30 -0
- tdd_cli-0.13.0/tests/test_split_runner.py +105 -0
- tdd_cli-0.13.0/tests/test_sudo_actor.py +70 -0
- tdd_cli-0.13.0/tests/test_suite_scope.py +261 -0
- tdd_cli-0.13.0/tests/test_target_only_runs.py +160 -0
- tdd_cli-0.13.0/tests/test_time_report.py +130 -0
- tdd_cli-0.13.0/tests/test_tree_hash.py +141 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_xctest_adapter.py +16 -1
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/.gitignore +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/LICENSE +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/examples/plan.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/config.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/contract.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/plan_paths.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/staging.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/src/tddcli/target_lint.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_advance_adoption.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_ancillary_files.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_artifact_regeneration.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_baseline_integrity.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_baseline_sanity.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_batch_collection.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_concurrent_advance.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_config_and_staging.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_contract.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_doctor_blockers.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_evidence_extraction.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_executor_attribution.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_executor_notes.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_failure_clipping.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_id_normalisation.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_named_leases_and_timeout.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_plan_paths.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_project_env.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_pytest_xdist_group.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_sensitivity_evidence.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_snapshot_and_identity.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_suite_overrides.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_target_lint.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_timing_visibility.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_undeclared_close_gate.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_undeclared_dedup.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_vitest_adapter.py +0 -0
- {tdd_cli-0.12.2 → tdd_cli-0.13.0}/tests/test_worker_leases.py +0 -0
|
@@ -6,6 +6,64 @@ and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.13.0] - 2026-09-23
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **Where a run's time went** (#147). `tdd metrics` gains a per-run `time` object:
|
|
14
|
+
`wall_clock_s`, `suite_s`, `suite_share` and `by_phase` (runs and suite seconds for
|
|
15
|
+
every phase recorded). A live run's wall clock runs to now. The friction log gains a
|
|
16
|
+
`## Time` section with the same split, a per-phase table and a per-cycle table
|
|
17
|
+
(wall clock, suite time and suite runs). Nothing new is recorded: these are the
|
|
18
|
+
ledger's existing timestamps and `invocation.duration_ms`.
|
|
19
|
+
- **Split mode: a ledger the agent's account cannot open** (#142). With
|
|
20
|
+
`/etc/tdd-cli/runner.toml` in place, a separate runner account owns the ledger
|
|
21
|
+
in a mode-700 directory. The agent's `tdd` forwards every verb except `docs`
|
|
22
|
+
and `init` to it through `sudo`, and the runner executes every suite, gate,
|
|
23
|
+
hook and git command as the agent. Executor identity is recorded as `operator`
|
|
24
|
+
when the runner's config assigns the calling account a model, and `claimed`
|
|
25
|
+
otherwise. `tdd runner import <ledger>` brings an existing ledger under the
|
|
26
|
+
runner, marked `pre_split_import`, which `tdd metrics` reports. Setup and a
|
|
27
|
+
verification checklist: `tdd docs split`.
|
|
28
|
+
- `tdd doctor` reports `mode` (`single` or `split`). In split mode it probes,
|
|
29
|
+
live and as the agent, that the uid drop works, that the agent cannot read the
|
|
30
|
+
ledger, and that it cannot write the install.
|
|
31
|
+
|
|
32
|
+
### Changed
|
|
33
|
+
|
|
34
|
+
- **GREEN runs the target first** (#149). An `AWAITING_IMPL` advance whose target
|
|
35
|
+
still fails now says so after running only the target. A passing target is followed
|
|
36
|
+
by the whole-suite run that decides GREEN, as before. `tdd metrics` counts
|
|
37
|
+
`impl_attempts` from each advance's target-only run only, and still counts every
|
|
38
|
+
`AWAITING_IMPL` run in cycles recorded before this change.
|
|
39
|
+
- `tdd doctor` on a single-user machine no longer lets "ledger outside
|
|
40
|
+
worktree" stand for isolation: a `ledger isolation` notice says the ledger is
|
|
41
|
+
owned by the same uid that runs the agent. The README and SECURITY.md say the
|
|
42
|
+
same. Single-user installs behave exactly as before.
|
|
43
|
+
- **RED runs only the target test** (#148). `AWAITING_TEST`, `AWAITING_PIN` and
|
|
44
|
+
`tdd sensitivity check` now run the target alone: pytest runs the owning suite's
|
|
45
|
+
collect command with the node id, and vitest filters to the file and an anchored,
|
|
46
|
+
escaped `-t` name. A failure elsewhere no longer blocks RED. It is caught at
|
|
47
|
+
GREEN, which runs the whole suite for **every** adapter. A collection error in
|
|
48
|
+
the target's own file is still `not_collected`. A target adopted in place of the
|
|
49
|
+
declared id is run in the same advance. The ledger is now schema 12:
|
|
50
|
+
`invocation.others_observed` is `0` for a target-only run.
|
|
51
|
+
|
|
52
|
+
### Fixed
|
|
53
|
+
|
|
54
|
+
- **GREEN ran only the target on cargo, gradle, xctest and exec** (#153), so a
|
|
55
|
+
regression elsewhere reached the close sweep unseen, and the sweep could skip the
|
|
56
|
+
cycle's own project without the whole suite ever running. These adapters now
|
|
57
|
+
narrow only when asked for a target-only run.
|
|
58
|
+
|
|
59
|
+
- **The close sweep re-ran a tree that had just passed** (#146): `gitutil.tree_hash` hashed git
|
|
60
|
+
state (index entries plus the unstaged diff), so the same content hashed differently before
|
|
61
|
+
and after the GREEN commit, and the §6.1 skip of a cycle's own suites never fired. It now
|
|
62
|
+
hashes working-tree content through a throwaway index, independent of what is staged or
|
|
63
|
+
committed. A skipped close sweep still runs the cycle project's lint/typecheck gates; before
|
|
64
|
+
this fix, the skip would have dropped them too. `--reuse-baselines` cache entries written by
|
|
65
|
+
earlier versions no longer match and are re-probed once.
|
|
66
|
+
|
|
9
67
|
## [0.12.2] - 2026-09-17
|
|
10
68
|
|
|
11
69
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.13.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -48,8 +48,15 @@ stop silently mid-plan; and no record comparable across runs, plans, or models.
|
|
|
48
48
|
|
|
49
49
|
This tool removes the agent from the reporting path. It runs the suites itself, computes
|
|
50
50
|
every phase transition from what the tests observably did, and records the whole run in a
|
|
51
|
-
ledger the
|
|
52
|
-
|
|
51
|
+
ledger outside the worktree — which is what the friction logs and metrics at the end rest
|
|
52
|
+
on.
|
|
53
|
+
|
|
54
|
+
How far out of reach that ledger is depends on the machine. On a single-user install it
|
|
55
|
+
belongs to the same account that runs the agent: an agent that edits files in its checkout
|
|
56
|
+
cannot touch it, but one that goes looking for it can. In **split mode** a separate runner
|
|
57
|
+
account owns the ledger and runs the agent's suites *as the agent*, so the agent's account
|
|
58
|
+
cannot open it at all. `tdd doctor` says which mode a machine is in; `tdd docs split` sets
|
|
59
|
+
the second one up.
|
|
53
60
|
|
|
54
61
|
## Install
|
|
55
62
|
|
|
@@ -323,7 +330,8 @@ cycle's RED path empirically, assigns cycle kinds, and authors the contract —
|
|
|
323
330
|
back to the **planning** process. It reports plan fidelity (declared vs delivered vs
|
|
324
331
|
skipped vs never-reached cycles, human interventions) and, per cycle: the target, suite
|
|
325
332
|
runs by phase, the first-run outcome against expectation, sensitivity checks, commits,
|
|
326
|
-
and integrity events.
|
|
333
|
+
and integrity events. A `## Time` section says where the run's time went: the wall clock,
|
|
334
|
+
suite time and its share, per phase and per cycle.
|
|
327
335
|
|
|
328
336
|
Every observable fact in it is projected from recorded events. The agent that did the
|
|
329
337
|
work cannot compose it — that is what makes it worth reading, and why the log is
|
|
@@ -348,7 +356,8 @@ rendered, never written. Judgement enters in exactly two ways:
|
|
|
348
356
|
narrative as the agent's opinion.
|
|
349
357
|
|
|
350
358
|
`tdd metrics` is the quantitative companion: attempts per cycle, RED-first violation
|
|
351
|
-
rate, fidelity, blockers, interventions
|
|
359
|
+
rate, fidelity, blockers, interventions, and suite time per phase against the run's wall
|
|
360
|
+
clock. Cross-plan aggregates are deliberately labelled
|
|
352
361
|
non-comparable — cycle difficulty varies too much — so compare runs of the same contract
|
|
353
362
|
only (e.g. the same plan executed by two models).
|
|
354
363
|
|
|
@@ -487,7 +496,7 @@ and is never reclassified as a pin.
|
|
|
487
496
|
| `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
|
|
488
497
|
| `tdd resume [--unblock --note]` | reconstruct position; human intervention |
|
|
489
498
|
| `tdd log render [--out]` | project the ledger into a friction log |
|
|
490
|
-
| `tdd metrics` | fidelity, attempts, violations, interventions |
|
|
499
|
+
| `tdd metrics` | fidelity, attempts, violations, interventions, time |
|
|
491
500
|
| `tdd fleet [--json]` | all active runs across every worktree; read-only |
|
|
492
501
|
|
|
493
502
|
## Scoped baseline capture (R9.5c)
|
|
@@ -670,6 +679,24 @@ One SQLite ledger **per repository**, in `~/.local/share/tdd-cli/` (override wit
|
|
|
670
679
|
from the current directory, never committed. `worktree_path` is a column, so concurrent runs
|
|
671
680
|
in separate worktrees are isolated without a pruned worktree orphaning its history.
|
|
672
681
|
|
|
682
|
+
That file is owned by whoever runs `tdd`. On a single-user machine that is the agent's own
|
|
683
|
+
account, so "outside the worktree" is all the separation there is, and `tdd doctor` reports
|
|
684
|
+
`mode: single` and says so. **Split mode** gives the ledger to a runner account the agent
|
|
685
|
+
cannot become:
|
|
686
|
+
|
|
687
|
+
- the agent's `tdd` forwards every verb except `docs` and `init` to the runner, through one
|
|
688
|
+
`sudoers` line;
|
|
689
|
+
- the runner keeps the ledger in a mode-700 directory of its own and ignores the caller's
|
|
690
|
+
`TDD_LEDGER_HOME`;
|
|
691
|
+
- suites, gates, hooks and git all run as the agent, in the agent's worktree;
|
|
692
|
+
- executor identity is `operator` when the runner's config assigns the calling account a
|
|
693
|
+
model, and `claimed` otherwise;
|
|
694
|
+
- `tdd runner import <ledger>` brings an existing ledger under the runner, marked as
|
|
695
|
+
pre-split history that `tdd metrics` reports.
|
|
696
|
+
|
|
697
|
+
`tdd docs split` ([docs/split-runner.md](docs/split-runner.md)) has the setup, the
|
|
698
|
+
verification checklist, and what split mode does not protect against.
|
|
699
|
+
|
|
673
700
|
## Enforcement boundary
|
|
674
701
|
|
|
675
702
|
The CLI cannot compel an agent — only the harness can.
|
|
@@ -20,8 +20,15 @@ stop silently mid-plan; and no record comparable across runs, plans, or models.
|
|
|
20
20
|
|
|
21
21
|
This tool removes the agent from the reporting path. It runs the suites itself, computes
|
|
22
22
|
every phase transition from what the tests observably did, and records the whole run in a
|
|
23
|
-
ledger the
|
|
24
|
-
|
|
23
|
+
ledger outside the worktree — which is what the friction logs and metrics at the end rest
|
|
24
|
+
on.
|
|
25
|
+
|
|
26
|
+
How far out of reach that ledger is depends on the machine. On a single-user install it
|
|
27
|
+
belongs to the same account that runs the agent: an agent that edits files in its checkout
|
|
28
|
+
cannot touch it, but one that goes looking for it can. In **split mode** a separate runner
|
|
29
|
+
account owns the ledger and runs the agent's suites *as the agent*, so the agent's account
|
|
30
|
+
cannot open it at all. `tdd doctor` says which mode a machine is in; `tdd docs split` sets
|
|
31
|
+
the second one up.
|
|
25
32
|
|
|
26
33
|
## Install
|
|
27
34
|
|
|
@@ -295,7 +302,8 @@ cycle's RED path empirically, assigns cycle kinds, and authors the contract —
|
|
|
295
302
|
back to the **planning** process. It reports plan fidelity (declared vs delivered vs
|
|
296
303
|
skipped vs never-reached cycles, human interventions) and, per cycle: the target, suite
|
|
297
304
|
runs by phase, the first-run outcome against expectation, sensitivity checks, commits,
|
|
298
|
-
and integrity events.
|
|
305
|
+
and integrity events. A `## Time` section says where the run's time went: the wall clock,
|
|
306
|
+
suite time and its share, per phase and per cycle.
|
|
299
307
|
|
|
300
308
|
Every observable fact in it is projected from recorded events. The agent that did the
|
|
301
309
|
work cannot compose it — that is what makes it worth reading, and why the log is
|
|
@@ -320,7 +328,8 @@ rendered, never written. Judgement enters in exactly two ways:
|
|
|
320
328
|
narrative as the agent's opinion.
|
|
321
329
|
|
|
322
330
|
`tdd metrics` is the quantitative companion: attempts per cycle, RED-first violation
|
|
323
|
-
rate, fidelity, blockers, interventions
|
|
331
|
+
rate, fidelity, blockers, interventions, and suite time per phase against the run's wall
|
|
332
|
+
clock. Cross-plan aggregates are deliberately labelled
|
|
324
333
|
non-comparable — cycle difficulty varies too much — so compare runs of the same contract
|
|
325
334
|
only (e.g. the same plan executed by two models).
|
|
326
335
|
|
|
@@ -459,7 +468,7 @@ and is never reclassified as a pin.
|
|
|
459
468
|
| `tdd blocker --kind --detail` | typed blocker; releases the stop hook |
|
|
460
469
|
| `tdd resume [--unblock --note]` | reconstruct position; human intervention |
|
|
461
470
|
| `tdd log render [--out]` | project the ledger into a friction log |
|
|
462
|
-
| `tdd metrics` | fidelity, attempts, violations, interventions |
|
|
471
|
+
| `tdd metrics` | fidelity, attempts, violations, interventions, time |
|
|
463
472
|
| `tdd fleet [--json]` | all active runs across every worktree; read-only |
|
|
464
473
|
|
|
465
474
|
## Scoped baseline capture (R9.5c)
|
|
@@ -642,6 +651,24 @@ One SQLite ledger **per repository**, in `~/.local/share/tdd-cli/` (override wit
|
|
|
642
651
|
from the current directory, never committed. `worktree_path` is a column, so concurrent runs
|
|
643
652
|
in separate worktrees are isolated without a pruned worktree orphaning its history.
|
|
644
653
|
|
|
654
|
+
That file is owned by whoever runs `tdd`. On a single-user machine that is the agent's own
|
|
655
|
+
account, so "outside the worktree" is all the separation there is, and `tdd doctor` reports
|
|
656
|
+
`mode: single` and says so. **Split mode** gives the ledger to a runner account the agent
|
|
657
|
+
cannot become:
|
|
658
|
+
|
|
659
|
+
- the agent's `tdd` forwards every verb except `docs` and `init` to the runner, through one
|
|
660
|
+
`sudoers` line;
|
|
661
|
+
- the runner keeps the ledger in a mode-700 directory of its own and ignores the caller's
|
|
662
|
+
`TDD_LEDGER_HOME`;
|
|
663
|
+
- suites, gates, hooks and git all run as the agent, in the agent's worktree;
|
|
664
|
+
- executor identity is `operator` when the runner's config assigns the calling account a
|
|
665
|
+
model, and `claimed` otherwise;
|
|
666
|
+
- `tdd runner import <ledger>` brings an existing ledger under the runner, marked as
|
|
667
|
+
pre-split history that `tdd metrics` reports.
|
|
668
|
+
|
|
669
|
+
`tdd docs split` ([docs/split-runner.md](docs/split-runner.md)) has the setup, the
|
|
670
|
+
verification checklist, and what split mode does not protect against.
|
|
671
|
+
|
|
645
672
|
## Enforcement boundary
|
|
646
673
|
|
|
647
674
|
The CLI cannot compel an agent — only the harness can.
|
|
@@ -16,6 +16,16 @@ Other properties worth knowing:
|
|
|
16
16
|
plain SQLite. It records test ids, file paths, diffs (for sensitivity
|
|
17
17
|
checks), and executor identity — treat it with the same sensitivity as the
|
|
18
18
|
repository itself.
|
|
19
|
+
- **On a single-user install the ledger is writable by the account that runs
|
|
20
|
+
the agent.** It has no triggers, hash chain or signature: an agent that looks
|
|
21
|
+
for it can edit it, and nothing records that it did. `tdd doctor` reports
|
|
22
|
+
this as `mode: single`. Where the record has to hold against the agent
|
|
23
|
+
itself, use split mode (`tdd docs split`): a separate runner account owns the
|
|
24
|
+
ledger, reached through one `sudoers` rule, and runs the agent's commands as
|
|
25
|
+
the agent. Split mode adds a privilege boundary to your machine; review the
|
|
26
|
+
two sudoers lines it needs as you would any other.
|
|
27
|
+
- Split mode does not make a test report trustworthy. Suites run as the agent,
|
|
28
|
+
and the runner parses what they write as data.
|
|
19
29
|
- `tdd fleet` opens the ledger read-only and never writes.
|
|
20
30
|
- Worker leases live in `~/.cache/tdd-cli/leases` and contain hostnames and
|
|
21
31
|
pids, nothing else.
|
|
@@ -134,6 +134,12 @@ visible at the moment it can still be fixed.
|
|
|
134
134
|
- **R5.2** Agents never supply executor identity by any path. Step 3 is a human affordance.
|
|
135
135
|
- **R5.3** Resolution requires the tool to run on the same host as the agent. Remote or CI
|
|
136
136
|
execution can set `TDD_EXECUTOR_MODEL` to declare the answer explicitly.
|
|
137
|
+
- **R5.3a** In split mode (R13.10) every source above is the agent's to write, and none of it
|
|
138
|
+
is in the runner's environment or home. The runner therefore records one of two things. If
|
|
139
|
+
its own config assigns the calling account (`SUDO_USER`) a model, that model with source
|
|
140
|
+
`operator`. Otherwise the ordinary resolution runs on the agent's side and its answer is
|
|
141
|
+
recorded with source `claimed`, so a comparison between models can exclude it. `unknown`
|
|
142
|
+
stays `unknown`. R5.2 holds across the boundary: no verb carries an identity.
|
|
137
143
|
|
|
138
144
|
### Cycle
|
|
139
145
|
| Field | Notes |
|
|
@@ -155,6 +161,7 @@ One execution of one project's test suite. **Never overwritten; append-only.**
|
|
|
155
161
|
| `target_failure_excerpt` | truncated |
|
|
156
162
|
| `total_passed`, `total_failed` | |
|
|
157
163
|
| `other_failures` | after baseline subtraction (§9.2) |
|
|
164
|
+
| `others_observed` | `1` when the run could see failures outside the target; `0` for a target-only run (R9.1a), whose `other_failures` is `[]` because nothing else ran, not because nothing else failed |
|
|
158
165
|
| `lint_outcome`, `typecheck_outcome` | per §9.4 |
|
|
159
166
|
| `duration_ms`, `started_at` | |
|
|
160
167
|
|
|
@@ -223,8 +230,8 @@ AWAITING_TEST ──advance──▶ AWAITING_IMPL ──advance──▶ AWAITI
|
|
|
223
230
|
|
|
224
231
|
| Phase | Agent's job | `advance` passes when |
|
|
225
232
|
|---|---|---|
|
|
226
|
-
| `AWAITING_TEST` | write exactly one failing test | target **fails
|
|
227
|
-
| `AWAITING_IMPL` | write the minimum implementation | target **passes**, no new failures elsewhere |
|
|
233
|
+
| `AWAITING_TEST` | write exactly one failing test | target **fails**; only the target runs (R9.1a) |
|
|
234
|
+
| `AWAITING_IMPL` | write the minimum implementation | target **passes**, no new failures elsewhere; the target runs first, and the whole suite only once it passes (R9.1b, R9.1e) |
|
|
228
235
|
| `AWAITING_REFACTOR` | refactor, or nothing | lint/typecheck clean and the close sweep green (R9.2). The cycle's own suites are **skipped** when the tree hash is unchanged since the GREEN commit — they just passed on an identical tree. The downstream sweep always runs, having not run yet. |
|
|
229
236
|
|
|
230
237
|
Outcomes that are not simple advancement:
|
|
@@ -343,8 +350,9 @@ Requirements:
|
|
|
343
350
|
- **R7.8** Generator tools that are not hand-edited during TDD — `codegen` being the motivating
|
|
344
351
|
case — are modelled as artifact regeneration commands, never as projects.
|
|
345
352
|
- **R7.12** A project declares its own suite command (`test_command`, and `collect_command` for
|
|
346
|
-
collection). Adapters append only reporting flags
|
|
347
|
-
|
|
353
|
+
collection). Adapters append only reporting flags — and, for a target-only run (R9.1a),
|
|
354
|
+
the selector that picks the target: parallelism, plugins and markers stay exactly as the
|
|
355
|
+
project declared, so the suite under TDD is the suite the team trusts.
|
|
348
356
|
- **R7.13** A project may declare **per-pattern suite overrides**: an alternate `test_command`
|
|
349
357
|
(plus optional `collect_command` and `env`) for files matching a root-relative pattern.
|
|
350
358
|
Collection and suite runs union the default suite with every override suite, so a cycle can
|
|
@@ -567,9 +575,34 @@ For the passed-on-arrival case, which occurred in 4 of 8 executed cycles in the
|
|
|
567
575
|
## 9. Execution semantics
|
|
568
576
|
|
|
569
577
|
### 9.1 Regression scope
|
|
570
|
-
- **R9.1**
|
|
571
|
-
|
|
572
|
-
|
|
578
|
+
- **R9.1** A cycle's phases run in the **union of projects named by the cycle's targets**. For an
|
|
579
|
+
ordinary cycle that is one project; for a contract cycle (§9.3) it is all projects the cycle
|
|
580
|
+
spans. How much of each project's suite runs depends on the phase (R9.1a–R9.1c).
|
|
581
|
+
- **R9.1a** `AWAITING_TEST` and `AWAITING_PIN` run **only the target tests**, for every adapter
|
|
582
|
+
that can select a single test id. RED asks one question: does the named test fail? A collection
|
|
583
|
+
error in the target's own file is still caught, because it is reported for the target itself,
|
|
584
|
+
and it remains `not_collected` (§6.1). Failures elsewhere are **not observed** at RED: the
|
|
585
|
+
invocation records `others_observed = 0`, and RED is never blocked by failures outside the
|
|
586
|
+
cycle. A target adopted in place of the declared id (R8.9) is run in the same advance when
|
|
587
|
+
the target-only run did not execute it.
|
|
588
|
+
- **R9.1b** `AWAITING_IMPL` runs the **whole suite** of each project, for **every** adapter,
|
|
589
|
+
and reads the target outcome from that run. This is what makes R9.1a safe. A RED that breaks
|
|
590
|
+
a shared fixture or helper is caught at GREEN, and it is blamed on the same cycle, because the
|
|
591
|
+
GREEN invocation carries the cycle id. It is also what lets the close sweep skip the cycle's own
|
|
592
|
+
suites when the tree is unchanged since GREEN (§6.1).
|
|
593
|
+
- **R9.1c** `tdd sensitivity check` runs only the target tests. It asks whether the target fails
|
|
594
|
+
with the mutation in place, and nothing else.
|
|
595
|
+
- **R9.1d** Rationale: on a 33-second suite, running the whole suite at RED cost 16 minutes of a
|
|
596
|
+
72-minute run (#145), and the time grows with suite size. The guarantee that every phase sees
|
|
597
|
+
the whole suite is given up at RED only. GREEN keeps it, because a whole-suite run at GREEN is
|
|
598
|
+
what blames a regression on a cycle. Decided in #148.
|
|
599
|
+
- **R9.1e** Each `AWAITING_IMPL` advance runs **the targets alone first**. If any target does
|
|
600
|
+
not pass, the reply is `write_implementation` and nothing else runs: a still-failing target is
|
|
601
|
+
all a failed attempt has to report. Only when every target passes does the whole-suite run of
|
|
602
|
+
R9.1b follow, and it alone decides GREEN, so a passing GREEN proves exactly what R9.1b says. That
|
|
603
|
+
whole-suite run is the attempt's last invocation, which is the one the close sweep's skip
|
|
604
|
+
compares against (§6.1). A target that fails while something else also broke is reported first;
|
|
605
|
+
the other breakage is reported once the target passes. Decided in #149.
|
|
573
606
|
- **R9.2** On the transition out of `AWAITING_REFACTOR`, the close sweep runs: the cycle's own
|
|
574
607
|
projects, plus every project **downstream of an artifact the cycle modified** per the
|
|
575
608
|
`consumed_by` edges in §7.1, plus lint and typecheck for each. Projects with
|
|
@@ -645,6 +678,8 @@ For the passed-on-arrival case, which occurred in 4 of 8 executed cycles in the
|
|
|
645
678
|
wrong reused baseline is recoverable via `resume --unblock --accept-failures` (R9.5b),
|
|
646
679
|
which treats a reused baseline row as an ordinary row, for tests that fail at the start sha.
|
|
647
680
|
- **R9.6** Baseline failures are subtracted from `other_failures` in every subsequent invocation.
|
|
681
|
+
A target-only invocation (R9.1a) has nothing to subtract from: it records `others_observed = 0`,
|
|
682
|
+
and its empty `other_failures` must not be read as a clean run.
|
|
648
683
|
A baseline row is only ever captured from `run.start_sha`, never from the current tree.
|
|
649
684
|
- **R9.7** A baseline failure that starts passing is recorded, not ignored.
|
|
650
685
|
- **R9.5g** `run start` refuses a baseline whose failure ratio exceeds the implausibility threshold
|
|
@@ -835,14 +870,15 @@ Adapter.typecheck(project) -> GateResult
|
|
|
835
870
|
| Fact | Source |
|
|
836
871
|
|---|---|
|
|
837
872
|
| RED-first violation | invocation verdict in `AWAITING_TEST` |
|
|
838
|
-
| Implementation attempts | `COUNT(invocation)` where phase = `AWAITING_IMPL` |
|
|
873
|
+
| Implementation attempts | `COUNT(invocation)` where phase = `AWAITING_IMPL` and `others_observed = 0`: the target-only run every attempt starts with (R9.1e). A cycle recorded before R9.1e has none, and counts every `AWAITING_IMPL` invocation |
|
|
839
874
|
| Convergence vs thrash | trajectory of `other_failures` across attempts in a cycle |
|
|
840
875
|
| Implementation written during RED | staged-set classification at the RED commit (R9.14) — exact, both languages |
|
|
841
876
|
| Stub-only *content* during RED | Python: `ast` check that added function bodies are sentinels. TypeScript: line-count heuristic only. **The metric is labelled partial for TS rather than reported as uniform.** Secondary to R9.14 |
|
|
842
877
|
| Tests removed or weakened | `collect()` set diff + assertion-count diff, minus `modifies_tests` |
|
|
843
878
|
| Files outside declared blast radius | `git diff --name-only` vs contract |
|
|
844
879
|
| Commits per cycle | commit trailers (§13.3) |
|
|
845
|
-
| Suite duration
|
|
880
|
+
| Suite duration and wall clock | `invocation.duration_ms` summed per phase and per cycle; wall clock from `run.started_at`/`ended_at` (to now while live) and `cycle.opened_at`/`closed_at` — reported by `tdd metrics` (`time`) and the friction log's `## Time` section |
|
|
881
|
+
| Cost | not recorded |
|
|
846
882
|
| Plan fidelity | declared vs executed cycles and tests |
|
|
847
883
|
|
|
848
884
|
### 11.2 Contributed by the agent
|
|
@@ -905,6 +941,21 @@ depend on any of them being installed.
|
|
|
905
941
|
- **R13.5** Rationale: keying by worktree path orphans a run's entire history when the worktree is
|
|
906
942
|
pruned — a live condition in the motivating repository. A repo-level ledger also makes
|
|
907
943
|
cross-worktree metrics a `GROUP BY` rather than a merge.
|
|
944
|
+
- **R13.3a** The "per-user data directory" is the directory of whoever runs `tdd`. On a
|
|
945
|
+
single-user install that is the agent's own account: the ledger is outside the worktree and
|
|
946
|
+
no further out of reach than that, and `tdd doctor` must say so (`mode: single`, the
|
|
947
|
+
`ledger isolation` notice) rather than claim isolation.
|
|
948
|
+
- **R13.10** **Split mode.** A machine with `/etc/tdd-cli/runner.toml` has a *runner* account
|
|
949
|
+
that owns the ledger in a mode-700 directory and ignores the caller's `TDD_LEDGER_HOME`. Every
|
|
950
|
+
other account is a *client*: its `tdd` answers `docs` and `init` itself and forwards every
|
|
951
|
+
other verb, verbatim, to the runner through `sudo`. The runner executes every suite, gate,
|
|
952
|
+
hook and git command as the calling account and observes the results from outside; it never
|
|
953
|
+
acts as itself in the agent's territory, and with no caller it refuses to act. The config is
|
|
954
|
+
trusted only when owned by root or by the account reading it and not writable by others.
|
|
955
|
+
`tdd doctor` verifies, live and as the agent, that the uid drop works, that the agent cannot
|
|
956
|
+
read the ledger, and that it cannot write the install. `tdd runner import` brings an existing
|
|
957
|
+
ledger under the runner and marks it `pre_split_import`. Split mode is one machine's
|
|
958
|
+
arrangement; it adds no daemon, so R13.7 below stands.
|
|
908
959
|
|
|
909
960
|
### 13.3 Git integration
|
|
910
961
|
- **R13.6** Commits made during a run carry `TDD-Run`, `TDD-Cycle` and `TDD-Phase` trailers.
|
|
@@ -71,8 +71,8 @@ skill that duplicates any of it will fight the ledger.
|
|
|
71
71
|
|---|---|---|
|
|
72
72
|
| `write_test` | the cycle needs its test (or the test still fails to fail correctly) | write the declared target test — for a pin cycle, a characterisation test that passes on arrival — then `tdd advance` |
|
|
73
73
|
| `create_stub` | the target test cannot be collected: the module it imports does not exist | create the declared stub file(s) with no logic — the `detail` quotes the language's idiom — then `tdd advance` |
|
|
74
|
-
| `write_implementation` | RED is confirmed and committed | write the minimum code to pass the target, then `tdd advance` |
|
|
75
|
-
| `fix_regression` | tests outside the cycle are failing, or the close sweep / lint / typecheck gates failed | fix them without breaking the target, then `tdd advance` |
|
|
74
|
+
| `write_implementation` | RED is confirmed and committed, or the target still fails at GREEN (only the target ran) | write the minimum code to pass the target, then `tdd advance` |
|
|
75
|
+
| `fix_regression` | tests outside the cycle are failing at GREEN, or the close sweep / lint / typecheck gates failed | fix them without breaking the target, then `tdd advance` |
|
|
76
76
|
| `run_sensitivity_check` | a test passed where proof it *can* fail is required | `tdd sensitivity begin`, mutate the behaviour under test, `tdd sensitivity check`, then `tdd sensitivity end` |
|
|
77
77
|
| `name_target_test` | several new tests appeared; a cycle covers one behaviour | pick the intended one: `tdd target <id>` |
|
|
78
78
|
| `refactor_or_advance` | GREEN is committed, or the tool simply wants `tdd advance` next | refactor only if the plan calls for it, then `tdd advance` |
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
# Split mode: a ledger the agent cannot reach
|
|
2
|
+
|
|
3
|
+
On a single-user machine the ledger is a SQLite file owned by the account that runs
|
|
4
|
+
`tdd`, which is the account the agent runs as. The agent cannot reach it by editing its
|
|
5
|
+
worktree, but it can reach it by looking: `sqlite3 ~/.local/share/tdd-cli/*.sqlite3`
|
|
6
|
+
opens it, and nothing in the file says afterwards that anyone did. `tdd doctor` reports
|
|
7
|
+
this as `mode: single` and says so in its `ledger isolation` check.
|
|
8
|
+
|
|
9
|
+
Split mode moves the ledger, and the process that observes the tests, to a second
|
|
10
|
+
account:
|
|
11
|
+
|
|
12
|
+
- A **runner** account owns the ledger, in a directory only it can open (mode 700).
|
|
13
|
+
- The agent's `tdd` does nothing itself except `docs` and `init`. For every other verb
|
|
14
|
+
it asks the runner, through `sudo`.
|
|
15
|
+
- The runner runs every suite, gate, artifact hook and git command **as the agent**,
|
|
16
|
+
in the agent's worktree, and reads the results from outside. The agent's code never
|
|
17
|
+
runs as the account that owns the ledger.
|
|
18
|
+
|
|
19
|
+
The command surface is unchanged. No verb accepts a phase, a cycle number or an
|
|
20
|
+
identity on either side of the boundary.
|
|
21
|
+
|
|
22
|
+
This is one machine's arrangement. Each machine has its own runner and its own ledger.
|
|
23
|
+
|
|
24
|
+
## What it does and does not protect
|
|
25
|
+
|
|
26
|
+
It stops the agent's account from reading or writing the ledger, and from running code
|
|
27
|
+
as the account that can.
|
|
28
|
+
|
|
29
|
+
It does not stop:
|
|
30
|
+
|
|
31
|
+
- **A forged test report.** The suite runs as the agent, so a test can write whatever
|
|
32
|
+
report it likes. The runner parses reports as data and never executes them, but it
|
|
33
|
+
cannot know a passing report is honest. This is the same on a single-user machine.
|
|
34
|
+
- **A client-side downgrade.** An agent can point its own `tdd` at a config of its own
|
|
35
|
+
(`TDD_RUNNER_CONFIG`), or install a second copy of tdd-cli, and record a run in a
|
|
36
|
+
ledger it owns. Nothing is forged by that: the run is simply absent from the runner's
|
|
37
|
+
ledger, which is the only one to trust. Read runs from the runner's ledger.
|
|
38
|
+
|
|
39
|
+
## Setup
|
|
40
|
+
|
|
41
|
+
You need root for this. Examples use `tdd-runner` for the runner account and `agent1`
|
|
42
|
+
for the agent's.
|
|
43
|
+
|
|
44
|
+
### 1. The runner account
|
|
45
|
+
|
|
46
|
+
Create a system account with a home directory and no login shell you hand out.
|
|
47
|
+
On Linux: `useradd --system --create-home --shell /usr/sbin/nologin tdd-runner`.
|
|
48
|
+
On macOS: `sysadminctl -addUser tdd-runner -home /var/tdd-runner` (then hide it if you
|
|
49
|
+
wish).
|
|
50
|
+
|
|
51
|
+
### 2. Install tdd-cli where the agent cannot write
|
|
52
|
+
|
|
53
|
+
A runner whose code the agent can edit is the agent. Install for the runner as root:
|
|
54
|
+
|
|
55
|
+
python3 -m venv /opt/tdd-cli
|
|
56
|
+
/opt/tdd-cli/bin/pip install tdd-cli
|
|
57
|
+
|
|
58
|
+
It must be readable and executable by the agent's account: the runner runs small
|
|
59
|
+
helpers from this install *as the agent* (`python -m tddcli.agentfs`,
|
|
60
|
+
`python -m tddcli.identity`). It must not be writable by it.
|
|
61
|
+
|
|
62
|
+
The agent's own `tdd` can be this same binary or a separate install; both read the same
|
|
63
|
+
config and become a client.
|
|
64
|
+
|
|
65
|
+
### 3. sudoers
|
|
66
|
+
|
|
67
|
+
Two lines, in a file under `/etc/sudoers.d/` (edit with `visudo -f`):
|
|
68
|
+
|
|
69
|
+
agent1 ALL=(tdd-runner) NOPASSWD: /opt/tdd-cli/bin/tdd
|
|
70
|
+
tdd-runner ALL=(agent1) NOPASSWD:SETENV: ALL
|
|
71
|
+
|
|
72
|
+
The first lets the agent ask the runner, and only through that one binary. The second
|
|
73
|
+
lets the runner act as the agent. Add one pair per agent account.
|
|
74
|
+
|
|
75
|
+
Four things to get right:
|
|
76
|
+
|
|
77
|
+
- **`SETENV:` is required** on the second line. The runner passes the agent's own
|
|
78
|
+
environment to the agent's commands with `sudo -E`; without the tag sudo answers
|
|
79
|
+
"sorry, you are not allowed to preserve the environment" and `tdd doctor` fails its
|
|
80
|
+
`runs as the agent` check.
|
|
81
|
+
- **`NOPASSWD` is required on both.** `tdd` always passes `-n`, so a rule that would
|
|
82
|
+
prompt fails at once instead of hanging an unattended agent.
|
|
83
|
+
- **Leave `env_reset` on** (it is sudo's default). It is what strips the agent's
|
|
84
|
+
`TDD_LEDGER_HOME`, `TDD_RUNNER_CONFIG` and `PYTHONPATH` on the way to the runner.
|
|
85
|
+
- **Nothing to configure for `HOME` or the working directory.** The client passes `-H`,
|
|
86
|
+
because macOS sudo would otherwise leave `HOME` pointing at the agent's and the
|
|
87
|
+
runner's ledger home would follow it. sudo keeps the caller's working directory, which
|
|
88
|
+
is how the runner knows which worktree it was asked about.
|
|
89
|
+
|
|
90
|
+
### 4. The config file
|
|
91
|
+
|
|
92
|
+
`/etc/tdd-cli/runner.toml`, owned by root, mode 644:
|
|
93
|
+
|
|
94
|
+
[runner]
|
|
95
|
+
user = "tdd-runner"
|
|
96
|
+
command = "/opt/tdd-cli/bin/tdd" # the path the first sudoers line names
|
|
97
|
+
# sudo = "/usr/bin/sudo" # the default
|
|
98
|
+
# ledger_home = "/var/lib/tdd-cli" # default: ~tdd-runner/.local/share/tdd-cli
|
|
99
|
+
|
|
100
|
+
[executor] # optional, see "Executor identity"
|
|
101
|
+
agent1 = "claude-fable-5-1"
|
|
102
|
+
|
|
103
|
+
A process whose account is `user` is the runner. Every other process on the machine is
|
|
104
|
+
a client. With no file, the machine is single-user and nothing changes.
|
|
105
|
+
|
|
106
|
+
**The file is trusted only if it is owned by root or by the account reading it, and is
|
|
107
|
+
not group- or world-writable.** Anything else is refused. That is what stops an agent
|
|
108
|
+
handing the runner a config of its own: even if a sudoers rule let `TDD_RUNNER_CONFIG`
|
|
109
|
+
through, the file it named would belong to the agent, and the runner refuses it.
|
|
110
|
+
|
|
111
|
+
### 5. Let the runner read the worktree
|
|
112
|
+
|
|
113
|
+
The runner reads the worktree as itself: `tdd.toml`, the test files it globs for, JUnit
|
|
114
|
+
XML under `build/`. The agent's worktree, and the directories above it, must be readable
|
|
115
|
+
and traversable by the runner account (`chmod -R go+rX`, or a shared group). On macOS
|
|
116
|
+
that rules out `~/Desktop`, `~/Documents` and `~/Downloads`, which are mode 700.
|
|
117
|
+
|
|
118
|
+
The runner never writes there as itself. Commits, the friction log, sensitivity-check
|
|
119
|
+
restores and temporary probe worktrees are all made as the agent.
|
|
120
|
+
|
|
121
|
+
If a file is unreadable the runner says which, with `reason: "worktree_unreadable"`.
|
|
122
|
+
|
|
123
|
+
## Verify it
|
|
124
|
+
|
|
125
|
+
Run these as the agent's account, in a worktree:
|
|
126
|
+
|
|
127
|
+
1. `tdd doctor` reports `mode: "split"`, `healthy: true`, and these three checks `ok`:
|
|
128
|
+
- `runs as the agent`: `id -u`, asked through sudo, answered with the agent's uid.
|
|
129
|
+
- `ledger out of the agent's reach`: `test -r <ledger>`, as the agent, was refused.
|
|
130
|
+
- `install not writable by the agent`: `test -w <install>`, as the agent, was refused.
|
|
131
|
+
2. `cat ~tdd-runner/.local/share/tdd-cli/*.sqlite3` (or your `ledger_home`) is denied.
|
|
132
|
+
3. Start a run and close a cycle. `git log -1 --format='%an %cn'` shows the agent as
|
|
133
|
+
author and committer, and `ls -l` on the committed files shows the agent as owner.
|
|
134
|
+
4. `sudo -u tdd-runner /opt/tdd-cli/bin/tdd status`, run by hand from the agent's
|
|
135
|
+
account, works; the same command naming any other binary is denied.
|
|
136
|
+
|
|
137
|
+
The project's own test suite cannot do this for you. It runs without root, so it uses a
|
|
138
|
+
stand-in for sudo that does everything except change uid. The three doctor checks are
|
|
139
|
+
the only place the uid change itself is verified, and they run against your real sudo.
|
|
140
|
+
|
|
141
|
+
## Executor identity
|
|
142
|
+
|
|
143
|
+
`tdd run start` records which model did the work. Everything it reads on a single-user
|
|
144
|
+
machine (the session id, the transcript, `TDD_EXECUTOR_MODEL`) is the agent's to write,
|
|
145
|
+
so in split mode the runner trusts none of it:
|
|
146
|
+
|
|
147
|
+
- If the calling account is in `[executor]`, the run records that model with
|
|
148
|
+
`executor_source: "operator"`. One account per agent slot makes this exact.
|
|
149
|
+
- Otherwise the ordinary resolution runs on the agent's side and the run records what
|
|
150
|
+
it found with `executor_source: "claimed"`. Leave `claimed` runs out of any comparison
|
|
151
|
+
between models that has to be defended.
|
|
152
|
+
|
|
153
|
+
`--executor`, the human label, is a claim like any other in split mode.
|
|
154
|
+
|
|
155
|
+
## Existing ledgers
|
|
156
|
+
|
|
157
|
+
A ledger written before the split was reachable by the agent the whole time. The runner
|
|
158
|
+
can keep it, marked as such. From a root shell (so that sudo reports uid 0), or logged
|
|
159
|
+
in as the runner with no sudo at all:
|
|
160
|
+
|
|
161
|
+
sudo -u tdd-runner /opt/tdd-cli/bin/tdd runner import /home/agent1/.local/share/tdd-cli/<repo>.sqlite3
|
|
162
|
+
|
|
163
|
+
The runner must be able to read that file. It copies the ledger into the
|
|
164
|
+
runner's home under the same name, migrates it to the current schema, and records
|
|
165
|
+
`pre_split_import` with the source path, the time, and `last_run_id`. `tdd metrics`
|
|
166
|
+
reports that marker, so anyone comparing runs can see which predate the split.
|
|
167
|
+
|
|
168
|
+
It refuses when:
|
|
169
|
+
|
|
170
|
+
- the machine is not split, or the caller is not the runner (`not_split`);
|
|
171
|
+
- an agent's account is the caller (`operator_only`): an agent that could import could
|
|
172
|
+
hand the runner a history it wrote itself;
|
|
173
|
+
- the runner already holds a ledger of that name (`ledger_exists`). There is no merge.
|
|
174
|
+
The runner's copy may hold runs recorded out of the agent's reach, and a second import
|
|
175
|
+
would trade them for ones that were not.
|
|
176
|
+
|
|
177
|
+
Afterwards, remove the agent's copy so that nothing reads it by mistake.
|
|
178
|
+
|
|
179
|
+
## Fleet, leases and other details
|
|
180
|
+
|
|
181
|
+
- Worker leases move with the runner: they live under the runner's `~/.cache/tdd-cli`,
|
|
182
|
+
so one budget covers every agent account on the machine.
|
|
183
|
+
- `tdd fleet`, `tdd metrics` and `tdd log render` are forwarded like everything else
|
|
184
|
+
and read the runner's ledger.
|
|
185
|
+
- A runner invoked directly, with no sudo in between, acts for nobody: every verb
|
|
186
|
+
except `runner import` is refused with `reason: "no_agent"`. It never falls back to
|
|
187
|
+
running suites as itself.
|
|
@@ -49,6 +49,7 @@ packages = ["src/tddcli"]
|
|
|
49
49
|
# tests/test_docs_command.py fails if a topic is not included here.
|
|
50
50
|
[tool.hatch.build.targets.wheel.force-include]
|
|
51
51
|
"docs/harness-integration.md" = "tddcli/_docs/docs/harness-integration.md"
|
|
52
|
+
"docs/split-runner.md" = "tddcli/_docs/docs/split-runner.md"
|
|
52
53
|
"examples/plan.md" = "tddcli/_docs/examples/plan.md"
|
|
53
54
|
"examples/skills/tdd-drive/SKILL.md" = "tddcli/_docs/examples/skills/tdd-drive/SKILL.md"
|
|
54
55
|
"examples/skills/tdd-handoff/SKILL.md" = "tddcli/_docs/examples/skills/tdd-handoff/SKILL.md"
|