froid-loop 0.11.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- froid_loop/__init__.py +11 -0
- froid_loop/__main__.py +12 -0
- froid_loop/adapters/__init__.py +3 -0
- froid_loop/adapters/base.py +254 -0
- froid_loop/adapters/entrypoints.py +63 -0
- froid_loop/adapters/env_fault.py +290 -0
- froid_loop/adapters/generic.py +2013 -0
- froid_loop/adapters/mock.py +49 -0
- froid_loop/adapters/multiplexer.py +914 -0
- froid_loop/adapters/opencode_http.py +1687 -0
- froid_loop/adapters/profile.py +650 -0
- froid_loop/adapters/psmux_backend.py +1428 -0
- froid_loop/adapters/registry.py +322 -0
- froid_loop/adapters/tmux_backend.py +35 -0
- froid_loop/adapters/tmux_base.py +630 -0
- froid_loop/checks.py +187 -0
- froid_loop/cli.py +5041 -0
- froid_loop/data/__init__.py +0 -0
- froid_loop/data/froid_loop_hook.py +228 -0
- froid_loop/data/froid_loop_probe_hook.py +88 -0
- froid_loop/data/plugins/example/plugin.toml +21 -0
- froid_loop/data/plugins/tea/plugin.toml +184 -0
- froid_loop/data/plugins/tea/tea_plugin.py +258 -0
- froid_loop/data/plugins/unity/plugin.toml +140 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
- froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
- froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
- froid_loop/data/plugins/unity/unity_facts.md +17 -0
- froid_loop/data/plugins/unity/unity_plugin.py +415 -0
- froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
- froid_loop/data/plugins/unity/unity_ready.py +230 -0
- froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
- froid_loop/data/plugins/unity/unity_setup.py +551 -0
- froid_loop/data/plugins/unity/unity_teardown.py +362 -0
- froid_loop/data/profiles/antigravity.toml +52 -0
- froid_loop/data/profiles/claude.toml +85 -0
- froid_loop/data/profiles/codex.toml +22 -0
- froid_loop/data/profiles/copilot.toml +52 -0
- froid_loop/data/profiles/gemini.toml +26 -0
- froid_loop/data/profiles/opencode.toml +54 -0
- froid_loop/data/settings/core.toml +458 -0
- froid_loop/data/skills/README.md +93 -0
- froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
- froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
- froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
- froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
- froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
- froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
- froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
- froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
- froid_loop/decisions.py +202 -0
- froid_loop/deferredwork.py +2282 -0
- froid_loop/devcontract.py +892 -0
- froid_loop/diagnostics.py +1104 -0
- froid_loop/documents.py +532 -0
- froid_loop/engine.py +7732 -0
- froid_loop/envvars.py +111 -0
- froid_loop/escalation.py +225 -0
- froid_loop/events.py +266 -0
- froid_loop/fences.py +103 -0
- froid_loop/froidconfig.py +226 -0
- froid_loop/frontmatter.py +526 -0
- froid_loop/gates.py +133 -0
- froid_loop/install.py +2936 -0
- froid_loop/journal.py +178 -0
- froid_loop/machine.py +148 -0
- froid_loop/model.py +898 -0
- froid_loop/operatoractions.py +474 -0
- froid_loop/platform_util.py +1490 -0
- froid_loop/plugins/__init__.py +64 -0
- froid_loop/plugins/bus.py +259 -0
- froid_loop/plugins/context.py +319 -0
- froid_loop/plugins/loader.py +145 -0
- froid_loop/plugins/manifest.py +279 -0
- froid_loop/plugins/model.py +296 -0
- froid_loop/plugins/registry.py +245 -0
- froid_loop/plugins/trust.py +75 -0
- froid_loop/policy.py +1569 -0
- froid_loop/probe.py +1044 -0
- froid_loop/process_host.py +408 -0
- froid_loop/recovery_flow.py +1561 -0
- froid_loop/resolve.py +283 -0
- froid_loop/runs.py +4715 -0
- froid_loop/runsetup.py +1293 -0
- froid_loop/sanitize.py +593 -0
- froid_loop/settings_schema.py +276 -0
- froid_loop/signals.py +160 -0
- froid_loop/sprintstatus.py +609 -0
- froid_loop/statemachine.py +57 -0
- froid_loop/stories.py +615 -0
- froid_loop/stories_engine.py +796 -0
- froid_loop/sweep.py +1892 -0
- froid_loop/tokens.py +196 -0
- froid_loop/tui/__init__.py +11 -0
- froid_loop/tui/app.py +1584 -0
- froid_loop/tui/data.py +840 -0
- froid_loop/tui/launch.py +1003 -0
- froid_loop/tui/screens/__init__.py +1 -0
- froid_loop/tui/screens/dashboard.py +1071 -0
- froid_loop/tui/screens/modals.py +943 -0
- froid_loop/tui/screens/settings_screen.py +477 -0
- froid_loop/tui/settings.py +135 -0
- froid_loop/tui/widgets.py +981 -0
- froid_loop/verify.py +4545 -0
- froid_loop/workspace.py +320 -0
- froid_loop/worktree_flow.py +2301 -0
- froid_loop-0.11.1.dist-info/METADATA +728 -0
- froid_loop-0.11.1.dist-info/RECORD +116 -0
- froid_loop-0.11.1.dist-info/WHEEL +4 -0
- froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
- froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/runsetup.py
ADDED
|
@@ -0,0 +1,1293 @@
|
|
|
1
|
+
"""Run-composition layer for the CLI's ``run`` callback.
|
|
2
|
+
|
|
3
|
+
``cli.cmd_run`` used to build the :class:`~froid_loop.model.RunState`, wire the
|
|
4
|
+
:class:`~froid_loop.engine.Engine`, and stand up the coding-CLI adapters inline in
|
|
5
|
+
an argparse callback — logic that could only be exercised by round-tripping
|
|
6
|
+
through argv. This module lifts those pieces out as typed functions so a non-CLI
|
|
7
|
+
frontend (or a test) can compose a run directly:
|
|
8
|
+
|
|
9
|
+
* :func:`make_adapters` — the per-role adapter factory.
|
|
10
|
+
* :func:`platform_preflight` — the multiplexer/process-host readiness probe
|
|
11
|
+
``cmd_validate`` reports.
|
|
12
|
+
* :func:`build_run_state` / :func:`compose_run` — the RunState + Engine wiring
|
|
13
|
+
for ``cmd_run``.
|
|
14
|
+
* :func:`compose_sweep` — the same wiring for a ``sweep`` run (``cmd_sweep`` and
|
|
15
|
+
the auto-triggered child-sweep factory).
|
|
16
|
+
* :func:`compose_resume` — rebuilds the engine for a paused/interrupted run
|
|
17
|
+
(``cmd_resume`` and ``resolve``'s re-arm), selecting the sweep/stories/plain
|
|
18
|
+
variant from persisted run state.
|
|
19
|
+
* :func:`config_digest` — the integrity pin over the agent-writable config that
|
|
20
|
+
reaches host code execution (issue #461 point 4).
|
|
21
|
+
|
|
22
|
+
The engine class and the adapter factory are *injected* into :func:`compose_run`
|
|
23
|
+
rather than referenced here directly: ``cli`` resolves ``Engine`` /
|
|
24
|
+
``StoriesEngine`` / ``_make_adapters`` from its own module namespace at call time,
|
|
25
|
+
so the test suite's ``monkeypatch.setattr(cli, "Engine", ...)`` (and friends)
|
|
26
|
+
still bites. ``cli`` re-exports :func:`make_adapters`, :func:`platform_preflight`,
|
|
27
|
+
:func:`mux_reason_label`, and :data:`ROLES` under their historical private names
|
|
28
|
+
so those seams stay importable and monkeypatchable from ``cli``.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import hashlib
|
|
34
|
+
import json
|
|
35
|
+
import sys
|
|
36
|
+
import time
|
|
37
|
+
from contextlib import suppress
|
|
38
|
+
from dataclasses import dataclass
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
from typing import TYPE_CHECKING, Protocol
|
|
41
|
+
|
|
42
|
+
from . import froidconfig
|
|
43
|
+
from . import policy as policy_mod
|
|
44
|
+
from . import runs
|
|
45
|
+
from .checks import Finding
|
|
46
|
+
from .journal import Journal, save_state
|
|
47
|
+
from .model import RunState
|
|
48
|
+
from .platform_util import atomic_replace, is_wsl_unc_path
|
|
49
|
+
from .runs import RUNS_DIR
|
|
50
|
+
|
|
51
|
+
if TYPE_CHECKING:
|
|
52
|
+
from collections.abc import Callable
|
|
53
|
+
|
|
54
|
+
from .adapters.base import CodingCLIAdapter
|
|
55
|
+
from .adapters.profile import CLIProfile
|
|
56
|
+
from .engine import Engine, SweepFactory
|
|
57
|
+
from .policy import Policy
|
|
58
|
+
from .stories_engine import StoriesEngine
|
|
59
|
+
from .sweep import SweepEngine
|
|
60
|
+
|
|
61
|
+
class MakeAdapters(Protocol):
|
|
62
|
+
"""Call shape of :func:`make_adapters`, which ``compose_*`` takes injected.
|
|
63
|
+
|
|
64
|
+
Spelled as a Protocol rather than a ``Callable`` alias only so the
|
|
65
|
+
keyword-only ``profiles`` freeze below is part of the injected contract:
|
|
66
|
+
a frontend that supplies its own factory has to accept the pre-resolved
|
|
67
|
+
profiles, or silently re-read them from disk."""
|
|
68
|
+
|
|
69
|
+
def __call__(
|
|
70
|
+
self,
|
|
71
|
+
project: Path,
|
|
72
|
+
run_dir: Path,
|
|
73
|
+
policy: Policy,
|
|
74
|
+
*,
|
|
75
|
+
profiles: dict[str, CLIProfile] | None = None,
|
|
76
|
+
) -> dict[str, CodingCLIAdapter]: ...
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
# The three adapter roles a run wires. Defined here (the composition layer that
|
|
80
|
+
# actually builds them) and re-exported as ``cli.ROLES``, which `cmd_validate`
|
|
81
|
+
# and the test suite resolve.
|
|
82
|
+
ROLES = ("dev", "review", "triage")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def resolve_profiles(policy: Policy, project: Path) -> dict[str, CLIProfile]:
|
|
86
|
+
"""Resolve every role's :class:`CLIProfile` from disk **once**, as a mapping
|
|
87
|
+
the caller can then hand to both :func:`config_digest` and
|
|
88
|
+
:func:`make_adapters` so the two agree on the same bytes.
|
|
89
|
+
|
|
90
|
+
This exists for the child-sweep gate (#461 point 4). ``config_digest`` and
|
|
91
|
+
``make_adapters`` each used to call ``get_profile`` on their own, which made
|
|
92
|
+
the gate a check-then-use over two *separate* reads of an agent-writable
|
|
93
|
+
file: a session that leaves a background writer flipping
|
|
94
|
+
``.froid-loop/profiles/*.toml`` between a benign and a hostile copy needs only
|
|
95
|
+
the digest's read to catch the benign one and the adapter's read to catch the
|
|
96
|
+
other. That race is cheap to repeat — a lost round raises
|
|
97
|
+
`sweep-auto-not-started`, which `_maybe_auto_sweep` swallows, so the parent
|
|
98
|
+
runs on and the next epic boundary deals a fresh hand — so "narrow window" is
|
|
99
|
+
not a defense. The repeat is the writer's, never the orchestrator's: #501
|
|
100
|
+
leaves a refused trigger unspent, but nothing re-asks that trigger (see
|
|
101
|
+
`_maybe_auto_sweep`'s docstring), and under ``[sweep] auto = "run-end"`` a run
|
|
102
|
+
has exactly one. It is `per-epic` that hands out the further rounds.
|
|
103
|
+
Resolving once and threading the result removes the second read rather than
|
|
104
|
+
shrinking the window.
|
|
105
|
+
|
|
106
|
+
``cmd_run`` and ``_resume_paused_run`` thread it too, for a DIFFERENT reason —
|
|
107
|
+
they stamp a baseline rather than compare against one, and at launch the
|
|
108
|
+
on-disk config is the trust anchor, so no race there grants an attacker
|
|
109
|
+
anything a plain pre-launch write does not. What the second read cost them was
|
|
110
|
+
accuracy: the pin they mint is what every later auto-sweep is held to, so a pin
|
|
111
|
+
over bytes the run did not launch makes those children refuse the config the
|
|
112
|
+
parent has been running all along. See ``cli._launch_profiles``.
|
|
113
|
+
|
|
114
|
+
The policy half needs no equivalent: ``cli._sweep_factory`` already loads
|
|
115
|
+
``policy.toml`` once and passes that one frozen ``Policy`` to both the gate
|
|
116
|
+
and the composition. Profiles were the only surface read twice.
|
|
117
|
+
|
|
118
|
+
Deduplicated by profile name, so the common single-CLI policy touches disk
|
|
119
|
+
once rather than three times. ``ProfileError`` propagates.
|
|
120
|
+
"""
|
|
121
|
+
from .adapters.profile import get_profile
|
|
122
|
+
|
|
123
|
+
by_name: dict[str, CLIProfile] = {}
|
|
124
|
+
for role in ROLES:
|
|
125
|
+
name = policy.adapter.resolved(role).name
|
|
126
|
+
if name not in by_name:
|
|
127
|
+
by_name[name] = get_profile(name, project)
|
|
128
|
+
return {role: by_name[policy.adapter.resolved(role).name] for role in ROLES}
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def config_digest(
|
|
132
|
+
policy: Policy, project: Path, *, profiles: dict[str, CLIProfile] | None = None
|
|
133
|
+
) -> str:
|
|
134
|
+
"""sha256 over the agent-writable config that reaches **host** code execution.
|
|
135
|
+
|
|
136
|
+
The driven sessions can write anywhere under the project tree, including
|
|
137
|
+
``.froid-loop/policy.toml`` and ``.froid-loop/profiles/*.toml`` — so a session
|
|
138
|
+
can rewrite the commands ``verify`` runs (``shell=True``), the ``binary`` a
|
|
139
|
+
later session is launched from, or the ``[plugins] enabled`` allowlist that
|
|
140
|
+
gates in-process Python import (issue #461 point 4). A run freezes its
|
|
141
|
+
``Policy`` at launch, so the parent loop is already pinned; this digest exists
|
|
142
|
+
for the one path that re-reads config mid-run with **no human present** — the
|
|
143
|
+
auto-triggered child sweep in ``cli._sweep_factory``.
|
|
144
|
+
|
|
145
|
+
Field-scoped on purpose. A whole-file hash would also fire on the benign
|
|
146
|
+
``[limits]`` live-edits #189 documents as supported, so this covers exactly
|
|
147
|
+
the exec-reachable surface:
|
|
148
|
+
|
|
149
|
+
* ``verify.commands`` — order-preserved; they run in sequence.
|
|
150
|
+
* ``sorted(plugins.enabled)`` — set semantics, so order is not meaningful.
|
|
151
|
+
* per :data:`ROLES`, every field that decides **which program runs and with
|
|
152
|
+
what flags and environment**. That rule, not a hand-picked list, is what
|
|
153
|
+
keeps this complete: walk ``GenericAdapter.interactive_argv`` and
|
|
154
|
+
``interactive_env`` and every token there traces back to one of
|
|
155
|
+
``binary`` / ``launch_args`` / ``bypass_args`` / ``model_flag`` /
|
|
156
|
+
``prompt_template`` / ``env`` on the *resolved* profile, or to
|
|
157
|
+
``extra_args`` on the resolved adapter. The opencode-http builder reads a
|
|
158
|
+
strict SUBSET of those — ``_serve_argv`` takes ``binary`` and the adapter's
|
|
159
|
+
``extra_args`` and nothing else, and ``_session_env`` layers
|
|
160
|
+
``profile.env`` plus one *generated* variable, which the ``skill_tree``
|
|
161
|
+
bullet below accounts for. See the union paragraph on why the subset does
|
|
162
|
+
not narrow what is hashed.
|
|
163
|
+
* ``adapter`` — the field naming the adapter KIND, because it decides *which
|
|
164
|
+
argv builder runs at all*. ``make_adapters`` resolves it against the adapter
|
|
165
|
+
registry (``adapters/registry.py``), so rewriting it does not add a token: it
|
|
166
|
+
swaps the entire builder, and with it every rule the bullets above assume.
|
|
167
|
+
See the paragraph below on why a hard-coded token is not the same thing as a
|
|
168
|
+
safe one — that argument was written about ``hookless`` and transferred here
|
|
169
|
+
intact when the registry made ``adapter`` the selector.
|
|
170
|
+
* ``hookless`` — the transport. It no longer *selects* a builder, but it still
|
|
171
|
+
decides what the opencode builder emits, and it is what gates hook
|
|
172
|
+
registration; kept for the same wholesale-rewrite reason.
|
|
173
|
+
|
|
174
|
+
Three of those are easy to lose, and each was lost in an earlier cut of this
|
|
175
|
+
function — which is why the rule above is stated rather than the list.
|
|
176
|
+
``binary`` (and its siblings) live in ``profiles/*.toml`` and never appear in
|
|
177
|
+
``policy_snapshot``, so a snapshot-only compare is blind to them.
|
|
178
|
+
``adapter.extra_args`` *replaces* ``profile.bypass_args`` when set — it is the
|
|
179
|
+
field that carries ``--permission-mode bypassPermissions`` — so hashing the
|
|
180
|
+
profile default alone leaves the flags actually launched unpinned; ``None``
|
|
181
|
+
there means "fall back to the profile", a different state from ``()``, and the
|
|
182
|
+
two are kept apart. And ``prompt_template`` reads like prompt *payload* but is
|
|
183
|
+
not: ``interactive_argv`` places ``render_prompt(spec.prompt)`` in the argv
|
|
184
|
+
list, and the template need not contain ``{prompt}`` at all, so a rewritten
|
|
185
|
+
template is a verbatim attacker-chosen argv token. ``build_command``
|
|
186
|
+
``shlex.quote``\\ s it, which bounds it to ONE token — no word-splitting — but
|
|
187
|
+
one token is enough for the ``--opt=value`` form.
|
|
188
|
+
|
|
189
|
+
The builder selector was the fourth, and it was excluded here on a reading
|
|
190
|
+
that turned out to be wrong, so the correction is worth keeping — it is now
|
|
191
|
+
the argument for ``adapter``, since ``hookless`` selected the builder only
|
|
192
|
+
until the registry took that job over: *a hard-coded argv token is not the
|
|
193
|
+
same thing as a safe one.* Flipping ``hooks.dialect`` to ``"none"``
|
|
194
|
+
does not add a token — it swaps the whole builder, dropping ``launch_args``,
|
|
195
|
+
the prompt and the ``bypass_args`` fallback and putting the literal ``"serve"``
|
|
196
|
+
at argv[1], which ``_spawn_server`` then runs with ``cwd`` at the workspace
|
|
197
|
+
root. To a CLI that is a subcommand and a bad one dies in the health poll. To
|
|
198
|
+
an *interpreter* — a profile whose ``binary`` is ``python``/``sh``/``node``
|
|
199
|
+
with the real program in ``launch_args``, which nothing forbids — argv[1] is a
|
|
200
|
+
**script path resolved against the agent-writable tree**, and the exec happens
|
|
201
|
+
before the health poll it fails (three times: ``SPAWN_ATTEMPTS``). ``binary``
|
|
202
|
+
being pinned does not save it: the attacker inherits whichever binary the
|
|
203
|
+
project configured and only has to write a file named ``serve``. So the token
|
|
204
|
+
is a literal, and the argv is still attacker-controlled — walking the consumer
|
|
205
|
+
means asking what the *launched program* does with a token, not only where the
|
|
206
|
+
token came from.
|
|
207
|
+
|
|
208
|
+
The payload is the UNION of those fields across transports, not the subset
|
|
209
|
+
the role's builder actually reads, and that costs a known false positive:
|
|
210
|
+
``adapter.extra_args`` REPLACES ``bypass_args`` rather than extending it, so
|
|
211
|
+
for a role that sets it the hashed ``bypass_args`` is dead, and rewriting the
|
|
212
|
+
dead field alone moves this digest without moving one token of the launched
|
|
213
|
+
argv. Under ``hookless``, ``bypass_args`` / ``launch_args`` / ``model_flag``
|
|
214
|
+
are dead the same way. Hashing the effective projection instead means
|
|
215
|
+
restating two builders' precedence rules inside the control that polices
|
|
216
|
+
them, where drift is silent and lands in the UNDER-covering direction — the
|
|
217
|
+
failure this function has already made four times by reasoning from one
|
|
218
|
+
builder. Over-coverage fails the other way, loudly: ``sweep-auto-not-started``
|
|
219
|
+
+ notify, with the message naming ``froid-loop sweep`` as the human-present
|
|
220
|
+
path. Not free — #501 stopped a refusal from *spending* the trigger, but that
|
|
221
|
+
is honest bookkeeping rather than a reprieve, since the same wrong answer
|
|
222
|
+
refuses the next trigger too. It needs a writer, though, and nothing under
|
|
223
|
+
``src/`` writes ``.froid-loop/profiles/*.toml``
|
|
224
|
+
at all — that overlay is hand-authored, and the TUI settings screen writes
|
|
225
|
+
``policy.toml`` (``extra_args`` included). So a dead-field rewrite arriving
|
|
226
|
+
mid-run is a config change nobody automated made under a running loop, which
|
|
227
|
+
is the condition this gate reports rather than a false alarm to suppress.
|
|
228
|
+
|
|
229
|
+
That completeness rule — *walk the builder; every token traces back to a
|
|
230
|
+
hashed field* — is only available for a builder whose code is ours, and since
|
|
231
|
+
the adapter registry that is no longer guaranteed: an out-of-tree kind arrives
|
|
232
|
+
through the ``froid_loop.adapters`` entry point, and its field reads cannot be
|
|
233
|
+
walked from here. What the rule becomes for such a kind:
|
|
234
|
+
|
|
235
|
+
* The reads are still drawn from a CLOSED set even though the builder is open.
|
|
236
|
+
An adapter is constructed from its kwargs and nothing else — the resolved
|
|
237
|
+
``CLIProfile``, the frozen ``Policy``, and the per-role ``extra_args`` /
|
|
238
|
+
``usage_grace_s`` / ``stop_without_result_nudges`` — so there is no field an
|
|
239
|
+
external builder can invent. But ``Policy`` is WIDER than the launch surface
|
|
240
|
+
hashed above: an external builder that read, say, a ``[limits]`` knob into an
|
|
241
|
+
argv token would be reading a field the exclusions below drop on the grounds
|
|
242
|
+
that *the bundled builders* cannot turn it into one. That reasoning is
|
|
243
|
+
builder-scoped, so for an external kind it does not carry.
|
|
244
|
+
* ``adapter`` being hashed bounds what that costs. A session cannot swap in an
|
|
245
|
+
unpinned builder mid-run — naming a different kind moves this digest. It can
|
|
246
|
+
only rewrite fields of the kind the run already launched under, and which of
|
|
247
|
+
those that kind reads was decided by that kind's own package.
|
|
248
|
+
|
|
249
|
+
So: derived for a bundled kind; for an external kind this pins the selector
|
|
250
|
+
plus the bundled launch surface, and the remainder is that package's own trust
|
|
251
|
+
boundary — the same boundary an enabled plugin's ``[python]`` module already
|
|
252
|
+
sits behind (see the plugin gaps at the end), not something a wider hash here
|
|
253
|
+
could close.
|
|
254
|
+
|
|
255
|
+
Deliberately EXCLUDED:
|
|
256
|
+
|
|
257
|
+
* The *bytes* behind ``binary``/``launch_args`` — this pins the launch
|
|
258
|
+
target's SPELLING, not its content. A project-local target (``binary`` a
|
|
259
|
+
path into the tree, or ``python`` with the program in ``launch_args``) can
|
|
260
|
+
be rewritten in place with no config field moving. The gap is real and
|
|
261
|
+
unguarded: ``profile.py`` requires only a non-empty ``binary`` string, while
|
|
262
|
+
the three sibling path fields (``hooks.config_path``, ``skill_tree``,
|
|
263
|
+
``seed_files``) all reject absolute and parent refs.
|
|
264
|
+
|
|
265
|
+
NOT excluded on "the parent execs it too" — that defence is false for the
|
|
266
|
+
``triage`` role. Base ``Engine`` wires only dev+review; ``sweep.py`` holds
|
|
267
|
+
the only ``adapters["triage"]`` assignment and the only two ``role="triage"``
|
|
268
|
+
dispatches, so a ``[adapter.triage]`` profile override's target is exec'd by
|
|
269
|
+
a sweep and by nothing else. ``sweep.auto = "run-end"`` and worktree
|
|
270
|
+
isolation give two more shapes where the child is the uniquely exposed one.
|
|
271
|
+
|
|
272
|
+
Excluded because a hash cannot identify the target. Which ``launch_args``
|
|
273
|
+
token names a file is undecidable (``-i`` vs ``tools/agent.py``);
|
|
274
|
+
digest-time resolution is not the tmux shell's; and one indirection defeats
|
|
275
|
+
it — this repo's own ``write_script_launcher`` is a stub that execs an
|
|
276
|
+
interpreter on a sidecar, so hashing the stub misses the payload. Nor is
|
|
277
|
+
the target ours to pin: it is normally a third-party CLI that self-updates,
|
|
278
|
+
and a mid-run update would move a content hash and refuse every auto-sweep
|
|
279
|
+
for the life of the run (the digest is pinned in memory at launch, so
|
|
280
|
+
nothing on disk can re-bless it).
|
|
281
|
+
Confinement is the instrument, not hashing — and as a ``validate`` warning
|
|
282
|
+
rather than a refusal, since "resolves inside the project" does not decide
|
|
283
|
+
it either: under an active project venv ``which("python")`` IS
|
|
284
|
+
``<project>/.venv/bin/python``, and this repo's own zero-token E2E gate
|
|
285
|
+
configures ``binary`` at ``<sandbox>/.froid-loop/fake-cli.sh``. Tracked as
|
|
286
|
+
#500, to land with #499's fix option 3; unreachable on stock config (all six
|
|
287
|
+
shipped profiles are bare PATH names).
|
|
288
|
+
* ``hooks.config_path`` — the relay is issue #461's points 1-3, hardened on
|
|
289
|
+
its own track; folding it in would fire on an ordinary ``froid-loop init``.
|
|
290
|
+
* ``adapter.model`` — it cannot introduce an argv token, only fill the value
|
|
291
|
+
slot behind ``model_flag``, which IS pinned here. Pinning it would refuse an
|
|
292
|
+
auto-sweep after a human's mid-run model change in the TUI.
|
|
293
|
+
* ``skill_tree`` — the one profile field reaching a launched session's env
|
|
294
|
+
without passing through argv. For a hookless role ``_config_content`` plants
|
|
295
|
+
``cwd/skill_tree`` in that ``OPENCODE_CONFIG_CONTENT`` as ``skills.paths``,
|
|
296
|
+
so a rewritten tree points the unattended child sweep at instructions of the
|
|
297
|
+
writer's choosing. Excluded because the pointer is not the door — the
|
|
298
|
+
content is, and the content is reachable with no config change at all, so
|
|
299
|
+
there is no digest to move either way. The triage session that consumes
|
|
300
|
+
``froid-loop-sweep`` runs at ``workspace.root`` — the main checkout, never a
|
|
301
|
+
unit worktree, since ``sweep.py`` swaps the workspace only around bundle
|
|
302
|
+
execution — and a driven session can write that copy: worktrees mount under
|
|
303
|
+
``.froid-loop/runs/`` INSIDE the main checkout, and nothing confines a
|
|
304
|
+
session's writes to its cwd (this docstring's opening premise). Nothing
|
|
305
|
+
hashes or reseeds a skill in place: ``_copy_skills`` skips an existing skill
|
|
306
|
+
dir absent ``--force-skills``.
|
|
307
|
+
|
|
308
|
+
Be precise about the worktree case, because the obvious reading is wrong:
|
|
309
|
+
under ``isolation = "worktree"`` a session editing the skill tree *in its
|
|
310
|
+
own worktree* does NOT reach that child. A gitignored tree is absent from a
|
|
311
|
+
fresh ``git worktree add`` checkout, so ``provision_worktree`` seeds it from
|
|
312
|
+
the WHEEL (``resources.files("froid_loop.data")``) — the per-file no-clobber
|
|
313
|
+
that ``test_provision_worktree_does_not_clobber_existing_skill`` pins only
|
|
314
|
+
preserves a file the destination already has, which is the *tracked*
|
|
315
|
+
skill-tree case that test's own docstring names. The two copies then never
|
|
316
|
+
share bytes: the worktree-local exclude blocks staging and ``git merge``
|
|
317
|
+
moves only tracked content. The route above (absolute path into the main
|
|
318
|
+
checkout) is what carries this bullet, not in-worktree editing.
|
|
319
|
+
|
|
320
|
+
Skill-content integrity is a real question and not one a config hash can
|
|
321
|
+
answer.
|
|
322
|
+
* ``usage_parser`` — and with it the rest of the token-budget surface. It
|
|
323
|
+
selects a read-only tally over a transcript the orchestrator opens anyway
|
|
324
|
+
and decides no program, flag, or variable. Rewriting it to ``"none"`` DOES
|
|
325
|
+
make the mid-session guard inert (``tokens.read_usage`` returns None for
|
|
326
|
+
anything it does not recognize, so ``_sample_weighted_usage`` never trips
|
|
327
|
+
and the session drops out of accounting) — but the guard's own controls are
|
|
328
|
+
``limits.session_budget_mode`` / ``max_tokens_per_session`` /
|
|
329
|
+
``cache_read_weight``, which sit in the ``[limits]`` excluded above and
|
|
330
|
+
which the child sweep re-reads fresh from disk. ``session_budget_mode =
|
|
331
|
+
"off"`` silences the guard in one line and more completely (it never
|
|
332
|
+
samples at all), so pinning the parser bolts a side door in a wall whose
|
|
333
|
+
main door is held open on purpose. Admitting it also widens the rule from
|
|
334
|
+
the launch surface to the launch surface *plus safety knobs*, which is
|
|
335
|
+
``[limits]`` — the whole-file hash the second paragraph exists to reject.
|
|
336
|
+
* ``[plugins.<name>]`` settings — an enabled plugin's resolved settings do
|
|
337
|
+
reach exec (``bus.py`` exports each as ``FROID_LOOP_SETTING_*`` into a
|
|
338
|
+
``shell=True`` hook, and the Unity plugin turns one into an ``--editor-path``
|
|
339
|
+
argv token), but the TUI settings screen writes those same tables, so
|
|
340
|
+
pinning them would refuse an auto-sweep after a supported human edit. They
|
|
341
|
+
are inside an *already-enabled* plugin's blast radius — the trust boundary
|
|
342
|
+
the gap below is about — rather than a way past the allowlist.
|
|
343
|
+
|
|
344
|
+
Known gaps, tracked separately (#496, #497). This pins the plugin allowlist by
|
|
345
|
+
NAME only, and the allowlist is not the whole plugin exec surface:
|
|
346
|
+
|
|
347
|
+
* A project-origin ``.froid-loop/plugins/<name>/`` overrides a same-named
|
|
348
|
+
builtin (``plugins/loader.py`` overlay precedence) and
|
|
349
|
+
``trust.require_enabled`` gates on the name, so a session can swap the module
|
|
350
|
+
behind an *already* enabled plugin without moving this digest.
|
|
351
|
+
* Adding a plugin is caught only for one that declares ``[python]``. A
|
|
352
|
+
*declarative* manifest (no ``[python]``) loads on folder-drop by design
|
|
353
|
+
(``plugins/trust.py``) and ``registry.hooks_for`` hands its ``[hooks.<stage>]
|
|
354
|
+
cmd`` to the bus, which runs it with ``shell=True`` — with no ``enabled``
|
|
355
|
+
entry, and from a directory this digest never reads.
|
|
356
|
+
|
|
357
|
+
Closing either is a plugin-trust-model change rather than a wider hash, so
|
|
358
|
+
neither belongs here.
|
|
359
|
+
|
|
360
|
+
Tuples are normalized to lists before ``json.dumps(sort_keys=True)`` for the
|
|
361
|
+
same reason ``cli._resume_paused_run``'s ``policy_changed`` compare does it:
|
|
362
|
+
the live policy carries TUPLES where a persisted round-trip yields lists, and
|
|
363
|
+
a raw compare then reports "changed" every single time. ``ProfileError``
|
|
364
|
+
propagates — an unresolvable profile already aborts at :func:`make_adapters`.
|
|
365
|
+
|
|
366
|
+
``profiles`` is an already-resolved mapping from :func:`resolve_profiles`.
|
|
367
|
+
Pass it wherever the digest gates — or becomes the baseline for — something
|
|
368
|
+
that then *runs* under the same config, so the bytes hashed here are the bytes
|
|
369
|
+
launched rather than a second read of a file the sessions can rewrite in
|
|
370
|
+
between. Both halves of that rule have a caller: ``cli._sweep_factory`` gates,
|
|
371
|
+
``cmd_run``/``_resume_paused_run`` baseline. Omit it to resolve fresh, which
|
|
372
|
+
``cmd_sweep`` does deliberately — a human started that one, and the pin it
|
|
373
|
+
stamps gates no child."""
|
|
374
|
+
profiles = profiles if profiles is not None else resolve_profiles(policy, project)
|
|
375
|
+
|
|
376
|
+
launch: dict[str, dict[str, object]] = {}
|
|
377
|
+
for role in ROLES:
|
|
378
|
+
cfg = policy.adapter.resolved(role)
|
|
379
|
+
prof = profiles[role]
|
|
380
|
+
launch[role] = {
|
|
381
|
+
"binary": prof.binary,
|
|
382
|
+
"launch_args": list(prof.launch_args),
|
|
383
|
+
"bypass_args": list(prof.bypass_args),
|
|
384
|
+
"model_flag": prof.model_flag,
|
|
385
|
+
# An argv element, not prompt payload: render_prompt returns this
|
|
386
|
+
# template formatted, and it need not reference {prompt} at all.
|
|
387
|
+
"prompt_template": prof.prompt_template,
|
|
388
|
+
"env": dict(prof.env),
|
|
389
|
+
# THE builder selector: `make_adapters` resolves this against the
|
|
390
|
+
# adapter registry and the kind it names decides which argv builder
|
|
391
|
+
# runs at all. Rewriting it swaps the whole launch shape without
|
|
392
|
+
# moving one of the fields above.
|
|
393
|
+
"adapter": prof.adapter,
|
|
394
|
+
# The transport. It no longer selects the builder (`adapter` does),
|
|
395
|
+
# but it still rewrites what the opencode builder emits WHOLESALE
|
|
396
|
+
# rather than adding a token: hookless drops launch_args/prompt/
|
|
397
|
+
# bypass_args and substitutes `serve --port … --print-logs`, whose
|
|
398
|
+
# literal "serve" an interpreter binary reads as a cwd-relative
|
|
399
|
+
# script path.
|
|
400
|
+
"hookless": prof.hookless,
|
|
401
|
+
# None (inherit profile.bypass_args) is NOT the same state as () (an
|
|
402
|
+
# explicit override to no flags at all); json.dumps keeps them apart.
|
|
403
|
+
"extra_args": None if cfg.extra_args is None else list(cfg.extra_args),
|
|
404
|
+
}
|
|
405
|
+
payload = {
|
|
406
|
+
"verify_commands": list(policy.verify.commands),
|
|
407
|
+
"plugins_enabled": sorted(policy.plugins.enabled),
|
|
408
|
+
"profiles": launch,
|
|
409
|
+
}
|
|
410
|
+
return hashlib.sha256(json.dumps(payload, sort_keys=True).encode("utf-8")).hexdigest()
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def make_adapters(
|
|
414
|
+
project: Path,
|
|
415
|
+
run_dir: Path,
|
|
416
|
+
policy,
|
|
417
|
+
*,
|
|
418
|
+
profiles: dict[str, CLIProfile] | None = None,
|
|
419
|
+
) -> dict[str, CodingCLIAdapter]:
|
|
420
|
+
"""Build the per-role adapters. ``profiles`` is an already-resolved mapping
|
|
421
|
+
from :func:`resolve_profiles`; when given, no profile is re-read from disk, so
|
|
422
|
+
a caller that gated on :func:`config_digest` launches the *same* bytes it
|
|
423
|
+
validated (#461 point 4). Omitted, each role resolves fresh as before.
|
|
424
|
+
|
|
425
|
+
Also the single resolution point for this run's out-of-tree events directory
|
|
426
|
+
(#494), handed to every family it builds — see the ``events_dir`` note below."""
|
|
427
|
+
from .adapters.multiplexer import fold_version, get_multiplexer, mux_usable
|
|
428
|
+
from .adapters.profile import ProfileError, get_profile
|
|
429
|
+
from .adapters.registry import AdapterError, get_adapter_kind
|
|
430
|
+
|
|
431
|
+
# The dev skill (froid-build-auto) writes no result.json: its adapter
|
|
432
|
+
# synthesizes the result from the spec, and so needs the project paths to
|
|
433
|
+
# find that spec — rebasing onto the active worktree's implementation-
|
|
434
|
+
# artifacts dir under isolation, not just the main checkout's.
|
|
435
|
+
paths = froidconfig.load_paths(project)
|
|
436
|
+
mux = None
|
|
437
|
+
adapters: dict[str, CodingCLIAdapter] = {}
|
|
438
|
+
by_cfg: dict = {}
|
|
439
|
+
for role in ROLES:
|
|
440
|
+
cfg = policy.adapter.resolved(role)
|
|
441
|
+
# Both the dev and review sessions are now froid-build-auto runs (the review
|
|
442
|
+
# session re-invokes the dev skill on the done spec for a follow-up pass),
|
|
443
|
+
# and the skill writes no result.json — its adapter synthesizes the result
|
|
444
|
+
# from the spec it leaves on disk, so it needs the project paths to find
|
|
445
|
+
# that spec and cannot be shared with the triage role even on identical
|
|
446
|
+
# config. `synthesizes` is a froid-build-auto pipeline concept (which variant
|
|
447
|
+
# of a family to build + whether to thread `paths`), NOT a per-family
|
|
448
|
+
# branch — it stays a documented contract for every registered adapter.
|
|
449
|
+
# `policy.dev.skill` below is the stable adapter DISCRIMINATOR (see
|
|
450
|
+
# policy.DevPolicy), NOT the invoked name — it keeps the pre-rename spelling.
|
|
451
|
+
synthesizes = role in ("dev", "review") and policy.dev.skill == "froid-dev-auto"
|
|
452
|
+
key = (cfg, synthesizes)
|
|
453
|
+
if key not in by_cfg:
|
|
454
|
+
if profiles is not None:
|
|
455
|
+
profile = profiles[role]
|
|
456
|
+
else:
|
|
457
|
+
try:
|
|
458
|
+
profile = get_profile(cfg.name, project)
|
|
459
|
+
except ProfileError as e:
|
|
460
|
+
raise SystemExit(f"error: {e}") from e
|
|
461
|
+
# Which adapter class drives this CLI is pure data — `profile.adapter`
|
|
462
|
+
# resolved against the registry. No adapter-name branching lives here;
|
|
463
|
+
# a new family plugs in with zero edits to this function. Note this
|
|
464
|
+
# reads the profile RESOLVED ABOVE, so under the `profiles is not None`
|
|
465
|
+
# path the kind comes from the same bytes `config_digest` pinned (#461
|
|
466
|
+
# point 4) rather than a second read of a file a session can rewrite in
|
|
467
|
+
# between. An unknown kind fails loud naming the profile.
|
|
468
|
+
try:
|
|
469
|
+
kind = get_adapter_kind(profile.adapter)
|
|
470
|
+
except AdapterError as e:
|
|
471
|
+
raise SystemExit(f"error: profile {profile.name!r}: {e}") from e
|
|
472
|
+
# The load thunk is where a family's classes — and any optional
|
|
473
|
+
# dependency they pull in — are first imported, and it is deliberately
|
|
474
|
+
# never invoked by `validate` or `froid-loop adapters` (both stay free
|
|
475
|
+
# of heavy imports), so a thunk that raises has had no earlier gate.
|
|
476
|
+
# By here `compose_run` has already written the run state and pid. An
|
|
477
|
+
# escaping ImportError used to strand that run directory behind a
|
|
478
|
+
# traceback, recorded as an accepted consequence; it no longer does —
|
|
479
|
+
# both composers unwind the whole composition on any escape (see
|
|
480
|
+
# `_unwind_composition`), and this raise is one of the six SystemExits
|
|
481
|
+
# that path exists for. What that changes is the run dir, not the
|
|
482
|
+
# message: the narrowing below is a separate decision and still holds.
|
|
483
|
+
# ImportError ONLY, on the same rule as `construct_error` below: a
|
|
484
|
+
# missing dependency is a lazy loader's DECLARED failure, while
|
|
485
|
+
# anything else is a bug in that package and must surface as itself
|
|
486
|
+
# rather than as a misleading `error:` line. Widening this to
|
|
487
|
+
# `Exception` would contradict the pin two tests down.
|
|
488
|
+
try:
|
|
489
|
+
builder = kind.load()
|
|
490
|
+
except ImportError as e:
|
|
491
|
+
raise SystemExit(
|
|
492
|
+
f"error: profile {profile.name!r}: adapter kind "
|
|
493
|
+
f"{profile.adapter!r} failed to load: {type(e).__name__}: {e}"
|
|
494
|
+
) from e
|
|
495
|
+
# Annotated: the literal below would otherwise fix the value type to
|
|
496
|
+
# `Path | CLIProfile`, and the `needs_mux` arm adds a multiplexer.
|
|
497
|
+
common: dict[str, object] = dict(
|
|
498
|
+
run_dir=run_dir,
|
|
499
|
+
policy=policy,
|
|
500
|
+
profile=profile,
|
|
501
|
+
extra_args=cfg.extra_args,
|
|
502
|
+
usage_grace_s=cfg.usage_grace_s,
|
|
503
|
+
stop_without_result_nudges=cfg.stop_without_result_nudges,
|
|
504
|
+
# The run's out-of-tree hook-event channel (#494). Resolved HERE,
|
|
505
|
+
# from the `project` this function is handed, because it is the
|
|
506
|
+
# only layer that holds both halves of the key — the adapter sees
|
|
507
|
+
# a run dir and nothing else. Handed to every family rather than
|
|
508
|
+
# gated like `mux`: this is a description of the run, not a
|
|
509
|
+
# capability, and unlike resolving a multiplexer it costs no probe
|
|
510
|
+
# and can refuse no host. The engine derives the same value from
|
|
511
|
+
# the same two inputs for the producing side.
|
|
512
|
+
events_dir=runs.events_dir_for(project, run_dir.name),
|
|
513
|
+
)
|
|
514
|
+
if kind.needs_mux:
|
|
515
|
+
# Resolve and probe the shared multiplexer only when a kind
|
|
516
|
+
# actually drives one; a self-hosted HTTP/SSE family needs no
|
|
517
|
+
# transport (and a test asserts it is never even resolved).
|
|
518
|
+
if mux is None:
|
|
519
|
+
mux = get_multiplexer()
|
|
520
|
+
if not mux_usable(mux):
|
|
521
|
+
try:
|
|
522
|
+
version = fold_version(mux.version())
|
|
523
|
+
except Exception: # diagnosing must not mask the refusal
|
|
524
|
+
version = None
|
|
525
|
+
raise SystemExit(
|
|
526
|
+
f"error: multiplexer backend {type(mux).__name__} is not usable on "
|
|
527
|
+
f"this host (reported version: {version}); its transport binary is "
|
|
528
|
+
"missing, the version is unsupported, or a required helper is "
|
|
529
|
+
"absent (psmux needs `pwsh` on PATH); see `froid-loop diagnose`"
|
|
530
|
+
)
|
|
531
|
+
common["mux"] = mux
|
|
532
|
+
# The synthesizing variant additionally needs `paths`; the plain
|
|
533
|
+
# variant does not accept it. `construct_error` is family-declared —
|
|
534
|
+
# `()` for a family that cannot fail construction (generic), or e.g.
|
|
535
|
+
# `(OpencodeServerError,)` for one that can — and becomes a SystemExit
|
|
536
|
+
# so a run aborts with a clean message instead of a traceback.
|
|
537
|
+
# `except ():` catches nothing, which is exactly right for the `()` case.
|
|
538
|
+
# A SIGNATURE mismatch is not a declared failure and no family names it,
|
|
539
|
+
# so it escaped both arms as a bare traceback until the second one below
|
|
540
|
+
# (#569): the bootstrap keyword set grows, and an out-of-tree class whose
|
|
541
|
+
# `__init__` does not accept a keyword this function passes is refused by
|
|
542
|
+
# the interpreter, not by the family. That arm keys on traceback DEPTH
|
|
543
|
+
# because depth is what separates the two TypeErrors — binding fails
|
|
544
|
+
# before any `__init__` frame is pushed, a raise from inside one carries
|
|
545
|
+
# that frame. ORDER MATTERS: a family that declares `TypeError` in its own
|
|
546
|
+
# `construct_error` keeps the `error: {e}` line above, unchanged.
|
|
547
|
+
cls = builder.dev if synthesizes else builder.plain
|
|
548
|
+
build_kwargs = {**common, "paths": paths} if synthesizes else common
|
|
549
|
+
try:
|
|
550
|
+
# heterogeneous **kwargs: pyright unions the dict values; per-arg error is spurious
|
|
551
|
+
by_cfg[key] = cls(**build_kwargs) # pyright: ignore[reportArgumentType]
|
|
552
|
+
except builder.construct_error as e:
|
|
553
|
+
raise SystemExit(f"error: {e}") from e
|
|
554
|
+
except TypeError as e:
|
|
555
|
+
# A binding failure is raised by the interpreter BEFORE any __init__
|
|
556
|
+
# frame is pushed, so the traceback holds this frame alone. A
|
|
557
|
+
# TypeError from inside a working __init__ carries that frame too and
|
|
558
|
+
# is a bug in that package: it must surface as itself, on the same
|
|
559
|
+
# rule the ImportError arm above states. Errs toward re-raising — a
|
|
560
|
+
# mismatch behind a Python-level metaclass `__call__` or a
|
|
561
|
+
# `super().__init__` call reads as deeper and re-raises, which is
|
|
562
|
+
# today's behavior; relabelling a real bug is the direction that would
|
|
563
|
+
# cost a diagnosis. Only valid while this `except` sits in the SAME
|
|
564
|
+
# FRAME as the call — do not extract the construct call into a helper
|
|
565
|
+
# or widen the `try`, either breaks it silently.
|
|
566
|
+
if e.__traceback__ is None or e.__traceback__.tb_next is not None:
|
|
567
|
+
raise
|
|
568
|
+
raise SystemExit(
|
|
569
|
+
f"error: profile {profile.name!r}: adapter kind "
|
|
570
|
+
f"{profile.adapter!r} rejected this run's adapter keywords: "
|
|
571
|
+
f"{type(e).__name__}: {e}"
|
|
572
|
+
) from e
|
|
573
|
+
adapters[role] = by_cfg[key]
|
|
574
|
+
return adapters
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def mux_reason_label(reason: str) -> str:
|
|
578
|
+
"""Human wording for a MuxBackendInfo.reason, shared by `mux` and validate."""
|
|
579
|
+
return {
|
|
580
|
+
"env": "forced by FROID_LOOP_MUX_BACKEND",
|
|
581
|
+
"policy": f"set by [mux] backend in {policy_mod.POLICY_FILE}",
|
|
582
|
+
"platform-default": f"platform default for {sys.platform}",
|
|
583
|
+
"first-match": "first available platform match",
|
|
584
|
+
# not "no registered backend is available": `_select` reaches `fallback` when no
|
|
585
|
+
# *available* backend matches this platform — an available backend registered for
|
|
586
|
+
# another platform leaves the reason here just the same.
|
|
587
|
+
"fallback": "fallback (no available backend matches this platform)",
|
|
588
|
+
}.get(reason, reason)
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
def platform_preflight(project: Path) -> list[Finding]:
|
|
592
|
+
"""Probe the platform-selected seams — the terminal multiplexer and the process
|
|
593
|
+
host — for `cmd_validate`, returning the findings in emission order.
|
|
594
|
+
|
|
595
|
+
A backend reports its own readiness through ``available()`` / ``version()``, so
|
|
596
|
+
a new OS or transport surfaces here by *registering* rather than by adding a
|
|
597
|
+
``sys.platform`` branch to validate. The process host is named so a
|
|
598
|
+
misselection (e.g. the Windows host picked on Linux) is visible at a glance.
|
|
599
|
+
|
|
600
|
+
``project`` is read only to name the host/interpreter mismatch behind #332 — a
|
|
601
|
+
win32 interpreter working on a WSL UNC path. Selection is unaffected by it: for
|
|
602
|
+
a win32 interpreter psmux *is* the right pick, so this warns rather than
|
|
603
|
+
re-chooses.
|
|
604
|
+
"""
|
|
605
|
+
from .adapters.multiplexer import (
|
|
606
|
+
detect_multiplexers,
|
|
607
|
+
external_backend_errors,
|
|
608
|
+
fold_version,
|
|
609
|
+
get_multiplexer,
|
|
610
|
+
)
|
|
611
|
+
from .process_host import get_process_host
|
|
612
|
+
|
|
613
|
+
found: list[Finding] = []
|
|
614
|
+
|
|
615
|
+
try:
|
|
616
|
+
backend = get_multiplexer()
|
|
617
|
+
label = type(backend).__name__
|
|
618
|
+
# Defensive fold: an out-of-tree backend can break the seam's
|
|
619
|
+
# single-line promise, and this string lands in an inline message.
|
|
620
|
+
version = fold_version(backend.version())
|
|
621
|
+
if backend.available():
|
|
622
|
+
found.append(
|
|
623
|
+
Finding(
|
|
624
|
+
"mux.backend",
|
|
625
|
+
"ok",
|
|
626
|
+
f"multiplexer {label} available" + (f" ({version})" if version else ""),
|
|
627
|
+
{"backend": label, "available": True, "version": version},
|
|
628
|
+
)
|
|
629
|
+
)
|
|
630
|
+
else:
|
|
631
|
+
found.append(
|
|
632
|
+
Finding(
|
|
633
|
+
"mux.backend",
|
|
634
|
+
"problem",
|
|
635
|
+
f"multiplexer {label} unavailable"
|
|
636
|
+
+ (f" (reports {version})" if version else "")
|
|
637
|
+
+ " — its transport binary is missing, the version is unsupported, or a "
|
|
638
|
+
"required helper is absent (psmux needs `pwsh` on PATH); "
|
|
639
|
+
"see `froid-loop diagnose`",
|
|
640
|
+
{"backend": label, "available": False, "version": version},
|
|
641
|
+
)
|
|
642
|
+
)
|
|
643
|
+
except Exception as e: # selection or readiness must not abort validate
|
|
644
|
+
found.append(Finding("mux.preflight", "problem", f"multiplexer preflight failed: {e}"))
|
|
645
|
+
|
|
646
|
+
try:
|
|
647
|
+
infos = detect_multiplexers()
|
|
648
|
+
except Exception as e:
|
|
649
|
+
# Advisory, so it must not abort validate — but it must not be silent either.
|
|
650
|
+
# Two findings below read `infos`: `mux.selection` vanishes entirely, and the
|
|
651
|
+
# #332 warning degrades to its no-backend wording. Without this line the report
|
|
652
|
+
# shows a healthy `mux.backend` (independent, from `get_multiplexer`) above a
|
|
653
|
+
# warning naming no backend — which reads as "selection failed" when what
|
|
654
|
+
# actually failed was detection.
|
|
655
|
+
found.append(Finding("mux.backends-detected", "warning", f"mux detection failed: {e}"))
|
|
656
|
+
infos = []
|
|
657
|
+
if len(infos) > 1: # a lone tmux needs no listing; keep single-backend output stable
|
|
658
|
+
listed = ", ".join(
|
|
659
|
+
i.name
|
|
660
|
+
+ ("*" if i.selected else "")
|
|
661
|
+
+ (
|
|
662
|
+
" (available" + (f", {i.version}" if i.version else "") + ")"
|
|
663
|
+
if i.available
|
|
664
|
+
else " (unavailable)"
|
|
665
|
+
)
|
|
666
|
+
for i in infos
|
|
667
|
+
)
|
|
668
|
+
# The text flattens each row into a suffix soup ("tmux*, psmux
|
|
669
|
+
# (unavailable)") whose trailing `*` a consumer would have to parse to
|
|
670
|
+
# learn which backend is selected. The detail keeps the rows themselves.
|
|
671
|
+
found.append(
|
|
672
|
+
Finding(
|
|
673
|
+
"mux.backends-detected",
|
|
674
|
+
"ok",
|
|
675
|
+
f"mux backends: {listed} — `froid-loop mux` for details",
|
|
676
|
+
{
|
|
677
|
+
"backends": [
|
|
678
|
+
{
|
|
679
|
+
"name": i.name,
|
|
680
|
+
"matches_platform": i.matches_platform,
|
|
681
|
+
"available": i.available,
|
|
682
|
+
"version": i.version,
|
|
683
|
+
"selected": i.selected,
|
|
684
|
+
"reason": i.reason,
|
|
685
|
+
}
|
|
686
|
+
for i in infos
|
|
687
|
+
]
|
|
688
|
+
},
|
|
689
|
+
)
|
|
690
|
+
)
|
|
691
|
+
chosen = next((i for i in infos if i.selected), None)
|
|
692
|
+
if chosen:
|
|
693
|
+
# Emitted for EVERY reason, not just the forced ones (#332): the reason that
|
|
694
|
+
# most needs naming is `platform-default`, which is how a win32 interpreter
|
|
695
|
+
# silently lands on psmux. detail keeps the raw enum, not mux_reason_label's
|
|
696
|
+
# prose: the label is wording ("set by [mux] backend in
|
|
697
|
+
# .froid-loop/policy.toml"), the enum is the value MuxBackendInfo.reason
|
|
698
|
+
# actually carries.
|
|
699
|
+
#
|
|
700
|
+
# Severity follows the reason. `fallback` is the one `_select` returns when no
|
|
701
|
+
# *available* backend matches this platform, and its label says exactly that — so
|
|
702
|
+
# emitting it at "ok" would print a green line whose own text contradicts it.
|
|
703
|
+
# It stays a warning rather than a problem because `mux.backend` above already
|
|
704
|
+
# carries the problem for that host; this line only names how it got there.
|
|
705
|
+
found.append(
|
|
706
|
+
Finding(
|
|
707
|
+
"mux.selection",
|
|
708
|
+
"warning" if chosen.reason == "fallback" else "ok",
|
|
709
|
+
f"multiplexer selection {mux_reason_label(chosen.reason)}",
|
|
710
|
+
{"backend": chosen.name, "reason": chosen.reason},
|
|
711
|
+
)
|
|
712
|
+
)
|
|
713
|
+
|
|
714
|
+
# A warning, not a problem and not a note: an installed package the operator
|
|
715
|
+
# asked for did not load, which is a real failure — but selection already
|
|
716
|
+
# degraded past it (a failed external can never be the selected backend), so
|
|
717
|
+
# the preflight outcome above is authoritative and the verdict must not flip.
|
|
718
|
+
# `cmd_mux` has always printed this same condition as `warning:`; validate was
|
|
719
|
+
# the outlier, pinned to "ok" because promoting inserts " warning: " into the
|
|
720
|
+
# text (render() keeps the double prefix by design) and the TUI rendered that
|
|
721
|
+
# text verbatim. Since #210 the TUI reads `validate --json` and styles from the
|
|
722
|
+
# severity field, so the severity is free to say what the message already does.
|
|
723
|
+
for ep_name, reason in sorted(external_backend_errors().items()):
|
|
724
|
+
found.append(
|
|
725
|
+
Finding(
|
|
726
|
+
"mux.external-backend",
|
|
727
|
+
"warning",
|
|
728
|
+
f"external mux backend '{ep_name}' failed to load: {reason}",
|
|
729
|
+
{"entry_point": ep_name, "error": reason},
|
|
730
|
+
)
|
|
731
|
+
)
|
|
732
|
+
|
|
733
|
+
try:
|
|
734
|
+
host = type(get_process_host()).__name__
|
|
735
|
+
found.append(Finding("host.process", "ok", f"process host: {host}", {"host": host}))
|
|
736
|
+
except Exception as e: # a bad FROID_LOOP_PROCESS_HOST must report, not crash
|
|
737
|
+
found.append(Finding("host.process", "problem", f"process host preflight failed: {e}"))
|
|
738
|
+
|
|
739
|
+
# A `warning`, never a `problem`: every seam above is healthy for this interpreter,
|
|
740
|
+
# so the verdict and the exit code must not flip — what is wrong is the interpreter,
|
|
741
|
+
# and only the operator can swap it.
|
|
742
|
+
if sys.platform == "win32" and is_wsl_unc_path(project):
|
|
743
|
+
# State only what the evidence supports: win32 + a distro path is NOT proof of a
|
|
744
|
+
# WSL shell — `cd \\wsl.localhost\...` from native PowerShell reaches the same
|
|
745
|
+
# condition, and the interop env markers do not survive the boundary (see
|
|
746
|
+
# `is_wsl_unc_path`) — so the WSL remedy stays conditional. The backend clause
|
|
747
|
+
# names what was *actually* chosen (a forced choice would otherwise contradict
|
|
748
|
+
# `mux.selection` above) and is dropped when selection failed (`chosen is None`):
|
|
749
|
+
# inventing a backend is the exact failure this check exists to stop.
|
|
750
|
+
picked = (
|
|
751
|
+
f"{chosen.name} was selected and the distro's own tmux is invisible to it"
|
|
752
|
+
if chosen
|
|
753
|
+
else "the distro's own tmux is invisible to it"
|
|
754
|
+
)
|
|
755
|
+
found.append(
|
|
756
|
+
Finding(
|
|
757
|
+
"host.win32-on-wsl-path",
|
|
758
|
+
"warning",
|
|
759
|
+
"the native-Windows build (this interpreter reports win32) is working on a "
|
|
760
|
+
f"WSL distro path — {picked}; if you are running from a WSL shell, install "
|
|
761
|
+
"froid-loop with the WSL/Linux Python instead",
|
|
762
|
+
# `project` is deliberately NOT carried here: `validate --json` is not a
|
|
763
|
+
# sanitized surface, and a distro path ends in the *Linux* username,
|
|
764
|
+
# which the egress redactor does not know.
|
|
765
|
+
{"backend": chosen.name if chosen else None, "platform": sys.platform},
|
|
766
|
+
)
|
|
767
|
+
)
|
|
768
|
+
|
|
769
|
+
return found
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
def build_run_state(
|
|
773
|
+
*,
|
|
774
|
+
run_id: str,
|
|
775
|
+
project: Path,
|
|
776
|
+
repo_root: Path,
|
|
777
|
+
policy: Policy,
|
|
778
|
+
epic_filter: int | None,
|
|
779
|
+
story_filter: str | None,
|
|
780
|
+
max_stories: int | None,
|
|
781
|
+
stories_on: bool,
|
|
782
|
+
spec_folder: str,
|
|
783
|
+
trusted_config_digest: str,
|
|
784
|
+
) -> RunState:
|
|
785
|
+
"""Assemble the launch-time :class:`RunState` for a fresh run.
|
|
786
|
+
|
|
787
|
+
``policy_snapshot`` freezes ``policy`` at launch so every later display reads
|
|
788
|
+
the weights the run actually launched under; ``source`` / ``spec_folder``
|
|
789
|
+
record which queue the run dispatches (a stories manifest vs sprint-status).
|
|
790
|
+
|
|
791
|
+
``trusted_config_digest`` is carried here **as well as** stamped out of the
|
|
792
|
+
tree by :func:`compose_run` (#498). The out-of-tree file is the one resume
|
|
793
|
+
trusts; this copy is the secondary that travels with the run directory — see
|
|
794
|
+
``RunState.trusted_config_digest`` for why a run that outlives its state key
|
|
795
|
+
needs one.
|
|
796
|
+
|
|
797
|
+
``repo_root`` records the git root code work happens in (``paths.repo_root``),
|
|
798
|
+
which equals ``project`` unless the FROID config sets a `repo_root:` override.
|
|
799
|
+
``runs.rearm_escalation`` runs out of process and reads it back to advance the
|
|
800
|
+
attempt baseline in the tree the proof-of-work gate actually measures."""
|
|
801
|
+
return RunState(
|
|
802
|
+
run_id=run_id,
|
|
803
|
+
project=str(project),
|
|
804
|
+
repo_root=str(repo_root),
|
|
805
|
+
started_at=time.strftime("%Y-%m-%dT%H:%M:%S"),
|
|
806
|
+
policy_snapshot=policy.to_dict(),
|
|
807
|
+
epic_filter=epic_filter,
|
|
808
|
+
story_filter=story_filter,
|
|
809
|
+
max_stories=max_stories,
|
|
810
|
+
source="stories" if stories_on else "sprint-status",
|
|
811
|
+
spec_folder=spec_folder if stories_on else "",
|
|
812
|
+
trusted_config_digest=trusted_config_digest,
|
|
813
|
+
)
|
|
814
|
+
|
|
815
|
+
|
|
816
|
+
@dataclass
|
|
817
|
+
class ComposedRun:
|
|
818
|
+
"""The composed-but-not-yet-run artifacts a ``compose_*`` returns for its
|
|
819
|
+
callback to render from — shared by :func:`compose_run`, :func:`compose_sweep`,
|
|
820
|
+
and :func:`compose_resume`.
|
|
821
|
+
|
|
822
|
+
``engine`` is ready to :meth:`run`; ``run_id`` names the run for the attach
|
|
823
|
+
hint. ``run_dir`` / ``state`` / ``journal`` are the persisted context a caller
|
|
824
|
+
other than the CLI can inspect."""
|
|
825
|
+
|
|
826
|
+
engine: Engine
|
|
827
|
+
run_id: str
|
|
828
|
+
run_dir: Path
|
|
829
|
+
state: RunState
|
|
830
|
+
journal: Journal
|
|
831
|
+
|
|
832
|
+
|
|
833
|
+
def _claim_run_dir(run_dir: Path) -> None:
|
|
834
|
+
"""Take exclusive ownership of a fresh run directory, refusing an id that
|
|
835
|
+
already names a run.
|
|
836
|
+
|
|
837
|
+
A **claim**, not a check, and that distinction is the whole point:
|
|
838
|
+
``exist_ok=False`` makes the directory's creation and the collision refusal one
|
|
839
|
+
atomic operation, so what follows may treat "this run dir is ours" as PROVEN
|
|
840
|
+
rather than inferred. :func:`_unwind_composition` deletes this directory
|
|
841
|
+
wholesale on a failed composition, and inference is not good enough to license
|
|
842
|
+
an ``rmtree``.
|
|
843
|
+
|
|
844
|
+
The hazard is not hypothetical. ``run_id`` is caller-supplied through the
|
|
845
|
+
hidden ``--run-id`` flag on both ``run`` and ``sweep``, and the composers ran
|
|
846
|
+
straight into ``Journal(run_dir)``, whose ``mkdir(parents=True,
|
|
847
|
+
exist_ok=True)`` adopts an existing directory without complaint. So pointing
|
|
848
|
+
``--run-id`` at a *pre-existing* paused, stopped or finished run published this
|
|
849
|
+
composition's ``state.json`` over that run's, and then — once ``make_adapters``
|
|
850
|
+
raised its reachable ``SystemExit`` — unwound the whole thing: journal, logs,
|
|
851
|
+
tasks and out-of-tree state, permanently. ``delete_run``'s guard does not cover
|
|
852
|
+
it, since that guard refuses only a *live* session and a paused or finished run
|
|
853
|
+
has none.
|
|
854
|
+
|
|
855
|
+
Refusing before anything is published is what makes that unreachable, so this
|
|
856
|
+
MUST stay outside the composers' ``try`` — a refusal that reached the unwind
|
|
857
|
+
arm would delete the very run it exists to protect. ``SystemExit`` matches the
|
|
858
|
+
other launch-time refusals an operator reads as an ``error:`` line
|
|
859
|
+
(``_reject_bad_run_id``, and ``make_adapters``' six sites).
|
|
860
|
+
|
|
861
|
+
Applied to a minted id too, not just a supplied one. ``new_run_id`` is a
|
|
862
|
+
timestamp plus two random bytes, so a same-second collision is remote rather
|
|
863
|
+
than impossible — and a guard that holds for every id lets callers state the
|
|
864
|
+
freshness of their run dir flatly instead of qualifying it by provenance."""
|
|
865
|
+
try:
|
|
866
|
+
run_dir.mkdir(parents=True, exist_ok=False)
|
|
867
|
+
except FileExistsError as e:
|
|
868
|
+
raise SystemExit(
|
|
869
|
+
f"error: run {run_dir.name} already exists — refusing to compose over it. "
|
|
870
|
+
"`--run-id` must name a run that does not exist yet."
|
|
871
|
+
) from e
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
def _unwind_composition(project: Path, run_dir: Path, journal: Journal | None) -> None:
|
|
875
|
+
"""Remove the run a failed ``compose_*`` had already published, so a launch
|
|
876
|
+
that aborts partway leaves nothing behind.
|
|
877
|
+
|
|
878
|
+
Safe as a wholesale removal only because :func:`_claim_run_dir` created this
|
|
879
|
+
directory with ``exist_ok=False`` moments earlier: the run being deleted is
|
|
880
|
+
provably this composition's, never a pre-existing one the caller named.
|
|
881
|
+
|
|
882
|
+
Reached from an ``except BaseException`` arm, because the failure it exists
|
|
883
|
+
for is a :class:`SystemExit`: :func:`make_adapters` raises one at six sites
|
|
884
|
+
(unresolvable profile, unknown adapter kind, a kind that fails to load, a
|
|
885
|
+
construction failure, an adapter class that rejects a bootstrap keyword, an
|
|
886
|
+
unusable multiplexer), every one of them *after* ``save_state`` has published
|
|
887
|
+
a run dir carrying ``finished=False`` / ``crashed=False`` and no
|
|
888
|
+
``run-start``. Nothing reconciles that shape —
|
|
889
|
+
:func:`runs.reconcile_stale_worktrees` only touches ``is_finished`` runs — so
|
|
890
|
+
it lingers as a resumable-looking empty run.
|
|
891
|
+
|
|
892
|
+
:func:`runs.delete_run` is the right primitive rather than a bare ``rmtree``
|
|
893
|
+
because it also drops the run's out-of-tree state dir (``_discard_state_dir``),
|
|
894
|
+
which is what covers the config-digest stamp the composers write between the
|
|
895
|
+
state and the pid.
|
|
896
|
+
|
|
897
|
+
``force=False``, deliberately. ``force`` is documented there as the
|
|
898
|
+
*operator's* explicit override, and there is no operator here — this is an
|
|
899
|
+
automatic unwind. What it would skip is the one guard protecting the one state
|
|
900
|
+
where a run dir is load-bearing: an untagged live ``froid-loop-<id>`` session,
|
|
901
|
+
for which that directory is the only ownership proof a later prune can read.
|
|
902
|
+
:func:`_claim_run_dir` rules out a session belonging to a *pre-existing run* at
|
|
903
|
+
this id — there is no such run — but not an orphaned session outliving the run
|
|
904
|
+
dir it was named for, which this launch would then be deleting the only
|
|
905
|
+
ownership proof of while never having spawned a session of its own. Narrower
|
|
906
|
+
than the case this paragraph used to argue, and still real. When the guard does
|
|
907
|
+
fire the cost is exactly the pre-fix behavior, a stranded run dir, which is no
|
|
908
|
+
worse than what this replaces; ``force=True`` would trade that bounded cost for
|
|
909
|
+
an unbounded one.
|
|
910
|
+
|
|
911
|
+
Best-effort, and never raising: the caller is already unwinding an exception
|
|
912
|
+
the operator has to see, and a cleanup failure replacing it is the one outcome
|
|
913
|
+
that must not happen. The enumerable failures are :class:`runs.LiveSessionError`
|
|
914
|
+
(the guard refusing), ``OSError`` (the removal, or ``project.resolve()`` on a
|
|
915
|
+
path the OS cannot canonicalize) and ``RuntimeError`` (how ``Path.resolve``
|
|
916
|
+
reports a symlink loop below 3.13 — see ``runs._discard_state_dir``). It is not
|
|
917
|
+
written as that tuple because ``delete_run`` reaches the multiplexer registry
|
|
918
|
+
through :func:`runs.live_session_may_be_ours`, an extension point an out-of-tree
|
|
919
|
+
backend can make raise anything, so an enumerated list is one a third-party
|
|
920
|
+
backend falsifies. ``Exception`` and not ``BaseException``: a
|
|
921
|
+
``KeyboardInterrupt`` arriving during the cleanup still belongs to the operator.
|
|
922
|
+
|
|
923
|
+
But not *silent*, which is a separate decision from not *raising* and was
|
|
924
|
+
previously conflated with it. "Repair writes must raise" (AGENTS.md) cannot be
|
|
925
|
+
honored literally here — raising is precisely what would swallow the launch
|
|
926
|
+
error — so the obligation it encodes is discharged by reporting instead.
|
|
927
|
+
Swallowing a failed unwind leaves exactly the resumable-looking ghost run this
|
|
928
|
+
function exists to prevent, and leaves it inferable only from the ABSENCE of an
|
|
929
|
+
effect: the operator reads the launch error, and nothing anywhere says the
|
|
930
|
+
cleanup after it did not happen."""
|
|
931
|
+
try:
|
|
932
|
+
runs.delete_run(project, run_dir)
|
|
933
|
+
except Exception as e:
|
|
934
|
+
detail = f"{type(e).__name__}: {e}"
|
|
935
|
+
print(
|
|
936
|
+
f"warning: could not remove the partially composed run {run_dir.name}: "
|
|
937
|
+
f"{detail} — it may look resumable; remove it with "
|
|
938
|
+
f"`froid-loop delete {run_dir.name}`",
|
|
939
|
+
file=sys.stderr,
|
|
940
|
+
)
|
|
941
|
+
# The journal lives INSIDE the run dir, so this lands for every failure that
|
|
942
|
+
# leaves one behind — the guard refusing, or a failed `rmtree` — which is
|
|
943
|
+
# also the only case where a ghost run is what the operator will find. When
|
|
944
|
+
# `_discard_state_dir` is instead what failed the dir is already gone, and
|
|
945
|
+
# `Journal.append` opens with "a" WITHOUT a mkdir, so it raises rather than
|
|
946
|
+
# resurrecting the run it just removed. Suppressed, and the stderr line
|
|
947
|
+
# above still carries the report.
|
|
948
|
+
#
|
|
949
|
+
# ``journal`` is None when the composer aborted between claiming the run dir
|
|
950
|
+
# and building the Journal — a window only a signal can realistically land
|
|
951
|
+
# in. Guarded explicitly rather than left to the ``suppress`` above: an
|
|
952
|
+
# AttributeError on None IS an Exception and would be swallowed, so the
|
|
953
|
+
# code would work by accident while reading as though a Journal were
|
|
954
|
+
# guaranteed. The stderr report is the part that matters and is unaffected.
|
|
955
|
+
if journal is not None:
|
|
956
|
+
with suppress(Exception):
|
|
957
|
+
journal.append("composition-unwind-failed", run_id=run_dir.name, error=detail)
|
|
958
|
+
|
|
959
|
+
|
|
960
|
+
def compose_run(
|
|
961
|
+
*,
|
|
962
|
+
project: Path,
|
|
963
|
+
paths: froidconfig.ProjectPaths,
|
|
964
|
+
policy: Policy,
|
|
965
|
+
run_id: str | None,
|
|
966
|
+
epic_filter: int | None,
|
|
967
|
+
story_filter: str | None,
|
|
968
|
+
max_stories: int | None,
|
|
969
|
+
stories_on: bool,
|
|
970
|
+
spec_folder: str,
|
|
971
|
+
sweep_factory: SweepFactory,
|
|
972
|
+
make_adapters: MakeAdapters,
|
|
973
|
+
engine_cls: type[Engine],
|
|
974
|
+
stories_engine_cls: type[StoriesEngine],
|
|
975
|
+
trusted_config_digest: str,
|
|
976
|
+
profiles: dict[str, CLIProfile] | None = None,
|
|
977
|
+
) -> ComposedRun:
|
|
978
|
+
"""Stand up a run: allocate the run dir, persist state + pid, build the
|
|
979
|
+
adapters, and wire the engine — everything ``cmd_run`` did inline between its
|
|
980
|
+
preflight gates and ``engine.run()``.
|
|
981
|
+
|
|
982
|
+
``profiles`` carries ``cmd_run``'s single :func:`resolve_profiles` resolution —
|
|
983
|
+
the same one ``trusted_config_digest`` was computed from — so the stamped
|
|
984
|
+
baseline describes the bytes these adapters are built from rather than a second
|
|
985
|
+
read of an agent-writable file (#461 point 4). ``None`` resolves fresh.
|
|
986
|
+
|
|
987
|
+
``trusted_config_digest`` is stamped into the run's out-of-tree state dir
|
|
988
|
+
(#498), so the baseline ``resume`` warns off is not sitting in the tree the
|
|
989
|
+
driven sessions write to, **and** onto the :class:`RunState` as the secondary
|
|
990
|
+
that travels with the run dir. The out-of-tree copy is preferred whenever it
|
|
991
|
+
exists, which is what keeps the in-tree one from being worth tampering with;
|
|
992
|
+
``RunState.trusted_config_digest`` states the split and why both are needed.
|
|
993
|
+
|
|
994
|
+
``make_adapters`` and the engine classes are injected (rather than imported
|
|
995
|
+
here) so ``cli`` supplies its own module-level names — keeping the test
|
|
996
|
+
suite's ``monkeypatch.setattr(cli, "Engine"/"_make_adapters", ...)`` effective.
|
|
997
|
+
"""
|
|
998
|
+
run_id = run_id or runs.new_run_id()
|
|
999
|
+
run_dir = project / RUNS_DIR / run_id
|
|
1000
|
+
# Outside the try below, and it must stay there: a collision refusal that
|
|
1001
|
+
# reached `_unwind_composition` would delete the run it exists to protect.
|
|
1002
|
+
_claim_run_dir(run_dir)
|
|
1003
|
+
# Composition is atomic from the first published artifact onward: everything
|
|
1004
|
+
# below either lands whole or is unwound (see :func:`_unwind_composition`,
|
|
1005
|
+
# which also states why the arm is `BaseException` and not `Exception`).
|
|
1006
|
+
# The guard opens on the statement immediately after the claim, because the
|
|
1007
|
+
# claim is what publishes that first artifact — the run DIRECTORY itself, which
|
|
1008
|
+
# is what a later `--run-id` collides with. Neither statement below can
|
|
1009
|
+
# realistically fail (`Journal` mkdirs `exist_ok=True` over a directory this
|
|
1010
|
+
# frame just created, and `build_run_state` is a pure constructor), but a
|
|
1011
|
+
# signal can land between any two statements, and the arm is `BaseException`
|
|
1012
|
+
# exactly so that case unwinds instead of stranding an empty run dir.
|
|
1013
|
+
journal: Journal | None = None
|
|
1014
|
+
try:
|
|
1015
|
+
journal = Journal(run_dir)
|
|
1016
|
+
state = build_run_state(
|
|
1017
|
+
run_id=run_id,
|
|
1018
|
+
project=project,
|
|
1019
|
+
repo_root=paths.repo_root,
|
|
1020
|
+
policy=policy,
|
|
1021
|
+
epic_filter=epic_filter,
|
|
1022
|
+
story_filter=story_filter,
|
|
1023
|
+
max_stories=max_stories,
|
|
1024
|
+
stories_on=stories_on,
|
|
1025
|
+
spec_folder=spec_folder,
|
|
1026
|
+
trusted_config_digest=trusted_config_digest,
|
|
1027
|
+
)
|
|
1028
|
+
save_state(run_dir, state)
|
|
1029
|
+
# After the run dir exists (Journal mkdir'd it above) and before the pid lands:
|
|
1030
|
+
# the ordering `reconcile_orphan_state_dirs` reads runs in, and a stamp that
|
|
1031
|
+
# cannot be written fails the launch before an observer can see a live run.
|
|
1032
|
+
runs.write_trusted_config_digest(project, run_id, trusted_config_digest)
|
|
1033
|
+
runs.write_pid(run_dir)
|
|
1034
|
+
adapters = make_adapters(project, run_dir, policy, profiles=profiles)
|
|
1035
|
+
journal.append(
|
|
1036
|
+
"run-start",
|
|
1037
|
+
run_id=run_id,
|
|
1038
|
+
source=state.source,
|
|
1039
|
+
adapter_dev=policy.adapter.resolved("dev").name,
|
|
1040
|
+
adapter_review=policy.adapter.resolved("review").name,
|
|
1041
|
+
)
|
|
1042
|
+
common = dict(
|
|
1043
|
+
paths=paths,
|
|
1044
|
+
policy=policy,
|
|
1045
|
+
adapter=adapters["dev"],
|
|
1046
|
+
review_adapter=adapters["review"],
|
|
1047
|
+
run_dir=run_dir,
|
|
1048
|
+
journal=journal,
|
|
1049
|
+
state=state,
|
|
1050
|
+
max_stories=max_stories,
|
|
1051
|
+
epic_filter=epic_filter,
|
|
1052
|
+
story_filter=story_filter,
|
|
1053
|
+
sweep_factory=sweep_factory,
|
|
1054
|
+
)
|
|
1055
|
+
# heterogeneous **kwargs: pyright unions the dict values; per-arg error is spurious
|
|
1056
|
+
engine: Engine = (
|
|
1057
|
+
stories_engine_cls(**common, spec_folder=spec_folder)
|
|
1058
|
+
if stories_on
|
|
1059
|
+
else engine_cls(**common) # pyright: ignore[reportArgumentType]
|
|
1060
|
+
)
|
|
1061
|
+
except BaseException:
|
|
1062
|
+
_unwind_composition(project, run_dir, journal)
|
|
1063
|
+
raise
|
|
1064
|
+
return ComposedRun(engine=engine, run_id=run_id, run_dir=run_dir, state=state, journal=journal)
|
|
1065
|
+
|
|
1066
|
+
|
|
1067
|
+
def compose_sweep(
|
|
1068
|
+
*,
|
|
1069
|
+
project: Path,
|
|
1070
|
+
paths: froidconfig.ProjectPaths,
|
|
1071
|
+
policy: Policy,
|
|
1072
|
+
run_id: str | None,
|
|
1073
|
+
prompting: bool,
|
|
1074
|
+
decisions_only: bool,
|
|
1075
|
+
max_bundles: int | None,
|
|
1076
|
+
repeat: bool | None,
|
|
1077
|
+
max_cycles: int | None,
|
|
1078
|
+
trigger: str,
|
|
1079
|
+
make_adapters: MakeAdapters,
|
|
1080
|
+
sweep_engine_cls: type[SweepEngine],
|
|
1081
|
+
trusted_config_digest: str,
|
|
1082
|
+
profiles: dict[str, CLIProfile] | None = None,
|
|
1083
|
+
on_started: Callable[[], None] | None = None,
|
|
1084
|
+
) -> ComposedRun:
|
|
1085
|
+
"""Stand up a sweep run: allocate the run dir, persist state + pid, record the
|
|
1086
|
+
sweep options, build the adapters, and wire the ``SweepEngine`` — everything
|
|
1087
|
+
``cli._start_sweep`` did inline before ``engine.run()``.
|
|
1088
|
+
|
|
1089
|
+
``profiles`` carries an already-resolved :func:`resolve_profiles` mapping down
|
|
1090
|
+
to ``make_adapters``. The child-sweep factory passes the same one it gated on,
|
|
1091
|
+
so the adapters are built from the validated bytes instead of a fresh read of
|
|
1092
|
+
an agent-writable file (#461 point 4); ``cmd_sweep`` (human-present) omits it.
|
|
1093
|
+
``trusted_config_digest`` lands in the run's out-of-tree state dir and, as the
|
|
1094
|
+
travelling secondary, on the :class:`RunState` — see :func:`compose_run`.
|
|
1095
|
+
|
|
1096
|
+
``sweep.json`` freezes the launch options so a resume rebuilds the same sweep
|
|
1097
|
+
(see :func:`compose_resume`). ``make_adapters`` and ``sweep_engine_cls`` are
|
|
1098
|
+
injected so ``cli`` supplies its own module-level names — keeping the test
|
|
1099
|
+
suite's ``monkeypatch.setattr(cli, "SweepEngine"/"_make_adapters", ...)``
|
|
1100
|
+
effective.
|
|
1101
|
+
|
|
1102
|
+
``on_started`` is the auto-sweep parent's latch (``engine.SweepFactory``'s
|
|
1103
|
+
``started`` thunk, threaded through ``cli._start_sweep``): a parent run spends
|
|
1104
|
+
its one trigger for this ``trigger`` string only if this fires.
|
|
1105
|
+
``cmd_sweep`` passes nothing — a human started that one, and there is no
|
|
1106
|
+
trigger to spend.
|
|
1107
|
+
|
|
1108
|
+
It fires as the LAST statement of the composition block, which is the boundary
|
|
1109
|
+
that makes "started" mean something the parent can act on: from here the child
|
|
1110
|
+
owns a published run dir, ``sweep.json`` and a live pid file, so a later
|
|
1111
|
+
failure leaves a run ``froid-loop resume`` can pick up rather than nothing at
|
|
1112
|
+
all. Before commit ``9c7a284`` the boundary had to sit at ``save_state``
|
|
1113
|
+
instead — an abort anywhere after it stranded a resumable-looking run dir, and
|
|
1114
|
+
:func:`compose_resume` will rebuild a sweep from ``state.json`` alone,
|
|
1115
|
+
tolerating a missing ``sweep.json``, so "it never got far enough to resume"
|
|
1116
|
+
was not true of the intervening steps. What moved it here is that block's
|
|
1117
|
+
``except BaseException`` arm, added by that commit, which unwinds the whole
|
|
1118
|
+
partial composition.
|
|
1119
|
+
|
|
1120
|
+
That premise has one documented exception, and it is worth reading rather than
|
|
1121
|
+
waving at: :func:`_unwind_composition` is best-effort — ``force=False`` leaves
|
|
1122
|
+
``runs.delete_run``'s live-session guard armed, and the call sits under
|
|
1123
|
+
``suppress(Exception)`` — so a refused or failed unwind CAN leave a resumable
|
|
1124
|
+
child behind while ``on_started`` never fired. On the auto path that needs a
|
|
1125
|
+
live ``froid-loop-<id>`` session at this run's id, and the path mints the id
|
|
1126
|
+
here: ``cli._sweep_factory`` calls ``_start_sweep`` with no ``run_id``, and the
|
|
1127
|
+
only caller that supplies one is ``cmd_sweep`` (``--run-id``), which passes no
|
|
1128
|
+
``on_started``. So reaching it needs a live session at an id that names no run
|
|
1129
|
+
of its own — an orphan outliving its run dir — because a collision with a run
|
|
1130
|
+
that still EXISTS is now refused before anything is published
|
|
1131
|
+
(:func:`_claim_run_dir`), and a :func:`runs.new_run_id` collision is remote to
|
|
1132
|
+
begin with. The latch boundary is not the place to answer what is left.
|
|
1133
|
+
|
|
1134
|
+
Firing inside the block rather than after it is deliberate for the same
|
|
1135
|
+
reason: should the latch itself raise, the unwind covers it, and the parent's
|
|
1136
|
+
in-memory flag — set BEFORE its write, see ``engine._maybe_auto_sweep`` —
|
|
1137
|
+
refuses a second attempt either way. At-most-once therefore holds independently
|
|
1138
|
+
of the unwind; what the unwind decides is only what that refusal costs. Normally
|
|
1139
|
+
it refuses a child that left nothing behind; under the refused unwind above it
|
|
1140
|
+
refuses one that is composed and resumable, which is the better of the two.
|
|
1141
|
+
Neither is a second launch, and that is the safe direction for a launcher."""
|
|
1142
|
+
run_id = run_id or runs.new_run_id()
|
|
1143
|
+
run_dir = project / RUNS_DIR / run_id
|
|
1144
|
+
# Same claim, same reason, same placement outside the try as in `compose_run`.
|
|
1145
|
+
_claim_run_dir(run_dir)
|
|
1146
|
+
# Atomic from the first published artifact onward, exactly as in `compose_run`
|
|
1147
|
+
# — same reason, same opening on the statement after the claim, and one more
|
|
1148
|
+
# artifact to unwind (`sweep.json`).
|
|
1149
|
+
journal: Journal | None = None
|
|
1150
|
+
try:
|
|
1151
|
+
journal = Journal(run_dir)
|
|
1152
|
+
state = RunState(
|
|
1153
|
+
run_id=run_id,
|
|
1154
|
+
project=str(project),
|
|
1155
|
+
repo_root=str(paths.repo_root),
|
|
1156
|
+
started_at=time.strftime("%Y-%m-%dT%H:%M:%S"),
|
|
1157
|
+
policy_snapshot=policy.to_dict(),
|
|
1158
|
+
run_type="sweep",
|
|
1159
|
+
trusted_config_digest=trusted_config_digest,
|
|
1160
|
+
)
|
|
1161
|
+
save_state(run_dir, state)
|
|
1162
|
+
# Out of the tree, same ordering and same reason as compose_run's stamp.
|
|
1163
|
+
runs.write_trusted_config_digest(project, run_id, trusted_config_digest)
|
|
1164
|
+
runs.write_pid(run_dir)
|
|
1165
|
+
options = {
|
|
1166
|
+
"prompting": prompting,
|
|
1167
|
+
"decisions_only": decisions_only,
|
|
1168
|
+
"max_bundles": max_bundles,
|
|
1169
|
+
"repeat": repeat,
|
|
1170
|
+
"max_cycles": max_cycles,
|
|
1171
|
+
"trigger": trigger,
|
|
1172
|
+
}
|
|
1173
|
+
# Persist the sweep options atomically (tmp + os.replace), the way save_state
|
|
1174
|
+
# writes state.json: a resume reads this back to rebuild the SweepEngine, so a
|
|
1175
|
+
# crash mid-write must not leave a torn file the recovery path then chokes on.
|
|
1176
|
+
sweep_path = run_dir / "sweep.json"
|
|
1177
|
+
sweep_tmp = sweep_path.with_suffix(".json.tmp")
|
|
1178
|
+
sweep_tmp.write_text(json.dumps(options, indent=2), encoding="utf-8")
|
|
1179
|
+
atomic_replace(sweep_tmp, sweep_path)
|
|
1180
|
+
adapters = make_adapters(project, run_dir, policy, profiles=profiles)
|
|
1181
|
+
journal.append("run-start", run_id=run_id, run_type="sweep", trigger=trigger)
|
|
1182
|
+
engine: Engine = sweep_engine_cls(
|
|
1183
|
+
paths=paths,
|
|
1184
|
+
policy=policy,
|
|
1185
|
+
adapter=adapters["dev"],
|
|
1186
|
+
review_adapter=adapters["review"],
|
|
1187
|
+
triage_adapter=adapters["triage"],
|
|
1188
|
+
run_dir=run_dir,
|
|
1189
|
+
journal=journal,
|
|
1190
|
+
state=state,
|
|
1191
|
+
prompting=prompting,
|
|
1192
|
+
decisions_only=decisions_only,
|
|
1193
|
+
max_bundles=max_bundles,
|
|
1194
|
+
repeat=repeat,
|
|
1195
|
+
max_cycles=max_cycles,
|
|
1196
|
+
)
|
|
1197
|
+
if on_started is not None:
|
|
1198
|
+
on_started()
|
|
1199
|
+
except BaseException:
|
|
1200
|
+
_unwind_composition(project, run_dir, journal)
|
|
1201
|
+
raise
|
|
1202
|
+
return ComposedRun(engine=engine, run_id=run_id, run_dir=run_dir, state=state, journal=journal)
|
|
1203
|
+
|
|
1204
|
+
|
|
1205
|
+
def compose_resume(
|
|
1206
|
+
*,
|
|
1207
|
+
project: Path,
|
|
1208
|
+
paths: froidconfig.ProjectPaths,
|
|
1209
|
+
run_dir: Path,
|
|
1210
|
+
state: RunState,
|
|
1211
|
+
policy: Policy,
|
|
1212
|
+
journal: Journal,
|
|
1213
|
+
sweep_factory: SweepFactory,
|
|
1214
|
+
make_adapters: MakeAdapters,
|
|
1215
|
+
engine_cls: type[Engine],
|
|
1216
|
+
stories_engine_cls: type[StoriesEngine],
|
|
1217
|
+
sweep_engine_cls: type[SweepEngine],
|
|
1218
|
+
profiles: dict[str, CLIProfile] | None = None,
|
|
1219
|
+
) -> ComposedRun:
|
|
1220
|
+
"""Rebuild the engine for a paused/interrupted run and return it ready to
|
|
1221
|
+
:meth:`run` — the adapter build + engine selection ``cli._resume_paused_run``
|
|
1222
|
+
did inline.
|
|
1223
|
+
|
|
1224
|
+
``state`` arrives already re-stamped and persisted by the caller: the resume
|
|
1225
|
+
policy-snapshot reconciliation and the pause/pid/graceful-stop bookkeeping stay
|
|
1226
|
+
CLI-side (their ordering is load-bearing — see ``_resume_paused_run``), so this
|
|
1227
|
+
lifts only the composition. The variant is selected from persisted run state:
|
|
1228
|
+
``run_type == "sweep"`` rebuilds a ``SweepEngine`` from ``sweep.json``;
|
|
1229
|
+
otherwise ``source`` picks ``StoriesEngine`` vs ``Engine``, restoring the
|
|
1230
|
+
launching scope + cap so a resumed ``--epic N`` run keeps its filter. The engine
|
|
1231
|
+
classes and ``make_adapters`` are injected so ``cli``'s ``monkeypatch.setattr``
|
|
1232
|
+
seams bite.
|
|
1233
|
+
|
|
1234
|
+
``profiles`` carries the caller's single :func:`resolve_profiles` resolution —
|
|
1235
|
+
the one the re-stamped ``state.trusted_config_digest`` was computed from — so
|
|
1236
|
+
the new baseline describes the bytes these adapters are built from rather than
|
|
1237
|
+
a second read of an agent-writable file (#461 point 4). ``None`` resolves
|
|
1238
|
+
fresh."""
|
|
1239
|
+
# drop any stale agent session so the run spins up a fresh one (a stopped or
|
|
1240
|
+
# interrupted run can leave a lingering froid-loop-<id> session behind).
|
|
1241
|
+
runs.kill_session(run_dir.name)
|
|
1242
|
+
adapters = make_adapters(project, run_dir, policy, profiles=profiles)
|
|
1243
|
+
if state.run_type == "sweep":
|
|
1244
|
+
opts_path = run_dir / "sweep.json"
|
|
1245
|
+
try:
|
|
1246
|
+
opts = json.loads(opts_path.read_text(encoding="utf-8")) if opts_path.is_file() else {}
|
|
1247
|
+
except (OSError, json.JSONDecodeError):
|
|
1248
|
+
# A torn/corrupt sweep.json (crash mid-write on an older run) must not
|
|
1249
|
+
# abort the recovery path — fall back to the same launch defaults as
|
|
1250
|
+
# the missing-file arm, mirroring tui.data's tolerant run-dir reads.
|
|
1251
|
+
opts = {}
|
|
1252
|
+
engine: Engine = sweep_engine_cls(
|
|
1253
|
+
paths=paths,
|
|
1254
|
+
policy=policy,
|
|
1255
|
+
adapter=adapters["dev"],
|
|
1256
|
+
review_adapter=adapters["review"],
|
|
1257
|
+
triage_adapter=adapters["triage"],
|
|
1258
|
+
run_dir=run_dir,
|
|
1259
|
+
journal=journal,
|
|
1260
|
+
state=state,
|
|
1261
|
+
prompting=bool(opts.get("prompting", False)),
|
|
1262
|
+
decisions_only=bool(opts.get("decisions_only", False)),
|
|
1263
|
+
max_bundles=opts.get("max_bundles"),
|
|
1264
|
+
repeat=opts.get("repeat"),
|
|
1265
|
+
max_cycles=opts.get("max_cycles"),
|
|
1266
|
+
)
|
|
1267
|
+
else:
|
|
1268
|
+
story_common = dict(
|
|
1269
|
+
paths=paths,
|
|
1270
|
+
policy=policy,
|
|
1271
|
+
adapter=adapters["dev"],
|
|
1272
|
+
review_adapter=adapters["review"],
|
|
1273
|
+
run_dir=run_dir,
|
|
1274
|
+
journal=journal,
|
|
1275
|
+
state=state,
|
|
1276
|
+
# restore the launching scope + cap so a resumed `--epic N` run keeps
|
|
1277
|
+
# picking within N instead of silently widening to every epic.
|
|
1278
|
+
epic_filter=state.epic_filter,
|
|
1279
|
+
story_filter=state.story_filter,
|
|
1280
|
+
max_stories=state.max_stories,
|
|
1281
|
+
sweep_factory=sweep_factory,
|
|
1282
|
+
)
|
|
1283
|
+
# stories mode is pinned in run state at launch, so resume rebuilds the
|
|
1284
|
+
# same picker (StoriesEngine) without any flag.
|
|
1285
|
+
# heterogeneous **kwargs: pyright unions the dict values; per-arg error is spurious
|
|
1286
|
+
engine = (
|
|
1287
|
+
stories_engine_cls(**story_common, spec_folder=state.spec_folder)
|
|
1288
|
+
if state.source == "stories"
|
|
1289
|
+
else engine_cls(**story_common) # pyright: ignore[reportArgumentType]
|
|
1290
|
+
)
|
|
1291
|
+
return ComposedRun(
|
|
1292
|
+
engine=engine, run_id=run_dir.name, run_dir=run_dir, state=state, journal=journal
|
|
1293
|
+
)
|